Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions lucene/CHANGES.txt
Original file line number Diff line number Diff line change
Expand Up @@ -187,6 +187,11 @@ Optimizations

Bug Fixes
---------------------
* GITHUB#15769: PathHierarchyTokenizer again emits all path tokens at the same position, so query
parsers treat them as synonyms and the documented "ancestor path" search works as designed. This
had been broken since 10.0 by GITHUB#12875, which made every token increment the position and
turned such queries into phrases. (Hoss Man, Vismay Tiwari)

* GITHUB#14049: Randomize KNN codec params in RandomCodec. Fixes scalar quantization div-by-zero
when all values are identical. (Mike Sokolov)

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -113,7 +113,16 @@ public PathHierarchyTokenizer(
public final boolean incrementToken() throws IOException {
clearAttributes();
termAtt.append(resultToken);
posIncAtt.setPositionIncrement(1);
// Emit every path token at the same position -- the first token advances the position and
// the rest keep it -- so query parsers treat them as synonyms: a query for "a/b/c" becomes
// (a OR a/b OR a/b/c), which is what the documented "ancestor path" use case relies on.
// #12875 had made every token increment the position, turning such a query into a phrase
// and breaking that use case (see #15769).
if (resultToken.length() == 0) {
posIncAtt.setPositionIncrement(1);
} else {
posIncAtt.setPositionIncrement(0);
}
int length = 0;
boolean added = false;
if (endDelimiter) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -22,12 +22,23 @@
import java.io.IOException;
import java.io.Reader;
import java.io.StringReader;
import java.util.Arrays;
import java.util.Random;
import org.apache.lucene.analysis.Analyzer;
import org.apache.lucene.analysis.Tokenizer;
import org.apache.lucene.analysis.charfilter.MappingCharFilter;
import org.apache.lucene.analysis.charfilter.NormalizeCharMap;
import org.apache.lucene.analysis.core.KeywordAnalyzer;
import org.apache.lucene.document.Document;
import org.apache.lucene.document.Field;
import org.apache.lucene.document.TextField;
import org.apache.lucene.index.DirectoryReader;
import org.apache.lucene.index.IndexReader;
import org.apache.lucene.index.IndexWriter;
import org.apache.lucene.search.IndexSearcher;
import org.apache.lucene.store.Directory;
import org.apache.lucene.tests.analysis.BaseTokenStreamTestCase;
import org.apache.lucene.util.QueryBuilder;

public class TestPathHierarchyTokenizer extends BaseTokenStreamTestCase {

Expand All @@ -42,7 +53,7 @@ public void testBasic() throws Exception {
new String[] {"/a", "/a/b", "/a/b/c"},
new int[] {0, 0, 0},
new int[] {2, 4, 6},
new int[] {1, 1, 1},
new int[] {1, 0, 0},
path.length());
}

Expand All @@ -57,7 +68,7 @@ public void testEndOfDelimiter() throws Exception {
new String[] {"/a", "/a/b", "/a/b/c", "/a/b/c/"},
new int[] {0, 0, 0, 0},
new int[] {2, 4, 6, 7},
new int[] {1, 1, 1, 1},
new int[] {1, 0, 0, 0},
path.length());
}

Expand All @@ -72,7 +83,7 @@ public void testStartOfChar() throws Exception {
new String[] {"a", "a/b", "a/b/c"},
new int[] {0, 0, 0},
new int[] {1, 3, 5},
new int[] {1, 1, 1},
new int[] {1, 0, 0},
path.length());
}

Expand All @@ -87,7 +98,7 @@ public void testStartOfCharEndOfDelimiter() throws Exception {
new String[] {"a", "a/b", "a/b/c", "a/b/c/"},
new int[] {0, 0, 0, 0},
new int[] {1, 3, 5, 6},
new int[] {1, 1, 1, 1},
new int[] {1, 0, 0, 0},
path.length());
}

Expand All @@ -112,7 +123,7 @@ public void testOnlyDelimiters() throws Exception {
new String[] {"/", "//"},
new int[] {0, 0},
new int[] {1, 2},
new int[] {1, 1},
new int[] {1, 0},
path.length());
}

Expand All @@ -126,7 +137,7 @@ public void testReplace() throws Exception {
new String[] {"\\a", "\\a\\b", "\\a\\b\\c"},
new int[] {0, 0, 0},
new int[] {2, 4, 6},
new int[] {1, 1, 1},
new int[] {1, 0, 0},
path.length());
}

Expand All @@ -140,7 +151,7 @@ public void testWindowsPath() throws Exception {
new String[] {"c:", "c:\\a", "c:\\a\\b", "c:\\a\\b\\c"},
new int[] {0, 0, 0, 0},
new int[] {2, 4, 6, 8},
new int[] {1, 1, 1, 1},
new int[] {1, 0, 0, 0},
path.length());
}

Expand All @@ -159,7 +170,7 @@ public void testNormalizeWinDelimToLinuxDelim() throws Exception {
new String[] {"c:", "c:/a", "c:/a/b", "c:/a/b/c"},
new int[] {0, 0, 0, 0},
new int[] {2, 4, 6, 8},
new int[] {1, 1, 1, 1},
new int[] {1, 0, 0, 0},
path.length());
}

Expand All @@ -173,7 +184,7 @@ public void testBasicSkip() throws Exception {
new String[] {"/b", "/b/c"},
new int[] {2, 2},
new int[] {4, 6},
new int[] {1, 1},
new int[] {1, 0},
path.length());
}

Expand All @@ -187,7 +198,7 @@ public void testEndOfDelimiterSkip() throws Exception {
new String[] {"/b", "/b/c", "/b/c/"},
new int[] {2, 2, 2},
new int[] {4, 6, 7},
new int[] {1, 1, 1},
new int[] {1, 0, 0},
path.length());
}

Expand All @@ -201,7 +212,7 @@ public void testStartOfCharSkip() throws Exception {
new String[] {"/b", "/b/c"},
new int[] {1, 1},
new int[] {3, 5},
new int[] {1, 1},
new int[] {1, 0},
path.length());
}

Expand All @@ -215,7 +226,7 @@ public void testStartOfCharEndOfDelimiterSkip() throws Exception {
new String[] {"/b", "/b/c", "/b/c/"},
new int[] {1, 1, 1},
new int[] {3, 5, 6},
new int[] {1, 1, 1},
new int[] {1, 0, 0},
path.length());
}

Expand Down Expand Up @@ -282,9 +293,137 @@ protected TokenStreamComponents createComponents(String fieldName) {
};

public void testTokenizerViaAnalyzerOutput() throws IOException {
assertAnalyzesTo(analyzer, "a/b/c", new String[] {"a", "a/b", "a/b/c"});
assertAnalyzesTo(analyzer, "a/b/c/", new String[] {"a", "a/b", "a/b/c", "a/b/c/"});
assertAnalyzesTo(analyzer, "/a/b/c", new String[] {"/a", "/a/b", "/a/b/c"});
assertAnalyzesTo(analyzer, "/a/b/c/", new String[] {"/a", "/a/b", "/a/b/c", "/a/b/c/"});
// The path tokens share a position (posInc 0 after the first), so their offsets overlap;
// pass graphOffsetsAreCorrect=false so the strict graph-offset check does not reject the
// legitimate overlapping-prefix output (see #15769).
assertAnalyzesTo(
analyzer,
"a/b/c",
new String[] {"a", "a/b", "a/b/c"},
null,
null,
null,
new int[] {1, 0, 0},
null,
false);
assertAnalyzesTo(
analyzer,
"a/b/c/",
new String[] {"a", "a/b", "a/b/c", "a/b/c/"},
null,
null,
null,
new int[] {1, 0, 0, 0},
null,
false);
assertAnalyzesTo(
analyzer,
"/a/b/c",
new String[] {"/a", "/a/b", "/a/b/c"},
null,
null,
null,
new int[] {1, 0, 0},
null,
false);
assertAnalyzesTo(
analyzer,
"/a/b/c/",
new String[] {"/a", "/a/b", "/a/b/c", "/a/b/c/"},
null,
null,
null,
new int[] {1, 0, 0, 0},
null,
false);
}

private static final Iterable<String> PATH_DOCS =
Arrays.asList(
"Books",
"Books/Fic",
"Books/Fic/Mystery",
"Books/NonFic",
"Books/NonFic/Law",
"Books/NonFic/Science/Physics");

/**
* "Descendant path" search: index with {@link PathHierarchyTokenizer}, query with a keyword
* analyzer. A query for {@code Books/NonFic} matches every document at or below that path.
*/
public void testDescendantQuery() throws Exception {
final String field = "descendant";
try (Directory dir = newDirectory()) {
final Analyzer a =
new Analyzer() {
@Override
protected TokenStreamComponents createComponents(String fieldName) {
return new TokenStreamComponents(new PathHierarchyTokenizer());
}
};
try (IndexWriter w = new IndexWriter(dir, newIndexWriterConfig(a))) {
for (String val : PATH_DOCS) {
Document doc = new Document();
doc.add(new TextField(field, val, Field.Store.YES));
w.addDocument(doc);
}
}
try (IndexReader r = DirectoryReader.open(dir)) {
final QueryBuilder parser = new QueryBuilder(new KeywordAnalyzer());
final IndexSearcher s = newSearcher(r);
assertEquals(
3, s.search(parser.createPhraseQuery(field, "Books/NonFic"), 100).totalHits.value());
assertEquals(
2, s.search(parser.createPhraseQuery(field, "Books/Fic"), 100).totalHits.value());
assertEquals(6, s.search(parser.createPhraseQuery(field, "Books"), 100).totalHits.value());
}
a.close();
}
}

/**
* "Ancestor path" search: index with a keyword analyzer, query with {@link
* PathHierarchyTokenizer}. A query for {@code Books/NonFic/Science/Physics} matches documents
* holding any ancestor path. Relies on the query tokens sharing a position so the parser builds a
* synonym (OR) query rather than a phrase (regression test for #15769).
*/
public void testAncestorQuery() throws Exception {
final String field = "ancestor";
try (Directory dir = newDirectory()) {
try (IndexWriter w = new IndexWriter(dir, newIndexWriterConfig(new KeywordAnalyzer()))) {
for (String val : PATH_DOCS) {
Document doc = new Document();
doc.add(new TextField(field, val, Field.Store.YES));
w.addDocument(doc);
}
}
try (IndexReader r = DirectoryReader.open(dir)) {
final QueryBuilder parser =
new QueryBuilder(
new Analyzer() {
@Override
protected TokenStreamComponents createComponents(String fieldName) {
return new TokenStreamComponents(new PathHierarchyTokenizer());
}
});
final IndexSearcher s = newSearcher(r);
assertEquals(
3,
s.search(parser.createPhraseQuery(field, "Books/NonFic/Science/Physics"), 100)
.totalHits
.value());
assertEquals(
2,
s.search(parser.createPhraseQuery(field, "Books/NonFic/Science"), 100)
.totalHits
.value());
assertEquals(
3,
s.search(parser.createPhraseQuery(field, "Books/Fic/Mystery/Noir"), 100)
.totalHits
.value());
assertEquals(1, s.search(parser.createPhraseQuery(field, "Books"), 100).totalHits.value());
}
}
}
}
Loading