Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -145,17 +145,27 @@ public enum CQLFields implements CQLFieldsInterface {
" .getAsLong()" +
" }"))
.order(order))),
// Phrase match on the field and its synonyms sub-field (search-time acronym expansion), fuzziness is not allowed
// for type phrase. Also used for quoted keywords, see fuzzy_title for why best of the two fields is scored.
title(
StacBasicField.Title.searchField,
StacBasicField.Title.displayField,
null,
(literal) -> MultiMatchQuery.of(m -> m
.type(TextQueryType.Phrase)
.fields(StacBasicField.Title.searchField + "^2", StacBasicField.Title.searchField + ".synonyms^2")
.tieBreaker(0.1)
.query(literal))._toQuery(),
null,
(order) -> new SortOptions.Builder()
.field(f -> f.field(StacBasicField.Title.sortField).order(order))),
description(
StacBasicField.Description.searchField,
StacBasicField.Description.displayField,
null,
(literal) -> MultiMatchQuery.of(m -> m
.type(TextQueryType.Phrase)
.fields(StacBasicField.Description.searchField, StacBasicField.Description.searchField + ".synonyms")
.tieBreaker(0.1)
.query(literal))._toQuery(),
null,
null),
providers(
Expand Down Expand Up @@ -266,50 +276,35 @@ public enum CQLFields implements CQLFieldsInterface {
null,
(order) -> new SortOptions.Builder()
.field(f -> f.field(StacSummeries.Score.sortField).order(order))),
// Fuzzy match on the field and its synonyms sub-field (search-time acronym expansion, e.g. "SOOP" -> "ships of opportunity").
// best_fields scores the better field plus tie_breaker times the other. Plain words pass the synonym analyzer
// unchanged and would otherwise be scored twice. Title fields are boosted 2, description fields 1, as before.
fuzzy_title(
null,
StacBasicField.Title.displayField,
(literal) -> MatchQuery.of(m -> m
(literal) -> MultiMatchQuery.of(m -> m
.type(TextQueryType.BestFields)
.fields(StacBasicField.Title.searchField + "^2", StacBasicField.Title.searchField + ".synonyms^2")
.tieBreaker(0.1)
.fuzziness("AUTO")
.field(StacBasicField.Title.searchField)
.prefixLength(4)// Use 4 to deal with NRMN short form may match NRM records
// Increase the relevance of matches in title
.boost(2.0F)
.operator(Operator.And)// ensure all terms are matched with fuzziness
.query(literal))._toQuery(),
null,
null),
fuzzy_desc(
null,
StacBasicField.Description.displayField,
(literal) -> MatchQuery.of(m -> m
(literal) -> MultiMatchQuery.of(m -> m
.type(TextQueryType.BestFields)
.fields(StacBasicField.Description.searchField, StacBasicField.Description.searchField + ".synonyms")
.tieBreaker(0.1)
.fuzziness("AUTO")
.field(StacBasicField.Description.searchField)
.prefixLength(4)// Use 4 to deal with NRMN short form may match NRM records
.operator(Operator.And)// ensure all terms are matched with fuzziness
.query(literal))._toQuery(),
null,
null),
// Acronym match on the synonyms sub-fields (search-time expansion), e.g. "SOOP" -> "ships of opportunity".
acronym_title(
StacBasicField.Title.searchField + ".synonyms",
StacBasicField.Title.displayField,
(literal) -> MatchQuery.of(m -> m
.field(StacBasicField.Title.searchField + ".synonyms")
.operator(Operator.And)// all expanded terms must match
.boost(2.0F)// align with fuzzy_title weighting
.query(literal))._toQuery(),
null,
null),
acronym_desc(
StacBasicField.Description.searchField + ".synonyms",
StacBasicField.Description.displayField,
(literal) -> MatchQuery.of(m -> m
.field(StacBasicField.Description.searchField + ".synonyms")
.operator(Operator.And)
.query(literal))._toQuery(),
null,
null),
// Contains cloud-optimized data
assets_summary(
StacBasicField.AssetsSummary.searchField,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -449,9 +449,6 @@ private static List<Query> createKeywordShouldClauses(List<String> keywords) {
should.add(CQLFields.organisation_vocabs.getPropertyEqualToQuery(term));
should.add(CQLFields.platform_vocabs.getPropertyEqualToQuery(term));
should.add(CQLFields.id.getPropertyEqualToQuery(term));
// Acronym match on the *.synonyms sub-fields, e.g. "SOOP" -> "ships of opportunity".
should.add(CQLFields.acronym_title.getPropertyEqualToQuery(term));
should.add(CQLFields.acronym_desc.getPropertyEqualToQuery(term));
// credit_contains uses match query by default, exact match is not applied here
should.add(CQLFields.credit_contains.getPropertyEqualToQuery(term));
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,12 @@ public class ExplainSimplifier {

protected static final String SYNONYM_PREFIX = "Synonym(";

// A dis_max (disjunction max) explanation includes every match, but it is scored as its winning child plus
// tie_breaker times the others, e.g. the best_fields multi_match on title and title.synonyms (see CQLFields.fuzzy_title).
// Lucene writes "max of:" for tie_breaker 0, otherwise "max plus 0.1 times others of:".
protected static final String MAX_OF_PREFIX = "max of:";
protected static final String MAX_PLUS_PREFIX = "max plus ";

protected static final String RELEVANCE_DESCRIPTION_PREFIX = "_score:";

protected static final Comparator<ExplainSimplifiedResponse.MatchedTerm> BY_SCORE_DESC =
Expand Down Expand Up @@ -214,6 +220,19 @@ protected static void collectScoreParts(List<ExplanationDetail> details,
for (ExplanationDetail detail : details) {
String description = detail.description();

// report only the winner, the tie_breaker share of the others is small and would list the same word twice
if (description != null
&& (description.startsWith(MAX_OF_PREFIX) || description.startsWith(MAX_PLUS_PREFIX))) {
if (detail.details() != null) {
// the highest child, not one equal to the parent, so float rounding cannot lose the winner; a tie keeps the first
detail.details().stream()
.max(Comparator.comparingDouble(ExplanationDetail::value))
.ifPresent(winner -> collectScoreParts(
List.of(winner), terms, filters));
}
continue;
}

// most nodes are idf/tf breakdowns, skip the regex for them
if (description != null && description.startsWith(WEIGHT_PREFIX)) {
Matcher matcher = WEIGHT_PATTERN.matcher(description);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -187,14 +187,16 @@ public void explainSimplifiedFormatReportsMultiWordMatches() throws IOException
assertFalse(term.path("term").asText().isBlank());
}

// both words of the query are reported, and separately for each field they hit,
// rather than being collapsed into one entry per field
// both words of the query are reported, rather than being collapsed into one entry per field
List<String> reported = StreamSupport.stream(terms.spliterator(), false)
.map(term -> term.path("field").asText() + ":" + term.path("term").asText())
.toList();
assertTrue(reported.contains("title:ocean"));
assertTrue(reported.contains("title:acidification"));
assertTrue(reported.size() > 2, "a multi word query hits more than one field");
for (String word : List.of("ocean", "acidification")) {
long titleHits = reported.stream()
.filter(r -> r.equals("title:" + word) || r.equals("title.synonyms:" + word))
.count();
assertEquals(1, titleHits, "'" + word + "' must be reported once for title, got: " + reported);
}
}

@Test
Expand Down Expand Up @@ -226,8 +228,8 @@ public void explainSimplifiedFormatReportsTheDatasetGroupPriorityInTheSortValues
public void explainSimplifiedFormatReportsQuotedPhraseMatches() throws IOException {
insertRecordsWithExplicitIds("7709f541-fc0c-4318-b5b9-9053aa474e0e.json");

// a quoted query routes title and description through MatchPhraseQuery, which elastic
// search renders as a single weight(title:"ocean acidification" in N) leaf
// a quoted query routes title and description through a phrase multi_match on the field and its
// synonyms sub-field, each child renders as a single weight(title:"ocean acidification" in N) leaf
URI simpleUri = explainUri("q", "\"ocean acidification\"", "format", "simple");

ResponseEntity<JsonNode> response = testRestTemplate.getForEntity(simpleUri, JsonNode.class);
Expand All @@ -237,9 +239,11 @@ public void explainSimplifiedFormatReportsQuotedPhraseMatches() throws IOExcepti
.path("hits").path(0).path("matched_terms");

// the phrase is reported as one term with its quotes stripped, a term only
// extraction would drop this leaf entirely
// extraction would drop this leaf entirely. A plain phrase scores the same on title and
// title.synonyms, so either can win the dis_max
assertTrue(StreamSupport.stream(terms.spliterator(), false)
.anyMatch(term -> "title".equals(term.path("field").asText())
.anyMatch(term -> ("title".equals(term.path("field").asText())
|| "title.synonyms".equals(term.path("field").asText()))
&& "ocean acidification".equals(term.path("term").asText())),
"a quoted phrase must be reported as a single matched term");
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -5,10 +5,12 @@
import co.elastic.clients.elasticsearch._types.ScriptSortType;
import co.elastic.clients.elasticsearch._types.SortOptions;
import co.elastic.clients.elasticsearch._types.SortOrder;
import co.elastic.clients.elasticsearch.core.explain.ExplanationDetail;
import co.elastic.clients.elasticsearch.core.search.Hit;
import com.fasterxml.jackson.databind.node.ObjectNode;
import org.junit.jupiter.api.Test;

import java.util.ArrayList;
import java.util.List;

import static org.junit.jupiter.api.Assertions.assertEquals;
Expand Down Expand Up @@ -86,6 +88,68 @@ public void sortValuesAreLeftOutWhenTheHitCarriesNone() {
assertNull(simplified.getSortValues());
}

@Test
public void disMaxReportsOnlyOneWinnerWhenPlainAndSynonymScoresTie() {
List<ExplainSimplifiedResponse.MatchedTerm> terms = collectDisMax("max of:", 12.1f,
detail(12.1f, "weight(title:wave in 1) [BM25], result of:"),
detail(12.1f, "weight(title.synonyms:wave in 1) [BM25], result of:"));

assertEquals(1, terms.size());
assertEquals("title", terms.get(0).getField());
assertEquals(12.1, terms.get(0).getScore(), 0.0001);
}

@Test
public void disMaxReportsSynonymWinnerAsOneEntry() {
List<ExplainSimplifiedResponse.MatchedTerm> terms = collectDisMax("max of:", 12.1f,
detail(4.2f, "weight(title:soop in 1) [BM25], result of:"),
detail(12.1f, "weight(Synonym(title.synonyms:soop title.synonyms:ships "
+ "title.synonyms:of title.synonyms:opportunity) in 1) [BM25], result of:"));

assertEquals(1, terms.size());
assertEquals("title.synonyms", terms.get(0).getField());
assertEquals("soop ships of opportunity", terms.get(0).getTerm());
assertEquals(12.1, terms.get(0).getScore(), 0.0001);
}

@Test
public void disMaxReportsHighestChildWhenNoChildEqualsParent() {
List<ExplainSimplifiedResponse.MatchedTerm> terms = collectDisMax("max of:", 12.1f,
detail(12.0999f, "weight(title:wave in 1) [BM25], result of:"),
detail(4.0f, "weight(title.synonyms:wave in 1) [BM25], result of:"));

assertEquals(1, terms.size());
assertEquals("title", terms.get(0).getField());
}

@Test
public void disMaxWithTieBreakerReportsOnlyTheWinner() {
// best_fields multi_match with tie_breaker 0.1: 36.3 + 0.1 * 12.1
List<ExplainSimplifiedResponse.MatchedTerm> terms = collectDisMax("max plus 0.1 times others of:", 37.51f,
detail(36.3f, "weight(title:wave in 1) [BM25], result of:"),
detail(12.1f, "weight(title.synonyms:wave in 1) [BM25], result of:"));

assertEquals(1, terms.size());
assertEquals("title", terms.get(0).getField());
assertEquals(36.3, terms.get(0).getScore(), 0.0001);
}

private static List<ExplainSimplifiedResponse.MatchedTerm> collectDisMax(String description, float value,
ExplanationDetail... children) {
List<ExplainSimplifiedResponse.MatchedTerm> terms = new ArrayList<>();
ExplainSimplifier.collectScoreParts(
List.of(detail(value, description, children)), terms, new ArrayList<>());
return terms;
}

private static ExplanationDetail detail(float value, String description,
ExplanationDetail... children) {
return ExplanationDetail.of(d -> d
.value(value)
.description(description)
.details(List.of(children)));
}

private static ExplainSimplifiedResponse.SortValue sortValue(String field, Object value) {
return ExplainSimplifiedResponse.SortValue.builder()
.field(field)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -357,7 +357,7 @@ public void verifyCorrectPageSizeAndScoreWithQuery() throws IOException {

// Call rest api directly and get query result with search on "dataset"
ResponseEntity<ExtendedCollections> collections = testRestTemplate.exchange(
getBasePath() + "/collections?q=dataset&filter=page_size=1 AND score>=1.3",
getBasePath() + "/collections?q=dataset&filter=page_size=1 AND score>=0.75",
HttpMethod.GET,
null,
new ParameterizedTypeReference<>() {
Expand Down Expand Up @@ -386,7 +386,7 @@ public void verifyCorrectPageSizeAndScoreWithQuery() throws IOException {

// Now the same search, same page but search_after the actual cursor returned above
collections = testRestTemplate.exchange(
getBasePath() + "/collections?q=dataset&filter=page_size=6 AND score>=1.3 AND search_after=" +
getBasePath() + "/collections?q=dataset&filter=page_size=6 AND score>=0.75 AND search_after=" +
String.format("'%s|| %s || %s || %s'",
collections.getBody().getSearchAfter().get(0),
collections.getBody().getSearchAfter().get(1),
Expand All @@ -401,7 +401,7 @@ public void verifyCorrectPageSizeAndScoreWithQuery() throws IOException {
assertEquals(HttpStatus.OK, collections.getStatusCode(), "Get status OK");

log.info("{}", collections.getBody());
// Remaining docs that clear min_score=1.3 after the first batch; the exact count is
// Remaining docs that clear min_score=0.75 after the first batch; the exact count is
// BM25-dependent and varies by env, so accept any non-empty result up to the remaining total.
int returnedSize = Objects.requireNonNull(collections.getBody()).getCollections().size();
assertTrue(returnedSize >= 1 && returnedSize <= 3,
Expand Down
Loading
Loading