Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,8 @@
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.searchablestructure.result.RowNumAndSimilarity;
import com.uber.ussi.searchablestructure.result.ResultHeaps;
import com.uber.ussi.searchablestructure.index.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.PopularTermDiscardPolicy;
import com.uber.ussi.searchablestructure.utils.metadata.PreFilteringResult;
import com.uber.ussi.searchablestructure.utils.parallel.ParallelRowScan;
import com.uber.ussi.searchablestructure.utils.parallel.SharedMinSimilarity;
Expand Down Expand Up @@ -52,13 +53,15 @@ public final class InvertedTermCache extends Cache {

public InvertedTermCache(NamespaceConfig namespaceConfig) {
super(namespaceConfig);
this.maxFractionIdsPerTerm = parseMaxFractionIdsPerTerm(namespaceConfig);
this.maxFractionIdsPerTerm =
PopularTermDiscardPolicy.maxFractionIdsPerTermFromCacheConfig(namespaceConfig);
this.popularTermDiscardScope = namespaceConfig.getCachePopularTermDiscardScope();
this.fullReevaluationCacheSizeDecreaseFraction =
parseFullReevaluationCacheSizeDecreaseFraction(namespaceConfig);
this.popularityConfidenceTester =
new MathUtils.ProportionConfidenceInterval1Sided(
parseMaxFractionIdsPerTermConfidence(namespaceConfig));
PopularTermDiscardPolicy.maxFractionIdsPerTermConfidenceFromCacheConfig(
namespaceConfig));
this.termAndRowNumsIndex = new LongObjectHashMap<>(CACHE_INITIAL_CAPACITY);
this.discardedTerms = new LongHashSet();
}
Expand Down Expand Up @@ -357,26 +360,11 @@ private void rebuildDiscardedTerms() {
}


private static double parseMaxFractionIdsPerTerm(NamespaceConfig namespaceConfig) {
return namespaceConfig.readDoubleCacheParam(
ConfigKeys.MAX_FRACTION_IDS_PER_TERM,
ConfigKeys.DEFAULT_MAX_FRACTION_IDS_PER_TERM);
}

private static double parseFullReevaluationCacheSizeDecreaseFraction(
NamespaceConfig namespaceConfig) {
return namespaceConfig.readDoubleCacheParam(
ConfigKeys.FULL_REEVALUATION_CACHE_SIZE_DECREASE_FRACTION,
ConfigKeys.DEFAULT_FULL_REEVALUATION_CACHE_SIZE_DECREASE_FRACTION);
}

/**
* Parses the one-sided confidence used to declare a term high-popularity. A confidence of 0.5
* degenerates to comparing the observed popularity against maxFractionIdsPerTerm directly.
*/
private static double parseMaxFractionIdsPerTermConfidence(NamespaceConfig namespaceConfig) {
return namespaceConfig.readDoubleCacheParam(
ConfigKeys.MAX_FRACTION_IDS_PER_TERM_CONFIDENCE,
ConfigKeys.DEFAULT_MAX_FRACTION_IDS_PER_TERM_CONFIDENCE);
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,9 @@
import com.uber.ussi.searchablestructure.index.inverted.generator.InvertedList;
import com.uber.ussi.searchablestructure.index.inverted.generator.MergeSearch;
import com.uber.ussi.searchablestructure.result.ResultHeaps;
import com.uber.ussi.searchablestructure.index.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndUniTransformedValue;
import com.uber.ussi.searchablestructure.utils.inverted.PopularTermDiscardPolicy;
import com.uber.ussi.searchablestructure.utils.metadata.MetadataFilteringStrategy;
import com.uber.ussi.searchablestructure.utils.parallel.ParallelShardSearch;
import com.uber.ussi.searchablestructure.utils.parallel.SharedMinSimilarity;
Expand Down Expand Up @@ -169,7 +171,8 @@ protected BaseInvertedIndex(
this.mergeScoresFromAccumulatedConjunction = sparsMerge.mergeScoresFromAccumulatedConjunction();
this.partialConjunctionPolicy = sparsMerge.partialConjunctionPolicy();
this.doesStoreMergePostingValues = sparsMerge.doesStoreMergePostingValues();
this.maxFractionIdsPerTerm = PopularTermDiscardPolicy.maxFractionIdsPerTerm(namespaceConfig);
this.maxFractionIdsPerTerm =
PopularTermDiscardPolicy.maxFractionIdsPerTermFromIndexConfig(namespaceConfig);
validateRows();
this.discardedTerms =
structureDiscardedTerms == null
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.searchablestructure.result.RowNumAndSimilarity;
import com.uber.ussi.searchablestructure.index.Index;
import com.uber.ussi.searchablestructure.utils.inverted.PopularTermDiscardPolicy;
import com.uber.ussi.utils.BoundedSizeMaxHeap;
import java.util.List;

Expand All @@ -26,7 +27,8 @@ public HybridIndex(
LongObjectHashMap<LongTermsAndValues> rowNumToTermsAndValuesMap,
LongObjectHashMap<LongMeta> rowNumToMetaMap) {
super(namespaceConfig);
double maxFractionIdsPerTerm = PopularTermDiscardPolicy.maxFractionIdsPerTerm(namespaceConfig);
double maxFractionIdsPerTerm =
PopularTermDiscardPolicy.maxFractionIdsPerTermFromIndexConfig(namespaceConfig);
LongObjectHashMap<LongTermsAndValues> exactRows = new LongObjectHashMap<>();
LongObjectHashMap<LongTermsAndValues> signatureRows = new LongObjectHashMap<>();
for (LongObjectCursor<LongTermsAndValues> entry : rowNumToTermsAndValuesMap) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import com.uber.ussi.entity.meta.LongMeta;
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.searchablestructure.index.IndexType;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndUniTransformedValue;
import java.util.Arrays;
import javax.annotation.Nullable;

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import com.uber.ussi.entity.meta.LongMeta;
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.searchablestructure.index.IndexType;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndUniTransformedValue;
import javax.annotation.Nullable;

/**
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.searchablestructure.result.RowNumAndSimilarity;
import com.uber.ussi.searchablestructure.result.ResultHeaps;
import com.uber.ussi.searchablestructure.index.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.parallel.SharedMinSimilarity;
import com.uber.ussi.utils.BoundedSizeMaxHeap;
import com.uber.ussi.utils.MathUtils;
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
/* AUTHOR: Shijie Lu (shijie@uber.com), Shalini Kedlaya (skedlaya@uber.com), Ahmed Metwally (ametwally@uber.com) */
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

/**
* Query key data ordered for low-cost prefix candidate generation. Shared by the inverted indexes
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
/* AUTHOR: Shijie Lu (shijie@uber.com), Shalini Kedlaya (skedlaya@uber.com), Ahmed Metwally (ametwally@uber.com) */
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

/** A key and its contribution to the comparator-specific unilateral value. */
public final class KeyAndUniTransformedValue {
Expand All @@ -11,11 +11,11 @@ public KeyAndUniTransformedValue(long key, double uniTransformedValue) {
this.uniTransformedValue = uniTransformedValue;
}

long getKey() {
public long getKey() {
return key;
}

double getUniTransformedValue() {
public double getUniTransformedValue() {
return uniTransformedValue;
}
}
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
/* AUTHOR: Ahmed Metwally (ametwally@uber.com) */
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

import com.carrotsearch.hppc.LongHashSet;
import com.carrotsearch.hppc.LongIntHashMap;
Expand All @@ -12,25 +12,52 @@
import java.util.Objects;

/**
* Popularity-based term discard policy for inverted term-keyed indexes.
* Popularity-based term discard policy for inverted term-keyed structures.
*
* <p>Hybrid and term indexes share this policy so a composite structure can compute discards over
* its term-keyed partition without depending on {@link BaseInvertedIndex}.
* <p>Graduated indexes read thresholds from {@code index_params}; the writable inverted term cache
* reads the same keys from {@code cache_params}. Batch discard over a complete row map applies only
* to index construction.
*/
public final class PopularTermDiscardPolicy {

private PopularTermDiscardPolicy() {}

/** Returns the configured maximum fraction of rows a term may appear in before it is discarded. */
public static double maxFractionIdsPerTerm(NamespaceConfig namespaceConfig) {
/**
* Returns the configured maximum fraction of rows a term may appear in before it is discarded, from
* {@code index_params}.
*/
public static double maxFractionIdsPerTermFromIndexConfig(NamespaceConfig namespaceConfig) {
Objects.requireNonNull(namespaceConfig, "namespaceConfig is null.");
return namespaceConfig.readDoubleIndexParam(
ConfigKeys.MAX_FRACTION_IDS_PER_TERM, ConfigKeys.DEFAULT_MAX_FRACTION_IDS_PER_TERM);
}

/** Returns whether any term may be discarded under the configured popularity threshold. */
public static boolean doesDiscardPopularTerms(NamespaceConfig namespaceConfig) {
return maxFractionIdsPerTerm(namespaceConfig) < 1.0;
/**
* Returns the configured maximum fraction of rows a term may appear in before it is discarded,
* from {@code cache_params}.
*/
public static double maxFractionIdsPerTermFromCacheConfig(NamespaceConfig namespaceConfig) {
Objects.requireNonNull(namespaceConfig, "namespaceConfig is null.");
return namespaceConfig.readDoubleCacheParam(
ConfigKeys.MAX_FRACTION_IDS_PER_TERM, ConfigKeys.DEFAULT_MAX_FRACTION_IDS_PER_TERM);
}

/**
* Returns the one-sided confidence used to declare a term high-popularity in the inverted term
* cache, from {@code cache_params}. A confidence of 0.5 degenerates to comparing observed
* popularity against {@link #maxFractionIdsPerTermFromCacheConfig} directly.
*/
public static double maxFractionIdsPerTermConfidenceFromCacheConfig(
NamespaceConfig namespaceConfig) {
Objects.requireNonNull(namespaceConfig, "namespaceConfig is null.");
return namespaceConfig.readDoubleCacheParam(
ConfigKeys.MAX_FRACTION_IDS_PER_TERM_CONFIDENCE,
ConfigKeys.DEFAULT_MAX_FRACTION_IDS_PER_TERM_CONFIDENCE);
}

/** Returns whether any term may be discarded under the index popularity threshold. */
public static boolean doesDiscardPopularTermsFromIndexConfig(NamespaceConfig namespaceConfig) {
return maxFractionIdsPerTermFromIndexConfig(namespaceConfig) < 1.0;
}

/** Returns whether any term may be discarded under {@code maxFractionIdsPerTerm}. */
Expand All @@ -39,14 +66,16 @@ public static boolean doesDiscardPopularTerms(double maxFractionIdsPerTerm) {
}

/**
* Returns the terms a structure holding {@code rowNumToTermsAndValuesMap} discards as popular.
* Returns the terms a graduated index holding {@code rowNumToTermsAndValuesMap} discards as
* popular, using {@code index_params}.
*/
public static LongHashSet discardedTermsOf(
NamespaceConfig namespaceConfig,
LongObjectHashMap<LongTermsAndValues> rowNumToTermsAndValuesMap) {
Objects.requireNonNull(namespaceConfig, "namespaceConfig is null.");
Objects.requireNonNull(rowNumToTermsAndValuesMap, "rowNumToTermsAndValuesMap is null.");
return discardedTermsOf(rowNumToTermsAndValuesMap, maxFractionIdsPerTerm(namespaceConfig));
return discardedTermsOf(
rowNumToTermsAndValuesMap, maxFractionIdsPerTermFromIndexConfig(namespaceConfig));
}

/**
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@
import com.uber.ussi.searchablestructure.index.IndexType;
import com.uber.ussi.searchablestructure.index.scan.ScanIndex;
import com.uber.ussi.searchablestructure.result.RowNumAndSimilarity;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndUniTransformedValue;
import com.uber.ussi.searchablestructure.utils.metadata.MetadataFilteringStrategy;
import com.uber.ussi.utils.ConfigKeys;
import java.lang.reflect.Field;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@
import com.uber.ussi.entity.termsandvalues.LongTermsAndValues;
import com.uber.ussi.entity.termsandvalues.LongTermsAndValuesTestFactory;
import com.uber.ussi.searchablestructure.result.RowNumAndSimilarity;
import com.uber.ussi.searchablestructure.index.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.inverted.KeyAndPrefixFilteringData;
import com.uber.ussi.searchablestructure.utils.parallel.SharedMinSimilarity;
import java.util.ArrayList;
import java.util.List;
Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertSame;
Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

import static org.junit.jupiter.api.Assertions.assertEquals;

Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
/* AUTHOR: Ahmed Metwally (ametwally@uber.com) */
package com.uber.ussi.searchablestructure.index.inverted;
package com.uber.ussi.searchablestructure.utils.inverted;

import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
Expand Down Expand Up @@ -31,10 +31,30 @@ void doesDiscardPopularTermsWhenMaxFractionIsBelowOne() {
.maxNumSearchableStructures(3)
.maxNumSimilarities(100)
.build();
assertTrue(PopularTermDiscardPolicy.doesDiscardPopularTerms(config));
assertTrue(PopularTermDiscardPolicy.doesDiscardPopularTermsFromIndexConfig(config));
assertFalse(PopularTermDiscardPolicy.doesDiscardPopularTerms(1.0));
}

@Test
void maxFractionIdsPerTermReadsIndexAndCacheParamBagsSeparately() {
NamespaceConfig config =
NamespaceConfig.builder()
.minTermsAndValuesLength(0)
.maxTermsAndValuesLength(100)
.maxCacheSize(100)
.cacheType("inverted_term")
.indexType("inverted_term")
.indexParams(Map.of(ConfigKeys.MAX_FRACTION_IDS_PER_TERM, "0.3"))
.cacheParams(Map.of(ConfigKeys.MAX_FRACTION_IDS_PER_TERM, "0.7"))
.comparatorType("jaccard")
.comparatorNormalizerType("complement")
.maxNumSearchableStructures(3)
.maxNumSimilarities(100)
.build();
assertEquals(0.3, PopularTermDiscardPolicy.maxFractionIdsPerTermFromIndexConfig(config));
assertEquals(0.7, PopularTermDiscardPolicy.maxFractionIdsPerTermFromCacheConfig(config));
}

@Test
void discardedTermsOfMarksTermsAboveTheRowFractionThreshold() {
LongObjectHashMap<LongTermsAndValues> rows = new LongObjectHashMap<>();
Expand Down
Loading