sampling_seed = 0 # unused (no sampling), required by parser load_format = "parquet" s3_base_path = "s3://paradedb-benchmarks/datasets/cohere" tables = [] [root_table] name = "cohere_wiki" primary_key = "_id" # used only if sampled later [params] # Only knobs a sweep cannot supply live here. Everything a sweep varies is declared in [sweeps] # alone; fixed non-recall settings are inlined as literals in the query SQL. # lists / vchord_lists -- CREATE INDEX parameters. A sweep measures many operating points # against one already-built index, so a build-time knob cannot be swept # without rebuilding per value. # max_scan_tuples_1pct -- bounds hnsw's filtered scan (not a recall knob) and scales with size. lists = "sqrt({{ dataset_size }})::int" vchord_lists = "CASE WHEN {{ dataset_size }} < 2000000 THEN 2000 WHEN {{ dataset_size }} < 100000000 THEN 10000 ELSE 80000 END" max_scan_tuples_1pct = "CASE {{ dataset_size }} WHEN 1000000 THEN 500000 WHEN 10000000 THEN 5000000 ELSE 200000 END" # Operating-point sweeps. Each arm's knob is measured across these values and the points that # reach 90/95/99% recall are the ones benchmarked -- so no value here needs to be "the" tuned # setting, and none needs a per-size CASE: the sweep picks a different point at 1m than at 10m. # # The ladders are ~1.3x geometric, subdividing every interval the first full run showed a target # falling into, so a selected point lands near its target rather than well past it. # pgvector ivfflat: probes (cells scanned). Extends past 640 -- 1% at 10m only reached 0.989 there. [sweeps.ivfflat.knn_top10_unfiltered] param = "probes_unfiltered" values = [ "10", "15", "22", "30", "45", "60", "80", "110", "145", "190", "250", "330", "440", "580", "780", "1000", ] [sweeps.ivfflat.knn_top10_10pct] param = "probes_10pct" values = [ "10", "15", "22", "30", "45", "60", "80", "110", "145", "190", "250", "330", "440", "580", "780", "1000", ] [sweeps.ivfflat.knn_top10_1pct] param = "probes_1pct" values = [ "10", "15", "22", "30", "45", "60", "80", "110", "145", "190", "250", "330", "440", "580", "780", "1000", ] # pgvector hnsw: ef_search (candidate queue size). pgvector caps this at 1000, so the ladder stops # there -- an arm that cannot reach a target within the cap is reported flagged rather than faked. [sweeps.hnsw.knn_top10_unfiltered] param = "ef_search_unfiltered" values = [ "20", "35", "55", "80", "110", "150", "200", "270", "350", "450", "580", "750", "1000", ] [sweeps.hnsw.knn_top10_10pct] param = "ef_search_10pct" values = [ "20", "35", "55", "80", "110", "150", "200", "270", "350", "450", "580", "750", "1000", ] [sweeps.hnsw.knn_top10_1pct] param = "ef_search_1pct" values = [ "20", "35", "55", "80", "110", "150", "200", "270", "350", "450", "580", "750", "1000", ] # VectorChord: probes (lists scanned). epsilon stays at the docs default. [sweeps.vchord.knn_top10_unfiltered] param = "vchord_probes_unfiltered" values = [ "20", "30", "45", "65", "90", "125", "170", "230", "310", "420", "570", "780", "1050", "1450", "2000", ] [sweeps.vchord.knn_top10_10pct_prefilter_off] param = "vchord_probes_10pct_off" values = [ "20", "30", "45", "65", "90", "125", "170", "230", "310", "420", "570", "780", "1050", "1450", "2000", ] [sweeps.vchord.knn_top10_10pct_prefilter_on] param = "vchord_probes_10pct_on" values = [ "20", "30", "45", "65", "90", "125", "170", "230", "310", "420", "570", "780", "1050", "1450", "2000", ] [sweeps.vchord.knn_top10_1pct_prefilter_off] param = "vchord_probes_1pct_off" values = [ "20", "30", "45", "65", "90", "125", "170", "230", "310", "420", "570", "780", "1050", "1450", "2000", ] [sweeps.vchord.knn_top10_1pct_prefilter_on] param = "vchord_probes_1pct_on" values = [ "20", "30", "45", "65", "90", "125", "170", "230", "310", "420", "570", "780", "1050", "1450", "2000", ] # pg_search: the work-budget ceiling (the bounds gate skips provably-useless # clusters within it, so the ceiling is the recall/latency knob). 1.0 allows # an exhaustive scan. [sweeps.pg_search.knn_top10_unfiltered] param = "pg_search_max_probe_unfiltered" values = [ "0.001", "0.002", "0.003", "0.005", "0.0075", "0.01", "0.015", "0.02", "0.03", "0.05", "0.08", "0.12", "0.2", "0.35", "0.5", "0.75", "1.0", ] [sweeps.pg_search.knn_top10_10pct] param = "pg_search_max_probe_10pct" values = [ "0.001", "0.002", "0.003", "0.005", "0.0075", "0.01", "0.015", "0.02", "0.03", "0.05", "0.08", "0.12", "0.2", "0.35", "0.5", "0.75", "1.0", ] [sweeps.pg_search.knn_top10_1pct] param = "pg_search_max_probe_1pct" values = [ "0.001", "0.002", "0.003", "0.005", "0.0075", "0.01", "0.015", "0.02", "0.03", "0.05", "0.08", "0.12", "0.2", "0.35", "0.5", "0.75", "1.0", ] # pgvectorscale StreamingDiskANN on the default SBQ-compressed build. Each arm sweeps whichever GUC # is its binding lever, and every ladder ends at that GUC's hard maximum -- so when a target is out # of reach the sweep's fallback reports the best achievable point (flagged) rather than a fabricated # one. Both maxima are from the extension's own GUC registration in 0.9.0: # diskann.query_rescore default 50, max 1000 # diskann.query_search_list_size default 100, max 10000 # # Unfiltered sweeps query_rescore: the graph is 1-bit quantized, so recall tracks how many # candidates get exact-distance rescoring, not beam width. It clears 95% well before the cap. # # The filtered arms cannot reach 90% on this build. Filtering is post-filter streaming, so only the # ~10%/1% of the beam that passes the predicate survives, and at most 1000 candidates are ever # exact-rescored. With rescore already pinned at that maximum in the query SQL, beam width is the # only lever left, so those two sweep query_search_list_size and report where they land. # # Their ladders stop short of the GUC's 10000 maximum on purpose. Recall plateaus far below it -- # measured around 0.88 at 10% and 0.67 at 1% -- while cost keeps climbing, so the top few values are # the most expensive points in the whole run and buy no recall. They end just past the plateau, # which is enough to show the curve flattening. Uncompressed `storage_layout = plain` would lift # recall, but is far too slow to build in CI. This arm is therefore NOT iso-recall with the others # -- that is the property being surfaced. [sweeps.pgvectorscale.knn_top10_unfiltered] param = "pgvectorscale_rescore_unfiltered" values = [ "50", "70", "90", "115", "145", "180", "230", "290", "370", "470", "600", "760", "1000", ] [sweeps.pgvectorscale.knn_top10_10pct] param = "pgvectorscale_search_list_size_10pct" values = [ "200", "300", "450", "650", "900", "1300", "1800", "2500", "3500", "5000", ] [sweeps.pgvectorscale.knn_top10_1pct] param = "pgvectorscale_search_list_size_1pct" values = ["500", "750", "1100", "1600", "2200", "3000", "4000", "5500"]