-- dict_dir: the cached dictionary-index directory (1.12.0, ROADMAP I7) must find -- exactly the dictionary page the on-disk chain walk finds, for every term: -- first and last in sort order, terms equal to an index page's first term, -- prefixes of indexed terms, absent terms between and beyond, several segments, -- a merge (directory rebuilt in the same backend), and with the cache disabled -- (pg_fts.doclen_cache_mb = 0, the chain walk). SET client_min_messages = warning; SET enable_seqscan = off; SET enable_bitmapscan = off; -- a dictionary index of many pages: each document adds a long term (190 chars), -- so a dictionary page holds ~36 terms and an index page ~40 entries; 20,000 -- documents give ~560 dictionary pages and ~15 index pages, and the chain walk -- reads several index pages for a late term CREATE TABLE dd (id int PRIMARY KEY, d ftsdoc); INSERT INTO dd SELECT g, to_ftsdoc('simple', 't' || lpad(g::text, 6, '0') || ' common ' || CASE WHEN g % 10 = 0 THEN 'tenth ' ELSE '' END || 'l' || lpad(g::text, 6, '0') || repeat('x', 183)) FROM generate_series(1, 30000) g; CREATE INDEX dd_fts ON dd USING fts (d); VACUUM ANALYZE dd; -- ranked top-k ids for a term, under a given doclen_cache_mb CREATE FUNCTION dd_ids(q text, mb int) RETURNS text LANGUAGE plpgsql AS $$ DECLARE r text; BEGIN EXECUTE format('SET LOCAL pg_fts.doclen_cache_mb = %s', mb); EXECUTE format('SELECT coalesce(string_agg(id::text, '','' ORDER BY rn), ''-'') FROM (SELECT id, row_number() OVER () rn FROM (SELECT id FROM dd WHERE d @@@ to_ftsquery(''simple'',%L) ORDER BY d <=> to_ftsquery(''simple'',%L), id LIMIT 5) s) z', q, q) INTO r; RETURN r; END $$; -- every probe: cached directory == chain walk, and the expected hit/miss CREATE TABLE probes (q text, want_hit bool); INSERT INTO probes VALUES ('t000001', true), ('t030000', true), ('t015000', true), ('t000255', true), ('t000256', true), ('t029999', true), ('common', true), ('tenth', true), ('a', false), ('t', false), ('t00000', false), ('t0000011', false), ('t030001', false), ('zzzz', false), ('t015000x', false), ('s999999', false), ('l000001' || repeat('x', 183), true), ('l030000' || repeat('x', 183), true), ('l015000' || repeat('x', 183), true), ('l015000' || repeat('x', 182), false), ('l015000' || repeat('x', 184), false), ('l030001' || repeat('x', 183), false); SELECT CASE WHEN length(q) > 20 THEN left(q, 7) || '+' || (length(q) - 7) || 'x' ELSE q END AS probe, dd_ids(q, 64) = dd_ids(q, 0) AS same, (dd_ids(q, 64) <> '-') = want_hit AS expected FROM probes ORDER BY q; -- every term of the index's first 3,000 documents, in both arms: this covers -- many dictionary-page FIRST terms -- the equal-key case of the directory's -- binary search (largest entry <= term), which a probe set of interior and -- absent terms never reaches. Reported as a count of disagreements. SELECT count(*) FILTER (WHERE dd_ids(q, 64) IS DISTINCT FROM dd_ids(q, 0)) AS disagree, count(*) FILTER (WHERE dd_ids(q, 64) = '-') AS cached_misses, count(*) AS probes FROM (SELECT 't' || lpad(g::text, 6, '0') AS q FROM generate_series(1, 3000) g UNION ALL SELECT 'l' || lpad(g::text, 6, '0') || repeat('x', 183) FROM generate_series(1, 3000) g UNION ALL -- the lowest-sorting terms: the directory's FIRST SELECT 'common' UNION ALL SELECT 'l000001' || repeat('x', 183)) p; -- entry -- the dictionary index really has several pages (the test means nothing otherwise) SELECT (SELECT nterms FROM fts_index_stats('dd_fts')) > 20000 AS many_terms; -- a second segment with new terms; the same backend must see them after the merge -- (the directory is rebuilt when the metapage generation moves) SELECT dd_ids('newseg000001', 64) AS before_insert; INSERT INTO dd SELECT 100000 + g, to_ftsdoc('simple', 'newseg' || lpad(g::text, 6, '0') || ' common') FROM generate_series(1, 5000) g; SELECT fts_merge('dd_fts') IS NOT NULL AS merged; SELECT dd_ids('newseg000001', 64) AS after_merge_hit, dd_ids('newseg000001', 64) = dd_ids('newseg000001', 0) AS same_new, dd_ids('t000001', 64) = dd_ids('t000001', 0) AS same_old, dd_ids('common', 64) = dd_ids('common', 0) AS same_common; -- multiple segments listed at once (pending flushed into its own segment) INSERT INTO dd SELECT 200000 + g, to_ftsdoc('simple', 'third' || g || ' common') FROM generate_series(1, 2000) g; VACUUM dd; SELECT fts_index_nsegments('dd_fts') >= 1 AS segs, dd_ids('third1', 64) = dd_ids('third1', 0) AS same_third, dd_ids('t029999', 64) = dd_ids('t029999', 0) AS same_old2, dd_ids('third1', 64) <> '-' AS third_found; -- the cache is actually used: a ranked lookup reads fewer index buffers with it -- (doclen_cache_mb 64) than with the chain walk (0). Fails if the directory is -- never built or never matched (the lookup silently falls back to the chain). CREATE FUNCTION dd_bufs(q text, mb int) RETURNS int LANGUAGE plpgsql AS $$ DECLARE l text; n int := 0; BEGIN EXECUTE format('SET LOCAL pg_fts.doclen_cache_mb = %s', mb); -- warm: the first ranked scan of a generation builds the chunk EXECUTE format('SELECT count(*) FROM (SELECT id FROM dd WHERE d @@@ to_ftsquery(''simple'',%L) ORDER BY d <=> to_ftsquery(''simple'',%L) LIMIT 5) s', q, q); FOR l IN EXECUTE format('EXPLAIN (ANALYZE, BUFFERS, COSTS OFF, TIMING OFF, SUMMARY OFF) SELECT id FROM dd WHERE d @@@ to_ftsquery(''simple'',%L) ORDER BY d <=> to_ftsquery(''simple'',%L) LIMIT 5', q, q) LOOP IF n = 0 AND l ~ 'Buffers: shared hit=' THEN n := substring(l from 'hit=([0-9]+)')::int; END IF; END LOOP; RETURN n; END $$; SELECT dd_bufs('t029999', 64) + 3 < dd_bufs('t029999', 0) AS cache_reads_fewer, dd_bufs('t029999', 64) > 0 AS measured; DROP FUNCTION dd_bufs(text, int); DROP TABLE probes; DROP TABLE dd; DROP FUNCTION dd_ids(text, int);