From 391da9b355b128a0ada7f124c7828b0fafc582b2 Mon Sep 17 00:00:00 2001 From: Jimmy Angelakos Date: Mon, 28 Sep 2026 19:29:31 +0100 Subject: [PATCH] fix: a retrain dropped rows from probed vector searches --- ci/journey.sh | 128 ++++--- docs/architecture_decoupled.md | 6 +- docs/architecture_vectors.md | 62 ++-- docs/changelog.md | 7 +- docs/usage_vectors.md | 35 +- extension/coldfront/Makefile | 2 +- extension/coldfront/coldfront--1.0.sql | 335 +++++++++--------- .../test/expected/adopt_iceberg_table.out | 12 +- .../coldfront/test/expected/vector_assign.out | 44 --- .../test/expected/vector_multicolumn.out | 9 + .../test/sql/adopt_iceberg_table.sql | 5 +- .../coldfront/test/sql/vector_assign.sql | 41 --- .../coldfront/test/sql/vector_multicolumn.sql | 5 + 13 files changed, 317 insertions(+), 374 deletions(-) delete mode 100644 extension/coldfront/test/expected/vector_assign.out delete mode 100644 extension/coldfront/test/sql/vector_assign.sql diff --git a/ci/journey.sh b/ci/journey.sh index 9534de2..830722c 100755 --- a/ci/journey.sh +++ b/ci/journey.sh @@ -137,6 +137,19 @@ ice_files() { q "$HOST" "SELECT coldfront.ensure_attached(); SELECT r['n'] FROM duckdb.query('SELECT count(*) AS n FROM iceberg_metadata(''$1'') WHERE status <> ''DELETED'' AND content NOT LIKE ''%DELETES'' AND regexp_matches(file_path, ''$2'')') AS t(r);" | tail -1 } +# ice_deletes : live delete files, counted the same way. +ice_deletes() { + q "$HOST" "SELECT coldfront.ensure_attached(); SELECT r['n'] FROM duckdb.query('SELECT count(*) AS n FROM iceberg_metadata(''$1'') WHERE status <> ''DELETED'' AND content LIKE ''%DELETES''') AS t(r);" | tail -1 +} + +# vec_misplaced : cold rows whose cluster is not the nearest live +# centroid, scored in PostgreSQL with the pgvector operator, independently of the +# generator the write paths use. A row with no embedding is skipped; a row with +# no assignment counts. Zero is the invariant every probe relies on. +vec_misplaced() { + q "$HOST" "SELECT coldfront.ensure_attached(); SELECT count(*) FROM iceberg_scan('ice.public.$1') r WHERE r['$2'] IS NOT NULL AND r['_cf_vec_list_$2']::int IS DISTINCT FROM (SELECT c.centroid_id FROM coldfront.vector_centroids c WHERE c.table_name = '$1' AND c.column_name = '$2' AND c.generation = (SELECT generation FROM coldfront.vector_config WHERE table_name = '$1' AND column_name = '$2') ORDER BY c.centroid <=> r['$2']::real[] LIMIT 1);" | tail -1 +} + # ice_spec : the table's partition spec as transform:column terms in # field order (e.g. "identity:region,month:ts"), empty for an unpartitioned # table or one that has no data yet. @@ -1138,6 +1151,12 @@ EOSQL ) assert_eq "training clamps nprobe to the trained count" "2/2" "$(extract CLAMP "$CL")" + # A row another engine appends straight to Iceberg carries no assignment. One + # such row, written under the claim as a foreign committer would and holding + # coldins's vector, is what the null arm is asserted with below. + local fts; fts=$(q "$HOST" "SELECT (date_trunc('month',now()) - interval '4 months' + interval '25 days')::text;") + q "$HOST" "SET duckdb.unsafe_allow_mixed_transactions = on; SELECT coldfront.ensure_attached(); SELECT coldfront._exec_iceberg_with_claim('\"ice\".\"public\".\"chunks\"', 'INSERT INTO \"ice\".\"public\".\"chunks\" (id, ts, body, embedding) VALUES (990001, TIMESTAMPTZ ''$fts'', ''foreign'', [4.0, 5.5, -6.0]::FLOAT[])');" >/dev/null + # With a generation live, every path that writes a vector into the cold tier # stamps the cluster in the same statement. Two paths, one vector, read back # from Iceberg: the view cannot project the column, by design. @@ -1171,15 +1190,17 @@ EOSQL ) assert_eq "the stamped cluster is the nearest centroid" "$(extract EXPECT "$E")" "$(extract ACTUAL "$E")" - # Rows archived before any training stay unassigned, which a probe reads - # through the null arm of its predicate rather than missing. + # Rows archived before any training are assigned by the training itself, to + # the centroid nearest them, the same as a row written after it. local U; U=$(qf "$HOST" <<'EOSQL' SELECT coldfront.ensure_attached(); -SELECT 'PRE:' || count(*) FROM iceberg_scan('ice.public.chunks') r - WHERE r['body'] = 'cold' AND r['_cf_vec_list_embedding'] IS NULL; +SELECT 'EXPECT:' || (SELECT centroid_id FROM coldfront.vector_centroids + WHERE table_name = 'chunks' AND generation = (SELECT generation FROM coldfront.vector_config WHERE table_name = 'chunks') + ORDER BY centroid <=> ARRAY[1.5,-2,3]::real[] LIMIT 1); +SELECT 'PRE:' || min(r['_cf_vec_list_embedding']::int) FROM iceberg_scan('ice.public.chunks') r WHERE r['body'] = 'cold'; EOSQL ) - assert_eq "a row written before training stays unassigned" "1" "$(extract PRE "$U")" + assert_eq "a row archived before training is assigned to its nearest centroid" "$(extract EXPECT "$U")" "$(extract PRE "$U")" # A cold UPDATE that sets a new embedding must re-derive the cluster in the # same statement. The row is found by its new cluster and no longer by its @@ -1244,7 +1265,7 @@ EOSQL ONE_ROWS=$(q "$HOST" "SET coldfront.vector_nprobe = 1; $topk LIMIT 100;" | grep -vx SET) assert_eq "a probe reads the cluster it aimed at" "1" "$(printf '%s\n' "$ONE_ROWS" | grep -c '^asg_trig$' || true)" assert_eq "a probe skips the cluster it did not look in" "0" "$(printf '%s\n' "$ONE_ROWS" | grep -c '^asg_bulk1$' || true)" - assert_eq "a probe keeps every unassigned row" "1" "$(printf '%s\n' "$ONE_ROWS" | grep -c '^coldins$' || true)" + assert_eq "a probe keeps every unassigned row" "1" "$(printf '%s\n' "$ONE_ROWS" | grep -c '^foreign$' || true)" assert_eq "a probe keeps every hot row" "1" "$(printf '%s\n' "$ONE_ROWS" | grep -c '^hot$' || true)" # The predicate reaches the Iceberg scan itself rather than filtering above it, @@ -1256,12 +1277,11 @@ EOSQL assert_eq "a declined shape carries no predicate" "0" "$(printf '%s\n' "$PLAN_PLAIN" | grep -c '_cf_vec_list_embedding' || true)" # The null arm carrying its weight, stated as an answer rather than a count: - # coldins holds this exact vector and was written before there was a generation, - # so a probe returns it only through that disjunct. Asserted as membership in the - # top few rather than as the single winner: asg_trig carries a centroid, and a - # centroid of a cluster with one member IS that member's vector, so the two can - # tie at distance zero and either is then a correct answer to LIMIT 1. - local NEAR; NEAR=$(q "$HOST" "SET coldfront.vector_nprobe = 1; SELECT body FROM chunks ORDER BY embedding <=> ARRAY[4,5.5,-6]::real[] LIMIT 3;" | grep -vx SET | grep -c '^coldins$' || true) + # the foreign row holds this exact vector and no assignment, so a probe returns + # it only through that disjunct. Asserted as membership in the top few rather + # than as the single winner: coldins holds the same vector, so the two tie at + # distance zero and either is a correct answer to LIMIT 1. + local NEAR; NEAR=$(q "$HOST" "SET coldfront.vector_nprobe = 1; SELECT body FROM chunks ORDER BY embedding <=> ARRAY[4,5.5,-6]::real[] LIMIT 3;" | grep -vx SET | grep -c '^foreign$' || true) assert_eq "a probed search still finds an unassigned row" "1" "$NEAR" # Grouping, aggregates, windows and DISTINCT ride the same narrowed scan: the @@ -1333,39 +1353,7 @@ EOSQL assert_eq "clusters under one row group are counted" "true" "$(extract FLOOR "$S")" assert_eq "a fixture this small draws advice" "true" "$(extract ADVICE "$S")" - # Assigning the rows that predate training. These are the rows vector_status - # just counted as unassigned, and they are read by every probe whatever clusters - # it looks in, so on a corpus tiered before training they are the entire cost of - # a search. One claimed UPDATE, reusing the SET item a cold UPDATE emits. - local unasg_before; unasg_before=$(extract REPORTS "$S" | cut -d/ -f2) - local A; A=$(qf "$HOST" <<'EOSQL' -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -SELECT coldfront.ensure_attached(); -SELECT 'LEFT:' || count(*) FROM iceberg_scan('ice.public.chunks') r - WHERE r['_cf_vec_list_embedding'] IS NULL; -SELECT 'ROWS:' || count(*) FROM iceberg_scan('ice.public.chunks') r; -SELECT 'MATCHES:' || count(*) FROM iceberg_scan('ice.public.chunks') r - WHERE r['_cf_vec_list_embedding']::int = ( - SELECT centroid_id FROM coldfront.vector_centroids c - WHERE c.table_name = 'chunks' - AND c.generation = (SELECT generation FROM coldfront.vector_config WHERE table_name = 'chunks') - ORDER BY c.centroid <=> r['embedding']::real[] LIMIT 1); -EOSQL -) - assert_gt "there were rows to assign" "0" "$unasg_before" - assert_eq "no cold row is left without a cluster" "0" "$(extract LEFT "$A")" - assert_eq "assigning rewrote rows, it did not add any" "7" "$(extract ROWS "$A")" - assert_eq "every row sits in its nearest cluster" "$(extract ROWS "$A")" "$(extract MATCHES "$A")" - - # Which is the point: with nothing unassigned, a narrow probe stops reading the - # whole table. - local narrowed - narrowed=$(q "$HOST" "SET coldfront.vector_nprobe = 1; $topk LIMIT 100;" | grep -vx SET | grep -c . || true) - assert_gt "a narrow probe now reads less than everything" "$narrowed" "$ALL" - - # Idempotent: nothing to do the second time, and no rewrite to pay for. - local A2; A2=$(q_may "$HOST" "CALL coldfront.vector_assign('public','chunks','embedding');") - assert_contains "a second assign has nothing to do" "already assigned" "$A2" + story_vector_retrain # A NULL embedding through the identity-omitted slow path. The Iceberg schema # declares one cluster column per vector column unconditionally, so the row's @@ -1394,6 +1382,56 @@ EOSQL story_vector_compaction } +# ─────────────────────────────────────────────────────────────────────────── +# Story 5b, retraining. A retrain replaces the centroids and, in the same +# transaction, re-assigns every cold row whose nearest centroid changed, so no +# probe ever reads a row against centroids it was not assigned under. At the +# same nlist the centroids keep their ids and move with their data, so the rows +# that stay put are not rewritten; at a new nlist every row moves. Every probe +# is asserted against the exact scan, which is the answer a probe may narrow +# but never lose. +# ─────────────────────────────────────────────────────────────────────────── +story_vector_retrain() { + step "5b. Retraining re-assigns the rows that moved" + local topk="SELECT body FROM chunks ORDER BY embedding <=> ARRAY[4,5.5,-6]::real[] LIMIT 100" + local live="SELECT generation FROM coldfront.vector_config WHERE table_name = 'chunks'" + local ids="SELECT string_agg(centroid_id::text, ',' ORDER BY centroid_id) FROM coldfront.vector_centroids WHERE table_name = 'chunks' AND generation = ($live)" + local exact; exact=$(q "$HOST" "SET coldfront.vector_probe = off; $topk;" | grep -vx SET | sort | tr '\n' ' ') + local gen; gen=$(q "$HOST" "$live;") + + # TC-195: a smaller nlist. Every row moves, the foreign row included, and an + # exhaustive probe (nprobe = nlist) must still return the exact answer. + local R; R=$(q_may "$HOST" "CALL coldfront.vector_train('public','chunks','embedding', 1);") + assert_contains "TC-195: the retrain wrote one centroid" "trained 1 centroids" "$R" + assert_eq "TC-195: an exhaustive probe after the retrain returns the exact answer" "$exact" \ + "$(q "$HOST" "SET coldfront.vector_nprobe = 1; $topk;" | grep -vx SET | sort | tr '\n' ' ')" + assert_eq "TC-195: every cold row sits in its nearest cluster" "0" "$(vec_misplaced chunks embedding)" + + # TC-196: a different nlist starts afresh, and a narrow probe narrows again. + R=$(q_may "$HOST" "CALL coldfront.vector_train('public','chunks','embedding', 2);") + assert_contains "TC-196: the retrain wrote two centroids" "trained 2 centroids" "$R" + assert_eq "TC-196: an exhaustive probe returns the exact answer" "$exact" \ + "$(q "$HOST" "SET coldfront.vector_nprobe = 2; $topk;" | grep -vx SET | sort | tr '\n' ' ')" + assert_eq "TC-196: every cold row sits in its nearest cluster" "0" "$(vec_misplaced chunks embedding)" + local all one + all=$(q "$HOST" "SET coldfront.vector_nprobe = 2; $topk;" | grep -vx SET | grep -c . || true) + one=$(q "$HOST" "SET coldfront.vector_nprobe = 1; $topk;" | grep -vx SET | grep -c . || true) + assert_gt "TC-196: a narrow probe reads less than everything" "$one" "$all" + + # TC-197: the same nlist keeps each centroid's identity. Nothing has changed + # since the last training, so no row moves and no delete file is written. + local ids_before deletes_before + ids_before=$(q "$HOST" "$ids;") + deletes_before=$(ice_deletes ice.public.chunks) + R=$(q_may "$HOST" "CALL coldfront.vector_train('public','chunks','embedding');") + assert_eq "TC-197: the retrain moved the pointer" "$((gen + 3))" "$(q "$HOST" "$live;")" + assert_eq "TC-197: the centroids kept their ids" "$ids_before" "$(q "$HOST" "$ids;")" + assert_eq "TC-197: no row moved, so no delete file was written" "$deletes_before" "$(ice_deletes ice.public.chunks)" + assert_eq "TC-197: every cold row sits in its nearest cluster" "0" "$(vec_misplaced chunks embedding)" + assert_eq "TC-197: an exhaustive probe returns the exact answer" "$exact" \ + "$(q "$HOST" "SET coldfront.vector_nprobe = 2; $topk;" | grep -vx SET | sort | tr '\n' ' ')" +} + # ─────────────────────────────────────────────────────────────────────────── # Story 5c — Compaction on a clustered table. Each cold write leaves a file # sorted within itself, so what accumulates is a set of files with overlapping diff --git a/docs/architecture_decoupled.md b/docs/architecture_decoupled.md index f2ea522..ad75589 100644 --- a/docs/architecture_decoupled.md +++ b/docs/architecture_decoupled.md @@ -337,9 +337,9 @@ ERROR: coldfront: "public.orders" is adopted read-only HINT: Release it with coldfront.release_iceberg_table() and adopt again with p_writable => true to arm INSERT/UPDATE/DELETE. ``` -Reads never consult the flag. `vector_train()`, `vector_assign()` and -`drop_iceberg_table()` refuse a read-only relation too, since each -rewrites or destroys the Iceberg table. The archiver and +Reads never consult the flag. `vector_train()` and `drop_iceberg_table()` +refuse a read-only relation too, since each rewrites or destroys the Iceberg +table. The archiver and `create_iceberg_table()` set the flag; adoption defaults it to false. ### One relation per Iceberg table diff --git a/docs/architecture_vectors.md b/docs/architecture_vectors.md index c002863..c67b1f1 100644 --- a/docs/architecture_vectors.md +++ b/docs/architecture_vectors.md @@ -128,8 +128,9 @@ spellings of the name. ## Assignment `coldfront._vec_list_expr(schema, table, column, vec_expr)` is the only place a -cluster assignment is defined. Given the text of an expression that yields the -vector as DuckDB sees it, it returns: +cluster assignment is defined for a write; its formula, `_vec_nearest_expr`, is +shared with the assignment at the end of training. Given the text of an +expression that yields the vector as DuckDB sees it, it returns: ```sql (SELECT arg_min(c.centroid_id, list_cosine_distance(c.centroid, )) @@ -223,8 +224,10 @@ Lloyd iterations run as DuckDB statements over a reservoir sample, with fixed `REPEATABLE` seeds so a retrain on unchanged data reproduces the same centroids. The mean recompute unnests the vector against a matching `range` so the two lists advance together, because DuckDB has no `WITH ORDINALITY`. The sample and the -working tables live in `memory.main`, which is session-scoped, so the whole loop -must run in one call. +working tables are DuckDB temporary tables, session-scoped, so the whole loop +must run in one call; temporary rather than `memory.main` because a DuckDB +transaction may write one attached database, the temporary database is exempt, +and the assignment at the end writes `ice`. The centroids return through a temporary heap table. A single `INSERT … SELECT FROM duckdb.query(…)` fails with `DuckDB does not support @@ -235,6 +238,26 @@ Empty clusters do not come back from the mean, so the stored count can be below `nlist`; the actual count is recorded rather than padded. A `vector_config` row must exist first, since it holds the generation pointer the procedure writes. +A retrain at the same `nlist` (a call without `p_nlist`, or with the trained +count) starts the iterations from the live centroids instead of fresh seeds, so +each centroid keeps its id and moves with its data. A first training, or a +changed `nlist`, has no set to start from and seeds with k-means++. + +Training ends with the loop's final step applied to the table rather than the +sample: every cold row is assigned to its nearest centroid of the generation just +written, in one claimed `UPDATE` whose `WHERE` is `cluster IS DISTINCT FROM +nearest`, so only a row whose cluster changed is rewritten and a row with no +cluster counts as changed. It scores against the session's copy of the new +centroids (`temp.main.cf_gen`), because the rows inserted into +`vector_centroids` in this transaction are invisible over `pglocal` until commit; +the formula is `_vec_nearest_expr`, the same one every write path uses. The +pointer and the assignments commit together, so no search ever reads a row +against centroids it was not assigned under. The claim is held from the sample +to the commit, as the compactor holds it across its read and rewrite, so a cold +write cannot land between the two with an assignment against the replaced set. +What the pass leaves behind is a merge-on-read delete per rewritten row, and +those rows in update order; compaction resolves both. + Seeds come from k-means++: one sample row at random, then each next drawn with probability proportional to its squared distance from the nearest seed already chosen, by an exponential race (the minimum of `-ln(u)/w` is a weighted draw, in one @@ -398,33 +421,6 @@ Two session knobs, both `PGC_USERSET`: | `coldfront.vector_probe` | `on` | `off` gives the exact scan a recall measurement compares against | | `coldfront.vector_nprobe` | `0` | `0` uses the column's configured `nprobe`; a value at or above `nlist` is exhaustive | -## Assigning what predates training - -`CALL coldfront.vector_assign(schema, table, column)` gives a cluster to the cold -rows that have none, which are exactly those written before the generation existed. -It is one claimed UPDATE whose SET item is the same generator a cold UPDATE uses -when a caller changes an embedding, applied to the embedding already stored, so it -serialises through the bakery like every other cold write and adds no new way to -compute an assignment. - -It refuses without a live generation rather than reporting success. The lookup would -resolve to NULL for every row, so the table would be rewritten in full and left -exactly as it was. - -Two constraints shape the body, and both bite anything else written here: - -- **A DuckDB transaction may write one attached database.** Staging the row count in - a `memory.main` table would spend this transaction's one database on `memory` and - leave the UPDATE unable to write `ice` at all. -- **`duckdb.query` needs a constant at plan time, not a literal in the source.** - Built through `EXECUTE format(...)` the argument is a constant again, which is how - a dynamic table name is read without staging anything. `vector_status` reads its - distribution the same way. - -What it leaves behind is a merge-on-read delete per rewritten row, and rows in -update order rather than cluster order. Compaction resolves both. The sequence is -train, assign, compact. - ## Reporting `CALL coldfront.vector_status([schema, table])` fills a session-lifetime temporary @@ -437,7 +433,9 @@ because a bare `CALL` is its own transaction. Nothing is staged on the way. DuckDB groups the table by cluster and aggregates that grouping in one query, so a row of scalars crosses back per table, read through -the same `EXECUTE format(...)` form `vector_assign` uses. The work list is a pair of +`EXECUTE format(...)`: `duckdb.query` needs a constant at plan time, not a literal +in the source, and a dynamic table name built into the executed text is one +again. The work list is a pair of key arrays walked by index, because a `FOR` over a query would hold a portal open for its body and pg_duckdb refuses a DuckDB read while one is. `vector_train`'s `memory.main` tables are its algorithm's own state, not a way to move a result diff --git a/docs/changelog.md b/docs/changelog.md index 1876b42..be1c8a4 100644 --- a/docs/changelog.md +++ b/docs/changelog.md @@ -16,8 +16,11 @@ and this project adheres to interface, and writes through the view work unchanged on both tiers. A table may carry several vector columns, with the Iceberg file layout sorted for one of them. -- `coldfront.vector_train` clusters a vector column with k-means. Every cold - write assigns each row to its nearest cluster, the compactor keeps that +- `coldfront.vector_train` clusters a vector column with k-means and assigns + the cold rows to the clusters; a retrain without a new `nlist` keeps each + centroid's identity and rewrites only the rows whose nearest centroid + changed. Every cold write assigns each row to its nearest cluster, the + compactor keeps that layout, and a nearest-neighbour search through the view reads the nearest clusters, `nlist` and `nprobe` being per-column settings, plus the rows that carry no assignment. Grouped, aggregated, windowed and `DISTINCT` diff --git a/docs/usage_vectors.md b/docs/usage_vectors.md index dd90f92..dfbed21 100644 --- a/docs/usage_vectors.md +++ b/docs/usage_vectors.md @@ -106,24 +106,21 @@ CALL coldfront.vector_train('public', 'chunks', 'embedding'); ``` `CALL`, not `SELECT`: this is a procedure. It samples the cold tier, trains -`nlist` centroids, and stores them as a new generation. Every cold write after it -stamps the row's nearest cluster in the same statement, on every write path, and a -retrain does not require regenerating anything. - -**Rows already in the cold tier are not clustered by training.** Training only -writes the centroids; the rows that predate it carry no cluster, and every search -reads all of them. If you are clustering a column that already has cold data, give -those rows a cluster and then compact: - -```sql -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -``` - -That rewrites the unassigned rows in one operation, serialised like any other cold -write. Compacting afterwards is what puts them in cluster order and clears the -delete markers the rewrite leaves; without it the rows are assigned but a search -still reads more of the table than it needs to. On a column clustered before any -data arrives there is nothing to assign and nothing to run. +`nlist` centroids, stores them as a new generation, and assigns every cold row to +its nearest centroid, rewriting the rows whose cluster changed. Every cold write +after it assigns the row in the same statement, on every write path. + +Compact afterwards. The assignment leaves a delete marker per rewritten row and +those rows in rewrite order; compaction is what puts them in cluster order and +clears the markers. Without it the rows are assigned but a search still reads +more of the table than it needs to. + +**Retraining.** Call `vector_train` again when the data has grown or you want a +different `nlist`. Without `p_nlist`, the new centroids start from the current +ones and keep their identities, so only the rows whose nearest centroid changed +are rewritten; with a new `nlist`, every row is. Either way the centroids and +the assignments change together, so a search is never wrong in between; it is +only slower until the next compaction. Choosing `nlist` is a floor rather than a formula: aim for at least one row group's worth of rows per cluster, roughly 2048. Below that, extra clusters stop @@ -233,7 +230,7 @@ holding that number up, and says which: | What it says | What to do | |---|---| | no trained generation | `CALL coldfront.vector_train(...)` | -| over half the rows predate training | `CALL coldfront.vector_assign(...)`, then compact | +| over half the rows have no assignment | rows another engine appended straight to Iceberg carry none; `CALL coldfront.vector_train(...)` assigns them, then compact | | over half the clusters hold less than one row group | retrain with a smaller `nlist` | | the largest clusters hold over 4x the median | expect some queries to be slower than `probe_fraction` suggests; uneven clusters are mostly a property of the embeddings and a retrain rarely changes it | diff --git a/extension/coldfront/Makefile b/extension/coldfront/Makefile index 2c413fd..c0d8601 100644 --- a/extension/coldfront/Makefile +++ b/extension/coldfront/Makefile @@ -28,7 +28,7 @@ REGRESS = load_order update_unregistered_view update_heap_table \ read_date_bin read_json_builders read_param_fold registry_snapshot \ adopt_type_map adopt_read_only adopt_iceberg_table \ cold_write_json_agg duckdb_temp_dir \ - vector_centroids vector_type_map vector_cold_render vector_param_render vector_ops vector_probe vector_status vector_assign vector_multicolumn \ + vector_centroids vector_type_map vector_cold_render vector_param_render vector_ops vector_probe vector_status vector_multicolumn \ partition_clause REGRESS_OPTS = --inputdir=test --outputdir=test diff --git a/extension/coldfront/coldfront--1.0.sql b/extension/coldfront/coldfront--1.0.sql index e48d350..34a135b 100644 --- a/extension/coldfront/coldfront--1.0.sql +++ b/extension/coldfront/coldfront--1.0.sql @@ -1662,13 +1662,94 @@ LANGUAGE sql IMMUTABLE STRICT AS $$ OR lower(trim(p_pg_type)) LIKE 'halfvec(%'; $$; --- Train a centroid set for one vector column and store it as a new generation. +-- k-means++ seeding over the session's sample (temp.main.cf_samp): one seed at +-- random, then each next drawn with probability proportional to its squared +-- distance from the nearest seed already chosen, so the seeds spread out instead +-- of clumping. Lloyd's iterations only move centroids locally, so they cannot +-- repair a clumped start. Leaves the seeds in temp.main.cf_cent as (cid, v). +-- +-- It matters most where dense clumps exist, which is the lower dimensionalities +-- many embedding models produce; at 1024 dimensions distances concentrate and the +-- spread start makes little difference. Seeding is a one-time training cost and +-- nothing a query pays. +-- +-- The draw is an exponential race: for weights w, the minimum of -ln(u)/w is +-- distributed exactly as a weighted draw, in one pass and with no cumulative sum. +-- d is maintained incrementally, folding in only the seed just added, so a round +-- is one pass over the sample rather than one per seed chosen so far. The first +-- seed's draw is fixed, like the sample's, so a retrain on unchanged data +-- reproduces the same centroids. +CREATE OR REPLACE FUNCTION coldfront._vec_seed_kmeanspp(p_nlist int) RETURNS void +LANGUAGE plpgsql AS $$ +DECLARE + i int; +BEGIN + PERFORM duckdb.raw_query( + 'CREATE OR REPLACE TEMP TABLE cf_seed AS ' + 'SELECT 1 AS seq, id, v FROM temp.main.cf_samp ' + 'USING SAMPLE reservoir(1 ROWS) REPEATABLE (7)'); + PERFORM duckdb.raw_query( + 'CREATE OR REPLACE TEMP TABLE cf_seed_d AS ' + 'SELECT s.id, s.v, ' + '(SELECT min(list_cosine_distance(s.v, p.v)) FROM temp.main.cf_seed p) AS d ' + 'FROM temp.main.cf_samp s'); + FOR i IN 2 .. p_nlist LOOP + -- A sample smaller than nlist, or one whose remaining rows all duplicate a + -- seed, leaves d = 0 everywhere and the draw returns nothing. The centroid + -- count the caller records is whatever came back, so that resolves itself. + PERFORM duckdb.raw_query(format( + 'INSERT INTO temp.main.cf_seed ' + 'SELECT %s, id, v FROM temp.main.cf_seed_d ' + 'WHERE d > 0 ORDER BY -ln(random()) / (d * d) LIMIT 1', i)); + -- Keyed on seq, because a sample id says nothing about insertion order. + PERFORM duckdb.raw_query(format( + 'UPDATE temp.main.cf_seed_d SET d = least(d, ' + 'list_cosine_distance(v, (SELECT p.v FROM temp.main.cf_seed p ' + 'WHERE p.seq = %s)))', i)); + END LOOP; + PERFORM duckdb.raw_query( + 'CREATE OR REPLACE TEMP TABLE cf_cent AS ' + 'SELECT row_number() OVER (ORDER BY seq) AS cid, v FROM temp.main.cf_seed'); +END; +$$; + +-- Lloyd's iterations over the session's sample, from the centroids in +-- temp.main.cf_cent and back into it: assign, then recompute each cluster's +-- mean, p_iterations times. The mean unnests the vector against a matching range +-- so the two lists advance together: DuckDB has no WITH ORDINALITY, and position +-- is what makes the average element-wise. Empty clusters do not come back from +-- the mean. +CREATE OR REPLACE FUNCTION coldfront._vec_lloyd(p_iterations int, p_dim int) RETURNS void +LANGUAGE plpgsql AS $$ +DECLARE + i int; +BEGIN + FOR i IN 1 .. p_iterations LOOP + PERFORM duckdb.raw_query( + 'CREATE OR REPLACE TEMP TABLE cf_asg AS ' + 'SELECT s.id, arg_min(c.cid, list_cosine_distance(s.v, c.v)) AS cid ' + 'FROM temp.main.cf_samp s CROSS JOIN temp.main.cf_cent c GROUP BY s.id'); + PERFORM duckdb.raw_query(format( + 'CREATE OR REPLACE TEMP TABLE cf_cent AS ' + 'SELECT cid, list(m ORDER BY i)::FLOAT[] AS v FROM (' + ' SELECT cid, i, avg(x) AS m FROM (' + ' SELECT a.cid, unnest(range(1, %s)) AS i, unnest(s.v) AS x ' + ' FROM temp.main.cf_samp s JOIN temp.main.cf_asg a USING (id))' + ' GROUP BY cid, i) GROUP BY cid', p_dim + 1)); + END LOOP; +END; +$$; + +-- Train a centroid set for one vector column, store it as a new generation, and +-- assign every cold row to its nearest centroid of that generation. -- -- Lloyd iterations over a reservoir sample, run as DuckDB statements: both the -- assignment and the mean are distance work over the cold corpus, which is what -- DuckDB is for and what pulling the vectors into PostgreSQL would waste. The -- sample and the working tables live in the session's own DuckDB instance, so the --- whole loop has to run in one call. +-- whole loop has to run in one call. They are temporary tables rather than tables +-- in memory.main: a DuckDB transaction may write one attached database, the +-- temporary database is exempt, and the assignment at the end writes ice. -- -- A PROCEDURE, not a function, and that is a hard requirement rather than a -- preference: pg_duckdb refuses to execute a DuckDB query inside a function @@ -1677,9 +1758,18 @@ $$; -- DuckDB channel with no access to PostgreSQL tables, so a function can move data -- in neither direction. CALL is what makes reading the result possible. -- +-- A retrain at the same nlist starts the iterations from the live centroids, so a +-- centroid keeps its id and moves with its data, and the assignment rewrites only +-- the rows whose nearest centroid changed. A first training, or a changed nlist, +-- has no set to start from and seeds with k-means++. +-- -- Empty clusters simply do not come back from the mean, so the stored count can be -- below p_nlist; it is recorded rather than padded, since a centroid nothing was -- assigned to routes nothing. +-- +-- The table's claim is held from the sample to the commit, as the compactor holds +-- it across its read and rewrite: a cold write landing in between would carry an +-- assignment against the centroids this call replaces. CREATE PROCEDURE coldfront.vector_train( p_schema text, p_table text, @@ -1690,13 +1780,15 @@ CREATE PROCEDURE coldfront.vector_train( ) LANGUAGE plpgsql AS $$ DECLARE - v_ice text; - v_writable boolean; - v_nlist int; - v_gen int; - v_n int; - v_dim int; - i int; + v_ice text; + v_writable boolean; + v_nlist int; + v_cfg_nlist int; + v_gen int; + v_n int; + v_dim int; + v_col text := quote_ident(coldfront._vec_list_col(p_column)); + v_nearest text := coldfront._vec_nearest_expr('temp.main.cf_gen c', quote_ident(p_column)); BEGIN PERFORM coldfront._reject_on_standby('train vector centroids'); @@ -1709,10 +1801,10 @@ BEGIN END IF; IF NOT v_writable THEN RAISE EXCEPTION 'coldfront.vector_train: "%.%" is adopted read-only', p_schema, p_table - USING HINT = 'Training stores a centroid generation the assignment then writes into the cold table. Release it and adopt again with p_writable => true.'; + USING HINT = 'Training rewrites the cold rows. Release it and adopt again with p_writable => true.'; END IF; - SELECT COALESCE(p_nlist, nlist), generation INTO v_nlist, v_gen + SELECT COALESCE(p_nlist, nlist), nlist, generation INTO v_nlist, v_cfg_nlist, v_gen FROM coldfront.vector_config WHERE schema_name = p_schema AND table_name = p_table AND column_name = p_column; IF v_nlist IS NULL THEN @@ -1724,80 +1816,45 @@ BEGIN SET LOCAL duckdb.unsafe_allow_mixed_transactions = on; PERFORM coldfront.ensure_attached(); + PERFORM coldfront._take_iceberg_claim(v_ice); - -- The sample, then the seeds drawn from it. Both seeds are fixed so a retrain - -- on unchanged data reproduces the same centroids. + -- The sample. Its seed is fixed so a retrain on unchanged data reproduces the + -- same centroids. PERFORM duckdb.raw_query(format( - 'CREATE OR REPLACE TABLE memory.main.cf_samp AS ' + 'CREATE OR REPLACE TEMP TABLE cf_samp AS ' 'SELECT row_number() OVER () AS id, v FROM (' ' SELECT %I AS v FROM %s WHERE %I IS NOT NULL ' ' USING SAMPLE reservoir(%s ROWS) REPEATABLE (42))', p_column, v_ice, p_column, p_sample)); SELECT r['n']::int, r['d']::int INTO v_n, v_dim - FROM duckdb.query('SELECT count(*) AS n, max(len(v)) AS d FROM memory.main.cf_samp') AS t(r); + FROM duckdb.query('SELECT count(*) AS n, max(len(v)) AS d FROM temp.main.cf_samp') AS t(r); IF COALESCE(v_n, 0) = 0 THEN RAISE EXCEPTION 'coldfront.vector_train: "%.%"."%" has no cold rows to train on', p_schema, p_table, p_column; END IF; - -- k-means++ seeding: one seed at random, then each next drawn with probability - -- proportional to its squared distance from the nearest seed already chosen, so - -- the seeds spread out instead of clumping. Lloyd's iterations below only move - -- centroids locally, so they cannot repair a clumped start. - -- - -- It matters most where dense clumps exist, which is the lower dimensionalities - -- many embedding models produce; at 1024 dimensions distances concentrate and the - -- spread start makes little difference. Seeding is a one-time training cost and - -- nothing a query pays. - -- - -- The draw is an exponential race: for weights w, the minimum of -ln(u)/w is - -- distributed exactly as a weighted draw, in one pass and with no cumulative sum. - -- d is maintained incrementally, folding in only the seed just added, so a round - -- is one pass over the sample rather than one per seed chosen so far. - PERFORM duckdb.raw_query( - 'CREATE OR REPLACE TABLE memory.main.cf_seed AS ' - 'SELECT 1 AS seq, id, v FROM memory.main.cf_samp ' - 'USING SAMPLE reservoir(1 ROWS) REPEATABLE (7)'); - PERFORM duckdb.raw_query( - 'CREATE OR REPLACE TABLE memory.main.cf_seed_d AS ' - 'SELECT s.id, s.v, ' - '(SELECT min(list_cosine_distance(s.v, p.v)) FROM memory.main.cf_seed p) AS d ' - 'FROM memory.main.cf_samp s'); - FOR i IN 2 .. v_nlist LOOP - -- A sample smaller than nlist, or one whose remaining rows all duplicate a - -- seed, leaves d = 0 everywhere and the draw returns nothing. The centroid - -- count recorded below is whatever came back, so that resolves itself. - PERFORM duckdb.raw_query(format( - 'INSERT INTO memory.main.cf_seed ' - 'SELECT %s, id, v FROM memory.main.cf_seed_d ' - 'WHERE d > 0 ORDER BY -ln(random()) / (d * d) LIMIT 1', i)); - -- Keyed on seq, because a sample id says nothing about insertion order. + IF v_gen > 0 AND v_nlist = v_cfg_nlist THEN + -- The live centroids are the seeds: same ids, and the iterations then + -- move each one with its data. + PERFORM coldfront.ensure_pg_attached(); PERFORM duckdb.raw_query(format( - 'UPDATE memory.main.cf_seed_d SET d = least(d, ' - 'list_cosine_distance(v, (SELECT p.v FROM memory.main.cf_seed p ' - 'WHERE p.seq = %s)))', i)); - END LOOP; - PERFORM duckdb.raw_query( - 'CREATE OR REPLACE TABLE memory.main.cf_cent AS ' - 'SELECT row_number() OVER (ORDER BY seq) AS cid, v FROM memory.main.cf_seed'); + 'CREATE OR REPLACE TEMP TABLE cf_cent AS ' + 'SELECT centroid_id AS cid, centroid AS v FROM pglocal.coldfront.vector_centroids ' + 'WHERE schema_name = %L AND table_name = %L AND column_name = %L AND generation = %s', + p_schema, p_table, p_column, v_gen)); + ELSE + PERFORM coldfront._vec_seed_kmeanspp(v_nlist); + END IF; + PERFORM coldfront._vec_lloyd(p_iterations, v_dim); - -- Assign, then recompute each cluster's mean. The mean unnests the vector - -- against a matching range so the two lists advance together: DuckDB has no - -- WITH ORDINALITY, and position is what makes the average element-wise. - FOR i IN 1 .. p_iterations LOOP - PERFORM duckdb.raw_query( - 'CREATE OR REPLACE TABLE memory.main.cf_asg AS ' - 'SELECT s.id, arg_min(c.cid, list_cosine_distance(s.v, c.v)) AS cid ' - 'FROM memory.main.cf_samp s CROSS JOIN memory.main.cf_cent c GROUP BY s.id'); - PERFORM duckdb.raw_query(format( - 'CREATE OR REPLACE TABLE memory.main.cf_cent AS ' - 'SELECT cid, list(m ORDER BY i)::FLOAT[] AS v FROM (' - ' SELECT cid, i, avg(x) AS m FROM (' - ' SELECT a.cid, unnest(range(1, %s)) AS i, unnest(s.v) AS x ' - ' FROM memory.main.cf_samp s JOIN memory.main.cf_asg a USING (id))' - ' GROUP BY cid, i) GROUP BY cid', v_dim + 1)); - END LOOP; + -- The generation as it is stored, in the columns the nearest-centroid + -- expression reads. It is what the assignment below scores against: the set + -- written to PostgreSQL in this transaction, which a read over pglocal could + -- not see yet. + PERFORM duckdb.raw_query( + 'CREATE OR REPLACE TEMP TABLE cf_gen AS ' + 'SELECT cid::INTEGER AS centroid_id, v AS centroid FROM temp.main.cf_cent'); -- The centroids land in a temporary heap first. A single INSERT reading a -- DuckDB scan is planned as a DuckDB statement, and DuckDB cannot write to a @@ -1808,14 +1865,14 @@ BEGIN DROP TABLE pg_temp.cf_cent_pg; END IF; CREATE TEMP TABLE cf_cent_pg ON COMMIT DROP AS - SELECT r['cid']::int AS cid, r['v']::real[] AS v - FROM duckdb.query('SELECT cid, v FROM memory.main.cf_cent') AS t(r); + SELECT r['centroid_id']::int AS centroid_id, r['centroid']::real[] AS centroid + FROM duckdb.query('SELECT centroid_id, centroid FROM temp.main.cf_gen') AS t(r); -- A generation is immutable, so this writes a new one and moves the pointer. v_gen := COALESCE(v_gen, 0) + 1; INSERT INTO coldfront.vector_centroids (schema_name, table_name, column_name, generation, centroid_id, centroid) - SELECT p_schema, p_table, p_column, v_gen, cid, v FROM cf_cent_pg; + SELECT p_schema, p_table, p_column, v_gen, centroid_id, centroid FROM cf_cent_pg; GET DIAGNOSTICS v_n = ROW_COUNT; -- The trained count can fall below the configured nprobe (empty clusters do @@ -1825,99 +1882,16 @@ BEGIN SET generation = v_gen, nlist = v_n, nprobe = LEAST(nprobe, v_n) WHERE schema_name = p_schema AND table_name = p_table AND column_name = p_column; - RAISE NOTICE 'coldfront: trained % centroids for "%.%"."%" as generation %', - v_n, p_schema, p_table, p_column, v_gen; -END; -$$; - --- Give a cluster to the cold rows that have none. --- --- Training writes centroids and every write after it is assigned in the statement --- that writes it, so the rows without an assignment are exactly those that predate --- the generation. A probe reads all of them whatever clusters it looks in, so on a --- corpus tiered before training they are the whole cost of the search. --- --- One claimed UPDATE, and nothing new: the SET item is the same generator a cold --- UPDATE uses when a caller changes an embedding, applied to the embedding already --- there. Serialised through the bakery like every other cold write, so a concurrent --- writer cannot land a row assigned under a different generation partway through. --- --- What it leaves behind is a merge-on-read delete per rewritten row, and rows in --- the order the update produced rather than in cluster order. Compaction resolves --- both: it applies the deletes and merges the result on the sort key. So the --- sequence is train, assign, compact. --- --- Fails rather than no-ops without a live generation. The lookup would resolve to --- NULL for every row, leaving the table exactly as it was after a full rewrite, and --- a wrong or absent cluster is invisible in a way a probe never reports. -CREATE PROCEDURE coldfront.vector_assign( - p_schema text, - p_table text, - p_column text -) -LANGUAGE plpgsql AS $$ -DECLARE - v_ice text; - v_writable boolean; - v_gen int; - v_item text; - v_col text := quote_ident(coldfront._vec_list_col(p_column)); - v_n bigint; -BEGIN - PERFORM coldfront._reject_on_standby('assign vector clusters'); - - SELECT iceberg_table, is_writable INTO v_ice, v_writable - FROM coldfront.tiered_views - WHERE schema_name = p_schema AND relname = p_table - AND p_column = ANY (COALESCE(vec_columns, '{}')); - IF v_ice IS NULL THEN - RAISE EXCEPTION 'coldfront.vector_assign: "%.%"."%" is not a registered clustered column', - p_schema, p_table, p_column; - END IF; - IF NOT v_writable THEN - RAISE EXCEPTION 'coldfront.vector_assign: "%.%" is adopted read-only', p_schema, p_table - USING HINT = 'Assigning rewrites every cold row. Release it and adopt again with p_writable => true.'; - END IF; - - SELECT NULLIF(generation, 0) INTO v_gen - FROM coldfront.vector_config - WHERE schema_name = p_schema AND table_name = p_table AND column_name = p_column; - IF v_gen IS NULL THEN - RAISE EXCEPTION 'coldfront.vector_assign: "%.%"."%" has no trained generation', - p_schema, p_table, p_column - USING HINT = 'CALL coldfront.vector_train(...) first: without centroids ' - 'every row would be assigned NULL, which is what it already is.'; - END IF; - - v_item := coldfront._vec_list_set_item(v_ice, p_column, quote_ident(p_column)); - - SET LOCAL duckdb.unsafe_allow_mixed_transactions = on; - PERFORM coldfront.ensure_attached(); - - -- Counted before the write, so the notice reports what this call did rather than - -- what the table looks like afterwards. - -- - -- Through EXECUTE, which is what makes a dynamic table name work here: a bare - -- duckdb.query(format(...)) is not a constant at plan time and is refused, while - -- staging the count in a memory.main table the way vector_train stages its - -- centroids would spend this transaction's one database on memory and leave the - -- UPDATE below unable to write ice at all ("a single transaction can only modify - -- one database"). Built as dynamic SQL, the argument is a literal again. - EXECUTE format('SELECT t.r[%L]::bigint FROM duckdb.query(%L) AS t(r)', 'n', - format('SELECT count(*) AS n FROM %s WHERE %s IS NULL', v_ice, v_col)) - INTO v_n; - - IF COALESCE(v_n, 0) = 0 THEN - RAISE NOTICE 'coldfront: every cold row of "%.%"."%" is already assigned', - p_schema, p_table, p_column; - RETURN; - END IF; - + -- The corpus, assigned: the loop's final step applied to the table rather than + -- the sample. Only a row whose cluster changed is rewritten, and a row with no + -- cluster counts as changed. The pointer and the assignments commit together. + -- What it leaves behind is a merge-on-read delete per rewritten row, and those + -- rows in update order rather than cluster order; compaction resolves both. PERFORM coldfront._exec_iceberg_with_claim(v_ice, format( - 'UPDATE %s SET %s WHERE %s IS NULL', v_ice, v_item, v_col)); + 'UPDATE %s SET %s = %s WHERE %s IS DISTINCT FROM %s', + v_ice, v_col, v_nearest, v_col, v_nearest)); - RAISE NOTICE 'coldfront: assigned % cold row(s) of "%.%"."%" to generation %; ' - 'compact the table to apply the deletes and restore cluster order', + RAISE NOTICE 'coldfront: trained % centroids for "%.%"."%" as generation % and assigned the cold rows; compact the table to apply the deletes and restore cluster order', v_n, p_schema, p_table, p_column, v_gen; END; $$; @@ -2041,10 +2015,9 @@ BEGIN -- -- Through EXECUTE because the table name is dynamic and duckdb.query needs a -- constant at plan time, not a literal in the source. Staging the grouping in - -- a memory.main table the way vector_train stages its centroids would work - -- here too, but it is a table to name, drop and read back for a result that - -- fits in one row, and on any path that also writes Iceberg it would spend - -- the transaction's one writable database (see vector_assign). + -- a DuckDB table the way vector_train stages its centroids would work here + -- too, but it is a table to name, drop and read back for a result that fits + -- in one row. EXECUTE format( 'SELECT t.r[%L]::bigint, t.r[%L]::bigint, t.r[%L]::int, ' 't.r[%L]::bigint, t.r[%L]::bigint, t.r[%L]::bigint, ' @@ -2096,7 +2069,7 @@ BEGIN WHEN vt.generation IS NULL THEN 'no trained generation: every row is unassigned and every probe reads the whole table' WHEN v_rows > 0 AND v_unasg::numeric / v_rows > 0.5 THEN - 'over half the rows predate training, and a probe reads all of them' + 'over half the rows have no assignment, and a probe reads all of them' WHEN v_occ > 0 AND v_floor::numeric / v_occ > 0.5 THEN 'over half the occupied clusters hold less than one row group: retrain with a smaller nlist' -- Against the median, not the minimum: one tiny cluster sets max/min @@ -2296,20 +2269,30 @@ $$; -- Before any training the config carries no generation, the inner query matches -- nothing, and the expression yields NULL: unassigned, which a probe reads -- through the null arm of its predicate rather than missing. A retrain cannot --- interleave with a cold write, since optimize() holds the table's claim for its --- duration and every cold write serialises on that same claim. +-- interleave with a cold write, since vector_train holds the table's claim from +-- its sample to its commit and every cold write serialises on that same claim. +-- +-- The formula itself is _vec_nearest_expr, shared with vector_train, which scores +-- the generation it has just trained from its own session: one formula, so a +-- write and the training can never disagree on a cluster. +CREATE OR REPLACE FUNCTION coldfront._vec_nearest_expr(p_from text, p_vec_expr text) +RETURNS text +LANGUAGE sql IMMUTABLE STRICT PARALLEL SAFE AS $$ + SELECT format('(SELECT arg_min(c.centroid_id, list_cosine_distance(c.centroid, %s)) FROM %s)', + p_vec_expr, p_from); +$$; + CREATE OR REPLACE FUNCTION coldfront._vec_list_expr( p_schema text, p_table text, p_column text, p_vec_expr text) RETURNS text LANGUAGE sql IMMUTABLE STRICT PARALLEL SAFE AS $$ - SELECT format( - '(SELECT arg_min(c.centroid_id, list_cosine_distance(c.centroid, %s)) ' - 'FROM pglocal.coldfront.vector_centroids c ' + SELECT coldfront._vec_nearest_expr(format( + 'pglocal.coldfront.vector_centroids c ' 'WHERE c.schema_name = %L AND c.table_name = %L AND c.column_name = %L ' 'AND c.generation = (SELECT vc.generation FROM pglocal.coldfront.vector_config vc ' 'WHERE vc.schema_name = %L AND vc.table_name = %L ' - 'AND vc.column_name = %L))', - p_vec_expr, p_schema, p_table, p_column, p_schema, p_table, p_column); + 'AND vc.column_name = %L)', + p_schema, p_table, p_column, p_schema, p_table, p_column), p_vec_expr); $$; -- The cluster columns for an Iceberg ref, quoted and comma-joined in schema order. diff --git a/extension/coldfront/test/expected/adopt_iceberg_table.out b/extension/coldfront/test/expected/adopt_iceberg_table.out index 6ab3d2c..0bcd19f 100644 --- a/extension/coldfront/test/expected/adopt_iceberg_table.out +++ b/extension/coldfront/test/expected/adopt_iceberg_table.out @@ -287,20 +287,16 @@ CONTEXT: PL/pgSQL function release_iceberg_table(text,text) line 17 at RAISE DROP VIEW public.events; ERROR: coldfront: cannot DROP "public.events": it has a cold tier in Iceberg HINT: Blocked by design: the Iceberg cold tier would be orphaned. Use coldfront.drop_iceberg_table() to remove the table and its cold tier, or coldfront.release_iceberg_table() to hand an adopted table back with its Iceberg table intact. --- Training and assigning both rewrite the cold table, so both refuse a relation --- adopted read-only, ahead of any DuckDB statement. +-- Training rewrites the cold table, so it refuses a relation adopted read-only, +-- ahead of any DuckDB statement. INSERT INTO coldfront.tiered_views(schema_name, relname, iceberg_table, is_iceberg_only, vec_columns, is_writable) VALUES ('public', 'chunks', 'ice.lake.chunks', true, ARRAY['embedding'], false); INSERT INTO coldfront.vector_config (schema_name, table_name, column_name, nlist, nprobe, generation) VALUES ('public', 'chunks', 'embedding', 2, 1, 1); CALL coldfront.vector_train('public', 'chunks', 'embedding'); ERROR: coldfront.vector_train: "public.chunks" is adopted read-only -HINT: Training stores a centroid generation the assignment then writes into the cold table. Release it and adopt again with p_writable => true. -CONTEXT: PL/pgSQL function vector_train(text,text,text,integer,integer,integer) line 21 at RAISE -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -ERROR: coldfront.vector_assign: "public.chunks" is adopted read-only -HINT: Assigning rewrites every cold row. Release it and adopt again with p_writable => true. -CONTEXT: PL/pgSQL function vector_assign(text,text,text) line 21 at RAISE +HINT: Training rewrites the cold rows. Release it and adopt again with p_writable => true. +CONTEXT: PL/pgSQL function vector_train(text,text,text,integer,integer,integer) line 23 at RAISE -- Cleanup. DELETE FROM coldfront.vector_config WHERE table_name = 'chunks'; DELETE FROM coldfront.tiered_views; diff --git a/extension/coldfront/test/expected/vector_assign.out b/extension/coldfront/test/expected/vector_assign.out deleted file mode 100644 index 302a5c8..0000000 --- a/extension/coldfront/test/expected/vector_assign.out +++ /dev/null @@ -1,44 +0,0 @@ --- vector_assign refuses before it writes. Both checks run ahead of any DuckDB --- statement, which is what lets pg_regress reach them with no Iceberg attached; the --- assignment itself is asserted in ci/journey.sh against a real cold tier. --- --- Suppress the run-order-dependent "already exists" NOTICE: in the shared regress --- db an earlier test may have created the extensions, standalone not. -SET client_min_messages = warning; -CREATE EXTENSION IF NOT EXISTS pg_duckdb; -CREATE EXTENSION IF NOT EXISTS coldfront; -CREATE EXTENSION IF NOT EXISTS vector; -RESET client_min_messages; -SET coldfront.warehouse = ''; -SET coldfront.lakekeeper_endpoint = ''; --- A column nothing registered. Naming the caller's own arguments back is what makes --- a typo in a scripted call readable. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -ERROR: coldfront.vector_assign: "public.chunks"."embedding" is not a registered clustered column -CONTEXT: PL/pgSQL function vector_assign(text,text,text) line 17 at RAISE -CREATE TABLE public._chunks (id bigint, ts timestamptz, embedding vector(3)); -INSERT INTO coldfront.tiered_views(schema_name, relname, hot_table, iceberg_table, partition_col, vec_columns) -VALUES ('public', 'chunks', 'public._chunks', 'ice.default.chunks', 'ts', ARRAY['embedding']); --- Registered, but the column named is not the clustered one. -CALL coldfront.vector_assign('public', 'chunks', 'other'); -ERROR: coldfront.vector_assign: "public.chunks"."other" is not a registered clustered column -CONTEXT: PL/pgSQL function vector_assign(text,text,text) line 17 at RAISE --- Registered and clustered, with nothing trained. Assigning here would rewrite --- every row to the NULL it already holds, so it refuses rather than reporting --- success over a full rewrite that changed nothing. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -ERROR: coldfront.vector_assign: "public.chunks"."embedding" has no trained generation -HINT: CALL coldfront.vector_train(...) first: without centroids every row would be assigned NULL, which is what it already is. -CONTEXT: PL/pgSQL function vector_assign(text,text,text) line 29 at RAISE -INSERT INTO coldfront.vector_config (schema_name, table_name, column_name, nlist, nprobe, generation) -VALUES ('public', 'chunks', 'embedding', 2, 1, 0); --- Generation 0 is the same case spelled differently: a configuration row exists but --- no generation was ever written. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -ERROR: coldfront.vector_assign: "public.chunks"."embedding" has no trained generation -HINT: CALL coldfront.vector_train(...) first: without centroids every row would be assigned NULL, which is what it already is. -CONTEXT: PL/pgSQL function vector_assign(text,text,text) line 29 at RAISE --- Cleanup. -DELETE FROM coldfront.vector_config WHERE table_name = 'chunks'; -DELETE FROM coldfront.tiered_views WHERE relname = 'chunks'; -DROP TABLE public._chunks; diff --git a/extension/coldfront/test/expected/vector_multicolumn.out b/extension/coldfront/test/expected/vector_multicolumn.out index d759d0c..794bea7 100644 --- a/extension/coldfront/test/expected/vector_multicolumn.out +++ b/extension/coldfront/test/expected/vector_multicolumn.out @@ -141,6 +141,15 @@ SELECT coldfront._vec_list_set_item('ice.default.docs', 'embedding', 'NEW_EXPR') t | t (1 row) +-- One formula for the nearest centroid whatever set it is scored against: the live +-- generation over pglocal for a cold write, the session's freshly trained set for +-- the assignment at the end of training. +SELECT coldfront._vec_nearest_expr('temp.main.cf_gen c', 'embedding'); + _vec_nearest_expr +------------------------------------------------------------------------------------------------------ + (SELECT arg_min(c.centroid_id, list_cosine_distance(c.centroid, embedding)) FROM temp.main.cf_gen c) +(1 row) + -- The trigger function is the contract. Both cluster expressions lead, each -- pairing with its own column's value, in column order; the identity column and -- the user-written generated column are positional NULLs the hot INSERT skips, diff --git a/extension/coldfront/test/sql/adopt_iceberg_table.sql b/extension/coldfront/test/sql/adopt_iceberg_table.sql index 928ee54..79c017b 100644 --- a/extension/coldfront/test/sql/adopt_iceberg_table.sql +++ b/extension/coldfront/test/sql/adopt_iceberg_table.sql @@ -159,15 +159,14 @@ SELECT coldfront.release_iceberg_table('public', 'events'); -- Plain DROP VIEW stays blocked, and the hint names both exits. DROP VIEW public.events; --- Training and assigning both rewrite the cold table, so both refuse a relation --- adopted read-only, ahead of any DuckDB statement. +-- Training rewrites the cold table, so it refuses a relation adopted read-only, +-- ahead of any DuckDB statement. INSERT INTO coldfront.tiered_views(schema_name, relname, iceberg_table, is_iceberg_only, vec_columns, is_writable) VALUES ('public', 'chunks', 'ice.lake.chunks', true, ARRAY['embedding'], false); INSERT INTO coldfront.vector_config (schema_name, table_name, column_name, nlist, nprobe, generation) VALUES ('public', 'chunks', 'embedding', 2, 1, 1); CALL coldfront.vector_train('public', 'chunks', 'embedding'); -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); -- Cleanup. DELETE FROM coldfront.vector_config WHERE table_name = 'chunks'; diff --git a/extension/coldfront/test/sql/vector_assign.sql b/extension/coldfront/test/sql/vector_assign.sql deleted file mode 100644 index db14c2d..0000000 --- a/extension/coldfront/test/sql/vector_assign.sql +++ /dev/null @@ -1,41 +0,0 @@ --- vector_assign refuses before it writes. Both checks run ahead of any DuckDB --- statement, which is what lets pg_regress reach them with no Iceberg attached; the --- assignment itself is asserted in ci/journey.sh against a real cold tier. --- --- Suppress the run-order-dependent "already exists" NOTICE: in the shared regress --- db an earlier test may have created the extensions, standalone not. -SET client_min_messages = warning; -CREATE EXTENSION IF NOT EXISTS pg_duckdb; -CREATE EXTENSION IF NOT EXISTS coldfront; -CREATE EXTENSION IF NOT EXISTS vector; -RESET client_min_messages; -SET coldfront.warehouse = ''; -SET coldfront.lakekeeper_endpoint = ''; - --- A column nothing registered. Naming the caller's own arguments back is what makes --- a typo in a scripted call readable. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); - -CREATE TABLE public._chunks (id bigint, ts timestamptz, embedding vector(3)); -INSERT INTO coldfront.tiered_views(schema_name, relname, hot_table, iceberg_table, partition_col, vec_columns) -VALUES ('public', 'chunks', 'public._chunks', 'ice.default.chunks', 'ts', ARRAY['embedding']); - --- Registered, but the column named is not the clustered one. -CALL coldfront.vector_assign('public', 'chunks', 'other'); - --- Registered and clustered, with nothing trained. Assigning here would rewrite --- every row to the NULL it already holds, so it refuses rather than reporting --- success over a full rewrite that changed nothing. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); - -INSERT INTO coldfront.vector_config (schema_name, table_name, column_name, nlist, nprobe, generation) -VALUES ('public', 'chunks', 'embedding', 2, 1, 0); - --- Generation 0 is the same case spelled differently: a configuration row exists but --- no generation was ever written. -CALL coldfront.vector_assign('public', 'chunks', 'embedding'); - --- Cleanup. -DELETE FROM coldfront.vector_config WHERE table_name = 'chunks'; -DELETE FROM coldfront.tiered_views WHERE relname = 'chunks'; -DROP TABLE public._chunks; diff --git a/extension/coldfront/test/sql/vector_multicolumn.sql b/extension/coldfront/test/sql/vector_multicolumn.sql index ce205dd..2966b32 100644 --- a/extension/coldfront/test/sql/vector_multicolumn.sql +++ b/extension/coldfront/test/sql/vector_multicolumn.sql @@ -92,6 +92,11 @@ SELECT coldfront._vec_list_prefix_for_ref('ice.default.docs', 's.') IS NOT NULL SELECT coldfront._vec_list_set_item('ice.default.docs', 'embedding', 'NEW_EXPR') IS NOT NULL AS vector_answers, coldfront._vec_list_set_item('ice.default.docs', 'body', 'NEW_EXPR') IS NULL AS non_vector_declines; +-- One formula for the nearest centroid whatever set it is scored against: the live +-- generation over pglocal for a cold write, the session's freshly trained set for +-- the assignment at the end of training. +SELECT coldfront._vec_nearest_expr('temp.main.cf_gen c', 'embedding'); + -- The trigger function is the contract. Both cluster expressions lead, each -- pairing with its own column's value, in column order; the identity column and -- the user-written generated column are positional NULLs the hot INSERT skips,