From 149a15b504c3697547b44c4d0efac01b1f56281c Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 11:51:08 +0900 Subject: [PATCH 1/9] Add serialized relation mutation set primitive --- docs/THREADING.md | 12 +- scripts/ci/check-threading-doc.sh | 2 +- tests/meson.build | 10 + tests/test_relation_mutation_set.c | 304 ++++++++++++++++++++++++ wirelog/columnar/internal.h | 82 +++++++ wirelog/columnar/relation.c | 359 +++++++++++++++++++++++++++++ wirelog/columnar/source_access.h | 15 ++ 7 files changed, 779 insertions(+), 5 deletions(-) create mode 100644 tests/test_relation_mutation_set.c diff --git a/docs/THREADING.md b/docs/THREADING.md index fb637fa8..742ff6ca 100644 --- a/docs/THREADING.md +++ b/docs/THREADING.md @@ -465,7 +465,7 @@ measured by `bench/bench_intern.c`; baselines are in `docs/INTERN_PERF.md` 21 + 4 + 5 + 19 + 1 + 1 + 1 + 37 + 5 + 7 + 3 = **104 atomic call sites** before the source-access contract below. -### 5.13 `wirelog/columnar/source_access.h` — relation source gate (20 rows) +### 5.13 `wirelog/columnar/source_access.h` — relation source gate (21 rows) This header-only gate protects relation descriptors and canonical source storage. It is linked into the production relation lifecycle and its gate state is zero-initialized. @@ -509,11 +509,12 @@ those slots and arena allocations are quiescent. | `source_access.h:wl_columnar_source_access_cohort_restore#2` | `cohort->source->state` | `atomic_compare_exchange_weak_explicit` | release/relaxed | Republish the source reader count after all descriptors are restored | | `source_access.h:wl_columnar_source_access_reader_release` | `gate->state` | `atomic_load_explicit` | acquire | Observe the active gate before releasing this reader | | `source_access.h:wl_columnar_source_access_writer_acquire` | `gate->state` | `atomic_compare_exchange_weak_explicit` | acquire/relaxed | Linearize exclusive writer admission and retry spurious failure | +| `source_access.h:wl_columnar_source_access_writer_validate` | `gate->state` | `atomic_load_explicit` | acquire | Validate that the address-bound, same-thread token still owns the exact writer sentinel before authorizing mutation without consuming the token | | `source_access.h:wl_columnar_source_access_writer_release` | `gate->state` | `atomic_load_explicit` | acquire | Validate the writer state before terminal publication | | `source_access.h:wl_columnar_source_access_writer_release#2` | `gate->state` | `atomic_compare_exchange_weak_explicit` | release/relaxed | Publish writer payload completion and retry spurious failure | | `source_access.h:wl_columnar_source_access_writer_move` | `src->owner->state` | `atomic_load_explicit` | acquire | Confirm that the source token still holds WRITER before moving its address-bound ownership to another token | -### 5.14 `wirelog/columnar/relation.c` and `session.c` — alias ownership and pool promotion (25 rows) +### 5.14 `wirelog/columnar/relation.c` and `session.c` — alias ownership and pool promotion (28 rows) The canonical owner's flattened alias count uses `wl_atomic_u64` because a quiesced worker can retire its alias while unrelated readers still hold the @@ -531,6 +532,9 @@ concurrent alias removals cannot underflow the count. | `relation.c:col_rel_storage_alias_borrow_acquire#2` | `storage_alias_borrows` | `atomic_compare_exchange_weak_explicit` | release/relaxed | Publish one new alias borrow without wrapping; retry with the observed value after a lost race | | `relation.c:col_rel_storage_alias_borrow_release` | `storage_alias_borrows` | `atomic_load_explicit` | acquire | Read the candidate count before refusing underflow or attempting retirement | | `relation.c:col_rel_storage_alias_borrow_release#2` | `storage_alias_borrows` | `atomic_compare_exchange_weak_explicit` | acq_rel/acquire | Retire one borrow atomically while pairing peer-visible descriptor publication with removal | +| `relation.c:col_rel_mutation_set_unwind` | `init->relation->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Restore the exact provisional legacy borrow snapshot under the descriptor writer; its later release publishes rollback | +| `relation.c:col_rel_mutation_set_acquire` | `relation->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Initialize provisional legacy ownership under descriptor exclusion; finish publishes commit or restores the snapshot before descriptor release | +| `relation.c:col_rel_storage_alias_release_locked` | `owner->storage_alias_borrows` | `atomic_fetch_sub_explicit` | acq_rel | Retire exactly one validated alias borrow as the final old-owner access; release publishes descriptor rebinding and acquire pairs with published borrow accounting | | `relation.c:col_rel_storage_alias_release` | `alias->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Clear child-borrow metadata while the alias descriptor is exclusively held | | `relation.c:col_rel_destroy_checked` | `r->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Leave an inert pool tombstone with no live alias borrows | | `relation.c:col_rel_destroy_checked#2` | `r->source_access.state` | `atomic_store_explicit` | release | Keep the retired pool slot closed until allocator reset or reuse | @@ -551,7 +555,7 @@ concurrent alias removals cannot underflow the count. | `session.c:session_pool_rel_promote#2` | `src->retained_reservation.owner_bits` | `atomic_load_explicit` | acquire | Promote a committed reservation only when the pool slot still owns it | | `session.c:session_pool_rel_promote#3` | `src->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Leave the closed pool tombstone with no child aliases | -104 + 20 + 25 = **149 atomic call sites**. +104 + 21 + 28 = **153 atomic call sites**. The `#N` suffix counts all atomic sites in a symbol, regardless of operation; the first site remains unsuffixed. `scripts/ci/check-threading-doc.sh` uses @@ -676,7 +680,7 @@ the committed token after publication and before a growth transaction. | `eval_dedup.c:wl_columnar_eval_dedup_test_fail_next_growth_alloc` | test-only fault flag | `atomic_store_explicit` | release | Arm one allocation refusal before a test invokes dedup growth; excluded from the production library | | `eval_dedup.c:wl_columnar_eval_dedup_set_grow` | test-only fault flag | `atomic_exchange_explicit` | acquire-release | Consume the one-shot fault safely when test workers grow dedup tables; excluded from the production library | -The complete source audit now contains **210 atomic call sites**. +The complete source audit now contains **214 atomic call sites**. --- diff --git a/scripts/ci/check-threading-doc.sh b/scripts/ci/check-threading-doc.sh index c693c7e6..8fa49bc7 100755 --- a/scripts/ci/check-threading-doc.sh +++ b/scripts/ci/check-threading-doc.sh @@ -19,7 +19,7 @@ fi rows="$tmp_dir/rows" sed -nE 's/^\| `([^`]+:[A-Za-z_][A-Za-z0-9_]*(#[0-9]+)?)` \| [^|]* \| `([^`]*)` \|.*/\1\t\3/p' "$doc" >"$rows" row_count=$(wc -l <"$rows") -expected_rows="${WIRELOG_THREADING_EXPECTED_ROWS:-210}" +expected_rows="${WIRELOG_THREADING_EXPECTED_ROWS:-214}" [ "$row_count" -eq "$expected_rows" ] || { echo "check-threading-doc: FAIL: expected $expected_rows audit rows, found $row_count" >&2 exit 1 diff --git a/tests/meson.build b/tests/meson.build index 181ec4db..0263744a 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -4266,6 +4266,16 @@ test_col_rel_deep_copy_exe = executable( test('col_rel_deep_copy', test_col_rel_deep_copy_exe, timeout: 120) # Relation identity/generation contract and mutation instrumentation (Issue #1441) +relation_mutation_set_exe = executable( + 'test_relation_mutation_set', + files('test_relation_mutation_set.c', '../wirelog/evaluation_control.c'), + backend_src, intern_src, arena_src, thread_src, workqueue_src, + include_directories: [wirelog_inc, wirelog_src_inc], + dependencies: [nanoarrow_dep, threads_dep, xxhash_dep, mbedtls_dep, math_dep], +) +test('relation_mutation_set', relation_mutation_set_exe, + suite: ['unit', 'tsan'], timeout: 60) + test_relation_generations_exe = executable( 'test_relation_generations', files('test_relation_generations.c', '../wirelog/evaluation_control.c'), diff --git a/tests/test_relation_mutation_set.c b/tests/test_relation_mutation_set.c new file mode 100644 index 00000000..c253e478 --- /dev/null +++ b/tests/test_relation_mutation_set.c @@ -0,0 +1,304 @@ +/* Relation mutation admission and lease authority (#2033). */ +#include "../wirelog/columnar/internal.h" +#include "../wirelog/thread.h" +#include +#include +#include +#include +#include + +typedef struct { + wl_columnar_relation_mutation_set_t set; + wl_columnar_relation_mutation_descriptor_t descriptors[4]; + wl_columnar_relation_mutation_owner_t owners[4]; + wl_columnar_relation_mutation_lease_t leases[4]; + wl_columnar_relation_mutation_initialization_t initializations[4]; +} fixture_t; + +static int +acquire(fixture_t *f, wl_columnar_relation_mutation_role_t *roles, size_t n) +{ + return col_rel_mutation_set_acquire(&f->set, roles, n, + f->descriptors, 4, f->owners, 4, f->leases, 4, + f->initializations, 4); +} + +static void +root_init(col_rel_t *r, uint64_t id) +{ + memset(r, 0, sizeof(*r)); + r->relation_identity = id; + r->storage_generation = 1; + r->view_generation = 1; + r->storage_owner = r; + r->storage_owner_identity = id; + r->storage_owner_generation = 1; + atomic_init(&r->storage_alias_borrows, 0); + wl_columnar_source_access_gate_init(&r->descriptor_access); + wl_columnar_source_access_gate_init(&r->source_access); +} + +static void +alias_init(col_rel_t *a, col_rel_t *r, uint64_t id) +{ + root_init(a, id); + a->storage_owner = r; + a->storage_owner_identity = r->relation_identity; + a->storage_owner_generation = r->storage_generation; + assert(col_rel_storage_alias_borrow_acquire(r) == 0); +} + +static void +assert_open(col_rel_t *r) +{ + assert(!wl_columnar_source_access_gate_busy(&r->descriptor_access)); + assert(!wl_columnar_source_access_gate_busy(&r->source_access)); +} + +static void * +cross_thread(void *arg) +{ + wl_columnar_relation_mutation_lease_t *lease = arg; + assert(col_rel_mutation_lease_validate(lease, lease->relation) == EINVAL); + assert(col_rel_storage_alias_release_locked(lease->relation, + lease) == EINVAL); + assert(col_rel_mutation_set_finish(lease->set, true) == EINVAL); + return NULL; +} + +static void +lease_tests(void) +{ + col_rel_t root, a, b; + root_init(&root, 1); + alias_init(&a, &root, 2); + alias_init(&b, &root, 3); + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t roles[] = { + {&a, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&b, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&a, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} + }; + assert(acquire(&f, roles, 3) == 0); + assert(f.set.descriptor_count == 2 && f.set.owner_count == 1); + assert(&f.leases[0] != &f.leases[2]); + for (size_t i = 0; i < 3; i++) + assert(col_rel_mutation_set_lease(&f.set, i) == &f.leases[i]); + wl_columnar_relation_mutation_set_t set_copy = f.set; + set_copy.identity = (uintptr_t)&set_copy; + assert(col_rel_mutation_set_lease(&set_copy, 0) == NULL); + assert(col_rel_mutation_set_finish(&set_copy, true) == EINVAL); + f.leases[0].role_flags = WL_COLUMNAR_RELATION_METADATA_DETACH; + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + f.leases[0].role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; + wl_columnar_relation_mutation_lease_t copy = f.leases[0]; + assert(col_rel_mutation_lease_validate(©, &a) == EINVAL); + copy.identity = (uintptr_t)© + assert(col_rel_mutation_lease_validate(©, &a) == EINVAL); + assert(col_rel_mutation_lease_validate(&f.leases[0], &b) == EINVAL); + a.relation_identity++; + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + a.relation_identity--; + a.storage_generation++; + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + a.storage_generation--; + root.storage_generation++; + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + root.storage_generation--; + wl_thread_t thread; + assert(wl_thread_create(&thread, cross_thread, &f.leases[0]) == 0); + assert(wl_thread_join(&thread) == 0); + assert(col_rel_storage_alias_release_locked(&a, &f.leases[0]) == 0); + assert(a.storage_owner == &a && + col_rel_storage_alias_borrow_count(&root) == 1); + assert(col_rel_storage_alias_release_locked(&a, &f.leases[0]) == EINVAL); + assert(col_rel_mutation_lease_validate(&f.leases[2], &a) == EINVAL); + assert(col_rel_storage_alias_release_locked(&b, &f.leases[1]) == 0); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert_open(&root); + assert_open(&a); + assert_open(&b); + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + /* A raw source writer has no descriptor/set authority. */ + wl_columnar_source_access_writer_t raw = {0}; + assert(wl_columnar_source_access_writer_acquire(&root.source_access, + &raw) == 0); + copy = (wl_columnar_relation_mutation_lease_t){0}; + assert(col_rel_storage_alias_release_locked(&a, ©) == EINVAL); + assert(wl_columnar_source_access_writer_release(&raw) == 0); +} + +static void +contention_tests(void) +{ + col_rel_t root, a; + root_init(&root, 4); + alias_init(&a, &root, 5); + wl_columnar_memory_resolution_t resolution = {0}; + resolution.budget_bytes = resolution.usable_bytes = 1024; + resolution.mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING; + resolution.source = WL_COLUMNAR_MEMORY_SOURCE_ENV; + resolution.status = WL_COLUMNAR_MEMORY_OK; + wl_columnar_memory_governor_t governor; + assert(wl_columnar_memory_governor_init(&governor, &resolution) + == WL_COLUMNAR_MEMORY_OK); + wl_columnar_memory_reservation_init(&a.retained_reservation); + assert(wl_columnar_memory_reserve(&governor, 64, &a.retained_reservation)); + assert(wl_columnar_memory_commit(&a.retained_reservation, &a)); + a.retained_reserved_bytes = 64; + int64_t payload = 42; + int64_t *columns[] = {&payload}; + root.columns = a.columns = columns; + root.ncols = a.ncols = root.nrows = a.nrows = 1; + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t role = {&a, + WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}; + wl_columnar_source_access_reader_t reader = {0}; + assert(wl_columnar_source_access_reader_acquire(&a.descriptor_access, + &reader) == 0); + col_rel_t before = a; + assert(acquire(&f, &role, 1) == EBUSY); + assert(memcmp(&a, &before, sizeof(a)) == 0); + assert(wl_columnar_memory_reserved(&governor) == 64); + assert(col_rel_storage_alias_borrow_count(&root) == 1); + assert(wl_columnar_source_access_reader_release(&reader) == 0); + assert(wl_columnar_source_access_reader_acquire(&root.source_access, + &reader) == 0); + before = a; + assert(acquire(&f, &role, 1) == EBUSY); + assert(memcmp(&a, &before, sizeof(a)) == 0); + assert(wl_columnar_memory_reserved(&governor) == 64); + assert(!wl_columnar_source_access_gate_busy(&a.descriptor_access)); + role.role_flags = WL_COLUMNAR_RELATION_METADATA_DETACH; + assert(acquire(&f, &role, 1) == 0); + assert(f.set.owner_count == 0); + assert(col_rel_storage_alias_release_locked(&a, &f.leases[0]) == 0); + assert(col_rel_storage_alias_borrow_count(&root) == 0); + assert(root.columns[0][0] == 42); + assert(col_rel_storage_owner_destroy_status(&root) == EBUSY); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert(wl_columnar_source_access_reader_release(&reader) == 0); + role.role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; + assert(acquire(&f, &role, 1) == 0); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert(wl_columnar_memory_reserved(&governor) == 64); + assert(wl_columnar_memory_release(&a.retained_reservation)); + assert(wl_columnar_memory_reserved(&governor) == 0); +} + +static void +rollback_tests(void) +{ + col_rel_t relations[2]; + root_init(&relations[0], 6); + root_init(&relations[1], 7); + relations[0].storage_owner = NULL; + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t roles[] = { + {&relations[1], WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&relations[0], WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} + }; + wl_columnar_source_access_reader_t reader = {0}; + col_rel_t before = relations[0]; + /* Legacy ownership must remain untouched while its descriptor is read. */ + assert(wl_columnar_source_access_reader_acquire( + &relations[0].descriptor_access, + &reader) == 0); + col_rel_t paused = relations[0]; + assert(acquire(&f, roles, 2) == EBUSY); + assert(memcmp(&relations[0], &paused, sizeof(paused)) == 0); + assert(wl_columnar_source_access_reader_release(&reader) == 0); + /* Sorted second descriptor is busy, regardless of input order. */ + assert(wl_columnar_source_access_reader_acquire( + &relations[1].descriptor_access, + &reader) == 0); + assert(acquire(&f, roles, 2) == EBUSY); + assert(memcmp(&relations[0], &before, sizeof(before)) == 0); + assert_open(&relations[0]); + assert(wl_columnar_source_access_reader_release(&reader) == 0); + /* Sorted second owner failure follows provisional initialization. */ + assert(wl_columnar_source_access_reader_acquire(&relations[1].source_access, + &reader) == 0); + assert(acquire(&f, roles, 2) == EBUSY); + assert(memcmp(&relations[0], &before, sizeof(before)) == 0); + assert_open(&relations[0]); + assert(!wl_columnar_source_access_gate_busy( + &relations[1].descriptor_access)); + assert(wl_columnar_source_access_reader_release(&reader) == 0); + assert(acquire(&f, roles, 2) == 0); + assert(relations[0].storage_owner == &relations[0]); + assert(col_rel_mutation_set_finish(&f.set, false) == 0); + assert(memcmp(&relations[0], &before, sizeof(before)) == 0); + assert(acquire(&f, roles, 2) == 0); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert(relations[0].storage_owner == &relations[0]); + assert_open(&relations[0]); + assert_open(&relations[1]); +} + +/* The metadata set must not retain a dereferenceable old-owner dependency + * after the final borrow drops. ASan checks this finish-after-retirement path. */ +static void +metadata_final_access_test(void) +{ + col_rel_t *root = malloc(sizeof(*root)); + assert(root); + root_init(root, 10); + col_rel_t alias; + alias_init(&alias, root, 11); + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t role = {&alias, + WL_COLUMNAR_RELATION_METADATA_DETACH}; + assert(acquire(&f, &role, 1) == 0); + assert(col_rel_storage_alias_release_locked(&alias, &f.leases[0]) == 0); + assert(col_rel_storage_alias_borrow_count(root) == 0); + free(root); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert_open(&alias); +} + +static void +invalid_tests(void) +{ + col_rel_t root, alias; + root_init(&root, 8); + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t role = {&root, + WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}; + assert(acquire(&f, &role, 0) == EINVAL); + assert(col_rel_mutation_set_finish(&f.set, true) == EINVAL); + assert(col_rel_mutation_set_lease(&f.set, 0) == NULL); + assert(col_rel_mutation_set_acquire(NULL, &role, 1, + f.descriptors, 4, f.owners, 4, f.leases, 4, + f.initializations, 4) == EINVAL); + assert(col_rel_mutation_set_acquire(&f.set, &role, SIZE_MAX, + f.descriptors, SIZE_MAX, f.owners, SIZE_MAX, f.leases, SIZE_MAX, + f.initializations, SIZE_MAX) == EOVERFLOW); + assert(acquire(&f, NULL, 1) == EINVAL); + role.relation = NULL; + assert(acquire(&f, &role, 1) == EINVAL); + role.relation = &root; + assert(col_rel_mutation_set_acquire(&f.set, &role, 1, + f.descriptors, 0, f.owners, 4, f.leases, 4, + f.initializations, 4) == EINVAL); + f.descriptors[0].writer.identity = 1; + assert(acquire(&f, &role, 1) == EINVAL); + memset(&f, 0, sizeof(f)); + alias_init(&alias, &root, 9); + assert(acquire(&f, &role, 1) == EBUSY); + assert_open(&root); + assert(root.storage_owner == &root); + assert(col_rel_storage_alias_borrow_count(&root) == 1); +} + +int +main(void) +{ + lease_tests(); + contention_tests(); + rollback_tests(); + invalid_tests(); + metadata_final_access_test(); + puts("relation mutation set tests passed"); + return 0; +} diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 83668e4a..3bd69ff1 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -549,6 +549,88 @@ typedef struct col_rel { struct wl_col_session_t *deferred_relation_session; } col_rel_t; +/* Roles and caller-owned mutation storage must remain unchanged at their + * original addresses until + * finish. Roles contain descriptors only; owners are resolved after exclusion. */ +typedef enum { + WL_COLUMNAR_RELATION_PAYLOAD_MUTATION = 1, + WL_COLUMNAR_RELATION_METADATA_DETACH = 2 +} wl_columnar_relation_mutation_role_flags_t; + +typedef struct { + col_rel_t *relation; + wl_columnar_relation_mutation_role_flags_t role_flags; +} wl_columnar_relation_mutation_role_t; + +typedef struct { + col_rel_t *relation; + wl_columnar_source_access_writer_t writer; +} wl_columnar_relation_mutation_descriptor_t; + +typedef struct { + col_rel_t *owner; + wl_columnar_source_access_writer_t writer; +} wl_columnar_relation_mutation_owner_t; + +typedef struct { + col_rel_t *relation; + uint64_t owner_identity; + uint64_t owner_generation; + uint64_t borrows; +} wl_columnar_relation_mutation_initialization_t; + +struct wl_columnar_relation_mutation_set; +typedef struct { + uintptr_t identity; + struct wl_columnar_relation_mutation_set *set; + col_rel_t *relation; + col_rel_t *owner; + size_t descriptor_slot; + size_t owner_slot; + uint64_t relation_identity; + uint64_t relation_generation; + uint64_t owner_identity; + uint64_t owner_generation; + wl_columnar_relation_mutation_role_flags_t role_flags; + bool detached; +} wl_columnar_relation_mutation_lease_t; + +typedef struct wl_columnar_relation_mutation_set { + uintptr_t identity; + const wl_columnar_relation_mutation_role_t *roles; + wl_columnar_relation_mutation_descriptor_t *descriptors; + wl_columnar_relation_mutation_owner_t *owners; + wl_columnar_relation_mutation_lease_t *leases; + wl_columnar_relation_mutation_initialization_t *initializations; + size_t descriptor_count; + size_t owner_count; + size_t lease_count; + size_t initialization_count; + size_t descriptors_acquired; + size_t owners_acquired; +} wl_columnar_relation_mutation_set_t; + +int col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, + const wl_columnar_relation_mutation_role_t *roles, size_t role_count, + wl_columnar_relation_mutation_descriptor_t *descriptors, + size_t descriptor_cap, + wl_columnar_relation_mutation_owner_t *owners, size_t owner_cap, + wl_columnar_relation_mutation_lease_t *leases, size_t lease_cap, + wl_columnar_relation_mutation_initialization_t *initializations, + size_t initialization_cap); +wl_columnar_relation_mutation_lease_t *col_rel_mutation_set_lease( + wl_columnar_relation_mutation_set_t *set, size_t index); +int col_rel_mutation_lease_validate( + const wl_columnar_relation_mutation_lease_t *lease, + const col_rel_t *expected_relation); +int col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, + bool commit); +/* Metadata detach publishes the independent binding before its final atomic + * old-owner borrow decrement. The caller must never access that old owner + * again through this lease after successful release. */ +int col_rel_storage_alias_release_locked(col_rel_t *alias, + wl_columnar_relation_mutation_lease_t *lease); + /* Prepared terminal retirement of an independent heap relation. The token * holds the descriptor and source writers from prepare through cancel or * commit; callers must keep the descriptor pointer stable and must not copy, diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 68e55e4d..c7abf7b9 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -376,6 +376,365 @@ col_rel_storage_owner_resolve(const col_rel_t *src, col_rel_t **out_owner) return 0; } +/* Mutation sets own only caller-provided storage. No allocation is permitted + * between descriptor admission and finish, including failure unwind. */ +static bool +col_rel_mutation_writer_inert(const wl_columnar_source_access_writer_t *writer) +{ + return !writer->owner && !writer->secondary_owner && !writer->identity + && !writer->thread_valid; +} + +static void +col_rel_mutation_set_unwind(wl_columnar_relation_mutation_set_t *set, + bool commit) +{ + while (set->owners_acquired) + (void)wl_columnar_source_access_writer_release( + &set->owners[--set->owners_acquired].writer); + if (!commit) { + for (size_t i = set->initialization_count; i > 0; i--) { + wl_columnar_relation_mutation_initialization_t *init + = &set->initializations[i - 1]; + init->relation->storage_owner = NULL; + init->relation->storage_owner_identity = init->owner_identity; + init->relation->storage_owner_generation = init->owner_generation; + atomic_store_explicit(&init->relation->storage_alias_borrows, + init->borrows, memory_order_relaxed); + } + } + while (set->descriptors_acquired) + (void)wl_columnar_source_access_writer_release( + &set->descriptors[--set->descriptors_acquired].writer); + memset(set->descriptors, 0, + set->descriptor_count * sizeof(*set->descriptors)); + memset(set->owners, 0, set->owner_count * sizeof(*set->owners)); + memset(set->leases, 0, set->lease_count * sizeof(*set->leases)); + memset(set->initializations, 0, + set->initialization_count * sizeof(*set->initializations)); + memset(set, 0, sizeof(*set)); +} + +int +col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, + const wl_columnar_relation_mutation_role_t *roles, size_t role_count, + wl_columnar_relation_mutation_descriptor_t *descriptors, + size_t descriptor_cap, + wl_columnar_relation_mutation_owner_t *owners, size_t owner_cap, + wl_columnar_relation_mutation_lease_t *leases, size_t lease_cap, + wl_columnar_relation_mutation_initialization_t *initializations, + size_t initialization_cap) +{ + int rc = EINVAL; + if (!set || set->identity || set->roles || set->descriptors || + set->owners || set->leases + || set->initializations || set->descriptor_count || set->owner_count + || set->lease_count || set->initialization_count + || set->descriptors_acquired || set->owners_acquired + || !role_count || !roles || !descriptors || !owners || !leases + || !initializations || descriptor_cap < role_count + || owner_cap < role_count || lease_cap < role_count + || initialization_cap < role_count) + return EINVAL; + if (role_count > SIZE_MAX / sizeof(*descriptors) + || role_count > SIZE_MAX / sizeof(*owners) + || role_count > SIZE_MAX / sizeof(*leases) + || role_count > SIZE_MAX / sizeof(*initializations)) + return EOVERFLOW; + /* Reject overlapping caller storage before writing any bookkeeping. */ + uintptr_t starts[] = {(uintptr_t)set, (uintptr_t)roles, + (uintptr_t)descriptors, (uintptr_t)owners, + (uintptr_t)leases, + (uintptr_t)initializations}; + size_t lengths[] = {sizeof(*set), role_count * sizeof(*roles), + role_count * sizeof(*descriptors), + role_count * sizeof(*owners), + role_count * sizeof(*leases), + role_count * sizeof(*initializations)}; + for (size_t i = 0; i < sizeof(starts) / sizeof(starts[0]); i++) { + if (lengths[i] > UINTPTR_MAX - starts[i]) + return EOVERFLOW; + for (size_t j = 0; j < i; j++) + if (starts[i] < starts[j] + lengths[j] + && starts[j] < starts[i] + lengths[i]) + return EINVAL; + } + for (size_t i = 0; i < role_count; i++) { + if (!roles[i].relation + || (roles[i].role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + && roles[i].role_flags != WL_COLUMNAR_RELATION_METADATA_DETACH) + || descriptors[i].relation || owners[i].owner + || !col_rel_mutation_writer_inert(&descriptors[i].writer) + || !col_rel_mutation_writer_inert(&owners[i].writer) + || leases[i].identity || leases[i].set || leases[i].relation + || leases[i].owner || leases[i].descriptor_slot + || leases[i].owner_slot || leases[i].relation_identity + || leases[i].relation_generation || leases[i].owner_identity + || leases[i].owner_generation || leases[i].role_flags + || leases[i].detached || initializations[i].relation + || initializations[i].owner_identity + || initializations[i].owner_generation || + initializations[i].borrows) + return EINVAL; + } + set->identity = (uintptr_t)set; + set->roles = roles; + set->descriptors = descriptors; + set->owners = owners; + set->leases = leases; + set->initializations = initializations; + set->lease_count = role_count; + for (size_t i = 0; i < role_count; i++) { + size_t j = 0; + while (j < set->descriptor_count + && descriptors[j].relation != roles[i].relation) + j++; + if (j == set->descriptor_count) + descriptors[set->descriptor_count++].relation = roles[i].relation; + } + for (size_t i = 1; i < set->descriptor_count; i++) { + col_rel_t *relation = descriptors[i].relation; + size_t j = i; + while (j && (uintptr_t)&descriptors[j - 1].relation->descriptor_access + > (uintptr_t)&relation->descriptor_access) { + descriptors[j].relation = descriptors[j - 1].relation; + j--; + } + descriptors[j].relation = relation; + } + for (size_t i = 0; i < set->descriptor_count; i++) { + rc = wl_columnar_source_access_writer_acquire( + &descriptors[i].relation->descriptor_access, + &descriptors[i].writer); + if (rc) + goto fail; + set->descriptors_acquired++; + } + for (size_t i = 0; i < set->descriptor_count; i++) { + col_rel_t *relation = descriptors[i].relation; + if (!relation->storage_owner) { + if (!relation->relation_identity + || col_rel_storage_alias_borrow_count(relation) + || !wl_columnar_relation_generation_valid( + relation->storage_generation)) { + rc = EINVAL; + goto fail; + } + wl_columnar_relation_mutation_initialization_t *init + = &initializations[set->initialization_count++]; + init->relation = relation; + init->owner_identity = relation->storage_owner_identity; + init->owner_generation = relation->storage_owner_generation; + init->borrows = col_rel_storage_alias_borrow_count(relation); + relation->storage_owner = relation; + relation->storage_owner_identity = relation->relation_identity; + relation->storage_owner_generation = relation->storage_generation; + atomic_store_explicit(&relation->storage_alias_borrows, 0, + memory_order_relaxed); + } + } + for (size_t i = 0; i < role_count; i++) { + col_rel_t *relation = roles[i].relation; + col_rel_t *owner; + rc = col_rel_storage_owner_resolve(relation, &owner); + if (rc) + goto fail; + if (!relation->relation_identity || !owner->relation_identity + || !wl_columnar_relation_generation_valid( + relation->storage_generation) + || !wl_columnar_relation_generation_valid(owner->storage_generation) + || relation->storage_owner_identity != owner->relation_identity + || relation->storage_owner_generation != + owner->storage_generation) { + rc = EINVAL; + goto fail; + } + wl_columnar_relation_mutation_lease_t *lease = &leases[i]; + lease->set = set; + lease->relation = relation; + lease->owner = owner; + lease->role_flags = roles[i].role_flags; + lease->relation_identity = relation->relation_identity; + lease->relation_generation = relation->storage_generation; + lease->owner_identity = owner->relation_identity; + lease->owner_generation = owner->storage_generation; + for (size_t j = 0; j < set->descriptor_count; j++) + if (descriptors[j].relation == relation) + lease->descriptor_slot = j; + if (roles[i].role_flags == WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) { + size_t j = 0; + while (j < set->owner_count && owners[j].owner != owner) + j++; + if (j == set->owner_count) + owners[set->owner_count++].owner = owner; + } + } + for (size_t i = 1; i < set->owner_count; i++) { + col_rel_t *owner = owners[i].owner; + size_t j = i; + while (j && (uintptr_t)&owners[j - 1].owner->source_access + > (uintptr_t)&owner->source_access) { + owners[j].owner = owners[j - 1].owner; + j--; + } + owners[j].owner = owner; + } + for (size_t i = 0; i < set->owner_count; i++) { + rc = wl_columnar_source_access_writer_acquire( + &owners[i].owner->source_access, &owners[i].writer); + if (rc) + goto fail; + set->owners_acquired++; + } + for (size_t i = 0; i < role_count; i++) { + wl_columnar_relation_mutation_lease_t *lease = &leases[i]; + if (lease->role_flags == WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) { + for (size_t j = 0; j < set->owner_count; j++) + if (owners[j].owner == lease->owner) + lease->owner_slot = j; + if (lease->relation == lease->owner + && col_rel_storage_alias_borrow_count(lease->owner)) { + rc = EBUSY; + goto fail; + } + } + lease->identity = (uintptr_t)lease; + } + /* Ownership snapshots must still match after source admission. */ + for (size_t i = 0; i < role_count; i++) { + rc = col_rel_mutation_lease_validate(&leases[i], roles[i].relation); + if (rc) + goto fail; + } + return 0; +fail: + col_rel_mutation_set_unwind(set, false); + return rc; +} + +wl_columnar_relation_mutation_lease_t * +col_rel_mutation_set_lease(wl_columnar_relation_mutation_set_t *set, + size_t index) +{ + if (!set || set->identity != (uintptr_t)set || !set->leases + || index >= set->lease_count || set->leases[index].set != set + || col_rel_mutation_lease_validate(&set->leases[index], + set->leases[index].relation)) + return NULL; + return &set->leases[index]; +} + +int +col_rel_mutation_lease_validate( + const wl_columnar_relation_mutation_lease_t *lease, + const col_rel_t *expected_relation) +{ + if (!lease || lease->identity != (uintptr_t)lease || !lease->set + || !expected_relation || lease->relation != expected_relation + || lease->detached) + return EINVAL; + const wl_columnar_relation_mutation_set_t *set = lease->set; + if (set->identity != (uintptr_t)set || !set->roles || !set->leases || + !set->descriptors + || !set->owners || set->descriptors_acquired != set->descriptor_count + || set->owners_acquired != set->owner_count + || lease->descriptor_slot >= set->descriptor_count) + return EINVAL; + bool member = false; + for (size_t i = 0; i < set->lease_count; i++) + if (&set->leases[i] == lease) { + if (set->roles[i].relation != expected_relation + || set->roles[i].role_flags != lease->role_flags) + return EINVAL; + member = true; + } + if (!member) + return EINVAL; + const wl_columnar_relation_mutation_descriptor_t *descriptor + = &set->descriptors[lease->descriptor_slot]; + if (descriptor->relation != expected_relation + || wl_columnar_source_access_writer_validate(&descriptor->writer, + &expected_relation->descriptor_access)) + return EINVAL; + col_rel_t *owner; + if (col_rel_storage_owner_resolve(expected_relation, &owner) + || owner != lease->owner + || expected_relation->relation_identity != lease->relation_identity + || expected_relation->storage_generation != lease->relation_generation + || expected_relation->storage_owner_identity != lease->owner_identity + || expected_relation->storage_owner_generation != + lease->owner_generation + || owner->relation_identity != lease->owner_identity + || owner->storage_generation != lease->owner_generation + || !wl_columnar_relation_generation_valid(lease->relation_generation) + || !wl_columnar_relation_generation_valid(lease->owner_generation)) + return EINVAL; + if (lease->role_flags == WL_COLUMNAR_RELATION_METADATA_DETACH) + return 0; + if (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || lease->owner_slot >= set->owner_count + || set->owners[lease->owner_slot].owner != owner) + return EINVAL; + return wl_columnar_source_access_writer_validate( + &set->owners[lease->owner_slot].writer, &owner->source_access); +} + +int +col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, + bool commit) +{ + if (!set || set->identity != (uintptr_t)set + || set->descriptors_acquired != set->descriptor_count + || set->owners_acquired != set->owner_count) + return EINVAL; + if (!set->leases || !set->roles || !set->descriptors || !set->owners + || !set->initializations || !set->lease_count || !set->descriptor_count) + return EINVAL; + for (size_t i = 0; i < set->lease_count; i++) + if (set->leases[i].set != set + || set->leases[i].identity != (uintptr_t)&set->leases[i]) + return EINVAL; + for (size_t i = 0; i < set->descriptor_count; i++) + if (wl_columnar_source_access_writer_validate( + &set->descriptors[i].writer, + &set->descriptors[i].relation->descriptor_access)) + return EINVAL; + for (size_t i = 0; i < set->owner_count; i++) + if (wl_columnar_source_access_writer_validate(&set->owners[i].writer, + &set->owners[i].owner->source_access)) + return EINVAL; + col_rel_mutation_set_unwind(set, commit); + return 0; +} + +int +col_rel_storage_alias_release_locked(col_rel_t *alias, + wl_columnar_relation_mutation_lease_t *lease) +{ + int rc = col_rel_mutation_lease_validate(lease, alias); + if (rc) + return rc; + col_rel_t *owner = lease->owner; + if (alias == owner) + return 0; + if (!col_rel_storage_alias_borrow_count(owner) + || col_rel_storage_alias_borrow_count(alias)) + return EINVAL; + alias->storage_owner = alias; + alias->storage_owner_identity = alias->relation_identity; + alias->storage_owner_generation = alias->storage_generation; + lease->detached = true; + /* This is deliberately the final access to owner. Our still-live borrow + * prevents teardown before the decrement; a peer reader owns its own gate. */ + uint64_t prior = atomic_fetch_sub_explicit(&owner->storage_alias_borrows, + 1, memory_order_acq_rel); + /* A zero prior value means an invariant was violated outside this + * descriptor's authority. Do not continue with a wrapped borrow count. */ + if (!prior) + abort(); + return 0; +} + int col_rel_replacement_cohort_validate( const col_rel_t *destination, diff --git a/wirelog/columnar/source_access.h b/wirelog/columnar/source_access.h index f22eb64d..82a07bb9 100644 --- a/wirelog/columnar/source_access.h +++ b/wirelog/columnar/source_access.h @@ -466,6 +466,21 @@ wl_columnar_source_access_writer_acquire( return 0; } +/* Validate exclusive authority without consuming it. */ +static inline int +wl_columnar_source_access_writer_validate( + const wl_columnar_source_access_writer_t *token, + const wl_columnar_source_access_gate_t *gate) +{ + if (!token || !gate || token->identity != (uintptr_t)token + || token->owner != gate || token->secondary_owner + || !wl_columnar_source_access_writer_thread_equal(token) + || atomic_load_explicit(&gate->state, memory_order_acquire) + != WL_COLUMNAR_SOURCE_ACCESS_WRITER) + return EINVAL; + return 0; +} + static inline int wl_columnar_source_access_writer_release( wl_columnar_source_access_writer_t *token) From 7eff9cb0a61dcf86585b02ccf12012fcc43e8df1 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 12:25:43 +0900 Subject: [PATCH 2/9] Guard single-relation payload mutation with leases --- tests/test_relation_generations.c | 287 +++++++++++++++++++++++++++++- wirelog/columnar/internal.h | 17 +- wirelog/columnar/merge.c | 7 +- wirelog/columnar/relation.c | 226 ++++++++++++++++------- 4 files changed, 460 insertions(+), 77 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index cabcd490..d4da6f70 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -135,11 +135,11 @@ set_transition_probe(col_rel_t *rel) && set_hook_owner->view_generation == set_hook_source_view_generation && set_hook_owner->storage_generation == set_hook_source_storage_generation - && rel->storage_owner == set_hook_owner + && rel->storage_owner == rel && rel->storage_generation != set_hook_storage_generation && rel->view_generation == set_hook_view_generation && rel->col_shared == NULL - && set_hook_owner->storage_alias_borrows == 1; + && set_hook_owner->storage_alias_borrows == 0; set_hook_reader_rc = col_rel_source_reader_acquire(rel, &reader); if (set_hook_reader_rc == 0) (void)col_rel_source_reader_release(&reader); @@ -4588,6 +4588,286 @@ rebind_paused_reader(void *opaque) return NULL; } +static int +mutation_set_or_cow(col_rel_t *relation, unsigned operation) +{ + if (operation == 0) + return col_rel_set(relation, 0, 0, 91); + return col_rel_cow_unshare(relation, + operation == 2 ? relation->capacity + 1 : 0); +} + +static void +test_set_cow_descriptor_and_owner_admission(void) +{ + for (unsigned operation = 0; operation < 3; operation++) { + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + int64_t value = 73; + CHECK(root && alias && col_rel_append_row(root, &value) == 0 + && col_rel_install_shared_view(alias, root) == 0, + "set/COW mutation alias setup"); + if (!root || !alias) + continue; + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 20, + .usable_bytes = 1u << 20, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(alias, ref) == 0, + "set/COW live governor setup"); + uint64_t credit = ref ? wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) : 0; + col_rel_t snapshot = *alias; + uint64_t root_view = root->view_generation; + uint64_t root_storage = root->storage_generation; + rebind_reader_pause_t pause = { .relation = alias }; + atomic_init(&pause.ready, false); + atomic_init(&pause.proceed, false); + rebind_reader_pause = &pause; + wl_columnar_relation_test_reader_after_descriptor + = rebind_pause_after_descriptor; + wl_thread_t thread; + bool started = wl_thread_create(&thread, rebind_paused_reader, + &pause) == 0; + CHECK(started, "set/COW paused source reader starts"); + if (started) { + while (!atomic_load_explicit(&pause.ready, memory_order_acquire)) + alias_accounting_yield(); + /* Snapshot includes the admitted descriptor reader itself. */ + snapshot = *alias; + CHECK(mutation_set_or_cow(alias, operation) == EBUSY + && memcmp(alias, &snapshot, sizeof(snapshot)) == 0 + && col_rel_storage_alias_borrow_count(root) == 1 + && root->columns[0][0] == value + && (!ref || wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == credit), + "paused descriptor reader rejects set/COW without mutation"); + atomic_store_explicit(&pause.proceed, true, memory_order_release); + CHECK(wl_thread_join(&thread) == 0 && pause.rc == 0, + "paused source reader drains successfully"); + } + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + snapshot = *alias; + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "set/COW peer owner reader setup"); + CHECK(mutation_set_or_cow(alias, operation) == EBUSY + && memcmp(alias, &snapshot, sizeof(snapshot)) == 0 + && col_rel_storage_alias_borrow_count(root) == 1 + && (!ref || wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == credit), + "peer owner reader rejects set/COW without mutation"); + wl_columnar_source_access_writer_t descriptor = { 0 }; + CHECK(wl_columnar_source_access_writer_acquire( + &alias->descriptor_access, + &descriptor) == 0 + && wl_columnar_source_access_writer_release(&descriptor) == 0, + "owner-phase denial fully releases descriptor writer"); + CHECK(col_rel_source_reader_release(&peer) == 0, + "set/COW peer reader release"); + col_rel_t root_snapshot = *root; + CHECK(col_rel_set(root, 0, 0, 92) == EBUSY + && col_rel_cow_unshare(root, 0) == EBUSY + && memcmp(root, &root_snapshot, sizeof(root_snapshot)) == 0, + "canonical mutation with live alias is rejected unchanged"); + wl_columnar_source_access_writer_t raw = { 0 }; + CHECK(col_rel_source_writer_acquire(alias, &raw) == 0, + "raw alias writer setup"); + CHECK(col_rel_cow_unshare_with_source_writer(alias, &raw) == EINVAL + && memcmp(alias, &snapshot, + offsetof(col_rel_t, source_access)) == 0, + "raw source writer cannot authorize alias COW"); + CHECK(wl_columnar_source_access_writer_release(&raw) == 0, + "raw alias writer release"); + wl_columnar_relation_test_fail_next_prepare_resize(); + CHECK(mutation_set_or_cow(alias, operation) == ENOMEM + && memcmp(alias, &snapshot, sizeof(snapshot)) == 0 + && col_rel_storage_alias_borrow_count(root) == 1 + && (!ref || wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == credit), + "lease-held preparation failure leaves relation and credit unchanged"); + CHECK(mutation_set_or_cow(alias, operation) == 0 + && alias->storage_owner == alias && !alias->col_shared + && col_rel_storage_alias_borrow_count(root) == 0 + && alias->storage_generation == snapshot.storage_generation + 1 + && alias->storage_owner_generation == alias->storage_generation + && alias->view_generation == snapshot.view_generation + + (operation == 0 ? 1 : 0) + && alias->columns[0][0] == (operation == 0 ? 91 : value) + && root->columns[0][0] == value + && root->view_generation == root_view + && root->storage_generation == root_storage, + "set/COW retry publishes one storage epoch and one borrow release"); + cleanup_relations(); + if (ref) { + CHECK(wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == 0, + "set/COW governed cleanup releases all credit"); + wl_columnar_memory_governor_ref_release(ref); + } + } +} + +/* Denial evidence is an error result channel, outside the transactional + * payload snapshot. Compare the payload and ownership fields explicitly. */ +static bool +mutation_payload_unchanged(const col_rel_t *rel, const col_rel_t *before) +{ + return rel->columns == before->columns && + rel->col_shared == before->col_shared + && rel->timestamps == before->timestamps + && rel->capacity == before->capacity && rel->nrows == before->nrows + && rel->ncols == before->ncols + && rel->timestamp_capacity == before->timestamp_capacity + && rel->arena_owned == before->arena_owned + && rel->relation_identity == before->relation_identity + && rel->view_generation == before->view_generation + && rel->storage_generation == before->storage_generation + && rel->storage_owner == before->storage_owner + && rel->storage_owner_identity == before->storage_owner_identity + && rel->storage_owner_generation == before->storage_owner_generation + && col_rel_storage_alias_borrow_count(rel) + == col_rel_storage_alias_borrow_count(before) + && rel->retained_reserved_bytes == before->retained_reserved_bytes + && rel->retained_reservation.identity == + before->retained_reservation.identity + && rel->retained_reservation.bytes == + before->retained_reservation.bytes + && atomic_load_explicit(&rel->retained_reservation.state, + memory_order_acquire) + == atomic_load_explicit(&before->retained_reservation.state, + memory_order_acquire) + && atomic_load_explicit(&rel->retained_reservation.owner_bits, + memory_order_acquire) + == atomic_load_explicit(&before->retained_reservation.owner_bits, + memory_order_acquire); +} + +static void +test_set_cow_denial_provenance(void) +{ + for (unsigned operation = 0; operation < 2; operation++) { + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + int64_t value = 67; + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 20, + .usable_bytes = 1u << 20, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(root && alias && ref && col_rel_append_row(root, &value) == 0 + && col_rel_install_shared_view(alias, root) == 0 + && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(alias, ref) == 0, + "set/COW provenance governed shared-alias setup"); + if (!root || !alias || !ref) { + cleanup_relations(); + if (ref) + wl_columnar_memory_governor_ref_release(ref); + continue; + } + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t alias_before = *alias; + col_rel_t root_before = *root; + /* Exhaust headroom without changing the live relation reservations. */ + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "set/COW quota denial blocker admitted"); + for (unsigned prior = 0; prior < 2; prior++) { + alias->memory_budget_denial_pending = (uint8_t)prior; + CHECK(mutation_set_or_cow(alias, operation) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && alias->columns[0][0] == value && root->columns[0][0] == value + && wl_columnar_memory_reserved(governor) == + resolution.usable_bytes, + "quota ENOMEM publishes denial evidence without payload/credit mutation"); + } + CHECK(wl_columnar_memory_release(&blocker) + && wl_columnar_memory_reserved(governor) == credit, + "set/COW quota blocker released for retry"); + for (unsigned prior = 0; prior < 2; prior++) { + for (unsigned failure = 0; failure < 2; failure++) { + alias->memory_budget_denial_pending = (uint8_t)prior; + if (failure == 0) + wl_columnar_relation_test_fail_next_prepare_resize(); + else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(mutation_set_or_cow(alias, operation) == ENOMEM + && alias->memory_budget_denial_pending == + (operation == 0 ? prior : 0) + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && alias->columns[0][0] == value && + root->columns[0][0] == value + && wl_columnar_memory_reserved(governor) == credit, + "nonbudget preparation/publication ENOMEM preserves payload and provenance"); + } + } + alias->memory_budget_denial_pending = true; + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "stale denial EBUSY peer admission"); + CHECK(mutation_set_or_cow(alias, operation) == EBUSY + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && root->columns[0][0] == value + && wl_columnar_memory_reserved(governor) == credit, + "guard EBUSY preserves stale denial evidence before eager clear"); + CHECK(col_rel_source_reader_release(&peer) == 0, + "stale denial EBUSY peer release"); + CHECK(mutation_set_or_cow(alias, operation) == 0 + && alias->memory_budget_denial_pending == (operation == 0 ? 1 : 0) + && alias->storage_owner == alias && !alias->col_shared + && col_rel_storage_alias_borrow_count(root) == 0 + && alias->storage_generation == alias_before.storage_generation + 1 + && alias->columns[0][0] == (operation == 0 ? 91 : value) + && root->columns[0][0] == value, + "successful retry clears COW evidence and preserves set evidence"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "provenance governed cleanup releases all credit"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_empty_append_all_legacy_cow_compatibility(void) +{ + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + col_rel_t *empty = new_relation(); + int64_t value = 81; + CHECK(root && alias && empty + && col_rel_append_row(root, &value) == 0 + && col_rel_install_shared_view(alias, root) == 0, + "empty-source legacy COW setup"); + CHECK(col_rel_append_all(alias, empty, NULL) == 0 + && alias->storage_owner == alias && alias->col_shared == NULL + && alias->columns[0][0] == value && alias->nrows == 1 + && root->columns[0][0] == value + && col_rel_storage_alias_borrow_count(root) == 0, + "empty append_all retains COW under its legacy raw writer"); + cleanup_relations(); +} + static void rebind_probe_descriptor_exclusion(const col_rel_t *relation) { @@ -4843,6 +5123,9 @@ int main(void) { #ifdef WL_TEST_RELATION_RESIZE_HOOK + test_set_cow_descriptor_and_owner_admission(); + test_set_cow_denial_provenance(); + test_empty_append_all_legacy_cow_compatibility(); test_leased_rebind_descriptor_exclusion(); test_root_leased_rebind_restores_descriptor(); #endif diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 3bd69ff1..5f34d08e 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2675,8 +2675,10 @@ void col_rel_retire_payload_credit(col_rel_t *r); * @new_cap <= capacity it admits the buffers the relation already owns. * ENOMEM with *@denied set is a governor verdict, clear is an allocation * failure (Issue #1446). EOVERFLOW reports size/accounting overflow; EINVAL - * reports invalid shape/state. Caller owns an unpublished relation or holds - * its writer. This operation resets transient denial evidence. */ + * reports invalid shape/state. Caller owns an unpublished/self-owned + * relation exclusively or holds its writer; the raw writer never authorizes + * alias detach through the mutation-set API. This operation resets transient + * denial evidence. */ int col_rel_reserve_capacity_admitted(col_rel_t *r, uint32_t new_cap, bool *denied); @@ -2689,7 +2691,8 @@ int col_rel_enable_timestamps_locked(col_rel_t *rel); /* Promote arena-backed relation columns to private heap storage. Admission * and copying are transactional; failure leaves the relation unchanged. - * Caller owns the relation exclusively. ENOMEM+pending denotes budget + * Caller owns the private/self-owned relation exclusively; this helper + * acquires no live mutation authority. ENOMEM+pending denotes budget * refusal; ENOMEM alone is allocation failure, EOVERFLOW/EINVAL stay typed. */ int col_rel_promote_arena_admitted(col_rel_t *rel); @@ -3072,11 +3075,17 @@ int col_rel_source_reader_acquire_transferable(const col_rel_t *, int col_rel_source_reader_release(wl_columnar_source_access_reader_t *); int col_rel_source_writer_acquire(const col_rel_t *, wl_columnar_source_access_writer_t *); +/* Canonical non-detach authority only: raw writers cannot detach aliases. */ int col_rel_cow_unshare_with_source_writer(col_rel_t *, const wl_columnar_source_access_writer_t *); +/* Temporary compatibility used only by the three unmigrated merge + * transactions. Remove as those complete transactions acquire mutation sets. */ +int col_rel_cow_unshare_legacy_with_source_writer(col_rel_t *, + const wl_columnar_source_access_writer_t *); + /* Checked single-cell mutation. The implementation lives in relation.c so - * it can hold canonical-owner source admission across COW and publication. */ + * it holds descriptor and canonical-owner writers across COW/publication. */ int col_rel_set(col_rel_t *, uint32_t row, uint32_t col, int64_t val); /* Test seam for the non-wrapping relation identity allocator. */ diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index 764ad579..cd3967bc 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -833,7 +833,7 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, * came from the validated source relation, so publish has no expected * validation failure and performs no allocation. */ if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_with_source_writer(rel, writer); + int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, writer); if (cow_rc != 0) { free(uniq_buf); HASH_RELEASE_SCRATCH(); @@ -1011,7 +1011,7 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, /* Every buffer allocation is now complete. A shared view may detach * before the first segment mutation; later sorting uses only workspace. */ if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_with_source_writer(rel, writer); + int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, writer); if (cow_rc != 0) { result = cow_rc; goto cleanup; @@ -2267,7 +2267,8 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, * view before those mutations so the deferred cleanup can retire its * alias borrow while the owner writer remains held. */ if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_with_source_writer(rel, rel_writer); + int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, + rel_writer); if (cow_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, cow_rc); diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index c7abf7b9..3983884f 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -276,20 +276,68 @@ wl_columnar_relation_radix_bench_enabled(void) /* ---- COW helpers --------------------------------------------------------- */ -static int col_rel_grow_owned_transition_impl(col_rel_t *r, +/* Single-entry storage lives with the operation, including its role mapping. */ +typedef struct { + wl_columnar_relation_mutation_set_t set; + wl_columnar_relation_mutation_role_t role; + wl_columnar_relation_mutation_descriptor_t descriptor; + wl_columnar_relation_mutation_owner_t owner; + wl_columnar_relation_mutation_lease_t lease; + wl_columnar_relation_mutation_initialization_t initialization; +} wl_columnar_relation_mutation_single_t; + +static int +col_rel_mutation_single_acquire(col_rel_t *relation, + wl_columnar_relation_mutation_single_t *single) +{ + single->role.relation = relation; + single->role.role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; + return col_rel_mutation_set_acquire(&single->set, &single->role, 1, + &single->descriptor, 1, &single->owner, 1, &single->lease, 1, + &single->initialization, 1); +} + +static int col_rel_grow_owned_transition_legacy_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release); typedef struct { int64_t **columns; bool *shared; } col_rel_cow_deferred_t; -static int col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, +static int col_rel_cow_unshare_legacy_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, bool defer_metadata_retirement, col_rel_cow_deferred_t *deferred_old); +static int col_rel_grow_owned_transition_publish_impl(col_rel_t *r, + uint32_t new_cap, bool defer_alias_release, + wl_columnar_relation_mutation_lease_t *lease); +static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, + bool defer_alias_release, bool defer_metadata_retirement, + col_rel_cow_deferred_t *deferred_old, + wl_columnar_relation_mutation_lease_t *lease); + +/* Temporary compatibility for unmigrated append/radix/batch transactions. + * NULL is an explicit legacy path, never a payload mutation lease. These + * private entry points disappear as their outer transactions acquire sets. */ +static int +col_rel_grow_owned_transition_legacy_impl(col_rel_t *r, uint32_t new_cap, + bool defer_alias_release) +{ + return col_rel_grow_owned_transition_publish_impl(r, new_cap, + defer_alias_release, NULL); +} + +static int +col_rel_cow_unshare_legacy_impl(col_rel_t *r, uint32_t new_cap, + bool defer_alias_release, bool defer_metadata_retirement, + col_rel_cow_deferred_t *deferred_old) +{ + return col_rel_cow_unshare_publish_impl(r, new_cap, defer_alias_release, + defer_metadata_retirement, deferred_old, NULL); +} static int col_rel_grow_owned_transition(col_rel_t *r, uint32_t new_cap) { - return col_rel_grow_owned_transition_impl(r, new_cap, false); + return col_rel_grow_owned_transition_legacy_impl(r, new_cap, false); } static void @@ -1015,59 +1063,45 @@ col_rel_published_writer_acquire(const col_rel_t *rel, int col_rel_set(col_rel_t *r, uint32_t row, uint32_t col, int64_t val) { - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_release_pending = false; - col_rel_t *owner = NULL; + wl_columnar_relation_mutation_single_t single = { 0 }; int rc; - - if (!r || !r->columns || row >= r->capacity || col >= r->ncols - || !r->columns[col]) + if (!r) return EINVAL; + rc = col_rel_mutation_single_acquire(r, &single); + if (rc) + return rc; + if (!r->columns || row >= r->capacity || col >= r->ncols + || !r->columns[col]) { + rc = EINVAL; + goto finish; + } if (r->column_types && r->column_types[col] == WIRELOG_TYPE_FLOAT) { - if (!wl_columnar_float_bits_valid(val)) - return EINVAL; + if (!wl_columnar_float_bits_valid(val)) { + rc = EINVAL; + goto finish; + } if (wl_columnar_float_bits_zero(val)) val = 0; } - - rc = col_rel_storage_owner_resolve(r, &owner); - if (rc != 0) - return rc; - rc = col_rel_source_writer_acquire(r, &writer); - if (rc != 0) - return rc; - if (r == owner && col_rel_storage_alias_borrow_count(owner) > 0) { - rc = EBUSY; - goto release_writer; - } - - /* Keep the canonical owner writer admission held through the detach, - * cell write, and logical generation publication. The deferred alias - * release prevents a new reader from entering through the detached view - * before this mutation is complete. */ - if (r->col_shared) { - rc = col_rel_cow_unshare_impl(r, 0, true, false, NULL); - if (rc != 0) - goto release_writer; - alias_release_pending = true; + bool borrowed = r->col_shared != NULL; + if (borrowed) { + rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, + &single.lease); + if (rc) + goto finish; } #ifdef WL_TEST_SET_HOOK - if (alias_release_pending && wl_columnar_set_transition_hook) + if (borrowed && wl_columnar_set_transition_hook) wl_columnar_set_transition_hook(r); #endif r->columns[col][row] = val; wl_columnar_relation_touch_view(r); rc = 0; - -release_writer: - if (alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(r); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; +finish: + { + int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); + return rc ? rc : finish_rc; } - if (wl_columnar_source_access_writer_release(&writer) != 0 && rc == 0) - rc = EINVAL; - return rc; } /* Prepare a complete private replacement for a relation resize. This is @@ -1940,8 +1974,8 @@ col_rel_reserve_transition(col_rel_t *r, uint32_t capacity, * prepared before the relation is changed, so admission or allocation * failure leaves the old ownership and reservation untouched. */ static int -col_rel_grow_owned_transition_impl(col_rel_t *r, uint32_t new_cap, - bool defer_alias_release) +col_rel_grow_owned_transition_publish_impl(col_rel_t *r, uint32_t new_cap, + bool defer_alias_release, wl_columnar_relation_mutation_lease_t *lease) { col_rel_payload_txn_t pending; wl_columnar_memory_reservation_t previous; @@ -1955,6 +1989,9 @@ col_rel_grow_owned_transition_impl(col_rel_t *r, uint32_t new_cap, bool old_arena; uint64_t ledger_before; + if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r))) + return EINVAL; wl_columnar_memory_reservation_init(&previous); if (!r || (r->ncols && !r->columns) || new_cap < r->nrows) @@ -2016,9 +2053,19 @@ col_rel_grow_owned_transition_impl(col_rel_t *r, uint32_t new_cap, col_rel_release_retired_reservation(&previous); col_rel_retire_payload_credit(r); col_rel_ledger_reconcile(r, ledger_before); - if (!defer_alias_release) - (void)col_rel_storage_alias_release(r); - wl_columnar_relation_touch_storage(r); + if (lease) { + if (!defer_alias_release) { + /* Publication cannot fail from here. Invalid authority means a + * broken invariant, never a recoverable partial-success return. */ + if (col_rel_storage_alias_release_locked(r, lease)) + abort(); + wl_columnar_relation_touch_storage(r); + } + } else { + if (!defer_alias_release) + (void)col_rel_storage_alias_release(r); + wl_columnar_relation_touch_storage(r); + } return 0; fail: @@ -2046,15 +2093,28 @@ col_rel_promote_arena_admitted(col_rel_t *r) int col_rel_cow_unshare(col_rel_t *r, uint32_t new_cap) { - if (r) - r->memory_budget_denial_pending = false; - return col_rel_cow_unshare_impl(r, new_cap, false, false, NULL); + wl_columnar_relation_mutation_single_t single = { 0 }; + int rc; + if (!r) + return 0; + rc = col_rel_mutation_single_acquire(r, &single); + if (rc) + return rc; + /* ENOMEM provenance is an outward result channel: quota refusal sets + * it again; allocation failure keeps it clear. Guard contention above + * leaves prior evidence untouched. It is not mutation rollback state. */ + r->memory_budget_denial_pending = false; + rc = col_rel_cow_unshare_publish_impl(r, new_cap, false, false, NULL, + &single.lease); + int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); + return rc ? rc : finish_rc; } static int -col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, +col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, bool defer_metadata_retirement, - col_rel_cow_deferred_t *deferred_old) + col_rel_cow_deferred_t *deferred_old, + wl_columnar_relation_mutation_lease_t *lease) { col_rel_payload_txn_t pending; wl_columnar_memory_reservation_t previous; @@ -2066,6 +2126,9 @@ col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, uint32_t capacity; uint64_t ledger_before; + if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r))) + return EINVAL; wl_columnar_memory_reservation_init(&previous); if (!r || !r->col_shared) @@ -2075,8 +2138,8 @@ col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, if (deferred_old) *deferred_old = (col_rel_cow_deferred_t){ 0 }; if (capacity > r->capacity) - return col_rel_grow_owned_transition_impl(r, capacity, - defer_alias_release); + return col_rel_grow_owned_transition_publish_impl(r, capacity, + defer_alias_release, lease); pending_rc = col_rel_reserve_transition(r, capacity, r->timestamps ? r->timestamp_capacity : 0, false, &pending, &new_bytes); @@ -2122,9 +2185,19 @@ col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, col_rel_retire_payload_credit(r); col_rel_ledger_reconcile(r, ledger_before); /* Private columns replaced the borrowed view: one storage epoch. */ - if (!defer_alias_release) - (void)col_rel_storage_alias_release(r); - wl_columnar_relation_touch_storage(r); + if (lease) { + if (!defer_alias_release) { + /* Publication cannot fail from here. Invalid authority means a + * broken invariant, never a recoverable partial-success return. */ + if (col_rel_storage_alias_release_locked(r, lease)) + abort(); + wl_columnar_relation_touch_storage(r); + } + } else { + if (!defer_alias_release) + (void)col_rel_storage_alias_release(r); + wl_columnar_relation_touch_storage(r); + } return 0; fail: @@ -2136,8 +2209,11 @@ col_rel_cow_unshare_impl(col_rel_t *r, uint32_t new_cap, return ENOMEM; } +/* Temporary legacy entry point for hash dedup, kway merge, and + * incremental delta consolidation in merge.c. Their complete transactions + * migrate in 2B/the batch unit; a raw writer is not a mutation-set lease. */ int -col_rel_cow_unshare_with_source_writer(col_rel_t *r, +col_rel_cow_unshare_legacy_with_source_writer(col_rel_t *r, const wl_columnar_source_access_writer_t *writer) { col_rel_t *owner = NULL; @@ -2152,7 +2228,19 @@ col_rel_cow_unshare_with_source_writer(col_rel_t *r, || writer->identity != (uintptr_t)writer || !wl_columnar_source_access_writer_thread_equal(writer)) return EINVAL; - return col_rel_cow_unshare_impl(r, 0, true, false, NULL); + return col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); +} + +/* A raw writer authorizes canonical non-detaching work only. */ +int +col_rel_cow_unshare_with_source_writer(col_rel_t *r, + const wl_columnar_source_access_writer_t *writer) +{ + col_rel_t *owner = NULL; + if (!r || col_rel_storage_owner_resolve(r, &owner) || owner != r + || r->col_shared) + return EINVAL; + return col_rel_cow_unshare_legacy_with_source_writer(r, writer); } int @@ -3968,7 +4056,7 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, /* Ownership transitions stage columns and timestamps together; * admission happens before any source buffer is copied and the * helper publishes the storage generation on success. */ - int transition_rc = col_rel_grow_owned_transition_impl(r, + int transition_rc = col_rel_grow_owned_transition_legacy_impl(r, new_cap, true); if (transition_rc != 0) { rc = transition_rc; @@ -4030,7 +4118,7 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, /* A shared view can still have spare capacity. Privatize it before the * in-place row write even when no capacity growth is needed. */ if (r->col_shared) { - rc = col_rel_cow_unshare_impl(r, 0, true, false, NULL); + rc = col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); if (rc != 0) goto release_writer; alias_release_pending = true; @@ -4273,14 +4361,14 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, if (r->storage_owner == r && col_rel_storage_alias_borrow_count(r) > 0) return EBUSY; - int rc = col_rel_grow_owned_transition_impl(r, r->capacity, + int rc = col_rel_grow_owned_transition_legacy_impl(r, r->capacity, true); if (rc == 0) *out_alias_release_pending = alias_release_pending; return rc; } if (r->col_shared && required <= r->capacity) { - int rc = col_rel_cow_unshare_impl(r, 0, true, false, NULL); + int rc = col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); if (rc == 0) *out_alias_release_pending = alias_release_pending; return rc; @@ -4298,7 +4386,7 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, new_cap *= 2u; } if (r->col_shared || r->arena_owned) { - int rc = col_rel_grow_owned_transition_impl(r, new_cap, true); + int rc = col_rel_grow_owned_transition_legacy_impl(r, new_cap, true); if (rc == 0) *out_alias_release_pending = alias_release_pending; return rc; @@ -4660,7 +4748,9 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, rc = col_rel_source_writer_acquire(dst, &writer); if (rc != 0) return rc; - rc = col_rel_cow_unshare(dst, 0); + /* Explicit temporary legacy path: append_all already owns its + * source writer and migrates as a complete read-source unit. */ + rc = col_rel_cow_unshare_legacy_impl(dst, 0, false, false, NULL); if (wl_columnar_source_access_writer_release(&writer) != 0 && rc == 0) rc = EINVAL; @@ -4886,7 +4976,7 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, } else if (dst->col_shared || dst->arena_owned || (dst->timestamps && new_nrows > dst->timestamp_capacity)) { - rc = col_rel_grow_owned_transition_impl(dst, new_cap, true); + rc = col_rel_grow_owned_transition_legacy_impl(dst, new_cap, true); if (rc != 0) goto cleanup; transitioned = true; @@ -4909,7 +4999,7 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, /* Bulk append also mutates spare capacity, so a shared view must be * privatized even when the destination does not grow. */ if (dst->col_shared) { - rc = col_rel_cow_unshare_impl(dst, 0, true, false, NULL); + rc = col_rel_cow_unshare_legacy_impl(dst, 0, true, false, NULL); if (rc != 0) goto cleanup; alias_release_pending = true; @@ -9588,7 +9678,7 @@ col_rel_radix_sort_impl(col_rel_t *r, uint32_t start_row, uint32_t nrows, * and releases it afterwards (or hands it back). Releasing inside * the COW and then rolling back would restore a borrowed view whose * borrow the owner no longer counts. */ - rc = col_rel_cow_unshare_impl(r, 0, true, true, &deferred); + rc = col_rel_cow_unshare_legacy_impl(r, 0, true, true, &deferred); if (rc != 0) goto cleanup_workspace; #ifdef WL_TEST_CONSOLIDATE_HOOK From 12c5bf3b4e8dc4512afb51fa6994c16054833441 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 13:06:00 +0900 Subject: [PATCH 3/9] fix(columnar): lease public row append through publication --- tests/test_relation_generations.c | 282 +++++++++++++++++++++++++++++- wirelog/columnar/internal.h | 8 +- wirelog/columnar/relation.c | 116 ++++++++++-- 3 files changed, 388 insertions(+), 18 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index d4da6f70..51d651a7 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -1734,7 +1734,7 @@ test_source_reader_blocks_append_row(void) view_reserved = view->retained_reserved_bytes; view_aliases = view->storage_alias_borrows; #ifdef WL_TEST_APPEND_HOOK - append_hook_owner = source; + append_hook_owner = view; append_hook_source = source; append_hook_expected = view; append_hook_called = false; @@ -1744,7 +1744,7 @@ test_source_reader_blocks_append_row(void) append_hook_source_expected_rows = source_rows; append_hook_expected_capacity = view_capacity; append_hook_source_columns = source_columns; - append_hook_source_aliases = source_aliases; + append_hook_source_aliases = source_aliases - 1; append_hook_view_generation = view_view; append_hook_storage_generation = view_storage; append_hook_reader_rc = 0; @@ -1831,7 +1831,7 @@ test_source_reader_blocks_append_row(void) view_reserved = view->retained_reserved_bytes; view_aliases = view->storage_alias_borrows; #ifdef WL_TEST_APPEND_HOOK - append_hook_owner = source; + append_hook_owner = view; append_hook_source = source; append_hook_expected = view; append_hook_called = false; @@ -1841,7 +1841,7 @@ test_source_reader_blocks_append_row(void) append_hook_source_expected_rows = source_rows; append_hook_expected_capacity = view_capacity; append_hook_source_columns = source_columns; - append_hook_source_aliases = source_aliases; + append_hook_source_aliases = source_aliases - 1; append_hook_view_generation = view_view; append_hook_storage_generation = view_storage; append_hook_reader_rc = 0; @@ -4750,6 +4750,277 @@ mutation_payload_unchanged(const col_rel_t *rel, const col_rel_t *before) memory_order_acquire); } +static void +test_public_append_mutation_admission(void) +{ + for (unsigned full = 0; full < 2; full++) { + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + int64_t value = 37; + CHECK(root && alias, "public append admission relations"); + uint32_t rows = full ? COL_REL_INIT_CAP : 1; + for (uint32_t i = 0; i < rows; i++) + CHECK(col_rel_append_row(root, &value) == 0, + "public append admission source rows"); + CHECK(col_rel_enable_timestamps(root) == 0 + && col_rel_install_shared_view(alias, root) == 0, + "public append admission shared timestamps"); + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 20, .usable_bytes = 1u << 20, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(alias, ref) == 0, + "public append admission governor"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t root_before = *root, alias_before = *alias; + root->memory_budget_denial_pending = true; + CHECK(col_rel_append_row(root, &value) == EBUSY + && root->memory_budget_denial_pending + && mutation_payload_unchanged(root, &root_before), + "public root append rejects live alias even with spare capacity"); + alias->memory_budget_denial_pending = true; + rebind_reader_pause_t pause = { .relation = alias }; + atomic_init(&pause.ready, false); + atomic_init(&pause.proceed, false); + rebind_reader_pause = &pause; + wl_columnar_relation_test_reader_after_descriptor = + rebind_pause_after_descriptor; + wl_thread_t thread; + int start_rc = wl_thread_create(&thread, rebind_paused_reader, &pause); + if (start_rc != 0) { + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + } + CHECK(start_rc == 0, "public append paused descriptor reader starts"); + while (!atomic_load_explicit(&pause.ready, memory_order_acquire)) + alias_accounting_yield(); + int paused_rc = col_rel_append_row(alias, &value); + bool paused_unchanged = mutation_payload_unchanged(alias, &alias_before) + && alias->memory_budget_denial_pending + && wl_columnar_memory_reserved(governor) == credit; + atomic_store_explicit(&pause.proceed, true, memory_order_release); + int join_rc = wl_thread_join(&thread); + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + CHECK(paused_rc == EBUSY && paused_unchanged && join_rc == 0 && + pause.rc == 0, + "public append descriptor denial preserves state and stale evidence"); + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "public append peer owner reader"); + int peer_rc = col_rel_append_row(alias, &value); + bool peer_unchanged = mutation_payload_unchanged(alias, &alias_before) + && alias->memory_budget_denial_pending; + CHECK(col_rel_source_reader_release(&peer) == 0 + && peer_rc == EBUSY && peer_unchanged, + "public append owner phase denial is unchanged"); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "public append budget blocker"); + alias->memory_budget_denial_pending = false; + CHECK(col_rel_append_row(alias, &value) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && wl_columnar_memory_reserved(governor) == resolution.usable_bytes, + "public alias append quota failure retains payload and credit"); + CHECK(wl_columnar_memory_release(&blocker), + "public append unblock budget"); + for (unsigned failure = 0; failure < 2; failure++) { + alias->memory_budget_denial_pending = true; + if (failure == 0) + wl_columnar_relation_test_fail_next_prepare_resize(); + else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_append_row(alias, &value) == ENOMEM + && !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && wl_columnar_memory_reserved(governor) == credit, + "public alias append nonbudget failures roll back under lease"); + } + CHECK(col_rel_append_row(alias, &value) == 0 + && alias->storage_owner == alias && !alias->col_shared + && alias->nrows == rows + 1 && alias->columns[0][rows] == value + && alias->storage_generation == alias_before.storage_generation + 1 + && alias->view_generation == alias_before.view_generation + 1 + && col_rel_storage_alias_borrow_count(root) == 0 + && root->nrows == rows && root->columns[0][0] == value + && alias->timestamps[rows].iteration == 0 + && alias->timestamps[rows].multiplicity == 0, + "public alias append retry detaches before row/view publication"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "public append admission cleanup"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_public_append_overlapping_input(void) +{ + for (unsigned wide = 0; wide < 2; wide++) { + uint32_t width = wide ? 32 : 1; + col_rel_t *rel = track_relation(col_rel_new_auto("append_overlap", + width)); + CHECK(rel, "public append overlap relation"); + int64_t row[32]; + for (uint32_t i = 0; i < COL_REL_INIT_CAP; i++) { + for (uint32_t c = 0; c < width; c++) + row[c] = (int64_t)i * 1000 + c; + CHECK(col_rel_append_row(rel, row) == 0, + "public overlap source rows"); + } + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 22, .usable_bytes = 1u << 22, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(rel, ref) == 0, + "public append overlap governed relation"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t before = *rel; + const int64_t *inside = &rel->columns[0][3]; + int64_t expected[32]; + memcpy(expected, inside, width * sizeof(*inside)); + if (wide) { + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "wide staging budget blocker"); + CHECK(col_rel_append_row(rel, inside) == ENOMEM + && rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == + resolution.usable_bytes, + "wide scratch admission failure is transactional and typed"); + CHECK(wl_columnar_memory_release(&blocker), "wide staging unblock"); +#ifdef WL_TEST_ALLOC_WRAP + allocation_calls = 0; + allocation_fail_at = 0; + int failed_rc = col_rel_append_row(rel, inside); + allocation_fail_at = -1; + CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == credit, + "wide staging malloc failure returns all scratch credit"); +#endif + } + CHECK(col_rel_append_row(rel, inside) == 0 + && rel->nrows == before.nrows + 1 + && rel->storage_generation == before.storage_generation + 1, + "overlapping input survives owned growth and retired old column"); + for (uint32_t c = 0; c < width; c++) + CHECK(rel->columns[c][before.nrows] == expected[c], + "whole overlapping tuple staged before replacement"); + /* Spare-capacity append still stages overlapping input; an external + * tuple uses no scratch and leaves a malloc fault unconsumed. */ + if (wide) { + credit = wl_columnar_memory_reserved(governor); +#ifdef WL_TEST_ALLOC_WRAP + allocation_calls = 0; + allocation_fail_at = 0; + int external_rc = col_rel_append_row(rel, row); + long external_calls = allocation_calls; + int inside_rc = col_rel_append_row(rel, rel->columns[0]); + allocation_fail_at = -1; + CHECK(external_rc == 0 && external_calls == 0 && inside_rc == ENOMEM + && !rel->memory_budget_denial_pending + && wl_columnar_memory_reserved(governor) == credit, + "no-overlap append does not consume a wide-scratch malloc fault"); +#endif + uint32_t index = rel->nrows; + memcpy(expected, rel->columns[0], width * sizeof(*inside)); + CHECK(col_rel_append_row(rel, rel->columns[0]) == 0 + && wl_columnar_memory_reserved(governor) == credit, + "wide overlapping spare append returns its temporary reservation"); + for (uint32_t c = 0; c < width; c++) + CHECK(rel->columns[c][index] == expected[c], + "wide overlapping spare append preserves tuple"); + } + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "public append overlap cleanup returns all credit"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_public_append_edges_and_locked_compatibility(void) +{ + col_rel_t *rel = new_relation(); + int64_t row = 43; + CHECK(rel && col_rel_append_row(rel, &row) == 0, + "public append edge setup"); + rel->memory_budget_denial_pending = true; + CHECK(col_rel_append_row(rel, + NULL) == EINVAL && rel->memory_budget_denial_pending, + "null input rejects before admission without clearing evidence"); + col_rel_t before = *rel; + CHECK(col_rel_append_row(rel, + (const int64_t *)(UINTPTR_MAX - 3)) == EOVERFLOW + && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, + &before), + "public append checks input uintptr overflow before loading row"); + uint32_t old_rows = rel->nrows, old_capacity = rel->capacity; + rel->nrows = rel->capacity = UINT32_MAX; + col_rel_t overflow_before = *rel; + CHECK(col_rel_append_row(rel, &row) == EOVERFLOW + && mutation_payload_unchanged(rel, &overflow_before), + "public append rejects row-count overflow without publication"); + rel->nrows = old_rows; rel->capacity = old_capacity; + rel->timestamp_capacity = 1; + CHECK(col_rel_append_row(rel, &row) == EINVAL, + "public append rejects inconsistent timestamp shape"); + rel->timestamp_capacity = 0; + cleanup_relations(); + rel = new_relation(); + wirelog_column_type_t type = WIRELOG_TYPE_FLOAT; + CHECK(rel && col_rel_set_column_types(rel, &type, 1) == 0, + "public append float shape"); + int64_t invalid = (int64_t)UINT64_C(0x7ff8000000000000); + before = *rel; + CHECK(col_rel_append_row(rel, &invalid) == EINVAL + && mutation_payload_unchanged(rel, &before), + "public append invalid float precedes publication"); + cleanup_relations(); + col_rel_t *root = new_relation(), *alias = new_relation(); + CHECK(root && alias && col_rel_append_row(root, &row) == 0 + && col_rel_install_shared_view(alias, root) == 0, + "locked append compatibility setup"); + wl_columnar_source_access_writer_t writer = { 0 }; + CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, + "locked append legacy source writer"); + CHECK(col_rel_append_row_locked(alias, &row, &writer) == 0 + && alias->storage_owner == alias && alias->nrows == 2 + && col_rel_storage_alias_borrow_count(root) == 0, + "locked append retains legacy admitted alias behavior"); + CHECK(wl_columnar_source_access_writer_release(&writer) == 0, + "locked append compatibility release"); + cleanup_relations(); + col_rel_t *empty = track_relation(col_rel_new_auto("append_nullary", 0)); + CHECK(empty && col_rel_append_row(empty, &row) == 0 && empty->nrows == 1, + "public append initialized nullary tuple"); + cleanup_relations(); +} + static void test_set_cow_denial_provenance(void) { @@ -5124,6 +5395,9 @@ main(void) { #ifdef WL_TEST_RELATION_RESIZE_HOOK test_set_cow_descriptor_and_owner_admission(); + test_public_append_mutation_admission(); + test_public_append_overlapping_input(); + test_public_append_edges_and_locked_compatibility(); test_set_cow_denial_provenance(); test_empty_append_all_legacy_cow_compatibility(); test_leased_rebind_descriptor_exclusion(); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 5f34d08e..4c2adc0b 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2868,8 +2868,12 @@ int wl_col_rel_inline_project_column(col_rel_t *dst, uint32_t dst_row, const col_rel_t *src, uint32_t src_row, uint32_t logical_col); -/* Row append and locked row reservation reset transient denial evidence - * under the source writer. Actual budget refusal retains legacy ENOMEM plus +/* Public row append owns descriptor and canonical source writers throughout + * publication. Input tuples may overlap column storage; the public path + * stages the complete tuple before any replacement. Caller keeps other input + * buffers stable for the call. Locked append/reservation retain their legacy + * source-writer contract. Outer attempts reset transient denial evidence. + * Actual budget refusal retains legacy ENOMEM plus * memory_budget_denial_pending; allocation ENOMEM, EOVERFLOW and EINVAL are * distinct. Lower nested admission helpers do not reset outer evidence. */ int diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 3983884f..2c5da384 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -3989,13 +3989,25 @@ col_rel_apply_compound_schema(col_rel_t *r, static int col_rel_append_row_impl(col_rel_t *r, const int64_t *row, - wl_columnar_source_access_writer_t *writer, bool writer_held) + wl_columnar_source_access_writer_t *writer, bool writer_held, + wl_columnar_relation_mutation_lease_t *lease) { bool alias_release_pending = false; + int64_t inline_row[16]; + int64_t *staged_row = NULL; + wl_columnar_memory_reservation_t row_admission; + bool row_credit_held = false; int rc; + wl_columnar_memory_reservation_init(&row_admission); + if (!r || !row || !writer) return EINVAL; + if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r))) + return EINVAL; + if (lease && (!col_rel_timestamp_shape_valid(r) || r->nrows > r->capacity)) + return EINVAL; /* Do all structural validation before any resize, COW, or timestamp * publication. In particular, a zero-column or partially initialized * relation must fail without consuming capacity or a generation epoch. */ @@ -4011,6 +4023,27 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, for (uint32_t c = 0; c < r->ncols; c++) if ((!r->columns || !r->columns[c]) && !empty_unallocated) return EINVAL; + bool row_overlaps = false; + size_t row_bytes = 0; + if (lease) { + if (r->nrows == UINT32_MAX) + return EOVERFLOW; + uint64_t bytes = (uint64_t)r->ncols * sizeof(*row); + uint64_t column_bytes = (uint64_t)r->capacity * sizeof(*row); + uintptr_t row_start = (uintptr_t)row; + if (bytes > SIZE_MAX || bytes > UINTPTR_MAX - row_start) + return EOVERFLOW; + row_bytes = (size_t)bytes; + for (uint32_t c = 0; r->columns && c < r->ncols; c++) { + uintptr_t column_start = (uintptr_t)r->columns[c]; + if (column_bytes > UINTPTR_MAX - column_start) + return EOVERFLOW; + if (row_bytes && column_bytes + && row_start < column_start + column_bytes + && column_start < row_start + row_bytes) + row_overlaps = true; + } + } if (r->column_types) { for (uint32_t c = 0; c < r->ncols; c++) { if (r->column_types[c] == WIRELOG_TYPE_FLOAT @@ -4023,13 +4056,50 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, * but before any resize, COW, timestamp, value, or generation mutation. * A reader on an alias therefore excludes append on every view of the * same backing storage. */ - if (!writer_held) { + if (!lease && !writer_held) { rc = col_rel_source_writer_acquire(r, writer); if (rc != 0) return rc; } - r->memory_budget_denial_pending = false; + if (!lease) + r->memory_budget_denial_pending = false; + + /* A valid input row may be a slice of a column that publication retires. + * Capture the complete tuple while its descriptor and source are closed, + * even when this append currently needs no replacement. */ + if (row_overlaps) { + if (r->ncols <= sizeof(inline_row) / sizeof(inline_row[0])) { + memcpy(inline_row, row, row_bytes); + row = inline_row; + } else { + if (r->memory_governor) { + wl_columnar_memory_admission_status_t status + = wl_columnar_memory_reserve_checked( + wl_columnar_memory_governor_ref_get(r->memory_governor), + row_bytes, &row_admission); + if (status != WL_COLUMNAR_MEMORY_ADMISSION_OK + && status != WL_COLUMNAR_MEMORY_ADMISSION_ADVISORY) { + if (status == WL_COLUMNAR_MEMORY_ADMISSION_DENIED) { + r->memory_budget_denial_pending = true; + rc = ENOMEM; + } else + rc = status == WL_COLUMNAR_MEMORY_ADMISSION_OVERFLOW + ? EOVERFLOW : EINVAL; + goto release_writer; + } + /* Rejected tokens retain bytes but own no governor credit. */ + row_credit_held = true; + if (!wl_columnar_memory_commit(&row_admission, &row_admission)) + goto enomem; + } + staged_row = malloc(row_bytes); + if (!staged_row) + goto enomem; + memcpy(staged_row, row, row_bytes); + row = staged_row; + } + } bool needs_resize = r->nrows >= r->capacity; bool timestamp_short = r->timestamps @@ -4056,8 +4126,10 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, /* Ownership transitions stage columns and timestamps together; * admission happens before any source buffer is copied and the * helper publishes the storage generation on success. */ - int transition_rc = col_rel_grow_owned_transition_legacy_impl(r, - new_cap, true); + int transition_rc = lease + ? col_rel_grow_owned_transition_publish_impl(r, new_cap, false, + lease) + : col_rel_grow_owned_transition_legacy_impl(r, new_cap, true); if (transition_rc != 0) { rc = transition_rc; goto release_writer; @@ -4118,7 +4190,9 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, /* A shared view can still have spare capacity. Privatize it before the * in-place row write even when no capacity growth is needed. */ if (r->col_shared) { - rc = col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); + rc = lease + ? col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease) + : col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); if (rc != 0) goto release_writer; alias_release_pending = true; @@ -4131,8 +4205,13 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, memset(&r->timestamps[r->nrows], 0, sizeof(col_delta_timestamp_t)); /* Structural and value validation above makes this raw copy * non-failing; retain the check as a defensive invariant. */ - if (col_rel_row_copy_in_raw(r, r->nrows, row) != 0) + if (col_rel_row_copy_in_raw(r, r->nrows, row) != 0) { + /* All shape/value checks and input staging preceded publication. + * A leased publication cannot report recoverable partial failure. */ + if (lease) + abort(); goto einval; + } if (r->column_types) { for (uint32_t c = 0; c < r->ncols; c++) { if (r->column_types[c] == WIRELOG_TYPE_FLOAT @@ -4151,12 +4230,15 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, einval: rc = EINVAL; release_writer: - if (alias_release_pending) { + free(staged_row); + if (row_credit_held) + col_rel_release_reservation_or_abort(&row_admission); + if (!lease && alias_release_pending) { int alias_rc = col_rel_storage_alias_release(r); if (alias_rc != 0 && rc == 0) rc = alias_rc; } - if (!writer_held + if (!lease && !writer_held && wl_columnar_source_access_writer_release(writer) != 0 && rc == 0) rc = EINVAL; return rc; @@ -4165,15 +4247,25 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, int col_rel_append_row(col_rel_t *r, const int64_t *row) { - wl_columnar_source_access_writer_t writer = { 0 }; - return col_rel_append_row_impl(r, row, &writer, false); + wl_columnar_relation_mutation_single_t single = { 0 }; + if (!r || !row) + return EINVAL; + int rc = col_rel_mutation_single_acquire(r, &single); + if (rc) + return rc; + /* Outer attempt provenance starts only after both admissions succeed. */ + r->memory_budget_denial_pending = false; + rc = col_rel_append_row_impl(r, row, &single.owner.writer, true, + &single.lease); + int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); + return rc ? rc : finish_rc; } int col_rel_append_row_locked(col_rel_t *r, const int64_t *row, wl_columnar_source_access_writer_t *writer) { - return col_rel_append_row_impl(r, row, writer, true); + return col_rel_append_row_impl(r, row, writer, true, NULL); } static int From 7f08f86bb67e763cba60e572ceeeefc553b85fee Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 13:38:42 +0900 Subject: [PATCH 4/9] fix(columnar): lease public batch append through publication --- tests/test_relation_generations.c | 540 ++++++++++++++++++++++++++++++ wirelog/columnar/relation.c | 202 ++++++++--- 2 files changed, 695 insertions(+), 47 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 51d651a7..905ad0ab 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -4962,6 +4962,541 @@ test_public_append_overlapping_input(void) } } +static void +test_public_batch_mutation_admission(void) +{ + for (unsigned full = 0; full < 2; full++) { + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + int64_t value = 37; + CHECK(root && alias, "public append admission relations"); + uint32_t rows = full ? COL_REL_INIT_CAP : 1; + for (uint32_t i = 0; i < rows; i++) + CHECK(col_rel_append_row(root, &value) == 0, + "public append admission source rows"); + CHECK(col_rel_enable_timestamps(root) == 0 + && col_rel_install_shared_view(alias, root) == 0, + "public append admission shared timestamps"); + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 20, .usable_bytes = 1u << 20, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(alias, ref) == 0, + "public append admission governor"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t root_before = *root, alias_before = *alias; + root->memory_budget_denial_pending = true; + CHECK(col_rel_append_rows_atomic(root, (int64_t[]){37, 37}, 2, 1, + NULL) == EBUSY + && root->memory_budget_denial_pending + && mutation_payload_unchanged(root, &root_before), + "public root append rejects live alias even with spare capacity"); + alias->memory_budget_denial_pending = true; + rebind_reader_pause_t pause = { .relation = alias }; + atomic_init(&pause.ready, false); + atomic_init(&pause.proceed, false); + rebind_reader_pause = &pause; + wl_columnar_relation_test_reader_after_descriptor = + rebind_pause_after_descriptor; + wl_thread_t thread; + int start_rc = wl_thread_create(&thread, rebind_paused_reader, &pause); + if (start_rc != 0) { + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + } + CHECK(start_rc == 0, "public append paused descriptor reader starts"); + while (!atomic_load_explicit(&pause.ready, memory_order_acquire)) + alias_accounting_yield(); + int paused_rc = col_rel_append_rows_atomic(alias, (int64_t[]){37, 37}, + 2, 1, NULL); + bool paused_unchanged = mutation_payload_unchanged(alias, &alias_before) + && alias->memory_budget_denial_pending + && wl_columnar_memory_reserved(governor) == credit; + atomic_store_explicit(&pause.proceed, true, memory_order_release); + int join_rc = wl_thread_join(&thread); + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + CHECK(paused_rc == EBUSY && paused_unchanged && join_rc == 0 && + pause.rc == 0, + "public append descriptor denial preserves state and stale evidence"); + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "public append peer owner reader"); + int peer_rc = col_rel_append_rows_atomic(alias, (int64_t[]){37, 37}, 2, + 1, NULL); + bool peer_unchanged = mutation_payload_unchanged(alias, &alias_before) + && alias->memory_budget_denial_pending; + CHECK(col_rel_source_reader_release(&peer) == 0 + && peer_rc == EBUSY && peer_unchanged, + "public append owner phase denial is unchanged"); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "public append budget blocker"); + alias->memory_budget_denial_pending = false; + CHECK(col_rel_append_rows_atomic(alias, (int64_t[]){37, 37}, 2, 1, + NULL) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && wl_columnar_memory_reserved(governor) == resolution.usable_bytes, + "public alias append quota failure retains payload and credit"); + CHECK(wl_columnar_memory_release(&blocker), + "public append unblock budget"); + for (unsigned failure = 0; failure < 2; failure++) { + alias->memory_budget_denial_pending = true; + if (failure == 0) + wl_columnar_relation_test_fail_next_prepare_resize(); + else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_append_rows_atomic(alias, (int64_t[]){37, 37}, 2, 1, + NULL) == ENOMEM + && !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && wl_columnar_memory_reserved(governor) == credit, + "public alias append nonbudget failures roll back under lease"); + } + CHECK(col_rel_append_rows_atomic(alias, (int64_t[]){37, 37}, 2, 1, + NULL) == 0 + && alias->storage_owner == alias && !alias->col_shared + && alias->nrows == rows + 2 && alias->columns[0][rows] == value + && alias->columns[0][rows + 1] == value + && alias->storage_generation == alias_before.storage_generation + 1 + && alias->view_generation == alias_before.view_generation + 1 + && col_rel_storage_alias_borrow_count(root) == 0 + && root->nrows == rows && root->columns[0][0] == value + && alias->timestamps[rows].iteration == 0 + && alias->timestamps[rows].multiplicity == 0, + "public alias append retry detaches before row/view publication"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "public append admission cleanup"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_public_batch_overlapping_input(void) +{ + for (unsigned wide = 0; wide < 2; wide++) { + uint32_t width = wide ? 32 : 1; + col_rel_t *rel = track_relation(col_rel_new_auto("append_overlap", + width)); + CHECK(rel, "public append overlap relation"); + int64_t row[32]; + for (uint32_t i = 0; i < COL_REL_INIT_CAP; i++) { + for (uint32_t c = 0; c < width; c++) + row[c] = (int64_t)i * 1000 + c; + CHECK(col_rel_append_rows_atomic(rel, row, 1, width, NULL) == 0, + "public overlap source rows"); + } + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 22, .usable_bytes = 1u << 22, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(rel, ref) == 0, + "public append overlap governed relation"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t before = *rel; + const int64_t *inside = &rel->columns[0][wide ? 0 : 3]; + int64_t expected[64]; + memcpy(expected, inside, 2 * width * sizeof(*inside)); + bool denied = false; + if (wide) { + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "wide staging budget blocker"); + CHECK(col_rel_append_rows_atomic(rel, inside, 2, width, + &denied) == ENOMEM + && denied && rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == + resolution.usable_bytes, + "wide scratch admission failure is transactional and typed"); + CHECK(wl_columnar_memory_release(&blocker), "wide staging unblock"); +#ifdef WL_TEST_ALLOC_WRAP + allocation_calls = 0; + allocation_fail_at = 0; + int failed_rc = col_rel_append_rows_atomic(rel, inside, 2, width, + &denied); + allocation_fail_at = -1; + CHECK(failed_rc == ENOMEM && !denied && + !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == credit, + "wide staging malloc failure returns all scratch credit"); +#endif + } + CHECK(col_rel_append_rows_atomic(rel, inside, 2, width, &denied) == 0 + && rel->nrows == before.nrows + 2 + && rel->storage_generation == before.storage_generation + 1, + "overlapping input survives owned growth and retired old column"); + for (uint32_t c = 0; c < 2 * width; c++) + CHECK(rel->columns[c % width][before.nrows + c / width] == + expected[c], + "whole overlapping tuple staged before replacement"); + /* Spare-capacity append still stages overlapping input; an external + * tuple uses no scratch and leaves a malloc fault unconsumed. */ + if (wide) { + credit = wl_columnar_memory_reserved(governor); +#ifdef WL_TEST_ALLOC_WRAP + allocation_calls = 0; + allocation_fail_at = 0; + int external_rc = col_rel_append_rows_atomic(rel, row, 1, width, + NULL); + long external_calls = allocation_calls; + int inside_rc = col_rel_append_rows_atomic(rel, rel->columns[0], 2, + width, NULL); + allocation_fail_at = -1; + CHECK(external_rc == 0 && external_calls == 0 && inside_rc == ENOMEM + && !rel->memory_budget_denial_pending + && wl_columnar_memory_reserved(governor) == credit, + "no-overlap append does not consume a wide-scratch malloc fault"); +#endif + uint32_t index = rel->nrows; + memcpy(expected, rel->columns[0], 2 * width * sizeof(*inside)); + CHECK(col_rel_append_rows_atomic(rel, rel->columns[0], 2, width, + NULL) == 0 + && wl_columnar_memory_reserved(governor) == credit, + "wide overlapping spare append returns its temporary reservation"); + for (uint32_t c = 0; c < 2 * width; c++) + CHECK(rel->columns[c % width][index + c / width] == expected[c], + "wide overlapping spare append preserves tuple"); + } + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "public append overlap cleanup returns all credit"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_public_batch_short_timestamp_alias(void) +{ + col_rel_t *root = new_relation(), *alias = new_relation(); + col_rel_t *sibling = new_relation(); + CHECK(root && alias && sibling, "short timestamp batch relations"); + int64_t source_rows[] = { 31, 32, 33 }; + CHECK(col_rel_append_rows_atomic(root, source_rows, 3, 1, NULL) == 0 + && col_rel_enable_timestamps(root) == 0, + "short timestamp batch source with spare columns"); + for (uint32_t i = 0; i < root->nrows; i++) { + root->timestamps[i].iteration = i + 7; + root->timestamps[i].multiplicity = i + 2; + } + CHECK(col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "short timestamp batch installs two aliases"); + col_delta_timestamp_t *short_timestamps = realloc(alias->timestamps, + (size_t)alias->nrows * sizeof(*alias->timestamps)); + CHECK(short_timestamps, + "short timestamp batch shrinks physical allocation"); + alias->timestamps = short_timestamps; + alias->timestamp_capacity = alias->nrows; + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 22, .usable_bytes = 1u << 22, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(alias, ref) == 0 + && col_rel_attach_memory_governor(sibling, ref) == 0, + "short timestamp batch governs complete physical shape"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t root_before = *root, alias_before = *alias; + col_rel_t sibling_before = *sibling; + col_delta_timestamp_t original_timestamps[3]; + memcpy(original_timestamps, root->timestamps, sizeof(original_timestamps)); + int64_t batch[] = { 81, 82 }; + CHECK(alias->nrows + 2 <= alias->capacity + && alias->timestamp_capacity == alias->nrows + && col_rel_storage_alias_borrow_count(root) == 2, + "short timestamp batch selects spare-column timestamp replacement"); + bool denied = true; + for (unsigned failure = 0; failure < 2; failure++) { + alias->memory_budget_denial_pending = true; + if (failure == 0) + wl_columnar_relation_test_fail_next_prepare_resize(); + else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_append_rows_atomic(alias, batch, 2, 1, &denied) == ENOMEM + && !denied && !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &alias_before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) == credit + && memcmp(root->columns[0], source_rows, sizeof(source_rows)) == 0 + && memcmp(root->timestamps, original_timestamps, + sizeof(original_timestamps)) == 0 + && memcmp(alias->timestamps, original_timestamps, + sizeof(original_timestamps)) == 0, + "short timestamp alias failures roll back batch, bindings and credit"); + } + CHECK(col_rel_append_rows_atomic(alias, batch, 2, 1, &denied) == 0 + && !denied && alias->storage_owner == alias && !alias->col_shared + && alias->nrows == 5 && alias->capacity == alias_before.capacity + && alias->timestamp_capacity == alias->capacity + && alias->storage_generation == alias_before.storage_generation + 1 + && alias->storage_owner_generation == alias->storage_generation + && alias->view_generation == alias_before.view_generation + 1 + && col_rel_storage_alias_borrow_count(root) == 1, + "short timestamp alias retry detaches one borrow and publishes exact epochs"); + /* The sole intentional root change is the released destination borrow. */ + atomic_store_explicit(&root_before.storage_alias_borrows, 1, + memory_order_relaxed); + CHECK(mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && sibling->storage_owner == root + && sibling->columns[0] == root->columns[0] + && memcmp(root->columns[0], source_rows, sizeof(source_rows)) == 0 + && memcmp(root->timestamps, original_timestamps, + sizeof(original_timestamps)) == 0 + && memcmp(sibling->timestamps, original_timestamps, + sizeof(original_timestamps)) == 0 + && memcmp(alias->columns[0], source_rows, sizeof(source_rows)) == 0 + && memcmp(alias->timestamps, original_timestamps, + sizeof(original_timestamps)) == 0 + && alias->columns[0][3] == batch[0] && alias->columns[0][4] == batch[1] + && alias->timestamps[3].iteration == 0 + && alias->timestamps[3].multiplicity == 0 + && alias->timestamps[4].iteration == 0 + && alias->timestamps[4].multiplicity == 0, + "short timestamp alias retry preserves root, sibling and original provenance"); + uint64_t live_bytes; + CHECK(col_rel_retained_live_bytes(alias, &live_bytes) + && alias->retained_reserved_bytes == live_bytes + && wl_columnar_memory_reserved(governor) == credit + - alias_before.retained_reserved_bytes - + alias_before.metadata_reserved_bytes + + alias->retained_reserved_bytes + alias->metadata_reserved_bytes, + "short timestamp alias retry has exact retained and total governor credit"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "short timestamp alias cleanup returns every reservation"); + wl_columnar_memory_governor_ref_release(ref); +} + +static void +test_public_batch_governed_rollback(void) +{ + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 22, .usable_bytes = 1u << 22, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref, "batch rollback governor"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + int64_t values[2] = { 71, 72 }; + for (unsigned lazy = 0; lazy < 2; lazy++) { + col_rel_t *rel = NULL; + if (lazy) { + CHECK(col_rel_alloc(&rel, "lazy-governed-batch") == 0, + "lazy governed batch descriptor"); + track_relation(rel); + } else { + rel = new_relation(); + CHECK(rel, "owned governed batch descriptor"); + for (uint32_t i = 0; i < COL_REL_INIT_CAP; i++) + CHECK(col_rel_append_row(rel, values) == 0, + "owned rollback fills capacity"); + } + CHECK(col_rel_attach_memory_governor(rel, ref) == 0, + "batch rollback governor attach"); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t before = *rel; + bool denied = false; + atomic_store_explicit(&governor->usable_bytes, credit, + memory_order_release); + int quota_rc = col_rel_append_rows_atomic(rel, values, 2, 1, &denied); + CHECK((quota_rc == ENOMEM || quota_rc == ENOSPC) && denied + && rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && rel->schema_ok == before.schema_ok + && wl_columnar_memory_reserved(governor) == credit, + "governed batch quota refusal leaves schema and whole batch unchanged"); + atomic_store_explicit(&governor->usable_bytes, resolution.usable_bytes, + memory_order_release); + for (unsigned failure = 0; failure < 2; failure++) { + if (failure == 0) { + if (lazy) + wl_columnar_relation_test_fail_next_metadata_alloc(); + else + wl_columnar_relation_test_fail_next_prepare_resize(); + } else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, + &denied) == ENOMEM + && !denied && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && rel->schema_ok == before.schema_ok + && wl_columnar_memory_reserved(governor) == credit, + "governed batch preparation/publication failure returns all credit"); + } + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == 0 + && !denied && rel->nrows == before.nrows + 2 + && rel->columns[0][before.nrows] == values[0] + && rel->columns[0][before.nrows + 1] == values[1] + && rel->view_generation == before.view_generation + (lazy ? 2 : 1) + && rel->storage_generation == before.storage_generation + !lazy, + "governed batch retry publishes every row and exact epochs"); + uint64_t live_bytes; + CHECK(col_rel_retained_live_bytes(rel, &live_bytes) + && rel->retained_reserved_bytes == live_bytes, + "governed batch retained payload credit matches physical shape"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "governed batch rollback cleanup returns all reservations"); + } + wl_columnar_memory_governor_ref_release(ref); +} + +static void +test_public_batch_lazy_and_timestamp_edges(void) +{ + col_rel_t *rel = NULL; + CHECK(col_rel_alloc(&rel, "lazy-batch") == 0, + "batch lazy descriptor allocation"); + track_relation(rel); + rel->column_types = malloc(sizeof(*rel->column_types)); + CHECK(rel->column_types, "prospective float metadata allocation"); + rel->column_types[0] = WIRELOG_TYPE_FLOAT; + rel->schema.n_children = 1; + int64_t values[2] = { 0, (int64_t)UINT64_C(0x7ff8000000000000) }; + bool denied = true; + col_rel_t before = *rel; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == EINVAL + && !denied && !rel->schema_ok && rel->ncols == 0 + && mutation_payload_unchanged(rel, &before), + "lazy batch checks final prospective float before schema publication"); + values[1] = 0; + rel->schema.n_children = 2; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == EINVAL + && !rel->schema_ok, "lazy batch rejects prospective metadata width"); + rel->schema.n_children = 1; + wl_columnar_relation_test_fail_next_metadata_alloc(); + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == ENOMEM + && !denied && !rel->schema_ok + && mutation_payload_unchanged(rel, &before), + "lazy batch allocator failure preserves unpublished schema"); + rel->view_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 2; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == EOVERFLOW + && !rel->schema_ok, "lazy batch reserves both view epochs"); + rel->view_generation = before.view_generation; + wl_columnar_relation_test_fail_next_prepare_resize(); + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == 0 + && !denied && rel->nrows == 2 && rel->ncols == 1 + && rel->view_generation == before.view_generation + 2 + && rel->storage_generation == before.storage_generation, + "lazy batch publishes complete schema then rows without reserve"); + /* The reserve fault must survive successful schema publication. */ + CHECK(col_rel_reserve_capacity_admitted(rel, rel->capacity * 2, + NULL) == ENOMEM, + "lazy batch leaves post-schema reserve fault unconsumed"); + cleanup_relations(); + + CHECK(col_rel_alloc(&rel, "lazy-short-timestamps") == 0, + "lazy timestamp descriptor"); + track_relation(rel); + rel->timestamps = calloc(1, sizeof(*rel->timestamps)); + CHECK(rel->timestamps, "lazy short timestamps allocation"); + rel->timestamp_capacity = 1; + before = *rel; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == 0 + && rel->nrows == 2 && rel->timestamp_capacity >= 2 + && rel->view_generation == before.view_generation + 2 + && rel->timestamps[1].iteration == 0, + "lazy schema prepares short existing timestamps before publication"); + cleanup_relations(); + + rel = new_relation(); + int64_t row = 9; + CHECK(rel && col_rel_append_row(rel, &row) == 0 + && col_rel_enable_timestamps(rel) == 0, + "batch timestamp-only replacement setup"); + rel->timestamp_capacity = rel->nrows; + rel->storage_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 1; + rel->storage_owner_generation = rel->storage_generation; + before = *rel; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == EOVERFLOW + && !denied && mutation_payload_unchanged(rel, &before), + "timestamp-only replacement needs storage epoch before preparation"); + rel->storage_generation = 17; + rel->storage_owner_generation = 17; + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == 0 + && rel->storage_generation == 18 && rel->nrows == 3 + && rel->columns[0][0] == row + && rel->timestamps[1].iteration == 0 + && rel->timestamps[2].multiplicity == 0, + "timestamp-only retry preserves values and publishes storage once"); + before = *rel; + CHECK(col_rel_append_rows_atomic(rel, + (const int64_t *)(UINTPTR_MAX - 3), 2, 1, &denied) == EOVERFLOW + && !denied && mutation_payload_unchanged(rel, &before), + "batch wrapped input fails before dereference"); + int64_t *column = rel->columns[0]; + rel->columns[0] = (int64_t *)(UINTPTR_MAX - 3); + CHECK(col_rel_append_rows_atomic(rel, values, 2, 1, &denied) == EOVERFLOW, + "batch wrapped allocation endpoint fails before dereference"); + rel->columns[0] = column; + before = *rel; + CHECK(col_rel_append_rows_atomic(rel, NULL, 0, 99, &denied) == 0 + && !denied && mutation_payload_unchanged(rel, &before), + "zero batch is a no-op regardless of width"); + cleanup_relations(); + + col_rel_t *root = new_relation(), *alias = new_relation(); + col_rel_t *sibling = new_relation(); + CHECK(root && alias && sibling, "batch sibling relations"); + for (uint32_t i = 0; i < 64; i++) { + row = i; + CHECK(col_rel_append_row(root, &row) == 0, "batch sibling input rows"); + } + CHECK(col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "batch sibling aliases installed"); + col_rel_t root_before = *root, sibling_before = *sibling; + const int64_t *inside = root->columns[0] + 3; + CHECK(col_rel_append_rows_atomic(alias, inside, 2, 1, &denied) == 0 + && !denied && alias->storage_owner == alias + && alias->nrows == 66 && alias->columns[0][64] == 3 + && alias->columns[0][65] == 4 + && col_rel_storage_alias_borrow_count(root) == 1 + && root->columns == root_before.columns && root->nrows == 64 + && sibling->columns == sibling_before.columns + && sibling->storage_owner == root && sibling->columns[0][3] == 3, + "batch overlapping COW releases exactly one of two root borrows"); + cleanup_relations(); +} + static void test_public_append_edges_and_locked_compatibility(void) { @@ -5395,6 +5930,11 @@ main(void) { #ifdef WL_TEST_RELATION_RESIZE_HOOK test_set_cow_descriptor_and_owner_admission(); + test_public_batch_short_timestamp_alias(); + test_public_batch_governed_rollback(); + test_public_batch_lazy_and_timestamp_edges(); + test_public_batch_mutation_admission(); + test_public_batch_overlapping_input(); test_public_append_mutation_admission(); test_public_append_overlapping_input(); test_public_append_edges_and_locked_compatibility(); diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 2c5da384..8dbf4b15 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -3231,7 +3231,8 @@ col_rel_set_schema_impl_capacity(col_rel_t *r, uint32_t ncols, /* Existing timestamps remain owned by the source until publication. */ image.timestamps = r->timestamps; image.timestamp_capacity = r->timestamp_capacity; - if (with_timestamps && image.timestamp_capacity < image.capacity) + if ((with_timestamps || image.timestamps) + && image.timestamp_capacity < image.capacity) image.timestamp_capacity = image.capacity; uint64_t incoming_grid = 0, incoming_scratch = 0; if (!col_rel_payload_private_bytes(r, ncols, image.capacity, @@ -3247,7 +3248,8 @@ col_rel_set_schema_impl_capacity(col_rel_t *r, uint32_t ncols, col_rel_reservation_rollback(&metadata_pending); return EOVERFLOW; } - if (with_timestamps && !image.timestamps && image.timestamp_capacity) { + if ((with_timestamps || image.timestamps) && image.timestamp_capacity + && (!image.timestamps || r->timestamp_capacity < image.capacity)) { uint64_t timestamp_bytes; if (!wl_columnar_memory_size_mul(image.timestamp_capacity, sizeof(col_delta_timestamp_t), ×tamp_bytes) @@ -3297,11 +3299,15 @@ col_rel_set_schema_impl_capacity(col_rel_t *r, uint32_t ncols, goto fail; } } - if (with_timestamps && !image.timestamps && image.capacity > 0) { + if ((with_timestamps || image.timestamps) && image.capacity > 0 + && (!image.timestamps || r->timestamp_capacity < image.capacity)) { image.timestamps = (col_delta_timestamp_t *)calloc(image.capacity, sizeof(*image.timestamps)); if (!image.timestamps) goto fail; + if (r->timestamps && r->nrows) + memcpy(image.timestamps, r->timestamps, + (size_t)r->nrows * sizeof(*image.timestamps)); image.timestamp_capacity = image.capacity; image_owns_timestamps = true; } @@ -3318,6 +3324,8 @@ col_rel_set_schema_impl_capacity(col_rel_t *r, uint32_t ncols, r->columns = image.columns; r->row_scratch = image.row_scratch; r->col_names = image.col_names; + if (image_owns_timestamps) + free(r->timestamps); r->timestamps = image.timestamps; r->timestamp_capacity = image.timestamp_capacity; r->schema = image.schema; @@ -4288,14 +4296,19 @@ col_rel_capacity_for_rows(uint32_t current, uint32_t required, static int wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, wl_columnar_source_access_writer_t *writer, - bool *out_alias_release_pending, bool begin_operation); + bool *out_alias_release_pending, bool begin_operation, + wl_columnar_relation_mutation_lease_t *lease); int col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, uint32_t num_rows, uint32_t num_cols, bool *denied) { - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_release_pending = false; + wl_columnar_relation_mutation_single_t single = { 0 }; + wl_columnar_memory_reservation_t scratch_admission; + int64_t inline_rows[32]; + int64_t *staged_rows = NULL; + bool scratch_credit_held = false; + bool overlaps = false; bool schema_was_unset; uint32_t required_rows; uint32_t capacity; @@ -4314,7 +4327,8 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, &input_bytes) || input_bytes > SIZE_MAX) return EOVERFLOW; - rc = col_rel_source_writer_acquire(r, &writer); + wl_columnar_memory_reservation_init(&scratch_admission); + rc = col_rel_mutation_single_acquire(r, &single); if (rc != 0) return rc; @@ -4322,6 +4336,10 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, * ENOMEM is called a budget denial only if this operation sets the bit. */ r->memory_budget_denial_pending = false; + if (!col_rel_timestamp_shape_valid(r) || r->nrows > r->capacity) { + rc = EINVAL; + goto finish; + } if (num_rows > UINT32_MAX - r->nrows) { rc = EOVERFLOW; goto finish; @@ -4334,7 +4352,8 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, } if (r->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - (schema_was_unset ? 2u : 1u) - || ((required_rows > r->capacity || r->col_shared || r->arena_owned) + || ((required_rows > r->capacity || r->col_shared || r->arena_owned + || (r->timestamps && required_rows > r->timestamp_capacity)) && r->storage_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u)) { rc = EOVERFLOW; goto finish; @@ -4355,24 +4374,91 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, goto finish; } } - if (r->column_types) { - for (uint32_t row = 0; row < num_rows; row++) { - const int64_t *values = rows + (size_t)row * num_cols; - for (uint32_t col = 0; col < r->ncols; col++) { - if (r->column_types[col] == WIRELOG_TYPE_FLOAT - && !wl_columnar_float_bits_valid(values[col])) { - rc = EINVAL; - goto finish; - } + } + /* Arrow metadata describes prospective types even before physical columns + * have been published. Never index an unverified prospective width. */ + if (r->column_types) { + if (schema_was_unset && (r->schema.n_children < 0 + || (uint64_t)r->schema.n_children != num_cols)) { + rc = EINVAL; + goto finish; + } + } + uintptr_t input_start = (uintptr_t)rows; + if (input_bytes > UINTPTR_MAX - input_start) { + rc = EOVERFLOW; + goto finish; + } + uint64_t column_bytes; + if (!wl_columnar_memory_size_mul(r->capacity, sizeof(int64_t), + &column_bytes) || column_bytes > SIZE_MAX) { + rc = EOVERFLOW; + goto finish; + } + for (uint32_t c = 0; r->columns && c < r->ncols; c++) { + uintptr_t column_start = (uintptr_t)r->columns[c]; + if (column_bytes > UINTPTR_MAX - column_start) { + rc = EOVERFLOW; + goto finish; + } + if (input_bytes && column_bytes + && input_start < column_start + column_bytes + && column_start < input_start + input_bytes) + overlaps = true; + } + if (r->column_types) { + for (uint32_t row = 0; row < num_rows; row++) { + for (uint32_t col = 0; col < num_cols; col++) { + if (r->column_types[col] == WIRELOG_TYPE_FLOAT + && !wl_columnar_float_bits_valid( + rows[(size_t)row * num_cols + col])) { + rc = EINVAL; + goto finish; + } + } + } + } + if (overlaps) { + if (input_bytes <= sizeof(inline_rows)) { + memcpy(inline_rows, rows, (size_t)input_bytes); + rows = inline_rows; + } else { + if (r->memory_governor) { + wl_columnar_memory_admission_status_t status + = wl_columnar_memory_reserve_checked( + wl_columnar_memory_governor_ref_get(r->memory_governor), + input_bytes, &scratch_admission); + if (status != WL_COLUMNAR_MEMORY_ADMISSION_OK + && status != WL_COLUMNAR_MEMORY_ADMISSION_ADVISORY) { + if (status == WL_COLUMNAR_MEMORY_ADMISSION_DENIED) { + r->memory_budget_denial_pending = true; + rc = ENOMEM; + } else + rc = status == WL_COLUMNAR_MEMORY_ADMISSION_OVERFLOW + ? EOVERFLOW : EINVAL; + goto finish; } + scratch_credit_held = true; + if (!wl_columnar_memory_commit(&scratch_admission, + &scratch_admission)) { + rc = ENOMEM; + goto finish; + } + } + staged_rows = malloc((size_t)input_bytes); + if (!staged_rows) { + rc = ENOMEM; + goto finish; } + memcpy(staged_rows, rows, (size_t)input_bytes); + rows = staged_rows; } } /* Publish a lazy schema only after every validation that can reject the * batch has completed. Schema construction itself performs admission * and rolls back on allocation failure; its capacity covers the batch, - * so the reserve below cannot expose a partially appended relation. */ + * so no fallible reserve is needed after schema publication. */ if (schema_was_unset) { rc = col_rel_capacity_for_rows(r->capacity, required_rows, &capacity); @@ -4384,12 +4470,16 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, goto finish; } - rc = wl_columnar_relation_reserve_rows_impl(r, num_rows, &writer, - &alias_release_pending, false); - if (rc != 0) { - if (denied && r->memory_budget_denial_pending) - *denied = true; - goto finish; + if (!schema_was_unset) { + rc = wl_columnar_relation_reserve_rows_impl(r, num_rows, NULL, + NULL, false, &single.lease); + if (rc != 0) + goto finish; + } else { + /* No recoverable work remains after schema publication. */ + assert(r->schema_ok && r->ncols == num_cols); + assert(r->capacity >= required_rows); + assert(col_rel_timestamp_shape_valid(r)); } /* Capacity/admission and COW are now complete. The prevalidated raw @@ -4409,11 +4499,12 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, *denied = false; finish: - if (alias_release_pending - && col_rel_storage_alias_release(r) != 0) - abort(); - if (wl_columnar_source_access_writer_release(&writer) != 0) - abort(); + free(staged_rows); + if (scratch_credit_held) + col_rel_release_reservation_or_abort(&scratch_admission); + int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); + if (rc == 0) + rc = finish_rc; if (rc != 0 && denied && r->memory_budget_denial_pending) *denied = true; return rc; @@ -4426,20 +4517,30 @@ col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, static int wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, wl_columnar_source_access_writer_t *writer, - bool *out_alias_release_pending, bool begin_operation) + bool *out_alias_release_pending, bool begin_operation, + wl_columnar_relation_mutation_lease_t *lease) { col_rel_t *owner = NULL; int rc; - if (!r || !writer || !out_alias_release_pending) + if (!r) return EINVAL; - rc = col_rel_storage_owner_resolve(r, &owner); - if (rc != 0 || writer->owner != &owner->source_access - || writer->identity != (uintptr_t)writer - || !wl_columnar_source_access_writer_thread_equal(writer)) - return rc != 0 ? rc : EINVAL; - if (begin_operation) - r->memory_budget_denial_pending = false; - *out_alias_release_pending = false; + if (lease) { + if (writer || out_alias_release_pending || begin_operation + || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + } else { + if (!writer || !out_alias_release_pending) + return EINVAL; + rc = col_rel_storage_owner_resolve(r, &owner); + if (rc != 0 || writer->owner != &owner->source_access + || writer->identity != (uintptr_t)writer + || !wl_columnar_source_access_writer_thread_equal(writer)) + return rc != 0 ? rc : EINVAL; + if (begin_operation) + r->memory_budget_denial_pending = false; + *out_alias_release_pending = false; + } if (additional > UINT32_MAX - r->nrows) return EOVERFLOW; uint32_t required = r->nrows + additional; @@ -4453,15 +4554,19 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, if (r->storage_owner == r && col_rel_storage_alias_borrow_count(r) > 0) return EBUSY; - int rc = col_rel_grow_owned_transition_legacy_impl(r, r->capacity, - true); - if (rc == 0) + int rc = lease + ? col_rel_grow_owned_transition_publish_impl(r, r->capacity, + false, lease) + : col_rel_grow_owned_transition_legacy_impl(r, r->capacity, true); + if (rc == 0 && !lease) *out_alias_release_pending = alias_release_pending; return rc; } if (r->col_shared && required <= r->capacity) { - int rc = col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); - if (rc == 0) + int rc = lease + ? col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease) + : col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); + if (rc == 0 && !lease) *out_alias_release_pending = alias_release_pending; return rc; } @@ -4478,8 +4583,11 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, new_cap *= 2u; } if (r->col_shared || r->arena_owned) { - int rc = col_rel_grow_owned_transition_legacy_impl(r, new_cap, true); - if (rc == 0) + int rc = lease + ? col_rel_grow_owned_transition_publish_impl(r, new_cap, false, + lease) + : col_rel_grow_owned_transition_legacy_impl(r, new_cap, true); + if (rc == 0 && !lease) *out_alias_release_pending = alias_release_pending; return rc; } @@ -4535,7 +4643,7 @@ col_rel_reserve_rows_locked(col_rel_t *r, uint32_t additional, bool *out_alias_release_pending) { return wl_columnar_relation_reserve_rows_impl(r, additional, writer, - out_alias_release_pending, true); + out_alias_release_pending, true, NULL); } int From c28fddf0fd319d430b4c4a526091989cf73408da Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 14:19:49 +0900 Subject: [PATCH 5/9] fix(columnar): preserve overlapping rows in locked append --- tests/test_relation_generations.c | 411 +++++++++++++++++++++++++++++- wirelog/columnar/internal.h | 14 +- wirelog/columnar/relation.c | 45 ++-- 3 files changed, 442 insertions(+), 28 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 905ad0ab..4aded50a 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -4866,8 +4866,24 @@ test_public_append_mutation_admission(void) } } +/* Exercise the same physical-overlap and governor contract through each + * append authority. The locked caller retains and releases its own writer. */ +static int +append_overlap_attempt(col_rel_t *rel, const int64_t *row, bool locked) +{ + if (!locked) + return col_rel_append_row(rel, row); + wl_columnar_source_access_writer_t writer = { 0 }; + int rc = col_rel_source_writer_acquire(rel, &writer); + if (rc) + return rc; + rc = col_rel_append_row_locked(rel, row, &writer); + int release_rc = wl_columnar_source_access_writer_release(&writer); + return rc ? rc : release_rc; +} + static void -test_public_append_overlapping_input(void) +test_append_overlapping_input(bool locked) { for (unsigned wide = 0; wide < 2; wide++) { uint32_t width = wide ? 32 : 1; @@ -4878,7 +4894,7 @@ test_public_append_overlapping_input(void) for (uint32_t i = 0; i < COL_REL_INIT_CAP; i++) { for (uint32_t c = 0; c < width; c++) row[c] = (int64_t)i * 1000 + c; - CHECK(col_rel_append_row(rel, row) == 0, + CHECK(append_overlap_attempt(rel, row, locked) == 0, "public overlap source rows"); } wl_columnar_memory_resolution_t resolution = { @@ -4905,7 +4921,7 @@ test_public_append_overlapping_input(void) resolution.usable_bytes - credit, &blocker) && wl_columnar_memory_commit(&blocker, &blocker), "wide staging budget blocker"); - CHECK(col_rel_append_row(rel, inside) == ENOMEM + CHECK(append_overlap_attempt(rel, inside, locked) == ENOMEM && rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == @@ -4915,7 +4931,7 @@ test_public_append_overlapping_input(void) #ifdef WL_TEST_ALLOC_WRAP allocation_calls = 0; allocation_fail_at = 0; - int failed_rc = col_rel_append_row(rel, inside); + int failed_rc = append_overlap_attempt(rel, inside, locked); allocation_fail_at = -1; CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) @@ -4923,7 +4939,27 @@ test_public_append_overlapping_input(void) "wide staging malloc failure returns all scratch credit"); #endif } - CHECK(col_rel_append_row(rel, inside) == 0 + if (locked) { +#ifdef WL_TEST_ALLOC_WRAP + for (long fail = wide ? 1 : 0; fail < (wide ? 4 : 2); fail++) { + allocation_calls = 0; + allocation_fail_at = fail; + int failed_rc = append_overlap_attempt(rel, inside, true); + allocation_fail_at = -1; + CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == credit, + "locked growth failure after staging rolls back all credit"); + } +#endif + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(append_overlap_attempt(rel, inside, true) == ENOMEM + && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before) + && wl_columnar_memory_reserved(governor) == credit, + "locked publication failure returns staged scratch credit"); + } + CHECK(append_overlap_attempt(rel, inside, locked) == 0 && rel->nrows == before.nrows + 1 && rel->storage_generation == before.storage_generation + 1, "overlapping input survives owned growth and retired old column"); @@ -4937,9 +4973,10 @@ test_public_append_overlapping_input(void) #ifdef WL_TEST_ALLOC_WRAP allocation_calls = 0; allocation_fail_at = 0; - int external_rc = col_rel_append_row(rel, row); + int external_rc = append_overlap_attempt(rel, row, locked); long external_calls = allocation_calls; - int inside_rc = col_rel_append_row(rel, rel->columns[0]); + int inside_rc = append_overlap_attempt(rel, rel->columns[0], + locked); allocation_fail_at = -1; CHECK(external_rc == 0 && external_calls == 0 && inside_rc == ENOMEM && !rel->memory_budget_denial_pending @@ -4948,7 +4985,7 @@ test_public_append_overlapping_input(void) #endif uint32_t index = rel->nrows; memcpy(expected, rel->columns[0], width * sizeof(*inside)); - CHECK(col_rel_append_row(rel, rel->columns[0]) == 0 + CHECK(append_overlap_attempt(rel, rel->columns[0], locked) == 0 && wl_columnar_memory_reserved(governor) == credit, "wide overlapping spare append returns its temporary reservation"); for (uint32_t c = 0; c < width; c++) @@ -4962,6 +4999,357 @@ test_public_append_overlapping_input(void) } } +static void +test_locked_append_alias_overlap(void) +{ + for (unsigned wide = 0; wide < 2; wide++) { + for (unsigned full = 0; full < 2; full++) { + uint32_t width = wide ? 32 : 1; + col_rel_t *root = track_relation(col_rel_new_auto("locked_root", + width)); + col_rel_t *alias = track_relation(col_rel_new_auto("locked_alias", + width)); + col_rel_t *sibling = + track_relation(col_rel_new_auto("locked_sibling", width)); + CHECK(root && alias && sibling, "locked overlap aliases"); + int64_t row[32], expected[32]; + uint32_t count = full ? COL_REL_INIT_CAP : 40; + for (uint32_t i = 0; i < count; i++) { + for (uint32_t c = 0; c < width; c++) + row[c] = (int64_t)i * 1000 + c; + CHECK(col_rel_append_row(root, row) == 0, + "locked alias source rows"); + } + CHECK(col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "locked overlap installs two aliases"); + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 22, .usable_bytes = 1u << 22, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(alias, ref) == 0, + "locked alias governed"); + wl_columnar_memory_governor_t *governor + = wl_columnar_memory_governor_ref_get(ref); + uint64_t credit = wl_columnar_memory_reserved(governor); + col_rel_t root_before = *root, before = *alias, + sibling_before = *sibling; + const int64_t *inside = &alias->columns[0][3]; + CHECK(3 + width <= alias->capacity, + "wide tuple wholly inside allocation"); + memcpy(expected, inside, width * sizeof(*inside)); + wl_columnar_source_access_writer_t writer = { 0 }; + CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, + "locked alias writer admission"); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit, &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "locked COW quota blocker"); + CHECK(col_rel_append_row_locked(alias, inside, &writer) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2, + "locked COW quota refusal preserves payload and alias borrows"); + CHECK(wl_columnar_memory_release(&blocker) + && wl_columnar_memory_reserved(governor) == credit, + "locked COW quota blocker releases exact credit"); +#ifdef WL_TEST_ALLOC_WRAP + for (long fail = 0; fail < (wide ? 3 : 2); fail++) { + allocation_calls = 0; + allocation_fail_at = fail; + int failed_rc = col_rel_append_row_locked(alias, inside, + &writer); + allocation_fail_at = -1; + CHECK(failed_rc == ENOMEM && + !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) == credit, + "locked COW allocation failure returns scratch and payload credit"); + } +#endif + wl_columnar_relation_test_fail_next_prepare_resize(); + CHECK(col_rel_append_row_locked(alias, inside, &writer) == ENOMEM + && !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) == credit, + "locked COW preparation failure preserves both aliases and credit"); + if (wide) { + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, + resolution.usable_bytes - credit - sizeof(expected), + &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "wide COW admits scratch but blocks payload"); + CHECK(col_rel_append_row_locked(alias, inside, + &writer) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) + == resolution.usable_bytes - sizeof(expected), + "later payload quota denial returns wide scratch credit"); + CHECK(wl_columnar_memory_release(&blocker), + "wide COW payload quota blocker release"); + } else { + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_append_row_locked(alias, inside, + &writer) == ENOMEM + && !alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before) + && mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) == credit, + "locked COW publication failure rolls back borrows and credit"); + } + CHECK(col_rel_append_row_locked(alias, inside, &writer) == 0 + && alias->storage_owner == alias && !alias->col_shared + && alias->nrows == before.nrows + 1 + && alias->view_generation == before.view_generation + 1 + && alias->storage_generation == before.storage_generation + 1 + && col_rel_storage_alias_borrow_count(root) == 1, + "locked COW immediately releases exactly one alias"); + for (uint32_t c = 0; c < width; c++) { + CHECK(alias->columns[c][before.nrows] == expected[c], + "locked COW copies complete overlapping tuple"); + for (uint32_t i = 0; i < count; i++) + CHECK(root->columns[c][i] == (int64_t)i * 1000 + c + && sibling->columns[c][i] == root->columns[c][i], + "locked COW leaves root and sibling data exact"); + } + atomic_store_explicit(&root_before.storage_alias_borrows, 1, + memory_order_relaxed); + CHECK(mutation_payload_unchanged(root, &root_before) + && mutation_payload_unchanged(sibling, &sibling_before), + "locked COW preserves root and sibling pointers and epochs"); + CHECK(writer.owner == &root->source_access + && wl_columnar_source_access_writer_release(&writer) == 0, + "locked append retains original owner writer for caller"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "locked alias cleanup releases all credit"); + wl_columnar_memory_governor_ref_release(ref); + } + } +} + +typedef struct { + col_rel_t *relation; + wl_columnar_source_access_writer_t *writer; + int rc; +} locked_append_thread_args_t; + +static void * +locked_append_foreign_thread(void *arg) +{ + locked_append_thread_args_t *args = arg; + /* An invalid row endpoint would overflow if payload validation ran. */ + args->rc = col_rel_append_row_locked(args->relation, + (const int64_t *)(UINTPTR_MAX - 3), args->writer); + return NULL; +} + +static void +test_locked_append_authority_and_edges(void) +{ + col_rel_t *rel = new_relation(), *other = new_relation(); + int64_t row = 17; + CHECK(rel && other && col_rel_append_row(rel, &row) == 0, + "locked edge relations"); + rel->memory_budget_denial_pending = true; + wl_columnar_source_access_writer_t writer = { 0 }, invalid = { 0 }; + CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), + &invalid) == EINVAL && rel->memory_budget_denial_pending, + "absent writer preserves evidence before row read"); + CHECK(col_rel_source_writer_acquire(other, &writer) == 0, + "locked foreign owner writer"); + CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), + &writer) == EINVAL && rel->memory_budget_denial_pending, + "foreign owner rejects before overlap arithmetic"); + CHECK(wl_columnar_source_access_writer_release(&writer) == 0 + && col_rel_source_writer_acquire(rel, &writer) == 0, + "locked correct writer"); + invalid = writer; + CHECK(col_rel_append_row_locked(rel, &row, &invalid) == EINVAL + && rel->memory_budget_denial_pending, + "copied writer rejects before admission"); + locked_append_thread_args_t args = { .relation = rel, .writer = &writer }; + wl_thread_t thread; + CHECK(wl_thread_create(&thread, locked_append_foreign_thread, &args) == 0 + && wl_thread_join(&thread) == 0 && args.rc == EINVAL + && rel->memory_budget_denial_pending, + "foreign thread rejects before payload read"); + + col_rel_t before = *rel; + CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), + &writer) == EOVERFLOW && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before), + "admitted row endpoint overflow clears evidence without dereference"); + int64_t *column = rel->columns[0]; + rel->columns[0] = (int64_t *)(UINTPTR_MAX - 3); + rel->memory_budget_denial_pending = true; + CHECK(col_rel_append_row_locked(rel, &row, &writer) == EOVERFLOW + && !rel->memory_budget_denial_pending, + "column endpoint overflow precedes dereference"); + rel->columns[0] = column; + rel->timestamp_capacity = 1; + rel->memory_budget_denial_pending = true; + CHECK(col_rel_append_row_locked(rel, &row, &writer) == EINVAL + && !rel->memory_budget_denial_pending, + "admitted shape rejection clears evidence"); + rel->timestamp_capacity = 0; + CHECK(wl_columnar_source_access_writer_release(&writer) == 0, + "locked edge writer release"); + cleanup_relations(); +} + +static void +test_locked_append_legacy_descriptor_topology(void) +{ + col_rel_t *root = new_relation(), *alias = new_relation(), + *sibling = new_relation(); + int64_t row = 37; + CHECK(root && alias && sibling && col_rel_append_row(root, &row) == 0 + && col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "legacy append descriptor topology relations"); + wl_columnar_source_access_writer_t writer = { 0 }; + CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, + "legacy topology first descriptor admission"); + wl_columnar_source_access_gate_t *owner_gate = writer.owner; + wl_columnar_source_access_gate_t *alias_gate = writer.secondary_owner; + CHECK(owner_gate == &root->source_access && + alias_gate == &alias->descriptor_access + && wl_columnar_source_access_writer_release(&writer) == 0, + "legacy first token pins alias descriptor and canonical owner"); + CHECK(col_rel_source_writer_acquire(sibling, &writer) == 0, + "legacy topology second descriptor admission"); + CHECK(writer.owner == owner_gate && + writer.secondary_owner == &sibling->descriptor_access + && writer.secondary_owner != alias_gate + && wl_columnar_source_access_writer_release(&writer) == 0, + "same owner does not imply the same admitted descriptor"); + /* TODO (#2033 Unit 2B3b2): exact target lease authority must reject an + * alias-A token used for alias B before reads, denial reset, or scratch. + * The legacy owner-only append contract provides no such guarantee; + * characterization deliberately performs no unsafe cross-descriptor write. */ + cleanup_relations(); +} + +static void +test_locked_append_generation_boundaries(void) +{ + /* spare heap, growing heap, shared COW, growing arena, short arena + * timestamps, short heap timestamps, and spare arena with timestamps. */ + for (unsigned kind = 0; kind < 7; kind++) { + bool arena_kind = kind == 3 || kind == 4 || kind == 6; + bool growth = kind == 1 || kind == 3; + bool short_ts = kind == 4 || kind == 5; + bool replacement = kind != 0 && kind != 6; + delta_pool_t *pool = arena_kind + ? delta_pool_create(128, sizeof(col_rel_t), 4096) : NULL; + wl_arena_t *arena = arena_kind ? wl_arena_create(65536) : NULL; + col_rel_t *root = NULL; + col_rel_t *rel = arena_kind + ? col_rel_pool_new_auto(pool, arena, "locked_arena", 32) + : col_rel_new_auto("locked_epoch", 32); + CHECK(rel && (!arena_kind || (pool && arena && rel->arena_owned)), + "locked epoch relation"); + int64_t row[32]; + for (uint32_t c = 0; c < 32; c++) + row[c] = 73; + uint32_t count = growth ? rel->capacity : 1; + for (uint32_t i = 0; i < count; i++) + CHECK(col_rel_append_row(rel, row) == 0, + "locked epoch source rows"); + if (kind == 2) { + root = rel; + rel = col_rel_new_auto("locked_epoch_alias", 32); + CHECK(rel && col_rel_install_shared_view(rel, root) == 0, + "locked epoch shared view"); + } + if (short_ts || kind == 6) { + CHECK(col_rel_enable_timestamps(rel) == 0, + "locked epoch timestamps"); + if (short_ts) { + col_delta_timestamp_t *ts = realloc(rel->timestamps, + rel->nrows * sizeof(*ts)); + CHECK(ts, "locked epoch short physical timestamps"); + rel->timestamps = ts; + rel->timestamp_capacity = rel->nrows; + } + } + wl_columnar_source_access_writer_t writer = { 0 }; + CHECK(col_rel_source_writer_acquire(rel, &writer) == 0, + "locked epoch writer"); + uint64_t view = rel->view_generation, storage = rel->storage_generation; + for (unsigned boundary = 0; boundary < (replacement ? 2u : 1u); + boundary++) { + uint64_t *epoch = + boundary ? &rel->storage_generation : &rel->view_generation; + *epoch = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + if (boundary && rel->storage_owner == rel) + rel->storage_owner_generation = *epoch; + col_rel_t before = *rel; + rel->memory_budget_denial_pending = true; +#ifdef WL_TEST_ALLOC_WRAP + allocation_calls = 0; + allocation_fail_at = 0; +#endif + CHECK(col_rel_append_row_locked(rel, rel->columns[0], + &writer) == EOVERFLOW + && !rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before), + "locked epoch boundary rejects before staging or publication"); +#ifdef WL_TEST_ALLOC_WRAP + CHECK(allocation_calls == 0, + "epoch rejection consumes no allocation"); + allocation_fail_at = -1; +#endif + rel->view_generation = view; + rel->storage_generation = storage; + if (rel->storage_owner == rel) + rel->storage_owner_generation = storage; + } + /* No replacement is required by spare arena storage alone. */ + if (!replacement) { + rel->storage_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + rel->storage_owner_generation = rel->storage_generation; + storage = rel->storage_generation; + } + uint32_t rows = rel->nrows; + CHECK(col_rel_append_row_locked(rel, rel->columns[0], &writer) == 0 + && rel->nrows == rows + 1 && rel->columns[0][rows] == row[0] + && rel->view_generation == view + 1 + && rel->storage_generation == storage + (replacement ? 1 : 0) + && (!short_ts || rel->timestamp_capacity == rel->capacity) + && (kind != 6 || rel->arena_owned), + "locked epoch retry publishes exact view and storage epochs"); + CHECK(wl_columnar_source_access_writer_release(&writer) == 0, + "locked epoch release"); + col_rel_destroy(rel); + col_rel_destroy(root); + delta_pool_destroy(pool); + wl_arena_free(arena); + } +} + static void test_public_batch_mutation_admission(void) { @@ -5936,7 +6324,12 @@ main(void) test_public_batch_mutation_admission(); test_public_batch_overlapping_input(); test_public_append_mutation_admission(); - test_public_append_overlapping_input(); + test_append_overlapping_input(false); + test_append_overlapping_input(true); + test_locked_append_alias_overlap(); + test_locked_append_authority_and_edges(); + test_locked_append_legacy_descriptor_topology(); + test_locked_append_generation_boundaries(); test_public_append_edges_and_locked_compatibility(); test_set_cow_denial_provenance(); test_empty_append_all_legacy_cow_compatibility(); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 4c2adc0b..9c10471f 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2870,9 +2870,17 @@ wl_col_rel_inline_project_column(col_rel_t *dst, uint32_t dst_row, /* Public row append owns descriptor and canonical source writers throughout * publication. Input tuples may overlap column storage; the public path - * stages the complete tuple before any replacement. Caller keeps other input - * buffers stable for the call. Locked append/reservation retain their legacy - * source-writer contract. Outer attempts reset transient denial evidence. + * stages the complete tuple before any replacement. Locked append also + * stages overlap under its validated caller-owned canonical writer and + * releases a detached alias immediately before returning. Its transitional + * owner resolver requires a stable descriptor binding from the caller. + * Raw writer validation covers the canonical gate, not exact target + * descriptor authority; owner-only consolidation tokens remain supported + * pending the two-role lease migration (#2033). Both #2033 and #2034 remain + * open until that migration and removal of the live raw alias bridges. + * Caller keeps other input buffers stable for the call. Locked reservation + * retains its legacy source-writer contract. Outer attempts reset transient + * denial evidence only after authority admission. * Actual budget refusal retains legacy ENOMEM plus * memory_budget_denial_pending; allocation ENOMEM, EOVERFLOW and EINVAL are * distinct. Lower nested admission helpers do not reset outer evidence. */ diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 8dbf4b15..30908433 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -4011,11 +4011,36 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, if (!r || !row || !writer) return EINVAL; + if (!lease) { + col_rel_t *owner = NULL; + /* Transitional raw authority: owner resolution follows its existing + * stable-binding caller contract; no descriptor gate is acquired. + * This checks canonical owner authority only: owner-only tokens and + * tokens pinning a different same-owner descriptor remain legacy + * callers. Exact target authority requires the two-role consolidation + * lease migration (#2033). Validate before shape or payload reads. */ + if (!writer_held || !writer->owner + || writer->identity != (uintptr_t)writer + || !writer->thread_valid + || !wl_columnar_source_access_writer_thread_equal(writer) + || col_rel_storage_owner_resolve(r, &owner) != 0 + || !wl_columnar_relation_retirement_writer_valid(writer, + &owner->source_access)) + return EINVAL; + r->memory_budget_denial_pending = false; + } if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r))) return EINVAL; - if (lease && (!col_rel_timestamp_shape_valid(r) || r->nrows > r->capacity)) + if (!col_rel_timestamp_shape_valid(r) || r->nrows > r->capacity) return EINVAL; + bool replaces_storage = r->col_shared || r->nrows >= r->capacity + || (r->timestamps && r->nrows >= r->timestamp_capacity); + if (r->nrows == UINT32_MAX + || r->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u + || (replaces_storage && r->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u)) + return EOVERFLOW; /* Do all structural validation before any resize, COW, or timestamp * publication. In particular, a zero-column or partially initialized * relation must fail without consuming capacity or a generation epoch. */ @@ -4033,7 +4058,7 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, return EINVAL; bool row_overlaps = false; size_t row_bytes = 0; - if (lease) { + { if (r->nrows == UINT32_MAX) return EOVERFLOW; uint64_t bytes = (uint64_t)r->ncols * sizeof(*row); @@ -4044,7 +4069,8 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, row_bytes = (size_t)bytes; for (uint32_t c = 0; r->columns && c < r->ncols; c++) { uintptr_t column_start = (uintptr_t)r->columns[c]; - if (column_bytes > UINTPTR_MAX - column_start) + if (column_bytes > SIZE_MAX + || column_bytes > UINTPTR_MAX - column_start) return EOVERFLOW; if (row_bytes && column_bytes && row_start < column_start + column_bytes @@ -4060,19 +4086,6 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, } } - /* Resolve and admit the canonical storage owner only after validation, - * but before any resize, COW, timestamp, value, or generation mutation. - * A reader on an alias therefore excludes append on every view of the - * same backing storage. */ - if (!lease && !writer_held) { - rc = col_rel_source_writer_acquire(r, writer); - if (rc != 0) - return rc; - } - - if (!lease) - r->memory_budget_denial_pending = false; - /* A valid input row may be a slice of a column that publication retires. * Capture the complete tuple while its descriptor and source are closed, * even when this append currently needs no replacement. */ From 3ab59cc1be7e9672deb1769194f87a49ab02ee61 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Wed, 30 Sep 2026 15:24:25 +0900 Subject: [PATCH 6/9] fix(columnar): serialize public radix sorting --- tests/meson.build | 8 +- tests/test_relation_generations.c | 864 ++++++++++++++++++++++++++++++ wirelog/columnar/internal.h | 29 + wirelog/columnar/relation.c | 618 ++++++++++++++------- 4 files changed, 1325 insertions(+), 194 deletions(-) diff --git a/tests/meson.build b/tests/meson.build index 0263744a..8dd5f827 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -4287,7 +4287,8 @@ test_relation_generations_exe = executable( include_directories: [wirelog_inc, wirelog_src_inc], dependencies: [nanoarrow_dep, threads_dep, xxhash_dep, mbedtls_dep, math_dep], c_args: ['-DWL_TEST_APPEND_HOOK=1', '-DWL_TEST_SET_HOOK=1', - '-DWL_TEST_RELATION_RESIZE_HOOK=1'], + '-DWL_TEST_RELATION_RESIZE_HOOK=1', '-DWL_TEST_CONSOLIDATE_ALLOC_HOOK=1', + '-DWL_SESSION_TEST_HOOKS=1'], ) test('relation_generations', test_relation_generations_exe) @@ -4304,9 +4305,10 @@ if host_machine.system() == 'linux' include_directories: [wirelog_inc, wirelog_src_inc], dependencies: [nanoarrow_dep, threads_dep, xxhash_dep, mbedtls_dep, math_dep], c_args: ['-DWL_TEST_ALLOC_WRAP=1', '-DWL_TEST_APPEND_HOOK=1', - '-DWL_TEST_SET_HOOK=1', '-DWL_TEST_RELATION_RESIZE_HOOK=1'], + '-DWL_TEST_SET_HOOK=1', '-DWL_TEST_RELATION_RESIZE_HOOK=1', '-DWL_TEST_CONSOLIDATE_ALLOC_HOOK=1', + '-DWL_SESSION_TEST_HOOKS=1'], link_args: [ - '-Wl,--wrap=malloc', '-Wl,--wrap=calloc', '-Wl,--wrap=realloc', + '-Wl,--wrap=malloc', '-Wl,--wrap=calloc', '-Wl,--wrap=realloc', '-Wl,--wrap=free', ], override_options: ['b_lto=false'], ) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 4aded50a..9c68e5ac 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -12,6 +12,23 @@ #include #endif +#ifdef WL_TEST_CONSOLIDATE_ALLOC_HOOK +wl_columnar_consolidate_alloc_hook_t wl_columnar_consolidate_alloc_hook; +#endif +#ifdef WL_SESSION_TEST_HOOKS +/* This standalone backend target has no generic session implementation. */ +void +wl_session_testhook_before_workqueue_drain(wl_session_t *session) +{ + (void)session; +} +void +wl_session_testhook_after_worker_lease_release(wl_session_t *session) +{ + (void)session; +} +#endif + #ifdef WL_TEST_ALLOC_WRAP void *__real_malloc(size_t size); void *__real_calloc(size_t count, size_t size); @@ -50,6 +67,18 @@ __wrap_realloc(void *ptr, size_t size) } #endif +#ifdef WL_TEST_ALLOC_WRAP +void __real_free(void *pointer); +static void (*radix_free_observer)(void *pointer); +void +__wrap_free(void *pointer) +{ + if (radix_free_observer) + radix_free_observer(pointer); + __real_free(pointer); +} +#endif + static int failures; static col_rel_t *owned_relations[32]; static size_t owned_relation_count; @@ -6313,6 +6342,834 @@ test_root_leased_rebind_restores_descriptor(void) } #endif +/* Exact range preparation and descriptor-first radix admission (#2033). */ +typedef struct { + wl_columnar_relation_mutation_set_t set; + wl_columnar_relation_mutation_role_t role; + wl_columnar_relation_mutation_descriptor_t descriptor; + wl_columnar_relation_mutation_owner_t owner; + wl_columnar_relation_mutation_lease_t lease; + wl_columnar_relation_mutation_initialization_t initialization; +} radix_test_lease_t; + +static int +radix_test_acquire(col_rel_t *rel, radix_test_lease_t *held) +{ + held->role = (wl_columnar_relation_mutation_role_t){ rel, + WL_COLUMNAR_RELATION_PAYLOAD_MUTATION }; + return col_rel_mutation_set_acquire(&held->set, &held->role, 1, + &held->descriptor, 1, &held->owner, 1, &held->lease, 1, + &held->initialization, 1); +} + +#ifdef WL_TEST_CONSOLIDATE_ALLOC_HOOK +static const char *radix_fail_site; +static unsigned radix_fault_calls; +static bool +radix_test_fault(const char *site) +{ + radix_fault_calls++; + return !radix_fail_site || strcmp(site, radix_fail_site) == 0; +} +#endif +#ifdef WL_SESSION_TEST_HOOKS +static unsigned radix_admission_calls; +static void +radix_test_admission(const col_rel_t *rel, uint64_t physical, + uint64_t additional, uint64_t total, + wl_columnar_radix_admission_phase_t phase, + wl_columnar_memory_admission_status_t status) +{ + (void)rel; (void)physical; (void)additional; (void)total; + (void)phase; (void)status; + radix_admission_calls++; +} +#endif + +static void +test_leased_radix_kernels(void) +{ + const uint32_t counts[] = { 10, 70, 50000, 70 }; + for (unsigned family = 0; family < 4; family++) { + for (unsigned shared = 0; shared < 2; shared++) { + for (unsigned timed = 0; timed < 2; timed++) { + uint32_t count = counts[family]; + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + col_rel_t *sibling = new_relation(); + CHECK(root && alias && sibling, "radix family relations"); + if (!root || !alias || !sibling) + continue; + if (family == 3) { + wirelog_column_type_t type = WIRELOG_TYPE_FLOAT; + CHECK(col_rel_set_column_types(root, &type, 1) == 0, + "float kernel exact type layout"); + } + for (uint32_t i = 0; i < count + 2; i++) { + int64_t value = count + 2 - i; + CHECK(col_rel_append_row(root, &value) == 0, + "radix family seed"); + } + if (timed) { + root->timestamps = calloc(root->capacity, + sizeof(*root->timestamps)); + root->timestamp_capacity = root->capacity; + CHECK(root->timestamps != NULL, "radix timestamp seed"); + for (uint32_t i = 0; i < root->nrows; i++) + root->timestamps[i].iteration = i; + } + col_rel_t *rel = root; + if (shared) { + CHECK(col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "radix two aliases install"); + rel = alias; + } + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 25, .usable_bytes = 1u << 25, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref + = wl_columnar_memory_governor_ref_create(&resolution); + CHECK(ref && col_rel_attach_memory_governor(rel, ref) == 0, + "radix workspace governed relation"); + radix_test_lease_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + CHECK(radix_test_acquire(rel, &held) == 0 + && wl_columnar_relation_radix_workspace_prepare_with_lease( + rel, 1, count, &workspace, &held.lease) == 0, + "radix exact preparation"); + for (unsigned malformed = 0; malformed < 3; malformed++) { + wl_columnar_radix_workspace_t bad = workspace; + if (family == 0 || family == 3) { + if (malformed == 0) bad.insertion_rows = NULL; + if (malformed == 1) bad.insertion_capacity = 1; + if (malformed == 2) bad.insertion_bytes = 1; + } else if (family == 2) { + if (malformed == 0) bad.count16 = NULL; + if (malformed == 1) bad.k16_capacity = 1; + if (malformed == 2) bad.short_values = NULL; + } else { + if (malformed == 0) bad.perm_a = NULL; + if (malformed == 1) bad.k8_capacity = 1; + if (malformed == 2) bad.byte_values = NULL; + } + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, + count, &bad, &held.lease) == EINVAL + && rel->columns[0][1] == count + 1 + && rel->storage_owner == (shared ? root : rel), + "every family rejects malformed physical scratch before detach"); + } + col_rel_t before = *rel; + col_rel_t root_before = *root; + col_rel_t sibling_before = *sibling; +#ifdef WL_TEST_CONSOLIDATE_ALLOC_HOOK + radix_fault_calls = 0; + radix_fail_site = NULL; + wl_columnar_consolidate_alloc_hook = radix_test_fault; +#endif +#ifdef WL_SESSION_TEST_HOOKS + radix_admission_calls = 0; + wl_columnar_radix_admission_test_hook = radix_test_admission; +#endif + int rc = wl_columnar_relation_radix_sort_with_lease(rel, 1, + count, &workspace, &held.lease); +#ifdef WL_TEST_CONSOLIDATE_ALLOC_HOOK + wl_columnar_consolidate_alloc_hook = NULL; + CHECK(radix_fault_calls == 0, + "prepared radix never reaches allocation hook"); +#endif +#ifdef WL_SESSION_TEST_HOOKS + wl_columnar_radix_admission_test_hook = NULL; + CHECK(radix_admission_calls == 0, + "prepared radix never reaches scratch admission hook"); +#endif + CHECK(rc == 0 && + rel->view_generation == before.view_generation + 1 + && rel->storage_generation == + before.storage_generation + shared, + "radix publishes exactly one view and shared storage epoch"); + CHECK(rel->columns[0][0] == count + 2 + && rel->columns[0][count + 1] == 1, + "radix outside range bytes preserved"); + bool ordered = true; + for (uint32_t i = 1; i <= count; i++) { + ordered &= rel->columns[0][i] == (int64_t)i + 1; + if (timed) + ordered &= rel->timestamps[i].iteration == + count + 1 - i; + } + CHECK(ordered, "radix range order and timestamp provenance"); + if (shared) + CHECK(root->columns[0][1] == count + 1 + && mutation_payload_unchanged(sibling, &sibling_before) + && root->view_generation == root_before.view_generation + && root->storage_generation == + root_before.storage_generation + && col_rel_storage_alias_borrow_count(root) == 1 + && rel->storage_owner == rel, + "radix detaches one of two aliases and preserves peers"); + if (!shared) { + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, + count, &workspace, &held.lease) == EINVAL, + "successful radix makes workspace provenance stale"); + wl_columnar_radix_workspace_destroy(&workspace); + CHECK( + wl_columnar_relation_radix_workspace_prepare_with_lease( + rel, 1, count, &workspace, &held.lease) == 0 + && wl_columnar_relation_radix_sort_with_lease(rel, 1, + count, &workspace, &held.lease) == 0, + "destroy and reprepare permits already sorted retry"); + } + wl_columnar_radix_workspace_destroy(&workspace); + CHECK(col_rel_mutation_set_finish(&held.set, rc == 0) == 0, + "radix set finish"); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == 0, + "radix releases retained and scratch credits"); + wl_columnar_memory_governor_ref_release(ref); + } + } + } +} + +static void +test_leased_radix_provenance(void) +{ + col_rel_t *rel = new_relation(); + col_rel_t *other = new_relation(); + for (int64_t value = 12; value > 0; value--) + CHECK(col_rel_append_row(rel, &value) == 0, "radix provenance seed"); + radix_test_lease_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + CHECK(radix_test_acquire(rel, &held) == 0 + && wl_columnar_relation_radix_workspace_prepare_with_lease(rel, 0, + 12, &workspace, &held.lease) == 0, "radix provenance prepare"); + col_rel_t before = *rel; + wl_columnar_relation_mutation_lease_t copy = held.lease; + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 0, 12, + &workspace, ©) == EINVAL + && wl_columnar_relation_radix_sort_with_lease(other, 0, 0, + &workspace, &held.lease) == EINVAL, + "radix rejects copied and foreign relation leases"); + for (unsigned wrong = 0; wrong < 12; wrong++) { + wl_columnar_radix_workspace_t bad = workspace; + switch (wrong) { + case 0: bad.prepared_relation = other; break; + case 1: bad.prepared_identity++; break; + case 2: bad.prepared_view_generation++; break; + case 3: bad.prepared_storage_generation++; break; + case 4: bad.prepared_start++; break; + case 5: bad.prepared_count--; break; + case 6: bad.prepared_ncols++; break; + case 7: bad.prepared_capacity++; break; + case 8: bad.prepared_has_timestamps = true; break; + case 9: bad.prepared_has_types = true; break; + case 10: bad.prepared_type_fingerprint++; break; + default: bad.insertion_bytes = 0; break; + } + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 0, 12, + &bad, &held.lease) == EINVAL + && mutation_payload_unchanged(rel, &before) + && rel->columns[0][0] == 12, + "radix malformed workspace rejected before row mutation"); + } + uint64_t generation = held.lease.relation_generation; + held.lease.relation_generation++; + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 0, 12, + &workspace, &held.lease) == EINVAL, "radix stale lease rejected"); + held.lease.relation_generation = generation; + held.lease.role_flags = WL_COLUMNAR_RELATION_METADATA_DETACH; + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 0, 12, + &workspace, &held.lease) == EINVAL, "radix wrong role rejected"); + held.lease.role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; + wl_columnar_radix_workspace_destroy(&workspace); + CHECK(col_rel_mutation_set_finish(&held.set, false) == 0, + "radix invalid input set finish"); + cleanup_relations(); +} + +static void +test_radix_descriptor_admission(void) +{ + for (unsigned full = 0; full < 2; full++) { + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + for (int64_t value = 40; value > 0; value--) + CHECK(col_rel_append_row(root, &value) == 0, + "radix admission seed"); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix admission alias"); + alias->memory_budget_denial_pending = true; + alias->sorted_nrows = 7; + col_rel_t before = *alias; + rebind_reader_pause_t pause = { .relation = alias }; + atomic_init(&pause.ready, false); + atomic_init(&pause.proceed, false); + rebind_reader_pause = &pause; + wl_columnar_relation_test_reader_after_descriptor = + rebind_pause_after_descriptor; + wl_thread_t thread; + int started = wl_thread_create(&thread, rebind_paused_reader, &pause); + CHECK(started == 0, "radix descriptor reader starts"); + if (!started) { + while (!atomic_load_explicit(&pause.ready, memory_order_acquire)) + alias_accounting_yield(); + int rc = full ? col_rel_radix_sort_int64(alias) + : col_rel_radix_sort(alias, 0, 40); + CHECK(rc == EBUSY && mutation_payload_unchanged(alias, &before) + && alias->sorted_nrows == 7 && + alias->memory_budget_denial_pending, + "paused descriptor rejects radix without clearing denial"); + atomic_store_explicit(&pause.proceed, true, memory_order_release); + CHECK(wl_thread_join(&thread) == 0 && pause.rc == 0, + "radix descriptor reader drains"); + } + wl_columnar_relation_test_reader_after_descriptor = NULL; + rebind_reader_pause = NULL; + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "radix owner reader starts"); + int rc = full ? col_rel_radix_sort_int64(alias) + : col_rel_radix_sort(alias, 0, 40); + CHECK(rc == EBUSY && mutation_payload_unchanged(alias, &before) + && alias->sorted_nrows == 7 && alias->memory_budget_denial_pending, + "owner reader rejects radix and leaves sorted prefix intact"); + CHECK(col_rel_source_reader_release(&peer) == 0, + "radix owner reader drains"); + CHECK(col_rel_radix_sort(root, 0, 0) == EBUSY + && col_rel_radix_sort(root, 0, 1) == EBUSY, + "live root borrow rejects trivial radix before shortcut"); + rc = full ? col_rel_radix_sort_int64(alias) : col_rel_radix_sort(alias, + 0, 40); + CHECK(rc == 0 && alias->columns[0][0] == 1 + && alias->storage_owner == alias + && !alias->memory_budget_denial_pending + && alias->sorted_nrows == (full ? 40 : 7), + "radix admitted retry publishes sort metadata"); + cleanup_relations(); + } +} + +/* Failure snapshots include row/timestamp bytes as well as binding and credit. */ +typedef struct { + col_rel_t descriptor; + int64_t *rows; + col_delta_timestamp_t *timestamps; + uint64_t credit; +} radix_snapshot_t; + +static void +radix_snapshot_take(col_rel_t *rel, radix_snapshot_t *snapshot) +{ + snapshot->descriptor = *rel; + snapshot->rows = malloc((size_t)rel->nrows * sizeof(*snapshot->rows)); + memcpy(snapshot->rows, rel->columns[0], + (size_t)rel->nrows * sizeof(*snapshot->rows)); + snapshot->timestamps = NULL; + if (rel->timestamps) { + snapshot->timestamps = malloc((size_t)rel->nrows * + sizeof(*snapshot->timestamps)); + memcpy(snapshot->timestamps, rel->timestamps, + (size_t)rel->nrows * sizeof(*snapshot->timestamps)); + } + snapshot->credit = rel->memory_governor ? wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(rel->memory_governor)) : 0; +} + +static bool +radix_snapshot_matches(col_rel_t *rel, const radix_snapshot_t *snapshot) +{ + return mutation_payload_unchanged(rel, &snapshot->descriptor) + && rel->sorted_nrows == snapshot->descriptor.sorted_nrows + && memcmp(rel->columns[0], snapshot->rows, + (size_t)rel->nrows * sizeof(*snapshot->rows)) == 0 + && (!rel->timestamps || memcmp(rel->timestamps, snapshot->timestamps, + (size_t)rel->nrows * sizeof(*snapshot->timestamps)) == 0) + && (!rel->memory_governor || wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(rel->memory_governor)) == + snapshot->credit); +} + +static void +radix_snapshot_destroy(radix_snapshot_t *snapshot) +{ + free(snapshot->rows); + free(snapshot->timestamps); +} + +static wl_columnar_memory_governor_ref_t * +radix_test_governor(col_rel_t *rel) +{ + wl_columnar_memory_resolution_t resolution = { + .budget_bytes = 1u << 25, .usable_bytes = 1u << 25, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *ref = + wl_columnar_memory_governor_ref_create(&resolution); + if (!ref || col_rel_attach_memory_governor(rel, ref)) + return NULL; + return ref; +} + +static void +test_radix_preparation_failure_gates(void) +{ + const uint32_t counts[] = { 16, 70, 50000, 70 }; + const char *sites[] = { "radix_workspace_timestamps", + "radix_workspace_perm_a", + "radix_workspace_perm_b", + "radix_workspace_temp_column", + "radix_workspace_bucket_values", + "radix_workspace_count16", + "radix_workspace_insertion_rows" }; + for (unsigned family = 0; family < 4; family++) { + col_rel_t *rel = new_relation(); + CHECK(rel, "radix failure relation"); + if (family == 3) { + wirelog_column_type_t type = WIRELOG_TYPE_FLOAT; + CHECK(col_rel_set_column_types(rel, &type, 1) == 0, + "radix float failure layout"); + } + for (uint32_t i = 0; i < counts[family]; i++) { + int64_t value = counts[family] - i; + CHECK(col_rel_append_row(rel, &value) == 0, "radix failure rows"); + } + CHECK(col_rel_enable_timestamps(rel) == 0, "radix failure timestamps"); + for (uint32_t i = 0; i < rel->nrows; i++) + rel->timestamps[i].iteration = i; + rel->sorted_nrows = 3; + wl_columnar_memory_governor_ref_t *ref = radix_test_governor(rel); + CHECK(ref, "radix failure governor"); + wl_columnar_memory_governor_t *governor = + wl_columnar_memory_governor_ref_get(ref); + radix_snapshot_t before; + radix_snapshot_take(rel, &before); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, (1u << 25) - before.credit, + &blocker) + && wl_columnar_memory_commit(&blocker, &blocker), + "radix scratch quota blocker"); + rel->memory_budget_denial_pending = true; + CHECK(col_rel_radix_sort_int64(rel) == ENOMEM && + rel->memory_budget_denial_pending + && mutation_payload_unchanged(rel, &before.descriptor) + && rel->sorted_nrows == 3 && memcmp(rel->columns[0], before.rows, + rel->nrows * sizeof(*before.rows)) == 0 + && memcmp(rel->timestamps, before.timestamps, + rel->nrows * sizeof(*before.timestamps)) == 0 + && wl_columnar_memory_reserved(governor) == (1u << 25), + "radix quota failure preserves bytes metadata and exact credit"); + CHECK(wl_columnar_memory_release(&blocker), "radix quota unblock"); + for (unsigned site = 0; site < 7; site++) { + bool insertion = family == 0 || family == 3; + bool used = site == 0 || (insertion ? site == 6 + : (site >= 1 && site <= 4) || (family == 2 && site == 5)); + if (!used) + continue; + radix_fail_site = sites[site]; + radix_fault_calls = 0; + wl_columnar_consolidate_alloc_hook = radix_test_fault; + rel->memory_budget_denial_pending = true; + int rc = col_rel_radix_sort_int64(rel); + wl_columnar_consolidate_alloc_hook = NULL; + CHECK(rc == ENOMEM && radix_fault_calls > 0 && + !rel->memory_budget_denial_pending + && radix_snapshot_matches(rel, &before), + "each radix allocator failure releases partial scratch and preserves state"); + } + CHECK(col_rel_radix_sort_int64(rel) == 0 && + !rel->memory_budget_denial_pending + && rel->sorted_nrows == counts[family] + && rel->view_generation == before.descriptor.view_generation + 1 + && rel->storage_generation == before.descriptor.storage_generation + && wl_columnar_memory_reserved(governor) == before.credit, + "radix allocation and quota retry returns scratch credit"); + radix_snapshot_destroy(&before); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(governor) == 0, + "radix failure cleanup all credit"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +typedef struct { + col_rel_t *rel; + const wl_columnar_radix_workspace_t *workspace; + wl_columnar_relation_mutation_lease_t *lease; + int rc; +} radix_wrong_thread_t; + +static void * +radix_test_wrong_thread(void *opaque) +{ + radix_wrong_thread_t *args = opaque; + args->rc = wl_columnar_relation_radix_sort_with_lease(args->rel, 1, + args->rel->nrows - 2, args->workspace, args->lease); + return NULL; +} + +static void +test_radix_exact_authority_and_shape(void) +{ + col_rel_t *rel = new_relation(); + col_rel_t *other = new_relation(); + for (int64_t value = 72; value > 0; value--) + CHECK(col_rel_append_row(rel, &value) == 0 + && col_rel_append_row(other, &value) == 0, + "radix exact shape seed"); + wirelog_column_type_t type = WIRELOG_TYPE_INT64; + CHECK(col_rel_set_column_types(rel, &type, 1) == 0 + && col_rel_enable_timestamps(rel) == 0, + "radix exact typed timestamp shape"); + wl_columnar_memory_governor_ref_t *ref = radix_test_governor(rel); + CHECK(ref, "radix exact governor"); + radix_test_lease_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + CHECK(radix_test_acquire(rel, &held) == 0 + && wl_columnar_relation_radix_workspace_prepare_with_lease(rel, 1, + 70, &workspace, &held.lease) == 0, "radix exact snapshot prepare"); + radix_snapshot_t before; + radix_snapshot_take(rel, &before); + rel->memory_budget_denial_pending = true; + radix_wrong_thread_t args = { rel, &workspace, &held.lease, 0 }; + wl_thread_t thread; + CHECK(wl_thread_create(&thread, radix_test_wrong_thread, &args) == 0 + && wl_thread_join(&thread) == 0 && args.rc == EINVAL + && radix_snapshot_matches(rel, + &before) && rel->memory_budget_denial_pending, + "wrong-thread radix lease preserves bytes credit and denial"); + for (unsigned mutation = 0; mutation < 9; mutation++) { + col_rel_t saved = *rel; + switch (mutation) { + case 0: rel->view_generation++; break; + case 1: rel->column_types[0] = WIRELOG_TYPE_FLOAT; break; + case 2: rel->timestamp_capacity++; break; + case 3: rel->timestamps = NULL; rel->timestamp_capacity = 0; break; + case 4: rel->ncols = 0; break; + case 5: rel->capacity++; break; + case 6: rel->nrows--; break; + case 7: rel->storage_generation++; break; + default: rel->column_types = NULL; break; + } + int rc = wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &workspace, &held.lease); + rel->view_generation = saved.view_generation; + rel->storage_generation = saved.storage_generation; + rel->ncols = saved.ncols; rel->nrows = saved.nrows; + rel->capacity = saved.capacity; + rel->timestamps = saved.timestamps; + rel->timestamp_capacity = saved.timestamp_capacity; + rel->column_types = saved.column_types; rel->column_types[0] = type; + CHECK(rc == EINVAL && radix_snapshot_matches(rel, &before) + && rel->memory_budget_denial_pending, + "actual relation shape/type/generation changes invalidate workspace before writes"); + } + for (unsigned malformed = 0; malformed < 9; malformed++) { + wl_columnar_radix_workspace_t bad = workspace; + switch (malformed) { + case 0: bad.perm_a = NULL; break; + case 1: bad.perm_b = NULL; break; + case 2: bad.temp_column = NULL; break; + case 3: bad.k8_capacity = 1; break; + case 4: bad.timestamp_capacity = 1; break; + case 5: bad.byte_values = NULL; break; + case 6: bad.prepared_timestamp_capacity++; break; + case 7: bad.prepared_nrows++; break; + default: bad.bucket_values = NULL; break; + } + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &bad, &held.lease) == EINVAL && radix_snapshot_matches(rel, + &before), + "radix selected scratch capacity malformed before mutation"); + } + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 0, 70, &workspace, + &held.lease) == EINVAL + && wl_columnar_relation_radix_sort_with_lease(rel, 1, 69, &workspace, + &held.lease) == EINVAL + && radix_snapshot_matches(rel, &before), + "radix exact range rejects compatible smaller range"); + wl_columnar_radix_workspace_t foreign_workspace = { 0 }; + uint32_t foreign_bounds[] = { 1, 71 }; + CHECK(wl_columnar_radix_workspace_prepare(other, foreign_bounds, 1, 0, + &foreign_workspace) == 0 + && wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &foreign_workspace, &held.lease) == EINVAL + && radix_snapshot_matches(rel, &before), + "radix real same-shape foreign preparation is refused"); + wl_columnar_radix_workspace_destroy(&foreign_workspace); + wl_columnar_source_access_writer_t writer_copy = held.descriptor.writer; + held.descriptor.writer = held.owner.writer; + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, &workspace, + &held.lease) == EINVAL, + "radix copied foreign descriptor token rejected"); + held.descriptor.writer = writer_copy; + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &workspace, &held.lease) == 0, "radix authority restored retry"); + CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &workspace, &held.lease) == EINVAL, + "radix successful generation consumes workspace"); + wl_columnar_radix_workspace_destroy(&workspace); + CHECK(wl_columnar_relation_radix_workspace_prepare_with_lease(rel, 1, 70, + &workspace, &held.lease) == 0 + && wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, + &workspace, &held.lease) == 0, "radix exact reprepare sorted retry"); + wl_columnar_radix_workspace_destroy(&workspace); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix exact finish"); + radix_snapshot_destroy(&before); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(wl_columnar_memory_governor_ref_get( + ref)) == 0, + "radix exact cleanup credit"); + wl_columnar_memory_governor_ref_release(ref); +} + +#ifdef WL_TEST_ALLOC_WRAP +static col_rel_t *radix_free_rel; +static col_rel_t *radix_free_root; +static void *radix_free_table; +static uint64_t radix_free_view; +static uint64_t radix_free_storage; +static bool radix_detach_observed; +static bool radix_sorted_observed; +static bool radix_boundary_ok; + +/* GNU free wrapping observes real retirement; it neither allocates nor alters + * the production allocator. Table retirement precedes the final old-owner + * borrow decrement; scratch retirement follows full-wrapper metadata publish. */ +static void +radix_free_boundary(void *pointer) +{ + if (!radix_free_rel) + return; + bool detach = pointer == radix_free_table; + bool sorted = radix_free_rel->view_generation == radix_free_view + 1; + if (!detach && !sorted) + return; + wl_columnar_source_access_reader_t reader = { 0 }; + int target_rc = col_rel_source_reader_acquire(radix_free_rel, &reader); + if (target_rc == 0) + (void)col_rel_source_reader_release(&reader); + int owner_rc = col_rel_source_reader_acquire(radix_free_root, &reader); + if (owner_rc == 0) + (void)col_rel_source_reader_release(&reader); + radix_boundary_ok &= target_rc == EBUSY && owner_rc == EBUSY; + if (detach) { + radix_detach_observed = true; + radix_boundary_ok &= radix_free_rel->col_shared == NULL + && radix_free_rel->storage_generation == radix_free_storage + && radix_free_rel->view_generation == radix_free_view + && col_rel_storage_alias_borrow_count(radix_free_root) == 2; + } + if (sorted) { + radix_sorted_observed = true; + radix_boundary_ok &= radix_free_rel->sorted_nrows == + radix_free_rel->nrows + && radix_free_rel->storage_generation == radix_free_storage + 1 + && radix_free_rel->storage_owner == radix_free_rel + && col_rel_storage_alias_borrow_count(radix_free_root) == 1; + } +} +#endif + +static void +test_radix_cow_failures_and_publication(void) +{ + for (unsigned family = 0; family < 3; family++) { + uint32_t count = family == 0 ? 16 : family == 1 ? 70 : 50000; + col_rel_t *root = new_relation(), *alias = new_relation(), + *sibling = new_relation(); + CHECK(root && alias && sibling, "radix COW test relations"); + for (uint32_t i = 0; i < count; i++) { + int64_t value = count - i; + CHECK(col_rel_append_row(root, &value) == 0, "radix COW rows"); + } + CHECK(col_rel_enable_timestamps(root) == 0 + && col_rel_install_shared_view(alias, root) == 0 + && col_rel_install_shared_view(sibling, root) == 0, + "radix COW two timed aliases"); + wl_columnar_memory_governor_ref_t *ref = radix_test_governor(alias); + CHECK(ref && col_rel_attach_memory_governor(root, ref) == 0 + && col_rel_attach_memory_governor(sibling, ref) == 0, + "radix COW governed peers"); + alias->sorted_nrows = 3; + radix_snapshot_t before, root_before, sibling_before; + radix_snapshot_take(alias, &before); + radix_snapshot_take(root, &root_before); + radix_snapshot_take(sibling, &sibling_before); + for (unsigned failure = 0; failure < 2; failure++) { + alias->memory_budget_denial_pending = true; + if (failure == 0) + wl_columnar_relation_test_fail_next_prepare_resize(); + else + wl_columnar_relation_test_fail_next_reservation_commit(); + CHECK(col_rel_radix_sort_int64(alias) == ENOMEM + && !alias->memory_budget_denial_pending + && radix_snapshot_matches(alias, &before) + && radix_snapshot_matches(root, &root_before) + && radix_snapshot_matches(sibling, &sibling_before) + && col_rel_storage_alias_borrow_count(root) == 2, + "radix COW prepare/publication failure restores peers bytes borrows and credit"); + } + radix_test_lease_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + CHECK(radix_test_acquire(alias, &held) == 0 + && wl_columnar_relation_radix_workspace_prepare_with_lease(alias, 0, + count, &workspace, &held.lease) == 0, + "radix COW preprepared scratch"); +#ifdef WL_TEST_ALLOC_WRAP + uint64_t prepared_credit = + wl_columnar_memory_reserved(wl_columnar_memory_governor_ref_get( + ref)); + for (long failure = 0; failure < 2; failure++) { + allocation_calls = 0; allocation_fail_at = failure; + int rc = wl_columnar_relation_radix_sort_with_lease(alias, 0, count, + &workspace, &held.lease); + allocation_fail_at = -1; + CHECK(rc == ENOMEM && mutation_payload_unchanged(alias, + &before.descriptor) + && memcmp(alias->columns[0], before.rows, + count * sizeof(*before.rows)) == 0 + && alias->sorted_nrows == 3 && + col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(ref)) == + prepared_credit, + "radix COW each column/table allocator failure keeps prepared credit exact"); + } +#endif + wl_columnar_memory_governor_t *governor = + wl_columnar_memory_governor_ref_get(ref); + uint64_t scratch_credit = wl_columnar_memory_reserved(governor); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + CHECK(wl_columnar_memory_reserve(governor, (1u << 25) - scratch_credit, + &blocker) && wl_columnar_memory_commit(&blocker, &blocker), + "radix prepared workspace leaves no COW quota"); + alias->memory_budget_denial_pending = false; + CHECK(wl_columnar_relation_radix_sort_with_lease(alias, 0, count, + &workspace, &held.lease) == ENOMEM + && alias->memory_budget_denial_pending + && mutation_payload_unchanged(alias, &before.descriptor) + && memcmp(alias->columns[0], before.rows, + count * sizeof(*before.rows)) == 0 + && memcmp(alias->timestamps, before.timestamps, + count * sizeof(*before.timestamps)) == 0 + && col_rel_storage_alias_borrow_count(root) == 2 + && wl_columnar_memory_reserved(governor) == (1u << 25), + "COW quota refusal preserves binding borrows bytes and prepared scratch"); + CHECK(wl_columnar_memory_release(&blocker) + && wl_columnar_memory_reserved(governor) == scratch_credit, + "COW quota unblock retains exact prepared credit"); + wl_columnar_radix_workspace_destroy(&workspace); + CHECK(col_rel_mutation_set_finish(&held.set, false) == 0 + && radix_snapshot_matches(alias, &before), + "radix COW failed scratch cleanup"); +#ifdef WL_TEST_ALLOC_WRAP + radix_free_rel = alias; radix_free_root = root; + radix_free_table = alias->columns; + radix_free_view = alias->view_generation; + radix_free_storage = alias->storage_generation; + radix_detach_observed = false; radix_sorted_observed = false; + radix_boundary_ok = true; + radix_free_observer = radix_free_boundary; +#endif + int rc = col_rel_radix_sort_int64(alias); +#ifdef WL_TEST_ALLOC_WRAP + radix_free_observer = NULL; radix_free_rel = NULL; + CHECK(radix_detach_observed && radix_sorted_observed && + radix_boundary_ok, + "radix retirement observers prove transition and sorted metadata remain excluded"); +#endif + CHECK(rc == 0 && alias->sorted_nrows == count + && alias->view_generation == before.descriptor.view_generation + 1 + && alias->storage_generation == + before.descriptor.storage_generation + 1 + && col_rel_storage_alias_borrow_count(root) == 1 + && alias->columns[0][0] == 1 && root->columns[0][0] == count + && sibling->columns[0][0] == count, + "radix COW retry publishes detach storage and sort once"); + radix_snapshot_destroy(&before); radix_snapshot_destroy(&root_before); + radix_snapshot_destroy(&sibling_before); + cleanup_relations(); + CHECK(wl_columnar_memory_reserved(wl_columnar_memory_governor_ref_get( + ref)) == 0, + "radix COW all governor cleanup"); + wl_columnar_memory_governor_ref_release(ref); + } +} + +static void +test_radix_opposite_address_contention(void) +{ + for (unsigned direction = 0; direction < 2; direction++) { + col_rel_t *first = new_relation(), *second = new_relation(); + CHECK(first && second, "radix address relation pair"); + col_rel_t *low = (uintptr_t)first < (uintptr_t)second ? first : second; + col_rel_t *high = low == first ? second : first; + col_rel_t *root = direction ? high : low; + col_rel_t *alias = direction ? low : high; + int64_t values[] = { 3, 2, 1 }; + for (unsigned i = 0; i < 3; i++) + CHECK(col_rel_append_row(root, &values[i]) == 0, + "radix address root rows"); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix address alias binding"); + radix_snapshot_t before; + radix_snapshot_take(alias, &before); + alias->sorted_nrows = 0; + alias->memory_budget_denial_pending = true; + for (unsigned repeat = 0; repeat < 16; repeat++) { + wl_columnar_source_access_reader_t peer = { 0 }; + CHECK(col_rel_source_reader_acquire(root, &peer) == 0, + "radix address owner reader"); + CHECK(col_rel_radix_sort_int64(alias) == EBUSY + && radix_snapshot_matches(alias, + &before) && alias->memory_budget_denial_pending, + "radix opposite owner phase denial remains exact"); + CHECK(col_rel_source_reader_release(&peer) == 0, + "radix address reader release"); + wl_columnar_source_access_reader_t descriptor_reader = { 0 }; + CHECK(wl_columnar_source_access_reader_acquire( + &alias->descriptor_access, + &descriptor_reader) == 0, + "radix address descriptor reader"); + CHECK(col_rel_radix_sort(alias, 0, 0) == EBUSY + && col_rel_radix_sort(alias, 0, 1) == EBUSY + && col_rel_radix_sort_int64(alias) == EBUSY + && radix_snapshot_matches(alias, + &before) && alias->memory_budget_denial_pending, + "radix descriptor contention validates authority before trivial shortcut"); + CHECK(wl_columnar_source_access_reader_release( + &descriptor_reader) == 0, + "radix address descriptor reader drains"); + CHECK(col_rel_radix_sort(root, 0, 0) == EBUSY + && col_rel_radix_sort(root, 0, 1) == EBUSY, + "radix root 0/1 ranges refuse live alias in both address directions"); + } + CHECK(col_rel_radix_sort_int64(alias) == 0 && alias->sorted_nrows == 3 + && alias->columns[0][0] == 1 && + col_rel_storage_alias_borrow_count(root) == 0, + "radix opposite addresses contention retry has no leaked gates"); + uint64_t view = alias->view_generation, + storage = alias->storage_generation; + CHECK(col_rel_radix_sort(alias, 0, 0) == 0 + && col_rel_radix_sort(alias, 0, 1) == 0 + && alias->view_generation == view && + alias->storage_generation == storage, + "radix admitted 0/1 ranges publish no epochs"); + radix_snapshot_destroy(&before); + cleanup_relations(); + } +} + int main(void) { @@ -6323,6 +7180,13 @@ main(void) test_public_batch_lazy_and_timestamp_edges(); test_public_batch_mutation_admission(); test_public_batch_overlapping_input(); + test_radix_preparation_failure_gates(); + test_radix_exact_authority_and_shape(); + test_radix_cow_failures_and_publication(); + test_radix_opposite_address_contention(); + test_radix_descriptor_admission(); + test_leased_radix_provenance(); + test_leased_radix_kernels(); test_public_append_mutation_admission(); test_append_overlapping_input(false); test_append_overlapping_input(true); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 9c10471f..47b010e8 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -3126,6 +3126,19 @@ typedef struct { uint32_t k16_capacity; uint32_t insertion_capacity; size_t insertion_bytes; /* Actual capacity, including the saved row. */ + const col_rel_t *prepared_relation; + uint64_t prepared_identity; + uint64_t prepared_view_generation; + uint64_t prepared_storage_generation; + uint64_t prepared_type_fingerprint; + uint32_t prepared_start; + uint32_t prepared_count; + uint32_t prepared_nrows; + uint32_t prepared_ncols; + uint32_t prepared_capacity; + uint32_t prepared_timestamp_capacity; + bool prepared_has_timestamps; + bool prepared_has_types; wl_columnar_memory_reservation_t admission; bool admission_active; } wl_columnar_radix_workspace_t; @@ -3171,6 +3184,22 @@ wl_columnar_relation_radix_sort_with_workspace(col_rel_t *rel, uint32_t nrows, const wl_columnar_source_access_writer_t *writer, const wl_columnar_radix_workspace_t *workspace); +/* Lease-scoped preparation allocates the complete scratch even for sorted + * ranges, and preflights authority, shape and epoch headroom before allocation. */ +WL_MUST_CHECK int +wl_columnar_relation_radix_workspace_prepare_with_lease(col_rel_t *rel, + uint32_t start_row, uint32_t nrows, + wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease); + +/* Exact prepared single-range snapshot, validated before any COW. Every call + * requires a workspace; mutations consume its generation provenance. The + * caller holds a PAYLOAD_MUTATION lease. Shared detach retires its borrow here. */ +WL_MUST_CHECK int +wl_columnar_relation_radix_sort_with_lease(col_rel_t *rel, uint32_t start_row, + uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease); + /** Stable LSD radix sort of a row-major int64_t buffer by a single key * column. Used by arrangement.c (sarr_build) and lftj.c * (lftj_iter_init) to replace the platform-specific qsort_r path -- diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 30908433..ec756977 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -8576,6 +8576,23 @@ col_radix_sort_rows_by_key(int64_t *data, uint32_t nrows, uint32_t ncols, * O(n^2) but low constant overhead — faster than radix sort for small N. */ static int +wl_columnar_radix_insertion_prepared(col_rel_t *r, uint32_t start_row, + uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps); + +static int +wl_columnar_radix_k16_prepared(col_rel_t *r, uint32_t start_row, uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps); + +static int +wl_columnar_radix_rows_prepared(col_rel_t *r, uint32_t start_row, + uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps); + +static int col_rel_insertion_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, col_delta_timestamp_t *timestamps) @@ -8597,70 +8614,12 @@ col_rel_insertion_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows, return EINVAL; if (!work) return ENOMEM; - for (uint32_t i = 0; i < nrows; i++) - col_rel_row_copy_out(r, start_row + i, work + (size_t)i * nc); - - if (timestamps) - memcpy(timestamps, r->timestamps + start_row, - (size_t)nrows * sizeof(*timestamps)); - for (uint32_t i = 1; i < nrows; i++) { - col_delta_timestamp_t saved_timestamp = { 0 }; - if (timestamps) - saved_timestamp = timestamps[i]; - int64_t *tbuf = work + (size_t)nrows * nc; - memcpy(tbuf, work + (size_t)i * nc, - (size_t)nc * sizeof(*work)); - uint32_t j = i; - while (j > 0) { - /* Compare r[start_row + j - 1] against saved key - * in tbuf (not the relation -- row i is overwritten - * after the first shift). */ - int cmp = 0; - for (uint32_t c = 0; c < nc; c++) { - int64_t va = work[(size_t)(j - 1) * nc + c]; - int64_t vb = tbuf[c]; - if (r->column_types - && r->column_types[c] == WIRELOG_TYPE_FLOAT) { - int fcmp = wl_columnar_float_compare_bits(va, vb); - if (fcmp != 0) { - cmp = fcmp; - break; - } - continue; - } - if (va > vb) { - cmp = 1; - break; - } - if (va < vb) { - cmp = -1; - break; - } - } - if (cmp <= 0) - break; - memcpy(work + (size_t)j * nc, - work + (size_t)(j - 1) * nc, - (size_t)nc * sizeof(*work)); - if (timestamps) - timestamps[j] = timestamps[j - 1]; - j--; - } - if (timestamps) - timestamps[j] = saved_timestamp; - memcpy(work + (size_t)j * nc, tbuf, - (size_t)nc * sizeof(*work)); - } - - for (uint32_t c = 0; c < nc; c++) - for (uint32_t i = 0; i < nrows; i++) - r->columns[c][start_row + i] = work[(size_t)i * nc + c]; - if (timestamps) - memcpy(r->timestamps + start_row, timestamps, - (size_t)nrows * sizeof(*timestamps)); + wl_columnar_radix_workspace_t prepared = { .insertion_rows = work }; + int rc = wl_columnar_radix_insertion_prepared(r, start_row, nrows, + &prepared, timestamps); if (owns_work) free(work); - return 0; + return rc; } /* ======================================================================== */ @@ -9085,11 +9044,8 @@ radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, col_delta_timestamp_t *timestamps) { - uint32_t nc = r->ncols; - const uint32_t radix_bits = 16u; - const uint32_t num_passes = 64u / radix_bits; /* 4 for k=16 */ - const uint32_t hist_size = 1u << radix_bits; /* 65536 for k=16 */ + const uint32_t hist_size = 65536u; bool owns_workspace = workspace == NULL; uint32_t *perm_a = workspace ? workspace->perm_a @@ -9121,6 +9077,249 @@ radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, : ENOMEM; } + int64_t *temp_col = workspace ? workspace->temp_column + : (int64_t *)wl_columnar_relation_radix_malloc( + nrows * sizeof(int64_t), "radix_temp_column"); + if (!temp_col) { + if (owns_workspace) { + free(perm_a); + free(perm_b); + free(bv_cache); + free(count); + } + return ENOMEM; + } + wl_columnar_radix_workspace_t prepared = { + .perm_a = perm_a, .perm_b = perm_b, .temp_column = temp_col, + .short_values = bv_cache, + .count16 = count, + }; + int rc = wl_columnar_radix_k16_prepared(r, start_row, nrows, &prepared, + timestamps); + if (owns_workspace) { + free(temp_col); + free(perm_a); + free(perm_b); + free(bv_cache); + free(count); + } + return rc; +} + +/* + * col_rel_radix_sort: index-permutation LSD radix sort (Phase B, Issue #330). + * + * Sort sub-range [start_row, start_row + nrows) of r in-place. + * Uses col_rel_get() for key extraction (layout-independent). + * Sorts a permutation array instead of scattering full rows. + * Permutation is applied once at the end via col_rel_row_copy_out/in. + * + * Optimizations (Issue #343): + * - Hybrid threshold: insertion sort for nrows <= 32 + * - Skip-pass: skip byte positions where all values have the same byte + * - Byte-value cache: read column data once per pass, reuse for scatter + * + * Adaptive radix width (Issue #363 Phase 5c): + * - nrows >= 50000: k=16 via radix_sort_k16() — 4 passes × 65536 buckets + * - nrows < 50000: k=8 with SIMD fused uniform+count — 8 passes × 256 buckets + * + * Falls back to insertion sort on allocation failure. + */ +static int +wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, + uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps) +{ + if (nrows <= 1) + return 0; + + /* IEEE-754 keys need a different sign transform from signed integers. + * Keep the established radix fast path for integer-only relations and + * use the typed comparator until the float radix path is introduced. */ + if (r->column_types) { + for (uint32_t c = 0; c < r->ncols; c++) { + if (r->column_types[c] == WIRELOG_TYPE_FLOAT) + goto insertion; + } + } + + /* Hybrid threshold: insertion sort for small segments (Issue #343) */ + if (nrows <= 32) + goto insertion; + + /* Adaptive radix width (Issue #363 Phase 5c): dispatch to k=16 for large + * arrays where fewer passes outweigh the larger histogram cost. + * + * Empirical threshold (Apple M-series, 1-col, 64-bit uniform-random keys): + * nrows=10K: k8=0.66ms k16=0.81ms k8 faster by 1.23x + * nrows=20K: k8=0.81ms k16=0.91ms k8 faster by 1.12x + * nrows=30K: k8=0.83ms k16=0.89ms k8 faster by 1.08x + * nrows=40K: k8=0.81ms k16=0.82ms near parity + * nrows=50K: k8=0.79ms k16=0.78ms k16 faster by 1.01x + * nrows=60K: k8=0.82ms k16=0.80ms k16 faster by 1.02x + * nrows=100K: k8=1.41ms k16=1.32ms k16 faster by 1.07x + * Crossover at ~40-50K rows; 50000 is a conservative round boundary. + * + * k=16 uses a scalar fused loop (not SIMD): the 4-pass reduction + * already yields fewer total iterations than 8-pass k=8+SIMD, and a + * 256KB histogram makes SIMD gather impractical (cache pressure). */ + if (nrows >= 50000u) { + int rc = radix_sort_k16(r, start_row, nrows, workspace, timestamps); + if (rc == 0) + wl_columnar_relation_touch_view(r); + return rc; + } + + /* k=8 SIMD path: 8 passes × 256 buckets (1KB histogram, stack-allocated). + * SIMD-dispatched fused uniform+count avoids a separate gather loop. */ + + bool owns_workspace = workspace == NULL; + uint32_t *perm_a = workspace ? workspace->perm_a + : (uint32_t *)wl_columnar_relation_radix_malloc( + nrows * sizeof(uint32_t), "radix_perm_a"); + uint32_t *perm_b = workspace ? workspace->perm_b + : (uint32_t *)wl_columnar_relation_radix_malloc( + nrows * sizeof(uint32_t), "radix_perm_b"); + uint8_t *bv_cache = workspace ? workspace->byte_values + : (uint8_t *)wl_columnar_relation_radix_malloc( + nrows, "radix_byte_values"); + if (workspace && workspace->k8_capacity < nrows) + return EINVAL; + if (!perm_a || !perm_b || !bv_cache) { + if (owns_workspace) { + free(perm_a); + free(perm_b); + free(bv_cache); + } + return nrows <= 32 ? col_rel_insertion_sort(r, start_row, nrows, + workspace, timestamps) + : ENOMEM; + } + + int64_t *temp_col = workspace ? workspace->temp_column + : (int64_t *)wl_columnar_relation_radix_malloc( + nrows * sizeof(int64_t), "radix_temp_column"); + if (!temp_col) { + if (owns_workspace) { + free(perm_a); + free(perm_b); + free(bv_cache); + } + return ENOMEM; + } + wl_columnar_radix_workspace_t prepared = { + .perm_a = perm_a, .perm_b = perm_b, .temp_column = temp_col, + .byte_values = bv_cache, + }; + int rc = wl_columnar_radix_rows_prepared(r, start_row, nrows, &prepared, + timestamps); + if (owns_workspace) { + free(temp_col); + free(perm_a); + free(perm_b); + free(bv_cache); + } + return rc; + +insertion: + { + int rc = col_rel_insertion_sort(r, start_row, nrows, workspace, + timestamps); + if (rc == 0) + wl_columnar_relation_touch_view(r); + return rc; + } +} + +/* Prepared-only kernels. Every buffer and capacity is validated before COW; + * no allocator, admission or fallback path is reachable from these kernels. */ +static int +wl_columnar_radix_insertion_prepared(col_rel_t *r, uint32_t start_row, + uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps) +{ + uint32_t nc = r->ncols; + int64_t *work = workspace->insertion_rows; + for (uint32_t i = 0; i < nrows; i++) + col_rel_row_copy_out(r, start_row + i, work + (size_t)i * nc); + + if (timestamps) + memcpy(timestamps, r->timestamps + start_row, + (size_t)nrows * sizeof(*timestamps)); + for (uint32_t i = 1; i < nrows; i++) { + col_delta_timestamp_t saved_timestamp = { 0 }; + if (timestamps) + saved_timestamp = timestamps[i]; + int64_t *tbuf = work + (size_t)nrows * nc; + memcpy(tbuf, work + (size_t)i * nc, + (size_t)nc * sizeof(*work)); + uint32_t j = i; + while (j > 0) { + /* Compare r[start_row + j - 1] against saved key + * in tbuf (not the relation -- row i is overwritten + * after the first shift). */ + int cmp = 0; + for (uint32_t c = 0; c < nc; c++) { + int64_t va = work[(size_t)(j - 1) * nc + c]; + int64_t vb = tbuf[c]; + if (r->column_types + && r->column_types[c] == WIRELOG_TYPE_FLOAT) { + int fcmp = wl_columnar_float_compare_bits(va, vb); + if (fcmp != 0) { + cmp = fcmp; + break; + } + continue; + } + if (va > vb) { + cmp = 1; + break; + } + if (va < vb) { + cmp = -1; + break; + } + } + if (cmp <= 0) + break; + memcpy(work + (size_t)j * nc, + work + (size_t)(j - 1) * nc, + (size_t)nc * sizeof(*work)); + if (timestamps) + timestamps[j] = timestamps[j - 1]; + j--; + } + if (timestamps) + timestamps[j] = saved_timestamp; + memcpy(work + (size_t)j * nc, tbuf, + (size_t)nc * sizeof(*work)); + } + + for (uint32_t c = 0; c < nc; c++) + for (uint32_t i = 0; i < nrows; i++) + r->columns[c][start_row + i] = work[(size_t)i * nc + c]; + if (timestamps) + memcpy(r->timestamps + start_row, timestamps, + (size_t)nrows * sizeof(*timestamps)); + return 0; +} +static int +wl_columnar_radix_k16_prepared(col_rel_t *r, uint32_t start_row, uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + col_delta_timestamp_t *timestamps) +{ + uint32_t nc = r->ncols; + + const uint32_t radix_bits = 16u; + const uint32_t num_passes = 64u / radix_bits; /* 4 for k=16 */ + const uint32_t hist_size = 1u << radix_bits; /* 65536 for k=16 */ + + uint32_t *perm_a = workspace->perm_a; + uint32_t *perm_b = workspace->perm_b; + uint16_t *bv_cache = workspace->short_values; + uint32_t *count = workspace->count16; for (uint32_t i = 0; i < nrows; i++) perm_a[i] = i; @@ -9189,18 +9388,7 @@ radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, #ifdef WL_RADIX_BENCH _t0 = now_ns(); #endif - int64_t *temp_col = workspace ? workspace->temp_column - : (int64_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(int64_t), "radix_temp_column"); - if (!temp_col) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - free(count); - } - return ENOMEM; - } + int64_t *temp_col = workspace->temp_column; for (uint32_t c = 0; c < nc; c++) { int64_t *col = r->columns[c]; for (uint32_t i = 0; i < nrows; i++) { @@ -9216,13 +9404,7 @@ radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, memcpy(r->timestamps + start_row, timestamps, (size_t)nrows * sizeof(*timestamps)); } - if (owns_workspace) { - free(temp_col); - free(perm_a); - free(perm_b); - free(bv_cache); - free(count); - } + #ifdef WL_RADIX_BENCH _tA = now_ns() - _t0; if (wl_columnar_relation_radix_bench_enabled()) { @@ -9241,27 +9423,8 @@ radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, return 0; } -/* - * col_rel_radix_sort: index-permutation LSD radix sort (Phase B, Issue #330). - * - * Sort sub-range [start_row, start_row + nrows) of r in-place. - * Uses col_rel_get() for key extraction (layout-independent). - * Sorts a permutation array instead of scattering full rows. - * Permutation is applied once at the end via col_rel_row_copy_out/in. - * - * Optimizations (Issue #343): - * - Hybrid threshold: insertion sort for nrows <= 32 - * - Skip-pass: skip byte positions where all values have the same byte - * - Byte-value cache: read column data once per pass, reuse for scatter - * - * Adaptive radix width (Issue #363 Phase 5c): - * - nrows >= 50000: k=16 via radix_sort_k16() — 4 passes × 65536 buckets - * - nrows < 50000: k=8 with SIMD fused uniform+count — 8 passes × 256 buckets - * - * Falls back to insertion sort on allocation failure. - */ static int -wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, +wl_columnar_radix_rows_prepared(col_rel_t *r, uint32_t start_row, uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, col_delta_timestamp_t *timestamps) @@ -9300,7 +9463,8 @@ wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, * already yields fewer total iterations than 8-pass k=8+SIMD, and a * 256KB histogram makes SIMD gather impractical (cache pressure). */ if (nrows >= 50000u) { - int rc = radix_sort_k16(r, start_row, nrows, workspace, timestamps); + int rc = wl_columnar_radix_k16_prepared(r, start_row, nrows, workspace, + timestamps); if (rc == 0) wl_columnar_relation_touch_view(r); return rc; @@ -9313,29 +9477,9 @@ wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, const uint32_t radix_bits = 8u; const uint32_t num_passes = 64u / radix_bits; /* 8 for k=8 */ - bool owns_workspace = workspace == NULL; - uint32_t *perm_a = workspace ? workspace->perm_a - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_a"); - uint32_t *perm_b = workspace ? workspace->perm_b - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_b"); - uint8_t *bv_cache = workspace ? workspace->byte_values - : (uint8_t *)wl_columnar_relation_radix_malloc( - nrows, "radix_byte_values"); - if (workspace && workspace->k8_capacity < nrows) - return EINVAL; - if (!perm_a || !perm_b || !bv_cache) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - } - return nrows <= 32 ? col_rel_insertion_sort(r, start_row, nrows, - workspace, timestamps) - : ENOMEM; - } - + uint32_t *perm_a = workspace->perm_a; + uint32_t *perm_b = workspace->perm_b; + uint8_t *bv_cache = workspace->byte_values; for (uint32_t i = 0; i < nrows; i++) perm_a[i] = i; @@ -9411,17 +9555,7 @@ wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, _t0 = now_ns(); #endif /* Apply permutation per-column (Issue #334): contiguous access pattern. */ - int64_t *temp_col = workspace ? workspace->temp_column - : (int64_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(int64_t), "radix_temp_column"); - if (!temp_col) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - } - return ENOMEM; - } + int64_t *temp_col = workspace->temp_column; for (uint32_t c = 0; c < nc; c++) { int64_t *col = r->columns[c]; /* Gather: prefetch 8 elements ahead (Issue #363 Phase 3). */ @@ -9438,12 +9572,7 @@ wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, memcpy(r->timestamps + start_row, timestamps, (size_t)nrows * sizeof(*timestamps)); } - if (owns_workspace) { - free(temp_col); - free(perm_a); - free(perm_b); - free(bv_cache); - } + #ifdef WL_RADIX_BENCH _tA = now_ns() - _t0; if (wl_columnar_relation_radix_bench_enabled()) { @@ -9464,7 +9593,8 @@ wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, insertion: { - int rc = col_rel_insertion_sort(r, start_row, nrows, workspace, + int rc = wl_columnar_radix_insertion_prepared(r, start_row, nrows, + workspace, timestamps); if (rc == 0) wl_columnar_relation_touch_view(r); @@ -9507,6 +9637,17 @@ wl_columnar_radix_workspace_destroy(wl_columnar_radix_workspace_t *workspace) memset(workspace, 0, sizeof(*workspace)); } +static uint64_t +wl_columnar_radix_type_fingerprint(const col_rel_t *rel) +{ + uint64_t hash = UINT64_C(14695981039346656037); + for (uint32_t c = 0; rel->column_types && c < rel->ncols; c++) { + hash ^= (uint64_t)rel->column_types[c]; + hash *= UINT64_C(1099511628211); + } + return hash; +} + static int wl_columnar_relation_radix_workspace_prepare(const col_rel_t *rel, const uint32_t *seg_boundaries, uint32_t seg_count, @@ -9536,7 +9677,8 @@ wl_columnar_relation_radix_workspace_prepare(const col_rel_t *rel, || workspace->temp_column || workspace->insertion_rows || workspace->timestamps || workspace->timestamp_capacity || workspace->k8_capacity || workspace->k16_capacity - || workspace->insertion_capacity || workspace->insertion_bytes) + || workspace->insertion_capacity || workspace->insertion_bytes + || workspace->prepared_relation) return EINVAL; /* Check the entire boundary sequence before comparing any row. Partial * relation ranges and empty segments are valid preparation inputs. */ @@ -9706,6 +9848,24 @@ wl_columnar_relation_radix_workspace_prepare(const col_rel_t *rel, workspace->insertion_capacity = insertion_rows; workspace->insertion_bytes = insertion_bytes; } + /* Exact provenance is available only for a complete single range. Raw + * multi-segment consumers continue using capacity validation. */ + if (seg_count == 1) { + workspace->prepared_relation = rel; + workspace->prepared_identity = rel->relation_identity; + workspace->prepared_view_generation = rel->view_generation; + workspace->prepared_storage_generation = rel->storage_generation; + workspace->prepared_start = seg_boundaries[0]; + workspace->prepared_count = seg_boundaries[1] - seg_boundaries[0]; + workspace->prepared_nrows = rel->nrows; + workspace->prepared_ncols = rel->ncols; + workspace->prepared_capacity = rel->capacity; + workspace->prepared_timestamp_capacity = rel->timestamp_capacity; + workspace->prepared_has_timestamps = rel->timestamps != NULL; + workspace->prepared_has_types = rel->column_types != NULL; + workspace->prepared_type_fingerprint = + wl_columnar_radix_type_fingerprint(rel); + } return 0; overflow: @@ -9935,7 +10095,8 @@ col_rel_radix_sort_impl(col_rel_t *r, uint32_t start_row, uint32_t nrows, return rc; } -/* Sort under a writer lease the caller already holds, so a wider mutation +/* Transitional owner-only authority for Unit 2B3b2 consolidation callers. + * Sort under a writer lease the caller already holds, so a wider mutation * transaction can keep admission across the sort. * * defer_alias_release selects what happens to the borrow the deferred COW @@ -9986,60 +10147,135 @@ col_rel_radix_sort_locked(col_rel_t *r, uint32_t start_row, uint32_t nrows, return rc; } -/* Standalone entry point: takes the canonical-owner lease itself and holds - * it across the whole COW, permutation and publication transaction. */ +/* Validate authority and physical shape before reading keys or allocating. */ +static int +wl_columnar_radix_lease_preflight(col_rel_t *r, uint32_t start, + uint32_t count, wl_columnar_relation_mutation_lease_t *lease) +{ + if (!lease || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + if (start > r->nrows || count > r->nrows - start + || r->nrows > r->capacity || !col_rel_timestamp_shape_valid(r) + || (r->ncols && !r->columns)) + return EINVAL; + for (uint32_t c = 0; c < r->ncols; c++) + if (!r->columns[c]) + return EINVAL; + if (!wl_columnar_relation_generation_valid(r->view_generation)) + return EINVAL; + if (r->ncols && count > 1 + && (r->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u + || (r->col_shared && r->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u))) + return EOVERFLOW; + return 0; +} + int -col_rel_radix_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows) +wl_columnar_relation_radix_workspace_prepare_with_lease(col_rel_t *r, + uint32_t start, uint32_t count, + wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease) { - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_release_pending = false; - int rc; + int rc = wl_columnar_radix_lease_preflight(r, start, count, lease); + if (rc) + return rc; + uint32_t bounds[] = { start, start + count }; + return wl_columnar_relation_radix_workspace_prepare(r, bounds, 1, 0, + workspace, false); +} - if (!r || start_row > r->nrows || nrows > r->nrows - start_row) +int +wl_columnar_relation_radix_sort_with_lease(col_rel_t *r, uint32_t start, + uint32_t count, const wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease) +{ + int rc = wl_columnar_radix_lease_preflight(r, start, count, lease); + if (rc) + return rc; + if (!workspace) return EINVAL; - rc = col_rel_source_writer_acquire(r, &writer); - if (rc != 0) + if (workspace->prepared_relation != r + || workspace->prepared_identity != r->relation_identity + || workspace->prepared_view_generation != r->view_generation + || workspace->prepared_storage_generation != r->storage_generation + || workspace->prepared_start != start || + workspace->prepared_count != count + || workspace->prepared_nrows != r->nrows + || workspace->prepared_ncols != r->ncols + || workspace->prepared_capacity != r->capacity + || workspace->prepared_timestamp_capacity != r->timestamp_capacity + || workspace->prepared_has_timestamps != (r->timestamps != NULL) + || workspace->prepared_has_types != (r->column_types != NULL) + || workspace->prepared_type_fingerprint != + wl_columnar_radix_type_fingerprint(r)) + return EINVAL; + if (!r->ncols || count <= 1) + return 0; + rc = wl_columnar_relation_radix_workspace_validate(r, count, workspace); + if (rc || workspace->byte_values != workspace->bucket_values + || (void *)workspace->short_values != workspace->bucket_values) + return rc ? rc : EINVAL; + if (r->col_shared) { + rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); + if (rc) + return rc; + } + /* After detach/first row move, prepared-only kernels cannot fail. */ + rc = wl_columnar_radix_rows_prepared(r, start, count, workspace, + r->timestamps ? workspace->timestamps : NULL); + if (rc) + abort(); + return 0; +} + +/* One descriptor-first mutation set covers range capture, preparation, + * detach, permutation and optional full-sort metadata publication. */ +static int +wl_columnar_radix_sort_admitted(col_rel_t *r, uint32_t start, + uint32_t count, bool full_relation) +{ + wl_columnar_relation_mutation_single_t single = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + if (!r) + return EINVAL; + if (!full_relation && count > UINT32_MAX - start) + return EINVAL; + int rc = col_rel_mutation_single_acquire(r, &single); + if (rc) return rc; r->memory_budget_denial_pending = false; - rc = col_rel_radix_sort_locked(r, start_row, nrows, &writer, false, - &alias_release_pending); - if (wl_columnar_source_access_writer_release(&writer) != 0 && rc == 0) - rc = EINVAL; + if (full_relation) { + start = 0; + count = r->nrows; + } + rc = wl_columnar_radix_lease_preflight(r, start, count, &single.lease); + if (!rc) { + rc = wl_columnar_relation_radix_workspace_prepare_with_lease(r, + start, count, &workspace, &single.lease); + } + if (!rc) + rc = wl_columnar_relation_radix_sort_with_lease(r, start, count, + &workspace, &single.lease); + if (!rc && full_relation) + r->sorted_nrows = count; + wl_columnar_radix_workspace_destroy(&workspace); + if (col_rel_mutation_set_finish(&single.set, rc == 0)) + abort(); return rc; } -/* - * col_rel_radix_sort_int64: sort all rows of r in-place using LSD radix sort. - * - * Sorts lexicographically by all ncols columns (column 0 is most significant). - * Handles signed int64_t by flipping the sign bit on the MSB of each column - * so that unsigned byte comparison yields the correct signed ordering. - * - * Complexity: O(ncols * 8 * nrows) time, O(nrows * ncols) extra space. - * Sets r->sorted_nrows = r->nrows on completion. - * Falls back to insertion sort on allocation failure. - */ +int +col_rel_radix_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows) +{ + return wl_columnar_radix_sort_admitted(r, start_row, nrows, false); +} + int col_rel_radix_sort_int64(col_rel_t *r) { - if (!r) { - return 0; - } - /* col_rel_radix_sort() detaches a shared view itself (through the - * admitted col_rel_cow_unshare) and rolls the detach back if the sort - * cannot allocate, so no COW happens here. It takes the canonical-owner - * lease, so it can also refuse with EBUSY when the relation is an owner - * with live alias borrows or the gate is contended. Callers must not - * treat the relation as sorted afterwards: sorted_nrows is left alone - * here, and the status is returned so a caller that depends on the - * ordering can stop rather than silently dedup an unsorted relation. */ - int rc = col_rel_radix_sort(r, 0, r->nrows); - if (rc == 0) - r->sorted_nrows = r->nrows; - else - WL_LOG(WL_LOG_SEC_CONSOLIDATION, WL_LOG_WARN, - "radix sort refused for %s: rc=%d", r->name ? r->name : "?", rc); - return rc; + return r ? wl_columnar_radix_sort_admitted(r, 0, 0, true) : 0; } int From 2a41e98f818082f72b3782af1e187a91b185164f Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sat, 3 Oct 2026 17:46:35 +0900 Subject: [PATCH 7/9] fix(columnar): preserve legacy consolidation cow path --- wirelog/columnar/merge.c | 2 +- wirelog/columnar/relation.c | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index cd3967bc..885da169 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -2190,7 +2190,7 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, bool compact = rel->run_count >= COL_MAX_RUNS; col_rel_compact_scratch_t scratch; if (rel->col_shared && compact) { - int cow_rc = col_rel_cow_unshare_with_source_writer(rel, + int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, rel_writer); if (cow_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index ec756977..3c1658fc 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -5033,7 +5033,8 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, * path so public COW may acquire its own gate independently. */ if (dst->col_shared) { dst->memory_budget_denial_pending = false; - rc = col_rel_cow_unshare_impl(dst, 0, false, false, NULL); + rc = col_rel_cow_unshare_legacy_impl(dst, 0, false, false, + NULL); } else { rc = 0; } From a19fce0f8fabd8dcb32e41df515bdefe482edd72 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sat, 3 Oct 2026 18:22:44 +0900 Subject: [PATCH 8/9] ci: make pre-stable binary size advisory --- docs/BINARY_SIZE.md | 22 +++++++++++--- scripts/ci/check-text-size.sh | 19 ++++++++++-- scripts/ci/run-size-comparison.sh | 16 +++++++++- scripts/ci/test-early-size-gate.sh | 28 +++++++++++------ scripts/ci/test-size-comparison.py | 47 +++++++++++++++++++++++++---- scripts/ci/test-text-size-policy.sh | 10 ++++-- scripts/ci/text-size-policy.py | 12 +++++--- tests/size_policy_mode.txt | 1 + 8 files changed, 126 insertions(+), 29 deletions(-) create mode 100644 tests/size_policy_mode.txt diff --git a/docs/BINARY_SIZE.md b/docs/BINARY_SIZE.md index dc21579b..ea95be0a 100644 --- a/docs/BINARY_SIZE.md +++ b/docs/BINARY_SIZE.md @@ -1,5 +1,16 @@ # Binary size monitoring and admission +## Pre-stable policy mode + +Until the user explicitly declares a stable version, the 5120-byte allowance +and resulting `.text` ceiling are reference values. PR CI reports a valid +overage as `over-budget` but exits successfully. Measurement failures, +production profile mismatches, malformed provenance, and tested-merge identity +failures remain blocking. The comparison reads the policy mode from the event +base; older bases without `tests/size_policy_mode.txt` use advisory mode. A +future switch to `enforced` requires the user's stable-version declaration and +a separate reviewed policy change. + `tests/baseline_size.txt` records 419135 bytes, the canonical Ubuntu x86_64 GCC `.text` measurement from eligible successful main ancestor `c6e263d492828d1208fa11005c18d19b05342233`, rather than current main or PR #2032. @@ -33,17 +44,18 @@ Historically, the 408989-byte baseline at `13d9244a` rose to 414051 bytes at `863e011e`, consuming 5062 bytes of the same fixed allowance. This reset replaces that ancestor measurement with the authenticated 419135-byte figure. -The production limit remains 5120 bytes. PR CI compares the production shared -library from the exact `pull_request.base.sha` tree with the library from the +The production reference allowance remains 5120 bytes. PR CI compares the +production shared library from the exact `pull_request.base.sha` tree with the library from the tested merge SHA. It verifies that the tested SHA is the merge commit and that its first parent is the event base. Both libraries are configured and built on the same runner with `-Dtests=true -DmbedTLS=disabled`, and the resolved Meson options, compiler/linker identity and version, target, platform, and effective wirelog build commands must match. A profile mismatch is an error. The -candidate's baseline file is used only after its main CI provenance is verified. +candidate baseline is used only after its CI provenance is verified. -Normally the head may be at most baseline + 5120 bytes. If the measured base -already exceeds that ceiling, the head may not exceed the measured base. A +The reference allowance normally places the head at baseline + 5120 bytes. If +the measured base already exceeds that ceiling, the reference size is the +measured base. In enforced mode, a head above that allowed size fails. A docs-only change receives no special exemption; equal measured binaries pass because they add no size to the base. diff --git a/scripts/ci/check-text-size.sh b/scripts/ci/check-text-size.sh index fd19385a..70dd819a 100755 --- a/scripts/ci/check-text-size.sh +++ b/scripts/ci/check-text-size.sh @@ -10,8 +10,11 @@ json_out= source_sha=${SOURCE_SHA:-unknown} profile_file=${SIZE_PROFILE_FILE:-} measure_only=no +mode_file="$repo_root/tests/size_policy_mode.txt" +mode=advisory +mode_explicit=no -usage() { echo "usage: $0 [--baseline-file FILE] [--json FILE] [--source-sha SHA] [--profile FILE]" >&2; exit 2; } +usage() { echo "usage: $0 [--baseline-file FILE] [--json FILE] [--source-sha SHA] [--profile FILE] [--mode advisory|enforced] [--mode-file FILE]" >&2; exit 2; } [ "$#" -ge 1 ] || usage library=$1; shift while [ "$#" -gt 0 ]; do @@ -20,6 +23,8 @@ while [ "$#" -gt 0 ]; do --json) [ "$#" -ge 2 ] || usage; json_out=$2; shift 2 ;; --source-sha) [ "$#" -ge 2 ] || usage; source_sha=$2; shift 2 ;; --profile) [ "$#" -ge 2 ] || usage; profile_file=$2; shift 2 ;; + --mode) [ "$#" -ge 2 ] || usage; mode=$2; mode_explicit=yes; shift 2 ;; + --mode-file) [ "$#" -ge 2 ] || usage; mode_file=$2; shift 2 ;; --measure-only) measure_only=yes; shift ;; *) usage ;; esac @@ -27,6 +32,12 @@ done fail() { printf 'error: %s\n' "$1" >&2; exit 2; } [ -f "$library" ] || fail "library not found: $library" [ "$measure_only" = yes ] || [ -f "$baseline_file" ] || fail "baseline file not found: $baseline_file" +if [ "$measure_only" != yes ]; then + if [ "$mode_explicit" = no ] && [ -f "$mode_file" ]; then + mode=$(cat "$mode_file") || fail "size policy mode file unreadable: $mode_file" + fi + case "$mode" in advisory|enforced) ;; *) fail "invalid size policy mode: $mode" ;; esac +fi case $(uname -s) in Linux) raw=$(size --format=sysv "$library") || fail "size failed for $library"; section=.text ;; @@ -65,8 +76,12 @@ fi if [ "$measure_only" = yes ]; then exit 0 fi -if [ "$status" = over-budget ]; then +if [ "$status" = over-budget ] && [ "$mode" = enforced ]; then printf '\nFAIL: .text growth (%+d) exceeds %s-byte budget\n' "$delta" "$threshold" >&2 exit 1 fi +if [ "$status" = over-budget ]; then + printf '\nADVISORY: .text growth (%+d) exceeds %s-byte reference budget\n' "$delta" "$threshold" + exit 0 +fi printf '\nPASS: .text delta within budget\n' diff --git a/scripts/ci/run-size-comparison.sh b/scripts/ci/run-size-comparison.sh index 676aba70..875e54de 100755 --- a/scripts/ci/run-size-comparison.sh +++ b/scripts/ci/run-size-comparison.sh @@ -4,6 +4,12 @@ set -eu script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd -P) repo_root=$(CDPATH= cd -- "$script_dir/../.." && pwd -P) +case "${TMPDIR:-}" in ''|/tmp|/tmp/*|/dev/shm|/dev/shm/*) + TMPDIR="${HOME:?HOME must be set}/.tmp" + export TMPDIR + ;; +esac +mkdir -p "$TMPDIR" || { printf 'size comparison setup error: cannot create TMPDIR %s\n' "$TMPDIR" >&2; exit 2; } head_build=${1:-} base_sha=${2:-} head_sha=${3:-} @@ -35,6 +41,14 @@ baseline=$(git show "$base_sha:tests/baseline_size.txt") || die "base-owned base printf '%s\n' "$baseline" >"$tmp/base-baseline.txt" case "$baseline" in ''|*[!0-9]*) die "base-owned baseline is invalid" ;; esac policy_baseline=$baseline +# Until the user declares a stable version, the size allowance is advisory. +# Older event bases lack the policy file and inherit that pre-stable mode. +policy_mode=advisory +if git cat-file -e "$base_sha:tests/size_policy_mode.txt" 2>/dev/null; then + policy_mode=$(git show "$base_sha:tests/size_policy_mode.txt") \ + || die "base-owned size policy mode is unreadable" +fi +case "$policy_mode" in advisory|enforced) ;; *) die "base-owned size policy mode is invalid" ;; esac git show "$head_sha:tests/baseline_size.provenance.json" >"$tmp/head-baseline.provenance.json" 2>/dev/null \ || die "candidate baseline provenance sidecar is missing" @@ -67,4 +81,4 @@ base_profile=$(python3 -c 'import json,sys; d=json.load(open(sys.argv[1])); d.po head_profile=$(python3 -c 'import json,sys; d=json.load(open(sys.argv[1])); d.pop("source_sha",None); import hashlib; print(hashlib.sha256(json.dumps(d,sort_keys=True,separators=(",",":")).encode()).hexdigest())' "$tmp/head-profile.json") python3 "$script_dir/text-size-policy.py" --base-size "$base_bytes" --head-size "$head_bytes" \ --baseline "$policy_baseline" --base-profile "$base_profile" --head-profile "$head_profile" \ - --base-sha "$base_sha" --head-sha "$head_sha" --output "$report" + --base-sha "$base_sha" --head-sha "$head_sha" --mode "$policy_mode" --output "$report" diff --git a/scripts/ci/test-early-size-gate.sh b/scripts/ci/test-early-size-gate.sh index 6b1400f5..f8ba86f8 100755 --- a/scripts/ci/test-early-size-gate.sh +++ b/scripts/ci/test-early-size-gate.sh @@ -11,12 +11,9 @@ # the workflow would leave the behavioural assertions green -- the # silent-downgrade shape this gate exists to prevent. # -# 2. Behaviour -- the real gate (scripts/ci/check-text-size.sh) FAILs on -# an intentionally oversize library and PASSes on a small one, against -# the committed baseline. A fixture that only ever supplied a passing -# library would pin nothing: the negative control must drive the gate's -# own fail path, which is what #1573 moved earlier in the job so a size -# regression fails within minutes instead of ~24. +# 2. Behaviour -- the real gate reports an intentionally oversize library +# as advisory in pre-stable mode and fails it in explicitly enforced +# mode. A small library passes in both modes. # # The wiring half runs FIRST and needs only awk + the workflow file; the # behavioural half needs cc/size and platform-specific size tools, and is the @@ -186,7 +183,9 @@ if ! command -v cc >/dev/null 2>&1 || ! command -v size >/dev/null 2>&1 \ exit 77 fi -tmp=$(mktemp -d "${TMPDIR:-/tmp}/wirelog-early-size.XXXXXX") +case "${TMPDIR:-}" in ''|/tmp|/tmp/*|/dev/shm|/dev/shm/*) TMPDIR="${HOME:?HOME must be set}/.tmp"; export TMPDIR ;; esac +mkdir -p "$TMPDIR" +tmp=$(mktemp -d "$TMPDIR/wirelog-early-size.XXXXXX") trap 'rm -rf "$tmp"' EXIT # Negative control: size it relative to the committed baseline so it remains @@ -240,12 +239,23 @@ else printf 'test-early-size-gate: generated fixture is not above the budget (%s <= %s)\n' "$actual_size" "$budget_limit" >&2 failures=$((failures + 1)) fi - negative_control() { + advisory_control() { local st=0 "$gate" "$big_so" >/dev/null 2>&1 || st=$? + [ "$st" = 0 ] + } + assert 'oversize library is advisory before stable declaration' advisory_control + enforced_control() { + local st=0 + "$gate" "$big_so" --mode enforced >/dev/null 2>&1 || st=$? [ "$st" = 1 ] } - assert 'oversize library FAILs the size gate (negative control)' negative_control + assert 'oversize library fails in enforced mode' enforced_control + advisory_override() { + printf 'enforced\n' >"$tmp/enforced-mode.txt" + "$gate" "$big_so" --mode-file "$tmp/enforced-mode.txt" --mode advisory >/dev/null 2>&1 + } + assert 'explicit advisory mode overrides the default mode file' advisory_override fi # Positive control: a small library must PASS against the same baseline. diff --git a/scripts/ci/test-size-comparison.py b/scripts/ci/test-size-comparison.py index 6847ecd6..c674e884 100755 --- a/scripts/ci/test-size-comparison.py +++ b/scripts/ci/test-size-comparison.py @@ -7,6 +7,11 @@ import sys import tempfile +if (os.environ.get("TMPDIR", "").startswith(("/tmp/", "/dev/shm/")) or + os.environ.get("TMPDIR") in ("/tmp", "/dev/shm", None, "")): + os.environ["TMPDIR"] = str(Path.home()/".tmp") +Path(os.environ["TMPDIR"]).mkdir(parents=True, exist_ok=True) + root=Path(sys.argv[1]).resolve() orchestrator=root/"scripts/ci/run-size-comparison.sh" if sys.platform != "linux": @@ -45,14 +50,14 @@ def write_source(nops): run(["git","add","."],cwd=repo); run(["git","commit","-qm","base"],cwd=repo) base=run(["git","rev-parse","HEAD"],cwd=repo).stdout.strip() - def commit_branch(name, edit): - run(["git","checkout","-q","-b",name,base],cwd=repo) + def commit_branch(name, edit, from_base=base): + run(["git","checkout","-q","-b",name,from_base],cwd=repo) edit() run(["git","add","."],cwd=repo); run(["git","commit","-qm",name],cwd=repo) return run(["git","rev-parse","HEAD"],cwd=repo).stdout.strip() - def merge_commit(pr_head): + def merge_commit(pr_head, merge_base=base): tree=run(["git","rev-parse",f"{pr_head}^{{tree}}"],cwd=repo).stdout.strip() - return run(["git","commit-tree",tree,"-p",base,"-p",pr_head,"-m","synthetic PR merge"],cwd=repo).stdout.strip() + return run(["git","commit-tree",tree,"-p",merge_base,"-p",pr_head,"-m","synthetic PR merge"],cwd=repo).stdout.strip() def checkout_merge(merge): run(["git","update-ref","refs/heads/main",merge],cwd=repo) run(["git","checkout","-q","-f","main"],cwd=repo) @@ -91,11 +96,41 @@ def compare(base_sha, pr_sha, merge_sha, expected=0, verifier_result=None): pr_growth=commit_branch("pr-growth",lambda: write_source(10100)) merge_growth=merge_commit(pr_growth); checkout_merge(merge_growth) - compare(base,pr_growth,merge_growth,1) + compare(base,pr_growth,merge_growth,0) growth_report=json.loads(report.read_text(encoding="utf-8")) assert growth_report["status"]=="over-budget" and growth_report["head_bytes"]>growth_report["base_bytes"] assert growth_report["allowed_head_bytes"]==growth_report["base_bytes"] assert growth_report["base_sha"]==base and growth_report["head_sha"]==merge_growth + assert growth_report["mode"]=="advisory" + + run(["git","checkout","-q","-b","enforced-base",base],cwd=repo) + (repo/"tests/size_policy_mode.txt").write_text("enforced\n",encoding="utf-8") + run(["git","add","."],cwd=repo); run(["git","commit","-qm","enforced policy"],cwd=repo) + enforced_base=run(["git","rev-parse","HEAD"],cwd=repo).stdout.strip() + enforced_growth=commit_branch("pr-enforced-growth",lambda: write_source(10100),enforced_base) + enforced_merge=merge_commit(enforced_growth,enforced_base); checkout_merge(enforced_merge) + compare(enforced_base,enforced_growth,enforced_merge,1) + enforced_report=json.loads(report.read_text(encoding="utf-8")) + assert enforced_report["mode"]=="enforced" and enforced_report["status"]=="over-budget" + + def candidate_downgrades_policy(): + (repo/"tests/size_policy_mode.txt").write_text("advisory\n",encoding="utf-8") + write_source(10100) + downgrade_head=commit_branch("pr-policy-downgrade",candidate_downgrades_policy,enforced_base) + downgrade_merge=merge_commit(downgrade_head,enforced_base); checkout_merge(downgrade_merge) + compare(enforced_base,downgrade_head,downgrade_merge,1) + downgrade_report=json.loads(report.read_text(encoding="utf-8")) + assert downgrade_report["mode"]=="enforced" and downgrade_report["status"]=="over-budget" + + run(["git","checkout","-q","-b","invalid-mode-base",base],cwd=repo) + (repo/"tests/size_policy_mode.txt").write_text("invalid\n",encoding="utf-8") + run(["git","add","."],cwd=repo); run(["git","commit","-qm","invalid policy mode"],cwd=repo) + invalid_base=run(["git","rev-parse","HEAD"],cwd=repo).stdout.strip() + invalid_head=commit_branch("pr-invalid-mode",lambda: write_source(10100),invalid_base) + invalid_merge=merge_commit(invalid_head,invalid_base); checkout_merge(invalid_merge) + invalid=compare(invalid_base,invalid_head,invalid_merge,2) + assert "base-owned size policy mode is invalid" in invalid.stderr + assert not report.exists(), "invalid event-base mode must fail before policy output" def inflate_candidate_baseline(): (repo/"tests/baseline_size.txt").write_text("999999999\n", encoding="utf-8") @@ -122,4 +157,4 @@ def inflate_candidate_baseline(): result=compare(pr_doc,pr_inflate,merge_inflate,2) assert "first parent differs" in result.stderr -print("test-size-comparison: exact merge, inherited debt, growth, and baseline authorization passed") +print("test-size-comparison: exact merge, advisory and enforced growth, inherited debt, and baseline authorization passed") diff --git a/scripts/ci/test-text-size-policy.sh b/scripts/ci/test-text-size-policy.sh index fda7d1c1..99e65ddd 100755 --- a/scripts/ci/test-text-size-policy.sh +++ b/scripts/ci/test-text-size-policy.sh @@ -1,23 +1,29 @@ #!/usr/bin/env bash set -euo pipefail ROOT=${1:-$(CDPATH= cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd -P)} +case "${TMPDIR:-}" in ''|/tmp|/tmp/*|/dev/shm|/dev/shm/*) TMPDIR="${HOME:?HOME must be set}/.tmp"; export TMPDIR ;; esac +mkdir -p "$TMPDIR" POLICY="$ROOT/scripts/ci/text-size-policy.py" PROFILE="$ROOT/scripts/ci/size-profile.py" VERIFY="$ROOT/scripts/ci/verify-size-baseline.py" -TMP=$(mktemp -d "${TMPDIR:-/tmp}/wirelog-size-policy.XXXXXX") +TMP=$(mktemp -d "$TMPDIR/wirelog-size-policy.XXXXXX") trap 'rm -rf -- "$TMP"' EXIT pass=0 fail=0 check() { local name=$1 expected=$2; shift 2; local rc=0; "$@" >/dev/null 2>&1 || rc=$?; if [ "$rc" -eq "$expected" ]; then echo "ok: $name"; pass=$((pass+1)); else echo "FAIL: $name (exit $rc, expected $expected)" >&2; fail=$((fail+1)); fi; } -policy() { python3 "$POLICY" --base-size "$1" --head-size "$2" --baseline "$3" --base-profile "$4" --head-profile "$5" --base-sha base123 --head-sha head123; } +policy() { python3 "$POLICY" --base-size "$1" --head-size "$2" --baseline "$3" --base-profile "$4" --head-profile "$5" --base-sha base123 --head-sha head123 --mode enforced; } +advisory_policy() { python3 "$POLICY" --base-size "$1" --head-size "$2" --baseline "$3" --base-profile "$4" --head-profile "$5" --base-sha base123 --head-sha head123 --mode advisory; } check 'exact baseline + 5120 boundary passes' 0 policy 12000 15120 10000 same same check 'baseline + 5121 fails' 1 policy 12000 15121 10000 same same +check 'advisory overage reports without blocking' 0 advisory_policy 12000 15121 10000 same same check 'identical head passes when base is already over budget' 0 policy 15121 15121 10000 same same check 'smaller head passes against over-budget base' 0 policy 15121 14999 10000 same same check 'new growth above an over-budget base fails' 1 policy 15121 15122 10000 same same check 'profile mismatch is fail-closed' 2 policy 10000 10000 10000 base-profile head-profile check 'invalid size is fail-closed' 2 policy 10000 nope 10000 same same +check 'advisory profile mismatch remains fail-closed' 2 advisory_policy 10000 10000 10000 base-profile head-profile +check 'advisory malformed measurement remains fail-closed' 2 advisory_policy 10000 nope 10000 same same check 'overflow is fail-closed' 2 policy 10000 10000 999999999999999999999999 same same python3 - "$TMP" <<'PY' diff --git a/scripts/ci/text-size-policy.py b/scripts/ci/text-size-policy.py index d1990bae..9eae9434 100755 --- a/scripts/ci/text-size-policy.py +++ b/scripts/ci/text-size-policy.py @@ -17,18 +17,21 @@ def byte_count(value, name): raise ValueError(f"{name} is invalid or out of range") return number -def compare(base_size, head_size, baseline, base_profile, head_profile, threshold=5120): +def compare(base_size, head_size, baseline, base_profile, head_profile, threshold=5120, + mode="advisory"): base_size = byte_count(base_size, "base size") head_size = byte_count(head_size, "head size") baseline = byte_count(baseline, "baseline") threshold = byte_count(threshold, "threshold") + if mode not in ("advisory", "enforced"): + raise ValueError("mode must be advisory or enforced") if not base_profile or base_profile != head_profile: raise ValueError("production profile mismatch between base and head") ceiling = baseline + threshold inherited_overage = base_size > ceiling allowed = base_size if inherited_overage else ceiling status = "pass" if head_size <= allowed else "over-budget" - return {"schema_version": 1, "status": status, "base_bytes": base_size, + return {"schema_version": 1, "status": status, "mode": mode, "base_bytes": base_size, "head_bytes": head_size, "baseline_bytes": baseline, "budget_bytes": threshold, "allowed_head_bytes": allowed, "delta_from_baseline_bytes": head_size - baseline, @@ -45,13 +48,14 @@ def main(): parser.add_argument("--base-sha", required=True) parser.add_argument("--head-sha", required=True) parser.add_argument("--threshold", default="5120") + parser.add_argument("--mode", choices=("advisory", "enforced"), default="advisory") parser.add_argument("--output") args = parser.parse_args() try: if not args.base_sha or not args.head_sha or args.base_sha == "unknown" or args.head_sha == "unknown": raise ValueError("base/head source SHA is missing") report = compare(args.base_size, args.head_size, args.baseline, - args.base_profile, args.head_profile, args.threshold) + args.base_profile, args.head_profile, args.threshold, args.mode) report.update({"base_sha": args.base_sha, "head_sha": args.head_sha}) text = json.dumps(report, sort_keys=True, indent=2) + "\n" if args.output: @@ -60,7 +64,7 @@ def main(): print(text, end="") if report["inherited_overage"]: print("inherited over-budget base; head may not exceed measured base size", file=sys.stderr) - return 0 if report["status"] == "pass" else 1 + return 0 if report["status"] == "pass" or report["mode"] == "advisory" else 1 except (OSError, ValueError) as exc: print(f"text-size-policy: {exc}", file=sys.stderr) return 2 diff --git a/tests/size_policy_mode.txt b/tests/size_policy_mode.txt new file mode 100644 index 00000000..c4543066 --- /dev/null +++ b/tests/size_policy_mode.txt @@ -0,0 +1 @@ +advisory From cd2eb595f5d44df3e1c3a828d83203daf8e232a5 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sat, 3 Oct 2026 18:31:26 +0900 Subject: [PATCH 9/9] ci: reset binary size baseline from PR 2037 evidence --- docs/BINARY_SIZE.md | 48 ++++++++------------- scripts/ci/test-size-baseline-provenance.py | 45 +++++++++++++++++++ scripts/ci/verify-size-baseline.py | 30 ++++++++++++- tests/baseline_size.provenance.json | 25 +++++++---- tests/baseline_size.txt | 2 +- 5 files changed, 111 insertions(+), 39 deletions(-) diff --git a/docs/BINARY_SIZE.md b/docs/BINARY_SIZE.md index ea95be0a..d28c8453 100644 --- a/docs/BINARY_SIZE.md +++ b/docs/BINARY_SIZE.md @@ -11,38 +11,27 @@ base; older bases without `tests/size_policy_mode.txt` use advisory mode. A future switch to `enforced` requires the user's stable-version declaration and a separate reviewed policy change. -`tests/baseline_size.txt` records 419135 bytes, the canonical Ubuntu x86_64 -GCC `.text` measurement from eligible successful main ancestor -`c6e263d492828d1208fa11005c18d19b05342233`, rather than current main or PR #2032. -The successful -[main workflow run 36673112247](https://github.com/semantic-reasoning/wirelog/actions/runs/36673112247) -produced `wirelog-size-monitor-ubuntu-latest` artifact `11079113586`. -Its source, run, artifact, profile, library digests and measurement are pinned -in `tests/baseline_size.provenance.json`. The unchanged verifier checks those -identities and independently rebuilds that source to reproduce its profile and -`.text` size through the normal `trusted-ci-artifact` policy. +## Current baseline -The report records `configure`, `build` and `test` each as success, with status -`within-budget` and a 5084-byte delta against the previous 414051-byte baseline. -The previous baseline measured main ancestor `863e011e`; subsequent changes -on main contribute to the new ancestor measurement. After #2040 found no -coherent reduction sufficient to admit PR #2032, the maintainer authorized -this measured reset. It grants fresh headroom rather than claiming a size -reduction, and does not change TDD eligibility. The fixed 5120-byte allowance -is unchanged, so the resulting ceiling is 424255 bytes. A rebuilt library may -have different non-`.text` bytes due to LTO metadata. +`tests/baseline_size.txt` records 428046 bytes, measured from the production +library for PR #2037 head `2a41e98f818082f72b3782af1e187a91b185164f` on the +canonical Ubuntu x86_64 GCC profile. The base was measured at 421283 bytes; +the head exceeded the previous 419135-byte reference by 8911 bytes. The exact +run `37110844170`, job `111168520922`, tested merge, profile, and job-log digest +are pinned in `tests/baseline_size.provenance.json`. The verifier reproduces +the source profile and `.text` measurement and limits this one-time exception +to the listed repair paths. The reset records the measured PR head as the +reference; it does not claim a size reduction. A rebuilt library may have +different non-`.text` bytes due to LTO metadata. -Main's measurement at any given tip is a moving figure and is deliberately -not tracked here. Read it from the most recent concluded `main` run's -`wirelog-size-monitor-ubuntu-latest` artifact, and compute headroom as 424255 -minus that measurement. Recording an ancestor stays valid as main advances: -the verifier requires that source to be an ancestor of the event base. The -artifact must remain unexpired, and its source must still reproduce with the -recorded toolchain profile. +Main's measurement at any given tip is a moving figure. Read it from the most +recent concluded `main` run's `wirelog-size-monitor-ubuntu-latest` artifact. +Main-branch measurements remain useful for tracking size changes even while +the PR ceiling is advisory. Historically, the 408989-byte baseline at `13d9244a` rose to 414051 bytes at `863e011e`, consuming 5062 bytes of the same fixed allowance. This reset -replaces that ancestor measurement with the authenticated 419135-byte figure. +replaces that ancestor measurement with the PR #2037 measurement above. The production reference allowance remains 5120 bytes. PR CI compares the production shared library from the exact `pull_request.base.sha` tree with the library from the @@ -51,7 +40,8 @@ its first parent is the event base. Both libraries are configured and built on the same runner with `-Dtests=true -DmbedTLS=disabled`, and the resolved Meson options, compiler/linker identity and version, target, platform, and effective wirelog build commands must match. A profile mismatch is an error. The -candidate baseline is used only after its CI provenance is verified. +candidate baseline is used only after its CI evidence or exact reviewed-PR +record is verified. The reference allowance normally places the head at baseline + 5120 bytes. If the measured base already exceeds that ceiling, the reference size is the @@ -89,7 +79,7 @@ grant additional budget only when the verified measurement comes from an eligibl main revision distinct from the candidate. If artifact access, toolchain reproduction, or provenance validation fails, the update is rejected. No workflow writes or commits baseline changes. The only reviewed-PR exceptions -are the exact, one-time records for PRs #1959 and #1961 embedded in the verifier; +are the exact, one-time records for PRs #1959, #1961, and #2037 embedded in the verifier; each is restricted to its recorded measurement and repair paths. They grant no general PR-based baseline eligibility and retain the same 5120-byte allowance. diff --git a/scripts/ci/test-size-baseline-provenance.py b/scripts/ci/test-size-baseline-provenance.py index ff53c0e8..8ba766e2 100755 --- a/scripts/ci/test-size-baseline-provenance.py +++ b/scripts/ci/test-size-baseline-provenance.py @@ -15,6 +15,11 @@ from unittest import mock from types import SimpleNamespace +if (os.environ.get("TMPDIR", "").startswith(("/tmp/", "/dev/shm/")) or + os.environ.get("TMPDIR") in ("/tmp", "/dev/shm", None, "")): + os.environ["TMPDIR"] = str(Path.home()/".tmp") +Path(os.environ["TMPDIR"]).mkdir(parents=True, exist_ok=True) + path = Path(__file__).with_name("verify-size-baseline.py") spec = importlib.util.spec_from_file_location("verify_size_baseline", path) mod = importlib.util.module_from_spec(spec); spec.loader.exec_module(mod) @@ -186,6 +191,46 @@ def assert_json_redirect_rejected(): if (mod.REVIEWED_PR1961_BASELINE["base_repository_baseline_bytes"], mod.REVIEWED_PR1961_BASELINE["source_baseline_bytes"], mod.REVIEWED_PR1961_BASELINE["budget_bytes"]) != (395540, 397579, 5120) else None) +reviewed2037_paths = { + "docs/BINARY_SIZE.md", "scripts/ci/verify-size-baseline.py", + "scripts/ci/test-size-baseline-provenance.py", "scripts/ci/text-size-policy.py", + "scripts/ci/run-size-comparison.sh", "scripts/ci/check-text-size.sh", + "scripts/ci/test-text-size-policy.sh", "scripts/ci/test-size-comparison.py", + "scripts/ci/test-early-size-gate.sh", "tests/size_policy_mode.txt", "tests/baseline_size.txt", + "tests/baseline_size.provenance.json", +} +repo_root = Path(__file__).resolve().parents[2] +checked_provenance = json.loads( + (repo_root / "tests/baseline_size.provenance.json").read_text(encoding="utf-8")) +check("checked-in PR 2037 sidecar matches the exact verifier record", + lambda: (_ for _ in ()).throw(AssertionError("sidecar differs from verifier record")) + if checked_provenance != mod.REVIEWED_PR2037_BASELINE else None) +check("checked-in baseline is the measured PR 2037 head size", + lambda: (_ for _ in ()).throw(AssertionError("baseline differs from measured head")) + if (repo_root / "tests/baseline_size.txt").read_text(encoding="ascii").strip() + != str(mod.REVIEWED_PR2037_BASELINE["baseline_bytes"]) else None) +check("PR 2037 exception has only its reviewed policy and baseline paths", + lambda: (_ for _ in ()).throw(AssertionError("path scope changed")) + if mod.reviewed_pr_repair_paths(mod.REVIEWED_PR2037_BASELINE) != reviewed2037_paths else None) +check("PR 2037 pins the measured base, head, budget, and run identity", + lambda: (_ for _ in ()).throw(AssertionError("pinned values changed")) + if (mod.REVIEWED_PR2037_BASELINE["pr_number"], + mod.REVIEWED_PR2037_BASELINE["baseline_bytes"], + mod.REVIEWED_PR2037_BASELINE["base_repository_baseline_bytes"], + mod.REVIEWED_PR2037_BASELINE["measured_base_bytes"], + mod.REVIEWED_PR2037_BASELINE["run_id"], + mod.REVIEWED_PR2037_BASELINE["job_id"], + mod.REVIEWED_PR2037_BASELINE["budget_bytes"]) + != (2037, 428046, 419135, 421283, 37110844170, 111168520922, 5120) else None) +for field, value in (("base_sha", "d" * 40), ("source_sha", "f" * 40), + ("tested_merge_sha", "e" * 40), + ("run_id", 37110844171), ("job_id", 111168520923), + ("profile_sha256", "0" * 64), ("job_log_sha256", "1" * 64)): + altered_2037_record = dict(mod.REVIEWED_PR2037_BASELINE, **{field: value}) + check(f"altered PR 2037 {field} is rejected", + lambda p=altered_2037_record: mod.authorize_reviewed_pr( + "semantic-reasoning/wirelog", "base", "candidate", 419135, 428046, p, "token"), + "exact approved record") check("legacy PR 1959 measurement remains a separate compatible record", lambda: (_ for _ in ()).throw(AssertionError("legacy record changed")) if (mod.REVIEWED_PR_BASELINE["pr_number"] != 1959 or diff --git a/scripts/ci/verify-size-baseline.py b/scripts/ci/verify-size-baseline.py index c88b9b8b..ccd045ed 100755 --- a/scripts/ci/verify-size-baseline.py +++ b/scripts/ci/verify-size-baseline.py @@ -56,11 +56,29 @@ "budget_bytes": 5120, "measurement_status": "over-budget", } +REVIEWED_PR2037_BASELINE = { + "schema_version": 1, "status": "trusted-reviewed-pr-size-job", + "authorization_id": "pr-2037-reviewed-head-2a41e98f", + "repository": "semantic-reasoning/wirelog", "pr_number": 2037, + "baseline_bytes": 428046, + "source_sha": "2a41e98f818082f72b3782af1e187a91b185164f", + "base_sha": "a8b5b5dd144054558cd5a93cbfe03c301d243164", + "tested_merge_sha": "8afc19a14f978589aac1f953b7cd6ad0e952b068", + "run_id": 37110844170, "job_id": 111168520922, + "profile_sha256": "cd6cc2f2c54520ba56c8efc0241b17d722d00c8b569cb6c6bb38f5e10d5e1508", + "job_log_sha256": "406f85e46bf713be7f807c1fafddb7fdb90786ba15653b2758c376442d1a3d2c", + "base_repository_baseline_bytes": 419135, + "source_baseline_bytes": 419135, "measured_base_bytes": 421283, + "reported_baseline_bytes": 419135, + "budget_bytes": 5120, "measurement_status": "over-budget", +} + def reviewed_pr_baselines(): # A function keeps the legacy #1959 fixture patchable without weakening # exact-record equality for production authorization. - return (REVIEWED_PR_BASELINE, REVIEWED_PR1961_BASELINE) + return (REVIEWED_PR_BASELINE, REVIEWED_PR1961_BASELINE, + REVIEWED_PR2037_BASELINE) def reviewed_pr_measurement(p): if p == REVIEWED_PR_BASELINE: @@ -174,6 +192,16 @@ def reviewed_pr_repair_paths(p): "scripts/ci/test-size-baseline-provenance.py", "tests/baseline_size.txt", "tests/baseline_size.provenance.json", } + if p == REVIEWED_PR2037_BASELINE: + return { + "docs/BINARY_SIZE.md", "scripts/ci/verify-size-baseline.py", + "scripts/ci/test-size-baseline-provenance.py", + "scripts/ci/text-size-policy.py", "scripts/ci/run-size-comparison.sh", + "scripts/ci/check-text-size.sh", "scripts/ci/test-text-size-policy.sh", + "scripts/ci/test-size-comparison.py", "scripts/ci/test-early-size-gate.sh", + "tests/size_policy_mode.txt", + "tests/baseline_size.txt", "tests/baseline_size.provenance.json", + } fail("reviewed PR baseline record is not recognized") def authorize_reviewed_pr(repo, base_sha, candidate_sha, base_value, diff --git a/tests/baseline_size.provenance.json b/tests/baseline_size.provenance.json index 6013a617..8e06bcbd 100644 --- a/tests/baseline_size.provenance.json +++ b/tests/baseline_size.provenance.json @@ -1,12 +1,21 @@ { "schema_version": 1, - "status": "trusted-ci-artifact", + "status": "trusted-reviewed-pr-size-job", + "authorization_id": "pr-2037-reviewed-head-2a41e98f", "repository": "semantic-reasoning/wirelog", - "baseline_bytes": 419135, - "source_sha": "c6e263d492828d1208fa11005c18d19b05342233", - "run_id": 36673112247, - "artifact_id": 11079113586, - "artifact_sha256": "aeeba27ed1c1fe5fbb73438c0ac779cf230bb3c0f030c604bc4abc3a49de27d5", - "profile_sha256": "e998de7827ab8b2c665e9f895e4e2e6e0f17b29450ae468c39efe41464aa97a7", - "library_sha256": "0fb8818fbd4ad89f44041b67f4b578c6d812abb472df798c52f59a6e3e6d1bf4" + "pr_number": 2037, + "baseline_bytes": 428046, + "source_sha": "2a41e98f818082f72b3782af1e187a91b185164f", + "base_sha": "a8b5b5dd144054558cd5a93cbfe03c301d243164", + "tested_merge_sha": "8afc19a14f978589aac1f953b7cd6ad0e952b068", + "run_id": 37110844170, + "job_id": 111168520922, + "profile_sha256": "cd6cc2f2c54520ba56c8efc0241b17d722d00c8b569cb6c6bb38f5e10d5e1508", + "job_log_sha256": "406f85e46bf713be7f807c1fafddb7fdb90786ba15653b2758c376442d1a3d2c", + "base_repository_baseline_bytes": 419135, + "source_baseline_bytes": 419135, + "measured_base_bytes": 421283, + "reported_baseline_bytes": 419135, + "budget_bytes": 5120, + "measurement_status": "over-budget" } diff --git a/tests/baseline_size.txt b/tests/baseline_size.txt index 4bd903fd..1e632d6b 100644 --- a/tests/baseline_size.txt +++ b/tests/baseline_size.txt @@ -1 +1 @@ -419135 +428046