From 72cd3520bea858edb934293991343a6b18850e58 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sat, 3 Oct 2026 23:32:49 +0900 Subject: [PATCH 01/15] Add validated relation storage continuation --- tests/meson.build | 1 + tests/test_relation_generations.c | 6 +- tests/test_relation_mutation_set.c | 66 ++++++-- wirelog/columnar/internal.h | 7 + wirelog/columnar/relation.c | 232 +++++++++++++++++++++++++++-- 5 files changed, 283 insertions(+), 29 deletions(-) diff --git a/tests/meson.build b/tests/meson.build index 8dd5f827..1c5f0636 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -4272,6 +4272,7 @@ relation_mutation_set_exe = executable( backend_src, intern_src, arena_src, thread_src, workqueue_src, include_directories: [wirelog_inc, wirelog_src_inc], dependencies: [nanoarrow_dep, threads_dep, xxhash_dep, mbedtls_dep, math_dep], + c_args: ['-DWL_TEST_MUTATION_SET_HOOK=1'], ) test('relation_mutation_set', relation_mutation_set_exe, suite: ['unit', 'tsan'], timeout: 60) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 9c68e5ac..caaad6ac 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -6347,7 +6347,7 @@ typedef struct { wl_columnar_relation_mutation_set_t set; wl_columnar_relation_mutation_role_t role; wl_columnar_relation_mutation_descriptor_t descriptor; - wl_columnar_relation_mutation_owner_t owner; + wl_columnar_relation_mutation_owner_t owners[2]; wl_columnar_relation_mutation_lease_t lease; wl_columnar_relation_mutation_initialization_t initialization; } radix_test_lease_t; @@ -6358,7 +6358,7 @@ radix_test_acquire(col_rel_t *rel, radix_test_lease_t *held) held->role = (wl_columnar_relation_mutation_role_t){ rel, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION }; return col_rel_mutation_set_acquire(&held->set, &held->role, 1, - &held->descriptor, 1, &held->owner, 1, &held->lease, 1, + &held->descriptor, 1, held->owners, 2, &held->lease, 1, &held->initialization, 1); } @@ -6905,7 +6905,7 @@ test_radix_exact_authority_and_shape(void) "radix real same-shape foreign preparation is refused"); wl_columnar_radix_workspace_destroy(&foreign_workspace); wl_columnar_source_access_writer_t writer_copy = held.descriptor.writer; - held.descriptor.writer = held.owner.writer; + held.descriptor.writer = held.owners[held.lease.owner_slot].writer; CHECK(wl_columnar_relation_radix_sort_with_lease(rel, 1, 70, &workspace, &held.lease) == EINVAL, "radix copied foreign descriptor token rejected"); diff --git a/tests/test_relation_mutation_set.c b/tests/test_relation_mutation_set.c index c253e478..5a37994b 100644 --- a/tests/test_relation_mutation_set.c +++ b/tests/test_relation_mutation_set.c @@ -10,7 +10,7 @@ typedef struct { wl_columnar_relation_mutation_set_t set; wl_columnar_relation_mutation_descriptor_t descriptors[4]; - wl_columnar_relation_mutation_owner_t owners[4]; + wl_columnar_relation_mutation_owner_t owners[8]; wl_columnar_relation_mutation_lease_t leases[4]; wl_columnar_relation_mutation_initialization_t initializations[4]; } fixture_t; @@ -19,7 +19,7 @@ static int acquire(fixture_t *f, wl_columnar_relation_mutation_role_t *roles, size_t n) { return col_rel_mutation_set_acquire(&f->set, roles, n, - f->descriptors, 4, f->owners, 4, f->leases, 4, + f->descriptors, 4, f->owners, 8, f->leases, 4, f->initializations, 4); } @@ -80,7 +80,14 @@ lease_tests(void) {&a, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} }; assert(acquire(&f, roles, 3) == 0); - assert(f.set.descriptor_count == 2 && f.set.owner_count == 1); + assert(f.set.descriptor_count == 2 && f.set.owner_count == 3); + assert(wl_columnar_source_access_gate_busy(&root.source_access)); + assert(wl_columnar_source_access_gate_busy(&a.source_access)); + assert(wl_columnar_source_access_gate_busy(&b.source_access)); + assert(f.owners[f.leases[0].owner_slot].owner == &root + && f.owners[f.leases[0].self_owner_slot].owner == &a + && f.owners[f.leases[1].owner_slot].owner == &root + && f.owners[f.leases[1].self_owner_slot].owner == &b); assert(&f.leases[0] != &f.leases[2]); for (size_t i = 0; i < 3; i++) assert(col_rel_mutation_set_lease(&f.set, i) == &f.leases[i]); @@ -108,17 +115,16 @@ lease_tests(void) wl_thread_t thread; assert(wl_thread_create(&thread, cross_thread, &f.leases[0]) == 0); assert(wl_thread_join(&thread) == 0); - assert(col_rel_storage_alias_release_locked(&a, &f.leases[0]) == 0); - assert(a.storage_owner == &a && - col_rel_storage_alias_borrow_count(&root) == 1); - assert(col_rel_storage_alias_release_locked(&a, &f.leases[0]) == EINVAL); - assert(col_rel_mutation_lease_validate(&f.leases[2], &a) == EINVAL); - assert(col_rel_storage_alias_release_locked(&b, &f.leases[1]) == 0); + assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == 0 + && col_rel_mutation_lease_validate(&f.leases[1], &b) == 0 + && col_rel_mutation_lease_validate(&f.leases[2], &a) == 0); + assert(col_rel_storage_alias_borrow_count(&root) == 2); assert(col_rel_mutation_set_finish(&f.set, true) == 0); assert_open(&root); assert_open(&a); assert_open(&b); assert(col_rel_mutation_lease_validate(&f.leases[0], &a) == EINVAL); + assert(col_rel_storage_alias_borrow_count(&root) == 2); /* A raw source writer has no descriptor/set authority. */ wl_columnar_source_access_writer_t raw = {0}; assert(wl_columnar_source_access_writer_acquire(&root.source_access, @@ -128,6 +134,35 @@ lease_tests(void) assert(wl_columnar_source_access_writer_release(&raw) == 0); } +static void +duplicate_role_publication_tests(void) +{ + col_rel_t relation; + root_init(&relation, 12); + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t roles[] = { + {&relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} + }; + assert(acquire(&f, roles, 2) == 0); + assert(f.set.descriptor_count == 1 && f.set.owner_count == 1); + + /* Model an irreversible same-owner resize publication. Advancing either + * role must advance both leases before failed-transaction cleanup. */ + relation.storage_generation = 2; + relation.storage_owner_generation = 2; + assert(wl_columnar_relation_test_mutation_lease_advance_storage( + &f.leases[0], 1) == 0); + assert(f.leases[0].storage_transitioned + && f.leases[1].storage_transitioned); + assert(col_rel_mutation_lease_validate(&f.leases[0], &relation) == 0 + && col_rel_mutation_lease_validate(&f.leases[1], &relation) == 0); + /* A later operation may fail after publication. Finish must retain the + * publication and validate its refreshed authority before unwinding. */ + assert(col_rel_mutation_set_finish(&f.set, false) == 0); + assert_open(&relation); +} + static void contention_tests(void) { @@ -162,6 +197,14 @@ contention_tests(void) assert(wl_columnar_memory_reserved(&governor) == 64); assert(col_rel_storage_alias_borrow_count(&root) == 1); assert(wl_columnar_source_access_reader_release(&reader) == 0); + assert(wl_columnar_source_access_reader_acquire(&a.source_access, + &reader) == 0); + before = a; + assert(acquire(&f, &role, 1) == EBUSY); + assert(memcmp(&a, &before, sizeof(a)) == 0); + assert(!wl_columnar_source_access_gate_busy(&root.source_access)); + assert(!wl_columnar_source_access_gate_busy(&a.descriptor_access)); + assert(wl_columnar_source_access_reader_release(&reader) == 0); assert(wl_columnar_source_access_reader_acquire(&root.source_access, &reader) == 0); before = a; @@ -269,7 +312,7 @@ invalid_tests(void) assert(col_rel_mutation_set_finish(&f.set, true) == EINVAL); assert(col_rel_mutation_set_lease(&f.set, 0) == NULL); assert(col_rel_mutation_set_acquire(NULL, &role, 1, - f.descriptors, 4, f.owners, 4, f.leases, 4, + f.descriptors, 4, f.owners, 8, f.leases, 4, f.initializations, 4) == EINVAL); assert(col_rel_mutation_set_acquire(&f.set, &role, SIZE_MAX, f.descriptors, SIZE_MAX, f.owners, SIZE_MAX, f.leases, SIZE_MAX, @@ -279,7 +322,7 @@ invalid_tests(void) assert(acquire(&f, &role, 1) == EINVAL); role.relation = &root; assert(col_rel_mutation_set_acquire(&f.set, &role, 1, - f.descriptors, 0, f.owners, 4, f.leases, 4, + f.descriptors, 0, f.owners, 8, f.leases, 4, f.initializations, 4) == EINVAL); f.descriptors[0].writer.identity = 1; assert(acquire(&f, &role, 1) == EINVAL); @@ -295,6 +338,7 @@ int main(void) { lease_tests(); + duplicate_role_publication_tests(); contention_tests(); rollback_tests(); invalid_tests(); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 47b010e8..4c9a939d 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -587,12 +587,14 @@ typedef struct { col_rel_t *owner; size_t descriptor_slot; size_t owner_slot; + size_t self_owner_slot; uint64_t relation_identity; uint64_t relation_generation; uint64_t owner_identity; uint64_t owner_generation; wl_columnar_relation_mutation_role_flags_t role_flags; bool detached; + bool storage_transitioned; } wl_columnar_relation_mutation_lease_t; typedef struct wl_columnar_relation_mutation_set { @@ -608,6 +610,7 @@ typedef struct wl_columnar_relation_mutation_set { size_t initialization_count; size_t descriptors_acquired; size_t owners_acquired; + bool published; } wl_columnar_relation_mutation_set_t; int col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, @@ -625,6 +628,10 @@ int col_rel_mutation_lease_validate( const col_rel_t *expected_relation); int col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, bool commit); +#ifdef WL_TEST_MUTATION_SET_HOOK +int wl_columnar_relation_test_mutation_lease_advance_storage( + wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation); +#endif /* Metadata detach publishes the independent binding before its final atomic * old-owner borrow decrement. The caller must never access that old owner * again through this lease after successful release. */ diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 3c1658fc..fa29a499 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -281,7 +281,7 @@ typedef struct { wl_columnar_relation_mutation_set_t set; wl_columnar_relation_mutation_role_t role; wl_columnar_relation_mutation_descriptor_t descriptor; - wl_columnar_relation_mutation_owner_t owner; + wl_columnar_relation_mutation_owner_t owners[2]; wl_columnar_relation_mutation_lease_t lease; wl_columnar_relation_mutation_initialization_t initialization; } wl_columnar_relation_mutation_single_t; @@ -293,7 +293,7 @@ col_rel_mutation_single_acquire(col_rel_t *relation, single->role.relation = relation; single->role.role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; return col_rel_mutation_set_acquire(&single->set, &single->role, 1, - &single->descriptor, 1, &single->owner, 1, &single->lease, 1, + &single->descriptor, 1, single->owners, 2, &single->lease, 1, &single->initialization, 1); } @@ -313,6 +313,8 @@ static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, bool defer_metadata_retirement, col_rel_cow_deferred_t *deferred_old, wl_columnar_relation_mutation_lease_t *lease); +static int col_rel_mutation_lease_advance_storage( + wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation); /* Temporary compatibility for unmigrated append/radix/batch transactions. * NULL is an explicit legacy path, never a payload mutation lease. These @@ -474,18 +476,24 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, size_t initialization_cap) { int rc = EINVAL; + size_t max_owners; + if (!role_count) + return EINVAL; + if (role_count > SIZE_MAX / 2u) + return EOVERFLOW; + max_owners = role_count * 2u; if (!set || set->identity || set->roles || set->descriptors || set->owners || set->leases || set->initializations || set->descriptor_count || set->owner_count || set->lease_count || set->initialization_count - || set->descriptors_acquired || set->owners_acquired + || set->descriptors_acquired || set->owners_acquired || set->published || !role_count || !roles || !descriptors || !owners || !leases || !initializations || descriptor_cap < role_count - || owner_cap < role_count || lease_cap < role_count + || owner_cap < max_owners || lease_cap < role_count || initialization_cap < role_count) return EINVAL; if (role_count > SIZE_MAX / sizeof(*descriptors) - || role_count > SIZE_MAX / sizeof(*owners) + || max_owners > SIZE_MAX / sizeof(*owners) || role_count > SIZE_MAX / sizeof(*leases) || role_count > SIZE_MAX / sizeof(*initializations)) return EOVERFLOW; @@ -496,7 +504,7 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, (uintptr_t)initializations}; size_t lengths[] = {sizeof(*set), role_count * sizeof(*roles), role_count * sizeof(*descriptors), - role_count * sizeof(*owners), + max_owners * sizeof(*owners), role_count * sizeof(*leases), role_count * sizeof(*initializations)}; for (size_t i = 0; i < sizeof(starts) / sizeof(starts[0]); i++) { @@ -511,20 +519,25 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, if (!roles[i].relation || (roles[i].role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION && roles[i].role_flags != WL_COLUMNAR_RELATION_METADATA_DETACH) - || descriptors[i].relation || owners[i].owner + || descriptors[i].relation || !col_rel_mutation_writer_inert(&descriptors[i].writer) - || !col_rel_mutation_writer_inert(&owners[i].writer) || leases[i].identity || leases[i].set || leases[i].relation || leases[i].owner || leases[i].descriptor_slot - || leases[i].owner_slot || leases[i].relation_identity + || leases[i].owner_slot || leases[i].self_owner_slot + || leases[i].relation_identity || leases[i].relation_generation || leases[i].owner_identity || leases[i].owner_generation || leases[i].role_flags - || leases[i].detached || initializations[i].relation + || leases[i].detached || leases[i].storage_transitioned + || initializations[i].relation || initializations[i].owner_identity || initializations[i].owner_generation || initializations[i].borrows) return EINVAL; } + for (size_t i = 0; i < max_owners; i++) + if (owners[i].owner + || !col_rel_mutation_writer_inert(&owners[i].writer)) + return EINVAL; set->identity = (uintptr_t)set; set->roles = roles; set->descriptors = descriptors; @@ -615,6 +628,11 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, j++; if (j == set->owner_count) owners[set->owner_count++].owner = owner; + j = 0; + while (j < set->owner_count && owners[j].owner != relation) + j++; + if (j == set->owner_count) + owners[set->owner_count++].owner = relation; } } for (size_t i = 1; i < set->owner_count; i++) { @@ -640,6 +658,14 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, for (size_t j = 0; j < set->owner_count; j++) if (owners[j].owner == lease->owner) lease->owner_slot = j; + for (size_t j = 0; j < set->owner_count; j++) + if (owners[j].owner == lease->relation) + lease->self_owner_slot = j; + if (lease->self_owner_slot >= set->owner_count + || owners[lease->self_owner_slot].owner != lease->relation) { + rc = EINVAL; + goto fail; + } if (lease->relation == lease->owner && col_rel_storage_alias_borrow_count(lease->owner)) { rc = EBUSY; @@ -721,12 +747,146 @@ col_rel_mutation_lease_validate( return 0; if (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || lease->owner_slot >= set->owner_count - || set->owners[lease->owner_slot].owner != owner) + || set->owners[lease->owner_slot].owner != owner + || lease->self_owner_slot >= set->owner_count + || set->owners[lease->self_owner_slot].owner != expected_relation + || wl_columnar_source_access_writer_validate( + &set->owners[lease->self_owner_slot].writer, + &expected_relation->source_access)) return EINVAL; return wl_columnar_source_access_writer_validate( &set->owners[lease->owner_slot].writer, &owner->source_access); } +/* A leased publisher may advance this lease only for the exact storage + * generation it just published. Mutation-set acquisition holds both the + * current owner and the prospective self-owner source gates, so alias COW + * can continue under a real token for the newly canonical relation. */ +static int +col_rel_mutation_lease_advance_storage( + wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation) +{ + if (!lease || lease->identity != (uintptr_t)lease || !lease->set + || (lease->detached && lease->owner == lease->relation) + || lease->storage_transitioned + || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || !lease->relation || !lease->owner + || lease->relation_generation != prior_generation + || prior_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u) + return EINVAL; + wl_columnar_relation_mutation_set_t *set = lease->set; + col_rel_t *relation = lease->relation; + if (set->identity != (uintptr_t)set || !set->leases || !set->roles + || !set->descriptors || !set->owners + || set->descriptors_acquired != set->descriptor_count + || set->owners_acquired != set->owner_count + || lease->descriptor_slot >= set->descriptor_count + || lease->owner_slot >= set->owner_count + || lease->self_owner_slot >= set->owner_count + || set->descriptors[lease->descriptor_slot].relation != relation + || set->owners[lease->owner_slot].owner != lease->owner + || set->owners[lease->self_owner_slot].owner != relation) + return EINVAL; + bool member = false; + col_rel_t *prior_owner = lease->owner; + uint64_t prior_owner_identity = lease->owner_identity; + uint64_t prior_owner_generation = lease->owner_generation; + for (size_t i = 0; i < set->lease_count; i++) + if (&set->leases[i] == lease + && set->roles[i].relation == relation + && set->roles[i].role_flags == lease->role_flags) { + member = true; + } + if (!member || !relation + || relation->relation_identity != lease->relation_identity + || wl_columnar_source_access_writer_validate( + &set->descriptors[lease->descriptor_slot].writer, + &relation->descriptor_access) + || wl_columnar_source_access_writer_validate( + &set->owners[lease->owner_slot].writer, + &lease->owner->source_access) + || wl_columnar_source_access_writer_validate( + &set->owners[lease->self_owner_slot].writer, + &relation->source_access)) + return EINVAL; + + uint64_t next_generation = prior_generation + 1u; + if (relation->storage_generation != next_generation + || relation->storage_owner != relation + || relation->storage_owner_identity != relation->relation_identity + || relation->storage_owner_generation != next_generation + || !wl_columnar_relation_generation_valid(next_generation)) + return EINVAL; + + if (prior_owner == relation) { + /* Same-owner reserve/resize: the identity and matching token stay + * fixed while storage and owner generation advance together. */ + if (lease->detached || lease->owner_slot != lease->self_owner_slot + || lease->owner_identity != relation->relation_identity + || lease->owner_generation != prior_generation + || relation->storage_owner_identity != lease->owner_identity) + return EINVAL; + } else { + /* Alias COW: only a just-released binding may move to the exact + * pre-acquired self-owner token; arbitrary stale leases cannot. */ + if (!lease->detached || prior_owner == NULL + || prior_owner_identity != prior_owner->relation_identity + || prior_owner_generation != prior_owner->storage_generation) + return EINVAL; + } + + /* A mutation set may hold the same descriptor in multiple payload + * roles. Validate every matching lease against the pre-publication + * binding before advancing any of them, then advance them together so + * commit validation observes one consistent published generation. */ + for (size_t i = 0; i < set->lease_count; i++) { + wl_columnar_relation_mutation_lease_t *same = &set->leases[i]; + if (set->roles[i].relation != relation + || set->roles[i].role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + continue; + if (same->set != set || same->identity != (uintptr_t)same + || same->relation != relation || same->storage_transitioned + || same->relation_identity != lease->relation_identity + || same->relation_generation != prior_generation + || same->owner != prior_owner + || same->owner_identity != prior_owner_identity + || same->owner_generation != prior_owner_generation + || same->descriptor_slot != lease->descriptor_slot + || same->owner_slot != lease->owner_slot + || same->self_owner_slot != lease->self_owner_slot + || same->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + return EINVAL; + } + for (size_t i = 0; i < set->lease_count; i++) { + wl_columnar_relation_mutation_lease_t *same = &set->leases[i]; + if (set->roles[i].relation != relation + || set->roles[i].role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + continue; + if (prior_owner != relation) { + same->owner = relation; + same->owner_slot = same->self_owner_slot; + same->owner_identity = relation->relation_identity; + same->detached = false; + } + same->relation_generation = next_generation; + same->owner_generation = next_generation; + same->storage_transitioned = true; + } + set->published = true; + return 0; +} + +#ifdef WL_TEST_MUTATION_SET_HOOK +int +wl_columnar_relation_test_mutation_lease_advance_storage( + wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation) +{ + return col_rel_mutation_lease_advance_storage(lease, prior_generation); +} +#endif + int col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, bool commit) @@ -751,7 +911,14 @@ col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, if (wl_columnar_source_access_writer_validate(&set->owners[i].writer, &set->owners[i].owner->source_access)) return EINVAL; - col_rel_mutation_set_unwind(set, commit); + if (commit || set->published) + for (size_t i = 0; i < set->lease_count; i++) + if (set->roles[i].role_flags + == WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + && col_rel_mutation_lease_validate(&set->leases[i], + set->roles[i].relation) != 0) + abort(); + col_rel_mutation_set_unwind(set, commit || set->published); return 0; } @@ -1988,10 +2155,14 @@ col_rel_grow_owned_transition_publish_impl(col_rel_t *r, uint32_t new_cap, bool *old_shared_flags; bool old_arena; uint64_t ledger_before; + uint64_t prior_generation; - if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + if (lease && (lease->storage_transitioned + || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r))) return EINVAL; + prior_generation = lease ? lease->relation_generation + : (r ? r->storage_generation : 0); wl_columnar_memory_reservation_init(&previous); if (!r || (r->ncols && !r->columns) || new_cap < r->nrows) @@ -2060,6 +2231,9 @@ col_rel_grow_owned_transition_publish_impl(col_rel_t *r, uint32_t new_cap, if (col_rel_storage_alias_release_locked(r, lease)) abort(); wl_columnar_relation_touch_storage(r); + if (col_rel_mutation_lease_advance_storage(lease, + prior_generation) != 0) + abort(); } } else { if (!defer_alias_release) @@ -2125,10 +2299,14 @@ col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, bool *old_shared; uint32_t capacity; uint64_t ledger_before; + uint64_t prior_generation; - if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + if (lease && (lease->storage_transitioned + || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r))) return EINVAL; + prior_generation = lease ? lease->relation_generation + : (r ? r->storage_generation : 0); wl_columnar_memory_reservation_init(&previous); if (!r || !r->col_shared) @@ -2192,6 +2370,9 @@ col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, if (col_rel_storage_alias_release_locked(r, lease)) abort(); wl_columnar_relation_touch_storage(r); + if (col_rel_mutation_lease_advance_storage(lease, + prior_generation) != 0) + abort(); } } else { if (!defer_alias_release) @@ -4006,6 +4187,7 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, wl_columnar_memory_reservation_t row_admission; bool row_credit_held = false; int rc; + uint64_t prior_storage_generation = 0; wl_columnar_memory_reservation_init(&row_admission); @@ -4032,6 +4214,8 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, if (lease && (lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r))) return EINVAL; + prior_storage_generation = lease ? lease->relation_generation + : r->storage_generation; if (!col_rel_timestamp_shape_valid(r) || r->nrows > r->capacity) return EINVAL; bool replaces_storage = r->col_shared || r->nrows >= r->capacity @@ -4126,6 +4310,10 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, bool timestamp_short = r->timestamps && r->nrows >= r->timestamp_capacity; if (needs_resize || timestamp_short) { + if (lease && lease->storage_transitioned) { + rc = EINVAL; + goto release_writer; + } /* Canonical-owner storage replacement frees the old buffers. Live * shared views still point at those buffers, so refuse growth until * the aliases are released. A shared-view destination is allowed @@ -4206,6 +4394,9 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, col_rel_retire_payload_credit(r); col_rel_ledger_reconcile(r, ledger_before); wl_columnar_relation_touch_storage(r); + if (lease && col_rel_mutation_lease_advance_storage(lease, + prior_storage_generation) != 0) + abort(); } } /* A shared view can still have spare capacity. Privatize it before the @@ -4276,7 +4467,8 @@ col_rel_append_row(col_rel_t *r, const int64_t *row) return rc; /* Outer attempt provenance starts only after both admissions succeed. */ r->memory_budget_denial_pending = false; - rc = col_rel_append_row_impl(r, row, &single.owner.writer, true, + rc = col_rel_append_row_impl(r, row, + &single.owners[single.lease.owner_slot].writer, true, &single.lease); int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); return rc ? rc : finish_rc; @@ -4535,6 +4727,7 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, { col_rel_t *owner = NULL; int rc; + uint64_t prior_storage_generation = 0; if (!r) return EINVAL; if (lease) { @@ -4542,6 +4735,7 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r)) return EINVAL; + prior_storage_generation = lease->relation_generation; } else { if (!writer || !out_alias_release_pending) return EINVAL; @@ -4563,6 +4757,11 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, * first locked append. Perform that fallible transition up front. */ bool timestamp_short = r->timestamps && required > r->timestamp_capacity; + bool will_replace_storage = (required <= r->capacity && timestamp_short) + || (r->col_shared && required <= r->capacity) + || required > r->capacity; + if (lease && lease->storage_transitioned && will_replace_storage) + return EINVAL; if (required <= r->capacity && timestamp_short) { if (r->storage_owner == r && col_rel_storage_alias_borrow_count(r) > 0) @@ -4647,6 +4846,9 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, col_rel_retire_payload_credit(r); col_rel_ledger_reconcile(r, ledger_before); wl_columnar_relation_touch_storage(r); + if (lease && col_rel_mutation_lease_advance_storage(lease, + prior_storage_generation) != 0) + abort(); return 0; } From 6c857d617b8002a4dc8af3d052b312e4f57716fb Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 00:06:00 +0900 Subject: [PATCH 02/15] Add terminal relation storage sequence --- tests/test_relation_mutation_set.c | 229 ++++++++++++++++++++ wirelog/columnar/internal.h | 56 +++++ wirelog/columnar/relation.c | 324 ++++++++++++++++++++++++++++- 3 files changed, 608 insertions(+), 1 deletion(-) diff --git a/tests/test_relation_mutation_set.c b/tests/test_relation_mutation_set.c index 5a37994b..e198e8d3 100644 --- a/tests/test_relation_mutation_set.c +++ b/tests/test_relation_mutation_set.c @@ -163,6 +163,234 @@ duplicate_role_publication_tests(void) assert_open(&relation); } +static void +terminal_sequence_tests(void) +{ + col_rel_t relation; + root_init(&relation, 20); + int64_t old_value = 1, new_value = 2; + int64_t *old_columns[] = { &old_value }; + int64_t *new_columns[] = { &new_value }; + relation.columns = old_columns; + relation.capacity = 3; + relation.merge_columns = new_columns; + relation.merge_buf_cap = 4; + relation.timestamps = malloc(sizeof(*relation.timestamps) * 3); + assert(relation.timestamps != NULL); + relation.timestamp_capacity = 3; + + fixture_t f = {0}; + wl_columnar_relation_mutation_role_t roles[] = { + {&relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} + }; + assert(acquire(&f, roles, 2) == 0); + wl_columnar_relation_terminal_sequence_t sequence = {0}; + assert(wl_columnar_relation_terminal_sequence_begin(&sequence, + &f.leases[0], 0) == EINVAL); + uint64_t original_generation = relation.storage_generation; + f.leases[1].storage_transitioned = true; + assert(wl_columnar_relation_terminal_sequence_begin(&sequence, + &f.leases[0], WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + | WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) == EINVAL); + assert(f.set.terminal_sequence == NULL + && f.set.terminal_expected_events == 0 + && f.set.terminal_observed_events == 0); + f.leases[1].storage_transitioned = false; + relation.storage_generation++; + relation.storage_owner_generation++; + assert(wl_columnar_relation_test_mutation_lease_advance_storage( + &f.leases[0], original_generation) == 0); + original_generation = relation.storage_generation; + assert(f.leases[0].storage_transitioned + && f.leases[1].storage_transitioned + && f.leases[0].relation_generation == original_generation + && f.leases[1].relation_generation == original_generation); + assert(wl_columnar_relation_terminal_sequence_begin(&sequence, + &f.leases[0], WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + | WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) == 0); + assert(wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + uint32_t saved_expected = sequence.expected_events; + sequence.expected_events = WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + sequence.expected_events = saved_expected; + uint32_t saved_role = roles[1].role_flags; + col_rel_t unrelated; + roles[1].relation = &unrelated; + roles[1].role_flags = WL_COLUMNAR_RELATION_METADATA_DETACH; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + roles[1].relation = &relation; + roles[1].role_flags = saved_role; + saved_role = f.leases[1].role_flags; + f.leases[1].role_flags = WL_COLUMNAR_RELATION_METADATA_DETACH; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + f.leases[1].role_flags = saved_role; + bool saved_detached = f.leases[1].detached; + f.leases[1].detached = !saved_detached; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + f.leases[1].detached = saved_detached; + bool saved_transitioned = f.leases[1].storage_transitioned; + f.leases[1].storage_transitioned = !saved_transitioned; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + f.leases[1].storage_transitioned = saved_transitioned; + assert(wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + + int64_t **columns = relation.columns; + uint32_t capacity = relation.capacity; + relation.columns = relation.merge_columns; + relation.capacity = relation.merge_buf_cap; + relation.merge_columns = columns; + relation.merge_buf_cap = capacity; + wl_columnar_relation_terminal_sequence_publish_grid_swap(&sequence); + columns = relation.columns; + capacity = relation.capacity; + relation.columns = relation.merge_columns; + relation.capacity = relation.merge_buf_cap; + relation.merge_columns = columns; + relation.merge_buf_cap = capacity; + assert(!wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + columns = relation.columns; + capacity = relation.capacity; + relation.columns = relation.merge_columns; + relation.capacity = relation.merge_buf_cap; + relation.merge_columns = columns; + relation.merge_buf_cap = capacity; + assert(wl_columnar_relation_test_terminal_sequence_validate(&sequence)); + free(relation.timestamps); + relation.timestamps = NULL; + relation.timestamp_capacity = 0; + wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + &sequence); + assert(wl_columnar_relation_terminal_sequence_finish(&sequence) == 0); + assert(relation.storage_generation == original_generation + 2 + && relation.storage_owner_generation == original_generation + 2 + && f.leases[0].relation_generation == original_generation + 2 + && f.leases[1].relation_generation == original_generation + 2 + && f.leases[0].owner_generation == original_generation + 2 + && f.leases[1].owner_generation == original_generation + 2); + wl_columnar_relation_terminal_sequence_t replay = {0}; + assert(wl_columnar_relation_terminal_sequence_begin(&replay, + &f.leases[0], WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) == EINVAL); + assert(relation.storage_generation == original_generation + 2); + assert(col_rel_mutation_set_finish(&f.set, true) == 0); + assert_open(&relation); + + col_rel_t owner, alias; + root_init(&owner, 21); + alias_init(&alias, &owner, 22); + alias.capacity = 1; + alias.timestamps = malloc(sizeof(*alias.timestamps)); + assert(alias.timestamps != NULL); + alias.timestamp_capacity = 1; + fixture_t af = {0}; + wl_columnar_relation_mutation_role_t alias_roles[] = { + {&alias, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION}, + {&alias, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION} + }; + assert(acquire(&af, alias_roles, 2) == 0); + wl_columnar_relation_terminal_sequence_t alias_sequence = {0}; + assert(wl_columnar_relation_terminal_sequence_begin(&alias_sequence, + &af.leases[0], + WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) == 0); + free(alias.timestamps); + alias.timestamps = NULL; + alias.timestamp_capacity = 0; + wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + &alias_sequence); + assert(wl_columnar_relation_terminal_sequence_finish(&alias_sequence) == 0); + assert(alias.storage_owner == &owner + && alias.storage_owner_identity == owner.relation_identity + && alias.storage_owner_generation == owner.storage_generation + && alias.storage_generation == af.leases[0].relation_generation + && af.leases[0].owner_generation == owner.storage_generation + && af.leases[1].owner_generation == owner.storage_generation); + assert(col_rel_mutation_set_finish(&af.set, true) == 0); + assert_open(&owner); + assert_open(&alias); + + root_init(&relation, 24); + relation.capacity = 2; + relation.merge_buf_cap = 3; + fixture_t nf = {0}; + wl_columnar_relation_mutation_role_t nullary_role = { + &relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + assert(acquire(&nf, &nullary_role, 1) == 0); + wl_columnar_relation_terminal_sequence_t nullary_sequence = {0}; + assert(wl_columnar_relation_terminal_sequence_begin(&nullary_sequence, + &nf.leases[0], WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) == 0); + uint32_t old_capacity = relation.capacity; + relation.capacity = relation.merge_buf_cap; + relation.merge_buf_cap = old_capacity; + wl_columnar_relation_terminal_sequence_publish_grid_swap( + &nullary_sequence); + assert(wl_columnar_relation_terminal_sequence_finish( + &nullary_sequence) == 0); + assert(relation.storage_generation == 2 + && relation.storage_owner_generation == 2); + assert(col_rel_mutation_set_finish(&nf.set, true) == 0); + assert_open(&relation); + + col_rel_t nullary_owner, nullary_alias; + root_init(&nullary_owner, 25); + alias_init(&nullary_alias, &nullary_owner, 26); + nullary_alias.capacity = 2; + nullary_alias.merge_buf_cap = 3; + fixture_t naf = {0}; + wl_columnar_relation_mutation_role_t nullary_alias_role = { + &nullary_alias, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + assert(acquire(&naf, &nullary_alias_role, 1) == 0); + wl_columnar_relation_terminal_sequence_t nullary_alias_sequence = {0}; + assert(wl_columnar_relation_terminal_sequence_begin( + &nullary_alias_sequence, &naf.leases[0], + WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) == 0); + old_capacity = nullary_alias.capacity; + nullary_alias.capacity = nullary_alias.merge_buf_cap; + nullary_alias.merge_buf_cap = old_capacity; + wl_columnar_relation_terminal_sequence_publish_grid_swap( + &nullary_alias_sequence); + assert(wl_columnar_relation_terminal_sequence_finish( + &nullary_alias_sequence) == 0); + assert(nullary_alias.storage_owner == &nullary_owner + && nullary_alias.storage_owner_generation + == nullary_owner.storage_generation + && nullary_alias.storage_generation == 2 + && naf.leases[0].relation_generation == 2 + && naf.leases[0].owner_generation == 1); + assert(col_rel_mutation_set_finish(&naf.set, true) == 0); + assert_open(&nullary_owner); + assert_open(&nullary_alias); + + fixture_t denied = {0}; + root_init(&relation, 23); + relation.columns = old_columns; + relation.capacity = 3; + relation.merge_columns = new_columns; + relation.merge_buf_cap = 4; + relation.timestamps = malloc(sizeof(*relation.timestamps) * 3); + assert(relation.timestamps != NULL); + relation.timestamp_capacity = 3; + relation.storage_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 2u; + relation.storage_owner_generation = relation.storage_generation; + wl_columnar_relation_mutation_role_t denied_role = { + &relation, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + assert(acquire(&denied, &denied_role, 1) == 0); + wl_columnar_relation_terminal_sequence_t denied_sequence = {0}; + assert(wl_columnar_relation_terminal_sequence_begin(&denied_sequence, + &denied.leases[0], WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + | WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) == EOVERFLOW); + assert(relation.storage_generation + == WL_COLUMNAR_REL_GENERATION_INVALID - 2u + && relation.timestamps != NULL); + free(relation.timestamps); + relation.timestamps = NULL; + relation.timestamp_capacity = 0; + assert(col_rel_mutation_set_finish(&denied.set, false) == 0); + assert_open(&relation); +} + static void contention_tests(void) { @@ -339,6 +567,7 @@ main(void) { lease_tests(); duplicate_role_publication_tests(); + terminal_sequence_tests(); contention_tests(); rollback_tests(); invalid_tests(); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 4c9a939d..bfb886d2 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -597,6 +597,7 @@ typedef struct { bool storage_transitioned; } wl_columnar_relation_mutation_lease_t; +struct wl_columnar_relation_terminal_sequence; typedef struct wl_columnar_relation_mutation_set { uintptr_t identity; const wl_columnar_relation_mutation_role_t *roles; @@ -610,9 +611,51 @@ typedef struct wl_columnar_relation_mutation_set { size_t initialization_count; size_t descriptors_acquired; size_t owners_acquired; + struct wl_columnar_relation_terminal_sequence *terminal_sequence; + uint32_t terminal_expected_events; + uint32_t terminal_observed_events; + bool terminal_sequence_consumed; bool published; } wl_columnar_relation_mutation_set_t; +typedef enum { + WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP = 1u << 0, + WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE = 1u << 1 +} wl_columnar_relation_terminal_event_t; + +/* A zero-initialized, transaction-local record for storage publications after + * incremental consolidation has completed every fallible preparation step. + * Its event mask is consumed by relation-specific publication helpers. */ +typedef struct wl_columnar_relation_terminal_sequence { + uintptr_t identity; + wl_columnar_relation_mutation_set_t *set; + wl_columnar_relation_mutation_lease_t *lease; + col_rel_t *relation; + col_rel_t *owner; + int64_t **columns_before; + int64_t **merge_columns_before; + col_delta_timestamp_t *timestamps_before; + size_t descriptor_slot; + size_t owner_slot; + size_t self_owner_slot; + uintptr_t descriptor_writer_identity; + uintptr_t owner_writer_identity; + uintptr_t self_writer_identity; + uint64_t relation_identity; + uint64_t owner_identity; + uint64_t relation_generation; + uint64_t owner_generation; + uint32_t capacity_before; + uint32_t merge_capacity_before; + uint32_t timestamp_capacity_before; + uint32_t expected_events; + uint32_t observed_events; + bool detached_before; + bool storage_transitioned_before; + bool armed; + bool consumed; +} wl_columnar_relation_terminal_sequence_t; + int col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, const wl_columnar_relation_mutation_role_t *roles, size_t role_count, wl_columnar_relation_mutation_descriptor_t *descriptors, @@ -628,6 +671,19 @@ int col_rel_mutation_lease_validate( const col_rel_t *expected_relation); int col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, bool commit); +int wl_columnar_relation_terminal_sequence_begin( + wl_columnar_relation_terminal_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease, uint32_t expected_events); +void wl_columnar_relation_terminal_sequence_publish_grid_swap( + wl_columnar_relation_terminal_sequence_t *sequence); +void wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + wl_columnar_relation_terminal_sequence_t *sequence); +int wl_columnar_relation_terminal_sequence_finish( + wl_columnar_relation_terminal_sequence_t *sequence); +#ifdef WL_TEST_MUTATION_SET_HOOK +bool wl_columnar_relation_test_terminal_sequence_validate( + const wl_columnar_relation_terminal_sequence_t *sequence); +#endif #ifdef WL_TEST_MUTATION_SET_HOOK int wl_columnar_relation_test_mutation_lease_advance_storage( wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation); diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index fa29a499..7a282776 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -315,6 +315,7 @@ static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, wl_columnar_relation_mutation_lease_t *lease); static int col_rel_mutation_lease_advance_storage( wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation); +static bool col_rel_timestamp_shape_valid(const col_rel_t *r); /* Temporary compatibility for unmigrated append/radix/batch transactions. * NULL is an explicit legacy path, never a payload mutation lease. These @@ -486,7 +487,10 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, set->owners || set->leases || set->initializations || set->descriptor_count || set->owner_count || set->lease_count || set->initialization_count - || set->descriptors_acquired || set->owners_acquired || set->published + || set->descriptors_acquired || set->owners_acquired + || set->terminal_sequence || set->terminal_expected_events + || set->terminal_observed_events || set->terminal_sequence_consumed + || set->published || !role_count || !roles || !descriptors || !owners || !leases || !initializations || descriptor_cap < role_count || owner_cap < max_owners || lease_cap < role_count @@ -895,6 +899,8 @@ col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, || set->descriptors_acquired != set->descriptor_count || set->owners_acquired != set->owner_count) return EINVAL; + if (set->terminal_sequence) + abort(); if (!set->leases || !set->roles || !set->descriptors || !set->owners || !set->initializations || !set->lease_count || !set->descriptor_count) return EINVAL; @@ -922,6 +928,322 @@ col_rel_mutation_set_finish(wl_columnar_relation_mutation_set_t *set, return 0; } +static uint32_t +wl_columnar_relation_terminal_event_count(uint32_t events) +{ + uint32_t count = 0; + for (uint32_t bits = events; bits; bits &= bits - 1u) + count++; + return count; +} + +static bool +wl_columnar_relation_terminal_tokens_valid( + const wl_columnar_relation_terminal_sequence_t *sequence, + bool validate_postconditions) +{ + if (!sequence || sequence->identity != (uintptr_t)sequence + || !sequence->armed || sequence->consumed || !sequence->set + || !sequence->lease || !sequence->relation || !sequence->owner) + return false; + const uint32_t allowed_events = + WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + | WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE; + wl_columnar_relation_mutation_set_t *set = sequence->set; + const wl_columnar_relation_mutation_lease_t *lease = sequence->lease; + col_rel_t *relation = sequence->relation; + if (set->identity != (uintptr_t)set || set->terminal_sequence != sequence + || set->terminal_sequence_consumed + || !sequence->expected_events + || (sequence->expected_events & ~allowed_events) + || set->terminal_expected_events != sequence->expected_events + || set->terminal_observed_events != sequence->observed_events + || (sequence->observed_events & ~sequence->expected_events) + || lease->set != set + || lease->identity != (uintptr_t)lease || lease->relation != relation + || lease->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || !set->roles || !set->leases || !set->descriptors || !set->owners + || lease->descriptor_slot != sequence->descriptor_slot + || lease->self_owner_slot != sequence->self_owner_slot + || set->descriptors_acquired != set->descriptor_count + || set->owners_acquired != set->owner_count + || sequence->descriptor_slot >= set->descriptor_count + || sequence->owner_slot >= set->owner_count + || sequence->self_owner_slot >= set->owner_count + || set->descriptors[sequence->descriptor_slot].relation != relation + || set->owners[sequence->owner_slot].owner != sequence->owner + || set->owners[sequence->self_owner_slot].owner != relation + || set->descriptors[sequence->descriptor_slot].writer.identity + != sequence->descriptor_writer_identity + || set->owners[sequence->owner_slot].writer.identity + != sequence->owner_writer_identity + || set->owners[sequence->self_owner_slot].writer.identity + != sequence->self_writer_identity + || wl_columnar_source_access_writer_validate( + &set->descriptors[sequence->descriptor_slot].writer, + &relation->descriptor_access) + || wl_columnar_source_access_writer_validate( + &set->owners[sequence->owner_slot].writer, + &sequence->owner->source_access) + || wl_columnar_source_access_writer_validate( + &set->owners[sequence->self_owner_slot].writer, + &relation->source_access)) + return false; + uint32_t observed_steps = + wl_columnar_relation_terminal_event_count(sequence->observed_events); + uint64_t observed_generation = sequence->relation_generation + + observed_steps; + bool lease_member = false; + for (size_t i = 0; i < set->lease_count; i++) { + const wl_columnar_relation_mutation_lease_t *same = &set->leases[i]; + const wl_columnar_relation_mutation_role_t *role = &set->roles[i]; + if (same == lease) { + if (role->relation != relation + || role->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || same->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + return false; + lease_member = true; + } + if (role->relation != relation && same->relation != relation) + continue; + if (role->relation != relation || same->relation != relation + || same->set != set || same->identity != (uintptr_t)same + || same->relation != relation + || same->role_flags != role->role_flags) + return false; + if (role->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + continue; + if (same->owner != sequence->owner + || same->relation_identity != sequence->relation_identity + || same->owner_identity != sequence->owner_identity + || same->relation_generation != sequence->relation_generation + || same->owner_generation != sequence->owner_generation + || same->descriptor_slot != sequence->descriptor_slot + || same->owner_slot != sequence->owner_slot + || same->self_owner_slot != sequence->self_owner_slot + || same->detached != sequence->detached_before + || same->storage_transitioned + != sequence->storage_transitioned_before) + return false; + } + bool grid_swapped = (sequence->observed_events + & WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) != 0; + bool timestamps_retired = (sequence->observed_events + & WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) != 0; + return lease_member + && relation->relation_identity == sequence->relation_identity + && relation->storage_owner == sequence->owner + && relation->storage_owner_identity == sequence->owner_identity + && (!validate_postconditions + || relation->storage_generation == observed_generation) + && relation->storage_owner_generation + == (sequence->owner == relation ? observed_generation + : sequence->owner_generation) + && sequence->owner->relation_identity == sequence->owner_identity + && sequence->owner->storage_generation + == (sequence->owner == relation ? observed_generation + : sequence->owner_generation) + && (!validate_postconditions || (grid_swapped + ? (relation->columns == sequence->merge_columns_before + && relation->capacity == sequence->merge_capacity_before + && relation->merge_columns == sequence->columns_before + && relation->merge_buf_cap == sequence->capacity_before) + : (relation->columns == sequence->columns_before + && relation->capacity == sequence->capacity_before + && relation->merge_columns + == sequence->merge_columns_before + && relation->merge_buf_cap + == sequence->merge_capacity_before))) + && (!validate_postconditions || (timestamps_retired + ? (relation->timestamps == NULL + && relation->timestamp_capacity == 0) + : (relation->timestamps == sequence->timestamps_before + && relation->timestamp_capacity + == sequence->timestamp_capacity_before))); +} + +int +wl_columnar_relation_terminal_sequence_begin( + wl_columnar_relation_terminal_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease, uint32_t expected_events) +{ + const uint32_t allowed = WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + | WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE; + if (!sequence || sequence->identity || sequence->armed + || sequence->consumed || !lease || (expected_events & ~allowed) + || !expected_events + || col_rel_mutation_lease_validate(lease, lease->relation) != 0) + return EINVAL; + col_rel_t *relation = lease->relation; + bool retires_timestamps = (expected_events + & WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) != 0; + if (retires_timestamps != (relation->timestamps != NULL) + || (retires_timestamps && (relation->timestamp_capacity == 0 + || !col_rel_timestamp_shape_valid(relation))) + || ((expected_events & WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) + && ((relation->ncols && (!relation->columns + || !relation->merge_columns)) + || (lease->owner != relation && relation->ncols != 0) + || relation->merge_buf_cap == 0))) + return EINVAL; + uint32_t steps = wl_columnar_relation_terminal_event_count( + expected_events); + if (steps == 0 + || relation->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - (uint64_t)steps) + return EOVERFLOW; + if (lease->set->terminal_sequence + || lease->set->terminal_sequence_consumed) + return EINVAL; + if (lease->set->terminal_expected_events + || lease->set->terminal_observed_events) + return EINVAL; + if (lease->owner_slot >= lease->set->owner_count + || lease->self_owner_slot >= lease->set->owner_count) + return EINVAL; + memset(sequence, 0, sizeof(*sequence)); + sequence->identity = (uintptr_t)sequence; + sequence->set = lease->set; + sequence->lease = lease; + sequence->relation = relation; + sequence->owner = lease->owner; + sequence->descriptor_slot = lease->descriptor_slot; + sequence->owner_slot = lease->owner_slot; + sequence->self_owner_slot = lease->self_owner_slot; + sequence->descriptor_writer_identity = + lease->set->descriptors[lease->descriptor_slot].writer.identity; + sequence->owner_writer_identity = + lease->set->owners[lease->owner_slot].writer.identity; + sequence->self_writer_identity = + lease->set->owners[lease->self_owner_slot].writer.identity; + sequence->relation_identity = lease->relation_identity; + sequence->owner_identity = lease->owner_identity; + sequence->relation_generation = lease->relation_generation; + sequence->owner_generation = lease->owner_generation; + sequence->columns_before = relation->columns; + sequence->merge_columns_before = relation->merge_columns; + sequence->timestamps_before = relation->timestamps; + sequence->capacity_before = relation->capacity; + sequence->merge_capacity_before = relation->merge_buf_cap; + sequence->timestamp_capacity_before = relation->timestamp_capacity; + sequence->expected_events = expected_events; + sequence->detached_before = lease->detached; + sequence->storage_transitioned_before = lease->storage_transitioned; + sequence->armed = true; + lease->set->terminal_sequence = sequence; + lease->set->terminal_expected_events = expected_events; + lease->set->terminal_observed_events = 0; + if (!wl_columnar_relation_terminal_tokens_valid(sequence, true)) { + lease->set->terminal_sequence = NULL; + lease->set->terminal_expected_events = 0; + lease->set->terminal_observed_events = 0; + memset(sequence, 0, sizeof(*sequence)); + return EINVAL; + } + lease->set->published = true; + return 0; +} + +static void +wl_columnar_relation_terminal_publish_event( + wl_columnar_relation_terminal_sequence_t *sequence, uint32_t event) +{ + if (!wl_columnar_relation_terminal_tokens_valid(sequence, false) + || !sequence->set->published + || (event != WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP + && event != WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE) + || !(sequence->expected_events & event) + || (sequence->observed_events & event)) + abort(); + col_rel_t *relation = sequence->relation; + if (event == WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) { + if (relation->columns != sequence->merge_columns_before + || relation->capacity != sequence->merge_capacity_before + || relation->merge_columns != sequence->columns_before + || relation->merge_buf_cap != sequence->capacity_before) + abort(); + } else if (relation->timestamps != NULL + || relation->timestamp_capacity != 0 + || sequence->timestamps_before == NULL + || sequence->timestamp_capacity_before == 0) { + abort(); + } + wl_columnar_relation_touch_storage(relation); + sequence->observed_events |= event; + sequence->set->terminal_observed_events |= event; +} + +#ifdef WL_TEST_MUTATION_SET_HOOK +bool +wl_columnar_relation_test_terminal_sequence_validate( + const wl_columnar_relation_terminal_sequence_t *sequence) +{ + return wl_columnar_relation_terminal_tokens_valid(sequence, true); +} +#endif + +void +wl_columnar_relation_terminal_sequence_publish_grid_swap( + wl_columnar_relation_terminal_sequence_t *sequence) +{ + wl_columnar_relation_terminal_publish_event(sequence, + WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP); +} + +void +wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + wl_columnar_relation_terminal_sequence_t *sequence) +{ + wl_columnar_relation_terminal_publish_event(sequence, + WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE); +} + +int +wl_columnar_relation_terminal_sequence_finish( + wl_columnar_relation_terminal_sequence_t *sequence) +{ + if (!wl_columnar_relation_terminal_tokens_valid(sequence, true) + || sequence->observed_events != sequence->expected_events) + return EINVAL; + wl_columnar_relation_mutation_set_t *set = sequence->set; + col_rel_t *relation = sequence->relation; + uint32_t steps = wl_columnar_relation_terminal_event_count( + sequence->expected_events); + uint64_t final_generation = sequence->relation_generation + steps; + if (relation->storage_generation != final_generation + || (sequence->owner == relation + && (relation->storage_owner != relation + || relation->storage_owner_identity != relation->relation_identity + || relation->storage_owner_generation != final_generation)) + || (sequence->owner != relation + && (relation->storage_owner != sequence->owner + || relation->storage_owner_identity != sequence->owner_identity + || relation->storage_owner_generation != sequence->owner_generation + || ((sequence->expected_events + & WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP) + && relation->ncols != 0)))) + return EINVAL; + for (size_t i = 0; i < set->lease_count; i++) { + wl_columnar_relation_mutation_lease_t *same = &set->leases[i]; + if (set->roles[i].relation != relation + || set->roles[i].role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) + continue; + same->relation_generation = final_generation; + if (sequence->owner == relation) + same->owner_generation = final_generation; + } + sequence->consumed = true; + sequence->armed = false; + set->terminal_sequence = NULL; + set->terminal_expected_events = 0; + set->terminal_observed_events = 0; + set->terminal_sequence_consumed = true; + set->published = true; + return 0; +} + int col_rel_storage_alias_release_locked(col_rel_t *alias, wl_columnar_relation_mutation_lease_t *lease) From 6551104773dbac96ac2646e3ba5285ccee977f0c Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 00:19:53 +0900 Subject: [PATCH 03/15] Migrate incremental consolidation to mutation leases --- tests/test_consolidate_incremental_delta.c | 4 +- wirelog/columnar/internal.h | 13 ++ wirelog/columnar/merge.c | 242 ++++++++++----------- wirelog/columnar/relation.c | 79 ++++++- 4 files changed, 201 insertions(+), 137 deletions(-) diff --git a/tests/test_consolidate_incremental_delta.c b/tests/test_consolidate_incremental_delta.c index e9e842df..f3816098 100644 --- a/tests/test_consolidate_incremental_delta.c +++ b/tests/test_consolidate_incremental_delta.c @@ -84,11 +84,11 @@ consolidation_pause_hook(col_rel_t *relation, return; state->hook_state_ok = stage == state->expected_stage && relation == state->expected_relation - && relation->storage_owner == state->expected_owner + && relation->storage_owner == relation && relation->col_shared == NULL && relation->columns[0] != state->old_columns && relation->nrows == state->expected_nrows - && state->expected_owner->storage_alias_borrows > 0; + && state->expected_owner->storage_owner == state->expected_owner; for (uint32_t i = 0; state->hook_state_ok && i < relation->nrows; i++) state->hook_state_ok = relation->columns[0][i] == state->expected_values[i]; diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index bfb886d2..593183f7 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2732,6 +2732,14 @@ col_rel_retained_live_bytes(const col_rel_t *r, uint64_t *out); /* Replace a persistent merge grid under the aggregate payload reservation. * On admission/allocation failure the relation and old grid are unchanged. */ int col_rel_reserve_merge_grid(col_rel_t *r, uint32_t capacity); +int col_rel_reserve_merge_grid_with_lease(col_rel_t *r, uint32_t capacity, + wl_columnar_relation_mutation_lease_t *lease); +int col_rel_cow_unshare_with_lease(col_rel_t *r, + wl_columnar_relation_mutation_lease_t *lease); +int col_rel_append_row_with_lease(col_rel_t *r, const int64_t *row, + wl_columnar_relation_mutation_lease_t *lease); +int col_rel_reserve_rows_with_lease(col_rel_t *r, uint32_t additional, + wl_columnar_relation_mutation_lease_t *lease); /* Retire aggregate credit after a payload allocation has been freed. */ void col_rel_retire_payload_credit(col_rel_t *r); /* Admit and grow @r to at least @new_cap rows as one transaction; with @@ -3262,6 +3270,11 @@ WL_MUST_CHECK int wl_columnar_relation_radix_sort_with_lease(col_rel_t *rel, uint32_t start_row, uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, wl_columnar_relation_mutation_lease_t *lease); +WL_MUST_CHECK int +wl_columnar_relation_radix_sort_consolidation_with_lease(col_rel_t *rel, + uint32_t start_row, uint32_t nrows, + const wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease); /** Stable LSD radix sort of a row-major int64_t buffer by a single key * column. Used by arrangement.c (sarr_build) and lftj.c diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index 2d19afd3..b2e2716e 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -2073,8 +2073,8 @@ col_op_consolidate_incremental_delta_fail(col_rel_t *delta_out, } static int -col_op_consolidate_incremental_delta_append_locked(col_rel_t *delta_out, - const int64_t *row, wl_columnar_source_access_writer_t *delta_writer, +col_op_consolidate_incremental_delta_append_with_lease(col_rel_t *delta_out, + const int64_t *row, wl_columnar_relation_mutation_lease_t *delta_lease, bool *test_hook_called) { #ifdef WL_TEST_CONSOLIDATE_HOOK @@ -2087,19 +2087,17 @@ col_op_consolidate_incremental_delta_append_locked(col_rel_t *delta_out, #else (void)test_hook_called; #endif - return col_rel_append_row_locked(delta_out, row, delta_writer); + return col_rel_append_row_with_lease(delta_out, row, delta_lease); } static int col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, col_rel_t *delta_out, int *out_fast_path, - wl_columnar_source_access_writer_t *delta_writer, - const wl_columnar_source_access_writer_t *rel_writer, - bool *out_rel_alias_release_pending) + wl_columnar_relation_mutation_lease_t *rel_lease, + wl_columnar_relation_mutation_lease_t *delta_lease) { - if (!out_rel_alias_release_pending) + if (!rel_lease || (delta_out && !delta_lease)) return EINVAL; - *out_rel_alias_release_pending = false; if (!wl_columnar_relation_float_values_valid(rel) || (delta_out && !wl_columnar_relation_float_values_valid(delta_out))) return EINVAL; @@ -2130,18 +2128,23 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, EOVERFLOW); - /* Phase 1: sort only the new delta rows using radix sort. Sorting can - * allocate permutation buffers, so do not continue with an unsorted - * source if admission fails. */ - /* rel_writer, never delta_writer: the latter is a lease on delta_out's - * canonical owner, which is a different gate whenever the two relations - * do not share one. */ - int sort_rc = col_rel_radix_sort_locked(rel, old_nrows, delta_count, - rel_writer, true, out_rel_alias_release_pending); + /* Sort only the new delta rows after preparing all permutation scratch. + * The exact source lease covers the relation being sorted, regardless of + * whether delta_out shares its owner. */ + wl_columnar_radix_workspace_t sort_workspace = { 0 }; + int sort_rc = wl_columnar_relation_radix_workspace_prepare_with_lease(rel, + old_nrows, delta_count, &sort_workspace, rel_lease); + if (sort_rc == 0) + sort_rc = wl_columnar_relation_radix_sort_consolidation_with_lease( + rel, old_nrows, delta_count, &sort_workspace, rel_lease); + wl_columnar_radix_workspace_destroy(&sort_workspace); if (sort_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, sort_rc); + /* From here on, partial progress is intentionally retained on errors. */ + rel_lease->set->published = true; + /* Phase 1b: dedup within delta */ uint32_t d_unique = 1; for (uint32_t i = 1; i < delta_count; i++) { @@ -2190,12 +2193,10 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, bool compact = rel->run_count >= COL_MAX_RUNS; col_rel_compact_scratch_t scratch; if (rel->col_shared && compact) { - int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, - rel_writer); + int cow_rc = col_rel_cow_unshare_with_lease(rel, rel_lease); if (cow_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, cow_rc); - *out_rel_alias_release_pending = true; } /* All d_unique rows are novel. Emit to delta_out and append as run. */ @@ -2208,8 +2209,8 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, for (uint32_t k = 0; k < d_unique; k++) { for (uint32_t c = 0; c < nc; c++) dr[c] = rel->columns[c][old_nrows + k]; - int rc = col_op_consolidate_incremental_delta_append_locked( - delta_out, dr, delta_writer, + int rc = col_op_consolidate_incremental_delta_append_with_lease( + delta_out, dr, delta_lease, &delta_append_hook_called); if (rc != 0) { col_row_buf_release(&drb); @@ -2251,11 +2252,21 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, } if (rel->timestamps) { + wl_columnar_relation_terminal_sequence_t terminal = { 0 }; + int terminal_rc = wl_columnar_relation_terminal_sequence_begin( + &terminal, rel_lease, + WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE); + if (terminal_rc != 0) + return col_op_consolidate_incremental_delta_fail(delta_out, + delta_initial_nrows, terminal_rc); free(rel->timestamps); rel->timestamps = NULL; rel->timestamp_capacity = 0; col_rel_retire_payload_credit(rel); - wl_columnar_relation_touch_storage(rel); + wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + &terminal); + if (wl_columnar_relation_terminal_sequence_finish(&terminal) != 0) + abort(); } if (out_fast_path) *out_fast_path = 1; @@ -2267,12 +2278,10 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, * view before those mutations so the deferred cleanup can retire its * alias borrow while the owner writer remains held. */ if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, - rel_writer); + int cow_rc = col_rel_cow_unshare_with_lease(rel, rel_lease); if (cow_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, cow_rc); - *out_rel_alias_release_pending = true; } /* Adaptive dispatch (#369): use binary-search dedup when D << N, @@ -2307,8 +2316,9 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, for (uint32_t c = 0; c < nc; c++) dr[c] = rel->columns[c][old_nrows + novel_count]; int rc - = col_op_consolidate_incremental_delta_append_locked( - delta_out, dr, delta_writer, + = col_op_consolidate_incremental_delta_append_with_lease + ( + delta_out, dr, delta_lease, &delta_append_hook_called); if (rc != 0) { col_row_buf_release(&drb); @@ -2368,11 +2378,21 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, rel->sorted_nrows = rel->nrows; if (rel->timestamps) { + wl_columnar_relation_terminal_sequence_t terminal = { 0 }; + int terminal_rc = wl_columnar_relation_terminal_sequence_begin( + &terminal, rel_lease, + WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE); + if (terminal_rc != 0) + return col_op_consolidate_incremental_delta_fail(delta_out, + delta_initial_nrows, terminal_rc); free(rel->timestamps); rel->timestamps = NULL; rel->timestamp_capacity = 0; col_rel_retire_payload_credit(rel); - wl_columnar_relation_touch_storage(rel); + wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + &terminal); + if (wl_columnar_relation_terminal_sequence_finish(&terminal) != 0) + abort(); } if (out_fast_path) *out_fast_path = 0; @@ -2389,7 +2409,8 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, : rel->merge_buf_cap * 2; if (new_cap < max_rows) new_cap = max_rows; - int grid_rc = col_rel_reserve_merge_grid(rel, new_cap); + int grid_rc = col_rel_reserve_merge_grid_with_lease(rel, new_cap, + rel_lease); if (grid_rc != 0) return col_op_consolidate_incremental_delta_fail(delta_out, delta_initial_nrows, grid_rc); @@ -2441,8 +2462,8 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, if (delta_out) { for (uint32_t c = 0; c < nc; c++) delta_row[c] = merged_cols[c][out]; - int rc = col_op_consolidate_incremental_delta_append_locked( - delta_out, delta_row, delta_writer, + int rc = col_op_consolidate_incremental_delta_append_with_lease( + delta_out, delta_row, delta_lease, &delta_append_hook_called); if (rc != 0) { col_row_buf_release(&delta_rb); @@ -2465,8 +2486,8 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, if (delta_out) { for (uint32_t c = 0; c < nc; c++) delta_row[c] = merged_cols[c][out]; - int rc = col_op_consolidate_incremental_delta_append_locked( - delta_out, delta_row, delta_writer, + int rc = col_op_consolidate_incremental_delta_append_with_lease( + delta_out, delta_row, delta_lease, &delta_append_hook_called); if (rc != 0) { col_row_buf_release(&delta_rb); @@ -2482,13 +2503,32 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, /* Swap merge_columns and columns to avoid O(N) memcpy (issue #218). */ { + wl_columnar_relation_terminal_sequence_t terminal = { 0 }; + uint32_t terminal_events = WL_COLUMNAR_RELATION_TERMINAL_GRID_SWAP; + if (rel->timestamps) + terminal_events |= WL_COLUMNAR_RELATION_TERMINAL_TIMESTAMP_RETIRE; + int terminal_rc = wl_columnar_relation_terminal_sequence_begin( + &terminal, rel_lease, terminal_events); + if (terminal_rc != 0) + return col_op_consolidate_incremental_delta_fail(delta_out, + delta_initial_nrows, terminal_rc); int64_t **old_cols = rel->columns; uint32_t old_cap = rel->capacity; rel->columns = rel->merge_columns; rel->capacity = rel->merge_buf_cap; rel->merge_columns = old_cols; rel->merge_buf_cap = old_cap; - wl_columnar_relation_touch_storage(rel); + wl_columnar_relation_terminal_sequence_publish_grid_swap(&terminal); + if (rel->timestamps) { + free(rel->timestamps); + rel->timestamps = NULL; + rel->timestamp_capacity = 0; + col_rel_retire_payload_credit(rel); + wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( + &terminal); + } + if (wl_columnar_relation_terminal_sequence_finish(&terminal) != 0) + abort(); } rel->nrows = out; wl_columnar_relation_touch_view(rel); @@ -2496,13 +2536,6 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, rel->run_count = 1; rel->run_ends[0] = out; - if (rel->timestamps) { - free(rel->timestamps); - rel->timestamps = NULL; - rel->timestamp_capacity = 0; - col_rel_retire_payload_credit(rel); - wl_columnar_relation_touch_storage(rel); - } if (out_fast_path) *out_fast_path = 0; return 0; @@ -2512,76 +2545,36 @@ int col_op_consolidate_incremental_delta(col_rel_t *rel, uint32_t old_nrows, col_rel_t *delta_out, int *out_fast_path) { - col_rel_t *rel_owner = NULL; - col_rel_t *delta_owner = NULL; - wl_columnar_source_access_writer_t rel_writer = { 0 }; - wl_columnar_source_access_writer_t delta_writer = { 0 }; - wl_columnar_source_access_writer_t *delta_writer_ptr = NULL; - bool rel_acquired = false; - bool delta_acquired = false; - bool rel_alias_release_pending = false; - bool delta_alias_release_pending = false; + wl_columnar_relation_mutation_role_t roles[2] = { 0 }; + wl_columnar_relation_mutation_descriptor_t descriptors[2] = { 0 }; + wl_columnar_relation_mutation_owner_t owners[4] = { 0 }; + wl_columnar_relation_mutation_lease_t leases[2] = { 0 }; + wl_columnar_relation_mutation_initialization_t initializations[2] = { 0 }; + wl_columnar_relation_mutation_set_t set = { 0 }; + size_t role_count = delta_out ? 2u : 1u; int rc; - if (!wl_columnar_relation_float_values_valid(rel) - || (delta_out && !wl_columnar_relation_float_values_valid(delta_out))) + if (!rel || delta_out == rel) return EINVAL; - if (delta_out == rel - || (delta_out && delta_out->ncols != rel->ncols)) - return EINVAL; - rc = col_rel_storage_owner_resolve(rel, &rel_owner); + roles[0] = (wl_columnar_relation_mutation_role_t) { + .relation = rel, + .role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + if (delta_out) + roles[1] = (wl_columnar_relation_mutation_role_t) { + .relation = delta_out, + .role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + rc = col_rel_mutation_set_acquire(&set, roles, role_count, + descriptors, 2, owners, 4, leases, 2, initializations, 2); if (rc != 0) return rc; - if (delta_out) { - rc = col_rel_storage_owner_resolve(delta_out, &delta_owner); - if (rc != 0) - return rc; - } - - /* Acquire the two canonical owners in address order. The same owner is - * acquired only once when rel and delta_out alias the same relation. */ - if (!delta_owner || rel_owner == delta_owner) { - rc = wl_columnar_source_access_writer_acquire( - &rel_owner->source_access, &rel_writer); - if (rc != 0) - return rc; - rel_acquired = true; - delta_writer_ptr = delta_out ? &rel_writer : NULL; - } else if ((uintptr_t)rel_owner < (uintptr_t)delta_owner) { - rc = wl_columnar_source_access_writer_acquire( - &rel_owner->source_access, &rel_writer); - if (rc != 0) - return rc; - rel_acquired = true; - rc = wl_columnar_source_access_writer_acquire( - &delta_owner->source_access, &delta_writer); - if (rc != 0) - goto cleanup; - delta_acquired = true; - delta_writer_ptr = &delta_writer; - } else { - rc = wl_columnar_source_access_writer_acquire( - &delta_owner->source_access, &delta_writer); - if (rc != 0) - return rc; - delta_acquired = true; - rc = wl_columnar_source_access_writer_acquire( - &rel_owner->source_access, &rel_writer); - if (rc != 0) - goto cleanup; - rel_acquired = true; - delta_writer_ptr = &delta_writer; - } - - /* Owner alias counts are policy state, not a safe preflight snapshot. - * All involved writer gates are held here, so a zero count cannot become - * stale before the in-place operation begins. */ - if ((rel_owner == rel - && col_rel_storage_alias_borrow_count(rel_owner) > 0) - || (delta_out && delta_owner == delta_out - && col_rel_storage_alias_borrow_count(delta_owner) > 0)) { - rc = EBUSY; - goto cleanup; + if (!wl_columnar_relation_float_values_valid(rel) + || (delta_out + && (!wl_columnar_relation_float_values_valid(delta_out) + || delta_out->ncols != rel->ncols))) { + rc = EINVAL; + goto finish; } if (delta_out) { @@ -2596,38 +2589,23 @@ col_op_consolidate_incremental_delta(col_rel_t *rel, uint32_t old_nrows, >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u - (uint64_t)delta_count) { rc = EOVERFLOW; - goto cleanup; + goto finish; } - rc = col_rel_reserve_rows_locked(delta_out, delta_count, - delta_writer_ptr, &delta_alias_release_pending); + rc = col_rel_reserve_rows_with_lease(delta_out, delta_count, + &leases[1]); if (rc != 0) - goto cleanup; + goto finish; } rc = col_op_consolidate_incremental_delta_impl(rel, old_nrows, - delta_out, out_fast_path, delta_writer_ptr, &rel_writer, - &rel_alias_release_pending); + delta_out, out_fast_path, &leases[0], + delta_out ? &leases[1] : NULL); -cleanup: - /* Keep each old canonical-owner gate closed until every mutation is - * complete, then end the deferred aliases before releasing its writer. */ - if (delta_alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(delta_out); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; - } - if (rel_alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(rel); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; +finish: + { + int finish_rc = col_rel_mutation_set_finish(&set, rc == 0); + if (rc == 0) + rc = finish_rc; } - if (delta_acquired - && wl_columnar_source_access_writer_release(&delta_writer) != 0 - && rc == 0) - rc = EINVAL; - if (rel_acquired - && wl_columnar_source_access_writer_release(&rel_writer) != 0 - && rc == 0) - rc = EINVAL; return rc; } diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 7a282776..b8eedc89 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -1964,6 +1964,17 @@ col_rel_reserve_merge_grid(col_rel_t *r, uint32_t capacity) return 0; } +int +col_rel_reserve_merge_grid_with_lease(col_rel_t *r, uint32_t capacity, + wl_columnar_relation_mutation_lease_t *lease) +{ + if (!r || !lease || lease->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + return col_rel_reserve_merge_grid(r, capacity); +} + static void col_rel_release_reservation_or_abort( wl_columnar_memory_reservation_t *reservation) @@ -2606,6 +2617,17 @@ col_rel_cow_unshare(col_rel_t *r, uint32_t new_cap) return rc ? rc : finish_rc; } +int +col_rel_cow_unshare_with_lease(col_rel_t *r, + wl_columnar_relation_mutation_lease_t *lease) +{ + if (!r || !lease || lease->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + return col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); +} + static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, bool defer_metadata_retirement, @@ -4796,6 +4818,18 @@ col_rel_append_row(col_rel_t *r, const int64_t *row) return rc ? rc : finish_rc; } +int +col_rel_append_row_with_lease(col_rel_t *r, const int64_t *row, + wl_columnar_relation_mutation_lease_t *lease) +{ + if (!r || !row || !lease || lease->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + return col_rel_append_row_impl(r, row, + &lease->set->owners[lease->owner_slot].writer, true, lease); +} + int col_rel_append_row_locked(col_rel_t *r, const int64_t *row, wl_columnar_source_access_writer_t *writer) @@ -5183,6 +5217,18 @@ col_rel_reserve_rows_locked(col_rel_t *r, uint32_t additional, out_alias_release_pending, true, NULL); } +int +col_rel_reserve_rows_with_lease(col_rel_t *r, uint32_t additional, + wl_columnar_relation_mutation_lease_t *lease) +{ + if (!r || !lease || lease->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + return wl_columnar_relation_reserve_rows_impl(r, additional, NULL, NULL, + false, lease); +} + int col_rel_reset_rows_locked(col_rel_t *r, wl_columnar_source_access_writer_t *writer) @@ -10711,10 +10757,10 @@ wl_columnar_relation_radix_workspace_prepare_with_lease(col_rel_t *r, workspace, false); } -int -wl_columnar_relation_radix_sort_with_lease(col_rel_t *r, uint32_t start, +static int +wl_columnar_relation_radix_sort_with_lease_impl(col_rel_t *r, uint32_t start, uint32_t count, const wl_columnar_radix_workspace_t *workspace, - wl_columnar_relation_mutation_lease_t *lease) + wl_columnar_relation_mutation_lease_t *lease, bool consolidation_hook) { int rc = wl_columnar_radix_lease_preflight(r, start, count, lease); if (rc) @@ -10746,6 +10792,14 @@ wl_columnar_relation_radix_sort_with_lease(col_rel_t *r, uint32_t start, rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); if (rc) return rc; +#ifdef WL_TEST_CONSOLIDATE_HOOK + if (consolidation_hook + && wl_columnar_consolidation_transition_hook) + wl_columnar_consolidation_transition_hook(r, + WL_COLUMNAR_CONSOLIDATION_TEST_SORT_AFTER_DETACH); +#else + (void)consolidation_hook; +#endif } /* After detach/first row move, prepared-only kernels cannot fail. */ rc = wl_columnar_radix_rows_prepared(r, start, count, workspace, @@ -10755,6 +10809,25 @@ wl_columnar_relation_radix_sort_with_lease(col_rel_t *r, uint32_t start, return 0; } +int +wl_columnar_relation_radix_sort_with_lease(col_rel_t *r, uint32_t start, + uint32_t count, const wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease) +{ + return wl_columnar_relation_radix_sort_with_lease_impl(r, start, count, + workspace, lease, false); +} + +int +wl_columnar_relation_radix_sort_consolidation_with_lease(col_rel_t *r, + uint32_t start, uint32_t count, + const wl_columnar_radix_workspace_t *workspace, + wl_columnar_relation_mutation_lease_t *lease) +{ + return wl_columnar_relation_radix_sort_with_lease_impl(r, start, count, + workspace, lease, true); +} + /* One descriptor-first mutation set covers range capture, preparation, * detach, permutation and optional full-sort metadata publication. */ static int From ed7981daa5d647fa0735ac2399c9d31cee21b634 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 00:29:57 +0900 Subject: [PATCH 04/15] Preflight consolidation before delta reservation --- tests/test_consolidate_incremental_delta.c | 91 ++++++++++++++++++++++ wirelog/columnar/merge.c | 42 ++++++---- 2 files changed, 120 insertions(+), 13 deletions(-) diff --git a/tests/test_consolidate_incremental_delta.c b/tests/test_consolidate_incremental_delta.c index f3816098..028010be 100644 --- a/tests/test_consolidate_incremental_delta.c +++ b/tests/test_consolidate_incremental_delta.c @@ -2688,6 +2688,95 @@ test_shared_view_storage_exhaustion(void) PASS(); } +static void +test_source_exhaustion_precedes_delta_reservation(void) +{ + TEST("source exhaustion precedes shared delta COW and clears stale denial"); + + col_rel_t *rel = test_rel_alloc(1); + col_rel_t *delta_owner = test_rel_alloc(1); + col_rel_t *delta_out = test_rel_alloc(1); + int64_t source_rows[] = { 10, 20 }; + int64_t owner_row = 99; + ASSERT(rel && delta_owner && delta_out, "source preflight relations"); + ASSERT(test_rel_append_row(rel, &source_rows[0]) == 0 + && test_rel_append_row(rel, &source_rows[1]) == 0 + && test_rel_append_row(delta_owner, &owner_row) == 0 + && col_rel_install_shared_view(delta_out, delta_owner) == 0, + "source and shared delta fixtures"); + + uint64_t last = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + rel->storage_generation = last; + rel->storage_owner_generation = last; + rel->memory_budget_denial_pending = true; + delta_out->memory_budget_denial_pending = true; + int64_t *delta_columns = delta_out->columns[0]; + uint64_t delta_generation = delta_out->storage_generation; + uint64_t delta_view_generation = delta_out->view_generation; + uint64_t delta_borrows = col_rel_storage_alias_borrow_count(delta_owner); + uint64_t owner_generation = delta_owner->storage_generation; + int fast_path = -1; + + ASSERT(col_op_consolidate_incremental_delta(rel, 1, delta_out, + &fast_path) == EOVERFLOW && fast_path == -1, + "exhausted source is refused before reserving output"); + ASSERT(!rel->memory_budget_denial_pending + && !delta_out->memory_budget_denial_pending, + "admitted non-budget failure clears stale denial evidence"); + ASSERT(rel->nrows == 2 && rel->columns[0][0] == 10 + && rel->columns[0][1] == 20 + && rel->storage_generation == last + && rel->storage_owner_generation == last, + "source remains unchanged at exhausted generation"); + ASSERT(delta_out->col_shared && delta_out->col_shared[0] + && delta_out->storage_owner == delta_owner + && delta_out->columns[0] == delta_columns + && delta_out->storage_generation == delta_generation + && delta_out->view_generation == delta_view_generation + && col_rel_storage_alias_borrow_count(delta_owner) == delta_borrows + && delta_owner->storage_generation == owner_generation + && delta_out->nrows == 1 && delta_out->columns[0][0] == owner_row, + "shared delta remains borrowed without COW or generation changes"); + + test_rel_free(delta_out); + test_rel_free(delta_owner); + test_rel_free(rel); + PASS(); +} + +static void +test_admission_failure_preserves_stale_denial(void) +{ + TEST("mutation admission failure preserves stale denial evidence"); + + col_rel_t *rel = test_rel_alloc(1); + col_rel_t *delta_out = test_rel_alloc(1); + int64_t source_row = 10, delta_row = 20; + wl_columnar_source_access_writer_t held = { 0 }; + ASSERT(rel && delta_out + && test_rel_append_row(rel, &source_row) == 0 + && test_rel_append_row(delta_out, &delta_row) == 0, + "admission failure relations"); + rel->memory_budget_denial_pending = true; + delta_out->memory_budget_denial_pending = true; + ASSERT(col_rel_source_writer_acquire(rel, &held) == 0, + "hold source writer to refuse mutation-set admission"); + int fast_path = -1; + int rc = col_op_consolidate_incremental_delta(rel, 0, delta_out, + &fast_path); + ASSERT(wl_columnar_source_access_writer_release(&held) == 0, + "release held source writer"); + ASSERT(rc == EBUSY && fast_path == -1 + && rel->memory_budget_denial_pending + && delta_out->memory_budget_denial_pending + && rel->nrows == 1 && delta_out->nrows == 1, + "failed admission preserves prior evidence and rows"); + + test_rel_free(delta_out); + test_rel_free(rel); + PASS(); +} + /* ================================================================ * Issue #2049: every view-generation advance an incremental consolidation * may make is reserved before the delta sort, as #2046 does for storage. @@ -3106,6 +3195,8 @@ main(void) test_binary_retirement_uses_last_generation(); test_fallback_exhaustion_without_timestamps(); test_shared_view_storage_exhaustion(); + test_source_exhaustion_precedes_delta_reservation(); + test_admission_failure_preserves_stale_denial(); /* View-generation headroom (Issue #2049) */ test_view_generation_headroom_reserved(); diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index b2e2716e..c919b300 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -2031,6 +2031,26 @@ col_op_consolidate_view_steps(const col_rel_t *rel, uint32_t old_nrows, + (compacting ? 3u : 1u); } +static int +col_op_consolidate_incremental_delta_generation_preflight( + const col_rel_t *rel, uint32_t old_nrows) +{ + if (rel->nrows == 0 || old_nrows >= rel->nrows) + return 0; + uint32_t delta_count = rel->nrows - old_nrows; + uint32_t storage_steps = col_op_consolidate_storage_steps(rel, + old_nrows, delta_count); + uint32_t view_steps = col_op_consolidate_view_steps(rel, old_nrows, + delta_count); + if ((storage_steps > 0 + && rel->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - (uint64_t)storage_steps) + || rel->view_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - (uint64_t)view_steps) + return EOVERFLOW; + return 0; +} + /* * col_op_consolidate_incremental_delta - Incremental consolidation with delta output * @@ -2116,18 +2136,6 @@ col_op_consolidate_incremental_delta_impl(col_rel_t *rel, uint32_t old_nrows, uint32_t delta_count = nr - old_nrows; - uint32_t storage_steps = col_op_consolidate_storage_steps(rel, - old_nrows, delta_count); - uint32_t view_steps = col_op_consolidate_view_steps(rel, old_nrows, - delta_count); - if ((storage_steps > 0 - && rel->storage_generation - >= WL_COLUMNAR_REL_GENERATION_INVALID - (uint64_t)storage_steps) - || rel->view_generation - >= WL_COLUMNAR_REL_GENERATION_INVALID - (uint64_t)view_steps) - return col_op_consolidate_incremental_delta_fail(delta_out, - delta_initial_nrows, EOVERFLOW); - /* Sort only the new delta rows after preparing all permutation scratch. * The exact source lease covers the relation being sorted, regardless of * whether delta_out shares its owner. */ @@ -2569,6 +2577,9 @@ col_op_consolidate_incremental_delta(col_rel_t *rel, uint32_t old_nrows, descriptors, 2, owners, 4, leases, 2, initializations, 2); if (rc != 0) return rc; + rel->memory_budget_denial_pending = false; + if (delta_out) + delta_out->memory_budget_denial_pending = false; if (!wl_columnar_relation_float_values_valid(rel) || (delta_out && (!wl_columnar_relation_float_values_valid(delta_out) @@ -2577,7 +2588,12 @@ col_op_consolidate_incremental_delta(col_rel_t *rel, uint32_t old_nrows, goto finish; } - if (delta_out) { + rc = col_op_consolidate_incremental_delta_generation_preflight(rel, + old_nrows); + if (rc != 0) + goto finish; + + if (delta_out && rel->nrows > old_nrows) { uint32_t delta_count = rel->nrows > old_nrows ? rel->nrows - old_nrows : 0; /* Reserve delta_out's view-generation advances before its first From 267438cf0ac822456ca16435c12d63f693d41380 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 00:39:01 +0900 Subject: [PATCH 05/15] Remove redundant terminal lease check --- wirelog/columnar/relation.c | 1 - 1 file changed, 1 deletion(-) diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index b8eedc89..14e9f0e7 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -1009,7 +1009,6 @@ wl_columnar_relation_terminal_tokens_valid( continue; if (role->relation != relation || same->relation != relation || same->set != set || same->identity != (uintptr_t)same - || same->relation != relation || same->role_flags != role->role_flags) return false; if (role->role_flags != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION) From 8564d40eadad28769be6bbac485f5b02d24f3c08 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 00:53:20 +0900 Subject: [PATCH 06/15] Migrate worker delta reservation to admitted capacity --- tests/test_memory_admission_relation.c | 46 ++++++++++++++------------ tests/test_session.c | 35 +++++++++++++++----- wirelog/columnar/eval.c | 19 +++-------- wirelog/columnar/eval_serial.c | 13 ++------ wirelog/columnar/internal.h | 4 --- wirelog/columnar/relation.c | 9 ----- 6 files changed, 57 insertions(+), 69 deletions(-) diff --git a/tests/test_memory_admission_relation.c b/tests/test_memory_admission_relation.c index 3384dd9a..bbb784ab 100644 --- a/tests/test_memory_admission_relation.c +++ b/tests/test_memory_admission_relation.c @@ -4305,6 +4305,28 @@ test_retraction_backup_timestamp_admission(void) /* Issue #1991: exercise the real governor verdicts with small valid storage. * A separate accounting-only token fills the counter; no huge allocation or * forged column dimensions are needed to force arithmetic overflow. */ +static int +reserve_rows_with_exact_lease(col_rel_t *rel, uint32_t additional) +{ + wl_columnar_relation_mutation_role_t role = { + .relation = rel, + .role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + wl_columnar_relation_mutation_descriptor_t descriptor = { 0 }; + wl_columnar_relation_mutation_owner_t owners[2] = { 0 }; + wl_columnar_relation_mutation_lease_t lease = { 0 }; + wl_columnar_relation_mutation_initialization_t initialization = { 0 }; + wl_columnar_relation_mutation_set_t set = { 0 }; + int rc = col_rel_mutation_set_acquire(&set, &role, 1, &descriptor, 1, + owners, 2, &lease, 1, &initialization, 1); + if (rc != 0) + return rc; + rel->memory_budget_denial_pending = false; + rc = col_rel_reserve_rows_with_lease(rel, additional, &lease); + int finish_rc = col_rel_mutation_set_finish(&set, rc == 0); + return rc ? rc : finish_rc; +} + static int exercise_growth_boundary(col_rel_t *rel, unsigned boundary, bool *denied) { @@ -4318,18 +4340,7 @@ exercise_growth_boundary(col_rel_t *rel, unsigned boundary, bool *denied) return col_rel_append_row(rel, &row); if (boundary == 3) return col_rel_append_rows_atomic(rel, &row, 1u, 1u, denied); - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_pending = false; - int rc = col_rel_source_writer_acquire(rel, &writer); - if (rc != 0) - return rc; - rc = col_rel_reserve_rows_locked(rel, 1u, &writer, &alias_pending); - if (alias_pending) - CHECK(col_rel_storage_alias_release(rel) == 0, - "growth boundary releases deferred alias"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "growth boundary releases its writer"); - return rc; + return reserve_rows_with_exact_lease(rel, 1u); } static void @@ -4651,15 +4662,8 @@ test_no_growth_admission_provenance(void) == 0 && !denied && !rel->memory_budget_denial_pending && rel->columns == columns && rel->storage_generation == generation, "no-growth retry admits same physical image"); - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_pending = false; - CHECK(col_rel_source_writer_acquire(rel, &writer) == 0, - "row-count overflow writer"); - CHECK(col_rel_reserve_rows_locked(rel, UINT32_MAX, &writer, - &alias_pending) == EOVERFLOW && !alias_pending, - "locked reserve reports capacity arithmetic overflow before allocation"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "row-count overflow writer release"); + CHECK(reserve_rows_with_exact_lease(rel, UINT32_MAX) == EOVERFLOW, + "leased reserve reports capacity arithmetic overflow before allocation"); cleanup: col_rel_destroy(rel); if (ref) { diff --git a/tests/test_session.c b/tests/test_session.c index 1e436453..7342059e 100644 --- a/tests/test_session.c +++ b/tests/test_session.c @@ -9439,6 +9439,21 @@ test_worker_delta_preflight(uint32_t ncols, uint32_t rows) uint64_t source_storage = source->storage_generation; int64_t **source_columns = source->columns; uint64_t baseline = wl_columnar_memory_reserved(budget); + if (rows == 0) { + uint32_t initial_capacity = source->capacity; + DELTA_CHECK(wl_columnar_eval_test_prepare_worker_delta(&delta, + "$d$source", source, 0, sess->memory_governor) == 0 + && delta && delta->capacity == initial_capacity + && delta->timestamp_capacity == initial_capacity + && (!!delta->timestamps == (initial_capacity != 0)), + "empty worker delta keeps its initial capacity and timestamps"); + DELTA_CHECK(col_rel_destroy_checked(delta) == 0, + "empty worker delta teardown"); + delta = NULL; + DELTA_CHECK(wl_columnar_memory_reserved(budget) == baseline, + "empty worker delta returns its credit"); + goto cleanup; + } DELTA_CHECK(wl_columnar_eval_test_prepare_worker_delta(&delta, "$d$source", source, UINT32_MAX, sess->memory_governor) == EOVERFLOW && !delta @@ -9461,15 +9476,16 @@ test_worker_delta_preflight(uint32_t ncols, uint32_t rows) "$d$source", source, sess->memory_governor) == 0, "measure descriptor stage"); constructor_peak = wl_columnar_memory_reserved(budget); - wl_columnar_source_access_writer_t writer = { 0 }; - DELTA_CHECK(col_rel_source_writer_acquire(delta, &writer) == 0, - "measure writer"); - bool alias_release_pending = false; - int stage_rc = col_rel_reserve_rows_locked(delta, rows, &writer, - &alias_release_pending); - int release_rc = wl_columnar_source_access_writer_release(&writer); - DELTA_CHECK(stage_rc == 0 && release_rc == 0 - && !alias_release_pending, "measure grid stage"); + uint32_t target_capacity = delta->capacity + ? delta->capacity : COL_REL_INIT_CAP; + while (target_capacity < rows) { + DELTA_CHECK(target_capacity <= UINT32_MAX / 2u, + "measure capacity bound"); + target_capacity *= 2u; + } + int stage_rc = col_rel_reserve_capacity_admitted(delta, target_capacity, + NULL); + DELTA_CHECK(stage_rc == 0, "measure grid stage"); grid_peak = wl_columnar_memory_reserved(budget); DELTA_CHECK(constructor_peak > baseline && grid_peak >= constructor_peak && peak > grid_peak, "strict timestamp stage"); @@ -15722,6 +15738,7 @@ main(void) test_worker_delta_preflight(1, 1); test_worker_delta_preflight(1, 17); test_worker_delta_preflight(0, 17); + test_worker_delta_preflight(1, 0); #ifdef WL_TEST_ALLOC_WRAP test_recursive_delta_publication_failure(true); #endif diff --git a/wirelog/columnar/eval.c b/wirelog/columnar/eval.c index 3c96ab60..11c840bc 100644 --- a/wirelog/columnar/eval.c +++ b/wirelog/columnar/eval.c @@ -2002,21 +2002,10 @@ wl_columnar_eval_prepare_worker_delta(col_rel_t **out, const char *name, col_rel_destroy(delta); return EOVERFLOW; } - wl_columnar_source_access_writer_t writer = { 0 }; - rc = col_rel_source_writer_acquire(delta, &writer); - if (rc == 0) { - bool alias_release_pending = false; - rc = col_rel_reserve_rows_locked(delta, rows, &writer, - &alias_release_pending); - /* A newly created heap delta cannot borrow another relation. */ - if (rc == 0 && alias_release_pending) - rc = EBUSY; - if (rc == 0) - rc = col_rel_enable_timestamps_locked(delta); - int release_rc = wl_columnar_source_access_writer_release(&writer); - if (rc == 0) - rc = release_rc; - } + rc = rows > 0 + ? col_rel_reserve_capacity_admitted(delta, target_capacity, NULL) : 0; + if (rc == 0) + rc = col_rel_enable_timestamps(delta); if (rc != 0) { if (rc == ENOMEM && delta->memory_budget_denial_pending) rc = ENOSPC; diff --git a/wirelog/columnar/eval_serial.c b/wirelog/columnar/eval_serial.c index 20749ae1..83be8193 100644 --- a/wirelog/columnar/eval_serial.c +++ b/wirelog/columnar/eval_serial.c @@ -846,17 +846,8 @@ col_eval_stratum(const wl_plan_stratum_t *sp, wl_col_session_t *sess, * one timestamp slot admissible before consolidation can emit * their first tuple. */ if (delta->capacity == 0) { - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_release_pending = false; - int reserve_rc = col_rel_source_writer_acquire(delta, - &writer); - if (reserve_rc == 0) - reserve_rc = col_rel_reserve_rows_locked(delta, 1, - &writer, &alias_release_pending); - int release_rc = - wl_columnar_source_access_writer_release(&writer); - if (reserve_rc == 0) - reserve_rc = release_rc; + int reserve_rc = col_rel_reserve_capacity_admitted(delta, + COL_REL_INIT_CAP, NULL); if (reserve_rc != 0) { col_rel_destroy(delta); outer_rc = reserve_rc; diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 593183f7..4cf1be00 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2967,10 +2967,6 @@ int col_rel_append_rows_atomic(col_rel_t *r, const int64_t *rows, uint32_t num_rows, uint32_t num_cols, bool *denied); int -col_rel_reserve_rows_locked(col_rel_t *r, uint32_t additional, - wl_columnar_source_access_writer_t *writer, - bool *out_alias_release_pending); -int col_rel_reset_rows_locked(col_rel_t *r, wl_columnar_source_access_writer_t *writer); /* Detach only row storage under descriptor/owner exclusion. The caller must diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 14e9f0e7..50dc6d75 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -5207,15 +5207,6 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, return 0; } -int -col_rel_reserve_rows_locked(col_rel_t *r, uint32_t additional, - wl_columnar_source_access_writer_t *writer, - bool *out_alias_release_pending) -{ - return wl_columnar_relation_reserve_rows_impl(r, additional, writer, - out_alias_release_pending, true, NULL); -} - int col_rel_reserve_rows_with_lease(col_rel_t *r, uint32_t additional, wl_columnar_relation_mutation_lease_t *lease) From ac17356603953e6e271d59faf918920f8f205bcc Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 01:21:24 +0900 Subject: [PATCH 07/15] Add leased prepared radix sequences --- tests/test_relation_generations.c | 313 +++++++++++++++++++++++++++++ tests/test_relation_mutation_set.c | 12 ++ wirelog/columnar/internal.h | 41 ++++ wirelog/columnar/relation.c | 287 +++++++++++++++++++++++++- 4 files changed, 651 insertions(+), 2 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index caaad6ac..5988311d 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -6592,6 +6592,318 @@ test_leased_radix_provenance(void) cleanup_relations(); } +typedef struct { + wl_columnar_relation_radix_sequence_t *sequence; + wl_columnar_relation_mutation_lease_t *lease; + int rc; +} radix_sequence_wrong_thread_t; + +static void * +radix_sequence_test_wrong_thread(void *opaque) +{ + radix_sequence_wrong_thread_t *args = opaque; + args->rc = wl_columnar_relation_radix_sequence_execute_with_lease( + args->sequence, args->lease); + return NULL; +} + +static void +test_leased_radix_sequence(void) +{ + col_rel_t *root = new_relation(); + col_rel_t *alias = new_relation(); + for (int64_t value = 1; value <= 4; value++) + CHECK(col_rel_append_row(root, &value) == 0, + "radix sequence sorted seed"); + for (int64_t value = 8; value >= 5; value--) + CHECK(col_rel_append_row(root, &value) == 0, + "radix sequence unsorted seed"); + for (int64_t value = 12; value >= 9; value--) + CHECK(col_rel_append_row(root, &value) == 0, + "radix sequence second unsorted seed"); + CHECK(col_rel_enable_timestamps(root) == 0, + "radix sequence timestamp allocation"); + for (uint32_t i = 0; i < root->nrows; i++) + root->timestamps[i].iteration = (uint64_t)(100u + i); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix sequence alias"); + radix_test_lease_t held = { 0 }; + wl_columnar_relation_radix_sequence_t sequence = { 0 }; + uint32_t bounds[] = { 0, 4, 8, 12 }; + CHECK(radix_test_acquire(alias, &held) == 0, + "radix sequence acquire"); + uint64_t sequence_view = alias->view_generation; + alias->view_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + bounds, 3, &sequence, &held.lease) == EOVERFLOW + && sequence.identity == 0, + "radix sequence preflights the exact view event count"); + alias->view_generation = sequence_view; + held.lease.storage_transitioned = true; + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + bounds, 3, &sequence, &held.lease) == EINVAL + && sequence.identity == 0, + "shared radix sequence rejects a prior storage transition"); + held.lease.storage_transitioned = false; +#ifdef WL_TEST_ALLOC_WRAP + col_rel_t before_alloc_failure = *alias; + for (long fail_at = 0; fail_at < 4; fail_at++) { + allocation_calls = 0; + allocation_fail_at = fail_at; + int alloc_rc = wl_columnar_relation_radix_sequence_prepare_with_lease( + alias, bounds, 3, &sequence, &held.lease); + allocation_fail_at = -1; + CHECK(alloc_rc == ENOMEM && sequence.identity == 0 + && mutation_payload_unchanged(alias, &before_alloc_failure) + && alias->storage_generation == + before_alloc_failure.storage_generation, + "radix sequence allocation failure precedes all mutation"); + } +#endif + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + bounds, 3, &sequence, &held.lease) == 0, + "radix sequence prepare"); + radix_sequence_wrong_thread_t wrong_thread = { &sequence, &held.lease, 0 }; + wl_thread_t worker; + CHECK(wl_thread_create(&worker, radix_sequence_test_wrong_thread, + &wrong_thread) == 0 && wl_thread_join(&worker) == 0 + && wrong_thread.rc == EINVAL && !sequence.consumed, + "radix sequence rejects wrong-thread execution without consumption"); + col_rel_t *foreign = new_relation(); + radix_test_lease_t foreign_held = { 0 }; + CHECK(radix_test_acquire(foreign, &foreign_held) == 0 + && wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &foreign_held.lease) == EINVAL + && col_rel_mutation_set_finish(&foreign_held.set, false) == 0, + "radix sequence rejects a foreign lease"); + bounds[1] = 8; /* The token owns the original boundaries. */ + wl_columnar_relation_radix_sequence_t copy = sequence; + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(©, + &held.lease) == EINVAL, + "radix sequence rejects a copied token"); + uint64_t view_before = alias->view_generation; + uint64_t storage_before = alias->storage_generation; +#ifdef WL_TEST_RELATION_RESIZE_HOOK + col_rel_t before_cow_failure = *alias; + wl_columnar_relation_test_fail_next_prepare_resize(); + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == ENOMEM && !sequence.consumed + && mutation_payload_unchanged(alias, &before_cow_failure) + && alias->storage_generation == storage_before + && col_rel_storage_alias_borrow_count(root) == 1, + "failed sequence COW leaves token retryable and payload unchanged"); +#endif + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 + && alias->view_generation == view_before + 2 + && alias->storage_generation == storage_before + 1 + && alias->columns[0][0] == 1 && alias->columns[0][3] == 4 + && alias->columns[0][4] == 5 && alias->columns[0][7] == 8 + && alias->columns[0][8] == 9 && alias->columns[0][11] == 12 + && alias->timestamps[4].iteration == 107 + && alias->timestamps[7].iteration == 104 + && alias->timestamps[8].iteration == 111 + && alias->timestamps[11].iteration == 108 + && root->columns[0][4] == 8 + && col_rel_storage_alias_borrow_count(root) == 0, + "radix sequence sorts copied ranges after one alias detach"); + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == EINVAL, + "radix sequence rejects replay"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence finish"); + cleanup_relations(); + + root = new_relation(); + alias = new_relation(); + for (int64_t value = 1; value <= 3; value++) + CHECK(col_rel_append_row(root, &value) == 0, + "radix zero-sort seed"); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix zero-sort alias"); + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t invalid_bounds[] = { 0, 4, 3 }; + uint32_t sorted_bounds[] = { 0, 0, 1, 3 }; + CHECK(radix_test_acquire(alias, &held) == 0, + "radix zero-sort lease"); + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + invalid_bounds, 2, &sequence, &held.lease) == EINVAL + && sequence.identity == 0, + "radix sequence validates every boundary before preparing"); + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + sorted_bounds, 3, &sequence, &held.lease) == 0, + "radix zero-sort prepare accepts empty and singleton segments"); + view_before = alias->view_generation; + storage_before = alias->storage_generation; + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 + && alias->view_generation == view_before + && alias->storage_generation == storage_before + && alias->storage_owner == root + && col_rel_storage_alias_borrow_count(root) == 1, + "zero-sort sequence consumes without detaching or advancing epochs"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix zero-sort finish"); + memset(&sequence, 0, sizeof(sequence)); + memset(&held.set, 0, sizeof(held.set)); + CHECK(radix_test_acquire(alias, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + sorted_bounds, 3, &sequence, &held.lease) == 0, + "radix sequence prepares before reacquisition"); + uint64_t prior_nonce = held.set.acquisition_nonce; + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence prior set finish"); + memset(&held.set, 0, sizeof(held.set)); + CHECK(radix_test_acquire(alias, &held) == 0 + && held.set.acquisition_nonce != prior_nonce + && wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == EINVAL, + "sequence cannot revive after unchanged descriptor reacquisition"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence reacquired set finish"); + cleanup_relations(); + +#ifdef WL_SESSION_TEST_HOOKS + root = new_relation(); + for (int64_t value = 8; value > 0; value--) + CHECK(col_rel_append_row(root, &value) == 0, + "radix sequence admission seed"); + wl_columnar_memory_resolution_t tiny_resolution = { + .budget_bytes = 1u << 20, .usable_bytes = 1u << 20, + .mode = WL_COLUMNAR_MEMORY_MODE_ENFORCING, + .source = WL_COLUMNAR_MEMORY_SOURCE_ENV, + .status = WL_COLUMNAR_MEMORY_OK + }; + wl_columnar_memory_governor_ref_t *tiny_ref + = wl_columnar_memory_governor_ref_create(&tiny_resolution); + CHECK(tiny_ref && col_rel_attach_memory_governor(root, tiny_ref) == 0, + "radix sequence attach tiny governor"); + wl_columnar_memory_reservation_t blocker; + wl_columnar_memory_reservation_init(&blocker); + uint64_t already_reserved = wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(tiny_ref)); + CHECK(already_reserved < tiny_resolution.budget_bytes + && wl_columnar_memory_reserve_checked( + wl_columnar_memory_governor_ref_get(tiny_ref), + tiny_resolution.budget_bytes - already_reserved - 1u, + &blocker) == WL_COLUMNAR_MEMORY_ADMISSION_OK, + "radix sequence reserve all but one governor byte"); + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t admission_bounds[] = { 0, 8 }; + CHECK(radix_test_acquire(root, &held) == 0, + "radix sequence admission lease"); + col_rel_t before_denial = *root; + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(root, + admission_bounds, 1, &sequence, &held.lease) == ENOMEM + && root->memory_budget_denial_pending + && mutation_payload_unchanged(root, &before_denial) + && root->view_generation == before_denial.view_generation + && root->storage_generation == before_denial.storage_generation, + "radix sequence budget denial precedes mutation"); + CHECK(col_rel_mutation_set_finish(&held.set, false) == 0, + "radix sequence denial finish"); + CHECK(wl_columnar_memory_release(&blocker), + "radix sequence release quota blocker"); + cleanup_relations(); + wl_columnar_memory_governor_ref_release(tiny_ref); +#endif + + root = new_relation(); + CHECK(col_rel_set_schema(root, 0, NULL) == 0, + "radix sequence nullary schema"); + int64_t dummy = 0; + CHECK(col_rel_append_row(root, &dummy) == 0 + && col_rel_append_row(root, &dummy) == 0, + "radix sequence nullary rows"); + alias = new_relation(); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix sequence nullary alias"); + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t nullary_bounds[] = { 0, 1, 2 }; + CHECK(radix_test_acquire(alias, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + nullary_bounds, 2, &sequence, &held.lease) == 0, + "radix sequence nullary prepare"); + view_before = alias->view_generation; + storage_before = alias->storage_generation; + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 && alias->view_generation == view_before + && alias->storage_generation == storage_before + && alias->storage_owner == root, + "nullary shared sequence is a zero-sort no-op"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence nullary finish"); + cleanup_relations(); + + root = new_relation(); + alias = new_relation(); + CHECK(col_rel_install_shared_view(alias, root) == 0, + "radix sequence zero-capacity alias"); + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t empty_bounds[] = { 0, 0 }; + CHECK(radix_test_acquire(alias, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + empty_bounds, 1, &sequence, &held.lease) == 0, + "radix sequence zero-capacity prepare"); + view_before = alias->view_generation; + storage_before = alias->storage_generation; + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 && alias->view_generation == view_before + && alias->storage_generation == storage_before + && alias->storage_owner == root, + "zero-capacity alias sequence has no publication"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence zero-capacity finish"); + cleanup_relations(); + + root = new_relation(); + wirelog_column_type_t float_type = WIRELOG_TYPE_FLOAT; + CHECK(col_rel_set_column_types(root, &float_type, 1) == 0, + "radix sequence float type"); + const double float_values[] = { 4.0, 1.0, 3.0, 2.0 }; + for (size_t i = 0; i < sizeof(float_values) / sizeof(float_values[0]); + i++) { + int64_t bits; + memcpy(&bits, &float_values[i], sizeof(bits)); + CHECK(col_rel_append_row(root, &bits) == 0, + "radix sequence float seed"); + } + CHECK(col_rel_enable_timestamps(root) == 0, + "radix sequence float timestamps"); + for (uint32_t i = 0; i < root->nrows; i++) + root->timestamps[i].iteration = 20u + i; + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t float_bounds[] = { 0, 4 }; + CHECK(radix_test_acquire(root, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(root, + float_bounds, 1, &sequence, &held.lease) == 0 + && wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0, + "radix sequence sorts float typed segment"); + const uint64_t expected_timestamps[] = { 21, 23, 22, 20 }; + for (uint32_t i = 0; i < root->nrows; i++) { + double value; + memcpy(&value, &root->columns[0][i], sizeof(value)); + CHECK(value == (double)(i + 1u) + && root->timestamps[i].iteration == expected_timestamps[i], + "radix sequence float timestamp permutation"); + } + wl_columnar_relation_radix_sequence_destroy(&sequence); + CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, + "radix sequence float finish"); + cleanup_relations(); +} + static void test_radix_descriptor_admission(void) { @@ -7186,6 +7498,7 @@ main(void) test_radix_opposite_address_contention(); test_radix_descriptor_admission(); test_leased_radix_provenance(); + test_leased_radix_sequence(); test_leased_radix_kernels(); test_public_append_mutation_admission(); test_append_overlapping_input(false); diff --git a/tests/test_relation_mutation_set.c b/tests/test_relation_mutation_set.c index e198e8d3..50872ef6 100644 --- a/tests/test_relation_mutation_set.c +++ b/tests/test_relation_mutation_set.c @@ -572,6 +572,18 @@ main(void) rollback_tests(); invalid_tests(); metadata_final_access_test(); + { + col_rel_t root; + fixture_t f = {0}; + root_init(&root, 3); + wl_columnar_relation_mutation_role_t role = { &root, + WL_COLUMNAR_RELATION_PAYLOAD_MUTATION }; + wl_columnar_relation_test_set_mutation_nonce(UINT64_MAX); + assert(acquire(&f, &role, 1) == EOVERFLOW); + assert(f.set.identity == 0 && f.set.acquisition_nonce == 0 + && atomic_load_explicit(&root.descriptor_access.state, + memory_order_relaxed) == 0); + } puts("relation mutation set tests passed"); return 0; } diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 4cf1be00..9f62c35d 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -582,6 +582,7 @@ typedef struct { struct wl_columnar_relation_mutation_set; typedef struct { uintptr_t identity; + uint64_t acquisition_nonce; struct wl_columnar_relation_mutation_set *set; col_rel_t *relation; col_rel_t *owner; @@ -600,6 +601,7 @@ typedef struct { struct wl_columnar_relation_terminal_sequence; typedef struct wl_columnar_relation_mutation_set { uintptr_t identity; + uint64_t acquisition_nonce; const wl_columnar_relation_mutation_role_t *roles; wl_columnar_relation_mutation_descriptor_t *descriptors; wl_columnar_relation_mutation_owner_t *owners; @@ -681,6 +683,7 @@ void wl_columnar_relation_terminal_sequence_publish_timestamp_retirement( int wl_columnar_relation_terminal_sequence_finish( wl_columnar_relation_terminal_sequence_t *sequence); #ifdef WL_TEST_MUTATION_SET_HOOK +void wl_columnar_relation_test_set_mutation_nonce(uint64_t next_nonce); bool wl_columnar_relation_test_terminal_sequence_validate( const wl_columnar_relation_terminal_sequence_t *sequence); #endif @@ -3210,6 +3213,35 @@ typedef struct { bool admission_active; } wl_columnar_radix_workspace_t; +/* Owning prepared token for one whole multi-segment radix sequence. Initialize + * to zero, prepare once, execute once, then destroy even after failures. */ +typedef struct wl_columnar_relation_radix_sequence { + uintptr_t identity; + uint64_t acquisition_nonce; + wl_columnar_relation_mutation_set_t *set; + wl_columnar_relation_mutation_lease_t *lease; + col_rel_t *relation; + col_rel_t *owner; + uint64_t relation_identity; + uint64_t relation_generation; + uint64_t owner_identity; + uint64_t owner_generation; + uint64_t view_generation; + uint64_t type_fingerprint; + const int64_t *const *columns; + const uint32_t *column_types; + const col_delta_timestamp_t *timestamps; + uint32_t nrows, ncols, capacity, timestamp_capacity; + uint32_t *boundaries; + uint8_t *needs_sort; + uint32_t seg_count, needs_sort_count; + wl_columnar_memory_reservation_t metadata_admission; + bool metadata_admission_active; + wl_columnar_radix_workspace_t workspace; + bool prepared; + bool consumed; +} wl_columnar_relation_radix_sequence_t; + /* Prepare a zero-initialized or fully destroyed workspace. Live workspaces * are refused unchanged; destroy releases their buffers and admission once. * Partial, empty and sorted ranges are supported. All boundaries are checked @@ -3271,6 +3303,15 @@ wl_columnar_relation_radix_sort_consolidation_with_lease(col_rel_t *rel, uint32_t start_row, uint32_t nrows, const wl_columnar_radix_workspace_t *workspace, wl_columnar_relation_mutation_lease_t *lease); +WL_MUST_CHECK int wl_columnar_relation_radix_sequence_prepare_with_lease( + col_rel_t *rel, const uint32_t *seg_boundaries, uint32_t seg_count, + wl_columnar_relation_radix_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease); +WL_MUST_CHECK int wl_columnar_relation_radix_sequence_execute_with_lease( + wl_columnar_relation_radix_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease); +void wl_columnar_relation_radix_sequence_destroy( + wl_columnar_relation_radix_sequence_t *sequence); /** Stable LSD radix sort of a row-major int64_t buffer by a single key * column. Used by arrangement.c (sarr_build) and lftj.c diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 50dc6d75..320c0133 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -25,6 +25,36 @@ * which has no C11 atomics in its default C mode. Relaxed ordering is * sufficient: the counter only has to hand out distinct values. */ static wl_atomic_u64 wl_next_relation_identity = 1u; +static wl_atomic_u64 wl_next_mutation_set_nonce = 1u; + +#ifdef WL_TEST_MUTATION_SET_HOOK +void +wl_columnar_relation_test_set_mutation_nonce(uint64_t next_nonce) +{ + atomic_store_explicit(&wl_next_mutation_set_nonce, next_nonce, + memory_order_relaxed); +} +#endif + +static int +col_rel_mutation_set_nonce_allocate(uint64_t *nonce) +{ + uint64_t observed; + if (!nonce) + return EINVAL; + observed = atomic_load_explicit(&wl_next_mutation_set_nonce, + memory_order_relaxed); + for (;;) { + if (observed == 0 || observed == UINT64_MAX) + return EOVERFLOW; + uint64_t desired = observed + 1u; + if (atomic_compare_exchange_weak_explicit(&wl_next_mutation_set_nonce, + &observed, desired, memory_order_relaxed, memory_order_relaxed)) { + *nonce = observed; + return 0; + } + } +} #ifdef WL_TEST_APPEND_HOOK wl_columnar_append_transition_hook_t wl_columnar_append_transition_hook; @@ -484,6 +514,7 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, return EOVERFLOW; max_owners = role_count * 2u; if (!set || set->identity || set->roles || set->descriptors || + set->acquisition_nonce || set->owners || set->leases || set->initializations || set->descriptor_count || set->owner_count || set->lease_count || set->initialization_count @@ -525,7 +556,8 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, && roles[i].role_flags != WL_COLUMNAR_RELATION_METADATA_DETACH) || descriptors[i].relation || !col_rel_mutation_writer_inert(&descriptors[i].writer) - || leases[i].identity || leases[i].set || leases[i].relation + || leases[i].identity || leases[i].acquisition_nonce + || leases[i].set || leases[i].relation || leases[i].owner || leases[i].descriptor_slot || leases[i].owner_slot || leases[i].self_owner_slot || leases[i].relation_identity @@ -542,7 +574,12 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, if (owners[i].owner || !col_rel_mutation_writer_inert(&owners[i].writer)) return EINVAL; + uint64_t acquisition_nonce; + rc = col_rel_mutation_set_nonce_allocate(&acquisition_nonce); + if (rc) + return rc; set->identity = (uintptr_t)set; + set->acquisition_nonce = acquisition_nonce; set->roles = roles; set->descriptors = descriptors; set->owners = owners; @@ -616,6 +653,7 @@ col_rel_mutation_set_acquire(wl_columnar_relation_mutation_set_t *set, } wl_columnar_relation_mutation_lease_t *lease = &leases[i]; lease->set = set; + lease->acquisition_nonce = set->acquisition_nonce; lease->relation = relation; lease->owner = owner; lease->role_flags = roles[i].role_flags; @@ -708,11 +746,14 @@ col_rel_mutation_lease_validate( const col_rel_t *expected_relation) { if (!lease || lease->identity != (uintptr_t)lease || !lease->set + || !lease->acquisition_nonce || !expected_relation || lease->relation != expected_relation || lease->detached) return EINVAL; const wl_columnar_relation_mutation_set_t *set = lease->set; - if (set->identity != (uintptr_t)set || !set->roles || !set->leases || + if (set->identity != (uintptr_t)set || !set->acquisition_nonce + || lease->acquisition_nonce != set->acquisition_nonce + || !set->roles || !set->leases || !set->descriptors || !set->owners || set->descriptors_acquired != set->descriptor_count || set->owners_acquired != set->owner_count @@ -10733,6 +10774,248 @@ wl_columnar_radix_lease_preflight(col_rel_t *r, uint32_t start, return 0; } +void +wl_columnar_relation_radix_sequence_destroy( + wl_columnar_relation_radix_sequence_t *sequence) +{ + if (!sequence) + return; + free(sequence->boundaries); + free(sequence->needs_sort); + if (sequence->metadata_admission_active) + (void)wl_columnar_memory_release(&sequence->metadata_admission); + wl_columnar_radix_workspace_destroy(&sequence->workspace); + memset(sequence, 0, sizeof(*sequence)); +} + +static bool +wl_columnar_memory_reservation_inert( + const wl_columnar_memory_reservation_t *reservation) +{ + return reservation && !reservation->governor && !reservation->bytes + && !reservation->replacement_bytes && !reservation->transition_kind + && !atomic_load_explicit(&reservation->owner_bits, + memory_order_relaxed) + && !atomic_load_explicit(&reservation->state, memory_order_relaxed) + && !reservation->identity; +} + +static bool +wl_columnar_radix_workspace_inert(const wl_columnar_radix_workspace_t *w) +{ + return w && !w->perm_a && !w->perm_b && !w->bucket_values + && !w->byte_values && !w->short_values && !w->count16 + && !w->temp_column && !w->insertion_rows && !w->timestamps + && !w->timestamp_capacity && !w->k8_capacity && !w->k16_capacity + && !w->insertion_capacity && !w->insertion_bytes + && !w->prepared_relation && !w->prepared_identity + && !w->prepared_view_generation && !w->prepared_storage_generation + && !w->prepared_type_fingerprint && !w->prepared_start + && !w->prepared_count && !w->prepared_nrows && !w->prepared_ncols + && !w->prepared_capacity && !w->prepared_timestamp_capacity + && !w->prepared_has_timestamps && !w->prepared_has_types + && !w->admission_active + && wl_columnar_memory_reservation_inert(&w->admission); +} + +int +wl_columnar_relation_radix_sequence_prepare_with_lease(col_rel_t *r, + const uint32_t *bounds, uint32_t seg_count, + wl_columnar_relation_radix_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease) +{ + int rc; + size_t boundary_count, boundary_bytes, mask_bytes; + uint64_t extra; + if (!sequence || sequence->identity || sequence->acquisition_nonce + || sequence->set || sequence->lease || sequence->relation + || sequence->owner || sequence->relation_identity + || sequence->relation_generation || sequence->owner_identity + || sequence->owner_generation || sequence->view_generation + || sequence->type_fingerprint || sequence->columns + || sequence->column_types || sequence->timestamps || sequence->nrows + || sequence->ncols || sequence->capacity || sequence->timestamp_capacity + || sequence->boundaries || sequence->needs_sort || sequence->seg_count + || sequence->needs_sort_count || sequence->metadata_admission_active + || !wl_columnar_memory_reservation_inert( + &sequence->metadata_admission) + || sequence->prepared || sequence->consumed + || !wl_columnar_radix_workspace_inert(&sequence->workspace)) + return EINVAL; + if (!r || !bounds || !seg_count) + return EINVAL; + if (seg_count == UINT32_MAX) + return EOVERFLOW; +#if SIZE_MAX <= UINT32_MAX + if (seg_count > SIZE_MAX / sizeof(uint32_t) - 1u) + return EOVERFLOW; +#endif + rc = wl_columnar_radix_lease_preflight(r, 0, 0, lease); + if (rc) + return rc; + boundary_count = (size_t)seg_count + 1u; + if (boundary_count > SIZE_MAX / sizeof(uint32_t)) + return EOVERFLOW; + boundary_bytes = boundary_count * sizeof(uint32_t); + mask_bytes = (size_t)seg_count * sizeof(uint8_t); + if (boundary_bytes > UINT64_MAX - (uint64_t)mask_bytes) + return EOVERFLOW; + extra = (uint64_t)boundary_bytes + (uint64_t)mask_bytes; + for (uint32_t i = 0; i < seg_count; i++) + if (bounds[i] > bounds[i + 1] || bounds[i + 1] > r->nrows) + return EINVAL; + uint32_t needs_sort_count = 0; + for (uint32_t i = 0; i < seg_count; i++) { + uint32_t start = bounds[i]; + uint32_t end = bounds[i + 1]; + bool sorted = end - start <= 1u; + if (!sorted) { + sorted = true; + for (uint32_t row = start + 1u; row < end; row++) + if (col_rel_row_cmp(r, row - 1u, row) > 0) { + sorted = false; + break; + } + } + needs_sort_count += !sorted; + } + if (needs_sort_count + && (r->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID + - needs_sort_count + || (r->col_shared && r->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u))) + return EOVERFLOW; + if (needs_sort_count && r->col_shared && lease->storage_transitioned) + return EINVAL; + if (r->memory_governor && extra) { + wl_columnar_memory_reservation_init(&sequence->metadata_admission); + wl_columnar_memory_admission_status_t status + = wl_columnar_memory_reserve_checked( + wl_columnar_memory_governor_ref_get(r->memory_governor), + extra, &sequence->metadata_admission); + if (status != WL_COLUMNAR_MEMORY_ADMISSION_OK + && status != WL_COLUMNAR_MEMORY_ADMISSION_ADVISORY) { + if (status == WL_COLUMNAR_MEMORY_ADMISSION_DENIED) + r->memory_budget_denial_pending = true; + wl_columnar_relation_radix_sequence_destroy(sequence); + return status == WL_COLUMNAR_MEMORY_ADMISSION_DENIED ? ENOMEM + : status == WL_COLUMNAR_MEMORY_ADMISSION_OVERFLOW + ? EOVERFLOW : EINVAL; + } + sequence->metadata_admission_active = true; + } + sequence->boundaries = malloc(boundary_bytes); + sequence->needs_sort = calloc(seg_count, sizeof(*sequence->needs_sort)); + if (!sequence->boundaries || !sequence->needs_sort) { + wl_columnar_relation_radix_sequence_destroy(sequence); + return ENOMEM; + } + memcpy(sequence->boundaries, bounds, boundary_bytes); + sequence->seg_count = seg_count; + sequence->needs_sort_count = needs_sort_count; + for (uint32_t i = 0; i < seg_count; i++) { + uint32_t start = sequence->boundaries[i]; + uint32_t end = sequence->boundaries[i + 1]; + if (end - start <= 1u) + continue; + for (uint32_t row = start + 1u; row < end; row++) + if (col_rel_row_cmp(r, row - 1u, row) > 0) { + sequence->needs_sort[i] = 1u; + break; + } + } + rc = wl_columnar_relation_radix_workspace_prepare(r, + sequence->boundaries, seg_count, 0, &sequence->workspace, true); + if (rc) { + wl_columnar_relation_radix_sequence_destroy(sequence); + return rc; + } + sequence->set = lease->set; + sequence->lease = lease; + sequence->relation = r; + sequence->owner = lease->owner; + sequence->acquisition_nonce = lease->acquisition_nonce; + sequence->relation_identity = r->relation_identity; + sequence->relation_generation = r->storage_generation; + sequence->owner_identity = lease->owner_identity; + sequence->owner_generation = lease->owner_generation; + sequence->view_generation = r->view_generation; + sequence->type_fingerprint = wl_columnar_radix_type_fingerprint(r); + sequence->columns = (const int64_t *const *)r->columns; + sequence->column_types = r->column_types; + sequence->timestamps = r->timestamps; + sequence->nrows = r->nrows; + sequence->ncols = r->ncols; + sequence->capacity = r->capacity; + sequence->timestamp_capacity = r->timestamp_capacity; + sequence->identity = (uintptr_t)sequence; + sequence->prepared = true; + return 0; +} + +int +wl_columnar_relation_radix_sequence_execute_with_lease( + wl_columnar_relation_radix_sequence_t *sequence, + wl_columnar_relation_mutation_lease_t *lease) +{ + if (!sequence || sequence->identity != (uintptr_t)sequence + || !sequence->prepared || sequence->consumed || !lease + || !sequence->set || !lease->set + || sequence->lease != lease || sequence->set != lease->set + || sequence->acquisition_nonce != lease->acquisition_nonce + || sequence->relation != lease->relation + || lease->set->acquisition_nonce != sequence->acquisition_nonce + || col_rel_mutation_lease_validate(lease, sequence->relation)) + return EINVAL; + col_rel_t *r = sequence->relation; + if (r->relation_identity != sequence->relation_identity + || r->storage_generation != sequence->relation_generation + || r->view_generation != sequence->view_generation + || lease->owner != sequence->owner + || lease->owner_identity != sequence->owner_identity + || lease->owner_generation != sequence->owner_generation + || r->columns != (int64_t **)sequence->columns + || r->column_types != sequence->column_types + || r->timestamps != sequence->timestamps + || r->nrows != sequence->nrows || r->ncols != sequence->ncols + || r->capacity != sequence->capacity + || r->timestamp_capacity != sequence->timestamp_capacity + || wl_columnar_radix_type_fingerprint(r) != sequence->type_fingerprint) + return EINVAL; + if (!sequence->needs_sort_count) { + sequence->consumed = true; + return 0; + } + for (uint32_t i = 0; i < sequence->seg_count; i++) { + if (!sequence->needs_sort[i]) + continue; + uint32_t count = sequence->boundaries[i + 1] + - sequence->boundaries[i]; + int rc = wl_columnar_relation_radix_workspace_validate(r, count, + &sequence->workspace); + if (rc) + return rc; + } + if (r->col_shared) { + int rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, + lease); + if (rc) + return rc; + } + sequence->consumed = true; + for (uint32_t i = 0; i < sequence->seg_count; i++) { + if (!sequence->needs_sort[i]) + continue; + int rc = wl_columnar_radix_rows_prepared(r, + sequence->boundaries[i], sequence->boundaries[i + 1] + - sequence->boundaries[i], &sequence->workspace, + r->timestamps ? sequence->workspace.timestamps : NULL); + if (rc) + abort(); + } + return 0; +} + int wl_columnar_relation_radix_workspace_prepare_with_lease(col_rel_t *r, uint32_t start, uint32_t count, From ff97278099e2804c4bd90539bb2536cdade0d0aa Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 01:29:34 +0900 Subject: [PATCH 08/15] Mark prepared radix payload publication --- tests/test_relation_generations.c | 85 ++++++++++++++++++++++++++++++- wirelog/columnar/relation.c | 3 +- 2 files changed, 86 insertions(+), 2 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 5988311d..7ccf0bd6 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -6639,6 +6639,15 @@ test_leased_radix_sequence(void) && sequence.identity == 0, "radix sequence preflights the exact view event count"); alias->view_generation = sequence_view; + uint64_t sequence_storage = alias->storage_generation; + alias->storage_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + held.lease.relation_generation = alias->storage_generation; + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + bounds, 3, &sequence, &held.lease) == EOVERFLOW + && sequence.identity == 0, + "radix sequence preflights shared storage epoch exhaustion"); + alias->storage_generation = sequence_storage; + held.lease.relation_generation = sequence_storage; held.lease.storage_transitioned = true; CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, bounds, 3, &sequence, &held.lease) == EINVAL @@ -6745,6 +6754,20 @@ test_leased_radix_sequence(void) && col_rel_storage_alias_borrow_count(root) == 1, "zero-sort sequence consumes without detaching or advancing epochs"); wl_columnar_relation_radix_sequence_destroy(&sequence); + memset(&sequence, 0, sizeof(sequence)); + uint32_t sorted_multirow_bounds[] = { 0, 3 }; + CHECK(wl_columnar_relation_radix_sequence_prepare_with_lease(alias, + sorted_multirow_bounds, 1, &sequence, &held.lease) == 0, + "radix multirow sorted alias prepare"); + view_before = alias->view_generation; + storage_before = alias->storage_generation; + CHECK(wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 && alias->view_generation == view_before + && alias->storage_generation == storage_before + && alias->storage_owner == root + && col_rel_storage_alias_borrow_count(root) == 1, + "multirow sorted alias does not detach or advance generations"); + wl_columnar_relation_radix_sequence_destroy(&sequence); CHECK(col_rel_mutation_set_finish(&held.set, true) == 0, "radix zero-sort finish"); memset(&sequence, 0, sizeof(sequence)); @@ -6792,6 +6815,8 @@ test_leased_radix_sequence(void) tiny_resolution.budget_bytes - already_reserved - 1u, &blocker) == WL_COLUMNAR_MEMORY_ADMISSION_OK, "radix sequence reserve all but one governor byte"); + uint64_t reserved_before_refusal = wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(tiny_ref)); memset(&held, 0, sizeof(held)); memset(&sequence, 0, sizeof(sequence)); uint32_t admission_bounds[] = { 0, 8 }; @@ -6803,7 +6828,10 @@ test_leased_radix_sequence(void) && root->memory_budget_denial_pending && mutation_payload_unchanged(root, &before_denial) && root->view_generation == before_denial.view_generation - && root->storage_generation == before_denial.storage_generation, + && root->storage_generation == before_denial.storage_generation + && wl_columnar_memory_reserved( + wl_columnar_memory_governor_ref_get(tiny_ref)) + == reserved_before_refusal, "radix sequence budget denial precedes mutation"); CHECK(col_rel_mutation_set_finish(&held.set, false) == 0, "radix sequence denial finish"); @@ -6904,6 +6932,60 @@ test_leased_radix_sequence(void) cleanup_relations(); } +static void +test_radix_sequence_lazy_publication(void) +{ + col_rel_t *rel = new_relation(); + for (int64_t value = 4; value > 0; value--) + CHECK(col_rel_append_row(rel, &value) == 0, + "lazy radix sequence seed"); + rel->storage_owner = NULL; + rel->storage_owner_identity = 0; + rel->storage_owner_generation = 0; + radix_test_lease_t held = { 0 }; + wl_columnar_relation_radix_sequence_t sequence = { 0 }; + uint32_t bounds[] = { 0, 4 }; + CHECK(radix_test_acquire(rel, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(rel, + bounds, 1, &sequence, &held.lease) == 0 + && wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 && held.set.published, + "canonical radix sort marks mutation-set publication before moving rows"); + uint64_t identity = rel->storage_owner_identity; + uint64_t generation = rel->storage_owner_generation; + CHECK(col_rel_mutation_set_finish(&held.set, false) == 0 + && rel->storage_owner == rel + && rel->storage_owner_identity == identity + && rel->storage_owner_generation == generation, + "failed finish retains lazy owner metadata after radix publication"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + cleanup_relations(); + + rel = new_relation(); + for (int64_t value = 1; value <= 3; value++) + CHECK(col_rel_append_row(rel, &value) == 0, + "zero-sort lazy radix seed"); + rel->storage_owner = NULL; + rel->storage_owner_identity = 0; + rel->storage_owner_generation = 0; + memset(&held, 0, sizeof(held)); + memset(&sequence, 0, sizeof(sequence)); + uint32_t sorted_bounds[] = { 0, 3 }; + CHECK(radix_test_acquire(rel, &held) == 0 + && wl_columnar_relation_radix_sequence_prepare_with_lease(rel, + sorted_bounds, 1, &sequence, &held.lease) == 0 + && wl_columnar_relation_radix_sequence_execute_with_lease(&sequence, + &held.lease) == 0 && !held.set.published, + "zero-sort sequence leaves mutation set unpublished"); + CHECK(col_rel_mutation_set_finish(&held.set, false) == 0 + && rel->storage_owner == NULL + && rel->storage_owner_identity == 0 + && rel->storage_owner_generation == 0, + "zero-sort failed finish rolls lazy owner initialization back"); + wl_columnar_relation_radix_sequence_destroy(&sequence); + cleanup_relations(); +} + static void test_radix_descriptor_admission(void) { @@ -7499,6 +7581,7 @@ main(void) test_radix_descriptor_admission(); test_leased_radix_provenance(); test_leased_radix_sequence(); + test_radix_sequence_lazy_publication(); test_leased_radix_kernels(); test_public_append_mutation_admission(); test_append_overlapping_input(false); diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 320c0133..0fa85597 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -11002,10 +11002,11 @@ wl_columnar_relation_radix_sequence_execute_with_lease( if (rc) return rc; } - sequence->consumed = true; for (uint32_t i = 0; i < sequence->seg_count; i++) { if (!sequence->needs_sort[i]) continue; + lease->set->published = true; + sequence->consumed = true; int rc = wl_columnar_radix_rows_prepared(r, sequence->boundaries[i], sequence->boundaries[i + 1] - sequence->boundaries[i], &sequence->workspace, From bad702a22da5228c7421e68f0609116f1b0080af Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 01:48:48 +0900 Subject: [PATCH 09/15] Lease consolidation merge mutations --- tests/test_consolidate_kway_merge.c | 224 ++++++++++++++++++++++++++- wirelog/columnar/internal.h | 3 + wirelog/columnar/merge.c | 232 ++++++++++++++++++---------- wirelog/columnar/relation.c | 26 ++++ 4 files changed, 402 insertions(+), 83 deletions(-) diff --git a/tests/test_consolidate_kway_merge.c b/tests/test_consolidate_kway_merge.c index 9b5cab8e..8be18350 100644 --- a/tests/test_consolidate_kway_merge.c +++ b/tests/test_consolidate_kway_merge.c @@ -1662,6 +1662,50 @@ test_hash_heuristic_fallback_succeeds(void) PASS(); } +static void +test_hash_fallback_releases_credit_before_radix_path(void) +{ + const uint32_t row_count = 10001; + const uint32_t boundaries[] = { 0, row_count }; + col_rel_t *rel = test_rel_alloc(1); + wl_columnar_memory_governor_ref_t *ref = NULL; + TEST("hash heuristic releases maximum credit before radix fallback"); + if (!rel) + FAIL("relation allocation failed"); + for (uint32_t i = 0; i < row_count; i++) { + int64_t value = (int64_t)(row_count - i); + if (test_rel_append_row(rel, &value) != 0) { + test_rel_free(rel); + FAIL("failed to append unique hash fixture"); + } + } + ref = test_consolidate_governor_create(UINT64_C(1) << 30); + if (!ref || col_rel_attach_memory_governor(rel, ref) != 0) { + if (ref) wl_columnar_memory_governor_ref_release(ref); + test_rel_free(rel); + FAIL("failed to attach hash fallback governor"); + } + wl_columnar_memory_governor_t *governor = + wl_columnar_memory_governor_ref_get(ref); + uint64_t baseline = wl_columnar_memory_reserved(governor); + /* At this row count the hash reservation's maximum rehash transient is + * about 625 KiB. The ordinary one-segment radix path needs substantially + * less; this cap admits either phase but cannot admit both concurrently. */ + atomic_store_explicit(&governor->usable_bytes, baseline + 700u * 1024u, + memory_order_seq_cst); + int rc = col_op_consolidate_kway_merge(rel, boundaries, 1); + if (rc != 0 || rel->nrows != row_count + || !test_rel_is_sorted_unique(rel) + || wl_columnar_memory_reserved(governor) != baseline) { + test_rel_free(rel); + wl_columnar_memory_governor_ref_release(ref); + FAIL("fallback failed or retained hash/radix admission credit"); + } + test_rel_free(rel); + wl_columnar_memory_governor_ref_release(ref); + PASS(); +} + static void test_k16_workspace_is_preallocated(void) { @@ -1777,7 +1821,10 @@ test_float_insertion_workspace(void) * The last one is the size assertion, and the memory governor is the * seam that makes it observable: merged_alloc_bytes is admitted as part * of the consolidation scratch, so a budget covering only the two - * segment arrays must be refused while one extra int64_t admits. A + * segment arrays must be refused while one extra int64_t admits. The + * prepared sequence also keeps its copied 3-boundary array and 2-byte + * sort mask admitted at the same time, so the exact admission includes 14 + * bytes for that metadata. A * mutation that drops the non-zero guarantee is caught there -- it is * not caught by the allocation hook, which receives only a site name, * nor by malloc(0) itself, which returns non-NULL on glibc. @@ -1807,6 +1854,173 @@ zero_arity_fixture(void) return rel; } +static void +test_leased_merge_alias_epoch_paths(void) +{ + const uint32_t boundaries[] = { 0, 2, 4 }; + const int64_t fixtures[][4] = { + { 1, 2, 3, 4 }, /* no segment needs sorting */ + { 1, 3, 4, 2 }, /* first segment sorted, second unsorted */ + }; + TEST("leased merge privatizes sorted aliases and advances exact epochs"); + for (size_t mode = 0; mode < sizeof(fixtures) / sizeof(fixtures[0]); + mode++) { + col_rel_t *source = test_rel_alloc(1); + col_rel_t *view = test_rel_alloc(1); + if (!source || !view) { + test_rel_free(view); + test_rel_free(source); + FAIL("alias fixture allocation failed"); + } + for (uint32_t i = 0; i < 4; i++) { + if (test_rel_append_row(source, &fixtures[mode][i]) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("alias fixture append failed"); + } + } + if (col_rel_install_shared_view(view, source) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("alias installation failed"); + } + uint64_t view_generation = view->view_generation; + uint64_t storage_generation = view->storage_generation; + if (col_op_consolidate_kway_merge(view, boundaries, 2) != 0 + || view->nrows != 4 || view->col_shared + || view->storage_owner != view + || view->storage_generation != storage_generation + 1u + || view->view_generation != view_generation + + (mode == 0 ? 1u : 2u) + || col_rel_storage_alias_borrow_count(source) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("alias detach/sort/publish epochs differ from the path"); + } + for (uint32_t i = 0; i < 4; i++) { + if (view->columns[0][i] != (int64_t)i + 1 + || source->columns[0][i] != fixtures[mode][i]) { + test_rel_free(view); + test_rel_free(source); + FAIL("alias result or sibling storage changed"); + } + } + test_rel_free(view); + test_rel_free(source); + } + PASS(); +} + +static void +test_merge_epoch_headroom_preflight(void) +{ + const uint32_t boundaries[] = { 0, 2, 4 }; + const int64_t values[] = { 2, 1, 4, 3 }; + TEST("merge preflights all view events and alias storage epoch"); + col_rel_t *rel = test_rel_alloc(1); + ASSERT(rel != NULL, "relation allocation failed"); + for (uint32_t i = 0; i < 4; i++) { + if (test_rel_append_row(rel, &values[i]) != 0) { + test_rel_free(rel); + FAIL("failed to append epoch fixture"); + } + } + uint64_t view_generation = rel->view_generation; + /* Two segment sorts fit, but the final publication is the third event. */ + rel->view_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 3u; + int rc = col_op_consolidate_kway_merge(rel, boundaries, 2); + if (rc != EOVERFLOW || rel->nrows != 4 + || rel->view_generation != WL_COLUMNAR_REL_GENERATION_INVALID - 3u + || rel->columns[0][0] != values[0] + || rel->columns[0][1] != values[1]) { + rel->view_generation = view_generation; + test_rel_free(rel); + FAIL("sort plus publication headroom must be checked before sorting"); + } + rel->view_generation = view_generation; + test_rel_free(rel); + + col_rel_t *source = test_rel_alloc(1); + col_rel_t *view = test_rel_alloc(1); + if (!source || !view) { + test_rel_free(view); + test_rel_free(source); + FAIL("shared epoch fixture allocation failed"); + } + for (uint32_t i = 0; i < 4; i++) { + if (test_rel_append_row(source, &values[i]) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("failed to append shared epoch fixture"); + } + } + if (col_rel_install_shared_view(view, source) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("failed to install epoch alias"); + } + int64_t *old_column = view->columns[0]; + uint64_t old_storage_generation = view->storage_generation; + view->storage_generation = WL_COLUMNAR_REL_GENERATION_INVALID - 1u; + rc = col_op_consolidate_kway_merge(view, boundaries, 2); + bool unchanged = rc == EOVERFLOW && view->nrows == 4 + && view->columns[0] == old_column && view->col_shared + && view->storage_generation + == WL_COLUMNAR_REL_GENERATION_INVALID - 1u + && col_rel_storage_alias_borrow_count(source) == 1 + && source->columns[0] == old_column; + view->storage_generation = old_storage_generation; + test_rel_free(view); + test_rel_free(source); + if (!unchanged) + FAIL("alias privatization headroom must precede COW or publication"); + PASS(); +} + +static void +test_nullary_alias_large_hash_merge(void) +{ + const uint32_t row_count = 10001; + const uint32_t boundaries[] = { 0, row_count }; + col_rel_t *source = test_rel_alloc(0); + col_rel_t *view = test_rel_alloc(0); + TEST("large nullary hash merge releases borrowed owner binding"); + if (!source || !view) { + test_rel_free(view); + test_rel_free(source); + FAIL("nullary fixture allocation failed"); + } + for (uint32_t i = 0; i < row_count; i++) { + int64_t empty_tuple = 0; + if (col_rel_append_row(source, &empty_tuple) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("nullary fixture append failed"); + } + } + if (col_rel_install_shared_view(view, source) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("nullary alias installation failed"); + } + uint64_t view_generation = view->view_generation; + uint64_t storage_generation = view->storage_generation; + if (view->col_shared != NULL + || col_rel_storage_alias_borrow_count(source) != 1 + || col_op_consolidate_kway_merge(view, boundaries, 1) != 0 + || view->nrows != 1 || view->storage_owner != view + || view->storage_generation != storage_generation + 1u + || view->view_generation != view_generation + 1u + || col_rel_storage_alias_borrow_count(source) != 0) { + test_rel_free(view); + test_rel_free(source); + FAIL("nullary hash path did not privatize and publish exactly once"); + } + test_rel_free(view); + test_rel_free(source); + PASS(); +} + static void test_zero_arity_merge_output_is_never_zero_sized(void) { @@ -1815,6 +2029,8 @@ test_zero_arity_merge_output_is_never_zero_sized(void) * relation adds no radix workspace, so this is the whole scratch * requirement apart from the merge output itself. */ const uint64_t segment_scratch = 2u * 2u * sizeof(uint32_t); + const uint64_t sequence_metadata = 3u * sizeof(uint32_t) + + 2u * sizeof(uint8_t); wl_columnar_memory_governor_ref_t *ref = NULL; col_rel_t *rel = NULL; int rc; @@ -1862,7 +2078,7 @@ test_zero_arity_merge_output_is_never_zero_sized(void) * the identical empty tuples to a single row. */ rel = zero_arity_fixture(); ref = test_consolidate_governor_create(descriptor_bytes - + segment_scratch + sizeof(int64_t)); + + segment_scratch + sizeof(int64_t) + sequence_metadata); if (!rel || !ref || col_rel_attach_memory_governor(rel, ref) != 0) { if (ref) wl_columnar_memory_governor_ref_release(ref); @@ -3053,6 +3269,9 @@ main(void) test_shared_view_sorted_dedup_is_copy_on_write(); test_shared_view_merge_scatter_is_copy_on_write(); test_shared_view_merge_oom_preserves_view_state(); + test_leased_merge_alias_epoch_paths(); + test_merge_epoch_headroom_preflight(); + test_nullary_alias_large_hash_merge(); test_merge_output_oom_is_transactional(); test_zero_arity_merge_output_is_never_zero_sized(); test_merge_heap_oom_is_transactional(); @@ -3061,6 +3280,7 @@ main(void) test_hash_allocation_oom_is_not_fallback(); test_hash_float_signed_zero_lexicographic_order(); test_hash_heuristic_fallback_succeeds(); + test_hash_fallback_releases_credit_before_radix_path(); test_k16_workspace_is_preallocated(); test_float_insertion_workspace(); for (unsigned ownership = 0; ownership < 3; ownership++) { diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index 9f62c35d..b68a17fd 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2739,6 +2739,9 @@ int col_rel_reserve_merge_grid_with_lease(col_rel_t *r, uint32_t capacity, wl_columnar_relation_mutation_lease_t *lease); int col_rel_cow_unshare_with_lease(col_rel_t *r, wl_columnar_relation_mutation_lease_t *lease); +WL_MUST_CHECK int +wl_columnar_relation_privatize_shared_view_with_lease(col_rel_t *r, + wl_columnar_relation_mutation_lease_t *lease); int col_rel_append_row_with_lease(col_rel_t *r, const int64_t *row, wl_columnar_relation_mutation_lease_t *lease); int col_rel_reserve_rows_with_lease(col_rel_t *r, uint32_t additional, diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index c919b300..5dcefda2 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -646,16 +646,20 @@ col_op_consolidate_hash_rows_sort(const col_rel_t *rel, int64_t *rows, static int col_op_consolidate_hash_dedup(col_rel_t *rel, - const wl_columnar_source_access_writer_t *writer, - bool *alias_release_pending) + wl_columnar_relation_mutation_lease_t *lease) { uint32_t nc = rel->ncols; uint32_t nr = rel->nrows; - size_t row_bytes = (size_t)nc * sizeof(int64_t); + size_t row_bytes; + if (!col_op_consolidate_size_multiply(nc, sizeof(int64_t), &row_bytes)) + return EOVERFLOW; + size_t allocated_row_bytes = row_bytes ? row_bytes : sizeof(int64_t); uint64_t scratch_bytes = 0; uint64_t bytes = 0; uint64_t max_ht_cap = 8192; uint64_t max_uniq_cap = 4096; + uint64_t max_hash_live_cap; + uint64_t max_uniq_live_cap; wl_columnar_memory_reservation_t reservation; while (max_ht_cap <= (uint64_t)nr * 2u @@ -663,13 +667,22 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, max_ht_cap *= 2u; while (max_uniq_cap < nr && max_uniq_cap <= UINT32_MAX / 2u) max_uniq_cap *= 2u; - if (max_uniq_cap < nr - || !wl_columnar_memory_size_mul(max_ht_cap, row_bytes, &bytes) + if (max_uniq_cap < nr) + return EOVERFLOW; + /* Rehash allocates the doubled table while the old table remains live. */ + max_hash_live_cap = max_ht_cap + + (max_ht_cap > 8192u ? max_ht_cap / 2u : 0u); + max_uniq_live_cap = max_uniq_cap + + (max_uniq_cap > 4096u ? max_uniq_cap / 2u : 0u); + if (max_hash_live_cap < max_ht_cap || max_uniq_live_cap < max_uniq_cap + || !wl_columnar_memory_size_mul(max_hash_live_cap, + allocated_row_bytes, &bytes) || !wl_columnar_memory_size_add(scratch_bytes, bytes, &scratch_bytes) - || !wl_columnar_memory_size_add(scratch_bytes, max_ht_cap, + || !wl_columnar_memory_size_add(scratch_bytes, max_hash_live_cap, &scratch_bytes) - || !wl_columnar_memory_size_mul(max_uniq_cap, row_bytes, &bytes) + || !wl_columnar_memory_size_mul(max_uniq_live_cap, + allocated_row_bytes, &bytes) || !wl_columnar_memory_size_add(scratch_bytes, bytes, &scratch_bytes) || (nc > COL_STACK_MAX @@ -677,6 +690,8 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, || !wl_columnar_memory_size_add(scratch_bytes, bytes, &scratch_bytes)))) return EOVERFLOW; + if (scratch_bytes > SIZE_MAX) + return EOVERFLOW; int admission_rc = col_rel_merge_scratch_reserve(rel, scratch_bytes, &reservation); if (admission_rc != 0) @@ -687,7 +702,7 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, uint32_t ht_cap = 8192; uint32_t ht_mask = ht_cap - 1; int64_t *ht_vals = (int64_t *)col_op_consolidate_malloc( - (size_t)ht_cap * row_bytes, "hash_table"); + (size_t)ht_cap * allocated_row_bytes, "hash_table"); uint8_t *ht_used = (uint8_t *)col_op_consolidate_calloc(ht_cap, 1, "hash_table_used"); if (!ht_vals || !ht_used) { @@ -701,7 +716,7 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, uint32_t uniq_cap = 4096; uint32_t uniq_count = 0; int64_t *uniq_buf = (int64_t *)col_op_consolidate_malloc( - (size_t)uniq_cap * row_bytes, "hash_unique_rows"); + (size_t)uniq_cap * allocated_row_bytes, "hash_unique_rows"); if (!uniq_buf) { free(ht_vals); free(ht_used); @@ -756,7 +771,8 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, uint32_t new_cap = ht_cap * 2; uint32_t new_mask = new_cap - 1; int64_t *new_vals = (int64_t *)col_op_consolidate_malloc( - (size_t)new_cap * row_bytes, "hash_rehash_rows"); + (size_t)new_cap * allocated_row_bytes, + "hash_rehash_rows"); uint8_t *new_used = (uint8_t *)col_op_consolidate_calloc( new_cap, 1, "hash_rehash_used"); if (!new_vals || !new_used) { @@ -804,7 +820,8 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, if (uniq_count >= uniq_cap) { uniq_cap *= 2; int64_t *nb = (int64_t *)col_op_consolidate_realloc(uniq_buf, - (size_t)uniq_cap * row_bytes, "hash_unique_grow"); + (size_t)uniq_cap * allocated_row_bytes, + "hash_unique_grow"); if (!nb) { if (rb != _rb) free(rb); free(ht_vals); @@ -832,17 +849,18 @@ col_op_consolidate_hash_dedup(col_rel_t *rel, /* Detach only after hash/sort staging is complete. All emitted rows * came from the validated source relation, so publish has no expected * validation failure and performs no allocation. */ - if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, writer); + if (lease->owner != rel) { + int cow_rc = wl_columnar_relation_privatize_shared_view_with_lease( + rel, lease); if (cow_rc != 0) { free(uniq_buf); HASH_RELEASE_SCRATCH(); return cow_rc; } - *alias_release_pending = true; } /* Publish the complete sorted result only after all fallible work. */ + lease->set->published = true; for (uint32_t r = 0; r < uniq_count; r++) col_rel_row_copy_in_raw(rel, r, uniq_buf + (size_t)r * nc); rel->nrows = uniq_count; @@ -885,8 +903,7 @@ wl_columnar_merge_heap_compare(const col_rel_t *rel, uint32_t left, static int col_op_consolidate_kway_merge_impl(col_rel_t *rel, const uint32_t *seg_boundaries, uint32_t seg_count, - const wl_columnar_source_access_writer_t *writer, - bool *alias_release_pending) + wl_columnar_relation_mutation_lease_t *lease) { typedef struct { uint32_t seg; /* segment index */ @@ -899,8 +916,6 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, uint32_t nc = rel->ncols; uint32_t nr = rel->nrows; - if (nr <= 1) - return 0; if (!seg_boundaries || seg_count == 0 || seg_boundaries[0] != 0 || seg_boundaries[seg_count] != nr) return EINVAL; @@ -909,15 +924,17 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, || seg_boundaries[s + 1] > nr) return EINVAL; } + if (nr <= 1) + return 0; size_t segment_bytes; size_t heap_bytes; size_t row_bytes = 0; size_t merged_bytes = 0; size_t merged_alloc_bytes = 0; size_t timestamp_bytes = 0; - size_t segment_total_bytes; - size_t additional_scratch_bytes; - wl_columnar_radix_workspace_t sort_workspace = { 0 }; + size_t aggregate_scratch_bytes; + wl_columnar_memory_reservation_t merge_reservation; + wl_columnar_relation_radix_sequence_t sort_sequence = { 0 }; uint32_t *seg_starts = NULL; uint32_t *seg_ends = NULL; int64_t *merged = NULL; @@ -926,15 +943,17 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, heap_entry_t *heap = stack_heap; heap_entry_t *heap_storage = NULL; int result = 0; + wl_columnar_memory_reservation_init(&merge_reservation); if (!col_op_consolidate_size_multiply(seg_count, sizeof(uint32_t), &segment_bytes) || !col_op_consolidate_size_multiply(seg_count, sizeof(heap_entry_t), &heap_bytes)) return EOVERFLOW; + size_t segment_total_bytes; if (!col_op_consolidate_size_multiply(segment_bytes, 2u, &segment_total_bytes)) return EOVERFLOW; - additional_scratch_bytes = segment_total_bytes; + aggregate_scratch_bytes = segment_total_bytes; if (seg_count >= 2) { /* Keep one element of storage for zero-arity relations: malloc(0) * is permitted to return NULL, which would otherwise look like @@ -946,21 +965,24 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, || !col_op_consolidate_size_add(merged_bytes, sizeof(int64_t), &merged_alloc_bytes)) return EOVERFLOW; - if (!col_op_consolidate_size_add(additional_scratch_bytes, - merged_alloc_bytes, &additional_scratch_bytes)) + if (!col_op_consolidate_size_add(aggregate_scratch_bytes, + merged_alloc_bytes, &aggregate_scratch_bytes)) return EOVERFLOW; } if (rel->timestamps && seg_count >= 2) { if (!col_op_consolidate_size_multiply(nr, sizeof(*merged_timestamps), ×tamp_bytes) - || !col_op_consolidate_size_add(additional_scratch_bytes, - timestamp_bytes, &additional_scratch_bytes)) + || !col_op_consolidate_size_add(aggregate_scratch_bytes, + timestamp_bytes, &aggregate_scratch_bytes)) return EOVERFLOW; } - if (seg_count >= 3 - && !col_op_consolidate_size_add(additional_scratch_bytes, heap_bytes, - &additional_scratch_bytes)) - return EOVERFLOW; + if (seg_count >= 3) { + size_t total; + if (!col_op_consolidate_size_add(aggregate_scratch_bytes, + heap_bytes, &total)) + return EOVERFLOW; + aggregate_scratch_bytes = total; + } /* Hash-based dedup for large datasets (#369): O(N) scan + O(U log U) sort * where U is the unique count. When U << N (common in recursive Datalog @@ -969,8 +991,7 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, * including signed/zero multiplicity. The hash shortcut does not carry * that provenance, so retain stable segment sorting for these inputs. */ if (nr > 10000 && !rel->timestamps) { - int rc = col_op_consolidate_hash_dedup(rel, writer, - alias_release_pending); + int rc = col_op_consolidate_hash_dedup(rel, lease); if (rc == 0) return 0; if (rc != -1) @@ -978,11 +999,27 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, /* The unique-count heuristic requests the ordinary sort+merge path. */ } - int workspace_rc = wl_columnar_radix_workspace_prepare(rel, - seg_boundaries, seg_count, additional_scratch_bytes, - &sort_workspace); + int workspace_rc = wl_columnar_relation_radix_sequence_prepare_with_lease( + rel, seg_boundaries, seg_count, &sort_sequence, lease); if (workspace_rc != 0) return workspace_rc; + uint64_t publication_events = (uint64_t)sort_sequence.needs_sort_count + 1u; + if (rel->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID + - publication_events) { + result = EOVERFLOW; + goto cleanup; + } + if (lease->owner != rel && rel->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u) { + result = EOVERFLOW; + goto cleanup; + } + int admission_rc = col_rel_merge_scratch_reserve(rel, + aggregate_scratch_bytes, &merge_reservation); + if (admission_rc != 0) { + result = admission_rc; + goto cleanup; + } /* Allocate every buffer that can fail before the first in-place sort or * dedup mutation. The workspace reservation includes these buffers. */ @@ -1008,22 +1045,21 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, goto cleanup; } - /* Every buffer allocation is now complete. A shared view may detach - * before the first segment mutation; later sorting uses only workspace. */ - if (rel->col_shared) { - int cow_rc = col_rel_cow_unshare_legacy_with_source_writer(rel, writer); - if (cow_rc != 0) { - result = cow_rc; + /* Every buffer and reservation is live before the first mutation. */ + result = wl_columnar_relation_radix_sequence_execute_with_lease( + &sort_sequence, lease); + if (result != 0) + goto cleanup; + if (lease->owner != rel && !lease->detached) { + result = wl_columnar_relation_privatize_shared_view_with_lease(rel, + lease); + if (result != 0) goto cleanup; - } - *alias_release_pending = true; } - /* Sort each segment in-place using radix sort. - * Optimization (#369): skip sort for already-sorted segments (e.g., - * from consolidated IDB reads). Also dedup within each segment after - * sort to reduce merge input. Track per-segment unique counts. */ + /* Sort is complete. Deduplicate each segment to reduce merge input. */ + lease->set->published = true; for (uint32_t s = 0; s < seg_count; s++) { uint32_t start = seg_boundaries[s]; uint32_t end = seg_boundaries[s + 1]; @@ -1031,24 +1067,6 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, seg_starts[s] = start; if (count > 1) { - /* Quick sorted-check: bail on first out-of-order pair */ - bool already_sorted = true; - for (uint32_t r = start + 1; r < end; r++) { - if (col_rel_row_cmp(rel, r - 1, r) > 0) { - already_sorted = false; - break; - } - } - if (!already_sorted) { - int sort_rc = - wl_columnar_relation_radix_sort_with_workspace(rel, - start, count, writer, &sort_workspace); - if (sort_rc != 0) { - result = sort_rc; - goto cleanup; - } - } - /* Intra-segment dedup: compact unique rows to reduce merge */ uint32_t out_r = start + 1; for (uint32_t r = start + 1; r < end; r++) { @@ -1166,7 +1184,8 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, free(heap_storage); free(seg_starts); free(seg_ends); - wl_columnar_radix_workspace_destroy(&sort_workspace); + wl_columnar_relation_radix_sequence_destroy(&sort_sequence); + col_rel_merge_scratch_release(&merge_reservation); return result; } @@ -1175,53 +1194,104 @@ wl_columnar_merge_consolidate_checked(col_rel_t *rel, const uint32_t *seg_boundaries, uint32_t seg_count, bool finalize, bool *out_published) { - wl_columnar_source_access_writer_t writer = { 0 }; + wl_columnar_relation_mutation_set_t set = { 0 }; + wl_columnar_relation_mutation_role_t role = { 0 }; + wl_columnar_relation_mutation_descriptor_t descriptor = { 0 }; + wl_columnar_relation_mutation_owner_t owners[2] = { 0 }; + wl_columnar_relation_mutation_lease_t lease = { 0 }; + wl_columnar_relation_mutation_initialization_t initialization = { 0 }; col_rel_t *owner = NULL; - bool alias_release_pending = false; int rc; if (out_published) *out_published = false; if (!rel) return EINVAL; - rc = col_rel_source_writer_acquire(rel, &writer); + role.relation = rel; + role.role_flags = WL_COLUMNAR_RELATION_PAYLOAD_MUTATION; + rc = col_rel_mutation_set_acquire(&set, &role, 1, &descriptor, 1, + owners, 2, &lease, 1, &initialization, 1); if (rc != 0) return rc; rel->memory_budget_denial_pending = false; if (!wl_columnar_relation_float_values_valid(rel)) { rc = EINVAL; - goto release_writer; + goto finish; + } + if (rel->nrows > rel->capacity + || (rel->timestamps + && rel->timestamp_capacity < rel->nrows) + || (rel->ncols && !rel->columns)) { + rc = EINVAL; + goto finish; } + for (uint32_t c = 0; c < rel->ncols; c++) + if (!rel->columns[c]) { + rc = EINVAL; + goto finish; + } + if (!seg_boundaries || !seg_count || seg_count == UINT32_MAX) { + rc = seg_count == UINT32_MAX ? EOVERFLOW : EINVAL; + goto finish; + } + if (seg_boundaries[0] != 0 + || seg_boundaries[seg_count] != rel->nrows) { + rc = EINVAL; + goto finish; + } + for (uint32_t s = 0; s < seg_count; s++) + if (seg_boundaries[s] > seg_boundaries[s + 1] + || seg_boundaries[s + 1] > rel->nrows) { + rc = EINVAL; + goto finish; + } if (rel->nrows <= 1 && !finalize) { rc = 0; - goto release_writer; + goto finish; } rc = col_rel_storage_owner_resolve(rel, &owner); if (rc != 0) - goto release_writer; + goto finish; if (rel == owner && col_rel_storage_alias_borrow_count(owner) > 0) { rc = EBUSY; - goto release_writer; + goto finish; + } + if (!wl_columnar_relation_generation_valid(rel->view_generation)) { + rc = EINVAL; + goto finish; + } + if (lease.owner != rel && rel->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u) { + rc = EOVERFLOW; + goto finish; + } + /* Hash candidate publication has one view event and may privatize one + * alias. Check both before it reads or allocates candidate state. */ + if (rel->nrows > 10000 && !rel->timestamps + && (rel->view_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u + || (lease.owner != rel && rel->storage_generation + >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u))) { + rc = EOVERFLOW; + goto finish; } rc = col_op_consolidate_kway_merge_impl(rel, seg_boundaries, seg_count, - &writer, &alias_release_pending); + &lease); if (rc == 0 && finalize) { + set.published = true; rel->sorted_nrows = rel->nrows; rel->run_count = 1; rel->run_ends[0] = rel->nrows; } if (rc == 0 && out_published) *out_published = true; -release_writer: - if (alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(rel); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; +finish: + { + int finish_rc = col_rel_mutation_set_finish(&set, false); + if (rc == 0) + rc = finish_rc; } - if (wl_columnar_source_access_writer_release(&writer) != 0 && rc == 0) - rc = EINVAL; return rc; } diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 0fa85597..fbed581c 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -2668,6 +2668,32 @@ col_rel_cow_unshare_with_lease(col_rel_t *r, return col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); } +int +wl_columnar_relation_privatize_shared_view_with_lease(col_rel_t *r, + wl_columnar_relation_mutation_lease_t *lease) +{ + int rc; + if (!r || !lease || lease->role_flags + != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + || col_rel_mutation_lease_validate(lease, r)) + return EINVAL; + if (r->col_shared) + return col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, + lease); + if (lease->owner == r) + return 0; + if (r->storage_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u) + return EOVERFLOW; + uint64_t prior_generation = r->storage_generation; + rc = col_rel_storage_alias_release_locked(r, lease); + if (rc) + return rc; + wl_columnar_relation_touch_storage(r); + if (col_rel_mutation_lease_advance_storage(lease, prior_generation) != 0) + abort(); + return 0; +} + static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, bool defer_metadata_retirement, From b4d90c468f44deb1356d764ec3333c665e21b7ce Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 01:55:01 +0900 Subject: [PATCH 10/15] Use owned merge partition boundaries --- tests/test_consolidate_kway_merge.c | 66 +++++++++++++++++++++++++++++ wirelog/columnar/merge.c | 4 +- 2 files changed, 68 insertions(+), 2 deletions(-) diff --git a/tests/test_consolidate_kway_merge.c b/tests/test_consolidate_kway_merge.c index 8be18350..5b5b55ff 100644 --- a/tests/test_consolidate_kway_merge.c +++ b/tests/test_consolidate_kway_merge.c @@ -71,10 +71,21 @@ static const char *consolidate_fail_site = NULL; static bool consolidate_fail_used = false; static uint32_t consolidate_fail_match = 1; static uint32_t consolidate_seen_matches = 0; +static const char *consolidate_mutate_site = NULL; +static uint32_t *consolidate_mutate_boundaries = NULL; +static uint32_t consolidate_mutate_index = 0; +static uint32_t consolidate_mutate_value = 0; +static bool consolidate_mutation_used = false; static bool test_consolidate_alloc_should_fail(const char *site) { + if (consolidate_mutate_site && !consolidate_mutation_used + && strcmp(site, consolidate_mutate_site) == 0) { + consolidate_mutate_boundaries[consolidate_mutate_index] + = consolidate_mutate_value; + consolidate_mutation_used = true; + } if (consolidate_fail_site && !consolidate_fail_used && strcmp(site, consolidate_fail_site) == 0) { consolidate_seen_matches++; @@ -113,6 +124,25 @@ clear_consolidate_allocation_failure(void) consolidate_fail_site = NULL; } +static void +mutate_consolidate_boundaries_at(const char *site, uint32_t *boundaries, + uint32_t index, uint32_t value) +{ + consolidate_mutate_site = site; + consolidate_mutate_boundaries = boundaries; + consolidate_mutate_index = index; + consolidate_mutate_value = value; + consolidate_mutation_used = false; +} + +static void +clear_consolidate_boundary_mutation(void) +{ + consolidate_mutate_site = NULL; + consolidate_mutate_boundaries = NULL; + consolidate_mutation_used = false; +} + #define TEST(name) \ do { \ test_count++; \ @@ -1977,6 +2007,41 @@ test_merge_epoch_headroom_preflight(void) PASS(); } +static void +test_merge_uses_owned_prepared_boundaries(void) +{ + uint32_t boundaries[] = { 0, 2, 4 }; + const int64_t values[] = { 2, 1, 3, 0 }; + TEST("merge uses its copied partition after caller boundary mutation"); + col_rel_t *rel = test_rel_alloc(1); + if (!rel) + FAIL("relation allocation failed"); + for (uint32_t i = 0; i < 4; i++) { + if (test_rel_append_row(rel, &values[i]) != 0) { + test_rel_free(rel); + FAIL("failed to append partition fixture"); + } + } + uint64_t view_generation = rel->view_generation; + /* This callback runs at segment_starts allocation, after sequence + * preparation copied and validated [0, 2, 4]. The replacement remains + * monotone and in-range, but describes a different partition. */ + mutate_consolidate_boundaries_at("segment_starts", boundaries, 1, 1); + int rc = col_op_consolidate_kway_merge(rel, boundaries, 2); + bool hook_used = consolidate_mutation_used; + clear_consolidate_boundary_mutation(); + clear_consolidate_allocation_failure(); + if (rc != 0 || !hook_used || boundaries[1] != 1 || rel->nrows != 4 + || rel->view_generation != view_generation + 3u + || rel->columns[0][0] != 0 || rel->columns[0][1] != 1 + || rel->columns[0][2] != 2 || rel->columns[0][3] != 3) { + test_rel_free(rel); + FAIL("post-prepare caller mutation changed merge partition semantics"); + } + test_rel_free(rel); + PASS(); +} + static void test_nullary_alias_large_hash_merge(void) { @@ -3271,6 +3336,7 @@ main(void) test_shared_view_merge_oom_preserves_view_state(); test_leased_merge_alias_epoch_paths(); test_merge_epoch_headroom_preflight(); + test_merge_uses_owned_prepared_boundaries(); test_nullary_alias_large_hash_merge(); test_merge_output_oom_is_transactional(); test_zero_arity_merge_output_is_never_zero_sized(); diff --git a/wirelog/columnar/merge.c b/wirelog/columnar/merge.c index 5dcefda2..3150eece 100644 --- a/wirelog/columnar/merge.c +++ b/wirelog/columnar/merge.c @@ -1061,8 +1061,8 @@ col_op_consolidate_kway_merge_impl(col_rel_t *rel, lease->set->published = true; for (uint32_t s = 0; s < seg_count; s++) { - uint32_t start = seg_boundaries[s]; - uint32_t end = seg_boundaries[s + 1]; + uint32_t start = sort_sequence.boundaries[s]; + uint32_t end = sort_sequence.boundaries[s + 1]; uint32_t count = end - start; seg_starts[s] = start; From 8eb85bd2b1de46f78e0bb80348d365cb92c23129 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 02:33:16 +0900 Subject: [PATCH 11/15] Retire raw relation mutation bridges --- tests/test_consolidate_incremental_delta.c | 2 +- tests/test_consolidate_kway_merge.c | 132 ++-- tests/test_radix_sort.c | 25 +- tests/test_relation_generations.c | 709 +++++++-------------- wirelog/columnar/internal.h | 84 +-- wirelog/columnar/relation.c | 582 +---------------- 6 files changed, 355 insertions(+), 1179 deletions(-) diff --git a/tests/test_consolidate_incremental_delta.c b/tests/test_consolidate_incremental_delta.c index 028010be..217d73a4 100644 --- a/tests/test_consolidate_incremental_delta.c +++ b/tests/test_consolidate_incremental_delta.c @@ -2995,7 +2995,7 @@ test_view_generation_headroom_reserved(void) /* ================================================================ * Issue #2057: delta_out's view generation advances once per emitted row - * (col_rel_append_row_locked) and once more when a failure rolls the + * (col_rel_append_row_with_lease) and once more when a failure rolls the * emission back. Every such advance is reserved before delta_out is first * mutated, so exhaustion is refused with EOVERFLOW instead of saturating * delta_out's view generation behind a successful return. diff --git a/tests/test_consolidate_kway_merge.c b/tests/test_consolidate_kway_merge.c index 5b5b55ff..eef08da0 100644 --- a/tests/test_consolidate_kway_merge.c +++ b/tests/test_consolidate_kway_merge.c @@ -1329,6 +1329,34 @@ test_consolidate_governor_create(uint64_t usable_bytes) return wl_columnar_memory_governor_ref_create(&resolution); } +typedef struct { + wl_columnar_relation_mutation_set_t set; + wl_columnar_relation_mutation_role_t role; + wl_columnar_relation_mutation_descriptor_t descriptor; + wl_columnar_relation_mutation_owner_t owners[2]; + wl_columnar_relation_mutation_lease_t lease; + wl_columnar_relation_mutation_initialization_t initialization; +} consolidate_test_mutation_t; + +static int +consolidate_test_mutation_acquire(col_rel_t *rel, + consolidate_test_mutation_t *held) +{ + held->role = (wl_columnar_relation_mutation_role_t){ + rel, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + return col_rel_mutation_set_acquire(&held->set, &held->role, 1, + &held->descriptor, 1, held->owners, 2, &held->lease, 1, + &held->initialization, 1); +} + +static int +consolidate_test_mutation_finish(consolidate_test_mutation_t *held, + bool published) +{ + return col_rel_mutation_set_finish(&held->set, published); +} + static void test_consolidate_scratch_admission(void) { @@ -2577,7 +2605,7 @@ test_radix_workspace_preflight(uint32_t count, bool timestamped) wl_columnar_memory_governor_ref_t *ref = test_consolidate_governor_create(UINT64_C(1) << 30); wl_columnar_radix_workspace_t workspace = { 0 }; - wl_columnar_source_access_writer_t writer = { 0 }; + consolidate_test_mutation_t held = { 0 }; int64_t *before = malloc((size_t)count * sizeof(*before)); col_delta_timestamp_t *before_ts = timestamped ? calloc(count, sizeof(*before_ts)) : NULL; @@ -2689,11 +2717,12 @@ test_radix_workspace_preflight(uint32_t count, bool timestamped) (size_t)count * sizeof(*before_ts)) == 0) && rel->view_generation == view && rel->storage_generation == storage, "preparation must preserve source bytes and generations"); - WP_CHECK(col_rel_source_writer_acquire(rel, &writer) == 0, "writer"); - rc = wl_columnar_relation_radix_sort_with_workspace(rel, 0, count, - &writer, &workspace); - WP_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "release writer"); + WP_CHECK(consolidate_test_mutation_acquire(rel, &held) == 0, + "payload lease"); + rc = wl_columnar_relation_radix_sort_with_lease(rel, 0, count, + &workspace, &held.lease); + WP_CHECK(consolidate_test_mutation_finish(&held, rc == 0) == 0, + "finish payload lease"); WP_CHECK(rc == 0 && wl_columnar_memory_reserved(g) == baseline + scratch + extra, "prepared sort must not charge twice"); @@ -2712,8 +2741,8 @@ test_radix_workspace_preflight(uint32_t count, bool timestamped) "sorted and empty ranges"); cleanup: clear_consolidate_allocation_failure(); - if (writer.owner) - (void)wl_columnar_source_access_writer_release(&writer); + if (held.set.identity) + (void)consolidate_test_mutation_finish(&held, false); wl_columnar_radix_workspace_destroy(&workspace); col_rel_destroy(rel); free(before); @@ -2739,15 +2768,19 @@ test_radix_admitted_call(col_rel_t *rel, unsigned route) return col_rel_radix_sort_int64(rel); if (route == 0) return col_rel_radix_sort(rel, 0, rel->nrows); - wl_columnar_source_access_writer_t writer = { 0 }; - int rc = col_rel_source_writer_acquire(rel, &writer); + consolidate_test_mutation_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + int rc = consolidate_test_mutation_acquire(rel, &held); if (rc != 0) return rc; - bool pending = false; - rc = col_rel_radix_sort_locked(rel, 0, rel->nrows, &writer, false, - &pending); - int release_rc = wl_columnar_source_access_writer_release(&writer); - return rc != 0 ? rc : release_rc; + rc = wl_columnar_relation_radix_workspace_prepare_with_lease(rel, 0, + rel->nrows, &workspace, &held.lease); + if (rc == 0) + rc = wl_columnar_relation_radix_sort_with_lease(rel, 0, rel->nrows, + &workspace, &held.lease); + wl_columnar_radix_workspace_destroy(&workspace); + int finish_rc = consolidate_test_mutation_finish(&held, rc == 0); + return rc ? rc : finish_rc; } static void @@ -2926,13 +2959,13 @@ test_radix_direct_admission(uint32_t count, unsigned route, unsigned ownership, static void test_radix_workspace_width(uint32_t count, bool floating) { - TEST("insertion workspace checks byte capacity before COW"); + TEST("radix workspace snapshot checks precede COW"); col_rel_t *narrow = col_rel_new_auto("narrow", 1); col_rel_t *wide = col_rel_new_auto("wide", 2), *view = NULL; wl_columnar_memory_governor_ref_t *ref = test_consolidate_governor_create(UINT64_C(1) << 30); wl_columnar_radix_workspace_t workspace = { 0 }; - wl_columnar_source_access_writer_t writer = { 0 }; + consolidate_test_mutation_t held = { 0 }; const char *failure = NULL; #define WW_CHECK(c, m) do { if (!(c)) { failure = m; goto cleanup; } } while (0) WW_CHECK(narrow && wide && ref, "fixture"); @@ -2965,17 +2998,19 @@ test_radix_workspace_width(uint32_t count, bool floating) uint64_t generation = view->view_generation, storage = view->storage_generation; int64_t *old_column = view->columns[0]; - WW_CHECK(col_rel_source_writer_acquire(view, &writer) == 0, "view writer"); + WW_CHECK(consolidate_test_mutation_acquire(view, &held) == 0, + "view mutation lease"); /* If COW is tried first, this limit yields ENOMEM, not required EINVAL. */ atomic_store_explicit(&g->usable_bytes, admitted, memory_order_seq_cst); - WW_CHECK(wl_columnar_relation_radix_sort_with_workspace(view, 0, count, - &writer, &workspace) == EINVAL && view->col_shared + WW_CHECK(wl_columnar_relation_radix_sort_with_lease(view, 0, count, + &workspace, &held.lease) == EINVAL + && consolidate_test_mutation_finish(&held, false) == 0 + && view->col_shared && view->columns[0] == old_column && view->view_generation == generation && view->storage_generation == storage && col_rel_storage_alias_borrow_count(wide) == 1 && wl_columnar_memory_reserved(g) == admitted, - "width refusal before COW"); - WW_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, "release"); + "foreign-width workspace refusal before COW"); for (uint32_t i = 0; i < count; i++) { int64_t a = count - i, b = (count - i) * 10; if (floating) { @@ -2985,11 +3020,11 @@ test_radix_workspace_width(uint32_t count, bool floating) WW_CHECK(view->columns[0][i] == a && view->columns[1][i] == b, "wide bytes preserved"); } - WW_CHECK(col_rel_source_writer_acquire(narrow, &writer) == 0, - "narrow writer"); - WW_CHECK(wl_columnar_relation_radix_sort_with_workspace(narrow, 0, count, - &writer, &workspace) == 0, "original workspace still usable"); - WW_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, "release"); + WW_CHECK(consolidate_test_mutation_acquire(narrow, &held) == 0 + && wl_columnar_relation_radix_sort_with_lease(narrow, 0, count, + &workspace, &held.lease) == 0 + && consolidate_test_mutation_finish(&held, true) == 0, + "exact narrow workspace remains usable"); wl_columnar_radix_workspace_destroy(&workspace); WW_CHECK(workspace.insertion_bytes == 0 && wl_columnar_memory_reserved(g) == baseline, @@ -3000,16 +3035,17 @@ test_radix_workspace_width(uint32_t count, bool floating) &workspace) == 0, "wide reprepare"); admitted = wl_columnar_memory_reserved(g); atomic_store_explicit(&g->usable_bytes, admitted, memory_order_seq_cst); - WW_CHECK(col_rel_source_writer_acquire(narrow, &writer) == 0, - "narrow writer"); - WW_CHECK(wl_columnar_relation_radix_sort_with_workspace(narrow, 0, count, - &writer, &workspace) == 0 && wl_columnar_memory_reserved(g) == admitted, - "wider capacity may serve narrower input without another charge"); - WW_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, "release"); - for (uint32_t i = 0; i < count; i++) - WW_CHECK(narrow->columns[0][i] == (floating - ? wl_columnar_float_to_bits((double)i + 1) : (int64_t)i + 1), - "narrow sorted output"); + WW_CHECK(consolidate_test_mutation_acquire(narrow, &held) == 0, + "narrow lease for foreign workspace"); + WW_CHECK(wl_columnar_relation_radix_sort_with_lease(narrow, 0, count, + &workspace, &held.lease) == EINVAL + && consolidate_test_mutation_finish(&held, false) == 0 + && wl_columnar_memory_reserved(g) == admitted, + "workspace provenance rejects a different relation"); + wl_columnar_radix_workspace_destroy(&workspace); + WW_CHECK(wl_columnar_radix_workspace_prepare(view, bounds, 1, 0, + &workspace) == 0, "shared view exact workspace"); + admitted = wl_columnar_memory_reserved(g); /* The supplied workspace is admitted; force the later COW reservation * to overflow with a real padding token, and preserve its typed cause. */ @@ -3018,10 +3054,12 @@ test_radix_workspace_width(uint32_t count, bool floating) atomic_store_explicit(&g->usable_bytes, UINT64_MAX, memory_order_seq_cst); WW_CHECK(wl_columnar_memory_reserve_checked(g, UINT64_MAX - admitted, &padding) == WL_COLUMNAR_MEMORY_ADMISSION_OK, "COW overflow padding"); - int rc = col_rel_source_writer_acquire(view, &writer); - if (rc == 0) - rc = wl_columnar_relation_radix_sort_with_workspace(view, 0, count, - &writer, &workspace); + WW_CHECK(consolidate_test_mutation_acquire(view, &held) == 0, + "shared view lease"); + int rc = wl_columnar_relation_radix_sort_with_lease(view, 0, count, + &workspace, &held.lease); + WW_CHECK(consolidate_test_mutation_finish(&held, false) == 0, + "failed shared sort lease release"); bool released = wl_columnar_memory_release(&padding); atomic_store_explicit(&g->usable_bytes, UINT64_C(1) << 30, memory_order_seq_cst); @@ -3030,10 +3068,12 @@ test_radix_workspace_width(uint32_t count, bool floating) && view->view_generation == generation && view->storage_generation == storage, "COW overflow must keep local cause and ownership"); - WW_CHECK(wl_columnar_relation_radix_sort_with_workspace(view, 0, count, - &writer, &workspace) == 0 && !view->col_shared + WW_CHECK(consolidate_test_mutation_acquire(view, &held) == 0 + && wl_columnar_relation_radix_sort_with_lease(view, 0, count, + &workspace, &held.lease) == 0 + && consolidate_test_mutation_finish(&held, true) == 0 + && !view->col_shared && col_rel_storage_alias_borrow_count(wide) == 0, "COW exact retry"); - WW_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, "release"); for (uint32_t i = 0; i < count; i++) { int64_t a = i + 1, b = (i + 1) * 10; if (floating) { @@ -3044,8 +3084,8 @@ test_radix_workspace_width(uint32_t count, bool floating) "COW sorted result"); } cleanup: - if (writer.owner) - (void)wl_columnar_source_access_writer_release(&writer); + if (held.set.identity) + (void)consolidate_test_mutation_finish(&held, false); wl_columnar_radix_workspace_destroy(&workspace); col_rel_destroy(view); col_rel_destroy(narrow); diff --git a/tests/test_radix_sort.c b/tests/test_radix_sort.c index 4e3e2bdd..bde4e79b 100644 --- a/tests/test_radix_sort.c +++ b/tests/test_radix_sort.c @@ -588,7 +588,12 @@ test_timestamp_sort(uint32_t count, bool prepared, bool floating, bool alias) timestamp_sort_oracle_t *expected = calloc(count, sizeof(*expected)); wl_columnar_radix_workspace_t workspace = { 0 }; wl_columnar_source_access_reader_t reader = { 0 }; - wl_columnar_source_access_writer_t writer = { 0 }; + wl_columnar_relation_mutation_set_t mutation_set = { 0 }; + wl_columnar_relation_mutation_role_t mutation_role = { 0 }; + wl_columnar_relation_mutation_descriptor_t mutation_descriptor = { 0 }; + wl_columnar_relation_mutation_owner_t mutation_owners[2] = { 0 }; + wl_columnar_relation_mutation_lease_t mutation_lease = { 0 }; + wl_columnar_relation_mutation_initialization_t mutation_init = { 0 }; wl_columnar_memory_governor_ref_t *governor = NULL; const char *failure = NULL; #define TS_CHECK(c, m) do { if (!(c)) { failure = m; goto cleanup; } } while (0) @@ -660,11 +665,17 @@ test_timestamp_sort(uint32_t count, bool prepared, bool floating, bool alias) &workspace) == 0, "prepare workspace"); TS_CHECK(workspace.timestamps && workspace.timestamp_capacity >= count, "timestamp scratch admitted in workspace"); - TS_CHECK(col_rel_source_writer_acquire(r, &writer) == 0, "writer"); - int rc = wl_columnar_relation_radix_sort_with_workspace(r, 1, count, - &writer, &workspace); - TS_CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "release writer"); + mutation_role = (wl_columnar_relation_mutation_role_t){ + r, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + TS_CHECK(col_rel_mutation_set_acquire(&mutation_set, &mutation_role, + 1, &mutation_descriptor, 1, mutation_owners, 2, + &mutation_lease, 1, &mutation_init, 1) == 0, + "mutation lease"); + int rc = wl_columnar_relation_radix_sort_with_lease(r, 1, count, + &workspace, &mutation_lease); + TS_CHECK(col_rel_mutation_set_finish(&mutation_set, rc == 0) == 0, + "finish mutation lease"); TS_CHECK(rc == 0, "prepared sort"); } else { TS_CHECK(col_rel_radix_sort(r, 1, count) == 0, "sort retry"); @@ -693,8 +704,6 @@ test_timestamp_sort(uint32_t count, bool prepared, bool floating, bool alias) timestamp_sort_record(i)), "COW source changed"); } cleanup: - if (writer.owner) - (void)wl_columnar_source_access_writer_release(&writer); if (reader.owner) (void)col_rel_source_reader_release(&reader); wl_columnar_radix_workspace_destroy(&workspace); diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 7ccf0bd6..98b27e70 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -216,6 +216,46 @@ new_relation(void) return track_relation(rel); } +typedef struct { + wl_columnar_relation_mutation_set_t set; + wl_columnar_relation_mutation_role_t role; + wl_columnar_relation_mutation_descriptor_t descriptor; + wl_columnar_relation_mutation_owner_t owners[2]; + wl_columnar_relation_mutation_lease_t lease; + wl_columnar_relation_mutation_initialization_t initialization; +} generation_test_mutation_t; + +static int +generation_test_mutation_acquire(col_rel_t *rel, + generation_test_mutation_t *held) +{ + held->role = (wl_columnar_relation_mutation_role_t){ + rel, WL_COLUMNAR_RELATION_PAYLOAD_MUTATION + }; + return col_rel_mutation_set_acquire(&held->set, &held->role, 1, + &held->descriptor, 1, held->owners, 2, &held->lease, 1, + &held->initialization, 1); +} + +static int +generation_test_mutation_finish(generation_test_mutation_t *held, + bool published) +{ + return col_rel_mutation_set_finish(&held->set, published); +} + +static int +generation_test_append_with_lease(col_rel_t *rel, const int64_t *row) +{ + generation_test_mutation_t held = { 0 }; + int rc = generation_test_mutation_acquire(rel, &held); + if (rc != 0) + return rc; + rc = col_rel_append_row_with_lease(rel, row, &held.lease); + int finish_rc = generation_test_mutation_finish(&held, rc == 0); + return rc ? rc : finish_rc; +} + static void test_same_row_count_mutation(void) { @@ -822,175 +862,22 @@ test_source_reader_blocks_checked_destroy(void) owned_relation_count--; } -/* Issue #1594: the collapsed radix-sort entry point validates every caller - * uniformly. Its predecessor col_rel_radix_sort_deferred() took no writer - * and checked nothing, so a caller holding no lease could reorder an - * owner's storage while a live alias was borrowing those exact buffers. */ - -/* A foreign writer names a different gate. The fixture is a shared view so - * a rejection that arrived after the COW would be visible as a detached - * relation rather than an untouched one. */ -static void -test_radix_sort_locked_rejects_foreign_writer(void) -{ - col_rel_t *source = new_relation(); - col_rel_t *view = new_relation(); - col_rel_t *other = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t high = 8, low = 3; - bool pending = true; - int64_t **columns_before; - uint64_t view_before; - - CHECK(source && view && other, "foreign writer relations"); - CHECK(col_rel_append_row(source, &high) == 0, "foreign writer row 0"); - CHECK(col_rel_append_row(source, &low) == 0, "foreign writer row 1"); - CHECK(col_rel_install_shared_view(view, source) == 0, - "foreign writer shared view"); - CHECK(col_rel_source_writer_acquire(other, &writer) == 0, - "foreign writer acquisition"); - columns_before = view->columns; - view_before = view->view_generation; - CHECK(col_rel_radix_sort_locked(view, 0, view->nrows, &writer, false, - &pending) == EINVAL, - "foreign writer refused"); - CHECK(pending == false, "foreign writer clears the out parameter"); - CHECK(view->col_shared != NULL && view->columns == columns_before - && view->view_generation == view_before - && source->storage_alias_borrows == 1, - "foreign writer refusal precedes the copy-on-write"); - CHECK(col_rel_get(source, 0, 0) == high && col_rel_get(source, 1, 0) == low, - "foreign writer refusal leaves the source unsorted"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "foreign writer release"); - cleanup_relations(); -} - -/* NULL writer and NULL out parameter. The out parameter is mandatory in - * both modes: a caller that meant to defer but passed NULL would otherwise - * have its borrow released under a transaction that still holds one. */ -static void -test_radix_sort_locked_rejects_absent_writer_and_out(void) -{ - col_rel_t *rel = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t high = 5, low = 1; - bool pending = true; - - CHECK(rel != NULL, "absent writer relation"); - CHECK(col_rel_append_row(rel, &high) == 0, "absent writer row 0"); - CHECK(col_rel_append_row(rel, &low) == 0, "absent writer row 1"); - CHECK(col_rel_radix_sort_locked(rel, 0, rel->nrows, NULL, false, - &pending) == EINVAL, - "absent writer refused"); - CHECK(pending == false, "absent writer clears the out parameter"); - CHECK(col_rel_get(rel, 0, 0) == high, - "absent writer refusal leaves the relation unsorted"); - CHECK(col_rel_source_writer_acquire(rel, &writer) == 0, - "mandatory out acquisition"); - CHECK(col_rel_radix_sort_locked(rel, 0, rel->nrows, &writer, true, - NULL) == EINVAL, - "NULL out parameter refused even with a valid writer"); - CHECK(col_rel_get(rel, 0, 0) == high, - "NULL out refusal leaves the relation unsorted"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "mandatory out release"); - cleanup_relations(); -} - -/* Validation now runs at every row count, including the range shortcut. */ -static void -test_radix_sort_locked_validates_below_the_range_shortcut(void) -{ - col_rel_t *rel = new_relation(); - col_rel_t *other = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t only = 4; - bool pending = true; - - CHECK(rel && other, "range shortcut relations"); - CHECK(col_rel_append_row(rel, &only) == 0, "range shortcut row"); - CHECK(col_rel_source_writer_acquire(other, &writer) == 0, - "range shortcut foreign acquisition"); - CHECK(col_rel_radix_sort_locked(rel, 0, rel->nrows, &writer, false, - &pending) == EINVAL, - "single-row sort still refuses a foreign writer"); - CHECK(col_rel_radix_sort_locked(rel, 0, 0, &writer, false, - &pending) == EINVAL, - "empty sort still refuses a foreign writer"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "range shortcut release"); - cleanup_relations(); -} - -/* Every radix-sort entry point must reject a contended or malformed trivial - * range before treating it as a no-op. The implicit-writer wrappers are - * checked separately with a source reader because a standalone call cannot - * be supplied a foreign writer. */ +/* Public mutation APIs validate contention before trivial-range shortcuts. */ static void test_radix_sort_family_validates_trivial_ranges(void) { - col_rel_t *rel = new_relation(); - col_rel_t *other = new_relation(); + col_rel_t *rel = NULL; col_rel_t *owner = new_relation(); col_rel_t *alias = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - wl_columnar_source_access_writer_t malformed = { 0 }; wl_columnar_source_access_reader_t reader = { 0 }; - wl_columnar_radix_workspace_t workspace = { 0 }; - const uint32_t boundaries[] = { 0u, 1u }; int64_t value = 7; - bool pending = true; uint32_t sorted_before; - CHECK(rel && other && owner && alias, "trivial-range family relations"); - CHECK(col_rel_append_row(rel, &value) == 0, - "trivial-range family row"); - CHECK(wl_columnar_radix_workspace_prepare(rel, boundaries, 1, 0, - &workspace) == 0, "trivial-range workspace preparation"); - CHECK(col_rel_source_writer_acquire(other, &writer) == 0, - "trivial-range foreign writer acquisition"); - CHECK(col_rel_radix_sort_locked(rel, 0, 1, &writer, false, &pending) - == EINVAL, "locked sort rejects foreign writer for one row"); - CHECK(pending == false, "locked sort clears pending on rejection"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(rel, 0, 1, - &writer, &workspace) == EINVAL, - "workspace sort rejects foreign writer for one row"); - malformed.owner = &rel->source_access; - malformed.identity = (uintptr_t)&malformed; - CHECK(col_rel_radix_sort_locked(rel, 0, 1, &malformed, false, &pending) - == EINVAL, "locked sort rejects invalid writer thread token"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(rel, 0, 1, - &malformed, &workspace) == EINVAL, - "workspace sort rejects invalid writer thread token"); - CHECK(col_rel_radix_sort_locked(rel, 0, 0, &writer, false, &pending) - == EINVAL, "locked sort rejects foreign writer for empty range"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "trivial-range foreign writer release"); - wl_columnar_radix_workspace_destroy(&workspace); - + CHECK(owner && alias, "trivial-range relations"); CHECK(col_rel_append_row(owner, &value) == 0, "trivial-range owner row"); CHECK(col_rel_install_shared_view(alias, owner) == 0, "trivial-range owner shared view"); - wl_columnar_radix_workspace_destroy(&workspace); - CHECK(wl_columnar_radix_workspace_prepare(owner, boundaries, 1, 0, - &workspace) == 0, "trivial-range owner workspace preparation"); - CHECK(col_rel_source_writer_acquire(owner, &writer) == 0, - "trivial-range owner writer acquisition"); - CHECK(col_rel_radix_sort_locked(owner, 0, 0, &writer, false, &pending) - == EBUSY, "locked sort reports owner borrow for empty range"); - CHECK(col_rel_radix_sort_locked(owner, 0, 1, &writer, false, &pending) - == EBUSY, "locked sort reports owner borrow for one row"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(owner, 0, 0, - &writer, &workspace) == EBUSY, - "workspace sort reports owner borrow for empty range"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(owner, 0, 1, - &writer, &workspace) == EBUSY, - "workspace sort reports owner borrow for one row"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "trivial-range owner writer release"); - wl_columnar_radix_workspace_destroy(&workspace); sorted_before = owner->sorted_nrows; CHECK(col_rel_radix_sort(owner, 0, 0) == EBUSY, "standalone sort reports owner borrow for empty range"); @@ -1028,9 +915,6 @@ test_radix_sort_family_validates_trivial_ranges(void) cleanup_relations(); col_rel_t *zero = NULL; - wl_columnar_source_access_writer_t zero_writer = { 0 }; - wl_columnar_radix_workspace_t zero_workspace = { 0 }; - const uint32_t zero_boundaries[] = { 0u, 3u }; CHECK(col_rel_alloc(&zero, "trivial-range-zero") == 0, "zero-column relation allocation"); CHECK(track_relation(zero) != NULL @@ -1039,19 +923,6 @@ test_radix_sort_family_validates_trivial_ranges(void) for (uint32_t i = 0; i < 3; i++) CHECK(col_rel_append_row(zero, &value) == 0, "zero-column relation row"); - CHECK(wl_columnar_radix_workspace_prepare(zero, zero_boundaries, 1, 0, - &zero_workspace) == 0, "zero-column workspace preparation"); - CHECK(col_rel_source_writer_acquire(zero, &zero_writer) == 0, - "zero-column writer acquisition"); - CHECK(col_rel_radix_sort_locked(zero, 0, zero->nrows, &zero_writer, - false, &pending) == 0, - "zero-column locked sort remains a validated no-op"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(zero, 0, - zero->nrows, &zero_writer, &zero_workspace) == 0, - "zero-column workspace sort remains a validated no-op"); - CHECK(wl_columnar_source_access_writer_release(&zero_writer) == 0, - "zero-column writer release"); - wl_columnar_radix_workspace_destroy(&zero_workspace); CHECK(col_rel_radix_sort_int64(zero) == 0 && zero->sorted_nrows == zero->nrows, "zero-column sort keeps its no-op behavior after validation"); @@ -1069,163 +940,6 @@ test_radix_sort_family_validates_trivial_ranges(void) cleanup_relations(); } -/* An owner with live borrows is refused with EBUSY. The source gate is - * deliberately left uncontended so the status can only come from the alias - * predicate, not from a reader holding the gate. */ -static void -test_radix_sort_locked_owner_with_borrows_is_busy(void) -{ - col_rel_t *owner = new_relation(); - col_rel_t *alias = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t high = 9, low = 2; - bool pending = true; - uint64_t view_before; - - CHECK(owner && alias, "busy owner relations"); - CHECK(col_rel_append_row(owner, &high) == 0, "busy owner row 0"); - CHECK(col_rel_append_row(owner, &low) == 0, "busy owner row 1"); - CHECK(col_rel_install_shared_view(alias, owner) == 0, - "busy owner shared view"); - CHECK(owner->storage_alias_borrows == 1, "busy owner borrow recorded"); - CHECK(col_rel_source_writer_acquire(owner, &writer) == 0, - "busy owner writer acquisition"); - view_before = owner->view_generation; - CHECK(col_rel_radix_sort_locked(owner, 0, owner->nrows, &writer, false, - &pending) == EBUSY, - "owner with live borrows refused"); - CHECK(pending == false, "busy owner clears the out parameter"); - CHECK(col_rel_get(owner, 0, 0) == high && col_rel_get(owner, 1, 0) == low - && owner->view_generation == view_before - && owner->storage_alias_borrows == 1, - "busy owner refusal reorders nothing"); - CHECK(col_rel_storage_alias_release(alias) == 0, "busy owner release"); - CHECK(col_rel_radix_sort_locked(owner, 0, owner->nrows, &writer, false, - &pending) == 0, - "owner sorts once the borrow is retired"); - CHECK(col_rel_get(owner, 0, 0) == low && col_rel_get(owner, 1, 0) == high, - "retired borrow admits the sort"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "busy owner writer release"); - cleanup_relations(); -} - -/* defer_alias_release == true hands the borrow back to the caller. */ -static void -test_radix_sort_locked_hands_back_the_alias(void) -{ - col_rel_t *source = new_relation(); - col_rel_t *view = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t high = 7, low = 0; - bool pending = false; - int64_t **source_columns; - - CHECK(source && view, "hand-back relations"); - CHECK(col_rel_append_row(source, &high) == 0, "hand-back row 0"); - CHECK(col_rel_append_row(source, &low) == 0, "hand-back row 1"); - CHECK(col_rel_install_shared_view(view, source) == 0, - "hand-back shared view"); - source_columns = source->columns; - CHECK(col_rel_source_writer_acquire(source, &writer) == 0, - "hand-back writer acquisition"); - CHECK(col_rel_radix_sort_locked(view, 0, view->nrows, &writer, true, - &pending) == 0, - "hand-back sort admitted"); - CHECK(pending == true, "hand-back reports the pending release"); - CHECK(source->storage_alias_borrows == 1, - "hand-back keeps the borrow live"); - CHECK(view->storage_owner == source, - "hand-back leaves the borrow on the original owner"); - CHECK(view->col_shared == NULL, "hand-back detached the view"); - CHECK(col_rel_get(view, 0, 0) == low && col_rel_get(view, 1, 0) == high, - "hand-back sorted the view"); - CHECK(source->columns == source_columns - && col_rel_get(source, 0, 0) == high - && col_rel_get(source, 1, 0) == low, - "hand-back left the owner's storage alone"); - CHECK(col_rel_storage_alias_release(view) == 0, "hand-back release"); - CHECK(source->storage_alias_borrows == 0, "hand-back borrow retired"); - CHECK(view->storage_owner == view, "hand-back view owns its storage"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "hand-back writer release"); - cleanup_relations(); -} - -/* defer_alias_release == false releases the borrow here. This is what - * replaces the old "NULL means never defer" guarantee. */ -static void -test_radix_sort_locked_releases_the_alias_itself(void) -{ - col_rel_t *source = new_relation(); - col_rel_t *view = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - int64_t high = 6, low = 1; - bool pending = true; - - CHECK(source && view, "self-release relations"); - CHECK(col_rel_append_row(source, &high) == 0, "self-release row 0"); - CHECK(col_rel_append_row(source, &low) == 0, "self-release row 1"); - CHECK(col_rel_install_shared_view(view, source) == 0, - "self-release shared view"); - CHECK(col_rel_source_writer_acquire(source, &writer) == 0, - "self-release writer acquisition"); - CHECK(col_rel_radix_sort_locked(view, 0, view->nrows, &writer, false, - &pending) == 0, - "self-release sort admitted"); - CHECK(pending == false, "self-release reports no pending release"); - CHECK(source->storage_alias_borrows == 0, - "self-release retired the borrow"); - CHECK(view->storage_owner == view, "self-release view owns its storage"); - CHECK(col_rel_get(view, 0, 0) == low && col_rel_get(view, 1, 0) == high, - "self-release sorted the view"); - CHECK(col_rel_get(source, 0, 0) == high - && col_rel_get(source, 1, 0) == low, - "self-release left the owner's storage alone"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "self-release writer release"); - cleanup_relations(); -} - -/* Issue #1594: the workspace variant reported a live alias borrow as EINVAL - * while the rest of the family reported the same condition as EBUSY, so one - * family answered a caller two different ways. The k-way merge wrapper - * pre-checks the condition, which is why nothing caught the split. */ -static void -test_radix_sort_with_workspace_owner_with_borrows_is_busy(void) -{ - col_rel_t *owner = new_relation(); - col_rel_t *alias = new_relation(); - wl_columnar_radix_workspace_t workspace = { 0 }; - wl_columnar_source_access_writer_t writer = { 0 }; - const uint32_t boundaries[] = { 0u, 2u }; - int64_t high = 11, low = 4; - bool prepared = false; - - CHECK(owner && alias, "workspace busy relations"); - CHECK(col_rel_append_row(owner, &high) == 0, "workspace busy row 0"); - CHECK(col_rel_append_row(owner, &low) == 0, "workspace busy row 1"); - CHECK(col_rel_install_shared_view(alias, owner) == 0, - "workspace busy shared view"); - CHECK(owner->storage_alias_borrows == 1, "workspace busy borrow"); - /* One segment spanning both rows; the boundaries array carries - * seg_count + 1 entries. */ - prepared = wl_columnar_radix_workspace_prepare(owner, boundaries, 1, 0, - &workspace) == 0; - CHECK(prepared, "workspace busy prepare"); - CHECK(col_rel_source_writer_acquire(owner, &writer) == 0, - "workspace busy writer acquisition"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(owner, 0, - owner->nrows, &writer, &workspace) == EBUSY, - "workspace sort reports a live borrow as EBUSY"); - CHECK(col_rel_get(owner, 0, 0) == high && col_rel_get(owner, 1, 0) == low, - "workspace busy refusal reorders nothing"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "workspace busy writer release"); - wl_columnar_radix_workspace_destroy(&workspace); - cleanup_relations(); -} - static void test_legacy_storage_owner_metadata_initialization(void) { @@ -1437,7 +1151,7 @@ test_workspace_sort_cows_shared_view(void) { col_rel_t *source = new_relation(); col_rel_t *view = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; + generation_test_mutation_t held = { 0 }; wl_columnar_radix_workspace_t workspace = { 0 }; uint32_t boundaries[2]; bool *old_shared; @@ -1453,21 +1167,20 @@ test_workspace_sort_cows_shared_view(void) boundaries[1] = view->nrows; CHECK(wl_columnar_radix_workspace_prepare(view, boundaries, 1, 0, &workspace) == 0, "workspace shared-view preparation"); - CHECK(wl_columnar_source_access_writer_acquire( - &source->source_access, &writer) == 0, - "workspace shared-view writer admission"); + CHECK(generation_test_mutation_acquire(view, &held) == 0, + "workspace shared-view mutation admission"); old_column = view->columns[0]; old_shared = view->col_shared; - CHECK(wl_columnar_relation_radix_sort_with_workspace(view, 0, - view->nrows, &writer, &workspace) == 0, + CHECK(wl_columnar_relation_radix_sort_with_lease(view, 0, + view->nrows, &workspace, &held.lease) == 0, "workspace sort COWs shared view"); CHECK(view->columns[0] != old_column && view->col_shared == NULL && source->storage_alias_borrows == 0 && view->columns[0][0] == 1 && view->columns[0][view->nrows - 1u] == 40, "workspace sort publishes detached sorted storage"); CHECK(old_shared != NULL, "workspace sort started from shared storage"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "workspace shared-view writer release"); + CHECK(generation_test_mutation_finish(&held, true) == 0, + "workspace shared-view mutation finish"); wl_columnar_radix_workspace_destroy(&workspace); cleanup_relations(); } @@ -1477,8 +1190,9 @@ test_source_writer_cow_shared_view(void) { col_rel_t *source = new_relation(); col_rel_t *view = new_relation(); - wl_columnar_source_access_writer_t writer = { 0 }; - bool alias_release_pending = true; + generation_test_mutation_t held = { 0 }; + wl_columnar_radix_workspace_t workspace = { 0 }; + uint32_t boundaries[2]; int64_t source_first; CHECK(source && view, "source-writer shared-view relation allocation"); @@ -1488,25 +1202,24 @@ test_source_writer_cow_shared_view(void) CHECK(col_rel_install_shared_view(view, source) == 0, "source-writer shared-view install"); source_first = source->columns[0][0]; - CHECK(wl_columnar_source_access_writer_acquire( - &source->source_access, &writer) == 0, - "source-writer shared-view admission"); - /* defer_alias_release == false is the self-releasing mode the wrapper - * this test was written against provided before #1594 collapsed the - * radix-sort lease wrappers into col_rel_radix_sort_locked. */ - CHECK(col_rel_radix_sort_locked(view, 0, view->nrows, &writer, false, - &alias_release_pending) == 0, + CHECK(generation_test_mutation_acquire(view, &held) == 0, + "source-writer shared-view mutation admission"); + boundaries[0] = 0; + boundaries[1] = view->nrows; + CHECK(wl_columnar_radix_workspace_prepare(view, boundaries, 1, 0, + &workspace) == 0, "source-writer shared-view workspace"); + CHECK(wl_columnar_relation_radix_sort_with_lease(view, 0, view->nrows, + &workspace, &held.lease) == 0, "source-writer sorts shared view through COW"); - CHECK(alias_release_pending == false, - "source-writer COW released the alias itself"); CHECK(source->columns[0][0] == source_first && view->col_shared == NULL && source->storage_alias_borrows == 0 && view->columns[0][0] == 1 && view->columns[0][view->nrows - 1u] == 40, "source-writer COW preserves source and detaches view"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "source-writer shared-view release"); + CHECK(generation_test_mutation_finish(&held, true) == 0, + "source-writer shared-view mutation finish"); + wl_columnar_radix_workspace_destroy(&workspace); cleanup_relations(); } @@ -1522,7 +1235,7 @@ test_workspace_sort_rejects_undersized_workspace_and_rolls_back(void) col_rel_t *view = new_relation(); col_rel_t *sorted = new_relation(); wl_columnar_radix_workspace_t undersized = { 0 }; - wl_columnar_source_access_writer_t writer = { 0 }; + generation_test_mutation_t held = { 0 }; uint32_t boundaries[2]; int64_t low = 1; int64_t high = 2; @@ -1567,11 +1280,10 @@ test_workspace_sort_rejects_undersized_workspace_and_rolls_back(void) && undersized.insertion_capacity == 0, "undersized workspace really is undersized"); - CHECK(wl_columnar_source_access_writer_acquire( - &source->source_access, &writer) == 0, - "undersized workspace writer admission"); - CHECK(wl_columnar_relation_radix_sort_with_workspace(view, 0, - view->nrows, &writer, &undersized) == EINVAL, + CHECK(generation_test_mutation_acquire(view, &held) == 0, + "undersized workspace mutation admission"); + CHECK(wl_columnar_relation_radix_sort_with_lease(view, 0, + view->nrows, &undersized, &held.lease) == EINVAL, "undersized workspace refused rather than silently reallocated"); CHECK(view->col_shared != NULL && view->storage_owner == source && source->storage_alias_borrows == 1, @@ -1586,8 +1298,8 @@ test_workspace_sort_rejects_undersized_workspace_and_rolls_back(void) wl_columnar_memory_governor_ref_get(governor)) == old_reserved, "refused workspace sort restores exact retained token and credit"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "undersized workspace writer release"); + CHECK(generation_test_mutation_finish(&held, false) == 0, + "undersized workspace mutation rollback"); wl_columnar_radix_workspace_destroy(&undersized); cleanup_relations(); if (governor) { @@ -4706,15 +4418,17 @@ test_set_cow_descriptor_and_owner_admission(void) && col_rel_cow_unshare(root, 0) == EBUSY && memcmp(root, &root_snapshot, sizeof(root_snapshot)) == 0, "canonical mutation with live alias is rejected unchanged"); - wl_columnar_source_access_writer_t raw = { 0 }; - CHECK(col_rel_source_writer_acquire(alias, &raw) == 0, - "raw alias writer setup"); - CHECK(col_rel_cow_unshare_with_source_writer(alias, &raw) == EINVAL + generation_test_mutation_t alias_mutation = { 0 }; + CHECK(generation_test_mutation_acquire(alias, &alias_mutation) == 0, + "alias COW mutation lease setup"); + wl_columnar_relation_mutation_lease_t copied_lease + = alias_mutation.lease; + CHECK(col_rel_cow_unshare_with_lease(alias, &copied_lease) == EINVAL && memcmp(alias, &snapshot, offsetof(col_rel_t, source_access)) == 0, - "raw source writer cannot authorize alias COW"); - CHECK(wl_columnar_source_access_writer_release(&raw) == 0, - "raw alias writer release"); + "copied lease cannot authorize alias COW"); + CHECK(generation_test_mutation_finish(&alias_mutation, false) == 0, + "alias COW mutation lease release"); wl_columnar_relation_test_fail_next_prepare_resize(); CHECK(mutation_set_or_cow(alias, operation) == ENOMEM && memcmp(alias, &snapshot, sizeof(snapshot)) == 0 @@ -4895,20 +4609,15 @@ test_public_append_mutation_admission(void) } } -/* Exercise the same physical-overlap and governor contract through each - * append authority. The locked caller retains and releases its own writer. */ +/* Exercise overlap through the public single-operation and explicit-lease + * admitted paths. */ static int -append_overlap_attempt(col_rel_t *rel, const int64_t *row, bool locked) +append_overlap_attempt(col_rel_t *rel, const int64_t *row, + generation_test_mutation_t *held) { - if (!locked) + if (!held) return col_rel_append_row(rel, row); - wl_columnar_source_access_writer_t writer = { 0 }; - int rc = col_rel_source_writer_acquire(rel, &writer); - if (rc) - return rc; - rc = col_rel_append_row_locked(rel, row, &writer); - int release_rc = wl_columnar_source_access_writer_release(&writer); - return rc ? rc : release_rc; + return col_rel_append_row_with_lease(rel, row, &held->lease); } static void @@ -4919,11 +4628,12 @@ test_append_overlapping_input(bool locked) col_rel_t *rel = track_relation(col_rel_new_auto("append_overlap", width)); CHECK(rel, "public append overlap relation"); + generation_test_mutation_t held = { 0 }; int64_t row[32]; for (uint32_t i = 0; i < COL_REL_INIT_CAP; i++) { for (uint32_t c = 0; c < width; c++) row[c] = (int64_t)i * 1000 + c; - CHECK(append_overlap_attempt(rel, row, locked) == 0, + CHECK(append_overlap_attempt(rel, row, NULL) == 0, "public overlap source rows"); } wl_columnar_memory_resolution_t resolution = { @@ -4938,6 +4648,9 @@ test_append_overlapping_input(bool locked) "public append overlap governed relation"); wl_columnar_memory_governor_t *governor = wl_columnar_memory_governor_ref_get(ref); + if (locked) + CHECK(generation_test_mutation_acquire(rel, &held) == 0, + "append overlap mutation admission"); uint64_t credit = wl_columnar_memory_reserved(governor); col_rel_t before = *rel; const int64_t *inside = &rel->columns[0][3]; @@ -4950,7 +4663,8 @@ test_append_overlapping_input(bool locked) resolution.usable_bytes - credit, &blocker) && wl_columnar_memory_commit(&blocker, &blocker), "wide staging budget blocker"); - CHECK(append_overlap_attempt(rel, inside, locked) == ENOMEM + CHECK(append_overlap_attempt(rel, inside, + locked ? &held : NULL) == ENOMEM && rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == @@ -4960,7 +4674,8 @@ test_append_overlapping_input(bool locked) #ifdef WL_TEST_ALLOC_WRAP allocation_calls = 0; allocation_fail_at = 0; - int failed_rc = append_overlap_attempt(rel, inside, locked); + int failed_rc = append_overlap_attempt(rel, inside, + locked ? &held : NULL); allocation_fail_at = -1; CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) @@ -4973,7 +4688,7 @@ test_append_overlapping_input(bool locked) for (long fail = wide ? 1 : 0; fail < (wide ? 4 : 2); fail++) { allocation_calls = 0; allocation_fail_at = fail; - int failed_rc = append_overlap_attempt(rel, inside, true); + int failed_rc = append_overlap_attempt(rel, inside, &held); allocation_fail_at = -1; CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) @@ -4982,13 +4697,14 @@ test_append_overlapping_input(bool locked) } #endif wl_columnar_relation_test_fail_next_reservation_commit(); - CHECK(append_overlap_attempt(rel, inside, true) == ENOMEM + int publish_rc = append_overlap_attempt(rel, inside, &held); + CHECK(publish_rc == ENOMEM && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == credit, "locked publication failure returns staged scratch credit"); } - CHECK(append_overlap_attempt(rel, inside, locked) == 0 + CHECK(append_overlap_attempt(rel, inside, locked ? &held : NULL) == 0 && rel->nrows == before.nrows + 1 && rel->storage_generation == before.storage_generation + 1, "overlapping input survives owned growth and retired old column"); @@ -5002,10 +4718,11 @@ test_append_overlapping_input(bool locked) #ifdef WL_TEST_ALLOC_WRAP allocation_calls = 0; allocation_fail_at = 0; - int external_rc = append_overlap_attempt(rel, row, locked); + int external_rc = append_overlap_attempt(rel, row, + locked ? &held : NULL); long external_calls = allocation_calls; int inside_rc = append_overlap_attempt(rel, rel->columns[0], - locked); + locked ? &held : NULL); allocation_fail_at = -1; CHECK(external_rc == 0 && external_calls == 0 && inside_rc == ENOMEM && !rel->memory_budget_denial_pending @@ -5014,13 +4731,17 @@ test_append_overlapping_input(bool locked) #endif uint32_t index = rel->nrows; memcpy(expected, rel->columns[0], width * sizeof(*inside)); - CHECK(append_overlap_attempt(rel, rel->columns[0], locked) == 0 + CHECK(append_overlap_attempt(rel, rel->columns[0], + locked ? &held : NULL) == 0 && wl_columnar_memory_reserved(governor) == credit, "wide overlapping spare append returns its temporary reservation"); for (uint32_t c = 0; c < width; c++) CHECK(rel->columns[c][index] == expected[c], "wide overlapping spare append preserves tuple"); } + if (locked) + CHECK(generation_test_mutation_finish(&held, true) == 0, + "append overlap mutation finish"); cleanup_relations(); CHECK(wl_columnar_memory_reserved(governor) == 0, "public append overlap cleanup returns all credit"); @@ -5029,7 +4750,7 @@ test_append_overlapping_input(bool locked) } static void -test_locked_append_alias_overlap(void) +test_leased_append_alias_overlap(void) { for (unsigned wide = 0; wide < 2; wide++) { for (unsigned full = 0; full < 2; full++) { @@ -5041,6 +4762,7 @@ test_locked_append_alias_overlap(void) col_rel_t *sibling = track_relation(col_rel_new_auto("locked_sibling", width)); CHECK(root && alias && sibling, "locked overlap aliases"); + generation_test_mutation_t held = { 0 }; int64_t row[32], expected[32]; uint32_t count = full ? COL_REL_INIT_CAP : 40; for (uint32_t i = 0; i < count; i++) { @@ -5064,6 +4786,8 @@ test_locked_append_alias_overlap(void) "locked alias governed"); wl_columnar_memory_governor_t *governor = wl_columnar_memory_governor_ref_get(ref); + CHECK(generation_test_mutation_acquire(alias, &held) == 0, + "locked alias mutation admission"); uint64_t credit = wl_columnar_memory_reserved(governor); col_rel_t root_before = *root, before = *alias, sibling_before = *sibling; @@ -5071,16 +4795,14 @@ test_locked_append_alias_overlap(void) CHECK(3 + width <= alias->capacity, "wide tuple wholly inside allocation"); memcpy(expected, inside, width * sizeof(*inside)); - wl_columnar_source_access_writer_t writer = { 0 }; - CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, - "locked alias writer admission"); wl_columnar_memory_reservation_t blocker; wl_columnar_memory_reservation_init(&blocker); CHECK(wl_columnar_memory_reserve(governor, resolution.usable_bytes - credit, &blocker) && wl_columnar_memory_commit(&blocker, &blocker), "locked COW quota blocker"); - CHECK(col_rel_append_row_locked(alias, inside, &writer) == ENOMEM + CHECK(col_rel_append_row_with_lease(alias, inside, + &held.lease) == ENOMEM && alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) @@ -5094,8 +4816,8 @@ test_locked_append_alias_overlap(void) for (long fail = 0; fail < (wide ? 3 : 2); fail++) { allocation_calls = 0; allocation_fail_at = fail; - int failed_rc = col_rel_append_row_locked(alias, inside, - &writer); + int failed_rc = col_rel_append_row_with_lease(alias, inside, + &held.lease); allocation_fail_at = -1; CHECK(failed_rc == ENOMEM && !alias->memory_budget_denial_pending @@ -5108,7 +4830,9 @@ test_locked_append_alias_overlap(void) } #endif wl_columnar_relation_test_fail_next_prepare_resize(); - CHECK(col_rel_append_row_locked(alias, inside, &writer) == ENOMEM + int prepare_rc = col_rel_append_row_with_lease(alias, inside, + &held.lease); + CHECK(prepare_rc == ENOMEM && !alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) @@ -5123,8 +4847,8 @@ test_locked_append_alias_overlap(void) &blocker) && wl_columnar_memory_commit(&blocker, &blocker), "wide COW admits scratch but blocks payload"); - CHECK(col_rel_append_row_locked(alias, inside, - &writer) == ENOMEM + CHECK(col_rel_append_row_with_lease(alias, inside, &held.lease) + == ENOMEM && alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) @@ -5137,8 +4861,8 @@ test_locked_append_alias_overlap(void) "wide COW payload quota blocker release"); } else { wl_columnar_relation_test_fail_next_reservation_commit(); - CHECK(col_rel_append_row_locked(alias, inside, - &writer) == ENOMEM + CHECK(col_rel_append_row_with_lease(alias, inside, &held.lease) + == ENOMEM && !alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) @@ -5147,7 +4871,7 @@ test_locked_append_alias_overlap(void) && wl_columnar_memory_reserved(governor) == credit, "locked COW publication failure rolls back borrows and credit"); } - CHECK(col_rel_append_row_locked(alias, inside, &writer) == 0 + CHECK(col_rel_append_row_with_lease(alias, inside, &held.lease) == 0 && alias->storage_owner == alias && !alias->col_shared && alias->nrows == before.nrows + 1 && alias->view_generation == before.view_generation + 1 @@ -5167,9 +4891,8 @@ test_locked_append_alias_overlap(void) CHECK(mutation_payload_unchanged(root, &root_before) && mutation_payload_unchanged(sibling, &sibling_before), "locked COW preserves root and sibling pointers and epochs"); - CHECK(writer.owner == &root->source_access - && wl_columnar_source_access_writer_release(&writer) == 0, - "locked append retains original owner writer for caller"); + CHECK(generation_test_mutation_finish(&held, true) == 0, + "locked alias mutation finish"); cleanup_relations(); CHECK(wl_columnar_memory_reserved(governor) == 0, "locked alias cleanup releases all credit"); @@ -5180,76 +4903,91 @@ test_locked_append_alias_overlap(void) typedef struct { col_rel_t *relation; - wl_columnar_source_access_writer_t *writer; + wl_columnar_relation_mutation_lease_t *lease; int rc; -} locked_append_thread_args_t; +} leased_append_thread_args_t; static void * -locked_append_foreign_thread(void *arg) +leased_append_foreign_thread(void *arg) { - locked_append_thread_args_t *args = arg; + leased_append_thread_args_t *args = arg; /* An invalid row endpoint would overflow if payload validation ran. */ - args->rc = col_rel_append_row_locked(args->relation, - (const int64_t *)(UINTPTR_MAX - 3), args->writer); + args->rc = col_rel_append_row_with_lease(args->relation, + (const int64_t *)(UINTPTR_MAX - 3), args->lease); return NULL; } static void -test_locked_append_authority_and_edges(void) +test_leased_append_authority_and_edges(void) { col_rel_t *rel = new_relation(), *other = new_relation(); int64_t row = 17; CHECK(rel && other && col_rel_append_row(rel, &row) == 0, - "locked edge relations"); + "leased edge relations"); rel->memory_budget_denial_pending = true; - wl_columnar_source_access_writer_t writer = { 0 }, invalid = { 0 }; - CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), - &invalid) == EINVAL && rel->memory_budget_denial_pending, - "absent writer preserves evidence before row read"); - CHECK(col_rel_source_writer_acquire(other, &writer) == 0, - "locked foreign owner writer"); - CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), - &writer) == EINVAL && rel->memory_budget_denial_pending, - "foreign owner rejects before overlap arithmetic"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0 - && col_rel_source_writer_acquire(rel, &writer) == 0, - "locked correct writer"); - invalid = writer; - CHECK(col_rel_append_row_locked(rel, &row, &invalid) == EINVAL + CHECK(col_rel_append_row_with_lease(rel, + (const int64_t *)(UINTPTR_MAX - 3), NULL) == EINVAL + && rel->memory_budget_denial_pending, + "absent lease preserves evidence before row read"); + generation_test_mutation_t foreign = { 0 }, held = { 0 }, + other_held = { 0 }; + CHECK(generation_test_mutation_acquire(other, &foreign) == 0, + "foreign relation lease"); + CHECK(col_rel_append_row_with_lease(rel, + (const int64_t *)(UINTPTR_MAX - 3), &foreign.lease) == EINVAL && rel->memory_budget_denial_pending, - "copied writer rejects before admission"); - locked_append_thread_args_t args = { .relation = rel, .writer = &writer }; + "foreign relation lease rejects before overlap arithmetic"); + CHECK(generation_test_mutation_finish(&foreign, false) == 0 + && generation_test_mutation_acquire(rel, &held) == 0 + && generation_test_mutation_acquire(other, &other_held) == 0, + "exact relation lease acquisition"); + wl_columnar_relation_mutation_lease_t copied = held.lease; + CHECK(col_rel_append_row_with_lease(rel, &row, &copied) == EINVAL + && rel->memory_budget_denial_pending, + "copied lease rejects before admission"); + CHECK(col_rel_append_row_with_lease(rel, + (const int64_t *)(UINTPTR_MAX - 3), &other_held.lease) == EINVAL + && rel->memory_budget_denial_pending, + "foreign lease rejects before payload arithmetic"); + CHECK(generation_test_mutation_finish(&other_held, false) == 0, + "foreign test lease release"); + leased_append_thread_args_t args = { .relation = rel, + .lease = &held.lease }; wl_thread_t thread; - CHECK(wl_thread_create(&thread, locked_append_foreign_thread, &args) == 0 + CHECK(wl_thread_create(&thread, leased_append_foreign_thread, &args) == 0 && wl_thread_join(&thread) == 0 && args.rc == EINVAL && rel->memory_budget_denial_pending, - "foreign thread rejects before payload read"); + "wrong thread rejects before payload read"); + CHECK(generation_test_mutation_finish(&held, false) == 0, + "finish provenance checks before stale denial admission"); + CHECK(generation_test_mutation_acquire(rel, &held) == 0, + "admitted non-budget failure mutation lease"); col_rel_t before = *rel; - CHECK(col_rel_append_row_locked(rel, (const int64_t *)(UINTPTR_MAX - 3), - &writer) == EOVERFLOW && !rel->memory_budget_denial_pending + rel->memory_budget_denial_pending = true; + CHECK(col_rel_append_row_with_lease(rel, + (const int64_t *)(UINTPTR_MAX - 3), &held.lease) == EOVERFLOW + && !rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before), "admitted row endpoint overflow clears evidence without dereference"); int64_t *column = rel->columns[0]; rel->columns[0] = (int64_t *)(UINTPTR_MAX - 3); - rel->memory_budget_denial_pending = true; - CHECK(col_rel_append_row_locked(rel, &row, &writer) == EOVERFLOW + CHECK(col_rel_append_row_with_lease(rel, &row, &held.lease) == EOVERFLOW && !rel->memory_budget_denial_pending, "column endpoint overflow precedes dereference"); rel->columns[0] = column; rel->timestamp_capacity = 1; - rel->memory_budget_denial_pending = true; - CHECK(col_rel_append_row_locked(rel, &row, &writer) == EINVAL + CHECK(col_rel_append_row_with_lease(rel, &row, &held.lease) == EINVAL && !rel->memory_budget_denial_pending, "admitted shape rejection clears evidence"); rel->timestamp_capacity = 0; - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "locked edge writer release"); + CHECK(generation_test_mutation_finish(&held, false) == 0, + "lease edge release"); cleanup_relations(); } static void -test_locked_append_legacy_descriptor_topology(void) +test_append_lease_descriptor_topology(void) { col_rel_t *root = new_relation(), *alias = new_relation(), *sibling = new_relation(); @@ -5257,32 +4995,25 @@ test_locked_append_legacy_descriptor_topology(void) CHECK(root && alias && sibling && col_rel_append_row(root, &row) == 0 && col_rel_install_shared_view(alias, root) == 0 && col_rel_install_shared_view(sibling, root) == 0, - "legacy append descriptor topology relations"); - wl_columnar_source_access_writer_t writer = { 0 }; - CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, - "legacy topology first descriptor admission"); - wl_columnar_source_access_gate_t *owner_gate = writer.owner; - wl_columnar_source_access_gate_t *alias_gate = writer.secondary_owner; - CHECK(owner_gate == &root->source_access && - alias_gate == &alias->descriptor_access - && wl_columnar_source_access_writer_release(&writer) == 0, - "legacy first token pins alias descriptor and canonical owner"); - CHECK(col_rel_source_writer_acquire(sibling, &writer) == 0, - "legacy topology second descriptor admission"); - CHECK(writer.owner == owner_gate && - writer.secondary_owner == &sibling->descriptor_access - && writer.secondary_owner != alias_gate - && wl_columnar_source_access_writer_release(&writer) == 0, - "same owner does not imply the same admitted descriptor"); - /* TODO (#2033 Unit 2B3b2): exact target lease authority must reject an - * alias-A token used for alias B before reads, denial reset, or scratch. - * The legacy owner-only append contract provides no such guarantee; - * characterization deliberately performs no unsafe cross-descriptor write. */ + "append lease descriptor topology relations"); + generation_test_mutation_t alias_lease = { 0 }, sibling_lease = { 0 }; + CHECK(generation_test_mutation_acquire(alias, &alias_lease) == 0 + && alias_lease.descriptor.relation == alias + && alias_lease.lease.owner == root, + "alias lease binds its exact descriptor and shared owner"); + CHECK(generation_test_mutation_finish(&alias_lease, false) == 0 + && generation_test_mutation_acquire(sibling, &sibling_lease) == 0 + && sibling_lease.descriptor.relation == sibling + && sibling_lease.lease.owner == root + && &sibling_lease.descriptor != &alias_lease.descriptor, + "same owner retains distinct alias descriptor leases"); + CHECK(generation_test_mutation_finish(&sibling_lease, false) == 0, + "append descriptor leases released"); cleanup_relations(); } static void -test_locked_append_generation_boundaries(void) +test_leased_append_generation_boundaries(void) { /* spare heap, growing heap, shared COW, growing arena, short arena * timestamps, short heap timestamps, and spare arena with timestamps. */ @@ -5324,9 +5055,6 @@ test_locked_append_generation_boundaries(void) rel->timestamp_capacity = rel->nrows; } } - wl_columnar_source_access_writer_t writer = { 0 }; - CHECK(col_rel_source_writer_acquire(rel, &writer) == 0, - "locked epoch writer"); uint64_t view = rel->view_generation, storage = rel->storage_generation; for (unsigned boundary = 0; boundary < (replacement ? 2u : 1u); boundary++) { @@ -5336,16 +5064,21 @@ test_locked_append_generation_boundaries(void) if (boundary && rel->storage_owner == rel) rel->storage_owner_generation = *epoch; col_rel_t before = *rel; - rel->memory_budget_denial_pending = true; + generation_test_mutation_t held = { 0 }; + CHECK(generation_test_mutation_acquire(rel, &held) == 0, + "epoch operation mutation admission"); #ifdef WL_TEST_ALLOC_WRAP allocation_calls = 0; allocation_fail_at = 0; #endif - CHECK(col_rel_append_row_locked(rel, rel->columns[0], - &writer) == EOVERFLOW - && !rel->memory_budget_denial_pending + rel->memory_budget_denial_pending = true; + int append_rc = col_rel_append_row_with_lease(rel, + rel->columns[0], &held.lease); + CHECK(append_rc == EOVERFLOW && mutation_payload_unchanged(rel, &before), "locked epoch boundary rejects before staging or publication"); + CHECK(generation_test_mutation_finish(&held, false) == 0, + "epoch operation lease rollback"); #ifdef WL_TEST_ALLOC_WRAP CHECK(allocation_calls == 0, "epoch rejection consumes no allocation"); @@ -5363,15 +5096,19 @@ test_locked_append_generation_boundaries(void) storage = rel->storage_generation; } uint32_t rows = rel->nrows; - CHECK(col_rel_append_row_locked(rel, rel->columns[0], &writer) == 0 + generation_test_mutation_t held = { 0 }; + CHECK(generation_test_mutation_acquire(rel, &held) == 0, + "epoch success mutation admission"); + CHECK(col_rel_append_row_with_lease(rel, rel->columns[0], + &held.lease) == 0 && rel->nrows == rows + 1 && rel->columns[0][rows] == row[0] && rel->view_generation == view + 1 && rel->storage_generation == storage + (replacement ? 1 : 0) && (!short_ts || rel->timestamp_capacity == rel->capacity) && (kind != 6 || rel->arena_owned), "locked epoch retry publishes exact view and storage epochs"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "locked epoch release"); + CHECK(generation_test_mutation_finish(&held, true) == 0, + "epoch success mutation finish"); col_rel_destroy(rel); col_rel_destroy(root); delta_pool_destroy(pool); @@ -5915,7 +5652,7 @@ test_public_batch_lazy_and_timestamp_edges(void) } static void -test_public_append_edges_and_locked_compatibility(void) +test_public_and_leased_append_edges(void) { col_rel_t *rel = new_relation(); int64_t row = 43; @@ -5956,16 +5693,11 @@ test_public_append_edges_and_locked_compatibility(void) col_rel_t *root = new_relation(), *alias = new_relation(); CHECK(root && alias && col_rel_append_row(root, &row) == 0 && col_rel_install_shared_view(alias, root) == 0, - "locked append compatibility setup"); - wl_columnar_source_access_writer_t writer = { 0 }; - CHECK(col_rel_source_writer_acquire(alias, &writer) == 0, - "locked append legacy source writer"); - CHECK(col_rel_append_row_locked(alias, &row, &writer) == 0 + "leased append compatibility setup"); + CHECK(generation_test_append_with_lease(alias, &row) == 0 && alias->storage_owner == alias && alias->nrows == 2 && col_rel_storage_alias_borrow_count(root) == 0, - "locked append retains legacy admitted alias behavior"); - CHECK(wl_columnar_source_access_writer_release(&writer) == 0, - "locked append compatibility release"); + "lease append retains admitted alias behavior"); cleanup_relations(); col_rel_t *empty = track_relation(col_rel_new_auto("append_nullary", 0)); CHECK(empty && col_rel_append_row(empty, &row) == 0 && empty->nrows == 1, @@ -6983,6 +6715,10 @@ test_radix_sequence_lazy_publication(void) && rel->storage_owner_generation == 0, "zero-sort failed finish rolls lazy owner initialization back"); wl_columnar_relation_radix_sequence_destroy(&sequence); + /* Restore a destroyable canonical descriptor after verifying rollback. */ + rel->storage_owner = rel; + rel->storage_owner_identity = rel->relation_identity; + rel->storage_owner_generation = rel->storage_generation; cleanup_relations(); } @@ -7586,11 +7322,11 @@ main(void) test_public_append_mutation_admission(); test_append_overlapping_input(false); test_append_overlapping_input(true); - test_locked_append_alias_overlap(); - test_locked_append_authority_and_edges(); - test_locked_append_legacy_descriptor_topology(); - test_locked_append_generation_boundaries(); - test_public_append_edges_and_locked_compatibility(); + test_leased_append_alias_overlap(); + test_leased_append_authority_and_edges(); + test_append_lease_descriptor_topology(); + test_leased_append_generation_boundaries(); + test_public_and_leased_append_edges(); test_set_cow_denial_provenance(); test_empty_append_all_legacy_cow_compatibility(); test_leased_rebind_descriptor_exclusion(); @@ -7611,14 +7347,7 @@ main(void) test_legacy_storage_owner_metadata_initialization(); test_peer_reader_alias_accounting_is_race_safe(); test_source_reader_blocks_radix_sort(); - test_radix_sort_locked_rejects_foreign_writer(); - test_radix_sort_locked_rejects_absent_writer_and_out(); - test_radix_sort_locked_validates_below_the_range_shortcut(); test_radix_sort_family_validates_trivial_ranges(); - test_radix_sort_locked_owner_with_borrows_is_busy(); - test_radix_sort_locked_hands_back_the_alias(); - test_radix_sort_locked_releases_the_alias_itself(); - test_radix_sort_with_workspace_owner_with_borrows_is_busy(); test_workspace_sort_cows_shared_view(); test_workspace_sort_rejects_undersized_workspace_and_rolls_back(); test_source_writer_cow_shared_view(); diff --git a/wirelog/columnar/internal.h b/wirelog/columnar/internal.h index b68a17fd..06dce90b 100644 --- a/wirelog/columnar/internal.h +++ b/wirelog/columnar/internal.h @@ -2945,27 +2945,12 @@ int wl_col_rel_inline_project_column(col_rel_t *dst, uint32_t dst_row, const col_rel_t *src, uint32_t src_row, uint32_t logical_col); -/* Public row append owns descriptor and canonical source writers throughout - * publication. Input tuples may overlap column storage; the public path - * stages the complete tuple before any replacement. Locked append also - * stages overlap under its validated caller-owned canonical writer and - * releases a detached alias immediately before returning. Its transitional - * owner resolver requires a stable descriptor binding from the caller. - * Raw writer validation covers the canonical gate, not exact target - * descriptor authority; owner-only consolidation tokens remain supported - * pending the two-role lease migration (#2033). Both #2033 and #2034 remain - * open until that migration and removal of the live raw alias bridges. - * Caller keeps other input buffers stable for the call. Locked reservation - * retains its legacy source-writer contract. Outer attempts reset transient - * denial evidence only after authority admission. - * Actual budget refusal retains legacy ENOMEM plus - * memory_budget_denial_pending; allocation ENOMEM, EOVERFLOW and EINVAL are - * distinct. Lower nested admission helpers do not reset outer evidence. */ +/* Public append acquires exact descriptor and canonical-owner authority. + * The lease form is used by wider transactions that already hold a mutation + * set. Both stage overlapping tuples before replacement and preserve the + * distinction between quota denial, allocator ENOMEM, EOVERFLOW and EINVAL. */ int col_rel_append_row(col_rel_t *r, const int64_t *row); -int -col_rel_append_row_locked(col_rel_t *r, const int64_t *row, - wl_columnar_source_access_writer_t *writer); /* Append a complete input batch under one relation writer. All row/type * validation and governor admission complete before the first row publishes. * On ENOMEM, *@denied distinguishes governor refusal from allocator failure. */ @@ -3160,15 +3145,6 @@ int col_rel_source_reader_acquire_transferable(const col_rel_t *, int col_rel_source_reader_release(wl_columnar_source_access_reader_t *); int col_rel_source_writer_acquire(const col_rel_t *, wl_columnar_source_access_writer_t *); -/* Canonical non-detach authority only: raw writers cannot detach aliases. */ -int col_rel_cow_unshare_with_source_writer(col_rel_t *, - const wl_columnar_source_access_writer_t *); - -/* Temporary compatibility used only by the three unmigrated merge - * transactions. Remove as those complete transactions acquire mutation sets. */ -int col_rel_cow_unshare_legacy_with_source_writer(col_rel_t *, - const wl_columnar_source_access_writer_t *); - /* Checked single-cell mutation. The implementation lives in relation.c so * it holds descriptor and canonical-owner writers across COW/publication. */ int col_rel_set(col_rel_t *, uint32_t row, uint32_t col, int64_t val); @@ -3259,33 +3235,6 @@ wl_columnar_radix_workspace_prepare(const col_rel_t *rel, wl_columnar_radix_workspace_t *workspace); void wl_columnar_radix_workspace_destroy(wl_columnar_radix_workspace_t *workspace); -/* Sort through a caller-supplied workspace. A still-shared view is - * accepted: it takes the same transactional copy-on-write path as - * col_rel_radix_sort_locked, threading the caller's scratch buffers - * through, and the alias borrow it takes is released here after - * publication. That unification is #1603; before it, this entry point - * refused a shared view with EINVAL while col_rel_radix_sort_locked - * unshared and sorted the same relation. - * - * A NULL relation or workspace, and a bad range, are refused at every row - * count. Writer, owner-linkage and contention checks also run before the - * nrows <= 1 shortcut, so a malformed or contended call is never masked by - * a trivial range. Workspace capacity is not needed for a trivial range. - * - * EBUSY here is the same predicate and the same status - * col_rel_radix_sort_locked reports. - * - * EINVAL: NULL relation or workspace; a bad range; an absent, foreign or - * wrong-thread writer; a stale canonical-owner linkage; or a - * workspace whose capacity was prepared for a smaller range than - * the one requested. - * EBUSY: rel is its own canonical owner and has live alias borrows. */ -int -wl_columnar_relation_radix_sort_with_workspace(col_rel_t *rel, - uint32_t start_row, - uint32_t nrows, const wl_columnar_source_access_writer_t *writer, - const wl_columnar_radix_workspace_t *workspace); - /* Lease-scoped preparation allocates the complete scratch even for sorted * ranges, and preflights authority, shape and epoch headroom before allocation. */ WL_MUST_CHECK int @@ -3336,31 +3285,6 @@ wl_columnar_relation_radix_sort_rows_by_key_typed(int64_t *data, * Phase C: permutation-apply uses col_rel_row_copy_out/in. */ int col_rel_radix_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows); -/* Sort [start_row, start_row + nrows) under a source-access writer the -* caller already holds on r's canonical storage owner. -* -* defer_alias_release == false: any alias borrow the deferred COW takes is -* released before this returns, and *out_alias_release_pending is false. -* defer_alias_release == true: the borrow is handed back. -* *out_alias_release_pending is true iff the caller must call -* col_rel_storage_alias_release(r) once its wider mutation transaction -* ends. Consolidation uses this to keep the borrow across the whole -* transaction and release it from its centralized cleanup path. -* -* out_alias_release_pending is mandatory in both modes and is written on -* every path, including every rejection. -* -* EINVAL: NULL out parameter, bad range, or an absent, foreign or -* wrong-thread writer -- checked at every row count. -* EBUSY: r is its own canonical owner and has live alias borrows. An -* alias of a borrowed owner is still admissible; that is the -* consolidation case. -* ENOMEM: COW or permutation allocation failed; r is left unchanged. */ -WL_MUST_CHECK int -col_rel_radix_sort_locked(col_rel_t *r, uint32_t start_row, uint32_t nrows, - const wl_columnar_source_access_writer_t *writer, - bool defer_alias_release, bool *out_alias_release_pending); - /* ======================================================================== */ /* Cache & Materialized Join (columnar/cache.c) */ /* ======================================================================== */ diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index fbed581c..6dd00000 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -329,27 +329,20 @@ col_rel_mutation_single_acquire(col_rel_t *relation, static int col_rel_grow_owned_transition_legacy_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release); -typedef struct { - int64_t **columns; - bool *shared; -} col_rel_cow_deferred_t; static int col_rel_cow_unshare_legacy_impl(col_rel_t *r, uint32_t new_cap, - bool defer_alias_release, bool defer_metadata_retirement, - col_rel_cow_deferred_t *deferred_old); + bool defer_alias_release); static int col_rel_grow_owned_transition_publish_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release, wl_columnar_relation_mutation_lease_t *lease); static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, - bool defer_alias_release, bool defer_metadata_retirement, - col_rel_cow_deferred_t *deferred_old, + bool defer_alias_release, wl_columnar_relation_mutation_lease_t *lease); static int col_rel_mutation_lease_advance_storage( wl_columnar_relation_mutation_lease_t *lease, uint64_t prior_generation); static bool col_rel_timestamp_shape_valid(const col_rel_t *r); -/* Temporary compatibility for unmigrated append/radix/batch transactions. - * NULL is an explicit legacy path, never a payload mutation lease. These - * private entry points disappear as their outer transactions acquire sets. */ +/* Append-all, batch and replacement transactions retain their existing + * private authority until those complete transactions migrate. */ static int col_rel_grow_owned_transition_legacy_impl(col_rel_t *r, uint32_t new_cap, bool defer_alias_release) @@ -360,11 +353,10 @@ col_rel_grow_owned_transition_legacy_impl(col_rel_t *r, uint32_t new_cap, static int col_rel_cow_unshare_legacy_impl(col_rel_t *r, uint32_t new_cap, - bool defer_alias_release, bool defer_metadata_retirement, - col_rel_cow_deferred_t *deferred_old) + bool defer_alias_release) { return col_rel_cow_unshare_publish_impl(r, new_cap, defer_alias_release, - defer_metadata_retirement, deferred_old, NULL); + NULL); } static int @@ -1614,8 +1606,7 @@ col_rel_set(col_rel_t *r, uint32_t row, uint32_t col, int64_t val) } bool borrowed = r->col_shared != NULL; if (borrowed) { - rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, - &single.lease); + rc = col_rel_cow_unshare_publish_impl(r, 0, false, &single.lease); if (rc) goto finish; } @@ -2651,8 +2642,7 @@ col_rel_cow_unshare(col_rel_t *r, uint32_t new_cap) * it again; allocation failure keeps it clear. Guard contention above * leaves prior evidence untouched. It is not mutation rollback state. */ r->memory_budget_denial_pending = false; - rc = col_rel_cow_unshare_publish_impl(r, new_cap, false, false, NULL, - &single.lease); + rc = col_rel_cow_unshare_publish_impl(r, new_cap, false, &single.lease); int finish_rc = col_rel_mutation_set_finish(&single.set, rc == 0); return rc ? rc : finish_rc; } @@ -2665,7 +2655,7 @@ col_rel_cow_unshare_with_lease(col_rel_t *r, != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r)) return EINVAL; - return col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); + return col_rel_cow_unshare_publish_impl(r, 0, false, lease); } int @@ -2678,8 +2668,7 @@ wl_columnar_relation_privatize_shared_view_with_lease(col_rel_t *r, || col_rel_mutation_lease_validate(lease, r)) return EINVAL; if (r->col_shared) - return col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, - lease); + return col_rel_cow_unshare_publish_impl(r, 0, false, lease); if (lease->owner == r) return 0; if (r->storage_generation >= WL_COLUMNAR_REL_GENERATION_INVALID - 1u) @@ -2696,8 +2685,7 @@ wl_columnar_relation_privatize_shared_view_with_lease(col_rel_t *r, static int col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, - bool defer_alias_release, bool defer_metadata_retirement, - col_rel_cow_deferred_t *deferred_old, + bool defer_alias_release, wl_columnar_relation_mutation_lease_t *lease) { col_rel_payload_txn_t pending; @@ -2723,8 +2711,6 @@ col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, return 0; capacity = new_cap ? new_cap : (r->capacity ? r->capacity : COL_REL_INIT_CAP); - if (deferred_old) - *deferred_old = (col_rel_cow_deferred_t){ 0 }; if (capacity > r->capacity) return col_rel_grow_owned_transition_publish_impl(r, capacity, defer_alias_release, lease); @@ -2755,22 +2741,14 @@ col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, r->columns = private_cols; r->col_shared = NULL; private_cols = NULL; - if (defer_metadata_retirement) { - if (!deferred_old) - abort(); - deferred_old->columns = old_columns; - deferred_old->shared = old_shared; - } else { - for (uint32_t c = 0; c < r->ncols; c++) - if (!old_shared[c]) - free(old_columns[c]); - free((void *)old_columns); - free(old_shared); - col_rel_retire_shared_table_credit(r); - } + for (uint32_t c = 0; c < r->ncols; c++) + if (!old_shared[c]) + free(old_columns[c]); + free((void *)old_columns); + free(old_shared); + col_rel_retire_shared_table_credit(r); col_rel_release_retired_reservation(&previous); - if (!defer_metadata_retirement) - col_rel_retire_payload_credit(r); + col_rel_retire_payload_credit(r); col_rel_ledger_reconcile(r, ledger_before); /* Private columns replaced the borrowed view: one storage epoch. */ if (lease) { @@ -2800,40 +2778,6 @@ col_rel_cow_unshare_publish_impl(col_rel_t *r, uint32_t new_cap, return ENOMEM; } -/* Temporary legacy entry point for hash dedup, kway merge, and - * incremental delta consolidation in merge.c. Their complete transactions - * migrate in 2B/the batch unit; a raw writer is not a mutation-set lease. */ -int -col_rel_cow_unshare_legacy_with_source_writer(col_rel_t *r, - const wl_columnar_source_access_writer_t *writer) -{ - col_rel_t *owner = NULL; - int rc; - - if (!r || !writer) - return EINVAL; - rc = col_rel_storage_owner_resolve(r, &owner); - if (rc != 0) - return rc; - if (writer->owner != &owner->source_access - || writer->identity != (uintptr_t)writer - || !wl_columnar_source_access_writer_thread_equal(writer)) - return EINVAL; - return col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); -} - -/* A raw writer authorizes canonical non-detaching work only. */ -int -col_rel_cow_unshare_with_source_writer(col_rel_t *r, - const wl_columnar_source_access_writer_t *writer) -{ - col_rel_t *owner = NULL; - if (!r || col_rel_storage_owner_resolve(r, &owner) || owner != r - || r->col_shared) - return EINVAL; - return col_rel_cow_unshare_legacy_with_source_writer(r, writer); -} - int col_rel_attach_memory_governor(col_rel_t *r, wl_columnar_memory_governor_ref_t *memory_governor) @@ -4813,8 +4757,8 @@ col_rel_append_row_impl(col_rel_t *r, const int64_t *row, * in-place row write even when no capacity growth is needed. */ if (r->col_shared) { rc = lease - ? col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease) - : col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); + ? col_rel_cow_unshare_publish_impl(r, 0, false, lease) + : col_rel_cow_unshare_legacy_impl(r, 0, true); if (rc != 0) goto release_writer; alias_release_pending = true; @@ -4892,17 +4836,11 @@ col_rel_append_row_with_lease(col_rel_t *r, const int64_t *row, != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r)) return EINVAL; + r->memory_budget_denial_pending = false; return col_rel_append_row_impl(r, row, &lease->set->owners[lease->owner_slot].writer, true, lease); } -int -col_rel_append_row_locked(col_rel_t *r, const int64_t *row, - wl_columnar_source_access_writer_t *writer) -{ - return col_rel_append_row_impl(r, row, writer, true, NULL); -} - static int col_rel_capacity_for_rows(uint32_t current, uint32_t required, uint32_t *out_capacity) @@ -5198,8 +5136,8 @@ wl_columnar_relation_reserve_rows_impl(col_rel_t *r, uint32_t additional, } if (r->col_shared && required <= r->capacity) { int rc = lease - ? col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease) - : col_rel_cow_unshare_legacy_impl(r, 0, true, false, NULL); + ? col_rel_cow_unshare_publish_impl(r, 0, false, lease) + : col_rel_cow_unshare_legacy_impl(r, 0, true); if (rc == 0 && !lease) *out_alias_release_pending = alias_release_pending; return rc; @@ -5590,7 +5528,7 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, return rc; /* Explicit temporary legacy path: append_all already owns its * source writer and migrates as a complete read-source unit. */ - rc = col_rel_cow_unshare_legacy_impl(dst, 0, false, false, NULL); + rc = col_rel_cow_unshare_legacy_impl(dst, 0, false); if (wl_columnar_source_access_writer_release(&writer) != 0 && rc == 0) rc = EINVAL; @@ -5660,8 +5598,7 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, * path so public COW may acquire its own gate independently. */ if (dst->col_shared) { dst->memory_budget_denial_pending = false; - rc = col_rel_cow_unshare_legacy_impl(dst, 0, false, false, - NULL); + rc = col_rel_cow_unshare_legacy_impl(dst, 0, false); } else { rc = 0; } @@ -5840,7 +5777,7 @@ col_rel_append_all_impl(col_rel_t *dst, const col_rel_t *src, /* Bulk append also mutates spare capacity, so a shared view must be * privatized even when the destination does not grow. */ if (dst->col_shared) { - rc = col_rel_cow_unshare_legacy_impl(dst, 0, true, false, NULL); + rc = col_rel_cow_unshare_legacy_impl(dst, 0, true); if (rc != 0) goto cleanup; alias_release_pending = true; @@ -9220,36 +9157,6 @@ wl_columnar_radix_rows_prepared(col_rel_t *r, uint32_t start_row, const wl_columnar_radix_workspace_t *workspace, col_delta_timestamp_t *timestamps); -static int -col_rel_insertion_sort(col_rel_t *r, uint32_t start_row, uint32_t nrows, - const wl_columnar_radix_workspace_t *workspace, - col_delta_timestamp_t *timestamps) -{ - uint32_t nc = r->ncols; -#if SIZE_MAX <= UINT32_MAX - if (nrows == UINT32_MAX) - return EOVERFLOW; -#endif - size_t row_count = (size_t)nrows + 1u; - if (nc != 0 && row_count > SIZE_MAX - / ((size_t)nc * sizeof(int64_t))) - return EOVERFLOW; - size_t work_count = row_count * nc; - bool owns_work = workspace == NULL; - int64_t *work = workspace ? workspace->insertion_rows - : (int64_t *)malloc(work_count * sizeof(*work)); - if (workspace && workspace->insertion_capacity < nrows) - return EINVAL; - if (!work) - return ENOMEM; - wl_columnar_radix_workspace_t prepared = { .insertion_rows = work }; - int rc = wl_columnar_radix_insertion_prepared(r, start_row, nrows, - &prepared, timestamps); - if (owns_work) - free(work); - return rc; -} - /* ======================================================================== */ /* Fused Uniform Check + Count Pass (Issue #363 Phase 1) */ /* ======================================================================== */ @@ -9660,206 +9567,6 @@ radix_uniform_count_fused_k16_scalar(const int64_t *col_data, #define radix_uniform_count_fused_k16_fast radix_uniform_count_fused_k16_scalar #endif -/* - * radix_sort_k16: LSD radix sort using 16-bit radix (Issue #363 Phase 5b/5c). - * - * 4 passes × 65536 buckets. Scalar fused uniform+count loop with - * WL_PREFETCH_R; in-place prefix sum avoids a separate 256KB prefix[]. - * Called for nrows >= 50000 where fewer passes justify the larger histogram. - */ -static int -radix_sort_k16(col_rel_t *r, uint32_t start_row, uint32_t nrows, - const wl_columnar_radix_workspace_t *workspace, - col_delta_timestamp_t *timestamps) -{ - - const uint32_t hist_size = 65536u; - - bool owns_workspace = workspace == NULL; - uint32_t *perm_a = workspace ? workspace->perm_a - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_a"); - uint32_t *perm_b = workspace ? workspace->perm_b - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_b"); - uint16_t *bv_cache = workspace ? workspace->short_values - : (uint16_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint16_t), "radix_short_values"); - uint32_t *count = workspace ? workspace->count16 - : (uint32_t *)wl_columnar_relation_radix_malloc( - hist_size * sizeof(uint32_t), "radix_count16"); - if (workspace && workspace->k16_capacity < nrows) - return EINVAL; - if (!perm_a || !perm_b || !bv_cache || !count) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - free(count); - } - /* The transactional insertion fallback is intentionally limited to - * small inputs. Running O(n^2) insertion sort for a failed 50K-row - * k16 allocation would turn an allocation failure into a timeout. */ - return nrows <= 32 ? col_rel_insertion_sort(r, start_row, nrows, - workspace, timestamps) - : ENOMEM; - } - - int64_t *temp_col = workspace ? workspace->temp_column - : (int64_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(int64_t), "radix_temp_column"); - if (!temp_col) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - free(count); - } - return ENOMEM; - } - wl_columnar_radix_workspace_t prepared = { - .perm_a = perm_a, .perm_b = perm_b, .temp_column = temp_col, - .short_values = bv_cache, - .count16 = count, - }; - int rc = wl_columnar_radix_k16_prepared(r, start_row, nrows, &prepared, - timestamps); - if (owns_workspace) { - free(temp_col); - free(perm_a); - free(perm_b); - free(bv_cache); - free(count); - } - return rc; -} - -/* - * col_rel_radix_sort: index-permutation LSD radix sort (Phase B, Issue #330). - * - * Sort sub-range [start_row, start_row + nrows) of r in-place. - * Uses col_rel_get() for key extraction (layout-independent). - * Sorts a permutation array instead of scattering full rows. - * Permutation is applied once at the end via col_rel_row_copy_out/in. - * - * Optimizations (Issue #343): - * - Hybrid threshold: insertion sort for nrows <= 32 - * - Skip-pass: skip byte positions where all values have the same byte - * - Byte-value cache: read column data once per pass, reuse for scatter - * - * Adaptive radix width (Issue #363 Phase 5c): - * - nrows >= 50000: k=16 via radix_sort_k16() — 4 passes × 65536 buckets - * - nrows < 50000: k=8 with SIMD fused uniform+count — 8 passes × 256 buckets - * - * Falls back to insertion sort on allocation failure. - */ -static int -wl_columnar_relation_radix_sort_rows(col_rel_t *r, uint32_t start_row, - uint32_t nrows, - const wl_columnar_radix_workspace_t *workspace, - col_delta_timestamp_t *timestamps) -{ - if (nrows <= 1) - return 0; - - /* IEEE-754 keys need a different sign transform from signed integers. - * Keep the established radix fast path for integer-only relations and - * use the typed comparator until the float radix path is introduced. */ - if (r->column_types) { - for (uint32_t c = 0; c < r->ncols; c++) { - if (r->column_types[c] == WIRELOG_TYPE_FLOAT) - goto insertion; - } - } - - /* Hybrid threshold: insertion sort for small segments (Issue #343) */ - if (nrows <= 32) - goto insertion; - - /* Adaptive radix width (Issue #363 Phase 5c): dispatch to k=16 for large - * arrays where fewer passes outweigh the larger histogram cost. - * - * Empirical threshold (Apple M-series, 1-col, 64-bit uniform-random keys): - * nrows=10K: k8=0.66ms k16=0.81ms k8 faster by 1.23x - * nrows=20K: k8=0.81ms k16=0.91ms k8 faster by 1.12x - * nrows=30K: k8=0.83ms k16=0.89ms k8 faster by 1.08x - * nrows=40K: k8=0.81ms k16=0.82ms near parity - * nrows=50K: k8=0.79ms k16=0.78ms k16 faster by 1.01x - * nrows=60K: k8=0.82ms k16=0.80ms k16 faster by 1.02x - * nrows=100K: k8=1.41ms k16=1.32ms k16 faster by 1.07x - * Crossover at ~40-50K rows; 50000 is a conservative round boundary. - * - * k=16 uses a scalar fused loop (not SIMD): the 4-pass reduction - * already yields fewer total iterations than 8-pass k=8+SIMD, and a - * 256KB histogram makes SIMD gather impractical (cache pressure). */ - if (nrows >= 50000u) { - int rc = radix_sort_k16(r, start_row, nrows, workspace, timestamps); - if (rc == 0) - wl_columnar_relation_touch_view(r); - return rc; - } - - /* k=8 SIMD path: 8 passes × 256 buckets (1KB histogram, stack-allocated). - * SIMD-dispatched fused uniform+count avoids a separate gather loop. */ - - bool owns_workspace = workspace == NULL; - uint32_t *perm_a = workspace ? workspace->perm_a - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_a"); - uint32_t *perm_b = workspace ? workspace->perm_b - : (uint32_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(uint32_t), "radix_perm_b"); - uint8_t *bv_cache = workspace ? workspace->byte_values - : (uint8_t *)wl_columnar_relation_radix_malloc( - nrows, "radix_byte_values"); - if (workspace && workspace->k8_capacity < nrows) - return EINVAL; - if (!perm_a || !perm_b || !bv_cache) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - } - return nrows <= 32 ? col_rel_insertion_sort(r, start_row, nrows, - workspace, timestamps) - : ENOMEM; - } - - int64_t *temp_col = workspace ? workspace->temp_column - : (int64_t *)wl_columnar_relation_radix_malloc( - nrows * sizeof(int64_t), "radix_temp_column"); - if (!temp_col) { - if (owns_workspace) { - free(perm_a); - free(perm_b); - free(bv_cache); - } - return ENOMEM; - } - wl_columnar_radix_workspace_t prepared = { - .perm_a = perm_a, .perm_b = perm_b, .temp_column = temp_col, - .byte_values = bv_cache, - }; - int rc = wl_columnar_radix_rows_prepared(r, start_row, nrows, &prepared, - timestamps); - if (owns_workspace) { - free(temp_col); - free(perm_a); - free(perm_b); - free(bv_cache); - } - return rc; - -insertion: - { - int rc = col_rel_insertion_sort(r, start_row, nrows, workspace, - timestamps); - if (rc == 0) - wl_columnar_relation_touch_view(r); - return rc; - } -} - /* Prepared-only kernels. Every buffer and capacity is validated before COW; * no allocator, admission or fallback path is reachable from these kernels. */ static int @@ -10543,238 +10250,6 @@ wl_columnar_relation_radix_workspace_validate(const col_rel_t *rel, workspace->k8_capacity >= nrows ? 0 : EINVAL; } -static int -col_rel_radix_sort_raw(col_rel_t *r, uint32_t start_row, uint32_t nrows, - const wl_columnar_radix_workspace_t *workspace) -{ - if (!r->timestamps || nrows <= 1) - return wl_columnar_relation_radix_sort_rows(r, start_row, nrows, - workspace, NULL); - if (workspace) { - if (!workspace->timestamps || workspace->timestamp_capacity < nrows) - return EINVAL; - return wl_columnar_relation_radix_sort_rows(r, start_row, nrows, - workspace, - workspace->timestamps); - } - size_t bytes = 0; - if (!col_rel_size_multiply(nrows, sizeof(col_delta_timestamp_t), &bytes)) - return ENOMEM; - wl_columnar_memory_reservation_t admission; - wl_columnar_memory_reservation_init(&admission); - bool admitted = false; - if (r->memory_governor) { - wl_columnar_memory_admission_status_t status = - wl_columnar_memory_reserve_checked( - wl_columnar_memory_governor_ref_get(r->memory_governor), - bytes, &admission); - if (status != WL_COLUMNAR_MEMORY_ADMISSION_OK - && status != WL_COLUMNAR_MEMORY_ADMISSION_ADVISORY) { - if (status == WL_COLUMNAR_MEMORY_ADMISSION_DENIED) - r->memory_budget_denial_pending = true; - return ENOMEM; - } - admitted = true; - } - col_delta_timestamp_t *timestamps = wl_columnar_relation_radix_malloc( - bytes, "radix_timestamps"); - int rc = timestamps ? wl_columnar_relation_radix_sort_rows(r, start_row, - nrows, - NULL, timestamps) : ENOMEM; - free(timestamps); - if (admitted) - (void)wl_columnar_memory_release(&admission); - return rc; -} - -static int -col_rel_radix_sort_impl(col_rel_t *r, uint32_t start_row, uint32_t nrows, - bool defer_alias_release, bool *out_alias_release_pending, - const wl_columnar_radix_workspace_t *workspace); - -/* Sort under a writer lease, using the caller's scratch workspace. Shared -* views take the same transactional COW path as the locked/consolidation -* entry point; the alias lease is released after successful publication. */ -int -wl_columnar_relation_radix_sort_with_workspace(col_rel_t *r, - uint32_t start_row, - uint32_t nrows, const wl_columnar_source_access_writer_t *writer, - const wl_columnar_radix_workspace_t *workspace) -{ - col_rel_t *owner = NULL; - int rc; - - if (!r || !workspace || start_row > r->nrows - || nrows > r->nrows - start_row) - return EINVAL; - rc = col_rel_storage_owner_resolve(r, &owner); - if (rc != 0) - return rc; - if (!writer || writer->owner != &owner->source_access - || writer->identity != (uintptr_t)writer - || !wl_columnar_source_access_writer_thread_equal(writer)) - return EINVAL; - /* Contention is reported as EBUSY, matching col_rel_radix_sort_locked. - * Folding it into the EINVAL conjunction above made one family answer a - * caller two different ways for the same condition. */ - if (r == owner && col_rel_storage_alias_borrow_count(owner) > 0) - return EBUSY; - if (r->ncols == 0 || nrows <= 1) - return 0; - bool alias_release_pending = false; - rc = col_rel_radix_sort_impl(r, start_row, nrows, false, - &alias_release_pending, workspace); - if (alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(r); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; - } - return rc; -} - -/* defer_alias_release describes the transactional shape of this call: the - * alias borrow taken by the deferred copy-on-write outlives it and the - * caller retires it later. That is the window the consolidation test hook - * probes, so the hook is keyed on it rather than on which wrapper called in - * -- a caller-identity flag would silently lose hook coverage the moment a - * new consolidation entry point routed through a different wrapper. */ -static int -col_rel_radix_sort_impl(col_rel_t *r, uint32_t start_row, uint32_t nrows, - bool defer_alias_release, bool *out_alias_release_pending, - const wl_columnar_radix_workspace_t *workspace) -{ - if (out_alias_release_pending) - *out_alias_release_pending = false; - if (!r || start_row > r->nrows || nrows > r->nrows - start_row) - return EINVAL; - if (nrows <= 1) - return 0; - - wl_columnar_radix_workspace_t local_workspace = { 0 }; - bool owns_workspace = !workspace && r->memory_governor; - int rc; - if (owns_workspace) { - uint32_t bounds[] = { start_row, start_row + nrows }; - rc = wl_columnar_relation_radix_workspace_prepare(r, bounds, 1, 0, - &local_workspace, false); - if (rc != 0) - return rc; - workspace = &local_workspace; - } - if (workspace) { - rc = wl_columnar_relation_radix_workspace_validate(r, nrows, workspace); - if (rc != 0) - goto cleanup_workspace; - } - col_rel_cow_deferred_t deferred = { 0 }; - uint64_t ledger_before = 0; - uint64_t old_view = 0; - uint64_t old_storage = 0; - bool borrowed = r->col_shared != NULL; - if (borrowed) { - ledger_before = col_rel_owned_ledger_bytes(r); - old_view = r->view_generation; - old_storage = r->storage_generation; - /* Always defer: every caller now holds the alias across the sort - * and releases it afterwards (or hands it back). Releasing inside - * the COW and then rolling back would restore a borrowed view whose - * borrow the owner no longer counts. */ - rc = col_rel_cow_unshare_legacy_impl(r, 0, true, true, &deferred); - if (rc != 0) - goto cleanup_workspace; -#ifdef WL_TEST_CONSOLIDATE_HOOK - if (defer_alias_release - && wl_columnar_consolidation_transition_hook) - wl_columnar_consolidation_transition_hook(r, - WL_COLUMNAR_CONSOLIDATION_TEST_SORT_AFTER_DETACH); -#endif - } - - rc = col_rel_radix_sort_raw(r, start_row, nrows, workspace); - if (rc != 0 && borrowed) { - col_columns_free(r->columns, r->ncols); - r->columns = deferred.columns; - r->col_shared = deferred.shared; - deferred.columns = NULL; - deferred.shared = NULL; - col_rel_retire_payload_credit(r); - col_rel_ledger_reconcile(r, ledger_before); - /* The detach was rolled back; its storage epoch goes with it. */ - r->view_generation = old_view; - r->storage_generation = old_storage; - } else if (borrowed) { - if (!deferred.columns || !deferred.shared) - abort(); - for (uint32_t c = 0; c < r->ncols; c++) - if (!deferred.shared[c]) - free(deferred.columns[c]); - free((void *)deferred.columns); - free(deferred.shared); - deferred.columns = NULL; - deferred.shared = NULL; - col_rel_retire_shared_table_credit(r); - col_rel_retire_payload_credit(r); - if (out_alias_release_pending) - *out_alias_release_pending = true; - } -cleanup_workspace: - if (owns_workspace) - wl_columnar_radix_workspace_destroy(&local_workspace); - return rc; -} - -/* Transitional owner-only authority for Unit 2B3b2 consolidation callers. - * Sort under a writer lease the caller already holds, so a wider mutation - * transaction can keep admission across the sort. - * - * defer_alias_release selects what happens to the borrow the deferred COW - * takes: false releases it here, true hands it back through - * *out_alias_release_pending for the caller's own cleanup path. The out - * parameter is mandatory in both modes -- a caller that meant to defer but - * passed NULL would otherwise have its borrow released out from under a - * transaction that still believes it holds one, so that mistake is refused - * rather than guessed at. - * - * Validation runs before the range shortcut, so a foreign, absent or - * wrong-thread writer is refused at every row count. The alias predicate - * is deliberately r == owner rather than the unconditioned form used by - * col_rel_reset_rows_locked: sorting an alias while its owner has live - * borrows is exactly the consolidation case and must stay admissible. */ -int -col_rel_radix_sort_locked(col_rel_t *r, uint32_t start_row, uint32_t nrows, - const wl_columnar_source_access_writer_t *writer, - bool defer_alias_release, bool *out_alias_release_pending) -{ - col_rel_t *owner = NULL; - int rc; - - if (!out_alias_release_pending) - return EINVAL; - *out_alias_release_pending = false; - if (!r || start_row > r->nrows || nrows > r->nrows - start_row) - return EINVAL; - rc = col_rel_storage_owner_resolve(r, &owner); - if (rc != 0) - return rc; - if (!writer || writer->owner != &owner->source_access - || writer->identity != (uintptr_t)writer - || !wl_columnar_source_access_writer_thread_equal(writer)) - return EINVAL; - if (r == owner && col_rel_storage_alias_borrow_count(owner) > 0) - return EBUSY; - if (r->ncols == 0 || nrows <= 1) - return 0; - rc = col_rel_radix_sort_impl(r, start_row, nrows, defer_alias_release, - out_alias_release_pending, NULL); - if (!defer_alias_release && *out_alias_release_pending) { - int alias_rc = col_rel_storage_alias_release(r); - if (alias_rc != 0 && rc == 0) - rc = alias_rc; - *out_alias_release_pending = false; - } - return rc; -} - /* Validate authority and physical shape before reading keys or allocating. */ static int wl_columnar_radix_lease_preflight(col_rel_t *r, uint32_t start, @@ -11023,8 +10498,7 @@ wl_columnar_relation_radix_sequence_execute_with_lease( return rc; } if (r->col_shared) { - int rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, - lease); + int rc = col_rel_cow_unshare_publish_impl(r, 0, false, lease); if (rc) return rc; } @@ -11089,7 +10563,7 @@ wl_columnar_relation_radix_sort_with_lease_impl(col_rel_t *r, uint32_t start, || (void *)workspace->short_values != workspace->bucket_values) return rc ? rc : EINVAL; if (r->col_shared) { - rc = col_rel_cow_unshare_publish_impl(r, 0, false, false, NULL, lease); + rc = col_rel_cow_unshare_publish_impl(r, 0, false, lease); if (rc) return rc; #ifdef WL_TEST_CONSOLIDATE_HOOK From 458c7aa7562f231c9e7393c3dbfe3c75dcb0791b Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 02:41:48 +0900 Subject: [PATCH 12/15] Preserve denial evidence across leased appends --- tests/test_relation_generations.c | 55 +++++++++++++++++++++++-------- wirelog/columnar/relation.c | 1 - 2 files changed, 42 insertions(+), 14 deletions(-) diff --git a/tests/test_relation_generations.c b/tests/test_relation_generations.c index 98b27e70..cfc6ee86 100644 --- a/tests/test_relation_generations.c +++ b/tests/test_relation_generations.c @@ -251,6 +251,8 @@ generation_test_append_with_lease(col_rel_t *rel, const int64_t *row) int rc = generation_test_mutation_acquire(rel, &held); if (rc != 0) return rc; + /* This helper models a new outer admitted append attempt. */ + rel->memory_budget_denial_pending = false; rc = col_rel_append_row_with_lease(rel, row, &held.lease); int finish_rc = generation_test_mutation_finish(&held, rc == 0); return rc ? rc : finish_rc; @@ -4648,9 +4650,11 @@ test_append_overlapping_input(bool locked) "public append overlap governed relation"); wl_columnar_memory_governor_t *governor = wl_columnar_memory_governor_ref_get(ref); - if (locked) + if (locked) { CHECK(generation_test_mutation_acquire(rel, &held) == 0, "append overlap mutation admission"); + rel->memory_budget_denial_pending = false; + } uint64_t credit = wl_columnar_memory_reserved(governor); col_rel_t before = *rel; const int64_t *inside = &rel->columns[0][3]; @@ -4677,7 +4681,8 @@ test_append_overlapping_input(bool locked) int failed_rc = append_overlap_attempt(rel, inside, locked ? &held : NULL); allocation_fail_at = -1; - CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending + CHECK(failed_rc == ENOMEM + && rel->memory_budget_denial_pending == (locked && wide) && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == credit, "wide staging malloc failure returns all scratch credit"); @@ -4690,7 +4695,8 @@ test_append_overlapping_input(bool locked) allocation_fail_at = fail; int failed_rc = append_overlap_attempt(rel, inside, &held); allocation_fail_at = -1; - CHECK(failed_rc == ENOMEM && !rel->memory_budget_denial_pending + CHECK(failed_rc == ENOMEM + && rel->memory_budget_denial_pending == wide && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == credit, "locked growth failure after staging rolls back all credit"); @@ -4699,7 +4705,7 @@ test_append_overlapping_input(bool locked) wl_columnar_relation_test_fail_next_reservation_commit(); int publish_rc = append_overlap_attempt(rel, inside, &held); CHECK(publish_rc == ENOMEM - && !rel->memory_budget_denial_pending + && rel->memory_budget_denial_pending == wide && mutation_payload_unchanged(rel, &before) && wl_columnar_memory_reserved(governor) == credit, "locked publication failure returns staged scratch credit"); @@ -4725,7 +4731,7 @@ test_append_overlapping_input(bool locked) locked ? &held : NULL); allocation_fail_at = -1; CHECK(external_rc == 0 && external_calls == 0 && inside_rc == ENOMEM - && !rel->memory_budget_denial_pending + && rel->memory_budget_denial_pending == locked && wl_columnar_memory_reserved(governor) == credit, "no-overlap append does not consume a wide-scratch malloc fault"); #endif @@ -4788,6 +4794,7 @@ test_leased_append_alias_overlap(void) = wl_columnar_memory_governor_ref_get(ref); CHECK(generation_test_mutation_acquire(alias, &held) == 0, "locked alias mutation admission"); + alias->memory_budget_denial_pending = false; uint64_t credit = wl_columnar_memory_reserved(governor); col_rel_t root_before = *root, before = *alias, sibling_before = *sibling; @@ -4820,7 +4827,7 @@ test_leased_append_alias_overlap(void) &held.lease); allocation_fail_at = -1; CHECK(failed_rc == ENOMEM && - !alias->memory_budget_denial_pending + alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) && mutation_payload_unchanged(sibling, &sibling_before) @@ -4833,7 +4840,7 @@ test_leased_append_alias_overlap(void) int prepare_rc = col_rel_append_row_with_lease(alias, inside, &held.lease); CHECK(prepare_rc == ENOMEM - && !alias->memory_budget_denial_pending + && alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) && mutation_payload_unchanged(sibling, &sibling_before) @@ -4863,7 +4870,7 @@ test_leased_append_alias_overlap(void) wl_columnar_relation_test_fail_next_reservation_commit(); CHECK(col_rel_append_row_with_lease(alias, inside, &held.lease) == ENOMEM - && !alias->memory_budget_denial_pending + && alias->memory_budget_denial_pending && mutation_payload_unchanged(alias, &before) && mutation_payload_unchanged(root, &root_before) && mutation_payload_unchanged(sibling, &sibling_before) @@ -4967,23 +4974,43 @@ test_leased_append_authority_and_edges(void) rel->memory_budget_denial_pending = true; CHECK(col_rel_append_row_with_lease(rel, (const int64_t *)(UINTPTR_MAX - 3), &held.lease) == EOVERFLOW - && !rel->memory_budget_denial_pending + && rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before), - "admitted row endpoint overflow clears evidence without dereference"); + "nested row endpoint overflow preserves denial evidence without dereference"); int64_t *column = rel->columns[0]; rel->columns[0] = (int64_t *)(UINTPTR_MAX - 3); CHECK(col_rel_append_row_with_lease(rel, &row, &held.lease) == EOVERFLOW - && !rel->memory_budget_denial_pending, + && rel->memory_budget_denial_pending, "column endpoint overflow precedes dereference"); rel->columns[0] = column; rel->timestamp_capacity = 1; CHECK(col_rel_append_row_with_lease(rel, &row, &held.lease) == EINVAL - && !rel->memory_budget_denial_pending, - "admitted shape rejection clears evidence"); + && rel->memory_budget_denial_pending, + "nested shape rejection preserves denial evidence"); rel->timestamp_capacity = 0; CHECK(generation_test_mutation_finish(&held, false) == 0, "lease edge release"); cleanup_relations(); + + col_rel_t *nested = new_relation(); + generation_test_mutation_t nested_held = { 0 }; + CHECK(nested && generation_test_mutation_acquire(nested, &nested_held) == 0, + "nested append lease acquisition"); + nested->memory_budget_denial_pending = true; + CHECK(col_rel_append_row_with_lease(nested, &row, &nested_held.lease) == 0 + && nested->memory_budget_denial_pending, + "successful nested append preserves outer quota-denial evidence"); + CHECK(generation_test_mutation_finish(&nested_held, true) == 0, + "nested append lease finish"); + cleanup_relations(); + + col_rel_t *public_retry = new_relation(); + CHECK(public_retry != NULL, "public stale-evidence relation"); + public_retry->memory_budget_denial_pending = true; + CHECK(col_rel_append_row(public_retry, &row) == 0 + && !public_retry->memory_budget_denial_pending, + "public append starts a fresh denial-evidence attempt"); + cleanup_relations(); } static void @@ -5075,6 +5102,7 @@ test_leased_append_generation_boundaries(void) int append_rc = col_rel_append_row_with_lease(rel, rel->columns[0], &held.lease); CHECK(append_rc == EOVERFLOW + && rel->memory_budget_denial_pending && mutation_payload_unchanged(rel, &before), "locked epoch boundary rejects before staging or publication"); CHECK(generation_test_mutation_finish(&held, false) == 0, @@ -5099,6 +5127,7 @@ test_leased_append_generation_boundaries(void) generation_test_mutation_t held = { 0 }; CHECK(generation_test_mutation_acquire(rel, &held) == 0, "epoch success mutation admission"); + rel->memory_budget_denial_pending = false; CHECK(col_rel_append_row_with_lease(rel, rel->columns[0], &held.lease) == 0 && rel->nrows == rows + 1 && rel->columns[0][rows] == row[0] diff --git a/wirelog/columnar/relation.c b/wirelog/columnar/relation.c index 6dd00000..24410692 100644 --- a/wirelog/columnar/relation.c +++ b/wirelog/columnar/relation.c @@ -4836,7 +4836,6 @@ col_rel_append_row_with_lease(col_rel_t *r, const int64_t *row, != WL_COLUMNAR_RELATION_PAYLOAD_MUTATION || col_rel_mutation_lease_validate(lease, r)) return EINVAL; - r->memory_budget_denial_pending = false; return col_rel_append_row_impl(r, row, &lease->set->owners[lease->owner_slot].writer, true, lease); } From 82c298678924f71170fb35b13db79d0730e4584a Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 02:43:55 +0900 Subject: [PATCH 13/15] Document mutation and radix atomics --- docs/THREADING.md | 11 ++++++++--- scripts/ci/check-threading-doc.sh | 2 +- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/docs/THREADING.md b/docs/THREADING.md index 742ff6ca..405e8382 100644 --- a/docs/THREADING.md +++ b/docs/THREADING.md @@ -277,7 +277,7 @@ The 64-byte padding between `tail` and `head` cache-line ping-pong between producer and consumer collapses throughput by 2-10x. -### 5.3 Non-explicit atomic APIs — init and relation identity (5 rows) +### 5.3 Non-explicit atomic APIs — init and relation identity (8 rows) | Anchor (`file:function[#N]`) | Field | Op | Order | Justification | |---|---|---|---|---| @@ -286,6 +286,9 @@ throughput by 2-10x. | `relation.c:col_rel_new_identity` | `wl_next_relation_identity` | `atomic_load_explicit` | `relaxed` | Read the candidate identity before the non-wrapping CAS reservation loop; the counter only has to hand out distinct values, no other memory is published through it | | `relation.c:col_rel_new_identity#2` | `wl_next_relation_identity` | `atomic_compare_exchange_weak_explicit` | `relaxed`/`relaxed` | Reserve a unique relation identity and retry with the observed value after a lost race; uniqueness comes from the RMW, not from ordering | | `relation.c:col_rel_test_set_next_identity` | `wl_next_relation_identity` | `atomic_store_explicit` | `relaxed` | Test-only seam for selecting the terminal allocator state; production allocation is not concurrent with this reset | +| `relation.c:col_rel_mutation_set_nonce_allocate` | `wl_next_mutation_set_nonce` | `atomic_load_explicit` | `relaxed` | Read the candidate nonce before the nonwrapping CAS reservation loop; the counter only supplies unique admission provenance and publishes no other state | +| `relation.c:col_rel_mutation_set_nonce_allocate#2` | `wl_next_mutation_set_nonce` | `atomic_compare_exchange_weak_explicit` | `relaxed`/`relaxed` | Reserve a unique nonzero mutation-set nonce and retry with the observed value after a lost race; the RMW provides uniqueness without publishing payload state | +| `relation.c:wl_columnar_relation_test_set_mutation_nonce` | `wl_next_mutation_set_nonce` | `atomic_store_explicit` | `relaxed` | Test-only seam selects allocator exhaustion or retry states before test admissions; tests do not race this reset with nonce allocation | These are the sites in `wirelog/` that use the **non-explicit** atomic APIs (`atomic_load`/`atomic_store`); they default to `memory_order_seq_cst`. @@ -514,7 +517,7 @@ those slots and arena allocations are quiescent. | `source_access.h:wl_columnar_source_access_writer_release#2` | `gate->state` | `atomic_compare_exchange_weak_explicit` | release/relaxed | Publish writer payload completion and retry spurious failure | | `source_access.h:wl_columnar_source_access_writer_move` | `src->owner->state` | `atomic_load_explicit` | acquire | Confirm that the source token still holds WRITER before moving its address-bound ownership to another token | -### 5.14 `wirelog/columnar/relation.c` and `session.c` — alias ownership and pool promotion (28 rows) +### 5.14 `wirelog/columnar/relation.c` and `session.c` — alias ownership and pool promotion (30 rows) The canonical owner's flattened alias count uses `wl_atomic_u64` because a quiesced worker can retire its alias while unrelated readers still hold the @@ -540,6 +543,8 @@ concurrent alias removals cannot underflow the count. | `relation.c:col_rel_destroy_checked#2` | `r->source_access.state` | `atomic_store_explicit` | release | Keep the retired pool slot closed until allocator reset or reuse | | `relation.c:col_rel_destroy_checked#3` | `r->descriptor_access.state` | `atomic_store_explicit` | release | Keep the retired descriptor closed until allocator reset or reuse | | `relation.c:col_rel_install_shared_view_unprotected` | `dst->storage_alias_borrows` | `atomic_store_explicit` | relaxed | A newly installed alias descriptor has no child aliases of its own | +| `relation.c:wl_columnar_memory_reservation_inert` | `reservation->owner_bits` | `atomic_load_explicit` | relaxed | Confirm a prepared radix workspace reservation has no owner before reusing its caller-owned token storage; the helper is called while the exact mutation lease stabilizes the relation and workspace | +| `relation.c:wl_columnar_memory_reservation_inert#2` | `reservation->state` | `atomic_load_explicit` | relaxed | Confirm the token is inert before workspace preparation; this is an initialization check, not a concurrent ownership decision, under the caller's mutation lease | | `relation.c:wl_columnar_relation_rebind_permit_valid` | `lease->owner->state` | `atomic_load_explicit` | acquire | Validate that the upgraded canonical source gate holds WRITER before publication | | `relation.c:wl_columnar_relation_rebind_permit_valid#2` | `lease->secondary_owner->state` | `atomic_load_explicit` | acquire | Validate that the upgraded destination descriptor gate holds WRITER before publication | | `relation.c:wl_columnar_relation_install_shared_view_with_lease` | `dst->descriptor_access.state` | `atomic_compare_exchange_weak_explicit` | acquire/relaxed | Upgrade the sole transferable descriptor reader to exclusive descriptor admission and retry spurious failure | @@ -680,7 +685,7 @@ the committed token after publication and before a growth transaction. | `eval_dedup.c:wl_columnar_eval_dedup_test_fail_next_growth_alloc` | test-only fault flag | `atomic_store_explicit` | release | Arm one allocation refusal before a test invokes dedup growth; excluded from the production library | | `eval_dedup.c:wl_columnar_eval_dedup_set_grow` | test-only fault flag | `atomic_exchange_explicit` | acquire-release | Consume the one-shot fault safely when test workers grow dedup tables; excluded from the production library | -The complete source audit now contains **214 atomic call sites**. +The complete source audit now contains **219 atomic call sites**. --- diff --git a/scripts/ci/check-threading-doc.sh b/scripts/ci/check-threading-doc.sh index 8fa49bc7..d2b84c6f 100755 --- a/scripts/ci/check-threading-doc.sh +++ b/scripts/ci/check-threading-doc.sh @@ -19,7 +19,7 @@ fi rows="$tmp_dir/rows" sed -nE 's/^\| `([^`]+:[A-Za-z_][A-Za-z0-9_]*(#[0-9]+)?)` \| [^|]* \| `([^`]*)` \|.*/\1\t\3/p' "$doc" >"$rows" row_count=$(wc -l <"$rows") -expected_rows="${WIRELOG_THREADING_EXPECTED_ROWS:-214}" +expected_rows="${WIRELOG_THREADING_EXPECTED_ROWS:-219}" [ "$row_count" -eq "$expected_rows" ] || { echo "check-threading-doc: FAIL: expected $expected_rows audit rows, found $row_count" >&2 exit 1 From 99692f09eab97830d9d1880e04f74537828b981b Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 02:48:12 +0900 Subject: [PATCH 14/15] Correct threading inventory arithmetic and ordering notes --- docs/THREADING.md | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/docs/THREADING.md b/docs/THREADING.md index 405e8382..73b45c97 100644 --- a/docs/THREADING.md +++ b/docs/THREADING.md @@ -277,7 +277,7 @@ The 64-byte padding between `tail` and `head` cache-line ping-pong between producer and consumer collapses throughput by 2-10x. -### 5.3 Non-explicit atomic APIs — init and relation identity (8 rows) +### 5.3 Init and relation identity/nonce allocators (8 rows) | Anchor (`file:function[#N]`) | Field | Op | Order | Justification | |---|---|---|---|---| @@ -290,13 +290,14 @@ throughput by 2-10x. | `relation.c:col_rel_mutation_set_nonce_allocate#2` | `wl_next_mutation_set_nonce` | `atomic_compare_exchange_weak_explicit` | `relaxed`/`relaxed` | Reserve a unique nonzero mutation-set nonce and retry with the observed value after a lost race; the RMW provides uniqueness without publishing payload state | | `relation.c:wl_columnar_relation_test_set_mutation_nonce` | `wl_next_mutation_set_nonce` | `atomic_store_explicit` | `relaxed` | Test-only seam selects allocator exhaustion or retry states before test admissions; tests do not race this reset with nonce allocation | -These are the sites in `wirelog/` that use the **non-explicit** atomic APIs -(`atomic_load`/`atomic_store`); they default to `memory_order_seq_cst`. -The identity allocator uses the same default ordering because the CAS loop -must reserve each relation identity without reuse; the test-only store is -only used to exercise allocator exhaustion. +Only the two `io_adapter.c` rows use non-explicit atomic APIs, which default +to `memory_order_seq_cst`. The relation identity and mutation-set nonce +allocator rows use the explicit relaxed order shown in the table: their CAS +operations reserve unique nonwrapping values and do not publish payload state. +The test-only stores reset allocator state before tests and do not race with +allocation. -### 5.4 `wirelog/columnar/join.c` — keyed-join cancel/budget and typed output (20 rows) +### 5.4 `wirelog/columnar/join.c` — keyed-join cancel/budget and typed output (19 rows) | Anchor (`file:function[#N]`) | Field | Op | Order | Justification | |---|---|---|---|---| @@ -465,7 +466,7 @@ measured by `bench/bench_intern.c`; baselines are in `docs/INTERN_PERF.md` ### 5.12 Existing inventory total -21 + 4 + 5 + 19 + 1 + 1 + 1 + 37 + 5 + 7 + 3 = **104 atomic call sites** +21 + 4 + 8 + 19 + 2 + 2 + 3 + 37 + 5 + 7 + 3 = **111 atomic call sites** before the source-access contract below. ### 5.13 `wirelog/columnar/source_access.h` — relation source gate (21 rows) @@ -560,7 +561,7 @@ concurrent alias removals cannot underflow the count. | `session.c:session_pool_rel_promote#2` | `src->retained_reservation.owner_bits` | `atomic_load_explicit` | acquire | Promote a committed reservation only when the pool slot still owns it | | `session.c:session_pool_rel_promote#3` | `src->storage_alias_borrows` | `atomic_store_explicit` | relaxed | Leave the closed pool tombstone with no child aliases | -104 + 21 + 28 = **153 atomic call sites**. +111 + 21 + 30 = **162 atomic call sites**. The `#N` suffix counts all atomic sites in a symbol, regardless of operation; the first site remains unsuffixed. `scripts/ci/check-threading-doc.sh` uses From f237724b21da4b752a5d0cab2168cc0f8204a3b3 Mon Sep 17 00:00:00 2001 From: Justin Kim Date: Sun, 4 Oct 2026 02:50:07 +0900 Subject: [PATCH 15/15] Correct threading replacement inventory row count --- docs/THREADING.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/THREADING.md b/docs/THREADING.md index 73b45c97..c014a4d9 100644 --- a/docs/THREADING.md +++ b/docs/THREADING.md @@ -621,7 +621,7 @@ restores them after moving the image's current reservations to the source. | `relation.c:col_rel_mutable_image_commit#5` | `old.storage_alias_borrows` | `atomic_store_explicit` | relaxed | Clear the stack retirement copy's alias count before physical cleanup; the copy is private | ### 5.16 `wirelog/columnar/memory_governor.c` and `relation.c` — atomic -replacement admission and compaction (19 rows) +replacement admission and compaction (27 rows) Replacement admission temporarily accounts for the new footprint while the old reservation remains committed. The overlap CAS is the admission