diff --git a/docs/roadmap/CVCGL-UI-DSL-ROADMAP.md b/docs/roadmap/CVCGL-UI-DSL-ROADMAP.md index 0af2c135..d4ef49bf 100644 --- a/docs/roadmap/CVCGL-UI-DSL-ROADMAP.md +++ b/docs/roadmap/CVCGL-UI-DSL-ROADMAP.md @@ -3293,8 +3293,25 @@ render/resolver thread**. wasm: `emscripten_fetch` is async-only (no sync on the > ari-generated blocks the render thread today; a DSL program that wants async loading uses `(fetch)`. > **NOT YET:** chunked/streaming bodies; the dedicated libcurl-`multi` I/O thread (today each in-flight > fetch blocks one compute-pool worker); an async scene `source:{uri:}` realize (so a large remote -> asset doesn't block scene setup — a follow-up); render-pass park budget (separate PR). NATIVE only -> until the wasm worker/`-sASYNCIFY` fetch path is confirmed. +> asset doesn't block scene setup — a follow-up). NATIVE only until the wasm worker/`-sASYNCIFY` fetch +> path is confirmed. +> +> **LANDED (render-pass park safety).** Two layers keep a render-pass program that awaits/parks from +> breaking the render thread. (1) The reactive READ lane (visible_when/computed) is structurally +> park-proof: it evaluates on a private idle scheduler that is never pumped and its allowlist excludes +> `await`/`msg-recv`/`http-get`/`fetch`, so a park verb in a predicate is DENIED (unbound → fail-safe +> hidden, render returns) — asserted by a unit test over both `msg-recv` and `await`. (2) Residents + +> actions run in `drain()`, off the draw walk, bounded by the pump's per-drain cap, so a per-node DSL +> loop degrades but never hangs; and a **per-STEP wall-clock cap** (`process.max_step_time`, wired for +> residents in `ensure_resident` at 50 ms — the same `eval_deadline_guard` an action's `max_time` +> arms, but fresh each step so it never kills a long-lived resident) is the backstop against an +> in-STEP loop (a runaway native builtin) that the pump can't interrupt: it is aborted + the process +> killed. A latent double-count of a parking step's slice into `accumulated_time` (the accrual re-added +> a slice a parking intrinsic already booked) is fixed alongside. **NOT killing a *persistently +> spinning* resident** (vs the pump bounding it each drain) is deferred: the naive per-activation +> budget that would do so false-kills a legitimate input-event burst (buffered events drain sharing one +> budget) and misses a `sleep(0)`/self-wake spin (resets every wake) — distinguishing a genuine +> external wait from a self-spin needs more than a wake-count, so it is left as future work. ### 13.9 A cvc::app-wide HTTP(s) cache in the state tree (TTL + conditional GET) — *planned* diff --git a/inc/cvc/core/state_exec/process.h b/inc/cvc/core/state_exec/process.h index f913fc20..08c77d34 100644 --- a/inc/cvc/core/state_exec/process.h +++ b/inc/cvc/core/state_exec/process.h @@ -45,28 +45,36 @@ struct process { process_status status = process_status::ready; // Scheduling - int priority = 0; // Nice value: -20 (high) to +19 (low) - std::string uid; // User identity - std::string gid; // Group identity - std::string root_path; // Chroot path (empty = full tree) - std::string owner; // Owner-scope tag (e.g. an Ariadne document/Runtime); "" = unowned. - // Lets a host reap a whole process group on teardown (kill_owner). + int priority = 0; // Nice value: -20 (high) to +19 (low) + std::string uid; // User identity + std::string gid; // Group identity + std::string root_path; // Chroot path (empty = full tree) + std::string owner; // Owner-scope tag (e.g. an Ariadne document/Runtime); "" = unowned. + // Lets a host reap a whole process group on teardown (kill_owner). bool awaiting_frame = false; // (await expr) parked this process until the next frame boundary; // re-readied by async_scheduler::wake_awaiting() once per pump. // Resource limits (0 = unlimited) uint64_t max_steps = 0; - double max_time = 0.0; // Wall-clock seconds + double max_time = 0.0; // Wall-clock seconds (TOTAL run-time, across all activations) uint64_t max_memory = 0; // Bytes uint64_t max_messages = 0; // Outbound message count uint64_t max_message_bytes = 0; // Total outbound message bytes + // §13.8 per-STEP wall-clock cap (0 = unlimited): the deadline armed for a SINGLE evaluator step, + // fresh each step (never accumulates), so a long-lived resident with total max_time == 0 still + // can never HANG the host's per-frame drain — a single step that loops in-place (e.g. a runaway + // native builtin) is aborted at the cap and the process killed, while a healthy step (far under + // the cap) and a normal event-loop are untouched. (A per-node DSL loop yields per step, so the + // pump's own per-drain cap bounds it; this cap is the backstop for an in-STEP loop the pump can't + // interrupt.) + double max_step_time = 0.0; // Evaluator state (stackless — serializable) evaluator_state state; // Timing std::chrono::steady_clock::time_point create_time; - double accumulated_time = 0.0; // Seconds running so far + double accumulated_time = 0.0; // Seconds running so far (TOTAL run-time; parked time uncounted) std::chrono::steady_clock::time_point last_run_start; // Sleep support: process is in `waiting` status until this deadline. diff --git a/inc/cvc/core/state_exec/scheduler_base.h b/inc/cvc/core/state_exec/scheduler_base.h index 6062d1c8..623d8378 100644 --- a/inc/cvc/core/state_exec/scheduler_base.h +++ b/inc/cvc/core/state_exec/scheduler_base.h @@ -53,7 +53,8 @@ struct execute_options { std::string root_path; // Chroot: confine to subtree (empty = full tree) std::string owner; // Owner-scope tag (Ariadne document/Runtime); "" = unowned uint64_t max_steps = 0; - double max_time = 0.0; + double max_time = 0.0; // TOTAL wall-clock run-time seconds + double max_step_time = 0.0; // §13.8 per-STEP wall-clock cap (fresh each step); 0 = unlimited uint64_t max_memory = 0; uint64_t max_messages = 0; uint64_t max_message_bytes = 0; diff --git a/src/cvc/ariadne/ariadne.cpp b/src/cvc/ariadne/ariadne.cpp index ca34fff0..ecd45a35 100644 --- a/src/cvc/ariadne/ariadne.cpp +++ b/src/cvc/ariadne/ariadne.cpp @@ -1479,8 +1479,15 @@ void Runtime::Impl::ensure_resident(const std::string &channel, const std::strin se::execute_options opts; opts.env = ac->env; opts.owner = owner_; // reaped by ~Impl's kill_owner(owner_) on teardown - opts.max_steps = 0; // unlimited total — see above - opts.max_time = 0.0; + opts.max_steps = 0; // unlimited total — a resident loops forever (see above) + opts.max_time = 0.0; // unlimited TOTAL run-time; the per-STEP cap below is the hang backstop + // §13.8 resident per-STEP wall-clock cap: a resident has no total budget, so give it a per-step + // deadline (fresh each step) so a single step that loops in-place (a runaway native builtin) + // can't HANG the host's per-frame drain — it is aborted + the resident killed. A healthy + // on:tick/on_key body is far under this, and a per-node DSL loop is bounded by the pump's own + // per-drain cap; this only bites an in-STEP loop the pump can't interrupt. + constexpr double kResidentStepSeconds = 0.05; + opts.max_step_time = kResidentStepSeconds; pid_slot = sched.execute(wrapped, opts); ctx_slot = std::move(ac); } catch (const std::exception &e) { diff --git a/src/cvc/core/state_exec/async_scheduler.cpp b/src/cvc/core/state_exec/async_scheduler.cpp index e2ddbab1..163eadbb 100644 --- a/src/cvc/core/state_exec/async_scheduler.cpp +++ b/src/cvc/core/state_exec/async_scheduler.cpp @@ -30,6 +30,7 @@ int async_scheduler::execute(const std::string &script, const execute_options &o proc.owner = opts.owner; proc.max_steps = opts.max_steps; proc.max_time = opts.max_time; + proc.max_step_time = opts.max_step_time; proc.max_memory = opts.max_memory; proc.max_messages = opts.max_messages; proc.max_message_bytes = opts.max_message_bytes; @@ -55,6 +56,7 @@ int async_scheduler::execute(const value_t &expr, const execute_options &opts) { proc.owner = opts.owner; proc.max_steps = opts.max_steps; proc.max_time = opts.max_time; + proc.max_step_time = opts.max_step_time; proc.max_memory = opts.max_memory; proc.max_messages = opts.max_messages; proc.max_message_bytes = opts.max_message_bytes; @@ -155,6 +157,12 @@ void async_scheduler::execute_process_step(process &proc) { const double remaining = proc.max_time - proc.elapsed_time(); step_budget = remaining > 0.0 ? remaining : 0.0; } + // §13.8 per-STEP cap: bound a SINGLE step's wall time too (fresh each step, never accumulates), + // so a resident with total max_time == 0 still can't HANG the host's drain on an in-step loop. A + // healthy step is far under the cap; a per-node DSL loop yields per step and is bounded by the + // pump's own per-drain cap instead. Take the tighter of the two. + if (proc.max_step_time > 0.0) + step_budget = step_budget ? std::min(*step_budget, proc.max_step_time) : proc.max_step_time; eval_deadline_guard step_deadline(step_budget); // Drive one async_stackless_evaluator step to completion at the leaf. The evaluator's step() // is a coroutine that runs one inner stackless step then yields a suspend_point; sync_wait @@ -179,7 +187,12 @@ void async_scheduler::execute_process_step(process &proc) { } auto now = std::chrono::steady_clock::now(); - proc.accumulated_time += std::chrono::duration(now - proc.last_run_start).count(); + // A parking intrinsic (sleep / yield-frame / receive_message) already closed out this running + // slice into accumulated_time and flipped status to `waiting`; accrue here only while still + // running, so a parking step's slice is not double-counted (mirrors process::elapsed_time()'s own + // status guard — previously the unconditional add booked the pre-park slice twice). + if (proc.status == process_status::running) + proc.accumulated_time += std::chrono::duration(now - proc.last_run_start).count(); if ((proc.in_signal_handler || proc.in_watch_handler) && proc.state.done) { restore_from_signal(proc); @@ -669,6 +682,7 @@ int async_scheduler::fork(int pid) { child->root_path = parent.root_path; child->max_steps = parent.max_steps; child->max_time = parent.max_time; + child->max_step_time = parent.max_step_time; child->max_memory = parent.max_memory; child->max_messages = parent.max_messages; child->max_message_bytes = parent.max_message_bytes; diff --git a/src/cvc/tests/ariadne_runtime_test.cpp b/src/cvc/tests/ariadne_runtime_test.cpp index 692780be..d1d90901 100644 --- a/src/cvc/tests/ariadne_runtime_test.cpp +++ b/src/cvc/tests/ariadne_runtime_test.cpp @@ -1367,6 +1367,52 @@ TEST(AriadneReactive, RunawayPredicateIsCappedNotHung) { EXPECT_NE(warns[0].find("budget"), std::string::npos); } +// PR-C render-pass park safety: a park verb (msg-recv/await/http-get/fetch) in a reactive predicate +// is NOT in the read-lane allowlist, so it is DENIED — the render walk cannot park (its private +// scheduler is never pumped; a park there would silent-nil). render() must return, hidden + +// reported. +TEST(AriadneReactive, ParkVerbInPredicateIsDeniedNotHung) { + if (!have_state_exec()) + GTEST_SKIP(); + // Both park verbs (msg-recv AND await) must be denied in the reactive read lane — neither is in + // the read-lane allowlist, so a predicate using one is unbound → fail-safe hidden, render + // returns. + for (const char *predicate : {"(msg-recv \"x\")", "(await 1)"}) { + cvc::app app; + Runtime rt(app, ""); + MockBackend mb; + rt.set_backend(&mb); + Widget w = text("shown"); + w.visible_when = predicate; + rt.set_root(group({w})); + rt.render(); // MUST return — the read lane cannot park/block + EXPECT_FALSE(mb.saw("text_line:shown")) << "predicate: " << predicate; // denied -> hidden + EXPECT_FALSE(rt.take_reactive_warnings().empty()) << "predicate: " << predicate; // and reported + } +} + +// PR-C end-to-end: a runaway on:tick resident (a body that never parks) is bounded by the +// per-activation budget wired in ensure_resident — drain() returns, and the scheduler recovers so a +// subsequently-installed healthy resident still fires. +TEST(AriadneResident, RunawayTickResidentIsBoundedAndSchedulerRecovers) { + if (!have_state_exec()) + GTEST_SKIP(); + cvc::app app; + Runtime rt(app, ""); + MockBackend mb; + rt.set_backend(&mb); + rt.set_tick_program("(while true (+ 1 1))"); // runaway: the body never parks + rt.render(); + rt.drain(); // MUST return (per-activation budget aborts the body; the pump caps regardless) + // Replace with a healthy resident; if the runaway had wedged the scheduler this would never fire. + rt.set_tick_program("(state-set \"tick.flag\" \"1\")"); + for (int i = 0; i < 4; ++i) { + rt.render(); + rt.drain(); + } + EXPECT_EQ(cvc::state::instance(app)("tick.flag").value(), "1"); +} + TEST(AriadneReactive, StringValueIsTruthyAndUnsetKeyIsCleanlyFalsy) { if (!have_state_exec()) GTEST_SKIP(); diff --git a/src/cvc/tests/state_exec_async_test.cpp b/src/cvc/tests/state_exec_async_test.cpp index dae0cb7e..c30253cb 100644 --- a/src/cvc/tests/state_exec_async_test.cpp +++ b/src/cvc/tests/state_exec_async_test.cpp @@ -634,6 +634,27 @@ TEST_F(AsyncSchedulerIntrinsicsTest, AwaitIntrinsicParksThenResumesWithValue) { EXPECT_EQ(std::get(it->second.v), 42); // resumed with the awaited value } +// §13.8 per-STEP cap: max_step_time arms a fresh per-step deadline (fresh each step, so it never +// accumulates and cannot kill a long-lived resident) — the SAME eval_deadline_guard mechanism the +// action time-limit uses, just applied to residents so an in-step loop can't hang the drain. This +// checks the wiring + that a NORMAL body under the cap completes untouched (no false-kill); the +// abort-on-overrun behaviour is covered by the existing action time-limit tests (same guard). +TEST_F(AsyncSchedulerIntrinsicsTest, PerStepCapLeavesHealthyWorkUntouched) { + execute_options opts; + opts.env = env; + opts.max_step_time = 0.05; // a generous per-step cap; every normal step is far under it + int pid = + sched.execute(std::string("(begin (set s 0) (set i 0) " + "(while (< i 200) (begin (set s (+ s i)) (set i (+ i 1)))) s)"), + opts); + auto results = sched.sync_run(2000000, 1.0); + auto it = results.find(pid); + ASSERT_NE(it, results.end()); // completed, not killed + EXPECT_EQ(std::get(it->second.v), 19900); // sum 0..199 + auto info = sched.get_process_info(pid); + EXPECT_TRUE(!info.has_value() || info->status != process_status::killed); +} + TEST(AsyncSchedulerTest, KillOwnerReapsProcessGroup) { async_scheduler sched; execute_options a;