From 0dec6756a815aa13d779cca80f44a2d1dceb9fd6 Mon Sep 17 00:00:00 2001 From: Christina Quast Date: Sun, 4 Oct 2026 19:20:32 +0200 Subject: [PATCH] orchestrator: Say start, not arm, outside the watchdogs BootWatch::arm becomes BootWatch::start and Phase::Armed becomes Phase::Started. The trait doc already said "starts a fresh attempt", so the name now says what the method does. Arm stays where a deadline is involved: the boot and commit watchdogs, the timer, and BootWatchdogs in the server. It also stays in the i2c, i3c, usart and sgpiom drivers, which are not ours. The rest is prose. The trial-boot record is set pending, matching set_trial_pending and is_pending on the trait. The SPI write filter is enabled, which is what the register does. The GPIO boot monitor cannot clear the latch. Assisted-by: Claude --- .../orchestrator/orchestrator-platform.md | 2 +- .../adapters/hal/src/gpio_boot_monitor.rs | 2 +- .../orchestrator/adapters/walk/src/walk.rs | 57 ++++++++++--------- .../capabilities/src/boot_watch.rs | 4 +- .../capabilities/src/device_trial_boot.rs | 20 +++---- services/orchestrator/capabilities/src/lib.rs | 4 +- services/orchestrator/driver/README.md | 4 +- services/orchestrator/driver/src/driver.rs | 6 +- services/orchestrator/driver/src/tests.rs | 8 +-- target/ast10x0/board/src/bmc.rs | 6 +- .../tests/orchestrator/runtime/main.rs | 2 +- 11 files changed, 58 insertions(+), 57 deletions(-) diff --git a/docs/src/design/orchestrator/orchestrator-platform.md b/docs/src/design/orchestrator/orchestrator-platform.md index ddc9e3b5..251d99e1 100644 --- a/docs/src/design/orchestrator/orchestrator-platform.md +++ b/docs/src/design/orchestrator/orchestrator-platform.md @@ -44,7 +44,7 @@ Two rules follow: - **Protection survives its crash.** The SPI monitor filters flash traffic in hardware, on its own; the orchestrator only loads its rules at boot and is not in the data path, and the hardware write filter stays - armed until the device's first fetch, closing the + enabled until the device's first fetch, closing the time-of-check/time-of-use window. Busy or crashed, the orchestrator cannot be bypassed — there is nothing to bypass. diff --git a/services/orchestrator/adapters/hal/src/gpio_boot_monitor.rs b/services/orchestrator/adapters/hal/src/gpio_boot_monitor.rs index 3591d2b8..fa195e0d 100644 --- a/services/orchestrator/adapters/hal/src/gpio_boot_monitor.rs +++ b/services/orchestrator/adapters/hal/src/gpio_boot_monitor.rs @@ -89,7 +89,7 @@ impl From for MonitorError { /// device re-enters reset (typically by wiring the latch's clear to the /// device's reset line) — [`BootStatus`] requires that evidence from a /// previous boot never reads as [`BootStatus::Booted`], and this reader only -/// reads the line, it cannot re-arm it. +/// reads the line, it cannot clear the latch. /// /// [`HalBootControl`]: crate::HalBootControl pub struct GpioBootMonitor<'a, P: GpioPort> { diff --git a/services/orchestrator/adapters/walk/src/walk.rs b/services/orchestrator/adapters/walk/src/walk.rs index 0d1a762e..d55a9df8 100644 --- a/services/orchestrator/adapters/walk/src/walk.rs +++ b/services/orchestrator/adapters/walk/src/walk.rs @@ -13,8 +13,9 @@ use orchestrator_config::{BootCheckpoint, DeviceConfig}; /// (`Booting`), so a transient bus glitch does not kill a healthy boot. /// A lapsed window is a timeout with no final read: a device that has not /// reported cannot be judged on a race. Each checkpoint's deadline -/// starts from the first poll after arm, not from the arm call, so -/// time between arming and polling does not count against the window. +/// starts from the first poll after the start call, not from the call +/// itself, so time between starting and polling does not count against +/// the window. /// An unarmed or post-terminal poll /// returns `Waiting { deadline_millis: u64::MAX }` (no deadline). pub struct CheckpointWalk { @@ -25,7 +26,7 @@ pub struct CheckpointWalk { enum Phase { Idle, - Armed, + Started, Walking { cursor: usize, deadline_millis: u64 }, } @@ -48,12 +49,12 @@ impl CheckpointWalk { } impl, P> BootWatch for CheckpointWalk { - fn arm(&mut self) { - self.phase = Phase::Armed; + fn start(&mut self) { + self.phase = Phase::Started; } fn poll(&mut self, now_millis: u64) -> WalkVerdict { - if let Phase::Armed = self.phase { + if let Phase::Started = self.phase { let timeout_millis = self.checkpoints[0].timeout().as_millis() as u64; let deadline = now_millis.saturating_add(timeout_millis); self.phase = Phase::Walking { @@ -201,7 +202,7 @@ mod tests { fn two_checkpoint_walk_completes_when_both_pass() { let mut w = walk(); w.reader_mut().level = 2; - w.arm(); + w.start(); let v = w.poll(0); assert_eq!( @@ -220,7 +221,7 @@ mod tests { fn single_checkpoint_walk_completes_in_one_poll() { let mut w = CheckpointWalk::new(ProgressReader::new(), &ONE_CP_DEVICE); w.reader_mut().level = 1; - w.arm(); + w.start(); assert_eq!(w.poll(0), WalkVerdict::Complete); } @@ -228,7 +229,7 @@ mod tests { #[test] fn progress_between_polls_advances_the_walk() { let mut w = walk(); - w.arm(); + w.start(); let v = w.poll(0); assert_eq!( @@ -258,7 +259,7 @@ mod tests { #[test] fn first_checkpoint_times_out_when_device_is_silent() { let mut w = walk(); - w.arm(); + w.start(); let v = w.poll(0); assert_eq!( @@ -281,7 +282,7 @@ mod tests { #[test] fn second_checkpoint_times_out_after_first_passes() { let mut w = walk(); - w.arm(); + w.start(); w.reader_mut().level = 1; let v = w.poll(0); @@ -307,7 +308,7 @@ mod tests { #[test] fn retriable_failure_ends_the_walk_early() { let mut w = walk(); - w.arm(); + w.start(); w.reader_mut().fault = Some(BootStatus::FailedRetriable); let v = w.poll(0); @@ -323,7 +324,7 @@ mod tests { #[test] fn fatal_failure_ends_the_walk_early() { let mut w = walk(); - w.arm(); + w.start(); w.reader_mut().fault = Some(BootStatus::FailedFatal); let v = w.poll(0); @@ -341,7 +342,7 @@ mod tests { #[test] fn read_error_treated_as_silence() { let mut w = walk(); - w.arm(); + w.start(); w.reader_mut().fail_read = true; let v = w.poll(0); @@ -366,13 +367,13 @@ mod tests { assert_eq!(w.poll(10), WalkVerdict::Complete); } - // ── arm() rewinds ─────────────────────────────────────────────────── + // ── start() rewinds ───────────────────────────────────────────────── #[test] fn arm_rewinds_to_the_first_checkpoint() { let mut w = walk(); w.reader_mut().level = 2; - w.arm(); + w.start(); assert_eq!( w.poll(0), WalkVerdict::Waiting { @@ -381,16 +382,16 @@ mod tests { ); assert_eq!(w.poll(0), WalkVerdict::Complete); - // Re-arm: back to checkpoint 0. + // Started again: back to checkpoint 0. w.reader_mut().level = 0; - w.arm(); + w.start(); let v = w.poll(1000); assert_eq!( v, WalkVerdict::Waiting { deadline_millis: 1100 }, - "fresh deadline from the re-arm" + "fresh deadline from the restart" ); } @@ -398,10 +399,10 @@ mod tests { fn arm_mid_walk_restarts_from_the_beginning() { let mut w = walk(); w.reader_mut().level = 1; - w.arm(); + w.start(); w.poll(0); // passes bl1, now at kernel - w.arm(); // restart + w.start(); // from the top w.reader_mut().level = 0; let v = w.poll(500); assert_eq!( @@ -429,8 +430,8 @@ mod tests { #[test] fn idle_after_terminal_waits_indefinitely() { let mut w = walk(); - w.arm(); - w.poll(0); // Armed -> Walking, deadline = 100 + w.start(); + w.poll(0); // Started -> Walking, deadline = 100 let v = w.poll(100); // now >= deadline -> TimedOut assert!(matches!(v, WalkVerdict::Failed { .. })); @@ -447,7 +448,7 @@ mod tests { #[test] fn deadline_is_relative_to_first_poll_not_arm() { let mut w = walk(); - w.arm(); + w.start(); // First poll at t=1000: deadline should be 1000 + 100, not 0 + 100. let v = w.poll(1000); assert_eq!( @@ -462,7 +463,7 @@ mod tests { fn next_checkpoint_deadline_is_relative_to_the_passing_poll() { let mut w = walk(); w.reader_mut().level = 1; - w.arm(); + w.start(); // bl1 passes at t=50, kernel deadline = 50 + 200. let v = w.poll(50); @@ -479,8 +480,8 @@ mod tests { #[test] fn booted_at_expiry_is_still_timeout() { let mut w = walk(); - w.arm(); - w.poll(0); // Armed -> Walking, deadline = 100 + w.start(); + w.poll(0); // Started -> Walking, deadline = 100 w.reader_mut().level = 1; let v = w.poll(100); // device ready, but window already lapsed @@ -500,7 +501,7 @@ mod tests { fn device_fault_at_second_checkpoint_names_it() { let mut w = walk(); w.reader_mut().level = 1; - w.arm(); + w.start(); w.poll(0); // bl1 passes, now at kernel w.reader_mut().fault = Some(BootStatus::FailedFatal); diff --git a/services/orchestrator/capabilities/src/boot_watch.rs b/services/orchestrator/capabilities/src/boot_watch.rs index 66c703c9..9e86933b 100644 --- a/services/orchestrator/capabilities/src/boot_watch.rs +++ b/services/orchestrator/capabilities/src/boot_watch.rs @@ -17,7 +17,7 @@ pub trait BootWatch { /// every reset release, retries included. Takes no timestamp — reset /// actuation has no clock; the attempt starts at the next /// [`poll`](BootWatch::poll)'s `now_millis`. - fn arm(&mut self); + fn start(&mut self); /// Judges the walk at `now_millis` (monotonic). Never sleeps — time is /// injected, so every decision is host-testable. @@ -93,7 +93,7 @@ mod tests { } impl BootWatch for ScriptedWalk { - fn arm(&mut self) { + fn start(&mut self) { self.next = 0; } diff --git a/services/orchestrator/capabilities/src/device_trial_boot.rs b/services/orchestrator/capabilities/src/device_trial_boot.rs index 6f93e3ea..5ba3a35e 100644 --- a/services/orchestrator/capabilities/src/device_trial_boot.rs +++ b/services/orchestrator/capabilities/src/device_trial_boot.rs @@ -18,7 +18,7 @@ //! has to read what the previous one left behind. That needs durable state //! this trait deliberately does not carry, and the verdict is the //! orchestrator's own supervised boot coming up, not a call from the image -//! that armed the trial. +//! that set the trial pending. //! //! A board that wants one gate for both can implement this trait over its //! [`SelfUpdate`] session; nothing here forbids it. What does not work is one @@ -49,11 +49,11 @@ /// plus the state a verdict reached after a reset needs. /// /// `confirm` and `revert` take no arguments. A device has at most one trial -/// open at a time, the one the last activation armed, so there is nothing to +/// open at a time, the one the last activation set pending, so there is nothing to /// name. Which slot is which stays behind the seam, as it does in /// `Updatable`. /// -/// The arming counts for the next boot only, but the record of the trial +/// The record counts for the next boot only, but the trial /// outlives it. An image that hangs, or an eRoT that loses power during the /// trial, boots the confirmed slot again without anyone calling anything. /// What makes that happen (a boot-select register the boot ROM clears, or @@ -74,7 +74,7 @@ /// anything; they only move slot metadata. Restarting the device is /// [`BootControl`](crate::BootControl). /// -/// `is_pending` is a yes or no. It does not say whether the armed image has +/// `is_pending` is a yes or no. It does not say whether the pending image has /// booted, and this trait cannot tell a `confirm` that came after a watched /// boot from one that did not. Nothing needs that difference: a half-done /// update is rerun from the start, not picked up where it left off. @@ -135,7 +135,7 @@ mod tests { } /// The one flow both implementations go through: apply the verdict to - /// whatever the last activation armed, then check the record came out + /// whatever the last activation set pending, then check the record came out /// clear. A record left open means a later boot finds a trial nobody /// owns. Generic over `DeviceTrialBoot`, so every device runs the same code /// whether the orchestrator held the instance all along or built it @@ -152,7 +152,7 @@ mod tests { Ok(()) } - /// A slot-selection record whose arming counts for the next boot only, + /// A slot-selection record that counts for the next boot only, /// the way a device's eRoT-held store behaves. struct SlotRecord { confirmed_slot: u8, @@ -176,7 +176,7 @@ mod tests { } /// Boots the device and returns the slot it ran: the trial slot if - /// the next boot is still armed, the confirmed slot otherwise. + /// the next boot is still pending, the confirmed slot otherwise. fn boot(&mut self) -> u8 { match self.trial_slot { Some(slot) if self.next_boot_armed => { @@ -270,7 +270,7 @@ mod tests { assert_eq!( device.record.boot(), 0, - "the arming is one-shot: no confirm, no second trial boot" + "the record is one-shot: no confirm, no second trial boot" ); assert_eq!( device.is_pending(), @@ -286,8 +286,8 @@ mod tests { let mut record = SlotRecord::new(0); record.activate(1); - // The boot that arming triggered. The armed image is now running, - // and the orchestrator that armed it has since restarted. + // The boot the record triggered. The pending image is now running, + // and the orchestrator that set it has since restarted. assert_eq!(record.boot(), 1); let mut trial = TrialRecord::from(&mut record); diff --git a/services/orchestrator/capabilities/src/lib.rs b/services/orchestrator/capabilities/src/lib.rs index 7f1b83f2..fcdad523 100644 --- a/services/orchestrator/capabilities/src/lib.rs +++ b/services/orchestrator/capabilities/src/lib.rs @@ -33,8 +33,8 @@ //! outlives the judge: the eRoT resets into the candidate, so the verdict is //! reached by a boot that has to read what the previous one left behind. It is //! one durable session, one at a time, carrying the state and the verified SVN -//! together, so the next boot can tell a session that was never armed from a -//! confirmed trial whose floor advance had not run yet. Downstream devices +//! together, so the next boot can tell a session whose trial never started +//! from a confirmed trial whose floor advance had not run yet. Downstream devices //! need no session: a reset loses the observation that would judge them, so //! abandoning at boot gets the same result with no storage. //! diff --git a/services/orchestrator/driver/README.md b/services/orchestrator/driver/README.md index cf2ef4d5..1bde9121 100644 --- a/services/orchestrator/driver/README.md +++ b/services/orchestrator/driver/README.md @@ -15,7 +15,7 @@ Everything device-specific arrives through the seams in `board.rs` Synchronous results (the verification verdict) return through `execute` and settle within the same dispatch run — there is no driver-side event queue. -Boot-walk verdicts are the one asynchronous read. `ReleaseReset` arms the +Boot-walk verdicts are the one asynchronous read. `ReleaseReset` starts the component's walk; the run loop polls and dispatches until quiet, then sleeps until the earliest walk deadline: @@ -31,6 +31,6 @@ loop { ``` Implemented executors: `ReadFirmware`, `VerifyFirmware`, `ReleaseReset` -(arms the boot walk), `AssertReset` (stops it). Everything else fails secure +(starts the boot walk), `AssertReset` (stops it). Everything else fails secure until its pillar lands (recovery, update path, attestation, reporting, lockdown latch). diff --git a/services/orchestrator/driver/src/driver.rs b/services/orchestrator/driver/src/driver.rs index 09d5d03f..a5f7c371 100644 --- a/services/orchestrator/driver/src/driver.rs +++ b/services/orchestrator/driver/src/driver.rs @@ -506,9 +506,9 @@ impl PlatformDriver { .ok_or(DriverError::UnknownComponent) } - /// Release `id` from reset and arm its boot walk; + /// Release `id` from reset and start its boot walk; /// [`poll_boot_walks`](Self::poll_boot_walks) feeds the verdict back - /// as `ComponentReady(id)`/`Booted(id)`/`BootFailed { id, .. }`. Arms on every + /// as `ComponentReady(id)`/`Booted(id)`/`BootFailed { id, .. }`. Starts on every /// release: a retry re-release starts a fresh walk. pub fn release_reset(&mut self, id: ComponentId) -> Result<(), DriverError> { self.boot_control(id)? @@ -516,7 +516,7 @@ impl PlatformDriver { .map_err(|_| DriverError::BootControlFault)?; let idx = id.get() as usize; // In bounds: boot_control(id) above already rejected unknown ids. - self.board.boot_watches[idx].arm(); + self.board.boot_watches[idx].start(); self.watching[idx] = true; Ok(()) } diff --git a/services/orchestrator/driver/src/tests.rs b/services/orchestrator/driver/src/tests.rs index 332a82ed..61bb13e5 100644 --- a/services/orchestrator/driver/src/tests.rs +++ b/services/orchestrator/driver/src/tests.rs @@ -203,9 +203,9 @@ impl orchestrator_capabilities::BootControl for MockReset { } /// Boot walk without a device; scripted verdicts. An exhausted script -/// holds its last verdict; an empty script waits forever. `arm` rewinds +/// holds its last verdict; an empty script waits forever. `start` rewinds /// to the script start, so a fresh attempt is observable from the -/// verdicts alone — no poll or arm counters needed. +/// verdicts alone — no poll or start counters needed. struct MockWalk { verdicts: std::vec::Vec, next: usize, @@ -225,7 +225,7 @@ impl MockWalk { } impl BootWatch for MockWalk { - fn arm(&mut self) { + fn start(&mut self) { self.next = 0; } @@ -1092,7 +1092,7 @@ fn only_released_components_are_watched() { assert_eq!(driver.poll_boot_walks(0).event, Some(Event::Booted(C0))); } -// Every release re-arms the walk: a retry judges a new attempt from the +// Every release starts the walk again: a retry judges a new attempt from // first checkpoint, not the failed one resumed. With a script of // [Failed, Complete], a resumed walk would report Complete on the second // attempt; a fresh one reports Failed again. diff --git a/target/ast10x0/board/src/bmc.rs b/target/ast10x0/board/src/bmc.rs index 48b5d172..6a7bc414 100644 --- a/target/ast10x0/board/src/bmc.rs +++ b/target/ast10x0/board/src/bmc.rs @@ -290,7 +290,7 @@ mod tests { #[test] fn a_booted_bmc_completes_the_walk_on_the_first_poll() { let mut walk = walk(ReadyLine::booted()); - walk.arm(); + walk.start(); assert_eq!(walk.poll(0), WalkVerdict::Complete); } @@ -298,7 +298,7 @@ mod tests { #[test] fn a_bmc_still_booting_holds_the_walk_until_its_line_rises() { let mut walk = walk(ReadyLine::late(2)); - walk.arm(); + walk.start(); let waiting = WalkVerdict::Waiting { deadline_millis: READY_WINDOW.as_millis() as u64, @@ -312,7 +312,7 @@ mod tests { #[test] fn a_bmc_that_never_reports_ready_times_out() { let mut walk = walk(ReadyLine::hung()); - walk.arm(); + walk.start(); let deadline = READY_WINDOW.as_millis() as u64; assert_eq!( diff --git a/target/ast10x0/tests/orchestrator/runtime/main.rs b/target/ast10x0/tests/orchestrator/runtime/main.rs index 741c2913..3238d8a9 100644 --- a/target/ast10x0/tests/orchestrator/runtime/main.rs +++ b/target/ast10x0/tests/orchestrator/runtime/main.rs @@ -241,7 +241,7 @@ fn walk_device( id: ComponentId, behavior: DeviceBehavior, ) -> Result { - walk.arm(); + walk.start(); let mut k = 0usize; loop {