From add6075b61d74a87f12e31c203e2dca2967a147b Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Sat, 26 Sep 2026 00:38:55 -0500 Subject: [PATCH 1/3] selection: composite_fitness + rank/select_best over NavScorecard (Track 4 Phase 1, PR-A) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The base SELECTION layer the roadmap Phase 1 asks for: turn a recorded NavScorecard into a scalar fitness and rank a set of candidates. Kept in a new grl_snam/selection.py so scorecard.py stays a pure schema/reducer — the composite is a FREE FUNCTION, never a serialized scorecard field, so it can't perturb the C++<->Python parity surface. RF-free (the DBG side composes an RF fitness on top in grl_snam_dbg); reads finished scorecards only, never the loss/rollout (that is Phase 2). - FitnessWeights: per-field weights, "more is better" (an _up field added, a _down cost subtracted). DEFAULTS reduce the composite to exactly arrival_rate — today's ranking signal (raw reach_rate) — so adoption changes NO ranking until a term is opted in. Phase-1 fields wired: formation (form_arrival/mission up, slot_error down), belief (explored/believed_free up, phantom down), grip-margin (mean_mu up, mean_mrisk down), and a SEPARATE default-0 progress term (closest_approach/stall; default-0 because mean_closest_approach_m==0 is reached-vs-unmeasured ambiguous). - composite_fitness / rank / select_best. Total over any card (a partial from_dict row never raises). - `python -m grl_snam.selection ... [--weights '{...}']` — the offline SELECTION entry point: loads recorded scorecard_json rows via NavScorecard.from_json (Phase 0's reader — its first consumer) and prints best-first. Recorded rows are where formation/coverage/grip are non-zero (the native collector fills them). Tests: default==arrival_rate, every field's direction, coverage sub-fields, rank + select_best ordering (+ a formation weighting flipping the winner), and totality over a partial row. 18/18 (with scorecard) pass; ruff clean; CLI verified end-to-end. --- grl_snam/selection.py | 140 ++++++++++++++++++++++++++++++++++++++++ tests/test_selection.py | 111 +++++++++++++++++++++++++++++++ 2 files changed, 251 insertions(+) create mode 100644 grl_snam/selection.py create mode 100644 tests/test_selection.py diff --git a/grl_snam/selection.py b/grl_snam/selection.py new file mode 100644 index 0000000..7857c64 --- /dev/null +++ b/grl_snam/selection.py @@ -0,0 +1,140 @@ +"""Checkpoint SELECTION over nav scorecards — turn a recorded ``NavScorecard`` into a scalar +fitness and rank a set of candidates (Track 4 Phase 1, SELECTION only). + +This is deliberately separate from :mod:`grl_snam.scorecard` (which stays a pure schema + reducer): +selection reads finished ``NavScorecard`` rows and never serializes back, so it can never perturb the +C++<->Python parity surface (the composite is a free function, NOT a scorecard field). It also stays +RF-free — the DBG campaign composes an RF fitness on top of ``composite_fitness`` in ``grl_snam_dbg``; +nothing here touches the loss or the rollout (that is Phase 2). + +Typical use — rank recorded corpus rows (the C++/native collector fills the formation/coverage/grip +fields that the Python base eval leaves at zero): + + from grl_snam.scorecard import NavScorecard + from grl_snam.selection import FitnessWeights, rank + cards = [NavScorecard.from_json(open(p).read()) for p in card_paths] + best = rank(cards, FitnessWeights(w_form_arrival=1.0))[0] +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass + +from .scorecard import NavScorecard + + +@dataclass +class FitnessWeights: + """Weights for :func:`composite_fitness` (higher fitness = better checkpoint). Each term is added + with the sign that makes "more is better": an ``_up`` field rewards a larger value, a ``_down`` + field (a cost) is subtracted so a smaller value scores higher. + + DEFAULTS reduce the composite to exactly ``arrival_rate`` — today's ranking signal (the raw + ``reach_rate`` the base eval sorts on) — so enabling this changes NO ranking until a caller opts a + term in. The Phase-1 fields (formation / belief / grip-margin) are structurally zero in the Python + base corpus, so their weights move the ranking only on recorded (``from_dict``) rows the native + collector filled. + """ + + # arrival / economy / safety (the base terms; only arrival is on by default = today's behavior) + w_arrival: float = 1.0 # arrival_rate — up + w_success: float = 0.0 # success_rate (every vehicle arrived) — up + w_time: float = 0.0 # mean_time_to_goal_s — down (economy) + w_path: float = 0.0 # mean_path_ratio — down (economy) + w_penetration: float = 0.0 # mean_penetration_pct — down (safety) + w_contacts: float = 0.0 # veh_contacts_per_run — down (safety) + w_min_sep: float = 0.0 # mean_min_sep_m — up (safety) + # formation holding + w_form_arrival: float = 0.0 # form_arrival_rate — up + w_form_mission: float = 0.0 # form_mission_rate — up + w_slot_error: float = 0.0 # mean_slot_error_m — down + # belief coverage + w_explored: float = 0.0 # mean_coverage.explored_frac — up + w_believed_free: float = 0.0 # mean_coverage.believed_free_frac — up + w_phantom: float = 0.0 # mean_coverage.phantom_frac — down (false-belief) + # grip-margin (drive telemetry): grip up, material risk down + w_grip_mu: float = 0.0 # mean_mu — up (more underfoot grip) + w_grip_mrisk: float = 0.0 # mean_mrisk — down (material risk) + # progress — kept a SEPARATE term, default-0: mean_closest_approach_m == 0 is ambiguous + # (0 = reached AND = unmeasured / no finite-distance run), so an all-zero corpus must not be + # rewarded. Enable only when the corpus is known to carry finite closest-approach. + w_closest: float = 0.0 # mean_closest_approach_m — down (closer to goal) + w_stall: float = 0.0 # mean_stall_steps — down + + +def composite_fitness(card: NavScorecard, weights: FitnessWeights | None = None) -> float: + """Scalar fitness of one scorecard, higher = better. A weighted linear combination (weights carry + the scale); with default weights this is exactly ``card.arrival_rate``. Total over any card — a + partial ``from_dict`` row (missing keys defaulted) never raises.""" + w = weights or FitnessWeights() + cov = card.mean_coverage + return ( + w.w_arrival * card.arrival_rate + + w.w_success * card.success_rate + - w.w_time * card.mean_time_to_goal_s + - w.w_path * card.mean_path_ratio + - w.w_penetration * card.mean_penetration_pct + - w.w_contacts * card.veh_contacts_per_run + + w.w_min_sep * card.mean_min_sep_m + + w.w_form_arrival * card.form_arrival_rate + + w.w_form_mission * card.form_mission_rate + - w.w_slot_error * card.mean_slot_error_m + + w.w_explored * cov.explored_frac + + w.w_believed_free * cov.believed_free_frac + - w.w_phantom * cov.phantom_frac + + w.w_grip_mu * card.mean_mu + - w.w_grip_mrisk * card.mean_mrisk + - w.w_closest * card.mean_closest_approach_m + - w.w_stall * card.mean_stall_steps + ) + + +def rank( + cards: list[NavScorecard], weights: FitnessWeights | None = None +) -> list[NavScorecard]: + """Cards sorted best-first by :func:`composite_fitness` (stable; ties keep input order).""" + w = weights or FitnessWeights() + return sorted(cards, key=lambda c: composite_fitness(c, w), reverse=True) + + +def select_best( + cards: list[NavScorecard], weights: FitnessWeights | None = None +) -> NavScorecard | None: + """The single best card by :func:`composite_fitness`, or None for an empty list. On a tie the + first in input order wins (max is stable).""" + if not cards: + return None + w = weights or FitnessWeights() + return max(cards, key=lambda c: composite_fitness(c, w)) + + +def _main(argv: list[str] | None = None) -> int: + """`python -m grl_snam.selection card1.json card2.json [--weights '{"w_form_arrival":1.0}']` — + load recorded scorecard rows (cvc::nav ``scorecard_json`` / ``NavScorecard.to_json``) and print + them best-first by composite fitness. This is the offline SELECTION entry point; the recorded + (``from_json``) row is where the formation/coverage/grip fields are actually non-zero.""" + import argparse + + ap = argparse.ArgumentParser(description="Rank nav scorecards by composite fitness (SELECTION).") + ap.add_argument("cards", nargs="+", help="scorecard JSON files") + ap.add_argument( + "--weights", + default="", + help='JSON object of FitnessWeights overrides, e.g. \'{"w_form_arrival":1.0}\'', + ) + args = ap.parse_args(argv) + w = FitnessWeights(**json.loads(args.weights)) if args.weights else FitnessWeights() + rows = [] + for p in args.cards: + with open(p) as f: + sc = NavScorecard.from_json(f.read()) + rows.append((composite_fitness(sc, w), sc.checkpoint or p)) + rows.sort(key=lambda r: r[0], reverse=True) + for score, label in rows: + print(f"{score:.6f}\t{label}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(_main()) diff --git a/tests/test_selection.py b/tests/test_selection.py new file mode 100644 index 0000000..ce226ee --- /dev/null +++ b/tests/test_selection.py @@ -0,0 +1,111 @@ +"""Track 4 Phase 1 SELECTION — composite_fitness + rank/select_best over NavScorecard rows. + +Pins: (1) default weights reduce the composite to today's ranking signal (arrival_rate) so enabling +selection changes no ranking until a term is opted in; (2) each weighted term moves fitness in the +correct direction; (3) rank/select_best order best-first; (4) the fitness is total over a partial +(from_dict) row. +""" + +import math + +from grl_snam.scorecard import NavCoverage, NavScorecard +from grl_snam.selection import FitnessWeights, composite_fitness, rank, select_best + + +def test_default_fitness_is_arrival_rate(): + # Back-compat: with default weights the composite IS arrival_rate (the raw reach_rate the base + # eval sorts on today) — nothing else contributes, so no recorded ranking changes on adoption. + # EVERY non-arrival field set non-zero, so a stray nonzero default on ANY weight would break the + # == arrival_rate identity (pins the back-compat guarantee for the whole weight vector). + sc = NavScorecard( + arrival_rate=0.7, + success_rate=1.0, + mean_time_to_goal_s=99.0, + mean_path_ratio=1.4, + mean_penetration_pct=5.0, + veh_contacts_per_run=2.0, + mean_min_sep_m=3.0, + form_arrival_rate=1.0, + form_mission_rate=1.0, + mean_slot_error_m=50.0, + mean_sense_flips=9.0, + mean_mu=0.9, + mean_mrisk=0.4, + mean_closest_approach_m=6.0, + mean_stall_steps=8.0, + mean_coverage=NavCoverage( + explored_frac=1.0, visible_frac=1.0, believed_free_frac=1.0, phantom_frac=1.0 + ), + ) + assert math.isclose(composite_fitness(sc), 0.7) + + +def _dir(field, weight_kw, up): + """A field moves fitness the expected way: build two cards differing only in `field`, weight it, + and assert the larger-field card scores higher (up=True) or lower (up=False).""" + lo, hi = NavScorecard(), NavScorecard() + setattr(lo, field, 1.0) + setattr(hi, field, 2.0) + w = FitnessWeights(**{weight_kw: 1.0}) + d = composite_fitness(hi, w) - composite_fitness(lo, w) + assert (d > 0) if up else (d < 0), f"{field} via {weight_kw}: delta {d} wrong sign" + + +def test_per_field_directions(): + # up = larger value -> higher fitness; down (a cost) = larger value -> lower fitness. + _dir("arrival_rate", "w_arrival", up=True) + _dir("success_rate", "w_success", up=True) + _dir("mean_time_to_goal_s", "w_time", up=False) + _dir("mean_path_ratio", "w_path", up=False) + _dir("mean_penetration_pct", "w_penetration", up=False) + _dir("veh_contacts_per_run", "w_contacts", up=False) + _dir("mean_min_sep_m", "w_min_sep", up=True) + _dir("form_arrival_rate", "w_form_arrival", up=True) + _dir("form_mission_rate", "w_form_mission", up=True) + _dir("mean_slot_error_m", "w_slot_error", up=False) + _dir("mean_mu", "w_grip_mu", up=True) # grip up + _dir("mean_mrisk", "w_grip_mrisk", up=False) # material risk down + _dir("mean_closest_approach_m", "w_closest", up=False) # closer = smaller = better + _dir("mean_stall_steps", "w_stall", up=False) + + +def test_coverage_directions(): + # nested mean_coverage fields, weighted. + def card(explored=0.0, believed=0.0, phantom=0.0): + return NavScorecard( + mean_coverage=NavCoverage( + explored_frac=explored, believed_free_frac=believed, phantom_frac=phantom + ) + ) + + we = FitnessWeights(w_explored=1.0) + assert composite_fitness(card(explored=0.8), we) > composite_fitness(card(explored=0.2), we) + wb = FitnessWeights(w_believed_free=1.0) + assert composite_fitness(card(believed=0.8), wb) > composite_fitness(card(believed=0.2), wb) + wp = FitnessWeights(w_phantom=1.0) # phantom is a cost -> more phantom scores LOWER + assert composite_fitness(card(phantom=0.8), wp) < composite_fitness(card(phantom=0.2), wp) + + +def test_rank_and_select_best(): + a = NavScorecard(checkpoint="a", arrival_rate=0.5) + b = NavScorecard(checkpoint="b", arrival_rate=0.9) + c = NavScorecard(checkpoint="c", arrival_rate=0.7) + ordered = rank([a, b, c]) # default weights -> by arrival_rate + assert [s.checkpoint for s in ordered] == ["b", "c", "a"] + assert select_best([a, b, c]).checkpoint == "b" + assert select_best([]) is None + # a weighting can flip the order: reward formation instead of arrival + a.form_arrival_rate, b.form_arrival_rate, c.form_arrival_rate = 1.0, 0.0, 0.0 + w = FitnessWeights(w_arrival=0.0, w_form_arrival=1.0) + assert select_best([a, b, c], w).checkpoint == "a" + + +def test_fitness_total_over_partial_row(): + # a partial recorded row (from_dict defaults the missing keys) must not raise, even with every + # Phase-1 term weighted. + sc = NavScorecard.from_dict({"checkpoint": "p", "arrival_rate": 0.5}) + w = FitnessWeights( + w_form_arrival=1.0, w_explored=1.0, w_grip_mu=1.0, w_slot_error=1.0, w_closest=1.0 + ) + assert isinstance(composite_fitness(sc, w), float) + assert math.isclose(composite_fitness(sc), 0.5) # only arrival present -> default == arrival From 0e1d7ce83216915b7d29eb1d5128d70ccbb07d10 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Sat, 26 Sep 2026 12:50:36 -0500 Subject: [PATCH 2/3] ci: hermetic dev tools on cp311/cp312; guard the pip fallback to cp313 only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The black/pytest/sympy/torch cvcpkg columns were republished (cy-pca/cvcpkg) carrying their declared transitive deps, so install-deps grl-snam-cpXXX now pulls mpmath/pluggy/pathspec/typing_extensions/... on cp311/cp312 — the non-hermetic pip dev-tool install is no longer needed there and is removed. cp313 keeps the pip fallback: the whole cp313 pure-Python column ecosystem is currently mis-installed to lib/python3.13t/ (the free-threaded python313t leaked a bin/python3.13 that shadowed the non-t interpreter in the fleet's shared build prefix). ROOT CAUSE FIXED at the source (build-python.sh strips the non-t names from a free-threaded column; python313t cvc.7 verified leak-free), but propagating it is an ecosystem-wide cp313 rebuild tracked as a fleet-ops batch. mpmath pinned <1.4 in the fallback to match sympy. TODO(cp313-ecosystem-rebuild): drop the cp313 branch once the mislaid columns are rebuilt. --- .github/workflows/ci.yml | 36 +++++++++++++++++++----------------- 1 file changed, 19 insertions(+), 17 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6c872f0..c7d2ca0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -72,23 +72,25 @@ jobs: # "libcvc.so.3: cannot open shared object file". export LD_LIBRARY_PATH="${PREFIX}/lib${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}" - # The cvcpkg dev-tool + runtime closure is missing a chain of transitive deps: - # black-cp3XX lacks its runtime deps (mypy_extensions/pathspec/platformdirs/ - # pytokens) and is absent on cp313; pytest-cp3XX lacks pluggy; and the sympy in - # the closure lacks mpmath, so the torch/sympy-using tests raise ImportError. - # Supply a working black + pytest (full dep trees) plus mpmath as dev tools into a - # throwaway dir on PYTHONPATH (the same pattern coverage uses below); black is - # pinned to the closure's 26.5.1 (the repo is clean against it). ruff (a - # self-contained binary) is unaffected. - # TODO: fix the black-cp3XX / pytest-cp3XX / sympy recipe closures in libcvc-deps. - # (cp313's closure is more broken still: torch can't import for lack of - # typing_extensions, erroring test collection, so supply torch's import-time deps - # too.) - echo "== install dev tools black + pytest + missing closure deps ==" - DEVTOOLS="${RUNNER_TEMP}/devtools" - "${PY}" -m pip install --quiet --disable-pip-version-check --target "${DEVTOOLS}" \ - 'black==26.5.1' pytest mpmath typing_extensions filelock networkx jinja2 fsspec - export PYTHONPATH="${DEVTOOLS}:${PYTHONPATH}" + # HERMETIC on cp311/cp312: the black/pytest/sympy/torch cvcpkg columns were republished + # (cy-pca/cvcpkg) carrying their declared transitive deps, so `install-deps grl-snam-cpXXX` + # now pulls mpmath/pluggy/pathspec/typing_extensions/... — no pip workaround needed. + # + # cp313 ONLY: the entire cp313 pure-Python column ecosystem is currently mis-installed to + # lib/python3.13t/site-packages (the free-threaded python313t leaked a bin/python3.13 that + # shadowed the non-t interpreter in the fleet's shared build prefix). ROOT CAUSE FIXED at the + # source (cy-pca/cvcpkg: build-python.sh strips the non-t names from a free-threaded column; + # python313t cvc.7 verified leak-free), but propagating it is an ecosystem-wide cp313 rebuild + # (dozens of columns incl. compiled torch/numpy) tracked as a fleet-ops batch. Until that + # lands, cp313 falls back to a pip-supplied dev-tool tree so CI stays green. + # TODO(cp313-ecosystem-rebuild): drop this branch once the mislaid cp313 columns are rebuilt. + if [ "${PY_DIGITS}" = "313" ]; then + echo "== cp313 pip dev-tool fallback (cp313 cvcpkg closure mislaid; see TODO) ==" + DEVTOOLS="${RUNNER_TEMP}/devtools" + "${PY}" -m pip install --quiet --disable-pip-version-check --target "${DEVTOOLS}" \ + 'black==26.5.1' pytest 'mpmath<1.4' typing_extensions filelock networkx jinja2 fsspec + export PYTHONPATH="${DEVTOOLS}:${PYTHONPATH}" + fi echo "== black --check ==" "${PY}" -m black --check grl_snam tests From 2ee417f4a0d13edc112a7b14c978ea4e1049eee4 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Sat, 26 Sep 2026 13:00:17 -0500 Subject: [PATCH 3/3] =?UTF-8?q?selection:=20black-format=20(line-length=20?= =?UTF-8?q?+=20quote=20style)=20=E2=80=94=20fixes=20CI=20black=20--check?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- grl_snam/selection.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/grl_snam/selection.py b/grl_snam/selection.py index 7857c64..c217ed5 100644 --- a/grl_snam/selection.py +++ b/grl_snam/selection.py @@ -90,9 +90,7 @@ def composite_fitness(card: NavScorecard, weights: FitnessWeights | None = None) ) -def rank( - cards: list[NavScorecard], weights: FitnessWeights | None = None -) -> list[NavScorecard]: +def rank(cards: list[NavScorecard], weights: FitnessWeights | None = None) -> list[NavScorecard]: """Cards sorted best-first by :func:`composite_fitness` (stable; ties keep input order).""" w = weights or FitnessWeights() return sorted(cards, key=lambda c: composite_fitness(c, w), reverse=True) @@ -116,12 +114,14 @@ def _main(argv: list[str] | None = None) -> int: (``from_json``) row is where the formation/coverage/grip fields are actually non-zero.""" import argparse - ap = argparse.ArgumentParser(description="Rank nav scorecards by composite fitness (SELECTION).") + ap = argparse.ArgumentParser( + description="Rank nav scorecards by composite fitness (SELECTION)." + ) ap.add_argument("cards", nargs="+", help="scorecard JSON files") ap.add_argument( "--weights", default="", - help='JSON object of FitnessWeights overrides, e.g. \'{"w_form_arrival":1.0}\'', + help="JSON object of FitnessWeights overrides, e.g. '{\"w_form_arrival\":1.0}'", ) args = ap.parse_args(argv) w = FitnessWeights(**json.loads(args.weights)) if args.weights else FitnessWeights()