Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 19 additions & 17 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -72,23 +72,25 @@ jobs:
# "libcvc.so.3: cannot open shared object file".
export LD_LIBRARY_PATH="${PREFIX}/lib${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"

# The cvcpkg dev-tool + runtime closure is missing a chain of transitive deps:
# black-cp3XX lacks its runtime deps (mypy_extensions/pathspec/platformdirs/
# pytokens) and is absent on cp313; pytest-cp3XX lacks pluggy; and the sympy in
# the closure lacks mpmath, so the torch/sympy-using tests raise ImportError.
# Supply a working black + pytest (full dep trees) plus mpmath as dev tools into a
# throwaway dir on PYTHONPATH (the same pattern coverage uses below); black is
# pinned to the closure's 26.5.1 (the repo is clean against it). ruff (a
# self-contained binary) is unaffected.
# TODO: fix the black-cp3XX / pytest-cp3XX / sympy recipe closures in libcvc-deps.
# (cp313's closure is more broken still: torch can't import for lack of
# typing_extensions, erroring test collection, so supply torch's import-time deps
# too.)
echo "== install dev tools black + pytest + missing closure deps =="
DEVTOOLS="${RUNNER_TEMP}/devtools"
"${PY}" -m pip install --quiet --disable-pip-version-check --target "${DEVTOOLS}" \
'black==26.5.1' pytest mpmath typing_extensions filelock networkx jinja2 fsspec
export PYTHONPATH="${DEVTOOLS}:${PYTHONPATH}"
# HERMETIC on cp311/cp312: the black/pytest/sympy/torch cvcpkg columns were republished
# (cy-pca/cvcpkg) carrying their declared transitive deps, so `install-deps grl-snam-cpXXX`
# now pulls mpmath/pluggy/pathspec/typing_extensions/... — no pip workaround needed.
#
# cp313 ONLY: the entire cp313 pure-Python column ecosystem is currently mis-installed to
# lib/python3.13t/site-packages (the free-threaded python313t leaked a bin/python3.13 that
# shadowed the non-t interpreter in the fleet's shared build prefix). ROOT CAUSE FIXED at the
# source (cy-pca/cvcpkg: build-python.sh strips the non-t names from a free-threaded column;
# python313t cvc.7 verified leak-free), but propagating it is an ecosystem-wide cp313 rebuild
# (dozens of columns incl. compiled torch/numpy) tracked as a fleet-ops batch. Until that
# lands, cp313 falls back to a pip-supplied dev-tool tree so CI stays green.
# TODO(cp313-ecosystem-rebuild): drop this branch once the mislaid cp313 columns are rebuilt.
if [ "${PY_DIGITS}" = "313" ]; then
echo "== cp313 pip dev-tool fallback (cp313 cvcpkg closure mislaid; see TODO) =="
DEVTOOLS="${RUNNER_TEMP}/devtools"
"${PY}" -m pip install --quiet --disable-pip-version-check --target "${DEVTOOLS}" \
'black==26.5.1' pytest 'mpmath<1.4' typing_extensions filelock networkx jinja2 fsspec
export PYTHONPATH="${DEVTOOLS}:${PYTHONPATH}"
fi

echo "== black --check =="
"${PY}" -m black --check grl_snam tests
Expand Down
140 changes: 140 additions & 0 deletions grl_snam/selection.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,140 @@
"""Checkpoint SELECTION over nav scorecards — turn a recorded ``NavScorecard`` into a scalar
fitness and rank a set of candidates (Track 4 Phase 1, SELECTION only).

This is deliberately separate from :mod:`grl_snam.scorecard` (which stays a pure schema + reducer):
selection reads finished ``NavScorecard`` rows and never serializes back, so it can never perturb the
C++<->Python parity surface (the composite is a free function, NOT a scorecard field). It also stays
RF-free — the DBG campaign composes an RF fitness on top of ``composite_fitness`` in ``grl_snam_dbg``;
nothing here touches the loss or the rollout (that is Phase 2).

Typical use — rank recorded corpus rows (the C++/native collector fills the formation/coverage/grip
fields that the Python base eval leaves at zero):

from grl_snam.scorecard import NavScorecard
from grl_snam.selection import FitnessWeights, rank
cards = [NavScorecard.from_json(open(p).read()) for p in card_paths]
best = rank(cards, FitnessWeights(w_form_arrival=1.0))[0]
"""

from __future__ import annotations

import json
from dataclasses import dataclass

from .scorecard import NavScorecard


@dataclass
class FitnessWeights:
"""Weights for :func:`composite_fitness` (higher fitness = better checkpoint). Each term is added
with the sign that makes "more is better": an ``_up`` field rewards a larger value, a ``_down``
field (a cost) is subtracted so a smaller value scores higher.

DEFAULTS reduce the composite to exactly ``arrival_rate`` — today's ranking signal (the raw
``reach_rate`` the base eval sorts on) — so enabling this changes NO ranking until a caller opts a
term in. The Phase-1 fields (formation / belief / grip-margin) are structurally zero in the Python
base corpus, so their weights move the ranking only on recorded (``from_dict``) rows the native
collector filled.
"""

# arrival / economy / safety (the base terms; only arrival is on by default = today's behavior)
w_arrival: float = 1.0 # arrival_rate — up
w_success: float = 0.0 # success_rate (every vehicle arrived) — up
w_time: float = 0.0 # mean_time_to_goal_s — down (economy)
w_path: float = 0.0 # mean_path_ratio — down (economy)
w_penetration: float = 0.0 # mean_penetration_pct — down (safety)
w_contacts: float = 0.0 # veh_contacts_per_run — down (safety)
w_min_sep: float = 0.0 # mean_min_sep_m — up (safety)
# formation holding
w_form_arrival: float = 0.0 # form_arrival_rate — up
w_form_mission: float = 0.0 # form_mission_rate — up
w_slot_error: float = 0.0 # mean_slot_error_m — down
# belief coverage
w_explored: float = 0.0 # mean_coverage.explored_frac — up
w_believed_free: float = 0.0 # mean_coverage.believed_free_frac — up
w_phantom: float = 0.0 # mean_coverage.phantom_frac — down (false-belief)
# grip-margin (drive telemetry): grip up, material risk down
w_grip_mu: float = 0.0 # mean_mu — up (more underfoot grip)
w_grip_mrisk: float = 0.0 # mean_mrisk — down (material risk)
# progress — kept a SEPARATE term, default-0: mean_closest_approach_m == 0 is ambiguous
# (0 = reached AND = unmeasured / no finite-distance run), so an all-zero corpus must not be
# rewarded. Enable only when the corpus is known to carry finite closest-approach.
w_closest: float = 0.0 # mean_closest_approach_m — down (closer to goal)
w_stall: float = 0.0 # mean_stall_steps — down


def composite_fitness(card: NavScorecard, weights: FitnessWeights | None = None) -> float:
"""Scalar fitness of one scorecard, higher = better. A weighted linear combination (weights carry
the scale); with default weights this is exactly ``card.arrival_rate``. Total over any card — a
partial ``from_dict`` row (missing keys defaulted) never raises."""
w = weights or FitnessWeights()
cov = card.mean_coverage
return (
w.w_arrival * card.arrival_rate
+ w.w_success * card.success_rate
- w.w_time * card.mean_time_to_goal_s
- w.w_path * card.mean_path_ratio
- w.w_penetration * card.mean_penetration_pct
- w.w_contacts * card.veh_contacts_per_run
+ w.w_min_sep * card.mean_min_sep_m
+ w.w_form_arrival * card.form_arrival_rate
+ w.w_form_mission * card.form_mission_rate
- w.w_slot_error * card.mean_slot_error_m
+ w.w_explored * cov.explored_frac
+ w.w_believed_free * cov.believed_free_frac
- w.w_phantom * cov.phantom_frac
+ w.w_grip_mu * card.mean_mu
- w.w_grip_mrisk * card.mean_mrisk
- w.w_closest * card.mean_closest_approach_m
- w.w_stall * card.mean_stall_steps
)


def rank(cards: list[NavScorecard], weights: FitnessWeights | None = None) -> list[NavScorecard]:
"""Cards sorted best-first by :func:`composite_fitness` (stable; ties keep input order)."""
w = weights or FitnessWeights()
return sorted(cards, key=lambda c: composite_fitness(c, w), reverse=True)


def select_best(
cards: list[NavScorecard], weights: FitnessWeights | None = None
) -> NavScorecard | None:
"""The single best card by :func:`composite_fitness`, or None for an empty list. On a tie the
first in input order wins (max is stable)."""
if not cards:
return None
w = weights or FitnessWeights()
return max(cards, key=lambda c: composite_fitness(c, w))


def _main(argv: list[str] | None = None) -> int:
"""`python -m grl_snam.selection card1.json card2.json [--weights '{"w_form_arrival":1.0}']` —
load recorded scorecard rows (cvc::nav ``scorecard_json`` / ``NavScorecard.to_json``) and print
them best-first by composite fitness. This is the offline SELECTION entry point; the recorded
(``from_json``) row is where the formation/coverage/grip fields are actually non-zero."""
import argparse

ap = argparse.ArgumentParser(
description="Rank nav scorecards by composite fitness (SELECTION)."
)
ap.add_argument("cards", nargs="+", help="scorecard JSON files")
ap.add_argument(
"--weights",
default="",
help="JSON object of FitnessWeights overrides, e.g. '{\"w_form_arrival\":1.0}'",
)
args = ap.parse_args(argv)
w = FitnessWeights(**json.loads(args.weights)) if args.weights else FitnessWeights()
rows = []
for p in args.cards:
with open(p) as f:
sc = NavScorecard.from_json(f.read())
rows.append((composite_fitness(sc, w), sc.checkpoint or p))
rows.sort(key=lambda r: r[0], reverse=True)
for score, label in rows:
print(f"{score:.6f}\t{label}")
return 0


if __name__ == "__main__":
raise SystemExit(_main())
111 changes: 111 additions & 0 deletions tests/test_selection.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
"""Track 4 Phase 1 SELECTION — composite_fitness + rank/select_best over NavScorecard rows.

Pins: (1) default weights reduce the composite to today's ranking signal (arrival_rate) so enabling
selection changes no ranking until a term is opted in; (2) each weighted term moves fitness in the
correct direction; (3) rank/select_best order best-first; (4) the fitness is total over a partial
(from_dict) row.
"""

import math

from grl_snam.scorecard import NavCoverage, NavScorecard
from grl_snam.selection import FitnessWeights, composite_fitness, rank, select_best


def test_default_fitness_is_arrival_rate():
# Back-compat: with default weights the composite IS arrival_rate (the raw reach_rate the base
# eval sorts on today) — nothing else contributes, so no recorded ranking changes on adoption.
# EVERY non-arrival field set non-zero, so a stray nonzero default on ANY weight would break the
# == arrival_rate identity (pins the back-compat guarantee for the whole weight vector).
sc = NavScorecard(
arrival_rate=0.7,
success_rate=1.0,
mean_time_to_goal_s=99.0,
mean_path_ratio=1.4,
mean_penetration_pct=5.0,
veh_contacts_per_run=2.0,
mean_min_sep_m=3.0,
form_arrival_rate=1.0,
form_mission_rate=1.0,
mean_slot_error_m=50.0,
mean_sense_flips=9.0,
mean_mu=0.9,
mean_mrisk=0.4,
mean_closest_approach_m=6.0,
mean_stall_steps=8.0,
mean_coverage=NavCoverage(
explored_frac=1.0, visible_frac=1.0, believed_free_frac=1.0, phantom_frac=1.0
),
)
assert math.isclose(composite_fitness(sc), 0.7)


def _dir(field, weight_kw, up):
"""A field moves fitness the expected way: build two cards differing only in `field`, weight it,
and assert the larger-field card scores higher (up=True) or lower (up=False)."""
lo, hi = NavScorecard(), NavScorecard()
setattr(lo, field, 1.0)
setattr(hi, field, 2.0)
w = FitnessWeights(**{weight_kw: 1.0})
d = composite_fitness(hi, w) - composite_fitness(lo, w)
assert (d > 0) if up else (d < 0), f"{field} via {weight_kw}: delta {d} wrong sign"


def test_per_field_directions():
# up = larger value -> higher fitness; down (a cost) = larger value -> lower fitness.
_dir("arrival_rate", "w_arrival", up=True)
_dir("success_rate", "w_success", up=True)
_dir("mean_time_to_goal_s", "w_time", up=False)
_dir("mean_path_ratio", "w_path", up=False)
_dir("mean_penetration_pct", "w_penetration", up=False)
_dir("veh_contacts_per_run", "w_contacts", up=False)
_dir("mean_min_sep_m", "w_min_sep", up=True)
_dir("form_arrival_rate", "w_form_arrival", up=True)
_dir("form_mission_rate", "w_form_mission", up=True)
_dir("mean_slot_error_m", "w_slot_error", up=False)
_dir("mean_mu", "w_grip_mu", up=True) # grip up
_dir("mean_mrisk", "w_grip_mrisk", up=False) # material risk down
_dir("mean_closest_approach_m", "w_closest", up=False) # closer = smaller = better
_dir("mean_stall_steps", "w_stall", up=False)


def test_coverage_directions():
# nested mean_coverage fields, weighted.
def card(explored=0.0, believed=0.0, phantom=0.0):
return NavScorecard(
mean_coverage=NavCoverage(
explored_frac=explored, believed_free_frac=believed, phantom_frac=phantom
)
)

we = FitnessWeights(w_explored=1.0)
assert composite_fitness(card(explored=0.8), we) > composite_fitness(card(explored=0.2), we)
wb = FitnessWeights(w_believed_free=1.0)
assert composite_fitness(card(believed=0.8), wb) > composite_fitness(card(believed=0.2), wb)
wp = FitnessWeights(w_phantom=1.0) # phantom is a cost -> more phantom scores LOWER
assert composite_fitness(card(phantom=0.8), wp) < composite_fitness(card(phantom=0.2), wp)


def test_rank_and_select_best():
a = NavScorecard(checkpoint="a", arrival_rate=0.5)
b = NavScorecard(checkpoint="b", arrival_rate=0.9)
c = NavScorecard(checkpoint="c", arrival_rate=0.7)
ordered = rank([a, b, c]) # default weights -> by arrival_rate
assert [s.checkpoint for s in ordered] == ["b", "c", "a"]
assert select_best([a, b, c]).checkpoint == "b"
assert select_best([]) is None
# a weighting can flip the order: reward formation instead of arrival
a.form_arrival_rate, b.form_arrival_rate, c.form_arrival_rate = 1.0, 0.0, 0.0
w = FitnessWeights(w_arrival=0.0, w_form_arrival=1.0)
assert select_best([a, b, c], w).checkpoint == "a"


def test_fitness_total_over_partial_row():
# a partial recorded row (from_dict defaults the missing keys) must not raise, even with every
# Phase-1 term weighted.
sc = NavScorecard.from_dict({"checkpoint": "p", "arrival_rate": 0.5})
w = FitnessWeights(
w_form_arrival=1.0, w_explored=1.0, w_grip_mu=1.0, w_slot_error=1.0, w_closest=1.0
)
assert isinstance(composite_fitness(sc, w), float)
assert math.isclose(composite_fitness(sc), 0.5) # only arrival present -> default == arrival
Loading