From 3ecb89949d30bd259de5bab6d9104b07ab17d14f Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Sun, 27 Sep 2026 16:24:05 -0500 Subject: [PATCH 1/2] =?UTF-8?q?coef:=20two-head=20sigmoid=20(lam=5Fsoft,?= =?UTF-8?q?=20lam=5Fhard)=20reroute=20=E2=80=94=20.cvcnav=20format=20v2?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend the deployable CoefMLP / coef_export / coef_train path with the paper's TWO learned, sigmoid-bounded reroute strengths (lam_soft AND lam_hard, App A.4 / material_nav.py:175-176: lam = lam_max*sigmoid(head)). Byte-compatible with the libcvc cvc::nav::coef_mlp v2 loader. sdf_nav.CoefMLP: - Add use_lam_hard (requires use_lam) -> out_dim = 3 + use_lam + use_lam_hard, plus lam_soft_max/lam_hard_max (5.0/10.0). A two-head net is lam_sigmoid: the lam columns are lam_max*sigmoid(head) with the head init in the last Linear bias (logit(init/max)); alpha/beta/gamma stay softplus(log(expm1)). full_out_bias() returns the 3 abg biases only in sigmoid mode. coeffs_and_lam returns (a,b,g,lam_soft[,lam_hard]). The v1 single-softplus lam path is unchanged. - add_lam_heads(model, soft_init, hard_init): basin-preserving lifter mirroring add_lam_head — append the lam_soft/lam_hard rows (zero weights, bias=logit(init/ max)), copy the abg rows verbatim so alpha/beta/gamma are bit-identical at init. coef_export.write_coef_mlp: a sigmoid net writes FORMAT_VERSION 2 + FLAG_LAM_SIGMOID, out_bias length 3 (abg only), then the two ceilings (struct '0) seeds the hard head. train_bicycle and eval_risk_exposure use the LEARNED lam_hard for a two-head net; the Swarm drives with both learned lambdas. Deploy note states the 5-out v2 net + the v2 libcvc host floor. tests: add_lam_heads identity + init + flags; two-head sigmoid bounds and out-of-range init rejection; exporter v2 layout + a pure-numpy decode/forward parity (~1e-4) that reproduces the C++ forward from the exported bytes; v1 single-softplus export back-compat; a two-head train_bicycle smoke; the CLI rejecting --lam-hard-init <= 0; and a pycvc-gated in-process C++ round-trip for the two-head net. --- grl_snam/sdf_nav.py | 139 ++++++++++++++++++++--- grl_snam/swarm.py | 8 +- grl_snam/tools/coef_export.py | 28 ++++- grl_snam/tools/coef_train.py | 74 +++++++++++-- tests/test_coef_mlp_parity.py | 22 ++++ tests/test_lam_head.py | 200 +++++++++++++++++++++++++++++++++- 6 files changed, 438 insertions(+), 33 deletions(-) diff --git a/grl_snam/sdf_nav.py b/grl_snam/sdf_nav.py index 984963a..cce0d4a 100644 --- a/grl_snam/sdf_nav.py +++ b/grl_snam/sdf_nav.py @@ -864,6 +864,12 @@ def ackermann_delta_max(L: float, delta_max: float, track_width: float) -> float return math.atan(L / (L / math.tan(delta_max) + 0.5 * float(track_width))) +def _logit(x: float) -> float: + """Inverse of sigmoid: the last-layer bias that makes a sigmoid lam head start at ``x*max``. + Caller guarantees 0 < x < 1 (see CoefMLP._check_sigmoid_init).""" + return math.log(x / (1.0 - x)) + + class CoefMLP(nn.Module): """Predict ``(alpha, beta, gamma)`` from local SDF features, biased toward the known-good navigating regime (``bias``) so the self-supervised optimizer starts @@ -878,6 +884,11 @@ def __init__( use_risk=False, use_lam=False, lam_init=0.4, + use_lam_hard=False, + lam_hard_init=1.0, + lam_sigmoid=False, + lam_soft_max=5.0, + lam_hard_max=10.0, ): super().__init__() #: 5 = the original [phi, goal_dist, gdir_x, gdir_y, align]; then the OPTIONAL @@ -890,11 +901,20 @@ def __init__( self.use_mu = bool(use_mu) self.use_risk = bool(use_risk) #: ``use_lam`` adds a 4th OUTPUT, the learned material-reroute strength lam_soft - #: (see :meth:`coeffs_and_lam`). The output stack stays a single uniform - #: ``softplus(net + log(expm1(out_bias)))`` so the exporter and the C++ forward - #: need no special case — only the drive reads output[3] as lam_soft. + #: (see :meth:`coeffs_and_lam`). ``use_lam_hard`` adds a 5th, lam_hard — the paper's + #: two-head reroute (App A.4). A two-head net is always ``lam_sigmoid`` (the paper form + #: lam = lam_max*sigmoid(head), material_nav.py:175-176), which the exporter writes as + #: .cvcnav format v2. A lone lam_soft (``use_lam`` only, no sigmoid) keeps the historical + #: single uniform ``softplus(net + log(expm1(out_bias)))`` (v1) so old blobs are unchanged. self.use_lam = bool(use_lam) - self.out_dim = 3 + (1 if self.use_lam else 0) + self.use_lam_hard = bool(use_lam_hard) + if self.use_lam_hard and not self.use_lam: + raise ValueError("use_lam_hard requires use_lam (lam_hard is the 5th output, col 4)") + #: A two-head net is the paper's sigmoid form; a single lam_soft can opt in explicitly. + self.lam_sigmoid = bool(lam_sigmoid) or self.use_lam_hard + self.lam_soft_max = float(lam_soft_max) + self.lam_hard_max = float(lam_hard_max) + self.out_dim = 3 + (1 if self.use_lam else 0) + (1 if self.use_lam_hard else 0) self.net = nn.Sequential( nn.Linear(self.in_dim, hidden), nn.SiLU(), @@ -903,40 +923,77 @@ def __init__( nn.Linear(hidden, self.out_dim), ) self.register_buffer("bias", torch.tensor(bias)) # abg raw bias (unchanged) - if self.use_lam: - # lam_soft init: softplus(0 + log(expm1(lam_init))) == lam_init at the zero column. - # Requires lam_init > 0: log(expm1(0)) is -inf (a DEAD lam head — 0 with zero - # gradient forever), and log(expm1(<0)) is NaN (poisons the drive/loss/.cvcnav). + if self.use_lam and not self.lam_sigmoid: + # v1 single-softplus lam_soft: softplus(0 + log(expm1(lam_init))) == lam_init at the + # zero column. Requires lam_init > 0: log(expm1(0)) is -inf (a DEAD lam head — 0 with + # zero gradient), and log(expm1(<0)) is NaN (poisons the drive/loss/.cvcnav). if not (float(lam_init) > 0.0): raise ValueError( f"use_lam requires lam_init > 0 (softplus init is -inf at 0, NaN below); " f"got {lam_init!r}" ) self.register_buffer("lam_raw_bias", torch.tensor(float(lam_init))) + if self.lam_sigmoid: + # Sigmoid lam heads: lam = lam_max*sigmoid(head), so the init bias lives in the LAST + # Linear layer's bias (logit(init/max)), NOT the softplus-fold buffer. Requires + # 0 < init < max (logit is +/-inf otherwise). A freshly-built net thus starts at the + # requested init (add_lam_heads re-asserts these when lifting a trained net). + self._check_sigmoid_init(float(lam_init), self.lam_soft_max, "lam_init/--lam-soft") + with torch.no_grad(): + self.net[-1].bias[3] = _logit(float(lam_init) / self.lam_soft_max) + if self.use_lam_hard: + self._check_sigmoid_init( + float(lam_hard_init), self.lam_hard_max, "lam_hard_init/--lam-hard-init" + ) + self.net[-1].bias[4] = _logit(float(lam_hard_init) / self.lam_hard_max) + + @staticmethod + def _check_sigmoid_init(init, mx, name): + if not (0.0 < init < mx): + raise ValueError( + f"a sigmoid lam head needs 0 < {name} < its max ({mx}); got {init!r} " + f"(logit(init/max) is +/-inf outside that range)" + ) def full_out_bias(self) -> torch.Tensor: - """The (out_dim,) RAW pre-softplus bias vector — abg bias, plus lam_init for a lam - head. The exporter writes this and the C++ folds log(expm1) at load, so the whole - output stack is one uniform softplus.""" + """The RAW pre-softplus bias the exporter writes as ``out_bias``. For a sigmoid net + (v2) it is the 3 abg biases only — the lam heads' init is carried in the last Linear + layer's bias, not here. For a v1 single-softplus lam net it is abg + lam_init, so the + whole output stack is one uniform softplus the C++ folds log(expm1) at load.""" + if self.lam_sigmoid: + return self.bias if self.use_lam: return torch.cat([self.bias, self.lam_raw_bias.reshape(1)]) return self.bias def _outputs(self, feat): - raw = self.net(feat) + torch.log(torch.expm1(self.full_out_bias())).unsqueeze(0) - return F.softplus(raw) # [N, out_dim] + net_out = self.net(feat) # [N, out_dim] + if not self.lam_sigmoid: + raw = net_out + torch.log(torch.expm1(self.full_out_bias())).unsqueeze(0) + return F.softplus(raw) + # Two-head sigmoid (v2): abg keep the softplus(log(expm1)) fold; the lam columns are + # lam_max*sigmoid(head) (the head bias is already in net[-1].bias). Byte-parity with the + # C++ coef_mlp forward (kFlagLamSigmoid) and material_nav.py:175-176. + abg = F.softplus(net_out[:, :3] + torch.log(torch.expm1(self.bias)).unsqueeze(0)) + cols = [abg, self.lam_soft_max * torch.sigmoid(net_out[:, 3:4])] + if self.use_lam_hard: + cols.append(self.lam_hard_max * torch.sigmoid(net_out[:, 4:5])) + return torch.cat(cols, dim=1) def forward(self, feat): c = self._outputs(feat) return c[:, 0], c[:, 1], c[:, 2] def coeffs_and_lam(self, feat): - """``(alpha, beta, gamma, lam_soft)`` — lam_soft is the LEARNED per-agent material - reroute strength (F_soft = -lam_soft * grad risk), the deployable replacement for the - fixed ``lam_soft`` dial. Requires ``use_lam``.""" + """``(alpha, beta, gamma, lam_soft)`` — or ``(..., lam_soft, lam_hard)`` for a two-head + net (``use_lam_hard``). lam_soft/lam_hard are the LEARNED per-agent reroute strengths + (F_soft = -lam_soft*grad risk, F_hard = -lam_hard*db*grad phi), the deployable + replacement for the fixed ``--lam-soft`` / ``--lam-hard`` dials. Requires ``use_lam``.""" if not self.use_lam: raise RuntimeError("this CoefMLP has no lam head (use_lam=False)") c = self._outputs(feat) + if self.use_lam_hard: + return c[:, 0], c[:, 1], c[:, 2], c[:, 3], c[:, 4] return c[:, 0], c[:, 1], c[:, 2], c[:, 3] @@ -1111,3 +1168,53 @@ def add_lam_head(model: CoefMLP, lam_init: float = 0.4) -> CoefMLP: out.net[4].bias[:3].copy_(model.net[4].bias) out.bias.copy_(model.bias) return out + + +def add_lam_heads( + model: CoefMLP, + soft_init: float = 0.4, + hard_init: float = 1.0, + lam_soft_max: float = 5.0, + lam_hard_max: float = 10.0, +) -> CoefMLP: + """Lift a trained ``CoefMLP`` into the paper's TWO-head sigmoid-bounded reroute net (the + deployable .cvcnav format v2): appends BOTH lam_soft (col 3) and lam_hard (col 4) as + ``lam_max*sigmoid(head)`` outputs (material_nav.py:175-176). Mirrors :func:`add_lam_head` + but basin-preserving for TWO heads: the appended final-layer weight rows are zeroed and + their biases set to ``logit(init/max)``, so each lam starts at its constant init; the + (alpha,beta,gamma) rows are copied verbatim so abg is bit-for-bit unchanged at init. Unlike + the v1 single-softplus :func:`add_lam_head`, the lam columns are sigmoid-bounded, so this net + exports as format v2 (kFlagLamSigmoid) and hard-fails to load on a pre-v2 libcvc host. + Returns a new model.""" + if getattr(model, "use_lam", False): + raise ValueError("add_lam_heads: model already has a lam head") + first = model.net[0] + hidden = first.out_features + out = CoefMLP( + hidden=hidden, + bias=tuple(model.bias.tolist()), + in_dim=model.in_dim, + use_mu=getattr(model, "use_mu", False), + use_risk=getattr(model, "use_risk", False), + use_lam=True, + lam_init=soft_init, + use_lam_hard=True, + lam_hard_init=hard_init, + lam_soft_max=lam_soft_max, + lam_hard_max=lam_hard_max, + ) + with torch.no_grad(): + out.net[0].weight.copy_(first.weight) + out.net[0].bias.copy_(first.bias) + out.net[2].weight.copy_(model.net[2].weight) + out.net[2].bias.copy_(model.net[2].bias) + # final layer: rows 0..2 = the abg head (copied verbatim), rows 3/4 = the lam heads with + # ZERO weights so they are position-independent at init, biases = logit(init/max) so they + # start at soft_init / hard_init (a sigmoid net's init lives in the last-layer bias). + out.net[4].weight.zero_() + out.net[4].weight[:3].copy_(model.net[4].weight) + out.net[4].bias[:3].copy_(model.net[4].bias) + out.net[4].bias[3] = _logit(float(soft_init) / lam_soft_max) + out.net[4].bias[4] = _logit(float(hard_init) / lam_hard_max) + out.bias.copy_(model.bias) + return out diff --git a/grl_snam/swarm.py b/grl_snam/swarm.py index 0516b4f..284768a 100644 --- a/grl_snam/swarm.py +++ b/grl_snam/swarm.py @@ -540,7 +540,13 @@ def step(self) -> None: else: feat = self._coef_feats(phi, nrm, carrot) mkw = self._material_kw() - if getattr(self.model, "use_lam", False): + if getattr(self.model, "use_lam_hard", False): + # TWO-head net: the net outputs BOTH lam_soft and lam_hard per agent; drive with + # both (as trained) instead of the fixed MaterialParams reroute strengths. + al, be, ga, lam_soft, lam_hard = self.model.coeffs_and_lam(feat) + if mkw: + mkw = dict(mkw, lam_soft=lam_soft, lam_hard=lam_hard) + elif getattr(self.model, "use_lam", False): # Deployable learned reroute: the net outputs lam_soft per agent; drive with # it (as trained) instead of the fixed MaterialParams lam_soft. al, be, ga, lam_soft = self.model.coeffs_and_lam(feat) diff --git a/grl_snam/tools/coef_export.py b/grl_snam/tools/coef_export.py index fc85aaf..44ce36c 100644 --- a/grl_snam/tools/coef_export.py +++ b/grl_snam/tools/coef_export.py @@ -14,7 +14,13 @@ u32 format_version, u32 flags, u32 in, u32 out, u32 num_layers, u64 arch_hash per layer: u32 rows, u32 cols, u32 act(0=identity,1=SiLU), f32 w[rows*cols], f32 b[rows] u32 out_bias_len, f32 out_bias[...] (raw bias; log(expm1) folded at load) + [v2 only, FLAG_LAM_SIGMOID] f32 lam_soft_max, f32 lam_hard_max (sigmoid ceilings) u32 meta_len, char meta[meta_len] (optional provenance) + +A two-head sigmoid net (CoefMLP.lam_sigmoid, i.e. use_lam_hard) writes format v2: out_bias_len +is 3 (the abg columns only; the lam heads' init lives in the last Linear bias), the two sigmoid +ceilings follow the out_bias block, and FLAG_LAM_SIGMOID is set so the C++ loader reads the lam +columns as lam_max*sigmoid(raw). A plain / single-softplus-lam net stays v1, byte-unchanged. """ from __future__ import annotations @@ -25,12 +31,16 @@ import numpy as np FORMAT_VERSION = 1 +FORMAT_VERSION_V2 = 2 # two-head sigmoid reroute (lam_soft, lam_hard) — see FLAG_LAM_SIGMOID FLAG_SOFTPLUS_LOG_EXPM1 = 1 << 0 # Input-feature-layout flags (must match cvc::nav::coef_mlp kFlagFeatMu/kFlagFeatRisk): a # 6-input net is ambiguous (grip mu vs terrain-risk), so the C++ drive reads the layout from # these flags rather than in_features(). Set from CoefMLP.use_mu / use_risk. FLAG_FEAT_MU = 1 << 1 FLAG_FEAT_RISK = 1 << 2 +# The lam OUTPUT columns (>= 3) are lam_max*sigmoid(raw), not softplus (matches +# cvc::nav::coef_mlp kFlagLamSigmoid). Set from CoefMLP.lam_sigmoid (use_lam_hard). +FLAG_LAM_SIGMOID = 1 << 3 _ACT_IDENTITY, _ACT_SILU = 0, 1 _U64 = (1 << 64) - 1 @@ -75,9 +85,11 @@ def write_coef_mlp(model, path, meta: bytes = b""): layers = _layers(model) in_f = int(layers[0][0].shape[1]) out_f = int(layers[-1][0].shape[0]) - # The RAW pre-softplus bias, one entry per output: (1,3,4) for abg, plus lam_init for a - # lam head. full_out_bias() keeps the whole output stack a single uniform softplus, so - # out_bias_len == out_f and the C++ forward needs no per-output special case. + # A two-head sigmoid net is format v2: the lam columns are lam_max*sigmoid(raw), not softplus. + sigmoid_lam = bool(getattr(model, "lam_sigmoid", False)) + # The RAW pre-softplus bias. For v1 this is one entry per output (abg + lam_init) so the whole + # stack is a single uniform softplus. For v2 full_out_bias() returns the 3 abg biases only — + # the lam heads' init lives in the last Linear bias, and the C++ folds log(expm1) for abg only. _ob = model.full_out_bias() if hasattr(model, "full_out_bias") else model.bias out_bias = _ob.detach().cpu().numpy().astype(np.float32) shape_act = [] @@ -86,16 +98,19 @@ def write_coef_mlp(model, path, meta: bytes = b""): ah = arch_hash(in_f, out_f, shape_act) # Encode the input-feature layout so the C++ drive builds the right columns for a widened # net (a bare 6-in net is grip; the risk flag disambiguates it). out_f already signals the - # lam head (out>=4), so it needs no flag. + # lam head (out>=4), so it needs no flag; FLAG_LAM_SIGMOID marks the sigmoid two-head form. flags = FLAG_SOFTPLUS_LOG_EXPM1 if getattr(model, "use_mu", False): flags |= FLAG_FEAT_MU if getattr(model, "use_risk", False): flags |= FLAG_FEAT_RISK + if sigmoid_lam: + flags |= FLAG_LAM_SIGMOID + version = FORMAT_VERSION_V2 if sigmoid_lam else FORMAT_VERSION with open(path, "wb") as f: f.write(b"CVNV") - f.write(struct.pack(" 0 (and < lam_hard_max=10); seeds the head, not a frozen dial.", + ) ap.add_argument( "--risk-lookahead", type=float, @@ -648,7 +669,7 @@ def main(argv=None): story.truth_grid(), story.bounds, meta["center"], meta["scale"], seed=args.seed ) risk_model = sdf_nav.add_risk_feature(sdf_nav.CoefMLP()) - if args.learned_lam: + if args.learned_lam or args.learned_lam_hard: # LEARNED reroute: the net outputs lam_soft per position instead of the fixed # --lam-soft dial — the deployable reroute lever (drive_step_material reads the 4th # output as per-agent lam_soft since libcvc #413). --lam-soft is the INIT here and @@ -658,7 +679,20 @@ def main(argv=None): "--learned-lam needs --lam-soft > 0 (it seeds the lam head; 0 would freeze " "it with a dead gradient). Use a small positive init, e.g. --lam-soft 0.4." ) - risk_model = sdf_nav.add_lam_head(risk_model, lam_init=args.lam_soft) + if args.learned_lam_hard: + # TWO-head (paper form): also learn lam_hard. Both lam columns become sigmoid- + # bounded (format v2). --lam-soft / --lam-hard-init seed the two heads and must + # be strictly inside (0, max) so logit(init/max) is finite. + if args.lam_hard_init <= 0.0: + raise SystemExit( + "--learned-lam-hard needs --lam-hard-init > 0 (it seeds the lam_hard " + "head; 0 folds to a +/-inf logit). Use e.g. --lam-hard-init 1.0." + ) + risk_model = sdf_nav.add_lam_heads( + risk_model, soft_init=args.lam_soft, hard_init=args.lam_hard_init + ) + else: + risk_model = sdf_nav.add_lam_head(risk_model, lam_init=args.lam_soft) curriculum = None if args.curriculum: from grl_snam.curriculum import build_city_curriculum @@ -686,12 +720,32 @@ def main(argv=None): model = train(args.steps, args.horizon, args.n, args.lr, args.seed) write_coef_mlp(model, args.out) if getattr(model, "use_risk", False) or getattr(model, "use_lam", False): - extra = f", out_dim={model.out_dim} (lam head)" if getattr(model, "use_lam", False) else "" + if getattr(model, "use_lam_hard", False): + extra = ( + f", out_dim={model.out_dim} (TWO-head sigmoid lam_soft+lam_hard, .cvcnav v2; " + f"maxes {model.lam_soft_max}/{model.lam_hard_max})" + ) + host = ( + "a libcvc host at or past the two-head sigmoid drive update (.cvcnav format v2 — " + "coef_mlp::has_lam_hard / kFlagLamSigmoid). A v2 blob HARD-FAILS to load on a " + "pre-v2 host, so land the host first" + ) + elif getattr(model, "use_lam", False): + extra = f", out_dim={model.out_dim} (lam head)" + host = ( + "a libcvc host at or past the risk-lookahead/learned-lam drive update " + "(transfix/libcvc #413)" + ) + else: + extra = "" + host = ( + "a libcvc host at or past the risk-lookahead/learned-lam drive update " + "(transfix/libcvc #413)" + ) print( - f"wrote {args.out} (grip/risk model, in_dim={model.in_dim}{extra}) — DEPLOYABLE on a " - "libcvc host at or past the risk-lookahead/learned-lam drive update (transfix/libcvc " - "#413): coef_feats builds the risk-lookahead column and drive_step_material reads the " - "lam output. Drive it through the MATERIAL path (drive_step_material / " + f"wrote {args.out} (grip/risk model, in_dim={model.in_dim}{extra}) — DEPLOYABLE on " + f"{host}: coef_feats builds the risk-lookahead column and drive_step_material reads " + "the lam output(s). Drive it through the MATERIAL path (drive_step_material / " "drive_step_material_ext), with a material stack + grip attached — the plain drive_step " "rejects a risk net. Round-trip + drive verified by libcvc nav_material_deploy_test." ) diff --git a/tests/test_coef_mlp_parity.py b/tests/test_coef_mlp_parity.py index ac668a2..0fc6507 100644 --- a/tests/test_coef_mlp_parity.py +++ b/tests/test_coef_mlp_parity.py @@ -81,3 +81,25 @@ def test_stale_arch_hash_is_rejected(tmp_path): bad.write_bytes(raw) with pytest.raises(Exception): nav_native.coef_mlp_forward(str(bad), np.zeros((1, 5), np.float32)) + + +def test_two_head_sigmoid_matches_torch(tmp_path): + # The paper's two-head reroute (format v2): the C++ coef_mlp forward of the exported .cvcnav + # must match torch on ALL 5 columns (abg softplus + lam_soft/lam_hard = max*sigmoid), the + # in-process byte+numeric cross-check for the v2 loader. Skips until pycvc is rebuilt on v2. + torch.manual_seed(2) + two = sdf_nav.add_lam_heads(sdf_nav.add_risk_feature(sdf_nav.CoefMLP()), 2.0, 4.0) + with torch.no_grad(): + two.net[-1].weight[3:].normal_(0, 0.5) # position-dependent lam heads + path = tmp_path / "two.cvcnav" + coef_export.write_coef_mlp(two, str(path)) + feats = np.random.default_rng(0).standard_normal((2000, two.in_dim)).astype(np.float32) + feats[0] = 50.0 # push the sigmoid + softplus tails + feats[1] = -50.0 + with torch.no_grad(): + ref = two._outputs(torch.from_numpy(feats)).numpy() # (N,5) + got = nav_native.coef_mlp_forward(str(path), feats) + assert got.shape == (2000, 5) + assert np.allclose(got, ref, rtol=1e-4, atol=1e-5), np.abs(got - ref).max() + assert (got[:, 3] >= 0).all() and (got[:, 3] <= 5.0 + 1e-4).all() # lam_soft bounded + assert (got[:, 4] >= 0).all() and (got[:, 4] <= 10.0 + 1e-4).all() # lam_hard bounded diff --git a/tests/test_lam_head.py b/tests/test_lam_head.py index 1fa7db2..16a1ec5 100644 --- a/tests/test_lam_head.py +++ b/tests/test_lam_head.py @@ -10,6 +10,7 @@ import struct import tempfile +import numpy as np import pytest import torch @@ -17,7 +18,11 @@ from grl_snam.fog_stories import STORIES, shrunk from grl_snam.material import city_material_grid from grl_snam.tools import coef_train -from grl_snam.tools.coef_export import write_coef_mlp +from grl_snam.tools.coef_export import ( + FLAG_LAM_SIGMOID, + FLAG_SOFTPLUS_LOG_EXPM1, + write_coef_mlp, +) def _scene(): @@ -137,3 +142,196 @@ def test_lam_net_scores_via_swarm(): assert 0.0 <= card.arrival_rate <= 1.0 with pytest.raises(ValueError, match="use_risk/use_lam net needs material"): evaluate(lam, scenes=1, agents=8, steps=20, material=False) + + +# ── two-head sigmoid (lam_soft, lam_hard) — .cvcnav format v2 ──────────────────────────────── +# The paper's two-head reroute (App A.4, material_nav.py:175-176): lam_soft = lam_soft_max* +# sigmoid(head), lam_hard = lam_hard_max*sigmoid(head). add_lam_heads lifts a trained abg net +# into it basin-preserving; the exporter writes it as format v2, byte-compatible with the C++ +# cvc::nav::coef_mlp v2 loader (nav_material_deploy_test.cpp). + + +def _decode_cvcnav(path): + """Pure-numpy .cvcnav decoder — an INDEPENDENT reference for the C++ loader/forward, so the + exporter's bytes can be checked without a pycvc build.""" + with open(path, "rb") as f: + b = f.read() + assert b[:4] == b"CVNV" + off = 4 + ver, flags, in_f, out_f, nl = struct.unpack_from(" 20.0, x, np.log1p(np.exp(x))) + + fold = np.log(np.expm1(dec["out_bias"])) + if dec["flags"] & FLAG_LAM_SIGMOID: + res = np.empty_like(a) + res[:, :3] = softplus(a[:, :3] + fold[None, :]) # abg keep the softplus fold + res[:, 3] = dec["lam_soft_max"] * (1.0 / (1.0 + np.exp(-a[:, 3]))) + if dec["out_f"] >= 5: + res[:, 4] = dec["lam_hard_max"] * (1.0 / (1.0 + np.exp(-a[:, 4]))) + return res + return softplus(a + fold[None, :]) + + +def test_add_lam_heads_identity_flags_and_init(): + field, grid = _scene() + mf = grid.field() + o = torch.tensor([[0.1, 0.0], [0.3, -0.2]]) + goal = torch.tensor([[1.0, 0.0], [0.5, 0.5]]) + feat = sdf_nav.coef_feats(field, o, goal, material=mf) + + risk = sdf_nav.add_risk_feature(sdf_nav.CoefMLP()) + two = sdf_nav.add_lam_heads(risk, soft_init=0.4, hard_init=1.0) + assert two.use_lam and two.use_lam_hard and two.lam_sigmoid + assert two.out_dim == 5 and two.use_risk and two.in_dim == risk.in_dim + assert two.lam_soft_max == 5.0 and two.lam_hard_max == 10.0 + + a0, b0, g0 = risk(feat) + a1, b1, g1, ls, lh = two.coeffs_and_lam(feat) + assert torch.allclose(a0, a1, atol=1e-6) # abg bit-identical after lifting + assert torch.allclose(b0, b1, atol=1e-6) + assert torch.allclose(g0, g1, atol=1e-6) + assert torch.allclose(ls, torch.tensor(0.4), atol=1e-5) # lam_soft starts at soft_init + assert torch.allclose(lh, torch.tensor(1.0), atol=1e-5) # lam_hard starts at hard_init + assert len(two(feat)) == 3 # forward() still returns abg + + with pytest.raises(ValueError, match="already has a lam head"): + sdf_nav.add_lam_heads(two) + + +def test_two_head_sigmoid_bounds_and_out_of_range_init(): + # sigmoid bounds: lam_soft in [0,5], lam_hard in [0,10] for any input. + two = sdf_nav.add_lam_heads(sdf_nav.add_risk_feature(sdf_nav.CoefMLP())) + with torch.no_grad(): # random extreme final-layer weights on the lam rows + two.net[-1].weight[3:].uniform_(-50, 50) + feat = torch.randn(500, two.in_dim) + _, _, _, ls, lh = two.coeffs_and_lam(feat) + assert (ls >= 0).all() and (ls <= 5.0 + 1e-4).all() + assert (lh >= 0).all() and (lh <= 10.0 + 1e-4).all() + # init strictly inside (0, max) — logit is +/-inf at the ends. + with pytest.raises(ValueError, match="0 < .* < its max"): + sdf_nav.CoefMLP(use_lam=True, use_lam_hard=True, lam_hard_init=0.0) + with pytest.raises(ValueError, match="0 < .* < its max"): + sdf_nav.CoefMLP(use_lam=True, use_lam_hard=True, lam_hard_init=10.0) + with pytest.raises(ValueError, match="requires use_lam"): + sdf_nav.CoefMLP(use_lam=False, use_lam_hard=True) + + +def test_two_head_export_v2_layout_and_forward_parity(tmp_path): + torch.manual_seed(1) + two = sdf_nav.add_lam_heads(sdf_nav.add_risk_feature(sdf_nav.CoefMLP()), 2.0, 4.0) + with torch.no_grad(): # make the lam heads position-dependent so parity is non-trivial + two.net[-1].weight[3:].normal_(0, 0.5) + p = str(tmp_path / "two_head.cvcnav") + write_coef_mlp(two, p) + + dec = _decode_cvcnav(p) + assert dec["ver"] == 2 + assert dec["flags"] & FLAG_LAM_SIGMOID + assert dec["flags"] & FLAG_SOFTPLUS_LOG_EXPM1 + assert dec["in_f"] == two.in_dim and dec["out_f"] == 5 + assert dec["out_bias"].shape[0] == 3 # abg only + assert dec["lam_soft_max"] == pytest.approx(5.0) + assert dec["lam_hard_max"] == pytest.approx(10.0) + assert dec["end"] == dec["size"] # trailer + meta consume the file exactly + + # byte+numeric parity: the numpy decode of the exported blob reproduces the torch forward. + feats = np.random.default_rng(0).standard_normal((256, two.in_dim)).astype(np.float32) + with torch.no_grad(): + out = two._outputs(torch.from_numpy(feats)).numpy() + ref = _forward_numpy(dec, feats) + assert np.allclose(out, ref, rtol=1e-4, atol=1e-5), np.abs(out - ref).max() + # and the exported lam columns are within the sigmoid ceilings. + assert (ref[:, 3] >= 0).all() and (ref[:, 3] <= 5.0 + 1e-4).all() + assert (ref[:, 4] >= 0).all() and (ref[:, 4] <= 10.0 + 1e-4).all() + + +def test_single_softplus_lam_still_exports_v1(tmp_path): + # BACK-COMPAT: a single-softplus lam net (add_lam_head) stays format v1, no sigmoid flag, + # out_bias_len == out_features — byte-unchanged from before the v2 bump. + lam = sdf_nav.add_lam_head(sdf_nav.add_risk_feature(sdf_nav.CoefMLP()), lam_init=0.5) + assert not getattr(lam, "lam_sigmoid", False) and not lam.use_lam_hard + p = str(tmp_path / "v1.cvcnav") + write_coef_mlp(lam, p) + dec = _decode_cvcnav(p) + assert dec["ver"] == 1 + assert not (dec["flags"] & FLAG_LAM_SIGMOID) + assert dec["out_f"] == 4 and dec["out_bias"].shape[0] == 4 # all-softplus, one bias per output + assert dec["lam_soft_max"] is None and dec["end"] == dec["size"] + + +def test_train_bicycle_learns_two_head_lam(): + _, grid = _scene() + model = sdf_nav.add_lam_heads(sdf_nav.add_risk_feature(sdf_nav.CoefMLP()), 0.4, 1.0) + m = coef_train.train_bicycle( + steps=12, horizon=12, n=48, grid=64, seed=1, model=model, material=grid, w_risk=6.0 + ) + field, g2 = _scene() + feat = sdf_nav.coef_feats(field, torch.zeros(4, 2), torch.ones(4, 2), material=g2.field()) + _, _, _, ls, lh = m.coeffs_and_lam(feat) + # both heads stay within their sigmoid ceilings after training + assert (ls >= 0).all() and (ls <= 5.0).all() + assert (lh >= 0).all() and (lh <= 10.0).all() + + +def test_cli_rejects_learned_lam_hard_with_nonpositive_init(): + with pytest.raises(SystemExit, match="lam-hard-init > 0"): + coef_train.main( + [ + "--rollout", + "bicycle", + "--w-risk", + "1", + "--learned-lam-hard", + "--lam-soft", + "0.4", + "--lam-hard-init", + "0", + "--steps", + "1", + ] + ) From 66882714284aae84bae1f2acedb53b997b7dc2b8 Mon Sep 17 00:00:00 2001 From: Joe Rivera Date: Sun, 27 Sep 2026 16:39:11 -0500 Subject: [PATCH 2/2] test(coef_mlp_parity): skip the pycvc v2 round-trip until pycvc is rebuilt on v2 libcvc The in-process C++<->torch round-trip loads a v2 .cvcnav through the INSTALLED pycvc, which still links a pre-v2 libcvc and raises 'unsupported .cvcnav format version' in hermetic CI. The comment already said it should skip until pycvc is rebuilt on v2 (deploy step 9); implement that skip. The pure-numpy decoder in test_lam_head already proves the v2 byte layout + forward math, so byte-parity coverage is unaffected. --- tests/test_coef_mlp_parity.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/tests/test_coef_mlp_parity.py b/tests/test_coef_mlp_parity.py index 0fc6507..4fb0b68 100644 --- a/tests/test_coef_mlp_parity.py +++ b/tests/test_coef_mlp_parity.py @@ -98,7 +98,16 @@ def test_two_head_sigmoid_matches_torch(tmp_path): feats[1] = -50.0 with torch.no_grad(): ref = two._outputs(torch.from_numpy(feats)).numpy() # (N,5) - got = nav_native.coef_mlp_forward(str(path), feats) + try: + got = nav_native.coef_mlp_forward(str(path), feats) + except RuntimeError as e: + # The installed pycvc still links a pre-v2 libcvc, which rejects the v2 .cvcnav + # ("unsupported .cvcnav format version"). This in-process round-trip runs once pycvc is + # rebuilt on the v2 libcvc (deploy step 9); until then the pure-numpy decoder in + # test_lam_head already proves the v2 byte layout + forward math, so skip rather than fail. + if "format version" in str(e): + pytest.skip(f"installed pycvc predates .cvcnav v2 (rebuild on v2 libcvc): {e}") + raise assert got.shape == (2000, 5) assert np.allclose(got, ref, rtol=1e-4, atol=1e-5), np.abs(got - ref).max() assert (got[:, 3] >= 0).all() and (got[:, 3] <= 5.0 + 1e-4).all() # lam_soft bounded