diff --git a/inc/cvc/nav/coef_mlp.h b/inc/cvc/nav/coef_mlp.h index f4078d45..de742202 100644 --- a/inc/cvc/nav/coef_mlp.h +++ b/inc/cvc/nav/coef_mlp.h @@ -49,8 +49,14 @@ namespace nav { class coef_mlp { public: - // The `.cvcnav` format this build reads/writes. - static constexpr std::uint32_t kFormatVersion = 1; + // The `.cvcnav` format this build WRITES. v2 adds the paper's two-head sigmoid + // reroute (kFlagLamSigmoid) and the lam_soft_max/lam_hard_max trailer. The loader + // ACCEPTS {1, 2}: a v1 file still loads and forwards identically, and a v2 file + // hard-fails on a pre-v2 host rather than silently misreading the trailer (safe). + // A plain all-softplus net (no sigmoid lam) is still written as v1 so it loads on + // old hosts; only a sigmoid-lam net bumps the version. + static constexpr std::uint32_t kFormatVersion = 2; + static constexpr std::uint32_t kFormatVersionLegacy = 1; // pre-sigmoid, all-softplus // flags bit 0: the output is softplus(net + log(expm1(out_bias))) — the // CoefMLP bias-toward-a-known-good-basin trick; the raw out_bias is stored and // log(expm1(.)) is folded once at load. @@ -63,6 +69,13 @@ class coef_mlp { // historical base-5 (or a bare 6-input grip net, treated as mu for back-compat). static constexpr std::uint32_t kFlagFeatMu = 1u << 1; static constexpr std::uint32_t kFlagFeatRisk = 1u << 2; + // flags bit 3: the lam OUTPUT columns (index >= 3: col 3 = lam_soft, col 4 = lam_hard) + // are `lam_max * sigmoid(raw)` — the paper's two-head sigmoid-bounded reroute + // (material_nav.py:175-176) — NOT the softplus(log(expm1)) fold used for cols 0..2 + // (alpha, beta, gamma). When set: the abg columns keep the softplus fold, the lam + // columns carry NO out_bias offset (their bias lives in the last Linear layer), and + // the file stores lam_soft_max / lam_hard_max as a v2 trailer after the out_bias block. + static constexpr std::uint32_t kFlagLamSigmoid = 1u << 3; // Max layer dimension the CPU forward() supports (its stack activation arrays). // load()/from_layers reject a wider net at the boundary rather than overflowing. // The CUDA drive (drive.cu d_mlp) has a tighter cap (64) it guards separately. @@ -96,12 +109,16 @@ class coef_mlp { // loaded .cvcnav. Throws if the shapes do not chain in -> ... -> out. // ``extra_flags`` OR-in the feature/output flags (kFlagFeatMu / kFlagFeatRisk) beyond the // always-on softplus fold — so a C++ trainer (or a test) can stamp a widened net's layout. + // When ``extra_flags`` includes kFlagLamSigmoid the net is a v2 two-head sigmoid policy: + // ``out_bias_raw`` is then the THREE abg raw biases only (the lam heads' init lives in the + // last layer's bias b, exactly as torch), and ``lam_soft_max`` / ``lam_hard_max`` bound the + // sigmoid lam columns. A non-sigmoid net stays v1 (fmt 1) so it loads on a pre-v2 host. static coef_mlp from_layers(int in, int out, const std::vector &rows, const std::vector &cols, const std::vector &act, const std::vector> &w, const std::vector> &b, - const std::vector &out_bias_raw, - std::uint32_t extra_flags = 0); + const std::vector &out_bias_raw, std::uint32_t extra_flags = 0, + float lam_soft_max = 5.0f, float lam_hard_max = 10.0f); // Serialize to the versioned `.cvcnav` (byte-identical layout to // grl_snam.tools.coef_export.write_coef_mlp), so a policy trained in pure C++ @@ -124,7 +141,12 @@ class coef_mlp { // with no risk flag is grip (the historical widen_coef_mlp output), so mu is implied there. bool has_risk() const { return (flags_ & kFlagFeatRisk) != 0; } bool has_mu() const { return (flags_ & kFlagFeatMu) != 0 || (in_ == 6 && !has_risk()); } - bool has_lam() const { return out_ >= 4; } // 4th output = learned reroute lam_soft + bool has_lam() const { return out_ >= 4; } // 4th output (col 3) = learned reroute lam_soft + bool has_lam_hard() const { return out_ >= 5; } // 5th output (col 4) = learned lam_hard + // True when the lam columns are lam_max*sigmoid(raw) (paper two-head form), not softplus. + bool lam_sigmoid() const { return (flags_ & kFlagLamSigmoid) != 0; } + float lam_soft_max() const { return lam_soft_max_; } + float lam_hard_max() const { return lam_hard_max_; } std::uint64_t arch_hash() const { return arch_hash_; } // The architecture hash (FNV-1a over in/out/num_layers and each layer's @@ -152,8 +174,12 @@ class coef_mlp { }; std::vector layers_; std::vector out_bias_off_; // = log(expm1(out_bias)), folded once at load + // (0 for sigmoid lam columns — bias is in the last layer) int in_ = 0, out_ = 0; std::uint32_t fmt_ = 0, flags_ = 0; + // Sigmoid lam ceilings (kFlagLamSigmoid): lam_soft = lam_soft_max_*sigmoid(raw), + // lam_hard = lam_hard_max_*sigmoid(raw). Defaults match coef_energy_net (5.0 / 10.0). + float lam_soft_max_ = 5.0f, lam_hard_max_ = 10.0f; std::uint64_t arch_hash_ = 0; void build_from_bytes(const std::uint8_t *p, std::size_t n); diff --git a/inc/cvc/nav/drive.h b/inc/cvc/nav/drive.h index 9537c0c8..29d14f47 100644 --- a/inc/cvc/nav/drive.h +++ b/inc/cvc/nav/drive.h @@ -240,8 +240,9 @@ void bicycle_rollout(const field_stack &f, float *o, float *th, float *sp, const // (nav_stats.h); keep the two PODs semantically in lockstep. struct drive_telemetry { // CoefMLP policy output this tick (the al/be/ga fed to the rollout; lam_soft is the - // effective material soft-weight, 0 on the non-material path). - float alpha = 0, beta = 0, gamma = 0, lam_soft = 0; + // effective material soft-weight, 0 on the non-material path; lam_hard is the ungated + // hard-hazard weight — learned per-agent for a two-head sigmoid net, else the fixed dial). + float alpha = 0, beta = 0, gamma = 0, lam_soft = 0, lam_hard = 0; float mu = 1; // underfoot grip at the agent cell (1 = dry / no grip field) float mrisk = 0; // material risk at the agent cell (0 = no material stack) float ext_fx = 0, ext_fy = 0; // applied external force in the drive frame (0 = no ext channel) diff --git a/src/cvc/nav/coef_mlp.cpp b/src/cvc/nav/coef_mlp.cpp index 4d320b1f..e83f87de 100644 --- a/src/cvc/nav/coef_mlp.cpp +++ b/src/cvc/nav/coef_mlp.cpp @@ -34,7 +34,13 @@ // u64 arch_hash // per layer: u32 rows, u32 cols, u32 act, f32 w[rows*cols], f32 b[rows] // u32 out_bias_len ; f32 out_bias[out_bias_len] (raw bias; log(expm1) at load) +// [v2 only, kFlagLamSigmoid] f32 lam_soft_max ; f32 lam_hard_max (sigmoid ceilings) // u32 meta_len ; char meta[meta_len] (provenance, ignored here) +// +// v2 (two-head sigmoid, kFlagLamSigmoid): out_bias_len == 3 (the abg raw biases only); +// the lam columns (>= 3) are lam_max*sigmoid(raw) with the raw head bias in the last +// Linear layer, so they carry no out_bias offset, and the two maxes follow the out_bias +// block. v1 stays out_bias_len == out_features, all columns softplus. #include #include @@ -65,6 +71,11 @@ inline float softplus(float x) { return x > 20.0f ? x : std::log1p(std::exp(x)); } +inline float sigmoidf(float x) { + // torch sigmoid; matches the lam_max*sigmoid two-head form in material_nav.py. + return 1.0f / (1.0f + std::exp(-x)); +} + // Read a trivially-copyable value from a little-endian byte cursor, advancing it. template T take(const std::uint8_t *&p, const std::uint8_t *end) { if (p + sizeof(T) > end) @@ -114,7 +125,10 @@ void coef_mlp::build_from_bytes(const std::uint8_t *p, std::size_t n) { if (std::memcmp(magic, "CVNV", 4) != 0) throw std::runtime_error("cvc::nav::coef_mlp: bad magic (not a .cvcnav file)"); fmt_ = take(p, end); - if (fmt_ != kFormatVersion) + // Accept {1, 2}: a v1 file still loads and forwards identically; a v2 file (two-head + // sigmoid) is read below via kFlagLamSigmoid. A newer version hard-fails here rather + // than silently misreading a trailer this build does not know. + if (fmt_ != kFormatVersionLegacy && fmt_ != kFormatVersion) throw std::runtime_error("cvc::nav::coef_mlp: unsupported .cvcnav format version"); flags_ = take(p, end); in_ = static_cast(take(p, end)); @@ -150,16 +164,35 @@ void coef_mlp::build_from_bytes(const std::uint8_t *p, std::size_t n) { const std::uint32_t obn = take(p, end); std::vector out_bias; take_floats(p, end, out_bias, obn); - out_bias_off_.resize(out_bias.size()); - if (flags_ & kFlagSoftplusLogExpm1) { - // Fold log(expm1(bias)) once, in float32, matching torch.log(torch.expm1(.)). - for (std::size_t i = 0; i < out_bias.size(); ++i) - out_bias_off_[i] = std::log(std::expm1(out_bias[i])); + const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0; + if (sigmoid_lam) { + // v2 two-head sigmoid: out_bias covers ONLY the 3 abg columns (softplus fold); the lam + // columns (>= 3) get a zero offset (their bias is in the last layer) and are read as + // lam_max*sigmoid(raw). The two ceilings follow the out_bias block. + if (fmt_ != kFormatVersion) + throw std::runtime_error("cvc::nav::coef_mlp: kFlagLamSigmoid requires .cvcnav format v2"); + if (out_ < 4) + throw std::runtime_error("cvc::nav::coef_mlp: sigmoid-lam net needs out_features >= 4"); + if (obn != 3) + throw std::runtime_error("cvc::nav::coef_mlp: v2 sigmoid out_bias must be the 3 abg columns"); + out_bias_off_.assign(static_cast(out_), 0.0f); + for (int i = 0; i < 3; ++i) + out_bias_off_[i] = + (flags_ & kFlagSoftplusLogExpm1) ? std::log(std::expm1(out_bias[i])) : out_bias[i]; + lam_soft_max_ = take(p, end); + lam_hard_max_ = take(p, end); } else { - out_bias_off_ = out_bias; // used as a plain additive offset + out_bias_off_.resize(out_bias.size()); + if (flags_ & kFlagSoftplusLogExpm1) { + // Fold log(expm1(bias)) once, in float32, matching torch.log(torch.expm1(.)). + for (std::size_t i = 0; i < out_bias.size(); ++i) + out_bias_off_[i] = std::log(std::expm1(out_bias[i])); + } else { + out_bias_off_ = out_bias; // used as a plain additive offset + } + if (static_cast(out_bias_off_.size()) != out_) + throw std::runtime_error("cvc::nav::coef_mlp: out_bias length != out_features"); } - if (static_cast(out_bias_off_.size()) != out_) - throw std::runtime_error("cvc::nav::coef_mlp: out_bias length != out_features"); // A meta trailer may follow; it is optional and ignored here. } @@ -171,7 +204,7 @@ coef_mlp coef_mlp::load_from_memory(const void *data, std::size_t nbytes) { coef_mlp coef_mlp::default_biased() { coef_mlp m; - m.fmt_ = kFormatVersion; + m.fmt_ = kFormatVersionLegacy; // a plain all-softplus abg net stays v1 (loads on old hosts) m.flags_ = kFlagSoftplusLogExpm1; m.in_ = 5; m.out_ = 3; @@ -198,16 +231,21 @@ coef_mlp coef_mlp::from_layers(int in, int out, const std::vector &rows, const std::vector &cols, const std::vector &act, const std::vector> &w, const std::vector> &b, - const std::vector &out_bias_raw, std::uint32_t extra_flags) { + const std::vector &out_bias_raw, std::uint32_t extra_flags, + float lam_soft_max, float lam_hard_max) { const int num_layers = static_cast(rows.size()); if (static_cast(cols.size()) != num_layers || static_cast(act.size()) != num_layers || static_cast(w.size()) != num_layers || static_cast(b.size()) != num_layers) throw std::runtime_error("cvc::nav::coef_mlp::from_layers: ragged layer arrays"); + const bool sigmoid_lam = (extra_flags & kFlagLamSigmoid) != 0; coef_mlp m; - m.fmt_ = kFormatVersion; + // A sigmoid-lam net is the v2 format; a plain net stays v1 so it loads on a pre-v2 host. + m.fmt_ = sigmoid_lam ? kFormatVersion : kFormatVersionLegacy; m.flags_ = kFlagSoftplusLogExpm1 | extra_flags; m.in_ = in; m.out_ = out; + m.lam_soft_max_ = lam_soft_max; + m.lam_hard_max_ = lam_hard_max; m.layers_.resize(num_layers); std::vector sa; int prev = in; @@ -233,11 +271,24 @@ coef_mlp coef_mlp::from_layers(int in, int out, const std::vector &rows, } if (prev != out) throw std::runtime_error("cvc::nav::coef_mlp::from_layers: last layer does not produce out"); - if (static_cast(out_bias_raw.size()) != out) - throw std::runtime_error("cvc::nav::coef_mlp::from_layers: out_bias length != out"); - m.out_bias_off_.resize(out); - for (int i = 0; i < out; ++i) - m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i])); + if (sigmoid_lam) { + // Two-head sigmoid: out_bias_raw is the 3 abg biases; the lam heads' init lives in the + // last layer's bias b (exactly as torch), so the lam columns carry a zero offset. + if (out < 4) + throw std::runtime_error("cvc::nav::coef_mlp::from_layers: sigmoid-lam net needs out >= 4"); + if (static_cast(out_bias_raw.size()) != 3) + throw std::runtime_error( + "cvc::nav::coef_mlp::from_layers: sigmoid net out_bias must be the 3 abg biases"); + m.out_bias_off_.assign(static_cast(out), 0.0f); + for (int i = 0; i < 3; ++i) + m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i])); + } else { + if (static_cast(out_bias_raw.size()) != out) + throw std::runtime_error("cvc::nav::coef_mlp::from_layers: out_bias length != out"); + m.out_bias_off_.resize(out); + for (int i = 0; i < out; ++i) + m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i])); + } m.arch_hash_ = compute_arch_hash(in, out, sa); return m; } @@ -267,12 +318,20 @@ void coef_mlp::save(const std::string &path, const std::string &meta) const { wf(L.b); } // The file stores the RAW out_bias; recover it from the folded offset - // (log(expm1(raw)) -> raw = softplus(off)) so save/load round-trips. - std::vector raw(out_bias_off_.size()); - for (std::size_t i = 0; i < raw.size(); ++i) + // (log(expm1(raw)) -> raw = softplus(off)) so save/load round-trips. In v2 sigmoid mode + // only the 3 abg columns carry a bias (the lam heads' bias is in the last layer), so the + // block is length 3 and the two ceilings follow it — byte-identical to the Python exporter. + const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0; + const std::size_t nbias = sigmoid_lam ? 3u : out_bias_off_.size(); + std::vector raw(nbias); + for (std::size_t i = 0; i < nbias; ++i) raw[i] = (flags_ & kFlagSoftplusLogExpm1) ? softplus(out_bias_off_[i]) : out_bias_off_[i]; w32(static_cast(raw.size())); wf(raw); + if (sigmoid_lam) { + f.write(reinterpret_cast(&lam_soft_max_), 4); + f.write(reinterpret_cast(&lam_hard_max_), 4); + } w32(static_cast(meta.size())); if (!meta.empty()) f.write(meta.data(), static_cast(meta.size())); @@ -387,8 +446,17 @@ void coef_mlp::forward(const float *feats, int n, float *out, int num_threads, a[o] = b[o]; } float *o = out + static_cast(s) * out_; - for (int k = 0; k < out_; ++k) - o[k] = softplus(a[k] + out_bias_off_[k]); + const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0; + for (int k = 0; k < out_; ++k) { + const float pre = a[k] + out_bias_off_[k]; + if (sigmoid_lam && k >= 3) { + // Two-head sigmoid reroute: col 3 = lam_soft, col 4 = lam_hard, each lam_max*sigmoid. + const float mx = (k == 3) ? lam_soft_max_ : lam_hard_max_; + o[k] = mx * sigmoidf(pre); + } else { + o[k] = softplus(pre); + } + } }); } diff --git a/src/cvc/nav/drive.cpp b/src/cvc/nav/drive.cpp index b2d4d609..b33d0e8c 100644 --- a/src/cvc/nav/drive.cpp +++ b/src/cvc/nav/drive.cpp @@ -331,7 +331,8 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl // Telemetry temporaries — write-only, never read by the drive math, so a null `tel` path is // byte-identical. They hold the LAST substep's value at loop exit (mu/mrisk/ext default to the // no-field case; steer/curv/clear are overwritten every substep). - float tel_mu = 1.0f, tel_mrisk = 0.0f, tel_lam = 0.0f, tel_ext_x = 0.0f, tel_ext_y = 0.0f; + float tel_mu = 1.0f, tel_mrisk = 0.0f, tel_lam = 0.0f, tel_lam_hard = 0.0f, tel_ext_x = 0.0f, + tel_ext_y = 0.0f; float tel_steer = 0.0f, tel_curv = 0.0f; for (int s = 0; s < v.nsub; ++s) { @@ -414,6 +415,7 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl const float db = -(1.0f / (1.0f + std::exp(-(mat->k_sharp * (mat->d_hat_m - mphi))))); const float ls = mat->lam_soft[i], lh = mat->lam_hard[i]; tel_lam = ls; + tel_lam_hard = lh; const float fsx = -ls * mgrx; const float fsy = -ls * mgry; const float fhx = (-lh * db) * mgpx; @@ -554,6 +556,7 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl t.beta = bei; t.gamma = gai; t.lam_soft = tel_lam; + t.lam_hard = tel_lam_hard; t.mu = tel_mu; t.mrisk = tel_mrisk; t.ext_fx = tel_ext_x; @@ -753,15 +756,23 @@ void drive_step_material(const field_stack &f, float *o, float *th, float *sp, c ga[i] = coef[static_cast(out_w) * i + 2]; } // Learned reroute: a lam-head net's 4th output is the per-agent lam_soft, overriding the - // material_drive's fixed/gated column (the deployable twin of grl_snam coeffs_and_lam). + // material_drive's fixed/gated column (the deployable twin of grl_snam coeffs_and_lam). A + // two-head sigmoid net (has_lam_hard) also learns lam_hard from its 5th output; lam_hard is + // never gated (material.h:37,161), so it mirrors lam_soft's extraction with no witness gate. material_drive md = mat; - std::vector lam_learned; + std::vector lam_learned, lam_hard_learned; if (model.has_lam()) { lam_learned.resize(n); for (int i = 0; i < n; ++i) lam_learned[i] = coef[static_cast(out_w) * i + 3]; md.lam_soft = lam_learned.data(); } + if (model.has_lam_hard()) { + lam_hard_learned.resize(n); + for (int i = 0; i < n; ++i) + lam_hard_learned[i] = coef[static_cast(out_w) * i + 4]; + md.lam_hard = lam_hard_learned.data(); + } rollout_impl(f, o, th, sp, carrot, al.data(), be.data(), ga.data(), n, map_id, v, md.stack ? &md : nullptr, nullptr, minclr_out, num_threads, nullptr, tel); } @@ -845,14 +856,22 @@ void drive_step_material_ext(const field_stack &f, float *o, float *th, float *s be[i] = coef[static_cast(out_w) * i + 1]; ga[i] = coef[static_cast(out_w) * i + 2]; } + // Same learned lam_soft (+ optional two-head lam_hard) extraction as drive_step_material; see + // there for the gate note. lam_hard is never gated. material_drive md = mat; - std::vector lam_learned; + std::vector lam_learned, lam_hard_learned; if (model.has_lam()) { lam_learned.resize(n); for (int i = 0; i < n; ++i) lam_learned[i] = coef[static_cast(out_w) * i + 3]; md.lam_soft = lam_learned.data(); } + if (model.has_lam_hard()) { + lam_hard_learned.resize(n); + for (int i = 0; i < n; ++i) + lam_hard_learned[i] = coef[static_cast(out_w) * i + 4]; + md.lam_hard = lam_hard_learned.data(); + } rollout_impl(f, o, th, sp, carrot, al.data(), be.data(), ga.data(), n, map_id, v, md.stack ? &md : nullptr, ext.sample ? &ext : nullptr, minclr_out, num_threads, nullptr, tel); diff --git a/src/cvc/tests/nav_material_deploy_test.cpp b/src/cvc/tests/nav_material_deploy_test.cpp index 39b84dfa..8344de93 100644 --- a/src/cvc/tests/nav_material_deploy_test.cpp +++ b/src/cvc/tests/nav_material_deploy_test.cpp @@ -15,6 +15,7 @@ #include #include #include +#include #include #include #include @@ -123,6 +124,25 @@ static std::string tmp_cvcnav(const char *tag) { return (fs::temp_directory_path() / (std::string("cvc_deploy_") + tag + ".cvcnav")).string(); } +// The two-head sigmoid (v2) twin of make_net: a single-layer net with ZERO linear weights whose +// lam heads' init lives in the last layer's bias (exactly as torch's add_lam_heads). With zero +// weights the pre-activation is that bias, so — by construction — forward gives the closed form +// abg=(1,3,4), lam_soft = smax*sigmoid(logit(soft/smax)) = soft, lam_hard = hmax*..(hard) = hard, +// which is the numpy reference the parity test checks against. out>=5 => two-head (lam_hard). +static coef_mlp make_sigmoid_net(int in, int out, std::uint32_t featflags, float soft_init, + float hard_init, float smax = 5.0f, float hmax = 10.0f) { + auto logit = [](float x) { return std::log(x / (1.0f - x)); }; + std::vector b(static_cast(out), 0.0f); + if (out >= 4) + b[3] = logit(soft_init / smax); + if (out >= 5) + b[4] = logit(hard_init / hmax); + const std::vector abg = {1.0f, 3.0f, 4.0f}; // v2 out_bias covers ONLY the abg columns + return coef_mlp::from_layers(in, out, {out}, {in}, {0}, + {std::vector((std::size_t)out * in, 0.0f)}, {b}, abg, + featflags | coef_mlp::kFlagLamSigmoid, smax, hmax); +} + } // namespace // (1) The deploy contract: a widened grip+risk+lam net serializes to .cvcnav and reloads with its @@ -320,3 +340,203 @@ TEST(NavMaterialDeploy, ShippedRiskOnlyLamNetRoundTripsAndDrives) { EXPECT_GT(moved, 0) << "the shipped 6-in risk-only+lam net did not drive the agents"; fs::remove(path); } + +// ── two-head sigmoid (.cvcnav format v2) ──────────────────────────────────────────────────── +// The paper's two-head reroute: col 3 = lam_soft, col 4 = lam_hard, each lam_max*sigmoid(raw) +// (material_nav.py:175-176). These pin the v2 format contract the Python exporter must match. + +// (6) A 5-output v2 net serializes to .cvcnav and reloads with its two-head layout AND the +// sigmoid ceilings intact — the deploy contract for a two-head policy. +TEST(NavMaterialDeploy, TwoHeadSigmoidNetRoundTripsAndKeepsMaxes) { + coef_mlp net = make_sigmoid_net(6, 5, coef_mlp::kFlagFeatRisk, 2.0f, 4.0f); + ASSERT_EQ(net.out_features(), 5); + EXPECT_TRUE(net.has_lam()); + EXPECT_TRUE(net.has_lam_hard()); + EXPECT_TRUE(net.lam_sigmoid()); + EXPECT_EQ(net.format_version(), 2u); + EXPECT_FLOAT_EQ(net.lam_soft_max(), 5.0f); + EXPECT_FLOAT_EQ(net.lam_hard_max(), 10.0f); + const std::string path = tmp_cvcnav("twohead"); + net.save(path); + coef_mlp back = coef_mlp::load(path); + EXPECT_EQ(back.format_version(), 2u); + EXPECT_EQ(back.out_features(), 5); + EXPECT_TRUE(back.has_lam_hard()); + EXPECT_TRUE(back.lam_sigmoid()); + EXPECT_FLOAT_EQ(back.lam_soft_max(), 5.0f); + EXPECT_FLOAT_EQ(back.lam_hard_max(), 10.0f); + // forward is byte-identical across the round-trip (save recovers the raw bias, load re-folds). + float feat[6] = {0.2f, 5.0f, 0.3f, -0.4f, 0.1f, 0.7f}; + float a[5], b2[5]; + net.forward(feat, 1, a, 1); + back.forward(feat, 1, b2, 1); + EXPECT_EQ(std::memcmp(a, b2, sizeof(a)), 0); + fs::remove(path); +} + +// (7) Forward PARITY vs the numpy reference (~1e-4): a zero-weight v2 net's outputs are the +// closed form abg=(1,3,4), lam_soft = smax*sigmoid(logit(soft/smax)) = soft, lam_hard = hard, +// independent of the input; and the lam columns are bounded in [0, max]. +TEST(NavMaterialDeploy, SigmoidLamColumnsMatchReferenceAndAreBounded) { + const float smax = 5.0f, hmax = 10.0f, soft = 2.0f, hard = 7.0f; + coef_mlp net = make_sigmoid_net(6, 5, coef_mlp::kFlagFeatRisk, soft, hard, smax, hmax); + std::mt19937 rng(3); + std::uniform_real_distribution u(-3.0f, 3.0f); + for (int t = 0; t < 200; ++t) { + float feat[6]; + for (float &x : feat) + x = u(rng); + float o[5]; + net.forward(feat, 1, o, 1); + EXPECT_NEAR(o[0], 1.0f, 1e-4f); + EXPECT_NEAR(o[1], 3.0f, 1e-4f); + EXPECT_NEAR(o[2], 4.0f, 1e-4f); + EXPECT_NEAR(o[3], soft, 1e-4f); // lam_soft = numpy reference + EXPECT_NEAR(o[4], hard, 1e-4f); // lam_hard = numpy reference + EXPECT_GE(o[3], 0.0f); + EXPECT_LE(o[3], smax); + EXPECT_GE(o[4], 0.0f); + EXPECT_LE(o[4], hmax); + } +} + +// (7b) The sigmoid ceiling actually clamps: a net with huge lam-row weights drives the +// pre-activation to +/- inf, and the lam columns must saturate to {0, max}, never beyond. +TEST(NavMaterialDeploy, SigmoidLamStaysBoundedUnderExtremeInput) { + const int in = 6, out = 5; + std::vector w((std::size_t)out * in, 0.0f); + for (int j = 0; j < in; ++j) { + w[(std::size_t)3 * in + j] = 100.0f; // lam_soft row: huge +ve -> sigmoid -> 1 + w[(std::size_t)4 * in + j] = -100.0f; // lam_hard row: huge -ve -> sigmoid -> 0 + } + const std::vector abg = {1.0f, 3.0f, 4.0f}; + coef_mlp net = coef_mlp::from_layers( + in, out, {out}, {in}, {0}, {w}, {std::vector((std::size_t)out, 0.0f)}, abg, + coef_mlp::kFlagFeatRisk | coef_mlp::kFlagLamSigmoid, 5.0f, 10.0f); + float fpos[6] = {9, 9, 9, 9, 9, 9}, fneg[6] = {-9, -9, -9, -9, -9, -9}, o[5]; + net.forward(fpos, 1, o, 1); + EXPECT_GE(o[3], 0.0f); + EXPECT_LE(o[3], 5.0f); + EXPECT_GE(o[4], 0.0f); + EXPECT_LE(o[4], 10.0f); + EXPECT_NEAR(o[3], 5.0f, 1e-3f); // saturates to lam_soft_max + EXPECT_NEAR(o[4], 0.0f, 1e-3f); + net.forward(fneg, 1, o, 1); + EXPECT_LE(o[3], 5.0f); + EXPECT_LE(o[4], 10.0f); + EXPECT_NEAR(o[3], 0.0f, 1e-3f); + EXPECT_NEAR(o[4], 10.0f, 1e-3f); // saturates to lam_hard_max +} + +// (8) The ON-DISK v2 byte layout the Python exporter must match: version 2, the sigmoid+softplus +// flags, out_bias_len == 3 (abg only), then the two f32 ceilings, then the meta trailer. +TEST(NavMaterialDeploy, V2ByteLayoutMatchesExporterContract) { + coef_mlp net = make_sigmoid_net(6, 5, coef_mlp::kFlagFeatRisk, 2.0f, 4.0f); + const std::string path = tmp_cvcnav("v2bytes"); + net.save(path); + std::ifstream f(path, std::ios::binary); + std::vector b((std::istreambuf_iterator(f)), + std::istreambuf_iterator()); + f.close(); + const std::uint8_t *p = b.data(); + std::size_t off = 0; + auto u32 = [&](std::size_t o) { + std::uint32_t v; + std::memcpy(&v, p + o, 4); + return v; + }; + auto f32 = [&](std::size_t o) { + float v; + std::memcpy(&v, p + o, 4); + return v; + }; + ASSERT_GE(b.size(), 32u); + EXPECT_EQ(std::memcmp(p, "CVNV", 4), 0); + EXPECT_EQ(u32(4), 2u); // format_version + const std::uint32_t flags = u32(8); + EXPECT_TRUE(flags & coef_mlp::kFlagLamSigmoid); + EXPECT_TRUE(flags & coef_mlp::kFlagSoftplusLogExpm1); + EXPECT_TRUE(flags & coef_mlp::kFlagFeatRisk); + EXPECT_EQ(u32(12), 6u); // in + EXPECT_EQ(u32(16), 5u); // out + EXPECT_EQ(u32(20), 1u); // num_layers + // header 32 + layer(12 + 5*6*4 + 5*4) = 32 + 12 + 120 + 20 = 184 -> out_bias_len + off = 184; + EXPECT_EQ(u32(off), 3u); // out_bias_len == 3 (abg only) + off += 4 + 3 * 4; // skip out_bias[3] + EXPECT_FLOAT_EQ(f32(off), 5.0f); // lam_soft_max + EXPECT_FLOAT_EQ(f32(off + 4), 10.0f); // lam_hard_max + EXPECT_EQ(u32(off + 8), 0u); // meta_len (no meta) + EXPECT_EQ(b.size(), off + 12); // exact end of file + fs::remove(path); +} + +// (9) BACK-COMPAT: a v1 (out <= 4, no sigmoid flag) net still loads and forwards all-softplus, +// with its lam column read as softplus(log(expm1(init))) exactly as before the v2 bump. +TEST(NavMaterialDeploy, V1NetsStillLoadAsLegacyAllSoftplus) { + coef_mlp v1 = make_net(6, 4, coef_mlp::kFlagFeatRisk, 0.5f); + EXPECT_EQ(v1.format_version(), 1u); + EXPECT_FALSE(v1.lam_sigmoid()); + const std::string path = tmp_cvcnav("v1compat"); + v1.save(path); + coef_mlp back = coef_mlp::load(path); + EXPECT_EQ(back.format_version(), 1u); + EXPECT_FALSE(back.lam_sigmoid()); + EXPECT_TRUE(back.has_lam()); + EXPECT_FALSE(back.has_lam_hard()); + float feat[6] = {0, 0, 0, 0, 0, 0}, o[4]; + back.forward(feat, 1, o, 1); + EXPECT_NEAR(o[0], 1.0f, 1e-4f); // abg basin (softplus fold) + EXPECT_NEAR(o[1], 3.0f, 1e-4f); + EXPECT_NEAR(o[2], 4.0f, 1e-4f); + EXPECT_NEAR(o[3], 0.5f, 1e-4f); // lam_soft is the v1 softplus fold, NOT a sigmoid + fs::remove(path); +} + +// (10) The learned lam_hard head is CONSUMED by the drive: with the fixed material lam columns +// zeroed, a two-head net whose lam_hard output is large reroutes harder away from a near hard +// hazard (phi_m small, grad_phi = +x) than one whose lam_hard is small. +TEST(NavMaterialDeploy, LearnedLamHardHeadDrivesHardReroute) { + deploy_world w; + const int hw = w.H * w.W; + std::vector store(6 * hw, 0.0f); + for (int r = 0; r < w.H; ++r) + for (int c = 0; c < w.W; ++c) { + store[0 * hw + r * w.W + c] = 0.0f; // risk ~0 (isolate the hard channel) + store[1 * hw + r * w.W + c] = 2.0f; // phi_m small -> inside d_hat_m, hard barrier active + store[4 * hw + r * w.W + c] = 1.0f; // grad_phi_x = +1 -> F_hard pushes +x (away from hazard) + } + material_stack ms; + ms.data = store.data(); + ms.M = 1; + ms.H = w.H; + ms.W = w.W; + ms.mnx = -10; + ms.mny = -10; + ms.mxx = 10; + ms.mxy = 10; + ms.cx = 0; + ms.cy = 0; + ms.S = 0.1; + std::vector ls(w.N, 0.0f), lh(w.N, 0.0f); // fixed columns 0: all reroute is the net's + material_drive md; + md.stack = &ms; + md.lam_soft = ls.data(); + md.lam_hard = lh.data(); + auto run = [&](float hard) { + coef_mlp net = make_sigmoid_net(6, 5, coef_mlp::kFlagFeatRisk, 0.01f, hard); + std::vector o = w.o, th = w.th, sp = w.sp, mc(w.N); + for (int step = 0; step < 8; ++step) + drive_step_material(w.fs, o.data(), th.data(), sp.data(), w.carrot.data(), net, w.N, nullptr, + w.v, md, mc.data(), 1); + double sx = 0; + for (int i = 0; i < w.N; ++i) + sx += o[2 * i]; + return sx / w.N; + }; + const double x_lo = run(0.05f); + const double x_hi = run(9.0f); + EXPECT_GT(x_hi, x_lo + 1e-4) + << "the learned lam_hard head did not increase hard-hazard reroute (+x) (lo=" << x_lo + << " hi=" << x_hi << ")"; +}