Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 31 additions & 5 deletions inc/cvc/nav/coef_mlp.h
Original file line number Diff line number Diff line change
Expand Up @@ -49,8 +49,14 @@ namespace nav {

class coef_mlp {
public:
// The `.cvcnav` format this build reads/writes.
static constexpr std::uint32_t kFormatVersion = 1;
// The `.cvcnav` format this build WRITES. v2 adds the paper's two-head sigmoid
// reroute (kFlagLamSigmoid) and the lam_soft_max/lam_hard_max trailer. The loader
// ACCEPTS {1, 2}: a v1 file still loads and forwards identically, and a v2 file
// hard-fails on a pre-v2 host rather than silently misreading the trailer (safe).
// A plain all-softplus net (no sigmoid lam) is still written as v1 so it loads on
// old hosts; only a sigmoid-lam net bumps the version.
static constexpr std::uint32_t kFormatVersion = 2;
static constexpr std::uint32_t kFormatVersionLegacy = 1; // pre-sigmoid, all-softplus
// flags bit 0: the output is softplus(net + log(expm1(out_bias))) — the
// CoefMLP bias-toward-a-known-good-basin trick; the raw out_bias is stored and
// log(expm1(.)) is folded once at load.
Expand All @@ -63,6 +69,13 @@ class coef_mlp {
// historical base-5 (or a bare 6-input grip net, treated as mu for back-compat).
static constexpr std::uint32_t kFlagFeatMu = 1u << 1;
static constexpr std::uint32_t kFlagFeatRisk = 1u << 2;
// flags bit 3: the lam OUTPUT columns (index >= 3: col 3 = lam_soft, col 4 = lam_hard)
// are `lam_max * sigmoid(raw)` — the paper's two-head sigmoid-bounded reroute
// (material_nav.py:175-176) — NOT the softplus(log(expm1)) fold used for cols 0..2
// (alpha, beta, gamma). When set: the abg columns keep the softplus fold, the lam
// columns carry NO out_bias offset (their bias lives in the last Linear layer), and
// the file stores lam_soft_max / lam_hard_max as a v2 trailer after the out_bias block.
static constexpr std::uint32_t kFlagLamSigmoid = 1u << 3;
// Max layer dimension the CPU forward() supports (its stack activation arrays).
// load()/from_layers reject a wider net at the boundary rather than overflowing.
// The CUDA drive (drive.cu d_mlp) has a tighter cap (64) it guards separately.
Expand Down Expand Up @@ -96,12 +109,16 @@ class coef_mlp {
// loaded .cvcnav. Throws if the shapes do not chain in -> ... -> out.
// ``extra_flags`` OR-in the feature/output flags (kFlagFeatMu / kFlagFeatRisk) beyond the
// always-on softplus fold — so a C++ trainer (or a test) can stamp a widened net's layout.
// When ``extra_flags`` includes kFlagLamSigmoid the net is a v2 two-head sigmoid policy:
// ``out_bias_raw`` is then the THREE abg raw biases only (the lam heads' init lives in the
// last layer's bias b, exactly as torch), and ``lam_soft_max`` / ``lam_hard_max`` bound the
// sigmoid lam columns. A non-sigmoid net stays v1 (fmt 1) so it loads on a pre-v2 host.
static coef_mlp from_layers(int in, int out, const std::vector<int> &rows,
const std::vector<int> &cols, const std::vector<std::uint32_t> &act,
const std::vector<std::vector<float>> &w,
const std::vector<std::vector<float>> &b,
const std::vector<float> &out_bias_raw,
std::uint32_t extra_flags = 0);
const std::vector<float> &out_bias_raw, std::uint32_t extra_flags = 0,
float lam_soft_max = 5.0f, float lam_hard_max = 10.0f);

// Serialize to the versioned `.cvcnav` (byte-identical layout to
// grl_snam.tools.coef_export.write_coef_mlp), so a policy trained in pure C++
Expand All @@ -124,7 +141,12 @@ class coef_mlp {
// with no risk flag is grip (the historical widen_coef_mlp output), so mu is implied there.
bool has_risk() const { return (flags_ & kFlagFeatRisk) != 0; }
bool has_mu() const { return (flags_ & kFlagFeatMu) != 0 || (in_ == 6 && !has_risk()); }
bool has_lam() const { return out_ >= 4; } // 4th output = learned reroute lam_soft
bool has_lam() const { return out_ >= 4; } // 4th output (col 3) = learned reroute lam_soft
bool has_lam_hard() const { return out_ >= 5; } // 5th output (col 4) = learned lam_hard
// True when the lam columns are lam_max*sigmoid(raw) (paper two-head form), not softplus.
bool lam_sigmoid() const { return (flags_ & kFlagLamSigmoid) != 0; }
float lam_soft_max() const { return lam_soft_max_; }
float lam_hard_max() const { return lam_hard_max_; }
std::uint64_t arch_hash() const { return arch_hash_; }

// The architecture hash (FNV-1a over in/out/num_layers and each layer's
Expand Down Expand Up @@ -152,8 +174,12 @@ class coef_mlp {
};
std::vector<Layer> layers_;
std::vector<float> out_bias_off_; // = log(expm1(out_bias)), folded once at load
// (0 for sigmoid lam columns — bias is in the last layer)
int in_ = 0, out_ = 0;
std::uint32_t fmt_ = 0, flags_ = 0;
// Sigmoid lam ceilings (kFlagLamSigmoid): lam_soft = lam_soft_max_*sigmoid(raw),
// lam_hard = lam_hard_max_*sigmoid(raw). Defaults match coef_energy_net (5.0 / 10.0).
float lam_soft_max_ = 5.0f, lam_hard_max_ = 10.0f;
std::uint64_t arch_hash_ = 0;

void build_from_bytes(const std::uint8_t *p, std::size_t n);
Expand Down
5 changes: 3 additions & 2 deletions inc/cvc/nav/drive.h
Original file line number Diff line number Diff line change
Expand Up @@ -240,8 +240,9 @@ void bicycle_rollout(const field_stack &f, float *o, float *th, float *sp, const
// (nav_stats.h); keep the two PODs semantically in lockstep.
struct drive_telemetry {
// CoefMLP policy output this tick (the al/be/ga fed to the rollout; lam_soft is the
// effective material soft-weight, 0 on the non-material path).
float alpha = 0, beta = 0, gamma = 0, lam_soft = 0;
// effective material soft-weight, 0 on the non-material path; lam_hard is the ungated
// hard-hazard weight — learned per-agent for a two-head sigmoid net, else the fixed dial).
float alpha = 0, beta = 0, gamma = 0, lam_soft = 0, lam_hard = 0;
float mu = 1; // underfoot grip at the agent cell (1 = dry / no grip field)
float mrisk = 0; // material risk at the agent cell (0 = no material stack)
float ext_fx = 0, ext_fy = 0; // applied external force in the drive frame (0 = no ext channel)
Expand Down
112 changes: 90 additions & 22 deletions src/cvc/nav/coef_mlp.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -34,7 +34,13 @@
// u64 arch_hash
// per layer: u32 rows, u32 cols, u32 act, f32 w[rows*cols], f32 b[rows]
// u32 out_bias_len ; f32 out_bias[out_bias_len] (raw bias; log(expm1) at load)
// [v2 only, kFlagLamSigmoid] f32 lam_soft_max ; f32 lam_hard_max (sigmoid ceilings)
// u32 meta_len ; char meta[meta_len] (provenance, ignored here)
//
// v2 (two-head sigmoid, kFlagLamSigmoid): out_bias_len == 3 (the abg raw biases only);
// the lam columns (>= 3) are lam_max*sigmoid(raw) with the raw head bias in the last
// Linear layer, so they carry no out_bias offset, and the two maxes follow the out_bias
// block. v1 stays out_bias_len == out_features, all columns softplus.

#include <cmath>
#include <cstdlib>
Expand Down Expand Up @@ -65,6 +71,11 @@ inline float softplus(float x) {
return x > 20.0f ? x : std::log1p(std::exp(x));
}

inline float sigmoidf(float x) {
// torch sigmoid; matches the lam_max*sigmoid two-head form in material_nav.py.
return 1.0f / (1.0f + std::exp(-x));
}

// Read a trivially-copyable value from a little-endian byte cursor, advancing it.
template <class T> T take(const std::uint8_t *&p, const std::uint8_t *end) {
if (p + sizeof(T) > end)
Expand Down Expand Up @@ -114,7 +125,10 @@ void coef_mlp::build_from_bytes(const std::uint8_t *p, std::size_t n) {
if (std::memcmp(magic, "CVNV", 4) != 0)
throw std::runtime_error("cvc::nav::coef_mlp: bad magic (not a .cvcnav file)");
fmt_ = take<std::uint32_t>(p, end);
if (fmt_ != kFormatVersion)
// Accept {1, 2}: a v1 file still loads and forwards identically; a v2 file (two-head
// sigmoid) is read below via kFlagLamSigmoid. A newer version hard-fails here rather
// than silently misreading a trailer this build does not know.
if (fmt_ != kFormatVersionLegacy && fmt_ != kFormatVersion)
throw std::runtime_error("cvc::nav::coef_mlp: unsupported .cvcnav format version");
flags_ = take<std::uint32_t>(p, end);
in_ = static_cast<int>(take<std::uint32_t>(p, end));
Expand Down Expand Up @@ -150,16 +164,35 @@ void coef_mlp::build_from_bytes(const std::uint8_t *p, std::size_t n) {
const std::uint32_t obn = take<std::uint32_t>(p, end);
std::vector<float> out_bias;
take_floats(p, end, out_bias, obn);
out_bias_off_.resize(out_bias.size());
if (flags_ & kFlagSoftplusLogExpm1) {
// Fold log(expm1(bias)) once, in float32, matching torch.log(torch.expm1(.)).
for (std::size_t i = 0; i < out_bias.size(); ++i)
out_bias_off_[i] = std::log(std::expm1(out_bias[i]));
const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0;
if (sigmoid_lam) {
// v2 two-head sigmoid: out_bias covers ONLY the 3 abg columns (softplus fold); the lam
// columns (>= 3) get a zero offset (their bias is in the last layer) and are read as
// lam_max*sigmoid(raw). The two ceilings follow the out_bias block.
if (fmt_ != kFormatVersion)
throw std::runtime_error("cvc::nav::coef_mlp: kFlagLamSigmoid requires .cvcnav format v2");
if (out_ < 4)
throw std::runtime_error("cvc::nav::coef_mlp: sigmoid-lam net needs out_features >= 4");
if (obn != 3)
throw std::runtime_error("cvc::nav::coef_mlp: v2 sigmoid out_bias must be the 3 abg columns");
out_bias_off_.assign(static_cast<std::size_t>(out_), 0.0f);
for (int i = 0; i < 3; ++i)
out_bias_off_[i] =
(flags_ & kFlagSoftplusLogExpm1) ? std::log(std::expm1(out_bias[i])) : out_bias[i];
lam_soft_max_ = take<float>(p, end);
lam_hard_max_ = take<float>(p, end);
} else {
out_bias_off_ = out_bias; // used as a plain additive offset
out_bias_off_.resize(out_bias.size());
if (flags_ & kFlagSoftplusLogExpm1) {
// Fold log(expm1(bias)) once, in float32, matching torch.log(torch.expm1(.)).
for (std::size_t i = 0; i < out_bias.size(); ++i)
out_bias_off_[i] = std::log(std::expm1(out_bias[i]));
} else {
out_bias_off_ = out_bias; // used as a plain additive offset
}
if (static_cast<int>(out_bias_off_.size()) != out_)
throw std::runtime_error("cvc::nav::coef_mlp: out_bias length != out_features");
}
if (static_cast<int>(out_bias_off_.size()) != out_)
throw std::runtime_error("cvc::nav::coef_mlp: out_bias length != out_features");
// A meta trailer may follow; it is optional and ignored here.
}

Expand All @@ -171,7 +204,7 @@ coef_mlp coef_mlp::load_from_memory(const void *data, std::size_t nbytes) {

coef_mlp coef_mlp::default_biased() {
coef_mlp m;
m.fmt_ = kFormatVersion;
m.fmt_ = kFormatVersionLegacy; // a plain all-softplus abg net stays v1 (loads on old hosts)
m.flags_ = kFlagSoftplusLogExpm1;
m.in_ = 5;
m.out_ = 3;
Expand All @@ -198,16 +231,21 @@ coef_mlp coef_mlp::from_layers(int in, int out, const std::vector<int> &rows,
const std::vector<int> &cols, const std::vector<std::uint32_t> &act,
const std::vector<std::vector<float>> &w,
const std::vector<std::vector<float>> &b,
const std::vector<float> &out_bias_raw, std::uint32_t extra_flags) {
const std::vector<float> &out_bias_raw, std::uint32_t extra_flags,
float lam_soft_max, float lam_hard_max) {
const int num_layers = static_cast<int>(rows.size());
if (static_cast<int>(cols.size()) != num_layers || static_cast<int>(act.size()) != num_layers ||
static_cast<int>(w.size()) != num_layers || static_cast<int>(b.size()) != num_layers)
throw std::runtime_error("cvc::nav::coef_mlp::from_layers: ragged layer arrays");
const bool sigmoid_lam = (extra_flags & kFlagLamSigmoid) != 0;
coef_mlp m;
m.fmt_ = kFormatVersion;
// A sigmoid-lam net is the v2 format; a plain net stays v1 so it loads on a pre-v2 host.
m.fmt_ = sigmoid_lam ? kFormatVersion : kFormatVersionLegacy;
m.flags_ = kFlagSoftplusLogExpm1 | extra_flags;
m.in_ = in;
m.out_ = out;
m.lam_soft_max_ = lam_soft_max;
m.lam_hard_max_ = lam_hard_max;
m.layers_.resize(num_layers);
std::vector<std::uint32_t> sa;
int prev = in;
Expand All @@ -233,11 +271,24 @@ coef_mlp coef_mlp::from_layers(int in, int out, const std::vector<int> &rows,
}
if (prev != out)
throw std::runtime_error("cvc::nav::coef_mlp::from_layers: last layer does not produce out");
if (static_cast<int>(out_bias_raw.size()) != out)
throw std::runtime_error("cvc::nav::coef_mlp::from_layers: out_bias length != out");
m.out_bias_off_.resize(out);
for (int i = 0; i < out; ++i)
m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i]));
if (sigmoid_lam) {
// Two-head sigmoid: out_bias_raw is the 3 abg biases; the lam heads' init lives in the
// last layer's bias b (exactly as torch), so the lam columns carry a zero offset.
if (out < 4)
throw std::runtime_error("cvc::nav::coef_mlp::from_layers: sigmoid-lam net needs out >= 4");
if (static_cast<int>(out_bias_raw.size()) != 3)
throw std::runtime_error(
"cvc::nav::coef_mlp::from_layers: sigmoid net out_bias must be the 3 abg biases");
m.out_bias_off_.assign(static_cast<std::size_t>(out), 0.0f);
for (int i = 0; i < 3; ++i)
m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i]));
} else {
if (static_cast<int>(out_bias_raw.size()) != out)
throw std::runtime_error("cvc::nav::coef_mlp::from_layers: out_bias length != out");
m.out_bias_off_.resize(out);
for (int i = 0; i < out; ++i)
m.out_bias_off_[i] = std::log(std::expm1(out_bias_raw[i]));
}
m.arch_hash_ = compute_arch_hash(in, out, sa);
return m;
}
Expand Down Expand Up @@ -267,12 +318,20 @@ void coef_mlp::save(const std::string &path, const std::string &meta) const {
wf(L.b);
}
// The file stores the RAW out_bias; recover it from the folded offset
// (log(expm1(raw)) -> raw = softplus(off)) so save/load round-trips.
std::vector<float> raw(out_bias_off_.size());
for (std::size_t i = 0; i < raw.size(); ++i)
// (log(expm1(raw)) -> raw = softplus(off)) so save/load round-trips. In v2 sigmoid mode
// only the 3 abg columns carry a bias (the lam heads' bias is in the last layer), so the
// block is length 3 and the two ceilings follow it — byte-identical to the Python exporter.
const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0;
const std::size_t nbias = sigmoid_lam ? 3u : out_bias_off_.size();
std::vector<float> raw(nbias);
for (std::size_t i = 0; i < nbias; ++i)
raw[i] = (flags_ & kFlagSoftplusLogExpm1) ? softplus(out_bias_off_[i]) : out_bias_off_[i];
w32(static_cast<std::uint32_t>(raw.size()));
wf(raw);
if (sigmoid_lam) {
f.write(reinterpret_cast<const char *>(&lam_soft_max_), 4);
f.write(reinterpret_cast<const char *>(&lam_hard_max_), 4);
}
w32(static_cast<std::uint32_t>(meta.size()));
if (!meta.empty())
f.write(meta.data(), static_cast<std::streamsize>(meta.size()));
Expand Down Expand Up @@ -387,8 +446,17 @@ void coef_mlp::forward(const float *feats, int n, float *out, int num_threads,
a[o] = b[o];
}
float *o = out + static_cast<std::size_t>(s) * out_;
for (int k = 0; k < out_; ++k)
o[k] = softplus(a[k] + out_bias_off_[k]);
const bool sigmoid_lam = (flags_ & kFlagLamSigmoid) != 0;
for (int k = 0; k < out_; ++k) {
const float pre = a[k] + out_bias_off_[k];
if (sigmoid_lam && k >= 3) {
// Two-head sigmoid reroute: col 3 = lam_soft, col 4 = lam_hard, each lam_max*sigmoid.
const float mx = (k == 3) ? lam_soft_max_ : lam_hard_max_;
o[k] = mx * sigmoidf(pre);
} else {
o[k] = softplus(pre);
}
}
});
}

Expand Down
27 changes: 23 additions & 4 deletions src/cvc/nav/drive.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -331,7 +331,8 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl
// Telemetry temporaries — write-only, never read by the drive math, so a null `tel` path is
// byte-identical. They hold the LAST substep's value at loop exit (mu/mrisk/ext default to the
// no-field case; steer/curv/clear are overwritten every substep).
float tel_mu = 1.0f, tel_mrisk = 0.0f, tel_lam = 0.0f, tel_ext_x = 0.0f, tel_ext_y = 0.0f;
float tel_mu = 1.0f, tel_mrisk = 0.0f, tel_lam = 0.0f, tel_lam_hard = 0.0f, tel_ext_x = 0.0f,
tel_ext_y = 0.0f;
float tel_steer = 0.0f, tel_curv = 0.0f;

for (int s = 0; s < v.nsub; ++s) {
Expand Down Expand Up @@ -414,6 +415,7 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl
const float db = -(1.0f / (1.0f + std::exp(-(mat->k_sharp * (mat->d_hat_m - mphi)))));
const float ls = mat->lam_soft[i], lh = mat->lam_hard[i];
tel_lam = ls;
tel_lam_hard = lh;
const float fsx = -ls * mgrx;
const float fsy = -ls * mgry;
const float fhx = (-lh * db) * mgpx;
Expand Down Expand Up @@ -554,6 +556,7 @@ void rollout_impl(const field_stack &f, float *o, float *th, float *sp, const fl
t.beta = bei;
t.gamma = gai;
t.lam_soft = tel_lam;
t.lam_hard = tel_lam_hard;
t.mu = tel_mu;
t.mrisk = tel_mrisk;
t.ext_fx = tel_ext_x;
Expand Down Expand Up @@ -753,15 +756,23 @@ void drive_step_material(const field_stack &f, float *o, float *th, float *sp, c
ga[i] = coef[static_cast<std::size_t>(out_w) * i + 2];
}
// Learned reroute: a lam-head net's 4th output is the per-agent lam_soft, overriding the
// material_drive's fixed/gated column (the deployable twin of grl_snam coeffs_and_lam).
// material_drive's fixed/gated column (the deployable twin of grl_snam coeffs_and_lam). A
// two-head sigmoid net (has_lam_hard) also learns lam_hard from its 5th output; lam_hard is
// never gated (material.h:37,161), so it mirrors lam_soft's extraction with no witness gate.
material_drive md = mat;
std::vector<float> lam_learned;
std::vector<float> lam_learned, lam_hard_learned;
if (model.has_lam()) {
lam_learned.resize(n);
for (int i = 0; i < n; ++i)
lam_learned[i] = coef[static_cast<std::size_t>(out_w) * i + 3];
md.lam_soft = lam_learned.data();
}
if (model.has_lam_hard()) {
lam_hard_learned.resize(n);
for (int i = 0; i < n; ++i)
lam_hard_learned[i] = coef[static_cast<std::size_t>(out_w) * i + 4];
md.lam_hard = lam_hard_learned.data();
}
rollout_impl(f, o, th, sp, carrot, al.data(), be.data(), ga.data(), n, map_id, v,
md.stack ? &md : nullptr, nullptr, minclr_out, num_threads, nullptr, tel);
}
Expand Down Expand Up @@ -845,14 +856,22 @@ void drive_step_material_ext(const field_stack &f, float *o, float *th, float *s
be[i] = coef[static_cast<std::size_t>(out_w) * i + 1];
ga[i] = coef[static_cast<std::size_t>(out_w) * i + 2];
}
// Same learned lam_soft (+ optional two-head lam_hard) extraction as drive_step_material; see
// there for the gate note. lam_hard is never gated.
material_drive md = mat;
std::vector<float> lam_learned;
std::vector<float> lam_learned, lam_hard_learned;
if (model.has_lam()) {
lam_learned.resize(n);
for (int i = 0; i < n; ++i)
lam_learned[i] = coef[static_cast<std::size_t>(out_w) * i + 3];
md.lam_soft = lam_learned.data();
}
if (model.has_lam_hard()) {
lam_hard_learned.resize(n);
for (int i = 0; i < n; ++i)
lam_hard_learned[i] = coef[static_cast<std::size_t>(out_w) * i + 4];
md.lam_hard = lam_hard_learned.data();
}
rollout_impl(f, o, th, sp, carrot, al.data(), be.data(), ga.data(), n, map_id, v,
md.stack ? &md : nullptr, ext.sample ? &ext : nullptr, minclr_out, num_threads,
nullptr, tel);
Expand Down
Loading
Loading