Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
139 changes: 123 additions & 16 deletions grl_snam/sdf_nav.py
Original file line number Diff line number Diff line change
Expand Up @@ -864,6 +864,12 @@ def ackermann_delta_max(L: float, delta_max: float, track_width: float) -> float
return math.atan(L / (L / math.tan(delta_max) + 0.5 * float(track_width)))


def _logit(x: float) -> float:
"""Inverse of sigmoid: the last-layer bias that makes a sigmoid lam head start at ``x*max``.
Caller guarantees 0 < x < 1 (see CoefMLP._check_sigmoid_init)."""
return math.log(x / (1.0 - x))


class CoefMLP(nn.Module):
"""Predict ``(alpha, beta, gamma)`` from local SDF features, biased toward the
known-good navigating regime (``bias``) so the self-supervised optimizer starts
Expand All @@ -878,6 +884,11 @@ def __init__(
use_risk=False,
use_lam=False,
lam_init=0.4,
use_lam_hard=False,
lam_hard_init=1.0,
lam_sigmoid=False,
lam_soft_max=5.0,
lam_hard_max=10.0,
):
super().__init__()
#: 5 = the original [phi, goal_dist, gdir_x, gdir_y, align]; then the OPTIONAL
Expand All @@ -890,11 +901,20 @@ def __init__(
self.use_mu = bool(use_mu)
self.use_risk = bool(use_risk)
#: ``use_lam`` adds a 4th OUTPUT, the learned material-reroute strength lam_soft
#: (see :meth:`coeffs_and_lam`). The output stack stays a single uniform
#: ``softplus(net + log(expm1(out_bias)))`` so the exporter and the C++ forward
#: need no special case — only the drive reads output[3] as lam_soft.
#: (see :meth:`coeffs_and_lam`). ``use_lam_hard`` adds a 5th, lam_hard — the paper's
#: two-head reroute (App A.4). A two-head net is always ``lam_sigmoid`` (the paper form
#: lam = lam_max*sigmoid(head), material_nav.py:175-176), which the exporter writes as
#: .cvcnav format v2. A lone lam_soft (``use_lam`` only, no sigmoid) keeps the historical
#: single uniform ``softplus(net + log(expm1(out_bias)))`` (v1) so old blobs are unchanged.
self.use_lam = bool(use_lam)
self.out_dim = 3 + (1 if self.use_lam else 0)
self.use_lam_hard = bool(use_lam_hard)
if self.use_lam_hard and not self.use_lam:
raise ValueError("use_lam_hard requires use_lam (lam_hard is the 5th output, col 4)")
#: A two-head net is the paper's sigmoid form; a single lam_soft can opt in explicitly.
self.lam_sigmoid = bool(lam_sigmoid) or self.use_lam_hard
self.lam_soft_max = float(lam_soft_max)
self.lam_hard_max = float(lam_hard_max)
self.out_dim = 3 + (1 if self.use_lam else 0) + (1 if self.use_lam_hard else 0)
self.net = nn.Sequential(
nn.Linear(self.in_dim, hidden),
nn.SiLU(),
Expand All @@ -903,40 +923,77 @@ def __init__(
nn.Linear(hidden, self.out_dim),
)
self.register_buffer("bias", torch.tensor(bias)) # abg raw bias (unchanged)
if self.use_lam:
# lam_soft init: softplus(0 + log(expm1(lam_init))) == lam_init at the zero column.
# Requires lam_init > 0: log(expm1(0)) is -inf (a DEAD lam head — 0 with zero
# gradient forever), and log(expm1(<0)) is NaN (poisons the drive/loss/.cvcnav).
if self.use_lam and not self.lam_sigmoid:
# v1 single-softplus lam_soft: softplus(0 + log(expm1(lam_init))) == lam_init at the
# zero column. Requires lam_init > 0: log(expm1(0)) is -inf (a DEAD lam head — 0 with
# zero gradient), and log(expm1(<0)) is NaN (poisons the drive/loss/.cvcnav).
if not (float(lam_init) > 0.0):
raise ValueError(
f"use_lam requires lam_init > 0 (softplus init is -inf at 0, NaN below); "
f"got {lam_init!r}"
)
self.register_buffer("lam_raw_bias", torch.tensor(float(lam_init)))
if self.lam_sigmoid:
# Sigmoid lam heads: lam = lam_max*sigmoid(head), so the init bias lives in the LAST
# Linear layer's bias (logit(init/max)), NOT the softplus-fold buffer. Requires
# 0 < init < max (logit is +/-inf otherwise). A freshly-built net thus starts at the
# requested init (add_lam_heads re-asserts these when lifting a trained net).
self._check_sigmoid_init(float(lam_init), self.lam_soft_max, "lam_init/--lam-soft")
with torch.no_grad():
self.net[-1].bias[3] = _logit(float(lam_init) / self.lam_soft_max)
if self.use_lam_hard:
self._check_sigmoid_init(
float(lam_hard_init), self.lam_hard_max, "lam_hard_init/--lam-hard-init"
)
self.net[-1].bias[4] = _logit(float(lam_hard_init) / self.lam_hard_max)

@staticmethod
def _check_sigmoid_init(init, mx, name):
if not (0.0 < init < mx):
raise ValueError(
f"a sigmoid lam head needs 0 < {name} < its max ({mx}); got {init!r} "
f"(logit(init/max) is +/-inf outside that range)"
)

def full_out_bias(self) -> torch.Tensor:
"""The (out_dim,) RAW pre-softplus bias vector — abg bias, plus lam_init for a lam
head. The exporter writes this and the C++ folds log(expm1) at load, so the whole
output stack is one uniform softplus."""
"""The RAW pre-softplus bias the exporter writes as ``out_bias``. For a sigmoid net
(v2) it is the 3 abg biases only — the lam heads' init is carried in the last Linear
layer's bias, not here. For a v1 single-softplus lam net it is abg + lam_init, so the
whole output stack is one uniform softplus the C++ folds log(expm1) at load."""
if self.lam_sigmoid:
return self.bias
if self.use_lam:
return torch.cat([self.bias, self.lam_raw_bias.reshape(1)])
return self.bias

def _outputs(self, feat):
raw = self.net(feat) + torch.log(torch.expm1(self.full_out_bias())).unsqueeze(0)
return F.softplus(raw) # [N, out_dim]
net_out = self.net(feat) # [N, out_dim]
if not self.lam_sigmoid:
raw = net_out + torch.log(torch.expm1(self.full_out_bias())).unsqueeze(0)
return F.softplus(raw)
# Two-head sigmoid (v2): abg keep the softplus(log(expm1)) fold; the lam columns are
# lam_max*sigmoid(head) (the head bias is already in net[-1].bias). Byte-parity with the
# C++ coef_mlp forward (kFlagLamSigmoid) and material_nav.py:175-176.
abg = F.softplus(net_out[:, :3] + torch.log(torch.expm1(self.bias)).unsqueeze(0))
cols = [abg, self.lam_soft_max * torch.sigmoid(net_out[:, 3:4])]
if self.use_lam_hard:
cols.append(self.lam_hard_max * torch.sigmoid(net_out[:, 4:5]))
return torch.cat(cols, dim=1)

def forward(self, feat):
c = self._outputs(feat)
return c[:, 0], c[:, 1], c[:, 2]

def coeffs_and_lam(self, feat):
"""``(alpha, beta, gamma, lam_soft)`` — lam_soft is the LEARNED per-agent material
reroute strength (F_soft = -lam_soft * grad risk), the deployable replacement for the
fixed ``lam_soft`` dial. Requires ``use_lam``."""
"""``(alpha, beta, gamma, lam_soft)`` — or ``(..., lam_soft, lam_hard)`` for a two-head
net (``use_lam_hard``). lam_soft/lam_hard are the LEARNED per-agent reroute strengths
(F_soft = -lam_soft*grad risk, F_hard = -lam_hard*db*grad phi), the deployable
replacement for the fixed ``--lam-soft`` / ``--lam-hard`` dials. Requires ``use_lam``."""
if not self.use_lam:
raise RuntimeError("this CoefMLP has no lam head (use_lam=False)")
c = self._outputs(feat)
if self.use_lam_hard:
return c[:, 0], c[:, 1], c[:, 2], c[:, 3], c[:, 4]
return c[:, 0], c[:, 1], c[:, 2], c[:, 3]


Expand Down Expand Up @@ -1111,3 +1168,53 @@ def add_lam_head(model: CoefMLP, lam_init: float = 0.4) -> CoefMLP:
out.net[4].bias[:3].copy_(model.net[4].bias)
out.bias.copy_(model.bias)
return out


def add_lam_heads(
model: CoefMLP,
soft_init: float = 0.4,
hard_init: float = 1.0,
lam_soft_max: float = 5.0,
lam_hard_max: float = 10.0,
) -> CoefMLP:
"""Lift a trained ``CoefMLP`` into the paper's TWO-head sigmoid-bounded reroute net (the
deployable .cvcnav format v2): appends BOTH lam_soft (col 3) and lam_hard (col 4) as
``lam_max*sigmoid(head)`` outputs (material_nav.py:175-176). Mirrors :func:`add_lam_head`
but basin-preserving for TWO heads: the appended final-layer weight rows are zeroed and
their biases set to ``logit(init/max)``, so each lam starts at its constant init; the
(alpha,beta,gamma) rows are copied verbatim so abg is bit-for-bit unchanged at init. Unlike
the v1 single-softplus :func:`add_lam_head`, the lam columns are sigmoid-bounded, so this net
exports as format v2 (kFlagLamSigmoid) and hard-fails to load on a pre-v2 libcvc host.
Returns a new model."""
if getattr(model, "use_lam", False):
raise ValueError("add_lam_heads: model already has a lam head")
first = model.net[0]
hidden = first.out_features
out = CoefMLP(
hidden=hidden,
bias=tuple(model.bias.tolist()),
in_dim=model.in_dim,
use_mu=getattr(model, "use_mu", False),
use_risk=getattr(model, "use_risk", False),
use_lam=True,
lam_init=soft_init,
use_lam_hard=True,
lam_hard_init=hard_init,
lam_soft_max=lam_soft_max,
lam_hard_max=lam_hard_max,
)
with torch.no_grad():
out.net[0].weight.copy_(first.weight)
out.net[0].bias.copy_(first.bias)
out.net[2].weight.copy_(model.net[2].weight)
out.net[2].bias.copy_(model.net[2].bias)
# final layer: rows 0..2 = the abg head (copied verbatim), rows 3/4 = the lam heads with
# ZERO weights so they are position-independent at init, biases = logit(init/max) so they
# start at soft_init / hard_init (a sigmoid net's init lives in the last-layer bias).
out.net[4].weight.zero_()
out.net[4].weight[:3].copy_(model.net[4].weight)
out.net[4].bias[:3].copy_(model.net[4].bias)
out.net[4].bias[3] = _logit(float(soft_init) / lam_soft_max)
out.net[4].bias[4] = _logit(float(hard_init) / lam_hard_max)
out.bias.copy_(model.bias)
return out
8 changes: 7 additions & 1 deletion grl_snam/swarm.py
Original file line number Diff line number Diff line change
Expand Up @@ -540,7 +540,13 @@ def step(self) -> None:
else:
feat = self._coef_feats(phi, nrm, carrot)
mkw = self._material_kw()
if getattr(self.model, "use_lam", False):
if getattr(self.model, "use_lam_hard", False):
# TWO-head net: the net outputs BOTH lam_soft and lam_hard per agent; drive with
# both (as trained) instead of the fixed MaterialParams reroute strengths.
al, be, ga, lam_soft, lam_hard = self.model.coeffs_and_lam(feat)
if mkw:
mkw = dict(mkw, lam_soft=lam_soft, lam_hard=lam_hard)
elif getattr(self.model, "use_lam", False):
# Deployable learned reroute: the net outputs lam_soft per agent; drive with
# it (as trained) instead of the fixed MaterialParams lam_soft.
al, be, ga, lam_soft = self.model.coeffs_and_lam(feat)
Expand Down
28 changes: 23 additions & 5 deletions grl_snam/tools/coef_export.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,13 @@
u32 format_version, u32 flags, u32 in, u32 out, u32 num_layers, u64 arch_hash
per layer: u32 rows, u32 cols, u32 act(0=identity,1=SiLU), f32 w[rows*cols], f32 b[rows]
u32 out_bias_len, f32 out_bias[...] (raw bias; log(expm1) folded at load)
[v2 only, FLAG_LAM_SIGMOID] f32 lam_soft_max, f32 lam_hard_max (sigmoid ceilings)
u32 meta_len, char meta[meta_len] (optional provenance)

A two-head sigmoid net (CoefMLP.lam_sigmoid, i.e. use_lam_hard) writes format v2: out_bias_len
is 3 (the abg columns only; the lam heads' init lives in the last Linear bias), the two sigmoid
ceilings follow the out_bias block, and FLAG_LAM_SIGMOID is set so the C++ loader reads the lam
columns as lam_max*sigmoid(raw). A plain / single-softplus-lam net stays v1, byte-unchanged.
"""

from __future__ import annotations
Expand All @@ -25,12 +31,16 @@
import numpy as np

FORMAT_VERSION = 1
FORMAT_VERSION_V2 = 2 # two-head sigmoid reroute (lam_soft, lam_hard) — see FLAG_LAM_SIGMOID
FLAG_SOFTPLUS_LOG_EXPM1 = 1 << 0
# Input-feature-layout flags (must match cvc::nav::coef_mlp kFlagFeatMu/kFlagFeatRisk): a
# 6-input net is ambiguous (grip mu vs terrain-risk), so the C++ drive reads the layout from
# these flags rather than in_features(). Set from CoefMLP.use_mu / use_risk.
FLAG_FEAT_MU = 1 << 1
FLAG_FEAT_RISK = 1 << 2
# The lam OUTPUT columns (>= 3) are lam_max*sigmoid(raw), not softplus (matches
# cvc::nav::coef_mlp kFlagLamSigmoid). Set from CoefMLP.lam_sigmoid (use_lam_hard).
FLAG_LAM_SIGMOID = 1 << 3
_ACT_IDENTITY, _ACT_SILU = 0, 1
_U64 = (1 << 64) - 1

Expand Down Expand Up @@ -75,9 +85,11 @@ def write_coef_mlp(model, path, meta: bytes = b""):
layers = _layers(model)
in_f = int(layers[0][0].shape[1])
out_f = int(layers[-1][0].shape[0])
# The RAW pre-softplus bias, one entry per output: (1,3,4) for abg, plus lam_init for a
# lam head. full_out_bias() keeps the whole output stack a single uniform softplus, so
# out_bias_len == out_f and the C++ forward needs no per-output special case.
# A two-head sigmoid net is format v2: the lam columns are lam_max*sigmoid(raw), not softplus.
sigmoid_lam = bool(getattr(model, "lam_sigmoid", False))
# The RAW pre-softplus bias. For v1 this is one entry per output (abg + lam_init) so the whole
# stack is a single uniform softplus. For v2 full_out_bias() returns the 3 abg biases only —
# the lam heads' init lives in the last Linear bias, and the C++ folds log(expm1) for abg only.
_ob = model.full_out_bias() if hasattr(model, "full_out_bias") else model.bias
out_bias = _ob.detach().cpu().numpy().astype(np.float32)
shape_act = []
Expand All @@ -86,23 +98,29 @@ def write_coef_mlp(model, path, meta: bytes = b""):
ah = arch_hash(in_f, out_f, shape_act)
# Encode the input-feature layout so the C++ drive builds the right columns for a widened
# net (a bare 6-in net is grip; the risk flag disambiguates it). out_f already signals the
# lam head (out>=4), so it needs no flag.
# lam head (out>=4), so it needs no flag; FLAG_LAM_SIGMOID marks the sigmoid two-head form.
flags = FLAG_SOFTPLUS_LOG_EXPM1
if getattr(model, "use_mu", False):
flags |= FLAG_FEAT_MU
if getattr(model, "use_risk", False):
flags |= FLAG_FEAT_RISK
if sigmoid_lam:
flags |= FLAG_LAM_SIGMOID
version = FORMAT_VERSION_V2 if sigmoid_lam else FORMAT_VERSION

with open(path, "wb") as f:
f.write(b"CVNV")
f.write(struct.pack("<IIIII", FORMAT_VERSION, flags, in_f, out_f, len(layers)))
f.write(struct.pack("<IIIII", version, flags, in_f, out_f, len(layers)))
f.write(struct.pack("<Q", ah))
for w, b, act in layers:
f.write(struct.pack("<III", int(w.shape[0]), int(w.shape[1]), int(act)))
f.write(np.ascontiguousarray(w, np.float32).tobytes())
f.write(np.ascontiguousarray(b, np.float32).tobytes())
f.write(struct.pack("<I", out_bias.shape[0]))
f.write(np.ascontiguousarray(out_bias, np.float32).tobytes())
if sigmoid_lam:
# v2 trailer: the sigmoid ceilings, right after the out_bias block (before meta).
f.write(struct.pack("<ff", float(model.lam_soft_max), float(model.lam_hard_max)))
f.write(struct.pack("<I", len(meta)))
f.write(meta)
return path
Expand Down
Loading
Loading