From 2f591b036243efb6e0a2dd67d1dae2ac15f53390 Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 02:30:52 -0700 Subject: [PATCH 1/6] mario: the package carries its configuration, objective, ledger and evidence renderer, and archives what it showed config.yaml (version 1) holds the instructions, gate, encoder, the tunables the rendering reads (horizons, the tall-wall height, the measured take-off windows) and the objective (pass on cleared, a failure per life lost, level_x as the locus, observations/ as the evidence). The environment reads its tunables from that file and archives every frame it shows under observations/ with a timestamp index. tools/evidence.py cuts the archive into a contact sheet around each failure. ledger.jsonl records version 1 and what the manual loop found. This is the first instance of docs/dual-loop.md in the harness repo. Plugin 0.4.0; the game logic is the released one. Co-Authored-By: Claude Fable 5.1 --- kits/mario/plugin/config.yaml | 28 +++++++ kits/mario/plugin/ledger.jsonl | 1 + kits/mario/plugin/plugin.json | 2 +- kits/mario/plugin/server/mario_env.py | 55 ++++++++++--- kits/mario/plugin/tools/evidence.py | 107 ++++++++++++++++++++++++++ 5 files changed, 182 insertions(+), 11 deletions(-) create mode 100644 kits/mario/plugin/config.yaml create mode 100644 kits/mario/plugin/ledger.jsonl create mode 100644 kits/mario/plugin/tools/evidence.py diff --git a/kits/mario/plugin/config.yaml b/kits/mario/plugin/config.yaml new file mode 100644 index 0000000..7ecef69 --- /dev/null +++ b/kits/mario/plugin/config.yaml @@ -0,0 +1,28 @@ +# The Super Mario reflex's configuration: everything that shapes its decisions except the model +# and the game. Versioned with the package; the outer loop (docs/dual-loop.md in the harness repo) +# changes it one thing at a time, with the evidence in ledger.jsonl beside it. +version: 1 +instructions: | + You play a side-scrolling platform game as Mario, deciding several times a second. Each action sets the keys for the next moment and they stay set until you change them: run_right holds right and run; jump_right holds right, run and jump, and a jump held in the air goes again the moment Mario lands, so holding it hops him along; jump holds jump alone; walk_left holds left; wait lets every key go. Keep moving right. The state ends with a line beginning "Now:" that says what the measured facts call for at this moment; follow it. + + Facts, measured on this game: a decision lands a fraction of a second after the state you read, and at a full run Mario covers 2 to 3 tiles in that time. A jump held for one decision clears 3 tiles, for two decisions 4 tiles; jump_right pressed again on the ground is a new jump, and held through a landing it is a new jump at once. A running jump lands about 9 tiles on. An enemy is jumped at a run with the take-off 1.5 to 4 tiles before it; under a row of blocks an earlier jump hits the blocks and drops Mario onto it, so when that take-off cannot be hit, stop with walk_left and jump_right when it is 1 to 2.5 tiles away, which always works. An enemy walking at a standing Mario is jumped with jump_right when it is 1 to 2.5 tiles away. An enemy on a ledge above walks off its edge and drops onto whoever runs under it: wait for it to come down, then jump_right over it. A pipe 4 tiles tall is cleared only at a full run, leaving the ground 2 to 3 tiles before it with the jump held for three decisions; from standing or at a jog it is never cleared, so if Mario is stopped at one, walk_left for four decisions, then run_right, then jump_right; but if an enemy is walking at his back, walking left meets it: stand and take jump (the standing jump) when it is 2 tiles behind him, it passes under, and only then walk_left. A step one tile tall is hopped with jump_right; a stair is a hop per step. Where the ground drops away ahead, jump_right from the edge lands past the drop, and walking off the edge lands in a slot with no way out. A gap is crossed by a running jump from its edge, and a jump still in the air with a gap coming is kept held so the next hop goes the moment Mario lands; a standing jump falls in; enemies waiting where the jump would land walk to the gap and fall in if Mario stands at the edge. A question block pays a coin when hit from below by a jump started 1 to 1.5 tiles before it at a run. + + When Mario has just died, wait. Finish when the level is cleared. Escalate when Mario has no lives left. +gate: {read: 0.5, write: 0.7, destructive: 0.9} +encoder: {history_steps: 3} +tunables: + enemy_horizon_tiles: 16 # how far ahead an enemy is named + gap_horizon_tiles: 14 # how far ahead a gap is named and the jump timed + tall_wall_tiles: 4 # a wall this tall is cleared only at a full run + pipe_takeoff_tiles: [1.8, 3.4] # measured: where the jump over a 4-tile pipe leaves the ground + enemy_takeoff_tiles: [1.5, 6.5] # measured: where the jump over an enemy at a run may leave the ground, in the open + enemy_takeoff_under_blocks_tiles: [1.5, 4.0] # measured: under the block row, farther hits the blocks and drops onto it + archive_frames: true # keep every frame the page shows under observations/ +objective: + pass: "cleared == true" + failure: "lives decreased" + metrics: + - {field: cleared, better: true} + - {field: level_x, better: higher} + locus: [level_x] + evidence: observations/ diff --git a/kits/mario/plugin/ledger.jsonl b/kits/mario/plugin/ledger.jsonl new file mode 100644 index 0000000..50dd71c --- /dev/null +++ b/kits/mario/plugin/ledger.jsonl @@ -0,0 +1 @@ +{"version": 1, "at": "2026-09-20", "change": "the released environment and instructions (starter-kit 28ce645)", "channel": "told,shown", "evidence": ["the manual loop of 2026-09-20: recordings and contact sheets of attempts 22 to 42"], "runs": ["local attempts 34, 41, 42; one hosted run on hr-test"], "metrics_before": null, "metrics_after": {"pass": 1.0, "failures": 4.3, "runs": 3}, "verdict": "kept", "note": "pipes cleared only at a full run from a measured take-off; the ground read from Mario's level; a held jump goes again on landing; the nearest thing ahead decides the advice"} diff --git a/kits/mario/plugin/plugin.json b/kits/mario/plugin/plugin.json index 1f3864c..b081aea 100644 --- a/kits/mario/plugin/plugin.json +++ b/kits/mario/plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "mario-env", - "version": "0.3.2", + "version": "0.4.0", "description": "Full Screen Mario as an environment for the System One base: the game's state as literal text each step, the keys it listens to as actions, and the browser's frame for the page to show.", "homepage": "https://github.com/HarnessRouter/starter-kit", "license": "SEE LICENSE IN ../LICENSE.md" diff --git a/kits/mario/plugin/server/mario_env.py b/kits/mario/plugin/server/mario_env.py index c00bc96..15b6d21 100644 --- a/kits/mario/plugin/server/mario_env.py +++ b/kits/mario/plugin/server/mario_env.py @@ -17,6 +17,7 @@ import asyncio import base64 +import json import glob import os import pathlib @@ -38,7 +39,25 @@ "so holding it hops him along; jump holds jump alone; walk_left holds left; wait lets every key go. The state says where Mario is, what he is doing, what is ahead and behind with distances in " "tiles, and which keys are held. Keep moving right and jump over what is within one step. Finish when " "the level is cleared. Escalate when Mario has no lives left.") -TALL = 4 # a wall this tall is cleared only at a run +TALL = 4 # a wall this tall is cleared only at a run (the default; config.yaml tunables override) +TUNABLES = {"enemy_horizon_tiles": 16, "gap_horizon_tiles": 14, "tall_wall_tiles": TALL, "pipe_takeoff_tiles": [1.8, 3.4], + "enemy_takeoff_tiles": [1.5, 6.5], "enemy_takeoff_under_blocks_tiles": [1.5, 4.0], "archive_frames": True} + + +def load_tunables() -> dict: + """The package's config.yaml tunables over the defaults: the numbers the rendering reads, which + the outer loop changes one at a time (docs/dual-loop.md in the harness repo).""" + tun = dict(TUNABLES) + for cand in (os.environ.get("SYSTEMONE_CONFIG"), str(pathlib.Path(__file__).resolve().parents[1] / "config.yaml")): + if cand and os.path.exists(cand): + try: + import yaml + d = yaml.safe_load(open(cand)) or {} + tun.update({k: v for k, v in (d.get("tunables") or {}).items() if k in tun}) + except Exception: # noqa: BLE001 - a bad file leaves the defaults + pass + break + return tun # The frame the kit's page shows: Chrome's own screencast, a JPEG for every third frame the game # draws (about twenty a second), each written whole to frame.jpg. The page reads that file. SCREENCAST = {"format": "jpeg", "quality": 45, "maxWidth": 960, "maxHeight": 600, "everyNthFrame": 3} @@ -63,7 +82,8 @@ if (!c.alive || c === player || c.title === undefined) return; if (['Coin','Mushroom','FireFlower','Star','Vine','Text','Shell','Fireball'].indexOf(c.title) >= 0) return; var dx = (c.left - p.right) / T, dy = (p.bottom - c.bottom) / T, back = (p.left - c.right) / T; - if (dx >= 0 && dx <= 16) enemies.push({kind: c.title, dx: Math.round(dx * 10) / 10, dy: Math.round(dy * 10) / 10, dir: (c.xvel || 0) < 0 ? 'toward' : 'away'}); + var EH = (window.__s1tun && window.__s1tun.enemy_horizon_tiles) || 16; + if (dx >= 0 && dx <= EH) enemies.push({kind: c.title, dx: Math.round(dx * 10) / 10, dy: Math.round(dy * 10) / 10, dir: (c.xvel || 0) < 0 ? 'toward' : 'away'}); else if (back >= 0 && back <= 8) behind.push({kind: c.title, dx: Math.round(back * 10) / 10, dy: Math.round(dy * 10) / 10, dir: (c.xvel || 0) > 0 ? 'toward' : 'away'}); }); // the ground ahead as a profile from where Mario stands: each quarter tile, the highest surface @@ -165,6 +185,9 @@ def __init__(self): self._session = None self._cdp = None self.frame_path = workspace_root() / "frame.jpg" + self.tun = load_tunables() + self.archive = (workspace_root() / "observations") if self.tun.get("archive_frames") else None + self._archived = 0 self.started_at = time.time() self.steps = 0 self.frames = 0 @@ -210,10 +233,21 @@ def _on_frame(self, ev: dict, session_id=None) -> None: (a temp file renamed over it, so a reader never sees half a picture), and the frame is acknowledged, without which Chrome stops sending.""" try: + data = base64.b64decode(ev["data"]) tmp = self.frame_path.with_suffix(".jpg.tmp") - tmp.write_bytes(base64.b64decode(ev["data"])) + tmp.write_bytes(data) os.replace(tmp, self.frame_path) self.frames += 1 + if self.archive is not None and self._archived < 12000: + # the observation archive the outer loop reads: every frame the page showed, with + # its time, under observations/ (about 7 KB a frame, eighteen a second) + if self._archived == 0: + self.archive.mkdir(parents=True, exist_ok=True) + name = f"{self._archived:06d}.jpg" + (self.archive / name).write_bytes(data) + with open(self.archive / "frames.jsonl", "a") as f: + f.write(json.dumps({"t": round(time.time(), 3), "file": name}) + "\n") + self._archived += 1 except Exception: # noqa: BLE001 - a missed frame is a missed frame, never a failed step pass self._loop.create_task(self._cdp.cdp_client.send.Page.screencastFrameAck( @@ -320,7 +354,7 @@ async def go(): if isinstance(st, dict) and st.get("ready"): break await asyncio.sleep(0.25) - await self._eval("window.unpause && unpause(); 'ok'") + await self._eval(f"window.__s1tun = {json.dumps(self.tun)}; window.unpause && unpause(); 'ok'") try: await self._screencast() except Exception: # noqa: BLE001 - already running across the navigation @@ -445,7 +479,7 @@ def _now(self, st: dict, reach: float, en: list, gaps: list, walls: list, blocks things.append((walls[0]["dx"], "wall")) if drops and drops[0]["dx"] <= 12: things.append((drops[0]["dx"], "drop")) - if gaps and gaps[0]["dx"] <= 14: + if gaps and gaps[0]["dx"] <= self.tun["gap_horizon_tiles"]: things.append((gaps[0]["dx"], "gap")) if near is not None and near["dx"] <= 9: things.append((near["dx"], "enemy")) @@ -453,7 +487,7 @@ def _now(self, st: dict, reach: float, en: list, gaps: list, walls: list, blocks kind = things[0][1] if things else None if kind == "wall": w = walls[0]; w_next = w["dx"] - lag - if w["height"] >= TALL: + if w["height"] >= self.tun["tall_wall_tiles"]: # measured: the pipe is cleared at near full speed (4.9 and up) with the jump held # long; at a jog the apex is level with its top and the side stops him (recorded) w = walls[0]; w_next = w["dx"] - lag @@ -465,9 +499,10 @@ def _now(self, st: dict, reach: float, en: list, gaps: list, walls: list, blocks return "too slow for the pipe from here: walk_left for two decisions, then run_right to full speed and jump_right at 2 to 3 tiles." if not full: return "run_right to full speed; jump_right when the pipe is about 5 tiles ahead and Mario is running flat out." - if 1.8 <= w_next <= 3.4: + lo_t, hi_t = self.tun["pipe_takeoff_tiles"] + if lo_t <= w_next <= hi_t: return "jump_right now, and keep it held for three decisions: the pipe's take-off point is here." - if w_next < 1.8: + if w_next < lo_t: return "jump_right now and hold it three decisions." return "run_right toward the pipe; jump_right when it is about 5 tiles ahead at this speed." if w_next <= 1.5: @@ -486,7 +521,7 @@ def _now(self, st: dict, reach: float, en: list, gaps: list, walls: list, blocks # when the next state would already be past it; under the blocks a miss is a death, # so there the stop and the standing jump (8 of 8) take over instead under_blocks = any(o["dx"] <= near["dx"] + 1 for o in overhead) - lo, hi = (1.5, 4.0) if under_blocks else (1.5, 6.5) + lo, hi = self.tun["enemy_takeoff_under_blocks_tiles"] if under_blocks else self.tun["enemy_takeoff_tiles"] if lo <= takeoff <= hi: return "jump_right now over the enemy: this is the take-off." if takeoff > hi: @@ -582,7 +617,7 @@ def _describe(self, st: dict) -> dict: if walls: w = walls[0] need = ("a jump held for three decisions from a run, leaving the ground 2 to 3 tiles before it; from against it or from standing it is never cleared" - if w["height"] >= TALL else "a jump held for two decisions" if w["height"] >= 3 else "a hop (jump_right)" if w["height"] <= 1 else "a jump") + if w["height"] >= self.tun["tall_wall_tiles"] else "a jump held for two decisions" if w["height"] >= 3 else "a hop (jump_right)" if w["height"] <= 1 else "a jump") parts.append(f"Wall ahead: {w['kind']} {w['dx']} tiles ahead, {w['height']} tiles tall; it takes {need}.") else: parts.append("No pipe or wall within 8 tiles.") diff --git a/kits/mario/plugin/tools/evidence.py b/kits/mario/plugin/tools/evidence.py new file mode 100644 index 0000000..baf5d31 --- /dev/null +++ b/kits/mario/plugin/tools/evidence.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""The evidence renderer for this environment: a contact sheet around each failure. + + evidence.py + +Reads the run's trace and the archived frames (observations/NNNNNN.jpg with frames.jsonl of +timestamps), finds each failure (a life lost), and writes failure_N.jpg: nine frames from one +second before the last live decision to a second and a half after, captioned with the step, the +action and the state. With Pillow missing it writes failure_N.html instead, the same frames as +images. The outer loop reads these; nothing here is specific to the calibration method. +""" +import json +import os +import re +import sys + + +def _frames(obs_dir): + idx = os.path.join(obs_dir, "frames.jsonl") + rows = [] + if os.path.exists(idx): + for line in open(idx): + try: + d = json.loads(line) + rows.append((float(d["t"]), os.path.join(obs_dir, d["file"]))) + except (ValueError, KeyError): + continue + else: + for name in sorted(os.listdir(obs_dir)): + m = re.match(r"^(\d+(?:\.\d+)?)\.jpg$", name) + if m: + rows.append((float(m.group(1)), os.path.join(obs_dir, name))) + rows.sort() + return rows + + +def _failures(trace): + steps = trace.get("steps") or [] + prev = None + last_live = None + out = [] + for st in steps: + f = ((st.get("result") or {}).get("fields") or {}) + lives = f.get("lives") + if lives is not None and prev is not None and lives < prev: + src = last_live or st + out.append(src) + if lives is not None: + prev = lives + if f.get("lives") is not None and not f.get("restarting") and not f.get("dead"): + last_live = st + return out + + +def _nearest(frames, t): + best = None + for ft, path in frames: + if best is None or abs(ft - t) < abs(best[0] - t): + best = (ft, path) + return best + + +def main(argv): + trace_path, obs_dir, out_dir = argv[1], argv[2], argv[3] + trace = json.load(open(trace_path)) + frames = _frames(obs_dir) + os.makedirs(out_dir, exist_ok=True) + fails = _failures(trace) + try: + from PIL import Image, ImageDraw + except ImportError: + Image = None + written = [] + for n, st in enumerate(fails, 1): + t0 = float(st.get("started_at") or 0) + times = [t0 - 1.0 + i * 0.3125 for i in range(9)] + picks = [_nearest(frames, t) for t in times] if frames else [] + text = str((st.get("result") or {}).get("text") or "") + cap = f"failure {n} before step {st.get('index')}: {st.get('action')}: {text[:200]}" + if Image and picks: + tiles = [] + for ft, path in picks: + im = Image.open(path).convert("RGB") + im.thumbnail((400, 250)) + tiles.append((ft - t0, im)) + w, h = tiles[0][1].size + sheet = Image.new("RGB", (w * 3, (h + 22) * 3 + 24), "white") + d = ImageDraw.Draw(sheet) + d.text((6, 4), cap[:180], fill="black") + for i, (dt, im) in enumerate(tiles): + x, y = (i % 3) * w, 24 + (i // 3) * (h + 22) + sheet.paste(im, (x, y)) + d.text((x + 4, y + h + 4), f"t{dt:+.1f}s", fill="black") + out = os.path.join(out_dir, f"failure_{n}.jpg") + sheet.save(out, quality=80) + else: + out = os.path.join(out_dir, f"failure_{n}.html") + imgs = "".join(f'
t{ft - t0:+.1f}s
' for ft, p in picks) + open(out, "w").write(f"

{cap}

{imgs}
") + written.append({"failure": n, "step": st.get("index"), "action": st.get("action"), "file": out, "text": text[:300]}) + json.dump(written, open(os.path.join(out_dir, "failures.json"), "w"), indent=1) + print(f"{len(written)} failures rendered into {out_dir}") + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) From b5baf753cf63bf7f33849b6c18f03ed6daf99e7f Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 02:32:48 -0700 Subject: [PATCH 2/6] plugins: calibrate, the outer loop as a Skill with its platform scripts Co-Authored-By: Claude Fable 5.1 --- plugins/calibrate/README.md | 21 ++ plugins/calibrate/plugin.json | 11 + plugins/calibrate/skills/calibrate/SKILL.md | 52 +++++ .../skills/calibrate/scripts/bench.py | 63 ++++++ .../calibrate/skills/calibrate/scripts/hr.py | 116 +++++++++++ .../skills/calibrate/scripts/metrics.py | 190 ++++++++++++++++++ .../skills/calibrate/scripts/probe.py | 35 ++++ .../skills/calibrate/scripts/publish.py | 63 ++++++ .../skills/calibrate/scripts/report.py | 31 +++ 9 files changed, 582 insertions(+) create mode 100644 plugins/calibrate/README.md create mode 100644 plugins/calibrate/plugin.json create mode 100644 plugins/calibrate/skills/calibrate/SKILL.md create mode 100755 plugins/calibrate/skills/calibrate/scripts/bench.py create mode 100755 plugins/calibrate/skills/calibrate/scripts/hr.py create mode 100755 plugins/calibrate/skills/calibrate/scripts/metrics.py create mode 100755 plugins/calibrate/skills/calibrate/scripts/probe.py create mode 100755 plugins/calibrate/skills/calibrate/scripts/publish.py create mode 100755 plugins/calibrate/skills/calibrate/scripts/report.py diff --git a/plugins/calibrate/README.md b/plugins/calibrate/README.md new file mode 100644 index 0000000..c659585 --- /dev/null +++ b/plugins/calibrate/README.md @@ -0,0 +1,21 @@ +# Calibrate + +The outer loop of the dual loop (the design is `docs/dual-loop.md` in the System One Harness +repository): a reasoning harness that improves another harness's configuration against the +objective that harness's package declares, one change per version, validated by that harness's +own model, with the evidence in a ledger. + +Install this package on a reasoning harness (a coding base) with platform access for one inner +harness: `HR_API_URL`, `HR_CALIBRATION_TOKEN` (a per-turn credential scoped to that harness) and +`HR_INNER_HARNESS` (its id). The Skill carries the method; the scripts do the platform work: + +| Script | What it does | +|---|---| +| `bench.py --runs 3 --package ` | K runs, one at a time; fetches each run's workspace; the objective's scoreboard and failure groups | +| `probe.py "a,b,c"` | one run driven by a fixed action sequence, to measure the environment | +| `publish.py --package ` | uploads the package as the inner harness's plugin; its instructions follow `config.yaml` | +| `report.py --package traces...` | the report over traces already on disk | + +The inner harness's package must carry `config.yaml` (with an `objective`) and `ledger.jsonl`; +the harness must write `trace.json` into its session workspace, and its environment may archive +what it showed under `observations/`. The Super Mario kit is the first such package. diff --git a/plugins/calibrate/plugin.json b/plugins/calibrate/plugin.json new file mode 100644 index 0000000..424df76 --- /dev/null +++ b/plugins/calibrate/plugin.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", + "name": "harnessrouter-calibrate", + "version": "0.1.0", + "description": "The outer loop: a reasoning harness calibrates another harness's configuration against its declared objective, one change at a time, with the evidence in a ledger.", + "author": {"name": "HarnessRouter", "url": "https://harnessrouter.ai"}, + "homepage": "https://github.com/HarnessRouter/starter-kit/tree/main/plugins/calibrate", + "repository": "https://github.com/HarnessRouter/starter-kit", + "license": "LicenseRef-HarnessRouter-Starter-Kit-1.0.0", + "keywords": ["harnessrouter", "calibration", "system-one", "system-two", "dual-loop"] +} diff --git a/plugins/calibrate/skills/calibrate/SKILL.md b/plugins/calibrate/skills/calibrate/SKILL.md new file mode 100644 index 0000000..0a4587f --- /dev/null +++ b/plugins/calibrate/skills/calibrate/SKILL.md @@ -0,0 +1,52 @@ +--- +name: calibrate +description: Calibrate another harness's configuration against its declared objective, one change per version, validated by that harness's own model. Use when asked to improve a harness, raise its pass rate, or find out why it fails. +--- + +# Calibrate a harness + +You are the outer loop. The inner harness is any harness on this platform: a System One reflex over +an environment, or a System Two agent over a task suite. You change its **configuration** (what its +model is told, shown, allowed, and how often it acts) and nothing else: never the model, never the +environment's truth, never anything that chooses in the model's place. + +Everything you need is in the inner harness's package: `config.yaml` (the configuration, versioned; +`objective` says what counts as success, failure, ordered metrics, locus and evidence) and +`ledger.jsonl` (every version so far, with its evidence and verdict). Read both first. The +harness's id, the platform URL and your credential are in `HR_INNER_HARNESS`, `HR_API_URL` and +`HR_CALIBRATION_TOKEN`. The scripts under `scripts/` do the platform work; read `--help`. + +## The method + +1. **Baseline.** `scripts/bench.py --runs 3` starts three runs, one at a time (never two on one + machine: shared machines drop frames and fake regressions), fetches each run's `trace.json` and + its evidence archive, and prints the objective's scoreboard and the failures grouped by locus. + Write the scoreboard down. +2. **Read.** Take the largest failure group. Render its evidence (the package may ship a renderer + under `tools/`, for example `tools/evidence.py `; read the images it + writes). Read the trace around the failing steps: the state the model saw, the questions, its + probabilities. Say in one sentence what killed the run there. +3. **Measure.** No change without a measurement. Probe the environment at that locus with + `scripts/probe.py "action,action,..."` (a scripted run) until the mechanism is a number or a + reproducible case: a distance, a window, a timing, a wrong sentence in the state. +4. **Change one thing.** Edit `config.yaml`: one fact or rule in `instructions`, one tunable, one + gate threshold, or one encoder setting. One change. Bump `version`. Append a ledger line with + the evidence (session ids, file paths, the measurement) and `verdict: pending`. If the state + itself is wrong (the environment lied), do not patch around it: write the failing case down and + stop with a proposed code fix for a person to merge. +5. **Publish and validate.** `scripts/publish.py` uploads the package as the harness's plugin and + relaunches it. Run the bench again (same run count). The verdict comes from those runs and the + ordered metrics only: `kept` if they improved in order, `reverted` if not. On `reverted`, put + the previous version back and publish again. Record the verdict in the ledger line. +6. **Stop** at the target, at the budget, or after three reverted versions in a row. Report the + scoreboard before and after, the ledger lines you added, and which limit stopped you: the + world's truth, the pace, or the model. + +## What never changes + +- One change per version. Batching four changes cost a night of untangling. +- The verdict comes from the inner harness's own model runs. A stand-in that follows the rendered + advice deterministically went to zero failures while the model went from three passes in three + to none in six. +- Read failures from the evidence, not from the summary text. +- Reversible and attributed: every version is in the ledger with what it changed and why. diff --git a/plugins/calibrate/skills/calibrate/scripts/bench.py b/plugins/calibrate/skills/calibrate/scripts/bench.py new file mode 100755 index 0000000..7328fad --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/bench.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +"""K runs of the inner harness, one at a time, then the objective's scoreboard and failure groups. + + bench.py --runs 3 --package [--goal ...] [--out traces/] + +Each run's workspace (trace.json, observations/) is fetched into /run-N/. The report is +printed and written to /report.json. Never start two runs at once.""" +import argparse +import json +import os +import sys + +import yaml + +sys.path.insert(0, os.path.dirname(__file__)) +import hr # noqa: E402 +import metrics # noqa: E402 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--runs", type=int, default=3) + ap.add_argument("--package", required=True, help="the inner harness's package directory (config.yaml, ledger.jsonl)") + ap.add_argument("--goal", default=None) + ap.add_argument("--model", default=None) + ap.add_argument("--out", default="traces") + a = ap.parse_args() + cfg = yaml.safe_load(open(os.path.join(a.package, "config.yaml"))) or {} + objective = cfg.get("objective") or {} + goal = a.goal or cfg.get("goal") or "Play." + traces = [] + sessions = [] + for i in range(1, a.runs + 1): + r = hr.start_run(goal, model=a.model) + rid = r["id"] + r = hr.wait_run(rid) + sid = hr.session_of(r) + print(f"run {i}: response {rid} status {r.get('status')} {((r.get('incomplete_details') or {}).get('reason') or '')} session {sid}", flush=True) + if not sid: + sys.exit("the response names no session; cannot fetch its workspace") + ws = hr.fetch_workspace(sid, os.path.join(a.out, f"run-{i}")) + if not ws["trace"]: + sys.exit(f"run {i}: no trace.json in the session's workspace ({ws['dir']}); the inner harness must write it") + t = json.load(open(ws["trace"])) + t["session_id"] = sid + t["observations"] = ws["observations"] + traces.append(t) + sessions.append(sid) + rep = metrics.report(traces, objective) + rep["sessions"] = sessions + rep["config_version"] = cfg.get("version") + os.makedirs(a.out, exist_ok=True) + json.dump(rep, open(os.path.join(a.out, "report.json"), "w"), indent=1, default=str) + print("\nscoreboard:", json.dumps(rep["scoreboard"])) + for g in rep["failure_groups"][:8]: + ex = g["examples"][0] if g["examples"] else {} + print(f" {g['count']:3d} x at {g['locus']}: run {ex.get('run')} step {ex.get('step')} {ex.get('action')}: {str(ex.get('text'))[:160]}") + print(f"\nreport: {os.path.join(a.out, 'report.json')}; traces under {a.out}/run-N/") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/calibrate/skills/calibrate/scripts/hr.py b/plugins/calibrate/skills/calibrate/scripts/hr.py new file mode 100755 index 0000000..07656e1 --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/hr.py @@ -0,0 +1,116 @@ +#!/usr/bin/env python3 +"""The platform client the calibration scripts share: start a run on the inner harness, wait for +it, fetch its workspace, read and write the harness. Credentials: HR_API_URL and +HR_CALIBRATION_TOKEN (a per-turn credential scoped to the inner harness); for a self-hosted box +in development, HR_AUTH_USER and HR_AUTH_PASSWORD log in instead.""" +import io +import json +import os +import sys +import time +import urllib.request +import urllib.error +import http.cookiejar +import zipfile + +API = os.environ.get("HR_API_URL", "").rstrip("/") +TOKEN = os.environ.get("HR_CALIBRATION_TOKEN", "") +HARNESS = os.environ.get("HR_INNER_HARNESS", "") + +_jar = http.cookiejar.CookieJar() +_opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(_jar)) +_logged_in = False + + +def _url(path: str) -> str: + base = API + if not base: + sys.exit("HR_API_URL is not set") + if path.startswith("/v1/") and "/api/harness" not in base and not base.endswith("/v1") and os.environ.get("HR_AUTH_USER"): + base = base + "/api/harness" # the self-hosted console's mount of the API + return base + path + + +def _login() -> None: + global _logged_in + if _logged_in or TOKEN or not os.environ.get("HR_AUTH_USER"): + return + body = json.dumps({"username": os.environ["HR_AUTH_USER"], "password": os.environ["HR_AUTH_PASSWORD"]}).encode() + req = urllib.request.Request(API + "/api/selfhost/login", data=body, headers={"content-type": "application/json"}, method="POST") + _opener.open(req, timeout=60).read() + _logged_in = True + + +def call(method: str, path: str, body=None, raw: bool = False, timeout: int = 120): + _login() + headers = {"content-type": "application/json", "accept": "application/json" if not raw else "*/*"} + if TOKEN: + headers["authorization"] = f"Bearer {TOKEN}" + data = json.dumps(body).encode() if body is not None else None + req = urllib.request.Request(_url(path), data=data, headers=headers, method=method) + try: + with _opener.open(req, timeout=timeout) as r: + payload = r.read() + except urllib.error.HTTPError as e: + sys.exit(f"{method} {path} -> {e.code}: {e.read()[:300].decode(errors='replace')}") + return payload if raw else (json.loads(payload) if payload else {}) + + +def start_run(goal: str, harness: str | None = None, model: str | None = None, script: list[str] | None = None, + max_step: int | None = None, timeout_seconds: int | None = None) -> dict: + """Start one run in the background; returns the response object.""" + meta = {"harness_id": harness or HARNESS} + if script: + meta["systemone"] = {"script": list(script)} + body = {"input": goal, "metadata": meta, "background": True, "store": True} + if model: + body["model"] = model + if max_step: + body["max_step"] = max_step + if timeout_seconds: + body["timeout_seconds"] = timeout_seconds + return call("POST", "/v1/responses", body) + + +def wait_run(response_id: str, poll: float = 5.0, limit: float = 1800.0) -> dict: + t0 = time.time() + while True: + r = call("GET", f"/v1/responses/{response_id}") + if r.get("status") != "in_progress": + return r + if time.time() - t0 > limit: + sys.exit(f"run {response_id} still in progress after {limit:.0f}s") + time.sleep(poll) + + +def session_of(response: dict) -> str | None: + meta = response.get("metadata") or {} + return meta.get("session_id") or response.get("session_id") or meta.get("harness_session_id") + + +def fetch_workspace(session_id: str, out_dir: str) -> dict: + """The session's files as one archive, unpacked; returns the paths of trace.json and observations/.""" + os.makedirs(out_dir, exist_ok=True) + blob = call("GET", f"/v1/sessions/{session_id}/files/archive", raw=True, timeout=600) + with zipfile.ZipFile(io.BytesIO(blob)) as z: + z.extractall(out_dir) + trace = None + obs = None + for root, dirs, files in os.walk(out_dir): + if "trace.json" in files and trace is None: + trace = os.path.join(root, "trace.json") + if os.path.basename(root) == "observations" and obs is None: + obs = root + return {"trace": trace, "observations": obs, "dir": out_dir} + + +def get_harness(harness: str | None = None) -> dict: + return call("GET", f"/v1/harnesses/{harness or HARNESS}") + + +def put_harness(body: dict, harness: str | None = None) -> dict: + return call("PUT", f"/v1/harnesses/{harness or HARNESS}", body) + + +if __name__ == "__main__": + print(json.dumps(get_harness(sys.argv[1] if len(sys.argv) > 1 else None), indent=1)[:2000]) diff --git a/plugins/calibrate/skills/calibrate/scripts/metrics.py b/plugins/calibrate/skills/calibrate/scripts/metrics.py new file mode 100755 index 0000000..2e93efe --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/metrics.py @@ -0,0 +1,190 @@ +"""The objective: what an environment declares as success, failure, ordered metrics and place. + +The outer loop reads only this declaration; it knows nothing about lives, levels or tests. The +predicates are small on purpose: + + pass: " == " | "" (truthy) | "terminal " + failure: " decreased" | " increased" | "terminal " | " == " + metrics: [{field: , better: true | false | higher | lower}, ...] ordered + locus: [, ...] where a failure happened + +`evaluate` reads a run record (the trace's dict) and returns the metrics and the failures; +`report` groups the failures of several runs by locus; `compare` says whether a run set improved. +""" +from __future__ import annotations + +import re +from statistics import mean + +ELAPSED = "elapsed" +FAILURES = "failures" + + +def _fields_of(step: dict) -> dict: + r = step.get("result") or {} + return dict(r.get("fields") or {}) if isinstance(r, dict) else {} + + +def _value(fields: dict, name: str): + return fields.get(name) + + +def _literal(text: str): + t = text.strip() + if t.lower() in ("true", "false"): + return t.lower() == "true" + if t.lower() in ("null", "none"): + return None + try: + return int(t) + except ValueError: + pass + try: + return float(t) + except ValueError: + return t.strip("'\"") + + +def _pass(run: dict, expr: str, final: dict) -> bool: + e = (expr or "").strip() + if not e: + return run.get("status") == "completed" + m = re.match(r"^terminal\s+(\S+)$", e) + if m: + return run.get("reason") == m.group(1) + m = re.match(r"^(\w+)\s*==\s*(.+)$", e) + if m: + return _value(final, m.group(1)) == _literal(m.group(2)) + return bool(_value(final, e)) + + +def _failure_steps(run: dict, expr: str) -> list[dict]: + """Each failure as {step, locus fields, last live text}: the step at which the failure signal fired.""" + e = (expr or "").strip() + steps = run.get("steps") or [] + out: list[dict] = [] + m = re.match(r"^(\w+)\s+(decreased|increased)$", e) + if m: + name, direction = m.group(1), m.group(2) + prev = None + last_live: dict | None = None + for st in steps: + f = _fields_of(st) + v = f.get(name) + if v is None: + continue + fired = prev is not None and ((direction == "decreased" and v < prev) or (direction == "increased" and v > prev)) + if fired: + src = last_live if last_live is not None else st + out.append({"step": int(src.get("index", 0)), "action": src.get("action"), "fields": _fields_of(src), + "text": str((src.get("result") or {}).get("text") or "")[:400]}) + prev = v + if not f.get("restarting") and not f.get("dead"): + last_live = st + return out + m = re.match(r"^terminal\s+(\S+)$", e) + if m: + if run.get("reason") == m.group(1) and steps: + st = steps[-1] + out.append({"step": int(st.get("index", 0)), "action": st.get("action"), "fields": _fields_of(st), + "text": str((st.get("result") or {}).get("text") or "")[:400]}) + return out + m = re.match(r"^(\w+)\s*==\s*(.+)$", e) + if m: + want = _literal(m.group(2)) + for st in steps: + if _fields_of(st).get(m.group(1)) == want: + out.append({"step": int(st.get("index", 0)), "action": st.get("action"), "fields": _fields_of(st), + "text": str((st.get("result") or {}).get("text") or "")[:400]}) + return out + return out + + +def evaluate(run: dict, objective: dict) -> dict: + """The metrics of one run under the objective, and its failures.""" + steps = run.get("steps") or [] + final: dict = {} + for st in steps: + f = _fields_of(st) + if f: + final = {**final, **f} + failures = _failure_steps(run, str(objective.get("failure") or "")) + started, finished = run.get("started_at"), run.get("finished_at") + elapsed = round(float(finished) - float(started), 1) if started is not None and finished is not None else None + metrics: dict = {"pass": _pass(run, str(objective.get("pass") or ""), final), FAILURES: len(failures), ELAPSED: elapsed} + for m in objective.get("metrics") or []: + name = m.get("field") + if name in (FAILURES, ELAPSED, "pass"): + continue + metrics[name] = final.get(name) + return {"metrics": metrics, "failures": failures, "final": final, "status": run.get("status"), "reason": run.get("reason"), + "steps": len(steps), "config_version": run.get("config_version")} + + +def _better(direction, a, b): + """True when a is better than b for the direction (true/false as target values, higher/lower).""" + if a is None or b is None: + return False + if isinstance(direction, bool) or str(direction).lower() in ("true", "false"): + want = direction if isinstance(direction, bool) else str(direction).lower() == "true" + return (a == want) and (b != want) + if str(direction).lower() == "higher": + return a > b + return a < b + + +def _aggregate(evals: list[dict], name: str): + vals = [e["metrics"].get(name) for e in evals] + vals = [v for v in vals if v is not None] + if not vals: + return None + if all(isinstance(v, bool) for v in vals): + return round(sum(1 for v in vals if v) / len(vals), 3) # a rate + return round(mean(float(v) for v in vals), 2) + + +def scoreboard(evals: list[dict], objective: dict) -> dict: + names = ["pass", FAILURES] + [m.get("field") for m in objective.get("metrics") or [] if m.get("field") not in ("pass", FAILURES, ELAPSED)] + [ELAPSED] + board = {n: _aggregate(evals, n) for n in names} + board["runs"] = len(evals) + return board + + +def compare(before: list[dict], after: list[dict], objective: dict) -> str: + """kept | reverted | same, by the ordered metrics; the first metric that differs decides.""" + b, a = scoreboard(before, objective), scoreboard(after, objective) + order = [("pass", True), (FAILURES, "lower")] + [(m.get("field"), m.get("better")) for m in objective.get("metrics") or [] + if m.get("field") not in ("pass", FAILURES)] + for name, direction in order: + x, y = a.get(name), b.get(name) + if x is None or y is None or x == y: + continue + if name == "pass": + return "kept" if x > y else "reverted" + if isinstance(direction, bool) or str(direction).lower() in ("true", "false"): + return "kept" if x > y else "reverted" + return "kept" if (str(direction).lower() == "higher" and x > y) or (str(direction).lower() == "lower" and x < y) else "reverted" + return "same" + + +def report(runs: list[dict], objective: dict) -> dict: + """The scoreboard over runs and the failures grouped by locus, largest group first.""" + evals = [evaluate(r, objective) for r in runs] + locus = list(objective.get("locus") or []) + groups: dict = {} + for i, e in enumerate(evals): + for f in e["failures"]: + key = tuple(_bucket(f["fields"].get(l)) for l in locus) if locus else ("*",) + g = groups.setdefault(key, {"locus": dict(zip(locus, key)) if locus else {}, "count": 0, "examples": []}) + g["count"] += 1 + if len(g["examples"]) < 3: + g["examples"].append({"run": i, "step": f["step"], "action": f["action"], "text": f["text"]}) + ordered = sorted(groups.values(), key=lambda g: -g["count"]) + return {"scoreboard": scoreboard(evals, objective), "runs": evals, "failure_groups": ordered} + + +def _bucket(v, width: float = 5.0): + """Numbers are grouped in bands so nearby failures fall together.""" + if isinstance(v, (int, float)) and not isinstance(v, bool): + return int(v // width) * width + return v diff --git a/plugins/calibrate/skills/calibrate/scripts/probe.py b/plugins/calibrate/skills/calibrate/scripts/probe.py new file mode 100755 index 0000000..5c9669f --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/probe.py @@ -0,0 +1,35 @@ +#!/usr/bin/env python3 +"""A probe: one run of the inner harness driven by a fixed action sequence instead of its model, +to measure the environment. Prints the trace's steps with their state text. + + probe.py "run_right,run_right,jump_right,jump_right" [--goal ...] [--out probe/]""" +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(__file__)) +import hr # noqa: E402 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("script") + ap.add_argument("--goal", default="Probe.") + ap.add_argument("--out", default="probe") + a = ap.parse_args() + actions = [x.strip() for x in a.script.split(",") if x.strip()] + r = hr.wait_run(hr.start_run(a.goal, script=actions, max_step=len(actions) + 2)["id"]) + sid = hr.session_of(r) + ws = hr.fetch_workspace(sid, a.out) + if not ws["trace"]: + sys.exit("no trace.json came back") + t = json.load(open(ws["trace"])) + for s in t.get("steps") or []: + print(f"{s.get('index'):3d} {s.get('action'):12s} {str((s.get('result') or {}).get('text') or '')[:200]}") + print(f"\nsession {sid}; workspace under {a.out}/") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/calibrate/skills/calibrate/scripts/publish.py b/plugins/calibrate/skills/calibrate/scripts/publish.py new file mode 100755 index 0000000..f386bb0 --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/publish.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +"""Publish the package as the inner harness's plugin: the new config.yaml and ledger.jsonl travel +with it, and the harness's instructions follow config.yaml. Every other harness setting is kept. + + publish.py --package """ +import argparse +import base64 +import json +import os +import sys + +import yaml + +sys.path.insert(0, os.path.dirname(__file__)) +import hr # noqa: E402 + +SKIP_DIRS = {"__pycache__", "node_modules", "observations", ".git"} +TEXT = {".py", ".md", ".json", ".yaml", ".yml", ".txt", ".sh", ".toml", ".cfg", ""} + + +def package_files(root: str) -> list[dict]: + out = [] + for dirpath, dirs, files in os.walk(root): + dirs[:] = [d for d in dirs if d not in SKIP_DIRS] + for name in files: + p = os.path.join(dirpath, name) + rel = os.path.relpath(p, root).replace(os.sep, "/") + data = open(p, "rb").read() + ext = os.path.splitext(name)[1].lower() + if ext in TEXT: + try: + out.append({"path": rel, "content": data.decode("utf-8")}) + continue + except UnicodeDecodeError: + pass + out.append({"path": rel, "content_b64": base64.b64encode(data).decode()}) + return out + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--package", required=True) + a = ap.parse_args() + cfg = yaml.safe_load(open(os.path.join(a.package, "config.yaml"))) or {} + manifest = json.load(open(os.path.join(a.package, "plugin.json"))) + h = hr.get_harness() + files = package_files(a.package) + others = [p for p in (h.get("plugins") or []) if p.get("name") != manifest["name"]] + body = {"name": h["name"], "base": h["base"], "defaultModel": h.get("defaultModel"), + "mcpServers": h.get("mcpServers", []), "skills": h.get("skills", []), + "plugins": others + [{"name": manifest["name"], "files": files, "enabled": True}], + "disabledTools": h.get("disabledTools", []), "additionalHeaders": h.get("additionalHeaders", []), + "timeoutSeconds": h.get("timeoutSeconds"), + "system_prompt": cfg.get("instructions") or h.get("systemPrompt") or "", + "max_step": h.get("maxStep") or h.get("max_step")} + r = hr.put_harness(body) + pv = [(p["name"], (p.get("manifest") or {}).get("version")) for p in r.get("plugins") or []] + print(f"published config v{cfg.get('version')} of {manifest['name']} {manifest.get('version')} on {r.get('id')}: plugins {pv}, prompt {len(r.get('systemPrompt') or '')} chars") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/calibrate/skills/calibrate/scripts/report.py b/plugins/calibrate/skills/calibrate/scripts/report.py new file mode 100755 index 0000000..211ee2f --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/report.py @@ -0,0 +1,31 @@ +#!/usr/bin/env python3 +"""The objective's scoreboard and failure groups over traces already on disk. + + report.py --package traces/run-*/trace.json""" +import argparse +import json +import os +import sys + +import yaml + +sys.path.insert(0, os.path.dirname(__file__)) +import metrics # noqa: E402 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--package", required=True) + ap.add_argument("traces", nargs="+") + a = ap.parse_args() + objective = (yaml.safe_load(open(os.path.join(a.package, "config.yaml"))) or {}).get("objective") or {} + rep = metrics.report([json.load(open(p)) for p in a.traces], objective) + print("scoreboard:", json.dumps(rep["scoreboard"])) + for g in rep["failure_groups"][:8]: + ex = g["examples"][0] if g["examples"] else {} + print(f" {g['count']:3d} x at {g['locus']}: run {ex.get('run')} step {ex.get('step')} {ex.get('action')}: {str(ex.get('text'))[:160]}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From 8e72923c27c8ce5052013a3eb386c7f964360aa5 Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 02:34:15 -0700 Subject: [PATCH 3/6] mario: the README names the package's configuration, objective and evidence Co-Authored-By: Claude Fable 5.1 --- kits/mario/README.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/kits/mario/README.md b/kits/mario/README.md index 3fd904f..0b901aa 100644 --- a/kits/mario/README.md +++ b/kits/mario/README.md @@ -138,6 +138,18 @@ npm run dev # against a console at the same origin, or set a proxy for /a npm run build # what the image serves at /kits/mario ``` +### The package's configuration, objective and evidence + +`plugin/config.yaml` is the reflex's configuration, versioned with the package: the instructions, +the gate, the encoder, the tunables the rendering reads (the enemy and gap horizons, the tall-wall +height, the measured take-off windows) and the objective (pass on `cleared`, a failure per life +lost, `level_x` as the locus, `observations/` as the evidence). The environment archives every +frame it shows under `observations/` beside the live `frame.jpg`, and `plugin/tools/evidence.py` +cuts that archive into a contact sheet around each failure. `plugin/ledger.jsonl` records every +version with its evidence and verdict. An outer harness (the `plugins/calibrate` package in this +repository) reads all of it to improve the configuration one version at a time; the design is +`docs/dual-loop.md` in the System One Harness repository. + ## Credits - The game is **Full Screen Mario**, played at [supermarioplay.com](https://supermarioplay.com/game/mario.html?v=1.0.1). Mario and its characters belong to Nintendo; this kit plays the page as a person would and ships none of the game's files. From f589039a3e6c3ea3a703d4dad278055eb371ef32 Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 02:52:38 -0700 Subject: [PATCH 4/6] calibrate: a run has ended only on a terminal status, and the bench takes the package's goal wait_run returned on any status but in_progress, and the server says running for work in progress, so the bench started its three runs at once on one machine (recorded on hr-test 2026-09-21 04:41). It now waits for completed, failed, incomplete or cancelled. session_of falls back to the inner harness's newest session. The Mario package carries its default goal for the bench. Co-Authored-By: Claude Fable 5.1 --- kits/mario/plugin/config.yaml | 1 + plugins/calibrate/plugin.json | 15 ++++++++++--- .../__pycache__/metrics.cpython-312.pyc | Bin 0 -> 13695 bytes .../calibrate/skills/calibrate/scripts/hr.py | 20 ++++++++++++++---- 4 files changed, 29 insertions(+), 7 deletions(-) create mode 100644 plugins/calibrate/skills/calibrate/scripts/__pycache__/metrics.cpython-312.pyc diff --git a/kits/mario/plugin/config.yaml b/kits/mario/plugin/config.yaml index 7ecef69..5f64790 100644 --- a/kits/mario/plugin/config.yaml +++ b/kits/mario/plugin/config.yaml @@ -2,6 +2,7 @@ # and the game. Versioned with the package; the outer loop (docs/dual-loop.md in the harness repo) # changes it one thing at a time, with the evidence in ledger.jsonl beside it. version: 1 +goal: "Play the level: keep running right, jump over enemies, gaps and pipes, hit the question blocks for coins, and get as far as you can." instructions: | You play a side-scrolling platform game as Mario, deciding several times a second. Each action sets the keys for the next moment and they stay set until you change them: run_right holds right and run; jump_right holds right, run and jump, and a jump held in the air goes again the moment Mario lands, so holding it hops him along; jump holds jump alone; walk_left holds left; wait lets every key go. Keep moving right. The state ends with a line beginning "Now:" that says what the measured facts call for at this moment; follow it. diff --git a/plugins/calibrate/plugin.json b/plugins/calibrate/plugin.json index 424df76..cef166f 100644 --- a/plugins/calibrate/plugin.json +++ b/plugins/calibrate/plugin.json @@ -1,11 +1,20 @@ { "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "harnessrouter-calibrate", - "version": "0.1.0", + "version": "0.1.1", "description": "The outer loop: a reasoning harness calibrates another harness's configuration against its declared objective, one change at a time, with the evidence in a ledger.", - "author": {"name": "HarnessRouter", "url": "https://harnessrouter.ai"}, + "author": { + "name": "HarnessRouter", + "url": "https://harnessrouter.ai" + }, "homepage": "https://github.com/HarnessRouter/starter-kit/tree/main/plugins/calibrate", "repository": "https://github.com/HarnessRouter/starter-kit", "license": "LicenseRef-HarnessRouter-Starter-Kit-1.0.0", - "keywords": ["harnessrouter", "calibration", "system-one", "system-two", "dual-loop"] + "keywords": [ + "harnessrouter", + "calibration", + "system-one", + "system-two", + "dual-loop" + ] } diff --git a/plugins/calibrate/skills/calibrate/scripts/__pycache__/metrics.cpython-312.pyc b/plugins/calibrate/skills/calibrate/scripts/__pycache__/metrics.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6bb20dfddd460c6646c54f48d531ecf955ae1f1d GIT binary patch literal 13695 zcmd@*Yj6}tdNcdbzO{O+K!}G&LISOT#k_6tFaiYF#xg?UEiA2eM$*D+SIo>Jgm=01 zos(OE5-(>i$Y&}Ru^lgGUo4khIXSti;7h6w*QF}68aWbkN|jTWO2w)Cz{Xc){3rRo zo}Jkhtd4W4@+)1nJv06Iy8G*|zi02ST`oHXiN8O7`tV~E^;`VVl2Lu3@1@}78pTp4 zD3)doA-dmi!a(D@F=RYpgr_NF>NlS-Lz($R5nFV^!j`j^mrWHQErDk#yM!%;XBoScUB;HZ zY(7ywWb{7fx4{J4`#2`jeVPl1!E;<=;9VAet;yrYHy%PzZ7?}dxN795)oiwTDFOCJW7 zH*WrP$QW5nHZmt)Bol6AwuSxu+>X^uHzxw9Mh51|0fZiZNWf=bu(uCz90M#5a(sL> z;kEYJW$3;*!o~V^b%U(IWfwI1)vCS4h`XV7TdyBdl%0QM0l}_>E$EQ z0kySO%S3t@0qBE=-GcE1rh5o50n6RWbOj>)13(*HjNl&?%ZZ9TC=w$ zc`h92i^4JJUAsHvA3Pf2&#XPjN04m@YXJriWVz-{P+U6@iuMB3wE=%9*v$hetQF1# zL!mhZ0X{eY-Lhov8yHgTo$Ay&BRzZxjJyzj!dA!<)Qr{nignnUs+6r&lU8Q5JJB@T zuyMTg7pAF(?IRmg^fmJpbE@xE_hd=k*m~KqLNYbqS`XPh=uQQ)h{Si?Aa#>!;bp!} zis_rwM%|==0C?o@UITA~hhoBmm_-in#*(5=!b$iF_*eozp%gNhwxjfw#^J^kD?1iT zro}v#&Y>9GYeF#$(8PR`c^4XCiqo+BI;E&(m}~VceU9SGAJN8erk(U6X9EZ7T zaJ}Z1n?Rh%_7pSWdVD1e;)SwsG!%k766P)#R@QjUidg_TG@zIXFI0@dFi`ptWcJNG zAK{^UDB>5fr^uZb6kkKC8vMbK zSo*}uM7?kULG3Vn{n4j)|$Nf+Ki=k#$A5bU^KgC-DRnDm(L|#-^@BHzM^Qe z>yFiV*9gyV@0yI}WwUNis`c{uq+6?BmW7;(UcNYKU5ZVYeJddCeqy&bRZyQ)*qW-1 zpD;A$b@rymsZSoS+T}9-(n&)e$R`)vGD2)WQi+XL84z1=6EvcV8)HVn1FS3`5&*)) z>THKKs$`WTm(x0dsBKAmQ_QF_wYVu};+M|@*Q^mh%ye2Or93!ATKyk1s^x)EH;WDp zR?P6I5w!V=F1=6TIa#$iY0vE3kUx+#XH*>zT5Dj^wiX3A3Gh)dDkwITSs_m3*l){T z&@_h*VO4!c+bVA@a&y#2zQE)2ZTSoJ9fPa99YVwQ?U2>`pv4!+^So9bxe;Mrih+k? z)9)7peTtdLPr|JTi7Te=NF)Tu#t0HoF>&Vy@Q9$|qtKi}_X{Y0h<2c|J98(hQ~YKq zeH?y58ZtN?HrJ)9RN1t(Dr2pZtxFP3cO1?ccNrdx)Op#>j6Iuiua&H8XRUVR=2Gd3 zG%Z^jpq<4!r(M%K`)=$@H@*2x#_f}=KE3_Qbg67zMcU80D^lEO>l;U3J37|-`U%-x z|K9%b)}J5!;OIo_hbQFKJ2UQ`l6B{-!+o_V6&>}#+(Uyq$UQ?qJSU13YfQrwIGed7nm$D zXUuvAunMbok~-@Hb$eiRSIotKFJ^nm5VM8pm_1$$r%yjuunXBF>SrcaTxedbjkRmo z($+9fONUmUx9(!S4COIN^YnIVEj3lG;A~uTTjdlywumjRqFDEJea{e2acg00vn3iF zo*tQ=r%h>In=b%^wvqobQ880no}ezsdCTU~7ytG=)LX8>pNG2imEi93iI|bU7&Ee- zMS#V){gJzAFNnWFyIHtpq21J}Sj_Ea;okojyZLC$S3H3EPsQ>q(NXi@SUL7W%kV1|QtO3qH{}L| z3y*11dsV&eJ&5TZ18*(g0FC+8khLf_y@g`an<(aBnB~s%D9e;0KlpAUVcv?Jsv!%e zKsWH3c;v7AHq20yB2|hVY(X^Fz`+7WANXT4hVfR}x+T%1fOb-~_>GmXtsJX< zytsVrI?^cx!kHuC$s;dHf+!8Xm=Rt~ zHs5i3W|ps(mOpVNIMJOl&Xkm2vtO~lwL5Km-*VG3ZkZ^R>$l3a+in~G$?~a1I(SUp ze|(CINCSe*iLYH!%$o0=n*0)KmZE|h9bojV* z2yf$?4M?G3~Q-8ifeHlqd#|-H0mH`aQRf$n^&%>kml>56QJ{Qu{IK@G+_OIPSqo`ggDg9rRr@;@C=XJmV-%JoB*j zil>@rum28oKA#IAr<3SfsR0~Mpml;?xB%peDbEF^%Lvp&K{AvEshII?I%a;GjvGfR zO?kRTj1ny{K1uT!^AG65tWlFZng$my0$l{{mGi#Cr7;M!p=+HAr!>dUM(&)$3V8g=+8c5`b4%QI^?i^baV^7Is4zZT21 z+L$e91Wl@-P0rf1IcvKJySD{)5B2v#7}Rwm)Zg=FppV8{dx)!Nq&8+O1Rio!TEM50|^VXr%F>I zulOeTRLtPq2C>Ru6`I(Jb#K$t2M!-K7Zae3ua zlPke%D=*(GyZ0sb&r~iZjM*tyu1O2;58fOckIQSDlgASKX148?wmmPkcS!AhBiz;I z(fU`Pe~U}+do!HzY#jGb6+a<`d1zL#`1;|o4euG#Prv`%&F3ao$g7(sSMR%ZNUq<1 zyG~wuaO%*Bsg6#mt55C-PMzT;@kPlKgMO|Bubdh_HTtCNs+)AJN|!;Fq|p%<9ge1s zUcLw-AN&gYulJ6flgfP=hi}H|nW|h6FuCi#i_w-i(?dNg8((7U%pBF-qemr*$5# zwa;@Z8^9IZ`1_&q;0pxO zZ-4{f80HHn+c8`Qv7a#GXCUmTM$v#(K%BFi>A{#Ha1C88C1F5rtr}bdTL9E$2$wL# z%@2`CHQ2V6xR<@Yy*$P$h&Ng(Qma7oM!yn|{xNc_`K9m%rMuyw(0+yP=AVUE0e2FQ zoO;qMo$9=JKNQEX{t{$x&@9%(P^u~Q^o-jxYbi~6ueVQGYTgQ@y>Eu6S2Si;G|DTs zNjqCJD_UoltQZ^msCH_}mf@jawB1_tw=Ya@Z_R9PmAAJ^M^9z8p91PFDNVIsu}!)g z#^|y38#c+>Fl(_VpZvkaY0Kh_WwC5oI#xMp@dC-G)(uCK&KZX@d3JbH;=sfD2eKDJ zi9VGQsXf#yhJc|LqAoyp^bZu2FM6EiiUBo?xG@g<4-iaQ2tb+WDF)wDQOpEw zY@`kNkp&1(_7l2S-zz{bKLQdWwT;jE~5C%LaB!p}kI5>R|rR}p6O?0|bU^GFPFn6}; zhtKw((bs628d+eXhNzzz2Wh?>06@q}CqoFH7<`ZO`=NMFxWl%s7b3Ud^WAY_X{W!p zm*;vRWarxkb}xjMclh+`Kf#_Da`_MHn*=4<*W5IXts7}f`LBDX9m{T-?h|7Ee$^6s z&WaFsWJM?f^%a5MiUBRqFN;_zJ9h`noyNJtw9)!)xk>RXJoBf^_dvlTCiaRdY0mF#u)GG*%uX3YJ*ZOn>S1hE7U%*?WdiITQUFYX{mtgFv&p7paQMKq zb4kXzWVHP=XHB{}u}5;Ozq4?$>{xR}OjVDbeJy^eDJ5d@nmZ0RSjfOzGDs}4ea)0* z4TQL^wq>aOHBv8o6{$`mF3&Cz;mW z|H?%W>!ha+&_pva8JWS0>2j1bK0f^OzD#i%_7ZUPIy7XD*nPQ?7MJ z2E+i2$xz7AmQ(igN6AI;LBU2MuX$8&U~U^tghgC{iRSwE()f*GIF&61k+C8@0$a?n zv=}&nHD>dwVFc(gZU&>fqcJa5P}9vg3!I!;Y@_%j`T zsXHiloR);4uVGR29s@k~8lE%U1%Pym;VX(d1rM;{>u%JgEpPh(%9jKRdrv#+zKaG}(I9KDueFS}v*0xN6f}qFHioob@amHNW~yqB*(Z zrRQcGC5Z#~zp~)W?kmoXB=-Nt@?EZ?PmJZ5uk;{!BB#-kPM&`byzv^9$JeQxw>Ym9 z^cZb(_jW1N7h5U%1qu}q;wC^s)@CrG;X6g*yugV2kPVj+@ zM?TX|6yqGXA%#kjss<}299yw!L#ulB3(yTNDZw1T%tr2!$C(hJ4=UXb1P2;E#J~ML z5KC7XRTsAG#7t#bWErZY+zq7)eO{r51cawDE|O=p^z4N)J^{6_!B6M~)ht2%o1-Gt zGa8f~_3t%|AD(zxZrGbR3wK3ai?8>kEz;7*CFeS+D>&WNpXmbc;Go=fezNPL6uT&O zUA#xpoBVVJK2LVcYw#mulfX-Wn)qqY6=i=$0%G{4pi_UW_9bzsecM*rj} zJ!%_kO;`WL5j0>O?$qW>J7$(H8!P?E;ne0y=lZlAGPtOc3clr;a;$i7c{=#6Z@OVe zreTNNuv6M|IMZl}VwWRuP8Raay|J6c75s@Vy zHyN7kMRR06?i7*v%YdwZ=s;I!o+$$Ex40pXrGgfq9cy4nC93I>GblhuD8OVMcB5)~ zVB7@mRC6Bt&a4w<2^?vIaDUpcM40OAfdMq>(@|_WwMs@hVx{0tn~tT zf&#{X0X2m@N1cte;p{a?EMXy_-Sk3lQggZ-N1B^rMZ*r(f&Jj=%syTC1b3b;YQS{4 zWTHl{fqS_^>kq3%!}@{rZuT2)H&gGGXsd(JD1syjoRj+Q!n>U86ccD!X54z>g5 z$|)fXT>s?Nd!u-D8QGfVEmSX?0^8Frlwu4JoB_bz$Tb>_?Fpg}@0vlJkht6Cm_Mlc zY(Rq!m8Kulx=m3H6q#qz4I5Yr7f?vY%-y2XFrT$MmggBvsOz=Bl{Ro8L<48w9yV%f zUL()JL!m>&87QC@!Y~QK##Bpck$R`LfVuYpv?VvqzEN$Fg}XmKJV9TFP{~;R4aYm~ z8}4yhX4X$vZut1HYWn?PWdjYDW$6YIlZ3Iqk7Jty;Lijm&Ykz;l_r7zA(X1oO*jJ- zAH)njk7}3_@PdAlbq|-$K`Rn95F33D9qV}5RWE}By*-8&b&Ng)1@q|@&~nbIp4M$4 ze}6aY-*KVxp`~YQ{u){fh`t0h-a0<`!HEx^C5{h(WjwVsc?7}sxIyb>^kvg_Q2i1rx-aqCqAtXyyIIFvlW9136ovJj0yJUE) znjeMQpTbWl2W)|X|F3J-$2}nA*UvgjQj4xFy|Q$4P%d3L z2`1Wl+3AD$X?aCr?`+AU(Yk*wVTMgfI=OD9v@F$lb!oEc&b|Xz>tv)M$fnA6ONWjn zch8h8m@cW#lvIx`|FrHS&qslYx?ANRMyA&7l@1@3x&zW#A+u7HOQH}&b-I$hm!BD} z9)AAvGvpd&>co}pGKA;AY^+0%R>Kw7G-&B1wdvLgdSb=Lw%e_@&wqLn!LX+@-Dfg& zq0dVCzjRP#kAWyHt^T^0f=gsauARDa>ebFn!&6hmI}`h6t7^v%r=OOqHs3N!$Gb8P zzhv@*ZPI=DRb3M8$&sXQkqr4ln8b=dK^W+hj zHWz^8wet^OFfO<_k=*GEJ6fXs-KbpQKOhJ(0KW=?V-uJe{9Pl$ggHMC{7dI);MWW> zVgy{8a1-9AIV&-e&oh{z+Y%JC{qRuCgF#m8o8yCd8QWm?H}DhisDknXZprGaf$N)6 z9pIXLuW{V}3(wT59b~IU%~u*Pe{ZZJ<6a?ISAg!3*smU_?EM6C2!IapPXOdZz>JWe zk#|)mYl2Uh5poCOPldaUghgOi^|ArBP-88m>~;Jk>nJMN66?RU<1syn_ufx|6`KQi z96RU0;2TVdSh?4+%7Yo1Qr7k*rGEAIX+C?VtKQer@WdHyG literal 0 HcmV?d00001 diff --git a/plugins/calibrate/skills/calibrate/scripts/hr.py b/plugins/calibrate/skills/calibrate/scripts/hr.py index 07656e1..3be816a 100755 --- a/plugins/calibrate/skills/calibrate/scripts/hr.py +++ b/plugins/calibrate/skills/calibrate/scripts/hr.py @@ -72,20 +72,32 @@ def start_run(goal: str, harness: str | None = None, model: str | None = None, s return call("POST", "/v1/responses", body) -def wait_run(response_id: str, poll: float = 5.0, limit: float = 1800.0) -> dict: +TERMINAL = ("completed", "failed", "incomplete", "cancelled") + + +def wait_run(response_id: str, poll: float = 5.0, limit: float = 3600.0) -> dict: + """Blocks until the run has ended. Only a terminal status ends the wait: a server may say + `running` or `queued` for work in progress, and treating anything but `in_progress` as done + started three runs at once on one machine (2026-09-21).""" t0 = time.time() while True: r = call("GET", f"/v1/responses/{response_id}") - if r.get("status") != "in_progress": + if r.get("status") in TERMINAL: return r if time.time() - t0 > limit: - sys.exit(f"run {response_id} still in progress after {limit:.0f}s") + sys.exit(f"run {response_id} still not finished after {limit:.0f}s") time.sleep(poll) def session_of(response: dict) -> str | None: + """The session the response ran in: named on the response, or else the inner harness's newest session.""" meta = response.get("metadata") or {} - return meta.get("session_id") or response.get("session_id") or meta.get("harness_session_id") + sid = meta.get("session_id") or response.get("session_id") or meta.get("harness_session_id") + if sid: + return sid + listing = call("GET", f"/v1/sessions?harness={HARNESS}&limit=1") + sessions = listing.get("sessions") or [] + return sessions[0].get("session_id") if sessions else None def fetch_workspace(session_id: str, out_dir: str) -> dict: From 71015f96a05745a1c049b51b76dfb9a4585ec107 Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 02:54:55 -0700 Subject: [PATCH 5/6] calibrate: fetch the inner package by name; a publish replaces the same-named package only The Calibrator took the harness's plugin export (a generated package named after the harness, with no version) as the package and published it back, which added a second package beside the environment's. fetch.py takes the package that carries the environment through the plugins files endpoint; publish.py refuses a manifest without a version and a name the harness does not carry unless told --new. Co-Authored-By: Claude Fable 5.1 --- plugins/calibrate/README.md | 1 + plugins/calibrate/plugin.json | 2 +- plugins/calibrate/skills/calibrate/SKILL.md | 9 ++- .../skills/calibrate/scripts/fetch.py | 61 +++++++++++++++++++ .../skills/calibrate/scripts/publish.py | 6 ++ 5 files changed, 75 insertions(+), 4 deletions(-) create mode 100644 plugins/calibrate/skills/calibrate/scripts/fetch.py diff --git a/plugins/calibrate/README.md b/plugins/calibrate/README.md index c659585..5f552dc 100644 --- a/plugins/calibrate/README.md +++ b/plugins/calibrate/README.md @@ -12,6 +12,7 @@ harness: `HR_API_URL`, `HR_CALIBRATION_TOKEN` (a per-turn credential scoped to t | Script | What it does | |---|---| | `bench.py --runs 3 --package ` | K runs, one at a time; fetches each run's workspace; the objective's scoreboard and failure groups | +| `fetch.py --out package` | the inner harness's package (the one carrying the environment), never the harness's export | | `probe.py "a,b,c"` | one run driven by a fixed action sequence, to measure the environment | | `publish.py --package ` | uploads the package as the inner harness's plugin; its instructions follow `config.yaml` | | `report.py --package traces...` | the report over traces already on disk | diff --git a/plugins/calibrate/plugin.json b/plugins/calibrate/plugin.json index cef166f..11c8508 100644 --- a/plugins/calibrate/plugin.json +++ b/plugins/calibrate/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "harnessrouter-calibrate", - "version": "0.1.1", + "version": "0.1.2", "description": "The outer loop: a reasoning harness calibrates another harness's configuration against its declared objective, one change at a time, with the evidence in a ledger.", "author": { "name": "HarnessRouter", diff --git a/plugins/calibrate/skills/calibrate/SKILL.md b/plugins/calibrate/skills/calibrate/SKILL.md index 0a4587f..efb2edc 100644 --- a/plugins/calibrate/skills/calibrate/SKILL.md +++ b/plugins/calibrate/skills/calibrate/SKILL.md @@ -12,9 +12,12 @@ environment's truth, never anything that chooses in the model's place. Everything you need is in the inner harness's package: `config.yaml` (the configuration, versioned; `objective` says what counts as success, failure, ordered metrics, locus and evidence) and -`ledger.jsonl` (every version so far, with its evidence and verdict). Read both first. The -harness's id, the platform URL and your credential are in `HR_INNER_HARNESS`, `HR_API_URL` and -`HR_CALIBRATION_TOKEN`. The scripts under `scripts/` do the platform work; read `--help`. +`ledger.jsonl` (every version so far, with its evidence and verdict). Fetch it first with +`scripts/fetch.py --out package` (the package that carries the environment), then read both. +Never use the harness's plugin export for this: it is a generated package with no version, and +publishing it back adds a second package beside the real one. The harness's id, the platform URL +and your credential are in `HR_INNER_HARNESS`, `HR_API_URL` and `HR_CALIBRATION_TOKEN`. The +scripts under `scripts/` do the platform work; read `--help`. ## The method diff --git a/plugins/calibrate/skills/calibrate/scripts/fetch.py b/plugins/calibrate/skills/calibrate/scripts/fetch.py new file mode 100644 index 0000000..8f15a8f --- /dev/null +++ b/plugins/calibrate/skills/calibrate/scripts/fetch.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +"""Fetch the inner harness's package (config.yaml, ledger.jsonl, the environment) into a directory. + + fetch.py [--name ] [--out package] + +Without --name, the package that carries the environment (the one with MCP servers) is taken. +Use this, not the harness's plugin export: the export is a generated package named after the +harness and has no version, and publishing it back adds a second package beside the real one.""" +import argparse +import base64 +import os +import sys + +sys.path.insert(0, os.path.dirname(__file__)) +import hr # noqa: E402 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--name", default=None) + ap.add_argument("--out", default="package") + a = ap.parse_args() + h = hr.get_harness() + plugins = h.get("plugins") or [] + if not plugins: + sys.exit("the inner harness carries no package") + chosen = None + if a.name: + chosen = next((p for p in plugins if p.get("name") == a.name), None) + else: + chosen = next((p for p in plugins if p.get("mcpServers")), None) or plugins[0] + if chosen is None: + sys.exit(f"no package named {a.name!r}; the harness carries {[p.get('name') for p in plugins]}") + listing = hr.call("GET", f"/v1/harnesses/{hr.HARNESS}/plugins/{chosen['name']}/files") + files = listing.get("files") if isinstance(listing, dict) else listing + if not files: + sys.exit(f"the package {chosen['name']!r} has no files to fetch") + os.makedirs(a.out, exist_ok=True) + n = 0 + for f in files: + rel = f.get("path") + if not rel or rel.startswith("/") or ".." in rel: + continue + p = os.path.join(a.out, rel) + os.makedirs(os.path.dirname(p) or ".", exist_ok=True) + if f.get("content") is not None: + open(p, "w").write(f["content"]) + elif f.get("content_b64") is not None: + open(p, "wb").write(base64.b64decode(f["content_b64"])) + else: + continue + n += 1 + print(f"fetched {chosen['name']} {(chosen.get('manifest') or {}).get('version')}: {n} files into {a.out}/") + for must in ("config.yaml", "ledger.jsonl", "plugin.json"): + if not os.path.exists(os.path.join(a.out, must)): + print(f" note: no {must} in the package") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/calibrate/skills/calibrate/scripts/publish.py b/plugins/calibrate/skills/calibrate/scripts/publish.py index f386bb0..395c799 100755 --- a/plugins/calibrate/skills/calibrate/scripts/publish.py +++ b/plugins/calibrate/skills/calibrate/scripts/publish.py @@ -40,11 +40,17 @@ def package_files(root: str) -> list[dict]: def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--package", required=True) + ap.add_argument("--new", action="store_true", help="add a package the harness does not carry yet") a = ap.parse_args() cfg = yaml.safe_load(open(os.path.join(a.package, "config.yaml"))) or {} manifest = json.load(open(os.path.join(a.package, "plugin.json"))) h = hr.get_harness() files = package_files(a.package) + names = [p.get("name") for p in (h.get("plugins") or [])] + if not manifest.get("version"): + sys.exit(f"{a.package}/plugin.json has no version; fetch the real package with fetch.py, never the harness's export") + if manifest["name"] not in names and not a.new: + sys.exit(f"the harness carries {names}, not {manifest['name']!r}; a publish replaces the same-named package (use --new to add one)") others = [p for p in (h.get("plugins") or []) if p.get("name") != manifest["name"]] body = {"name": h["name"], "base": h["base"], "defaultModel": h.get("defaultModel"), "mcpServers": h.get("mcpServers", []), "skills": h.get("skills", []), From 09a848b8783a69d7c17a22a3838af856e0520054 Mon Sep 17 00:00:00 2001 From: richard-epsilla Date: Mon, 21 Sep 2026 03:23:52 -0700 Subject: [PATCH 6/6] calibrate: a run has ended when its session is at rest, the archive is retried, and a run without a record counts as a failed run A response read failed while its turn was still running, and the archive, built from the checkpoint that lands when the turn ends, answered 404 right after; the bench exited and the Calibrator left the method to improvise. wait_run now believes a terminal status only once the session is no longer running, and a non-completed one only after four consecutive reads; fetch_workspace retries the archive on 404 for two minutes; every non-completed status read is logged to wait-log.jsonl for the platform to chase; a run without a record is a failed run in the scoreboard, not a crash. Co-Authored-By: Claude Fable 5.1 --- plugins/calibrate/plugin.json | 2 +- .../skills/calibrate/scripts/bench.py | 9 ++- .../calibrate/skills/calibrate/scripts/hr.py | 59 +++++++++++++++++-- 3 files changed, 61 insertions(+), 9 deletions(-) diff --git a/plugins/calibrate/plugin.json b/plugins/calibrate/plugin.json index 11c8508..6cea9a7 100644 --- a/plugins/calibrate/plugin.json +++ b/plugins/calibrate/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "harnessrouter-calibrate", - "version": "0.1.2", + "version": "0.1.3", "description": "The outer loop: a reasoning harness calibrates another harness's configuration against its declared objective, one change at a time, with the evidence in a ledger.", "author": { "name": "HarnessRouter", diff --git a/plugins/calibrate/skills/calibrate/scripts/bench.py b/plugins/calibrate/skills/calibrate/scripts/bench.py index 7328fad..c49da60 100755 --- a/plugins/calibrate/skills/calibrate/scripts/bench.py +++ b/plugins/calibrate/skills/calibrate/scripts/bench.py @@ -40,10 +40,15 @@ def main() -> int: sys.exit("the response names no session; cannot fetch its workspace") ws = hr.fetch_workspace(sid, os.path.join(a.out, f"run-{i}")) if not ws["trace"]: - sys.exit(f"run {i}: no trace.json in the session's workspace ({ws['dir']}); the inner harness must write it") - t = json.load(open(ws["trace"])) + # a run without a record is still a run: it counts as a failure with no locus, and + # the bench goes on; the reason is in wait-log.jsonl for the platform + print(f"run {i}: no trace.json in the session's workspace ({ws['dir']}); counted as a failed run", flush=True) + t = {"status": r.get("status"), "reason": ((r.get("incomplete_details") or {}).get("reason") or (r.get("error") or {}).get("message") if isinstance(r.get("error"), dict) else r.get("error")), "steps": [], "started_at": None, "finished_at": None} + else: + t = json.load(open(ws["trace"])) t["session_id"] = sid t["observations"] = ws["observations"] + t["response_id"] = rid traces.append(t) sessions.append(sid) rep = metrics.report(traces, objective) diff --git a/plugins/calibrate/skills/calibrate/scripts/hr.py b/plugins/calibrate/skills/calibrate/scripts/hr.py index 3be816a..d1f24c9 100755 --- a/plugins/calibrate/skills/calibrate/scripts/hr.py +++ b/plugins/calibrate/skills/calibrate/scripts/hr.py @@ -73,17 +73,51 @@ def start_run(goal: str, harness: str | None = None, model: str | None = None, s TERMINAL = ("completed", "failed", "incomplete", "cancelled") +LOG = os.environ.get("HR_CALIBRATION_LOG", "wait-log.jsonl") + + +def _log(kind: str, payload) -> None: + try: + with open(LOG, "a") as f: + f.write(json.dumps({"at": time.time(), "kind": kind, "payload": payload}, default=str)[:4000] + "\n") + except OSError: + pass + + +def session_running(session_id: str | None) -> bool: + if not session_id: + return False + try: + s = call("GET", f"/v1/sessions/{session_id}") + except SystemExit: + return False + return str(s.get("status") or "") in ("running", "in_progress", "queued") def wait_run(response_id: str, poll: float = 5.0, limit: float = 3600.0) -> dict: - """Blocks until the run has ended. Only a terminal status ends the wait: a server may say - `running` or `queued` for work in progress, and treating anything but `in_progress` as done - started three runs at once on one machine (2026-09-21).""" + """Blocks until the run has ended: the response carries a terminal status AND its session is no + longer running. A server may say `running` or `queued` for work in progress, and a transient + `failed` was seen on a response whose turn was still running (2026-09-21), so a non-completed + terminal status is re-read a few times before it is believed. Every status read that is not + `completed` is logged to wait-log.jsonl for the platform to chase.""" t0 = time.time() + confirmations = 0 while True: r = call("GET", f"/v1/responses/{response_id}") - if r.get("status") in TERMINAL: - return r + status = r.get("status") + if status != "completed": + _log("status", {"response_id": response_id, "status": status, "error": r.get("error"), + "incomplete_details": r.get("incomplete_details"), "metadata": r.get("metadata")}) + if status in TERMINAL: + sid = session_of(r) + if session_running(sid): + confirmations = 0 + elif status == "completed": + return r + else: + confirmations += 1 + if confirmations >= 4: # about twenty seconds of the same terminal status with the session at rest + return r if time.time() - t0 > limit: sys.exit(f"run {response_id} still not finished after {limit:.0f}s") time.sleep(poll) @@ -103,7 +137,20 @@ def session_of(response: dict) -> str | None: def fetch_workspace(session_id: str, out_dir: str) -> dict: """The session's files as one archive, unpacked; returns the paths of trace.json and observations/.""" os.makedirs(out_dir, exist_ok=True) - blob = call("GET", f"/v1/sessions/{session_id}/files/archive", raw=True, timeout=600) + blob = None + for attempt in range(12): + # the archive is built from the session's checkpoint, which lands once the turn has ended; + # a 404 right after the run means "not yet", so wait and ask again + try: + blob = call("GET", f"/v1/sessions/{session_id}/files/archive", raw=True, timeout=600) + break + except SystemExit as e: + _log("archive", {"session_id": session_id, "attempt": attempt, "error": str(e)[:300]}) + if "404" not in str(e) and "409" not in str(e): + raise + time.sleep(10) + if blob is None: + return {"trace": None, "observations": None, "dir": out_dir, "missing": True} with zipfile.ZipFile(io.BytesIO(blob)) as z: z.extractall(out_dir) trace = None