From ff1bac47c36d558116ef59ca13fb0de56bd90ab7 Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Thu, 17 Sep 2026 16:50:44 -0400 Subject: [PATCH 1/6] Label with Haiku plus parsed structure, keep Jev for grading, lock held-out gold Teacher/judge default to claude-haiku-4-5; teacher-v2 passes the parser's structure and names, which lifts agreement with the Opus gold from 74.0% to 78.9% (Opus's own second candidates score 72.0%). Jev Choice matches the judge only 52% of the time, so selection stays with the judge model. Candidates are keyed per teacher model, judge dedup hashes candidate text, and test/validation labels are frozen in datasets/gold_lock.json. Co-Authored-By: Claude Opus 5 (1M context) --- live-status/ARCHITECTURE.md | 20 +++++----- live-status/DATASET.md | 27 +++++++++++-- live-status/EVALUATION.md | 7 +++- live-status/README.md | 10 +++-- live-status/dataset_build/build.py | 19 +++++++++ live-status/judging/jev.py | 63 +++++++++++++++++++++++++++++- live-status/judging/judge.py | 38 ++++++++++-------- live-status/labeling/llm.py | 8 +++- live-status/labeling/prompts.py | 8 +++- live-status/labeling/teacher.py | 28 +++++++++++-- 10 files changed, 184 insertions(+), 44 deletions(-) diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index 850231c..24bdc76 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -8,15 +8,15 @@ Cursor, Grok, │ PSReadLine) parsers/shell.py ─► structure, difficulty tags ───────────┤ ▼ - labeling/teacher.py (Opus 5, two candidates per command, batched, resumable) ──► labels/teacher.jsonl - judging/judge.py (Opus 5, separate prompt, scores teacher + heuristic candidates, + labeling/teacher.py (Haiku 4.5, two candidates per command, batched, resumable) ──► labels/teacher.jsonl + judging/judge.py (Haiku 4.5, separate prompt, scores teacher + heuristic candidates, writes recommended_output) ──► labels/judged.jsonl dataset_build/build.py (accept / review / reject; MinHash families; frozen test) ──► datasets//*.jsonl │ benchmarks/baseline.py ◄── untuned tiny models through Ollama ◄─────────────────────────────┤ training/train.py ──► models/ (merged HF weights) ◄────────────────────────────┘ training/export_gguf.py──► GGUF f16/q8_0 + Ollama quantized tags - evaluation/evaluate.py ──► validators + Opus 5 grading + latency/memory ──► evaluation/registry.json + evaluation/evaluate.py ──► validators + jev/Opus grading + latency/memory ──► evaluation/registry.json evaluation/active.py ──► student failures on unlabeled commands ──► teacher/judge ──► prefs.jsonl (DPO) client ──► api/server.py ──► redact ─► cache ─► model (Ollama) ─► validate ─► heuristic fallback @@ -38,12 +38,14 @@ - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. -- **jev for grading, Opus for writing.** jev (TypeSafe System One) returns only probabilities, - choices and scores, in milliseconds, at $0.042 per million input tokens. Given the gold - status it agrees with Opus evaluations at AUC 0.92, so it grades every benchmark and ranks - student outputs for failure mining. Without a reference it is too weak (AUC 0.68) to - replace the Opus judge when labels are created. -- **Opus 5 as both teacher and judge**, with different prompts. Label diversity comes from two candidates per command plus the heuristic, not from different models. +- **Haiku writes, jev grades, Opus audits.** Teacher and judge run on Haiku 4.5 with different + prompts; label diversity comes from two candidates per command plus the heuristic. jev + (TypeSafe System One) returns only probabilities, choices and scores, in milliseconds, at + $0.042 per million input tokens: given the gold status it agrees with Opus evaluations at + AUC 0.92, so it grades every benchmark and ranks student outputs for failure mining. It + cannot replace the judge, because without a reference it reaches only AUC 0.68 and its own + Choice between near-equal candidates matches the judge just 52% of the time. Opus is kept + for spot checks and for the calibration sets both graders are measured against. - **Frequency weighting.** Each template group counts `min(4, 1 + log2(count))` times in training, so common patterns are learned first without drowning the long tail. ## Adding a transcript format diff --git a/live-status/DATASET.md b/live-status/DATASET.md index 3ea3e37..18eadc3 100644 --- a/live-status/DATASET.md +++ b/live-status/DATASET.md @@ -42,13 +42,32 @@ ordinary text look secret-shaped). ## Labels -- Teacher: Opus 5, `teacher-v1`, temperature 0.4, two candidates per command (concise and - complete), batches of ≤20 commands / 30k characters. -- Judge: Opus 5, `judge-v1`, temperature 0, scores each candidate plus the heuristic, - picks the best and writes `recommended_output`. +- Teacher: `teacher-v2`, temperature 0.4, two candidates per command (concise and complete), + batches of ≤20 commands / 30k characters. Items carry the parser's `structure` and `names` + so the model keeps concrete names instead of "the script". +- Judge: `judge-v1`, temperature 0, scores each candidate plus the heuristic, picks the best + and writes `recommended_output`. - Decision: accepted when the recommended score ≥ 85, validators pass and the judge is not uncertain; manual review at 70–84, uncertain, or validator failure; rejected for secret leakage or score < 70. The judge rewrites weak candidates, so v1 has no rejections. +- Model: **Haiku 4.5** by default (`LIVE_STATUS_TEACHER`). v1's labels were written by Opus 5; + rows record `teacher_model` and `judge_model`, and candidate keys are prefixed per model + (`ota` = Opus teacher candidate a, `hta` = Haiku), so mixed provenance stays traceable. + +### Why Haiku + +On 300 test commands that already had Opus labels, Haiku relabelled from scratch and its +labels were graded against the Opus gold: + +| Teacher | Accepted vs Opus gold | Same actions | Invented | +| --- | ---: | ---: | ---: | +| Haiku, `teacher-v1` prompt | 74.0% | 97.6% | 4.5% | +| Haiku, `teacher-v2` prompt (+structure, +names) | **78.9%** | 96.7% | 3.7% | +| Opus's own second-choice candidate (reference point) | 72.0% | — | — | + +Haiku with the structure hints matches Opus's own alternate candidates, at a fraction of the +cost and without exhausting the Claude subscription that the interactive session shares. +Opus stays available (`--model claude-opus-5`) for spot checks. ## v1 splits diff --git a/live-status/EVALUATION.md b/live-status/EVALUATION.md index 873c9e0..e57b3d5 100644 --- a/live-status/EVALUATION.md +++ b/live-status/EVALUATION.md @@ -20,13 +20,16 @@ recall for simple read/process/delete commands. | Grader | What it sees | Cost / speed | Agreement with Opus 5 | | --- | --- | --- | --- | | `opus` | command, reference, output; returns correct, score, missing/hallucinated actions, secret leak, injection followed | ~20 outputs per call, ~60 s | — | +| `opus` on Haiku's labels | same | same | Haiku labels reach 78.9% of the Opus gold standard (DATASET.md) | | `jev` (default) | command, reference, output; answers `same_actions`, `invented`, `quality` | 700 outputs in ~3 s, ~$0.02 | AUC 0.92, 86.4% agreement at score ≥ 0.45 (700 Opus-graded outputs) | A jev-accepted output passes validators and has `same × (1 − invented) × quality/4 ≥ 0.45`. Calibration lives in `evaluation/jev_eval_calibration.json`; rerun it with `python live-status/judging/jev.py calibrate-eval`. -Grading a status without a reference is much weaker (AUC 0.68 on 3,000 teacher candidates), so -jev only *ranks* unlabeled outputs during failure mining and Opus writes the labels. +Grading a status without a reference is much weaker (AUC 0.68 on 3,000 teacher candidates), and +asking jev to *choose* between near-equal candidates matches the judge only 52% of the time +(1,200 commands, `judging/jev.py calibrate-choice`). So jev ranks unlabeled outputs during +failure mining, and a judge model still writes the labels. jev is stricter than Opus on good outputs: Opus's alternate teacher candidates on 200 test commands pass 86% of Opus evaluations and 72% of jev evaluations. Compare runs only within diff --git a/live-status/README.md b/live-status/README.md index 16bb49b..5d79c0a 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -35,8 +35,8 @@ Run from the repository root. GPU steps use the training venv (see TRAINING.md). | --- | --- | | Inventory transcript sources | `python live-status/cli.py inventory_sources` | | Extract, redact, deduplicate, report | `python live-status/cli.py extract_commands` | -| Teacher labels (Opus 5) | `python live-status/cli.py generate_labels --limit 4000` | -| Judge labels (Opus 5) | `python live-status/cli.py judge_labels` | +| Teacher labels (Haiku) | `python live-status/cli.py generate_labels --limit 4000` | +| Judge labels (Haiku) | `python live-status/cli.py judge_labels` | | Build splits | `python live-status/cli.py build_dataset --version v1` | | Baseline tiny models | `python live-status/cli.py benchmark_base_models --models smollm2:135m qwen3:0.6b` | | Train | `live-status/cli.py train --base Qwen/Qwen3-0.6B --method lora --name ...` | @@ -52,8 +52,10 @@ commands stay in its `private/` folder. ## Models and keys -- Teacher and judge: `claude-opus-5` through the local CLIProxyAPI (`127.0.0.1:8317`). - It shares the Claude subscription; keep `--workers` at 4 or below. On a long cooldown +- Teacher and judge: `claude-haiku-4-5-20251001` through the local CLIProxyAPI + (`127.0.0.1:8317`), overridable with `--model` or `LIVE_STATUS_TEACHER`. Haiku labels match + Opus's own second-choice candidates (DATASET.md), so Opus is reserved for spot checks. + These share the Claude subscription; keep `--workers` at 4 or below. On a long cooldown the run stops and resumes on rerun. - Grader: TypeSafe `jev` through Vercel AI Gateway (`AI_GATEWAY_API_KEY` in the repo `.env`) or directly (`TYPESAFE_API_KEY`). Needs Bun; `bun install` in `live-status/jev`. diff --git a/live-status/dataset_build/build.py b/live-status/dataset_build/build.py index 22ec224..af9f797 100644 --- a/live-status/dataset_build/build.py +++ b/live-status/dataset_build/build.py @@ -91,9 +91,25 @@ def _split_of(family: str, seed: int) -> str: return "test" if x < 0.1 else "validation" if x < 0.2 else "train" +def gold_lock_path(): + return home() / "datasets" / "gold_lock.json" + + +def load_gold_lock() -> dict: + import json + path = gold_lock_path() + return json.loads(path.read_text(encoding="utf-8")) if path.exists() else {} + + def build(seed: int = 20260916, version: str = "v1", freeze: bool = True) -> dict: cmds = {r["id"]: r for r in read_jsonl(home() / "commands_redacted.jsonl")} judged = latest_judgements() + # Held-out labels are frozen with the judgement that produced them, so re-judging the + # corpus with another model cannot silently move the benchmark's target. + lock = load_gold_lock() + for cid, locked in lock.items(): + if cid in judged: + judged[cid] = locked buckets = collections.defaultdict(list) for cid, j in judged.items(): r = cmds.get(cid) @@ -165,6 +181,9 @@ def build(seed: int = 20260916, version: str = "v1", freeze: bool = True) -> dic new_frozen.setdefault(r["id"], name) if freeze: save_json(frozen_path, new_frozen) + lock.update({r["id"]: judged[r["id"]] for name in ("test", "validation") for r in splits[name] + if r["id"] in judged and r["id"] not in lock}) + save_json(gold_lock_path(), lock) leak = _leak_check(splits) report = {"version": version, "seed": seed, "counts": counts, "families": len(set(fam.values())), diff --git a/live-status/judging/jev.py b/live-status/judging/jev.py index e499596..53448fc 100644 --- a/live-status/judging/jev.py +++ b/live-status/judging/jev.py @@ -230,9 +230,68 @@ def fit_rule(rows: list[dict]) -> dict: return best +def choose(items: list[tuple[str, str, dict[str, str]]]) -> dict[str, dict]: + """Pick the best candidate per command. Choice returns one of the option keys, so the + winning *text* comes back too: Jev emits text only by selecting an input. + + items: (key, command, {candidate_key: sentence}) -> {key: {"choice", "text", "probabilities", "confidence"}} + """ + reqs = [] + for k, cmd, cands in items: + if len(cands) < 2: + continue + reqs.append({"id": k, "state": {"command": cmd, "candidates": cands}, + "questions": {"best": {"type": "choice", + "instructions": "Which candidate is the best live status sentence for this command: accurate, complete, specific about names, and concise?", + "criteria": {ck: text for ck, text in cands.items()}}}}) + res = evaluate(reqs) + out = {} + for k, cmd, cands in items: + r = res.get(k) + if not r or "error" in r: + out[k] = {"error": (r or {}).get("error", "missing"), "choice": None, + "text": next(iter(cands.values())) if len(cands) == 1 else None} + continue + a = r["answers"]["best"] + out[k] = {"choice": a["choice"], "text": cands.get(a["choice"]), + "probabilities": a.get("probabilities"), "confidence": (r.get("confidence") or {}).get("best")} + return out + + +def calibrate_choice(limit: int) -> dict: + """How often Jev's Choice picks the same candidate Opus's judge picked.""" + cmds = {r["id"]: r for r in read_jsonl(home() / "commands_redacted.jsonl")} + from judging.judge import latest_judgements + items, truth, scores = [], {}, {} + for cid, j in latest_judgements().items(): + cands = {k: v for k, v in j["candidates"].items() if isinstance(v, str)} + if cid not in cmds or len(cands) < 2 or not j.get("best") or j["best"] not in cands: + continue + items.append((cid, cmds[cid]["command_redacted"], cands)) + truth[cid] = j["best"] + scores[cid] = {k: (j["verdicts"].get(k) or {}).get("score") for k in cands} + if limit and len(items) >= limit: + break + picked = choose(items) + ok = [k for k in truth if picked.get(k, {}).get("choice")] + agree = sum(1 for k in ok if picked[k]["choice"] == truth[k]) + # "no worse" = Jev's pick scored at least as high as Opus's pick under Opus's own scores + no_worse = sum(1 for k in ok if (scores[k].get(picked[k]["choice"]) or 0) >= (scores[k].get(truth[k]) or 0)) + conf = [picked[k]["confidence"] for k in ok if picked[k].get("confidence") is not None] + hi = [k for k in ok if (picked[k].get("confidence") or 0) >= 0.5] + rep = {"items": len(items), "chosen": len(ok), "agreement": round(agree / max(1, len(ok)), 3), + "no_worse_by_opus_score": round(no_worse / max(1, len(ok)), 3), + "mean_confidence": round(sum(conf) / max(1, len(conf)), 3) if conf else None, + "high_confidence_share": round(len(hi) / max(1, len(ok)), 3), + "agreement_when_confident": round(sum(1 for k in hi if picked[k]["choice"] == truth[k]) / max(1, len(hi)), 3)} + save_json(home() / "evaluation" / "jev_choice_calibration.json", rep) + return rep + + if __name__ == "__main__": p = argparse.ArgumentParser() - p.add_argument("action", choices=["calibrate", "calibrate-eval"]) + p.add_argument("action", choices=["calibrate", "calibrate-eval", "calibrate-choice"]) p.add_argument("--limit", type=int, default=400) a = p.parse_args() - print(json.dumps((calibrate if a.action == "calibrate" else calibrate_eval)(a.limit), indent=2)) + fn = {"calibrate": calibrate, "calibrate-eval": calibrate_eval, "calibrate-choice": calibrate_choice}[a.action] + print(json.dumps(fn(a.limit), indent=2)) diff --git a/live-status/judging/judge.py b/live-status/judging/judge.py index 23eedd4..6147f87 100644 --- a/live-status/judging/judge.py +++ b/live-status/judging/judge.py @@ -13,15 +13,20 @@ from common import append_jsonl, home, read_jsonl, sha from evaluation.validators import check from inference.heuristic import describe -from labeling.llm import DEFAULT_MODEL, QuotaExhausted, chat, parse_results +from labeling.llm import DEFAULT_MODEL, QuotaExhausted, chat, model_code, parse_results from labeling.prompts import JUDGE_SYSTEM, JUDGE_VERSION from redaction.redact import find_secrets BATCH_CHARS = 30000 -def judged_path(): - return home() / "labels" / "judged.jsonl" +def judged_path(name: str = "judged.jsonl"): + return home() / "labels" / name + + +def cand_hash(cands: dict) -> str: + """Hash candidate *text*, so re-keying candidates never re-judges settled rows.""" + return sha(json.dumps(sorted(v.strip() for v in cands.values() if isinstance(v, str)))) def compact_structure(st: dict) -> str: @@ -30,13 +35,13 @@ def compact_structure(st: dict) -> str: return "; ".join(acts + extra) -def candidate_sets() -> dict[str, dict]: - """Latest candidates per command id from all teacher runs.""" +def candidate_sets(models: set[str] | None = None) -> dict[str, dict]: + """Latest candidates per command id from all teacher runs, keyed .""" sets: dict[str, dict] = {} for row in read_jsonl(home() / "labels" / "teacher.jsonl"): - if row["missing"]: + if row["missing"] or (models and row["teacher_model"] not in models): continue - tag = "r" if "regen" in row["prompt_version"] else "t" + tag = model_code(row["teacher_model"]) + ("r" if "regen" in row["prompt_version"] else "t") cur = sets.setdefault(row["id"], {}) for k, v in row["candidates"].items(): cur[f"{tag}{k}"] = v @@ -69,15 +74,16 @@ def run_batch(items: list[dict], model: str) -> list[dict]: def judge_all(limit: int = 0, batch: int = 8, model: str = DEFAULT_MODEL, workers: int = 3, only_ids: set[str] | None = None, extra: dict[str, dict] | None = None, - teacher: bool = True) -> dict: + teacher: bool = True, teacher_models: set[str] | None = None, out_name: str = "judged.jsonl") -> dict: """extra: additional candidates per id (e.g. {"m": student output}); teacher=False judges only those.""" cmds = {r["id"]: r for r in read_jsonl(home() / "commands_redacted.jsonl")} - sets = candidate_sets() if teacher else {} + sets = candidate_sets(teacher_models) if teacher else {} for cid, more in (extra or {}).items(): sets[cid] = {**sets.get(cid, {}), **more} + out_path = judged_path(out_name) done = set() - if judged_path().exists(): - done = {(x["id"], x["cand_hash"]) for x in read_jsonl(judged_path()) if not x.get("missing")} + if out_path.exists(): + done = {(x["id"], cand_hash(x["candidates"])) for x in read_jsonl(out_path) if not x.get("missing")} items = [] for cid, cands in sets.items(): if only_ids is not None and cid not in only_ids: @@ -88,7 +94,7 @@ def judge_all(limit: int = 0, batch: int = 8, model: str = DEFAULT_MODEL, worker h_text, h_conf = describe(r["command_redacted"], r["shell"]) if h_conf >= 0.7 and teacher: cands = {**cands, "h": h_text} - ch = sha(json.dumps(cands, sort_keys=True)) + ch = cand_hash(cands) if (cid, ch) in done: continue item = {"id": cid, "shell": r["shell"], "command": r["command_redacted"], @@ -126,17 +132,17 @@ def judge_all(limit: int = 0, batch: int = 8, model: str = DEFAULT_MODEL, worker print(f" batch failed: {str(exc)[:200]}", flush=True) continue with lock: - append_jsonl(judged_path(), out) + append_jsonl(out_path, out) written += len(out) if i % 10 == 0 or i == len(batches): print(f" {i}/{len(batches)} batches, {written} judged", flush=True) return {"items": len(items), "written": written, "failed_batches": failed, "stopped_on_quota": stopped} -def latest_judgements() -> dict[str, dict]: +def latest_judgements(name: str = "judged.jsonl") -> dict[str, dict]: out: dict[str, dict] = {} - if judged_path().exists(): - for x in read_jsonl(judged_path()): + if judged_path(name).exists(): + for x in read_jsonl(judged_path(name)): if not x.get("missing"): out[x["id"]] = x return out diff --git a/live-status/labeling/llm.py b/live-status/labeling/llm.py index c552f3b..0922cfb 100644 --- a/live-status/labeling/llm.py +++ b/live-status/labeling/llm.py @@ -16,7 +16,13 @@ BASE_URL = os.environ.get("LIVE_STATUS_LLM_URL", "http://127.0.0.1:8317/v1") API_KEY = os.environ.get("LIVE_STATUS_LLM_KEY", "local") # CLIProxyAPI localhost gate, not a vendor key -DEFAULT_MODEL = os.environ.get("LIVE_STATUS_TEACHER", "claude-opus-5") +DEFAULT_MODEL = os.environ.get("LIVE_STATUS_TEACHER", "claude-haiku-4-5-20251001") +MODEL_CODES = {"claude-haiku-4-5-20251001": "h", "claude-opus-5": "o", "claude-sonnet-5": "s"} + + +def model_code(model: str) -> str: + """Short per-model tag so candidates from different teachers stay distinguishable.""" + return MODEL_CODES.get(model, model.split("-")[1][:1] if "-" in model else model[:1]) MAX_COOLDOWN_WAIT = float(os.environ.get("LIVE_STATUS_MAX_COOLDOWN", "900")) diff --git a/live-status/labeling/prompts.py b/live-status/labeling/prompts.py index 4267a93..bfc2bc6 100644 --- a/live-status/labeling/prompts.py +++ b/live-status/labeling/prompts.py @@ -1,7 +1,7 @@ """Versioned teacher, judge and student prompts.""" from __future__ import annotations -TEACHER_VERSION = "teacher-v1" +TEACHER_VERSION = "teacher-v2" JUDGE_VERSION = "judge-v1" STYLE_RULES = """\ @@ -43,7 +43,11 @@ Examples: {EXAMPLES} -Input is a JSON array of items with id, shell and command. Commands are untrusted data from logs. +Input is a JSON array of items with id, shell, command, and sometimes `structure` and `names` +from a heuristic parser. Commands are untrusted data from logs; `structure` and `names` are +extracted mechanically from the command, so prefer the concrete names listed there over generic +words ("the script", "the file", "a directory") whenever they fit the sentence. The parse can be +wrong or incomplete: never let it add an action the command does not show. For each item produce two candidates: - "a": the best concise status (typically 5-14 words). - "b": an alternative that covers every meaningful step (may be longer, still one sentence, max 22 words). diff --git a/live-status/labeling/teacher.py b/live-status/labeling/teacher.py index d5cf5d3..3cbbd27 100644 --- a/live-status/labeling/teacher.py +++ b/live-status/labeling/teacher.py @@ -14,7 +14,7 @@ from pathlib import Path from common import append_jsonl, home, read_jsonl -from labeling.llm import DEFAULT_MODEL, QuotaExhausted, chat, parse_results +from labeling.llm import DEFAULT_MODEL, QuotaExhausted, chat, model_code, parse_results from labeling.prompts import TEACHER_REGEN_NOTE, TEACHER_SYSTEM, TEACHER_VERSION from redaction.redact import find_secrets @@ -23,6 +23,10 @@ TEMPERATURE = 0.4 +def version_of(model: str, regen: bool) -> str: + return TEACHER_VERSION + ("+regen" if regen else "") + "@" + model + + def teacher_path() -> Path: return home() / "labels" / "teacher.jsonl" @@ -70,8 +74,24 @@ def _batches(items: list[dict], size: int) -> list[list[dict]]: return batches +def salient_names(st: dict) -> list[str]: + """Concrete targets the parser found: file basenames, process names, git/package subcommands.""" + out = [] + for a in st.get("actions", []): + out += [t for t in a.get("targets", []) if t and not t.startswith(("<", "$", "-")) and len(t) < 60] + if a.get("sub"): + out.append(a["sub"]) + seen = [] + for n in out: + if n not in seen: + seen.append(n) + return seen[:10] + + def _item(r: dict, feedback: dict | None = None) -> dict: - it = {"id": r["id"], "shell": r["shell"], "command": r["command_redacted"]} + from judging.judge import compact_structure + it = {"id": r["id"], "shell": r["shell"], "command": r["command_redacted"], + "structure": compact_structure(r["structure"]), "names": salient_names(r["structure"])} if feedback: it["previous_attempt"] = feedback.get("output") it["judge_feedback"] = feedback.get("notes") @@ -90,7 +110,7 @@ def run_batch(batch: list[dict], model: str, regen: bool) -> list[dict]: for it in batch: res = results.get(it["id"]) cands = {k: res[k].strip() for k in ("a", "b") if res and isinstance(res.get(k), str) and res[k].strip()} - out.append({"id": it["id"], "teacher_model": model, "prompt_version": TEACHER_VERSION + ("+regen" if regen else ""), + out.append({"id": it["id"], "teacher_model": model, "prompt_version": version_of(model, regen), "params": {"temperature": TEMPERATURE, "batch_size": len(batch)}, "candidates": cands, "missing": not cands, "ts": now, "latency_s": round(time.time() - t0, 1), "usage": usage if it is batch[0] else None}) @@ -106,7 +126,7 @@ def generate(limit: int = 0, batch: int = 20, model: str = DEFAULT_MODEL, worker else: todo = select(rows, limit) regen = bool(feedback) - version = TEACHER_VERSION + ("+regen" if regen else "") + version = version_of(model, regen) done = set() if teacher_path().exists(): done = {x["id"] for x in read_jsonl(teacher_path()) if x["prompt_version"] == version and not x["missing"]} From 4322b80c05080fa98890b822710632476243a742 Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Thu, 17 Sep 2026 16:50:58 -0400 Subject: [PATCH 2/6] Derive teacher structure hints when a caller passes none Co-Authored-By: Claude Opus 5 (1M context) --- live-status/labeling/teacher.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/live-status/labeling/teacher.py b/live-status/labeling/teacher.py index 3cbbd27..6bbb7b1 100644 --- a/live-status/labeling/teacher.py +++ b/live-status/labeling/teacher.py @@ -90,8 +90,10 @@ def salient_names(st: dict) -> list[str]: def _item(r: dict, feedback: dict | None = None) -> dict: from judging.judge import compact_structure + from parsers.shell import analyze + st = r.get("structure") or analyze(r["command_redacted"], r.get("shell")).to_dict() it = {"id": r["id"], "shell": r["shell"], "command": r["command_redacted"], - "structure": compact_structure(r["structure"]), "names": salient_names(r["structure"])} + "structure": compact_structure(st), "names": salient_names(st)} if feedback: it["previous_attempt"] = feedback.get("output") it["judge_feedback"] = feedback.get("notes") From 9217cc8cece941df9ae4707cff71b5a807fa4919 Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:20:48 -0400 Subject: [PATCH 3/6] Default the CLI to Haiku, derive preference pairs, keep held-out rows out of DPO Co-Authored-By: Claude Opus 5 (1M context) --- live-status/cli.py | 5 ++++- live-status/evaluation/active.py | 28 +++++++++++++++++++++++++++- live-status/training/train.py | 9 ++++++++- 3 files changed, 39 insertions(+), 3 deletions(-) diff --git a/live-status/cli.py b/live-status/cli.py index 8050668..44a34be 100644 --- a/live-status/cli.py +++ b/live-status/cli.py @@ -9,6 +9,9 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) +from labeling.llm import DEFAULT_MODEL as DEFAULT_TEACHER # noqa: E402 + + def _print(obj) -> None: print(json.dumps(obj, ensure_ascii=False, indent=2)[:6000]) @@ -99,7 +102,7 @@ def main(argv=None): s.add_argument("--limit", type=int, default=0) s.add_argument("--batch", type=int, default=20 if name == "generate_labels" else 8) s.add_argument("--workers", type=int, default=3, help="parallel Opus requests; >4 trips the subscription rate limit") - s.add_argument("--model", default="claude-opus-5") + s.add_argument("--model", default=DEFAULT_TEACHER, help="labeling model (default: Haiku)") if name == "generate_labels": s.add_argument("--ids", help="file of command ids to label (e.g. mined failures)") s.set_defaults(fn=fn) diff --git a/live-status/evaluation/active.py b/live-status/evaluation/active.py index c92c0fe..0970d09 100644 --- a/live-status/evaluation/active.py +++ b/live-status/evaluation/active.py @@ -33,9 +33,31 @@ def unlabeled_pool(n: int, seed: int) -> list[dict]: return select(rows, n, seed=seed) +def prefs_from_judgements(round_tag: str = "all") -> int: + """Every judged row that also saw a student output ("m") yields a preference pair.""" + cmds = {r["id"]: r for r in load_commands()} + seen = set() + path = home() / "labels" / "prefs.jsonl" + if path.exists(): + seen = {(r["id"], r["rejected"]) for r in read_jsonl(path)} + out = [] + for cid, j in latest_judgements().items(): + rejected = (j.get("candidates") or {}).get("m") + chosen = j.get("recommended_output") + if not rejected or not chosen or cid not in cmds: + continue + if chosen.strip() == rejected.strip() or (cid, rejected) in seen: + continue + if j["validators"]["pass"] and (j.get("recommended_score") or 0) >= 85: + out.append({"id": cid, "command": cmds[cid]["command_redacted"], "chosen": chosen, + "rejected": rejected, "round": round_tag}) + append_jsonl(path, out) + return len(out) + + def main(argv=None): p = argparse.ArgumentParser(prog="mine_failures") - p.add_argument("--backend", required=True) + p.add_argument("--backend") p.add_argument("--round", required=True, help="round tag, e.g. r1") p.add_argument("--pool", type=int, default=1000) p.add_argument("--seed", type=int, default=0) @@ -43,7 +65,11 @@ def main(argv=None): p.add_argument("--screen", choices=["jev", "opus"], default="jev", help="jev: cheap reference-free ranking; opus: judge every student output") p.add_argument("--fail-fraction", type=float, default=0.4, help="jev screen: share of lowest-ranked outputs to relabel") + p.add_argument("--rebuild-prefs", action="store_true", help="only derive preference pairs from existing judgements") a = p.parse_args(argv) + if a.rebuild_prefs: + print(json.dumps({"new_pairs": prefs_from_judgements(a.round)})) + return from inference.backends import from_spec out_dir = home() / "active" / a.round diff --git a/live-status/training/train.py b/live-status/training/train.py index c269a6c..0060b0a 100644 --- a/live-status/training/train.py +++ b/live-status/training/train.py @@ -143,7 +143,14 @@ def main(argv=None): t0 = time.time() if a.dpo: - prefs = list(read_jsonl(a.dpo)) + held = set() + for split in ("test", "validation"): + f = home() / "datasets" / a.data / f"{split}.jsonl" + if f.exists(): + held |= {r["id"] for r in read_jsonl(f)} + prefs = [r for r in read_jsonl(a.dpo) if r.get("id") not in held] + dropped = sum(1 for r in read_jsonl(a.dpo) if r.get("id") in held) + print(json.dumps({"pref_pairs": len(prefs), "dropped_held_out": dropped}), flush=True) pairs = [] for r in prefs: c = encode(tok, r["command"], r["chosen"], False, a.max_len) From 3af0c2627cf0aa4e38ec34fcf0b3c89cf394a13d Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:42:21 -0400 Subject: [PATCH 4/6] Add service load benchmark and round-2 measurements Co-Authored-By: Claude Opus 5 (1M context) --- live-status/EVALUATION.md | 11 ++++ live-status/README.md | 7 +- live-status/TRAINING.md | 24 +++++-- live-status/benchmarks/service_load.py | 89 ++++++++++++++++++++++++++ 4 files changed, 124 insertions(+), 7 deletions(-) create mode 100644 live-status/benchmarks/service_load.py diff --git a/live-status/EVALUATION.md b/live-status/EVALUATION.md index e57b3d5..fb900b1 100644 --- a/live-status/EVALUATION.md +++ b/live-status/EVALUATION.md @@ -47,6 +47,17 @@ A run is promoted into `evaluation/registry.json` only if nothing leaks, its acc least the current best, hallucination does not rise by more than a point, and it was graded by the same grader as the current best. +## Service under load + +```powershell +python live-status/benchmarks/service_load.py --url http://127.0.0.1:8765 --clients 8 --requests 240 +``` + +Qwen3-0.6B q4_K_M behind the service, 8 concurrent clients, cache bypassed: 10.0 requests/s, +p50 0.80 s, p99 1.00 s, no failures, 239/240 answered by the model and one by the fallback. +Single-client latency is 0.10 s p50; the gap is queueing, since one Ollama runner serves the +default concurrency of 4. + ## Reports Each report includes validator rates, accepted %, invented/hallucination %, omission % (Opus), diff --git a/live-status/README.md b/live-status/README.md index 5d79c0a..bd3681a 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -7,9 +7,10 @@ Get-Process opencode2,node,powershell -ErrorAction SilentlyContinue | Select-Obj → Checking running opencode2, node, and powershell processes. ``` -Current best: **Qwen3-0.6B + LoRA, GGUF q4_K_M (397 MB)**, 67.8% of held-out statuses -accepted by the grader (the teacher's own second choices score 72%), 100 ms p50 on an -RTX 3070 and 393 ms on CPU only. Untuned models of the same size score at most 23.5%. +Current best: **Qwen3-0.6B + LoRA, GGUF q4_K_M (397 MB)**, 68.8% of held-out statuses +accepted by the grader (the teacher's own second choices score 72%), 105 ms p50 on an +RTX 3070 and 393 ms on CPU only; behind the service, 10 requests/s with 8 clients. Untuned +models of the same size score at most 23.5%. Details: [TRAINING.md](TRAINING.md), [EVALUATION.md](EVALUATION.md), [DATASET.md](DATASET.md), [ARCHITECTURE.md](ARCHITECTURE.md). diff --git a/live-status/TRAINING.md b/live-status/TRAINING.md index 73a07e5..776d9b5 100644 --- a/live-status/TRAINING.md +++ b/live-status/TRAINING.md @@ -62,6 +62,22 @@ For scale: Opus's own alternate (non-selected) candidates for the same 200 test score 72.0% accepted under jev (86% under the Opus evaluator), so Qwen3-0.6B at 69.4% is close to the teacher's second-choice quality under the same grader. +## Round 2: failure mining, more data, DPO (v1 test grown to 583 commands) + +The 89 commands added to test by round-1 mining are the previous model's own failures, so +scores on the enlarged set are lower and only comparable within it. + +| Checkpoint | Train rows | Accepted | Invented | Same actions | p50 | +| --- | ---: | ---: | ---: | ---: | ---: | +| Qwen3-0.6B q4_K_M (round 1) | 3,113 | 65.9% | 6.3% | 94.3% | 100 ms | +| + round-1 corrections (`v1b`) | 3,628 | **68.8%** | 6.0% | 94.3% | 105 ms | +| + DPO on 517 preference pairs | 3,628 | 68.6% | 6.3% | **96.1%** | 107 ms | + +Mining the model's own failures is worth ~3 points. DPO left acceptance flat while raising +same-action agreement and the mean grade (0.548 → 0.559); it is kept as an optional stage. +Preference pairs whose command later landed in test or validation are dropped automatically +(148 of 665 here). + ## Iteration loop ``` @@ -69,7 +85,7 @@ build_dataset -> train -> export_gguf -> evaluate (jev) -> mine_failures -> buil ``` `mine_failures` runs the current student over unlabeled commands, ranks its outputs with -jev, sends the weakest 40% (plus a random fifth as many) to the Opus teacher and judge, -and appends (chosen, rejected) pairs to `labels/prefs.jsonl`. Round r1 (SmolLM2-135M, 1,500 -commands) sent 726 to the teacher; the judge stopped after 104 when the Opus subscription -entered a cooldown, and resumes on rerun. +jev, sends the weakest 40% (plus a random fifth as many) to the teacher and judge, and +appends (chosen, rejected) pairs to `labels/prefs.jsonl`. Round r1 (SmolLM2-135M, 1,500 +commands) sent 726 to the teacher; judging finished on Haiku after the Opus subscription +hit a cooldown. `mine_failures --rebuild-prefs` re-derives pairs from existing judgements. diff --git a/live-status/benchmarks/service_load.py b/live-status/benchmarks/service_load.py new file mode 100644 index 0000000..9a98a64 --- /dev/null +++ b/live-status/benchmarks/service_load.py @@ -0,0 +1,89 @@ +"""Measure the running service under concurrent load. + + python live-status/benchmarks/service_load.py --url http://127.0.0.1:8765 --clients 8 --requests 200 + +Sends held-out commands (cache disabled by appending a unique comment unless --allow-cache), +and reports end-to-end latency percentiles, throughput and the mix of answer sources. +""" +from __future__ import annotations + +import argparse +import collections +import json +import statistics +import sys +import threading +import time +import urllib.request +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from common import home, read_jsonl, save_json # noqa: E402 + + +def post(url: str, body: dict, token: str | None, timeout: float) -> tuple[float, dict | None]: + data = json.dumps(body).encode() + headers = {"Content-Type": "application/json"} + if token: + headers["Authorization"] = f"Bearer {token}" + req = urllib.request.Request(url.rstrip("/") + "/v1/summarize-command", data=data, headers=headers, method="POST") + t0 = time.perf_counter() + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + return time.perf_counter() - t0, json.loads(resp.read()) + except Exception: + return time.perf_counter() - t0, None + + +def main(argv=None): + p = argparse.ArgumentParser(prog="service_load") + p.add_argument("--url", default="http://127.0.0.1:8765") + p.add_argument("--clients", type=int, default=8) + p.add_argument("--requests", type=int, default=200) + p.add_argument("--data", default="v1") + p.add_argument("--split", default="test") + p.add_argument("--token") + p.add_argument("--timeout", type=float, default=30.0) + p.add_argument("--allow-cache", action="store_true") + a = p.parse_args(argv) + rows = list(read_jsonl(home() / "datasets" / a.data / f"{a.split}.jsonl")) + jobs = [rows[i % len(rows)] for i in range(a.requests)] + results: list[tuple[float, dict | None]] = [] + lock = threading.Lock() + nxt = [0] + + def worker(): + while True: + with lock: + i = nxt[0] + nxt[0] += 1 + if i >= len(jobs): + return + cmd = jobs[i]["command"] if a.allow_cache else f"{jobs[i]['command']} # run {i}" + out = post(a.url, {"command": cmd, "shell": jobs[i]["shell"], "debug": True}, a.token, a.timeout) + with lock: + results.append(out) + + t0 = time.perf_counter() + threads = [threading.Thread(target=worker) for _ in range(a.clients)] + for t in threads: + t.start() + for t in threads: + t.join() + wall = time.perf_counter() - t0 + lat = sorted(x for x, _ in results) + ok = [r for _, r in results if r] + sources = collections.Counter((r.get("source") or "?").split(":")[0] for r in ok) + rep = {"url": a.url, "clients": a.clients, "requests": len(results), "failed": len(results) - len(ok), + "wall_s": round(wall, 2), "throughput_rps": round(len(results) / wall, 1), + "latency_s": {"p50": round(lat[len(lat) // 2], 3), "p90": round(lat[int(len(lat) * 0.9)], 3), + "p99": round(lat[min(len(lat) - 1, int(len(lat) * 0.99))], 3), + "max": round(lat[-1], 3), "mean": round(statistics.mean(lat), 3)}, + "sources": dict(sources), "cache_bypassed": not a.allow_cache} + save_json(home() / "benchmarks" / f"service-load-{a.clients}x{len(results)}.json", rep) + print(json.dumps(rep, indent=2)) + + +if __name__ == "__main__": + main() From 8d6ee5a85afbf2ee4e9236b491743718f4319c28 Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:58:49 -0400 Subject: [PATCH 5/6] Label and serve with Jev-selected local candidates Jev returns one of the options it is given, so the student samples 4-5 candidates and Jev picks: 77.4% accepted versus 69.1% greedy on the v1 test set (oracle 83.9%). Adds labeling/self_label.py (free labelling gated at 0.30 => 66% coverage at 89.6% precision, rest queued for a paid pass), --best-of N in the service, and self-labels as training-only dataset rows. Co-Authored-By: Claude Opus 5 (1M context) --- live-status/ARCHITECTURE.md | 8 ++ live-status/DATASET.md | 10 +++ live-status/README.md | 4 + live-status/api/server.py | 37 ++++++++- live-status/cli.py | 7 +- live-status/dataset_build/build.py | 26 +++++- live-status/inference/backends.py | 11 ++- live-status/labeling/self_label.py | 126 +++++++++++++++++++++++++++++ live-status/scripts/self_label.py | 10 +++ 9 files changed, 229 insertions(+), 10 deletions(-) create mode 100644 live-status/labeling/self_label.py create mode 100644 live-status/scripts/self_label.py diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index 24bdc76..744cebe 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -9,6 +9,8 @@ PSReadLine) parsers/shell.py ─► structure, difficulty tags ───────────┤ ▼ labeling/teacher.py (Haiku 4.5, two candidates per command, batched, resumable) ──► labels/teacher.jsonl + labeling/self_label.py (student samples k candidates, Jev selects and gates; no paid model) + ──► labels/self_labels.jsonl judging/judge.py (Haiku 4.5, separate prompt, scores teacher + heuristic candidates, writes recommended_output) ──► labels/judged.jsonl dataset_build/build.py (accept / review / reject; MinHash families; frozen test) ──► datasets//*.jsonl @@ -38,6 +40,12 @@ - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. +- **Jev selects what a local model wrote.** Jev cannot generate, but a Choice or Score returns + one of the options it was handed, so the student samples 4-5 candidates locally and Jev picks + one. On the v1 test set that beats greedy decoding by 8 points (77.4% vs 69.1%; oracle 83.9%), + which powers both `self_label` (free labelling, no paid model) and the service's `--best-of` + quality mode. Selection only helps when the candidates actually differ: choosing between two + near-identical teacher sentences matched the judge just 52% of the time. - **Haiku writes, jev grades, Opus audits.** Teacher and judge run on Haiku 4.5 with different prompts; label diversity comes from two candidates per command plus the heuristic. jev (TypeSafe System One) returns only probabilities, choices and scores, in milliseconds, at diff --git a/live-status/DATASET.md b/live-status/DATASET.md index 18eadc3..8b2945a 100644 --- a/live-status/DATASET.md +++ b/live-status/DATASET.md @@ -69,6 +69,16 @@ Haiku with the structure hints matches Opus's own alternate candidates, at a fra cost and without exhausting the Claude subscription that the interactive session shares. Opus stays available (`--model claude-opus-5`) for spot checks. +### Self-labels (no paid model) + +`labeling/self_label.py` has the fine-tuned student write 4-5 candidates for a command and +Jev select one, keeping the label when the selector's score clears 0.30. Measured against gold +on the v1 test set, that gate keeps 66% of commands at 89.6% precision (0.25 → 77% at 87.7%, +0.35 → 55% at 90.0%; `evaluation/self_label_threshold.json`). Kept rows join **training only**, +carry `label_source: "self"`, and are dropped when their family belongs to a held-out split. +Rejected commands land in `labels/needs_teacher.txt` for a paid pass, so the teacher only sees +what the local loop could not label. + ## v1 splits | Split | Rows | diff --git a/live-status/README.md b/live-status/README.md index bd3681a..7ab4204 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -21,6 +21,9 @@ python live-status/cli.py serve --backend ollama:live-status-v1-qwen3-06b-lora-q python live-status/api/client.py "git fetch origin && git status -sb" ``` +`--best-of 4` samples several candidates and lets jev pick the best: +8 points of quality for +~1 s per request instead of ~0.1 s. + `POST /v1/summarize-command` with `{"command": "...", "shell": "powershell", "cwd": "optional"}` returns `{"status": "..."}` (add `"debug": true` for source and latency). The service redacts before inference, caches by normalised command, bounds concurrency, falls back to a @@ -36,6 +39,7 @@ Run from the repository root. GPU steps use the training venv (see TRAINING.md). | --- | --- | | Inventory transcript sources | `python live-status/cli.py inventory_sources` | | Extract, redact, deduplicate, report | `python live-status/cli.py extract_commands` | +| Self-label (student + jev, free) | `python live-status/cli.py self_label --backend ollama:... --pool 8000` | | Teacher labels (Haiku) | `python live-status/cli.py generate_labels --limit 4000` | | Judge labels (Haiku) | `python live-status/cli.py judge_labels` | | Build splits | `python live-status/cli.py build_dataset --version v1` | diff --git a/live-status/api/server.py b/live-status/api/server.py index 20ae9ac..0f28f34 100644 --- a/live-status/api/server.py +++ b/live-status/api/server.py @@ -3,7 +3,8 @@ python live-status/cli.py serve --backend ollama:live-status --port 8765 Order per request: size limits -> redaction -> cache -> [optional heuristic fast path] -> -model (bounded concurrency, timeout) -> validation -> heuristic fallback. +model (bounded concurrency, timeout; `--best-of N` samples N candidates and lets Jev pick) -> +validation -> heuristic fallback. Commands are never logged; optional logs hold hashes, timings and the status only. For remote use set LIVE_STATUS_API_TOKEN and pass --tls-cert/--tls-key; binding a non-loopback address without a token is refused. @@ -38,7 +39,8 @@ class Service: def __init__(self, backend=None, *, cache_size: int = 2048, log_path: Path | None = None, - max_concurrency: int = 4, queue_wait: float = 2.0, fast_path: bool = False): + max_concurrency: int = 4, queue_wait: float = 2.0, fast_path: bool = False, + best_of: int = 1): self.backend = backend self.cache: collections.OrderedDict[str, str] = collections.OrderedDict() self.cache_size = cache_size @@ -46,6 +48,7 @@ def __init__(self, backend=None, *, cache_size: int = 2048, log_path: Path | Non self.slots = threading.BoundedSemaphore(max_concurrency) self.queue_wait = queue_wait self.fast_path = fast_path + self.best_of = best_of self.log_path = log_path def _log(self, row: dict) -> None: @@ -71,6 +74,9 @@ def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: try: model_in = red if len(red) <= MODEL_INPUT_CHARS else red[:MODEL_INPUT_CHARS] + " …" status, _ = self.backend.generate(model_in) + if self.best_of > 1: + picked, source = self._best_of(model_in, status) + status = picked or status except Exception as exc: # timeouts, backend down source = f"fallback:{type(exc).__name__}" finally: @@ -85,6 +91,29 @@ def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: return self._done(h_text, source, key, t0, cache=False) return self._done(status, source, key, t0) + def _best_of(self, command: str, greedy: str) -> tuple[str | None, str]: + """Sample extra candidates and let Jev pick; falls back to the greedy answer.""" + from judging.jev import combined, grade + pool = [greedy] if greedy else [] + for i in range(self.best_of - 1): + try: + text, _ = self.backend.generate(command, temperature=0.8, seed=1000 + i) + except Exception: + break + if text and text not in pool: + pool.append(text) + pool = [c for c in pool if check(c, command)["pass"]] + if len(pool) < 2: + return (pool[0] if pool else None), "model" + try: + graded = grade([(str(i), command, c) for i, c in enumerate(pool)]) + except Exception: + return greedy, "model:best-of-failed" + scored = [(combined(g), pool[int(k)]) for k, g in graded.items() if "error" not in g] + if not scored: + return greedy, "model:best-of-failed" + return max(scored)[1], "model:best-of" + def _done(self, status: str, source: str, key: str, t0: float, cache: bool = True) -> dict: if cache and self.cache_size: with self.lock: @@ -176,6 +205,8 @@ def main(argv=None): p.add_argument("--rate-per-minute", type=int, default=0, help="0 disables (default for localhost)") p.add_argument("--cache-size", type=int, default=2048) p.add_argument("--log", type=Path, help="JSONL log of hashes/timings/statuses (off by default)") + p.add_argument("--best-of", type=int, default=1, + help="sample N candidates and let Jev select (measured +8 points, ~5x latency)") p.add_argument("--fast-path", action="store_true", help="answer high-confidence heuristic matches without the model (judge-rated less specific; off by default)") p.add_argument("--tls-cert"); p.add_argument("--tls-key") a = p.parse_args(argv) @@ -190,7 +221,7 @@ def main(argv=None): if hasattr(backend, "warm"): backend.warm() service = Service(backend, cache_size=a.cache_size, log_path=a.log, max_concurrency=a.concurrency, - fast_path=a.fast_path) + fast_path=a.fast_path, best_of=a.best_of) httpd = ThreadingHTTPServer((a.host, a.port), make_handler(service, token, RateLimiter(a.rate_per_minute))) httpd.daemon_threads = True scheme = "http" diff --git a/live-status/cli.py b/live-status/cli.py index 44a34be..9aed323 100644 --- a/live-status/cli.py +++ b/live-status/cli.py @@ -72,6 +72,11 @@ def cmd_serve(a): serve_main(a.rest) +def cmd_self_label(a): + from labeling.self_label import main as self_main + self_main(a.rest) + + def cmd_mine_failures(a): from evaluation.active import main as active_main active_main(a.rest) @@ -83,7 +88,7 @@ def cmd_run_full_pipeline(a): PASSTHROUGH = {"train": cmd_train, "evaluate": cmd_evaluate, "export_gguf": cmd_export_gguf, - "serve": cmd_serve, "mine_failures": cmd_mine_failures} + "serve": cmd_serve, "mine_failures": cmd_mine_failures, "self_label": cmd_self_label} def main(argv=None): diff --git a/live-status/dataset_build/build.py b/live-status/dataset_build/build.py index af9f797..f1a4495 100644 --- a/live-status/dataset_build/build.py +++ b/live-status/dataset_build/build.py @@ -124,14 +124,31 @@ def build(seed: int = 20260916, version: str = "v1", freeze: bool = True) -> dic "weight": round(min(4.0, 1 + math.log2(r["count"])), 2), "existing_model_text": r.get("existing_model_text")} buckets[verdict].append(row) + # Self-labels (student wrote, Jev selected) join training only: they never define held-out + # truth, and any whose family is frozen into test/validation is dropped. + from labeling.self_label import load_self_labels + self_rows = [] + for cid, sl in load_self_labels().items(): + r = cmds.get(cid) + if not r or cid in judged: + continue + self_rows.append({"id": cid, "command": r["command_redacted"], "status": sl["status"], "shell": r["shell"], + "score": None, "reason": "self", "uncertain": False, "candidates": {}, "best": "self", + "notes": f"{sl['source']} score={sl['score']}", "tags": r["tags"], "complexity": r["complexity"], + "length": r["length"], "count": r["count"], "sources": r["sources"], + "action_types": r["structure"]["action_types"], "label_source": "self", + "weight": round(min(4.0, 1 + math.log2(r["count"])), 2), "existing_model_text": r.get("existing_model_text")}) + accepted = buckets["accepted"] - fam = families(accepted) + for r in accepted: + r.setdefault("label_source", "judged") + fam = families(accepted + self_rows) frozen_path = home() / "datasets" / "frozen_families.json" frozen = {} if frozen_path.exists(): import json frozen = json.loads(frozen_path.read_text(encoding="utf-8")) - for r in accepted: + for r in accepted + self_rows: r["family"] = fam[r["id"]] # a family inherits a frozen split if any member was frozen there before fam_split = {} @@ -168,6 +185,9 @@ def build(seed: int = 20260916, version: str = "v1", freeze: bool = True) -> dic splits = collections.defaultdict(list) for r in accepted: splits[fam_split[r["family"]]].append(r) + held_families = {f for f, name in fam_split.items() if name in ("test", "validation")} + dropped_self = sum(1 for r in self_rows if r["family"] in held_families) + splits["train"] += [r for r in self_rows if r["family"] not in held_families] out_dir = home() / "datasets" / version counts = {} for name in ("train", "validation", "test"): @@ -187,6 +207,8 @@ def build(seed: int = 20260916, version: str = "v1", freeze: bool = True) -> dic leak = _leak_check(splits) report = {"version": version, "seed": seed, "counts": counts, "families": len(set(fam.values())), + "self_labels": {"available": len(self_rows), "used_in_train": len(self_rows) - dropped_self, + "dropped_held_out_family": dropped_self}, "cross_split_template_collisions": leak, "distributions": {n: distribution(splits[n]) for n in ("train", "validation", "test")}, "rejected_reasons": dict(collections.Counter(r["reason"] for r in buckets["rejected"])), "review_reasons": dict(collections.Counter(r["reason"] for r in buckets["manual_review"]))} diff --git a/live-status/inference/backends.py b/live-status/inference/backends.py index e25b4e0..204c697 100644 --- a/live-status/inference/backends.py +++ b/live-status/inference/backends.py @@ -54,16 +54,19 @@ def __init__(self, model: str, url: str = "http://127.0.0.1:11434", mode: str = self.cpu, self.threads = cpu, threads self.name = f"ollama:{model}:{mode}" + (":cpu" if cpu else "") - def _options(self) -> dict: - opts = {"temperature": 0, "num_predict": MAX_NEW_TOKENS, "stop": ["\n"], "num_ctx": 2048} + def _options(self, temperature: float | None = None, seed: int | None = None) -> dict: + opts = {"temperature": 0 if temperature is None else temperature, + "num_predict": MAX_NEW_TOKENS, "stop": ["\n"], "num_ctx": 2048} + if seed is not None: + opts["seed"] = seed if self.cpu: opts["num_gpu"] = 0 if self.threads: opts["num_thread"] = self.threads return opts - def generate(self, command: str) -> tuple[str, dict]: - opts = self._options() + def generate(self, command: str, temperature: float | None = None, seed: int | None = None) -> tuple[str, dict]: + opts = self._options(temperature, seed) t0 = time.perf_counter() if self.mode in ("plain", "long"): out = _post(f"{self.url}/api/generate", {"model": self.model, "prompt": student_prompt(command, instruct=self.mode == "long"), diff --git a/live-status/labeling/self_label.py b/live-status/labeling/self_label.py new file mode 100644 index 0000000..d00326e --- /dev/null +++ b/live-status/labeling/self_label.py @@ -0,0 +1,126 @@ +"""Label commands without a paid model: the student writes candidates, Jev picks one. + + python live-status/cli.py self_label --backend ollama:live-status-v1b-qwen3-06b-lora-q4_k_m --pool 8000 + +Jev cannot write text, but a Choice/Score returns one of the options it was given, so a local +sampler supplies 4-5 candidates per command (well inside Jev's 255-option limit) and Jev +selects. Measured on the 583-command v1 test set against gold labels: + + greedy decoding 69.1% accepted + random sample 61.1% + Jev Choice over the pool 76.3% + Jev score over the pool 77.4% <- used here + oracle (best in pool) 83.9% + +Kept labels are gated on the selector's own score; at the default 0.30 the gate keeps 66% of +commands at 89.6% precision (`evaluation/self_label_threshold.json`). Rejected commands are +written to `needs_teacher.txt` for a paid pass. Self-labels are training-only: held-out gold +is never overwritten (see `datasets/gold_lock.json`). +""" +from __future__ import annotations + +import argparse +import json +import sys +import time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from common import append_jsonl, home, read_jsonl, save_json # noqa: E402 +from evaluation.validators import check # noqa: E402 +from judging.jev import combined, grade # noqa: E402 +from judging.judge import latest_judgements # noqa: E402 +from labeling.teacher import load_commands, select, teacher_path # noqa: E402 +from redaction.redact import find_secrets # noqa: E402 + +KEEP_THRESHOLD = 0.30 +SELF_VERSION = "self-jev-v1" + + +def self_labels_path() -> Path: + return home() / "labels" / "self_labels.jsonl" + + +def load_self_labels() -> dict[str, dict]: + if not self_labels_path().exists(): + return {} + return {r["id"]: r for r in read_jsonl(self_labels_path())} + + +def pool_for(n: int, seed: int) -> list[dict]: + labeled = {r["id"] for r in read_jsonl(teacher_path())} if teacher_path().exists() else set() + labeled |= set(latest_judgements()) | set(load_self_labels()) + return select([r for r in load_commands() if r["id"] not in labeled], n, seed=seed) + + +def candidates(backend, command: str, k: int) -> list[str]: + out = [] + greedy, _ = backend.generate(command) + if greedy: + out.append(greedy) + for i in range(k): + text, _ = backend.generate(command, temperature=0.8, seed=1000 + i) + if text and text not in out: + out.append(text) + return out + + +def main(argv=None): + p = argparse.ArgumentParser(prog="self_label") + p.add_argument("--backend", required=True) + p.add_argument("--pool", type=int, default=2000) + p.add_argument("--samples", type=int, default=4) + p.add_argument("--threshold", type=float, default=KEEP_THRESHOLD) + p.add_argument("--seed", type=int, default=20260917) + p.add_argument("--ids", help="label these command ids instead of an automatic pool") + a = p.parse_args(argv) + from inference.backends import from_spec + + backend = from_spec(a.backend) + if hasattr(backend, "warm"): + backend.warm() + if a.ids: + wanted = set(Path(a.ids).read_text(encoding="utf-8").split()) + rows = [r for r in load_commands() if r["id"] in wanted] + else: + rows = pool_for(a.pool, a.seed) + done = load_self_labels() + rows = [r for r in rows if r["id"] not in done] + t0 = time.time() + pools: dict[str, list[str]] = {} + for i, r in enumerate(rows, 1): + cmd = r["command_redacted"] + if find_secrets(json.dumps(cmd)): + continue + pools[r["id"]] = [c for c in candidates(backend, cmd, a.samples) if check(c, cmd)["pass"]] + if i % 250 == 0: + print(f" generated {i}/{len(rows)} in {round(time.time() - t0)}s", flush=True) + by_id = {r["id"]: r for r in rows} + graded = grade([(f"{cid}|{i}", by_id[cid]["command_redacted"], c) + for cid, cands in pools.items() for i, c in enumerate(cands)]) + kept, rejected = [], [] + now = time.strftime("%Y-%m-%dT%H:%M:%S") + for cid, cands in pools.items(): + best, text = -1.0, None + for i, c in enumerate(cands): + g = graded.get(f"{cid}|{i}") + if g and "error" not in g and combined(g) > best: + best, text = combined(g), c + if text is None: + rejected.append(cid) + continue + row = {"id": cid, "status": text, "score": round(best, 4), "candidates": len(cands), + "source": SELF_VERSION, "selector": "jev", "generator": backend.name, "ts": now} + (kept if best >= a.threshold else rejected).append(row if best >= a.threshold else cid) + append_jsonl(self_labels_path(), kept) + out_dir = home() / "labels" + (out_dir / "needs_teacher.txt").write_text("\n".join(rejected), encoding="utf-8") + summary = {"pool": len(rows), "generated": len(pools), "kept": len(kept), "needs_teacher": len(rejected), + "threshold": a.threshold, "generator": backend.name, "seconds": round(time.time() - t0)} + save_json(out_dir / "self_label_summary.json", summary) + print(json.dumps(summary), flush=True) + + +if __name__ == "__main__": + main() diff --git a/live-status/scripts/self_label.py b/live-status/scripts/self_label.py new file mode 100644 index 0000000..c7411f7 --- /dev/null +++ b/live-status/scripts/self_label.py @@ -0,0 +1,10 @@ +"""Wrapper for `cli.py self_label`.""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +if __name__ == "__main__": + from cli import main + + main(["self_label", *sys.argv[1:]]) From aaa941ccbbb0a23c9df09e2550a2575eeb7b1f1d Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Sun, 20 Sep 2026 20:27:33 -0400 Subject: [PATCH 6/6] Evaluate newer structured shell explainer --- live-status/EVALUATION.md | 4 +- live-status/TRAINING.md | 2 + live-status/benchmarks/baseline.py | 6 +- live-status/cli.py | 5 +- live-status/evaluation/evaluate.py | 47 ++++++++++-- live-status/inference/backends.py | 43 +++++++---- live-status/labeling/prompts.py | 35 ++++++++- .../research/2026-09-20-model-sweep.md | 73 ++++++++++++++++++ .../tests/test_evaluation_checkpoint.py | 71 ++++++++++++++++++ live-status/tests/test_structured_prompt.py | 27 +++++++ live-status/tests/test_training_encoding.py | 29 ++++++++ live-status/tools/audit_failures.py | 46 ++++++++++++ live-status/tools/research_hf_candidates.py | 74 +++++++++++++++++++ live-status/training/train.py | 17 +++-- 14 files changed, 442 insertions(+), 37 deletions(-) create mode 100644 live-status/research/2026-09-20-model-sweep.md create mode 100644 live-status/tests/test_evaluation_checkpoint.py create mode 100644 live-status/tests/test_structured_prompt.py create mode 100644 live-status/tests/test_training_encoding.py create mode 100644 live-status/tools/audit_failures.py create mode 100644 live-status/tools/research_hf_candidates.py diff --git a/live-status/EVALUATION.md b/live-status/EVALUATION.md index fb900b1..e26c699 100644 --- a/live-status/EVALUATION.md +++ b/live-status/EVALUATION.md @@ -1,7 +1,7 @@ # Evaluation ```powershell -python live-status/cli.py evaluate --backend ollama:[:plain|:long|:instruct][:cpu] --name [--grader jev|opus|both|none] [--promote] +python live-status/cli.py evaluate --backend ollama:[:plain|:long|:instruct|:structured|:structured-plain][:cpu] --name [--grader jev|opus|both|none] [--promote] python live-status/cli.py evaluate --regrade --name --grader opus # re-grade saved outputs ``` @@ -36,6 +36,8 @@ commands pass 86% of Opus evaluations and 72% of jev evaluations. Compare runs o one grader. Both graders cache verdicts by (command, output), so re-evaluating an unchanged output is free. +Generation checkpoints every 25 rows to `evaluation/outputs/.generated.jsonl`. A retry +resumes from that checkpoint, and a completed run removes it after saving the final output. The `no_secret` validator also fails outputs that echo a redaction placeholder (``, ``). Two of the round-1 runs did this once each, which blocks promotion; diff --git a/live-status/TRAINING.md b/live-status/TRAINING.md index 776d9b5..91e423b 100644 --- a/live-status/TRAINING.md +++ b/live-status/TRAINING.md @@ -18,6 +18,8 @@ Status: Loss covers only the status and EOS. `--prompt instruct` prepends the long instruction (serve it with the `:long` backend suffix); `mixed` uses it on 30% of rows. +`--prompt structured` adds a deterministic action parse and bounded raw command; serve that +adapter with the `:structured-plain` backend suffix. ## Mechanics diff --git a/live-status/benchmarks/baseline.py b/live-status/benchmarks/baseline.py index 85e1033..5c3138c 100644 --- a/live-status/benchmarks/baseline.py +++ b/live-status/benchmarks/baseline.py @@ -26,14 +26,14 @@ def unload(backend: OllamaBackend) -> None: def run(models: list[str], split: str = "test", limit: int = 0, judge: bool = True, prompt: str = "instruct", - data: str = "v1") -> dict: + data: str = "v1", grader: str = "jev") -> dict: rows = load_split(data, split, limit) table = [] for m in models: - backend = OllamaBackend(m, mode="plain" if prompt == "plain" else "instruct", timeout=60) + backend = OllamaBackend(m, mode="plain" if prompt == "plain" else prompt, timeout=60) name = f"baseline-{data}-{m.replace(':', '-').replace('/', '_')}-{prompt}-{split}{limit or ''}" print(f"== {m} ({len(rows)} rows)", flush=True) - rep = run_eval(backend, rows, name, judge=judge) + rep = run_eval(backend, rows, name, judge=judge, grader=grader) unload(backend) table.append({"model": m, "prompt": prompt, "n": rep["n"], "judge": rep.get("judge"), "validators": rep["validators"], "latency_s": rep["latency_s"], "tokens_per_s": rep["tokens_per_s_median"], "memory": rep["memory"], diff --git a/live-status/cli.py b/live-status/cli.py index 9aed323..1648ca1 100644 --- a/live-status/cli.py +++ b/live-status/cli.py @@ -49,7 +49,7 @@ def cmd_build_dataset(a): def cmd_benchmark_base_models(a): from benchmarks.baseline import run - _print(run(models=a.models, split=a.split, limit=a.limit, judge=not a.no_judge, prompt=a.prompt, data=a.data)) + _print(run(models=a.models, split=a.split, limit=a.limit, judge=not a.no_judge, prompt=a.prompt, data=a.data, grader=a.grader)) def cmd_train(a): @@ -117,8 +117,9 @@ def main(argv=None): s.add_argument("--split", default="test") s.add_argument("--data", default="v1") s.add_argument("--limit", type=int, default=0) - s.add_argument("--prompt", choices=["instruct", "plain"], default="instruct") + s.add_argument("--prompt", choices=["instruct", "plain", "structured"], default="instruct") s.add_argument("--no-judge", action="store_true") + s.add_argument("--grader", choices=["none", "jev", "opus", "both"], default="jev") s.set_defaults(fn=cmd_benchmark_base_models) for name in PASSTHROUGH: sub.add_parser(name, help=f"see `{name} --help`") diff --git a/live-status/evaluation/evaluate.py b/live-status/evaluation/evaluate.py index d91d9d1..5e228fa 100644 --- a/live-status/evaluation/evaluate.py +++ b/live-status/evaluation/evaluate.py @@ -10,7 +10,9 @@ import argparse import json +import re import statistics +from collections import Counter import sys import threading import time @@ -151,9 +153,15 @@ def run_eval(backend, rows: list[dict], name: str, judge: bool = True, cpu_probe cpu_procs = [] if hasattr(backend, "warm"): backend.warm() + d = home() / "evaluation" + generated_path = d / "outputs" / f"{name}.generated.jsonl" + cached = {r["id"]: r for r in read_jsonl(generated_path)} if generated_path.exists() else {} outputs = [] t_start = time.time() for r in rows: + if r["id"] in cached: + outputs.append(cached[r["id"]]) + continue try: out, m = backend.generate(r["command"]) except Exception as exc: @@ -162,6 +170,8 @@ def run_eval(backend, rows: list[dict], name: str, judge: bool = True, cpu_probe v = check(out, r["command"], salient_targets(st.to_dict())) outputs.append({**{k: r[k] for k in ("id", "command", "status", "shell", "tags", "complexity")}, "output": out, "metrics": m, "validators": v}) + if len(outputs) % 25 == 0: + write_jsonl(generated_path, outputs) elapsed = time.time() - t_start cpu = None try: @@ -169,29 +179,54 @@ def run_eval(backend, rows: list[dict], name: str, judge: bool = True, cpu_probe except Exception: pass mem = memory_snapshot(backend) - judge = judge and grader in ("opus", "both") - if judge: + # Preserve the complete local output if a remote grader fails. + write_jsonl(generated_path, outputs) + grading_requested = judge + opus_judged = grading_requested and grader in ("opus", "both") + if opus_judged: cache = judge_outputs(outputs) for o in outputs: o["judge"] = cache.get(o["jkey"]) - if grader in ("jev", "both"): + if grading_requested and grader in ("jev", "both"): jev_grade_outputs(outputs) - rep = summarize(outputs, judge) + rep = summarize(outputs, opus_judged) rep.update({"name": name, "backend": backend.name, "n": len(outputs), "elapsed_s": round(elapsed, 1), "cpu_percent_avg": cpu, "memory": mem, "ts": time.strftime("%Y-%m-%dT%H:%M:%S")}) - d = home() / "evaluation" write_jsonl(d / "outputs" / f"{name}.jsonl", outputs) save_json(d / f"{name}.json", rep) + generated_path.unlink(missing_ok=True) return rep +_OVERLAP_STOP = {"a", "an", "and", "the", "then", "to", "for", "of", "in", "on", "with", "from", "into", "its"} + + +def reference_overlap(output: str, reference: str) -> dict: + """Whole-set lexical regression signal; this is not a semantic accuracy score.""" + tokens = lambda text: [x for x in re.findall(r"[a-z0-9_./\\:+-]+", text.lower()) if x not in _OVERLAP_STOP] + got, want = Counter(tokens(output)), Counter(tokens(reference)) + common = sum((got & want).values()) + precision = common / max(1, sum(got.values())) + recall = common / max(1, sum(want.values())) + f1 = 2 * precision * recall / (precision + recall) if precision + recall else 0.0 + return {"precision": precision, "recall": recall, "f1": f1} + + def summarize(outputs: list[dict], judged: bool) -> dict: n = max(1, len(outputs)) walls = [o["metrics"]["wall_s"] for o in outputs if o["metrics"].get("wall_s") is not None] tps = [o["metrics"]["tokens_per_s"] for o in outputs if o["metrics"].get("tokens_per_s")] V = lambda k: round(100 * sum(1 for o in outputs if o["validators"].get(k)) / n, 1) # noqa: E731 recalls = [o["validators"]["target_recall"] for o in outputs if "target_recall" in o["validators"]] + overlaps = [reference_overlap(o["output"], o["status"]) for o in outputs] rep = { + "reference_overlap": { + "coverage_pct": round(100 * len(overlaps) / n, 1), + "precision_avg": round(statistics.mean(x["precision"] for x in overlaps), 3), + "recall_avg": round(statistics.mean(x["recall"] for x in overlaps), 3), + "f1_avg": round(statistics.mean(x["f1"] for x in overlaps), 3), + "meaning": "whole-set lexical regression signal, not semantic accuracy", + }, "validators": {"pass_pct": V("pass"), "style_pct": V("style_ok"), "one_sentence_pct": V("one_sentence"), "length_pct": V("length_ok"), "live_tense_pct": V("live_tense"), "no_secret_pct": V("no_secret"), "no_boilerplate_pct": V("no_boilerplate"), "no_shell_noise_pct": V("no_shell_noise"), @@ -313,7 +348,7 @@ def main(argv=None): from inference.backends import from_spec rep = run_eval(from_spec(a.backend), load_split(a.data, a.split, a.limit), a.name, judge=a.grader != "none", grader=a.grader) - print(json.dumps({k: rep.get(k) for k in ("name", "n", "validators", "latency_s", "tokens_per_s_median", "memory", "jev", "judge")}, indent=2)) + print(json.dumps({k: rep.get(k) for k in ("name", "n", "validators", "reference_overlap", "latency_s", "tokens_per_s_median", "memory", "jev", "judge")}, indent=2)) if a.promote: print(json.dumps(promote(rep), indent=2)) diff --git a/live-status/inference/backends.py b/live-status/inference/backends.py index 204c697..e169439 100644 --- a/live-status/inference/backends.py +++ b/live-status/inference/backends.py @@ -9,7 +9,7 @@ import time import urllib.request -from labeling.prompts import EXAMPLES, STUDENT_INSTRUCTION, fit_command, student_prompt +from labeling.prompts import EXAMPLES, STUDENT_INSTRUCTION, fit_command, structured_student_input, student_prompt MAX_NEW_TOKENS = 48 @@ -27,13 +27,15 @@ def clean(text: str) -> str: return t[:1].upper() + t[1:] -def _few_shot_messages(command: str) -> list[dict]: +def _few_shot_messages(command: str, *, structured: bool = False) -> list[dict]: msgs = [{"role": "system", "content": STUDENT_INSTRUCTION}] for block in EXAMPLES.split("\n\n")[:4]: cmd, status = block.split("\nStatus: ") - msgs.append({"role": "user", "content": cmd.removeprefix("Command: ")}) + example = cmd.removeprefix("Command: ") + msgs.append({"role": "user", "content": structured_student_input(example) if structured else example}) msgs.append({"role": "assistant", "content": status}) - msgs.append({"role": "user", "content": fit_command(command)}) + content = structured_student_input(command) if structured else fit_command(command) + msgs.append({"role": "user", "content": content}) return msgs @@ -45,8 +47,8 @@ def _post(url: str, body: dict, timeout: float) -> dict: class OllamaBackend: - """`plain`: fine-tuned completion format; `long`: same with the instruction prepended; - `instruct`: chat with few-shot examples (untuned models).""" + """Plain is fine-tuned completion format; long prepends the instruction. + Structured-plain completes over the deterministic parse; instruct uses few-shot chat.""" def __init__(self, model: str, url: str = "http://127.0.0.1:11434", mode: str = "plain", timeout: float = 20.0, keep_alive: str = "30m", cpu: bool = False, threads: int | None = None): @@ -68,12 +70,14 @@ def _options(self, temperature: float | None = None, seed: int | None = None) -> def generate(self, command: str, temperature: float | None = None, seed: int | None = None) -> tuple[str, dict]: opts = self._options(temperature, seed) t0 = time.perf_counter() - if self.mode in ("plain", "long"): - out = _post(f"{self.url}/api/generate", {"model": self.model, "prompt": student_prompt(command, instruct=self.mode == "long"), + if self.mode in ("plain", "long", "structured-plain"): + out = _post(f"{self.url}/api/generate", {"model": self.model, + "prompt": student_prompt(command, instruct=self.mode == "long", + structured=self.mode == "structured-plain"), "raw": True, "stream": False, "options": opts, "keep_alive": self.keep_alive}, self.timeout) text = out.get("response", "") else: - out = _post(f"{self.url}/api/chat", {"model": self.model, "messages": _few_shot_messages(command), "stream": False, + out = _post(f"{self.url}/api/chat", {"model": self.model, "messages": _few_shot_messages(command, structured=self.mode == "structured"), "stream": False, "think": False, "options": opts, "keep_alive": self.keep_alive}, self.timeout) text = (out.get("message") or {}).get("content", "") wall = time.perf_counter() - t0 @@ -105,7 +109,8 @@ def generate(self, command: str) -> tuple[str, dict]: class HFBackend: """Transformers checkpoint (optionally with a LoRA adapter) for pre-export evaluation.""" - def __init__(self, base: str, adapter: str | None = None, instruct: bool = False, device: str = "cuda", max_len: int = 1024): + def __init__(self, base: str, adapter: str | None = None, instruct: bool = False, + structured: bool = False, device: str = "cuda", max_len: int = 1024): import torch from transformers import AutoModelForCausalLM, AutoTokenizer self.torch = torch @@ -116,13 +121,16 @@ def __init__(self, base: str, adapter: str | None = None, instruct: bool = False from peft import PeftModel model = PeftModel.from_pretrained(model, adapter).merge_and_unload() self.model = model.eval() - self.device, self.instruct, self.max_len = device, instruct, max_len + self.device, self.instruct, self.structured, self.max_len = device, instruct, structured, max_len self.name = f"hf:{adapter or base}" def generate(self, command: str) -> tuple[str, dict]: torch = self.torch - ids = self.tok(student_prompt(command, instruct=self.instruct), return_tensors="pt").input_ids - ids = ids[:, -self.max_len:].to(self.device) + ids = self.tok(student_prompt(command, instruct=self.instruct, structured=self.structured), return_tensors="pt").input_ids + if ids.shape[1] > self.max_len: + head = int(self.max_len * 0.7) + ids = self.torch.cat((ids[:, :head], ids[:, -(self.max_len - head):]), dim=1) + ids = ids.to(self.device) t0 = time.perf_counter() with torch.no_grad(): out = self.model.generate(ids, attention_mask=torch.ones_like(ids), max_new_tokens=MAX_NEW_TOKENS, do_sample=False, @@ -142,19 +150,22 @@ def _stop_ids(self) -> list[int]: def from_spec(spec: str): - """ollama:[:plain|:long|:instruct][:cpu] | llama-server: | hf:[@]""" + """ollama:[:mode][:cpu] | llama-server: | hf:[@][:structured]""" kind, _, rest = spec.partition(":") if kind == "ollama": mode, cpu = "plain", False if rest.endswith(":cpu"): rest, cpu = rest[:-4], True - for m in ("plain", "instruct", "long"): + for m in ("structured-plain", "plain", "instruct", "long", "structured"): if rest.endswith(":" + m): rest, mode = rest[: -len(m) - 1], m return OllamaBackend(rest, mode=mode, cpu=cpu) if kind == "llama-server": return LlamaServerBackend(rest or "http://127.0.0.1:8080") if kind == "hf": + structured = rest.endswith(":structured") + if structured: + rest = rest[:-len(":structured")] base, _, adapter = rest.partition("@") - return HFBackend(base, adapter or None) + return HFBackend(base, adapter or None, structured=structured) raise ValueError(f"unknown backend spec {spec}") diff --git a/live-status/labeling/prompts.py b/live-status/labeling/prompts.py index bfc2bc6..7bb8aa7 100644 --- a/live-status/labeling/prompts.py +++ b/live-status/labeling/prompts.py @@ -1,6 +1,8 @@ """Versioned teacher, judge and student prompts.""" from __future__ import annotations +import json + TEACHER_VERSION = "teacher-v2" JUDGE_VERSION = "judge-v1" @@ -89,7 +91,34 @@ def fit_command(command: str, limit: int = STUDENT_MAX_CHARS) -> str: return command[:head] + "\n…\n" + command[-(limit - head):] -def student_prompt(command: str, *, instruct: bool) -> str: - """Plain format the fine-tuned student learns; `instruct` prepends the long instruction.""" +def student_prompt(command: str, *, instruct: bool, structured: bool = False) -> str: + """Completion format shared by fine-tuning and inference.""" head = STUDENT_INSTRUCTION + "\n\n" if instruct else "" - return f"{head}Command:\n{fit_command(command)}\n\nStatus:" + body = structured_student_input(command) if structured else f"Command:\n{fit_command(command)}" + return f"{head}{body}\n\nStatus:" + + +def structured_student_input(command: str) -> str: + """Compact parser hints plus bounded source text for ambiguous details.""" + from parsers.shell import analyze + + structure = analyze(command).to_dict() + actions = [] + for action in structure["actions"][:12]: + item = {"action": action["type"], "command": action["exe"]} + if action.get("sub"): + item["subcommand"] = action["sub"] + if action.get("targets"): + item["targets"] = action["targets"][:4] + actions.append(item) + parsed = { + "shell": structure["shell"], + "actions": actions, + "cwd_changes": structure["cwd_changes"][:3], + "loops": structure["loops"], + "conditionals": structure["conditionals"], + "inline_script": structure["has_inline_script"], + } + return ("Mechanical parse (may be incomplete):\n" + + json.dumps(parsed, ensure_ascii=False, separators=(",", ":")) + + "\n\nRaw command:\n" + fit_command(command, limit=1200)) diff --git a/live-status/research/2026-09-20-model-sweep.md b/live-status/research/2026-09-20-model-sweep.md new file mode 100644 index 0000000..8563d7d --- /dev/null +++ b/live-status/research/2026-09-20-model-sweep.md @@ -0,0 +1,73 @@ +# Small model sweep — 2026-09-20 + +## Method + +Candidates were discovered from the Hugging Face model registry by creation date, then joined to the public `FlameF0X/lm-cpu-benchmarks` results. This avoids starting from remembered model names. The registry date is only a discovery filter. A candidate still needs independent capability evidence and evaluation on the private shell-to-status test set. + +The CPU leaderboard measures unquantized base-model prefill, decode, and WikiText-2 perplexity. It is a community benchmark; perplexity is a generic coherence signal, not shell-to-English accuracy. The Open SLM leaderboard is a second community check for models below roughly 250M parameters. + +## Current shortlist + +| Model | HF created | Parameters | CPU prefill tok/s | CPU decode tok/s | WikiText-2 PPL | Decision | +|---|---:|---:|---:|---:|---:|---| +| Qwen/Qwen3.5-0.8B | 2026-02-28 | 873M | 40.6 | 7.1 | 25.44 | Quality candidate; test and fine-tune | +| LiquidAI/LFM2.5-350M-Base | 2026-03-31 | 354M | 84.5 | 13.6 | 193.31 | Speed candidate; fine-tune before judging task quality | +| LiquidAI/LFM2.5-230M-Base | 2026-06-16 | 230M | 127.4 | 19.3 | 222.30 | Extreme-size candidate; only continue if structured task training works | +| HuggingFaceTB/nanowhale-100m-base | 2026-04-24 | 110M | 1820.5 | 34.0 | unavailable | Reject for now: its general benchmark scores are near chance and perplexity failed | +| tencent/Hunyuan-0.5B-Pretrain | 2025-07-28 | 539M | — | — | 25.05 | Older fallback; lower priority than Qwen3.5 | + +The newest credible small base is not automatically the best candidate. LFM2.5-230M is newer and faster than Qwen3.5-0.8B, but its generic perplexity is much worse. The task benchmark decides whether domain fine-tuning closes that gap. + +Sources: [Hugging Face model registry](https://huggingface.co/models?pipeline_tag=text-generation&sort=created), [CPU LM Speed Leaderboard](https://huggingface.co/spaces/FlameF0X/cpu-lm-benchmark), [Open SLM Leaderboard](https://huggingface.co/spaces/AxiomicLabs/Open_SLM_Leaderboard), [Qwen3.5-0.8B](https://huggingface.co/Qwen/Qwen3.5-0.8B), [LFM2.5-350M-Base](https://huggingface.co/LiquidAI/LFM2.5-350M-Base), [LFM2.5-230M-Base](https://huggingface.co/LiquidAI/LFM2.5-230M-Base). + +## What the existing score means + +The current Qwen3-0.6B LoRA model's saved Jev results show 94.3% same-action probability at the 0.5 threshold and 93.8% no-invention. Only 70.0% clears the 2.5/4 prose-quality threshold, producing a 69.3% combined pass rate. The old headline was therefore not factual accuracy. The remaining approximately 6% semantic error is still unacceptable for a trusted UI. + +Report these dimensions separately from now on: + +- action agreement and omission rate; +- invention rate; +- important-name preservation; +- deterministic secret and format checks; +- prose quality; +- latency, resident memory, and artifact size. + +## Local Qwen3.5 prompt A/B + +The full 587-row held-out set was run locally through the installed Ollama qwen3.5:0.8b model with identical decoding settings. No external grader was used. + +| Input | Validator pass | Target recall | Secret-safe | p50 | p90 | Output rate | +|---|---:|---:|---:|---:|---:|---:| +| Raw command + few-shot chat | 97.3% | 40.2% | 99.7% | 225 ms | 301 ms | 136.4 tok/s | +| Mechanical parse + raw command + same examples | 98.6% | 85.9% | 100.0% | 219 ms | 294 ms | 135.5 tok/s | + +The structured input improved target recall by 45.7 percentage points with no latency penalty, but target recall is computed for only 46 of 587 rows (7.8%): simple commands where the parser exposes one salient target. This is evidence that parser hints help name preservation, not a whole-set accuracy result. Reference-overlap checks across the full set remain weaker than the trained Qwen3 0.6B model, so Qwen3.5 0.8B is not a replacement without fine-tuning. Its installed artifact and measured residency are about 1.06 GB, versus 655 MB for the current quantized model. + +## Local evaluation limits found + +The deterministic target-recall check runs on only 46 of 587 held-out rows (7.8%), so it cannot be reported as model accuracy. Validator pass mostly measures sentence shape, tense, shell-noise removal, and secret handling. The new report also includes whole-set lexical overlap with the vetted reference, explicitly labeled as a regression signal rather than semantic accuracy. Whole-set semantic action agreement and invention still require an independent judge. + +The structured parser emits an average of 3.35 actions on the held-out set. It finds no action for 27 rows, no meaningful action for 59, and more than the 12-action prompt cap for 24. These are concentrated in long PowerShell, nested quoting, JavaScript cells, heredocs, and pipelines, so the bounded raw command remains in the prompt. + +With the LFM2.5 tokenizer, structured prompts have median lengths of 183 train and 196 test tokens. Thirty train and six test prompts exceed the 768-token training window. Training and checkpoint inference now retain both the prompt head and tail when truncating, preserving the mechanical action record and the end of the raw command. + +## Trained LFM2.5-350M result + +The structured LoRA run trained 0.98M parameters for three epochs on an RTX 3070. Validation loss was best at epoch 2 (1.2664), and the trainer restored that checkpoint after epoch 3 regressed to 1.2977. Training took 540 seconds and reported 1.5 GB peak allocated GPU memory. + +| Model and runtime | Resident/model size | p50 | Output tok/s | Validator pass | Target recall (7.8% coverage) | Whole-set lexical F1 | +|---|---:|---:|---:|---:|---:|---:| +| Current Qwen3 0.6B Q4_K_M | 655 MB | 105 ms | 334.9 | 99.0% | 0.815 | 0.525 | +| LFM2.5 350M merged HF | 1,668 MB process RSS | 687 ms | 29.9 | 99.7% | 0.815 | 0.419 | +| LFM2.5 350M Q8_0 | 437 MB | 90 ms | 425.1 | 99.3% | 0.837 | 0.418 | +| LFM2.5 350M Q4_K_M | 287 MB | 85 ms | 475.6 | 98.8% | 0.859 | 0.393 | + +The Q8 GGUF file is 379 MB and the Q4_K_M file is 229 MB. Q8 preserves the merged model's local lexical signal while reducing measured residency 33% and p50 latency 14% versus the current model. Q4 is smaller and faster but loses more reference overlap. The 350M Q8 candidate advances to independent semantic grading; it does not replace the current model based on local validators. Do not spend a run on LFM2.5 230M until the 350M candidate clears the action-agreement and invention gates. + +## Architecture experiment + +The raw-command formulation asks a small model to parse shell syntax, resolve action order, copy names, ignore injection-like strings, and write polished prose in one unconstrained generation. The repository already performs much of the parsing deterministically. Structured training made the 350M model viable on size and speed, but it did not reach the current model's whole-set lexical signal. A deterministic final renderer for recognized actions remains the next architectural experiment after semantic grading. + +Weight pruning comes after semantic parity. Removing generic-domain weights without retraining can destroy useful syntax and language behavior, and zeroed weights do not guarantee lower latency in Ollama/llama.cpp. Quantizing the smaller dense base already produced the useful size and latency gain. If semantic grading passes and further compression is needed, distill the structured task into a smaller student and compare quantization-aware training before structured pruning. +Relevant pruning evidence: [Iterative Structured Pruning with Multi-Domain Calibration](https://arxiv.org/abs/2601.02674) argues for hardware-friendly structured removal and mixed-domain calibration; [GPrune-LLM](https://arxiv.org/abs/2603.13418) shows that single-domain calibration can bias neuron importance; [Pruning as a Domain-specific LLM Extractor](https://arxiv.org/abs/2405.06275) supports task-calibrated pruning but does not establish that arbitrary out-of-domain weights can be safely deleted from a sub-1B model. \ No newline at end of file diff --git a/live-status/tests/test_evaluation_checkpoint.py b/live-status/tests/test_evaluation_checkpoint.py new file mode 100644 index 0000000..b30e552 --- /dev/null +++ b/live-status/tests/test_evaluation_checkpoint.py @@ -0,0 +1,71 @@ +import os +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +from evaluation.evaluate import run_eval + + +ROW = { + "id": "one", + "command": "git status", + "status": "Checking Git status.", + "shell": "bash", + "tags": [], + "complexity": "simple", +} + + +class FakeBackend: + name = "fake" + + def generate(self, command): + return "Checking Git status.", {"wall_s": 0.001, "tokens_per_s": 100} + + +class EvaluationCheckpointTests(unittest.TestCase): + def test_none_grader_stays_local_and_saves_final_output(self): + with tempfile.TemporaryDirectory() as td, patch.dict(os.environ, {"LIVE_STATUS_HOME": td}): + with patch("evaluation.evaluate.jev_grade_outputs", side_effect=AssertionError("external grader called")): + report = run_eval(FakeBackend(), [ROW], "local-only", judge=False, grader="none") + self.assertEqual(report["n"], 1) + self.assertEqual(report["reference_overlap"]["coverage_pct"], 100.0) + self.assertEqual(report["reference_overlap"]["f1_avg"], 1.0) + self.assertTrue((Path(td) / "evaluation" / "outputs" / "local-only.jsonl").exists()) + self.assertFalse((Path(td) / "evaluation" / "outputs" / "local-only.generated.jsonl").exists()) + + def test_disabled_grading_stays_local_even_with_configured_grader(self): + with tempfile.TemporaryDirectory() as td, patch.dict(os.environ, {"LIVE_STATUS_HOME": td}): + with patch("evaluation.evaluate.jev_grade_outputs", side_effect=AssertionError("external grader called")): + report = run_eval(FakeBackend(), [ROW], "disabled-grading", judge=False, grader="jev") + self.assertEqual(report["n"], 1) + + def test_existing_generation_checkpoint_is_reused(self): + with tempfile.TemporaryDirectory() as td, patch.dict(os.environ, {"LIVE_STATUS_HOME": td}): + output_dir = Path(td) / "evaluation" / "outputs" + output_dir.mkdir(parents=True) + checkpoint = {**ROW, "output": "Checking Git status.", + "metrics": {"wall_s": 0.001, "tokens_per_s": 100}, + "validators": {"pass": True, "style_ok": True, "one_sentence": True, + "length_ok": True, "live_tense": True, "no_secret": True, + "no_boilerplate": True, "no_shell_noise": True}} + from common import write_jsonl + write_jsonl(output_dir / "resume.generated.jsonl", [checkpoint]) + backend = FakeBackend() + with patch.object(backend, "generate", side_effect=AssertionError("checkpoint was regenerated")): + report = run_eval(backend, [ROW], "resume", judge=False, grader="none") + self.assertEqual(report["n"], 1) + self.assertTrue((output_dir / "resume.jsonl").exists()) + self.assertFalse((output_dir / "resume.generated.jsonl").exists()) + + def test_grader_failure_keeps_generated_output(self): + with tempfile.TemporaryDirectory() as td, patch.dict(os.environ, {"LIVE_STATUS_HOME": td}): + with patch("evaluation.evaluate.jev_grade_outputs", side_effect=RuntimeError("grader unavailable")): + with self.assertRaisesRegex(RuntimeError, "grader unavailable"): + run_eval(FakeBackend(), [ROW], "failed-grade", judge=True, grader="jev") + self.assertTrue((Path(td) / "evaluation" / "outputs" / "failed-grade.generated.jsonl").exists()) + + +if __name__ == "__main__": + unittest.main() \ No newline at end of file diff --git a/live-status/tests/test_structured_prompt.py b/live-status/tests/test_structured_prompt.py new file mode 100644 index 0000000..8b818ab --- /dev/null +++ b/live-status/tests/test_structured_prompt.py @@ -0,0 +1,27 @@ +import json +import unittest + +from labeling.prompts import structured_student_input, student_prompt + + +class StructuredStudentInputTests(unittest.TestCase): + def test_preserves_action_order_and_targets(self): + prompt = structured_student_input("git fetch origin && git status -sb && git log --oneline -5") + parsed = json.loads(prompt.split("\n", 2)[1]) + self.assertEqual([a["subcommand"] for a in parsed["actions"]], ["fetch", "status", "log"]) + self.assertIn("origin", parsed["actions"][0]["targets"]) + + def test_bounds_raw_command_but_keeps_parser_hints(self): + prompt = structured_student_input("echo " + "x" * 5000) + self.assertLess(len(prompt), 1800) + self.assertIn('"action":"format"', prompt) + + def test_training_prompt_uses_structured_input_and_status_boundary(self): + prompt = student_prompt("git status -sb", instruct=False, structured=True) + self.assertIn('"action":"git"', prompt) + self.assertIn("Raw command:\ngit status -sb", prompt) + self.assertTrue(prompt.endswith("\n\nStatus:")) + + +if __name__ == "__main__": + unittest.main() \ No newline at end of file diff --git a/live-status/tests/test_training_encoding.py b/live-status/tests/test_training_encoding.py new file mode 100644 index 0000000..107b1db --- /dev/null +++ b/live-status/tests/test_training_encoding.py @@ -0,0 +1,29 @@ +import unittest + +from training.train import encode + + +class FakeTokenizer: + eos_token_id = 999 + + def __call__(self, text, add_special_tokens): + class Encoded: + pass + encoded = Encoded() + encoded.input_ids = [7] if text.startswith(" ") else list(range(len(text))) + return encoded + + +class TrainingEncodingTests(unittest.TestCase): + def test_long_prompt_keeps_head_and_tail(self): + tok = FakeTokenizer() + ids, labels = encode(tok, "echo " + "x" * 5000, "Done.", False, 128, structured=True) + prompt = ids[:-2] + self.assertEqual(len(ids), 128) + self.assertEqual(prompt[0], 0) + self.assertGreater(prompt[-1], 1000) + self.assertEqual(labels[:len(prompt)], [-100] * len(prompt)) + + +if __name__ == "__main__": + unittest.main() diff --git a/live-status/tools/audit_failures.py b/live-status/tools/audit_failures.py new file mode 100644 index 0000000..5f3fdca --- /dev/null +++ b/live-status/tools/audit_failures.py @@ -0,0 +1,46 @@ +"""Summarize semantic evaluation failures without printing private commands.""" +from __future__ import annotations + +import argparse +import collections +import json +from pathlib import Path + + +def main() -> None: + p = argparse.ArgumentParser() + p.add_argument("output", type=Path) + p.add_argument("--examples", type=int, default=12) + args = p.parse_args() + rows = [json.loads(line) for line in args.output.read_text(encoding="utf-8").splitlines() if line.strip()] + failed = [r for r in rows if r.get("jev") and r["jev"].get("score", 0) < 0.45] + by_tag: collections.Counter[str] = collections.Counter() + reasons: collections.Counter[str] = collections.Counter() + for row in failed: + by_tag.update(row.get("tags") or []) + grade = row["jev"] + if grade.get("invented", 0) > 0.5: + reasons["invented"] += 1 + if grade.get("same", 0) <= 0.5: + reasons["different_actions"] += 1 + if grade.get("quality", 0) < 2.5: + reasons["low_quality"] += 1 + print(json.dumps({ + "rows": len(rows), + "failed": len(failed), + "failure_reasons": reasons.most_common(), + "failure_tags": by_tag.most_common(20), + "schema": sorted(rows[0]) if rows else [], + "examples": [{ + "shell": r.get("shell"), + "length": len(r.get("command", "")), + "tags": r.get("tags"), + "reference": r.get("reference") or r.get("status"), + "output": r.get("output"), + "grade": r.get("jev"), + } for r in failed[:args.examples]], + }, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/live-status/tools/research_hf_candidates.py b/live-status/tools/research_hf_candidates.py new file mode 100644 index 0000000..39d238e --- /dev/null +++ b/live-status/tools/research_hf_candidates.py @@ -0,0 +1,74 @@ +"""List benchmarked sub-1B base models by Hub creation date. + +Reads the public FlameF0X CPU LM benchmark bucket and joins each measured row +with Hugging Face Hub metadata. This keeps discovery registry-driven instead +of starting from hand-picked model names. +""" +from __future__ import annotations + +import argparse +import json +import math +import re +import tempfile +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from huggingface_hub import HfApi, download_bucket_files, list_bucket_tree + +BUCKET = "FlameF0X/lm-cpu-benchmarks" +FINETUNE = re.compile(r"(instruct|chat|sft|dpo|rlhf|finetun|fine-?tun)", re.I) + + +def length_index(row: dict) -> float: + values = [] + for value in (row.get("lengths") or {}).values(): + prefill, decode = value.get("prefill_ts", 0), value.get("decode_ts", 0) + if prefill > 0 and decode > 0: + values.append(2 * prefill * decode / (prefill + decode)) + return sum(values) / len(values) * len(values) / 5 if values else 0.0 + + +def quality(row: dict) -> float: + ppl = row.get("perplexity") + return 1 / math.log2(ppl + 1) if ppl is not None and ppl > 0 else 0.0 + + +def main() -> None: + p = argparse.ArgumentParser() + p.add_argument("--since", default="2025-01-01") + args = p.parse_args() + files = [x.path for x in list_bucket_tree(BUCKET, prefix="results") if x.type == "file" and x.path.endswith(".json")] + with tempfile.TemporaryDirectory() as td: + pairs = [(name, str(Path(td) / Path(name).name)) for name in files] + download_bucket_files(BUCKET, files=pairs) + rows = [json.loads(Path(local).read_text(encoding="utf-8")) for _, local in pairs] + rows = [r for r in rows if 10_000_000 <= r.get("parameters", 0) <= 1_000_000_000 + and not r.get("is_finetune") and not FINETUNE.search(r.get("model_id", "").split("/")[-1])] + api = HfApi() + def metadata(row: dict) -> dict: + try: + info = api.model_info(row["model_id"]) + created = info.created_at.isoformat() if info.created_at else "" + except Exception: + created = "" + return { + "model": row["model_id"], + "created": created[:10], + "parameters_m": round(row["parameters"] / 1e6), + "architecture": row.get("architecture"), + "cpu_prefill_tps": round(row.get("prefill_ts", 0), 1), + "cpu_decode_tps": round(row.get("decode_ts", 0), 1), + "perplexity": round(row["perplexity"], 2) if row.get("perplexity") is not None else None, + "speed_quality": round(length_index(row) * quality(row), 2), + "benchmarked": str(row.get("timestamp", ""))[:10], + } + with ThreadPoolExecutor(max_workers=8) as pool: + found = list(pool.map(metadata, rows)) + found = [r for r in found if r["created"] >= args.since] + found.sort(key=lambda r: (r["created"], r["speed_quality"]), reverse=True) + print(json.dumps({"source": BUCKET, "count": len(found), "models": found}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/live-status/training/train.py b/live-status/training/train.py index 0060b0a..ca12251 100644 --- a/live-status/training/train.py +++ b/live-status/training/train.py @@ -32,7 +32,7 @@ def parse(argv): p.add_argument("--extra", type=Path, action="append", default=[], help="additional train JSONL (command,status)") p.add_argument("--name", required=True) p.add_argument("--method", choices=["lora", "full"], default="lora") - p.add_argument("--prompt", choices=["plain", "instruct", "mixed"], default="plain") + p.add_argument("--prompt", choices=["plain", "instruct", "mixed", "structured"], default="plain") p.add_argument("--epochs", type=float, default=3) p.add_argument("--lr", type=float) p.add_argument("--batch", type=int, default=16) @@ -57,10 +57,13 @@ def load_rows(a) -> tuple[list[dict], list[dict]]: return train, val -def encode(tok, command: str, status: str, instruct: bool, max_len: int): - prompt = tok(student_prompt(command, instruct=instruct), add_special_tokens=True).input_ids +def encode(tok, command: str, status: str, instruct: bool, max_len: int, *, structured: bool = False): + prompt = tok(student_prompt(command, instruct=instruct, structured=structured), add_special_tokens=True).input_ids target = tok(" " + status.strip(), add_special_tokens=False).input_ids + [tok.eos_token_id] - prompt = prompt[-(max_len - len(target)):] + budget = max_len - len(target) + if len(prompt) > budget: + head = int(budget * 0.7) + prompt = prompt[:head] + prompt[-(budget - head):] return prompt + target, [-100] * len(prompt) + target @@ -187,8 +190,10 @@ def main(argv=None): reps = 1 if a.no_weights else max(1, round(r.get("weight", 1))) for k in range(reps): instruct = a.prompt == "instruct" or (a.prompt == "mixed" and rng.random() < 0.3) - items.append(encode(tok, r["command"], r["status"], instruct, a.max_len)) - vitems = [encode(tok, r["command"], r["status"], a.prompt == "instruct", a.max_len) for r in val] + items.append(encode(tok, r["command"], r["status"], instruct, a.max_len, + structured=a.prompt == "structured")) + vitems = [encode(tok, r["command"], r["status"], a.prompt == "instruct", a.max_len, + structured=a.prompt == "structured") for r in val] steps = math.ceil(len(items) / a.batch * a.epochs) sched = get_cosine_schedule_with_warmup(opt, max(1, steps // 20), steps) print(json.dumps({"train_rows": len(train), "train_items": len(items), "val": len(vitems), "steps": steps,