From 1938cf3084676a0fa64343aa7cf8414d05c36fdf Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Sun, 20 Sep 2026 22:48:15 -0400 Subject: [PATCH 1/4] Prototype parser-first PowerShell statuses --- live-status/ARCHITECTURE.md | 8 +- live-status/README.md | 7 + live-status/api/server.py | 7 +- live-status/inference/backends.py | 6 +- live-status/inference/parser_first.py | 275 ++++++++++++++++++ live-status/inference/powershell_ast_host.ps1 | 45 +++ .../research/2026-09-20-model-sweep.md | 28 ++ live-status/tests/test_parser_first.py | 90 ++++++ live-status/tests/test_service.py | 8 + live-status/tools/benchmark_parser_first.py | 107 +++++++ 10 files changed, 576 insertions(+), 5 deletions(-) create mode 100644 live-status/inference/parser_first.py create mode 100644 live-status/inference/powershell_ast_host.ps1 create mode 100644 live-status/tests/test_parser_first.py create mode 100644 live-status/tools/benchmark_parser_first.py diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index 744cebe..182ba20 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -21,7 +21,9 @@ evaluation/evaluate.py ──► validators + jev/Opus grading + latency/memory ──► evaluation/registry.json evaluation/active.py ──► student failures on unlabeled commands ──► teacher/judge ──► prefs.jsonl (DPO) - client ──► api/server.py ──► redact ─► cache ─► model (Ollama) ─► validate ─► heuristic fallback + client ──► api/server.py ──► redact ─► cache ─┬► PowerShell AST ─► mapped renderer ─┐ + └► model (Ollama) ───────────────────┤ + validate ─► heuristic fallback ``` ## Boundaries @@ -37,6 +39,10 @@ ## Why these choices +- **Parser-first PowerShell prototype.** `parser:powershell` keeps one local PowerShell + process warm, extracts commands with PowerShell's real AST, and renders only mapped facts. + Parse errors and dynamic invocation abstain; unknown executables remain literal. It uses + no model or GPU. Coverage and grounding are reported separately from human usefulness. - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. diff --git a/live-status/README.md b/live-status/README.md index 7ab4204..b63471a 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -17,10 +17,17 @@ Details: [TRAINING.md](TRAINING.md), [EVALUATION.md](EVALUATION.md), ## Use it ```powershell +python live-status/cli.py serve --backend parser:powershell --port 8765 python live-status/cli.py serve --backend ollama:live-status-v1-qwen3-06b-lora-q4_k_m --port 8765 python live-status/api/client.py "git fetch origin && git status -sb" ``` +`parser:powershell` is the CPU-only prototype. It uses PowerShell's own AST, maps only +known commands, repeats unknown executable names literally, and abstains on malformed or +dynamic invocation. It does not load a model or use the GPU. The recent-session result and +its limitations are recorded in +[the model sweep](research/2026-09-20-model-sweep.md#parser-first-powershell-prototype). + `--best-of 4` samples several candidates and lets jev pick the best: +8 points of quality for ~1 s per request instead of ~0.1 s. diff --git a/live-status/api/server.py b/live-status/api/server.py index 0f28f34..b95319b 100644 --- a/live-status/api/server.py +++ b/live-status/api/server.py @@ -69,12 +69,13 @@ def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: h_text, h_conf = describe(red, shell) if (self.fast_path and h_conf >= FAST_PATH) or self.backend is None: return self._done(h_text, "heuristic" if h_conf >= FAST_PATH else "fallback", key, t0, cache=h_conf >= FAST_PATH) - status, source = None, "model" + status = None + source = getattr(self.backend, "source", "model") if self.slots.acquire(timeout=self.queue_wait): try: model_in = red if len(red) <= MODEL_INPUT_CHARS else red[:MODEL_INPUT_CHARS] + " …" status, _ = self.backend.generate(model_in) - if self.best_of > 1: + if self.best_of > 1 and source == "model": picked, source = self._best_of(model_in, status) status = picked or status except Exception as exc: # timeouts, backend down @@ -197,7 +198,7 @@ def do_POST(self): def main(argv=None): p = argparse.ArgumentParser(prog="serve") p.add_argument("--backend", default=os.environ.get("LIVE_STATUS_BACKEND", "ollama:live-status"), - help="ollama: | llama-server: | hf:@ | none") + help="parser:powershell | ollama: | llama-server: | hf:@ | none") p.add_argument("--host", default="127.0.0.1") p.add_argument("--port", type=int, default=8765) p.add_argument("--timeout", type=float, default=8.0) diff --git a/live-status/inference/backends.py b/live-status/inference/backends.py index e169439..9433046 100644 --- a/live-status/inference/backends.py +++ b/live-status/inference/backends.py @@ -150,8 +150,12 @@ def _stop_ids(self) -> list[int]: def from_spec(spec: str): - """ollama:[:mode][:cpu] | llama-server: | hf:[@][:structured]""" + """Create a configured inference or deterministic parser backend.""" kind, _, rest = spec.partition(":") + if kind == "parser" and rest == "powershell": + from inference.parser_first import PowerShellAstBackend + + return PowerShellAstBackend() if kind == "ollama": mode, cpu = "plain", False if rest.endswith(":cpu"): diff --git a/live-status/inference/parser_first.py b/live-status/inference/parser_first.py new file mode 100644 index 0000000..a42881d --- /dev/null +++ b/live-status/inference/parser_first.py @@ -0,0 +1,275 @@ +"""CPU-only PowerShell AST backend with conservative deterministic rendering.""" + +from __future__ import annotations + +import json +import re +import shutil +import subprocess +import threading +import time +from pathlib import Path +from typing import Any + +ABSTENTION = "Running a PowerShell command." +_SIMPLE_NAME = re.compile(r"^[A-Za-z0-9_.+-]+$") +_COMMAND_ACTIONS = { + "add-content": "appending to a file", "compare-object": "comparing objects", + "compress-archive": "creating an archive", "convertfrom-json": "parsing JSON", + "convertto-json": "formatting JSON", "copy-item": "copying an item", + "expand-archive": "extracting an archive", "findstr": "searching text", + "foreach-object": "processing pipeline items", + "format-list": "formatting output as a list", + "format-table": "formatting output as a table", + "get-ciminstance": "reading system information", + "get-childitem": "listing directory contents", "get-content": "reading a file", + "get-command": "inspecting available commands", + "get-date": "reading the current date and time", + "get-filehash": "calculating a file hash", + "get-item": "inspecting an item", "get-process": "listing processes", + "get-nettcpconnection": "listing network connections", + "get-service": "listing services", "group-object": "grouping objects", + "invoke-restmethod": "calling a REST endpoint", + "invoke-webrequest": "making an HTTP request", + "join-path": "building a path", + "measure-object": "measuring objects", "move-item": "moving an item", + "new-item": "creating an item", "out-file": "writing a file", + "out-null": "discarding output", "pop-location": "restoring the previous directory", + "push-location": "saving and changing directories", + "remove-item": "removing an item", "rename-item": "renaming an item", + "resolve-path": "resolving a path", "select-object": "selecting object properties", + "select-string": "searching text", "set-content": "writing a file", + "set-location": "changing directories", "sort-object": "sorting objects", + "start-process": "starting a process", "stop-process": "stopping a process", + "start-sleep": "waiting", + "split-path": "extracting part of a path", + "tee-object": "copying pipeline output", "test-path": "checking whether a path exists", + "where-object": "filtering pipeline items", "write-error": "writing an error", + "write-output": "writing output", "write-warning": "writing a warning", +} +_ALIASES = { + "%": "foreach-object", "?": "where-object", "cat": "get-content", + "cd": "set-location", "cp": "copy-item", "del": "remove-item", + "dir": "get-childitem", "echo": "write-output", "gc": "get-content", + "gci": "get-childitem", "gi": "get-item", "ls": "get-childitem", + "mi": "move-item", "mv": "move-item", "ni": "new-item", + "pwd": "get-location", "ren": "rename-item", "rg.exe": "rg", + "ri": "remove-item", "rm": "remove-item", "sls": "select-string", + "select": "select-object", "type": "get-content", "curl.exe": "curl", +} +_GIT_ACTIONS = { + "add": "staging Git changes", "branch": "inspecting Git branches", + "checkout": "switching Git revisions", "clean": "cleaning the Git worktree", + "clone": "cloning a Git repository", "commit": "committing Git changes", + "diff": "inspecting Git changes", "fetch": "fetching Git updates", + "log": "reading Git history", "merge": "merging Git changes", + "ls-files": "listing tracked Git files", + "pull": "pulling Git updates", "push": "pushing Git changes", + "rebase": "rebasing Git changes", "remote": "inspecting Git remotes", + "reset": "resetting Git state", "restore": "restoring Git files", + "rev-parse": "inspecting Git revision data", "show": "showing a Git revision", + "status": "checking Git status", "switch": "switching Git branches", + "tag": "inspecting Git tags", "worktree": "managing Git worktrees", +} +_GH_ACTIONS = { + ("pr", "checks"): "checking pull request status", + ("pr", "create"): "creating a pull request", + ("pr", "diff"): "inspecting a pull request diff", + ("pr", "list"): "listing pull requests", + ("pr", "merge"): "merging a pull request", + ("pr", "view"): "viewing a pull request", + ("run", "view"): "viewing a workflow run", + ("run", "watch"): "watching a workflow run", +} +_RUNTIME_ACTIONS = { + "bun": "running Bun", "bun.exe": "running Bun", "bunx": "running Bun", + "cargo": "running Cargo", + "cmd": "running Command Prompt", "cmd.exe": "running Command Prompt", + "deno": "running Deno", "dotnet": "running .NET", "go": "running Go", + "java": "running Java", "make": "running Make", "node": "running Node.js", + "node.exe": "running Node.js", "npm": "running npm", "npx": "running npx", + "pnpm": "running pnpm", "pwsh": "running PowerShell", + "pwsh.exe": "running PowerShell", "powershell": "running Windows PowerShell", + "powershell.exe": "running Windows PowerShell", "python": "running Python", + "python.exe": "running Python", "python3": "running Python", + "pytest": "running Python tests", "uv": "running uv", "yarn": "running Yarn", +} + + +def _clean(value: str) -> str: + value = value.strip() + if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": + return value[1:-1] + return value + + +def _subcommands(elements: list[str]) -> list[str]: + values: list[str] = [] + skip_value = False + for raw in elements[1:]: + value = _clean(raw) + if skip_value: + skip_value = False + continue + if value in ("-C", "--git-dir", "--work-tree"): + skip_value = True + continue + if value.startswith("-"): + continue + values.append(value.lower()) + return values + + +def _describe(node: dict[str, Any]) -> tuple[str | None, bool]: + raw_name = node.get("name") + if not raw_name: + return None, False + basename = Path(str(raw_name)).name.lower() + name = _ALIASES.get(basename, basename) + elements = [str(item) for item in node.get("elements", [])] + if name == "git": + args = _subcommands(elements) + if args: + return _GIT_ACTIONS.get(args[0], f"running git {args[0]}"), args[0] in _GIT_ACTIONS + return "running Git", False + if name == "gh": + args = _subcommands(elements) + pair = tuple(args[:2]) + if pair in _GH_ACTIONS: + return _GH_ACTIONS[pair], True + return (f"running gh {args[0]}" if args else "running GitHub CLI"), False + if name in ("rg", "ripgrep"): + return "searching text", True + if name == "curl": + return "making an HTTP request", True + if name in _COMMAND_ACTIONS: + return _COMMAND_ACTIONS[name], True + if name == "get-location": + return "reading the current directory", True + if name in _RUNTIME_ACTIONS: + return _RUNTIME_ACTIONS[name], True + if _SIMPLE_NAME.fullmatch(name): + return f"running {name}", False + return None, False + + +def _metrics(nodes, errors, abstained, facts, semantic, literal, total_actions=0): + return { + "abstained": abstained, + "fully_mapped": not abstained and literal == 0 and semantic > 0, + "literal_fallback": literal > 0, + "mapped_actions": semantic, "literal_actions": literal, + "facts": facts, "parse_errors": errors, "ast_nodes": len(nodes), + "total_actions": total_actions, + } + + +def render(parsed: dict[str, Any]) -> tuple[str, dict[str, Any]]: + nodes = list(parsed.get("nodes") or []) + errors = list(parsed.get("errors") or []) + if not parsed.get("ok") or errors or any(node.get("dynamic") for node in nodes): + return ABSTENTION, _metrics(nodes, errors, True, [], 0, 0) + facts: list[dict[str, str]] = [] + semantic = literal = 0 + seen: set[str] = set() + for node in nodes: + if node.get("kind") != "command": + continue + action, known = _describe(node) + if not action or action in seen: + continue + seen.add(action) + facts.append({"text": action, "evidence": str(node.get("evidence", ""))}) + semantic += int(known) + literal += int(not known) + if not facts: + return ABSTENTION, _metrics(nodes, [], True, [], 0, 0) + total_actions = len(facts) + rendered_facts = facts[:3] + phrases = [fact["text"] for fact in rendered_facts] + if len(phrases) == 1: + body = phrases[0] + elif len(phrases) == 2: + body = f"{phrases[0]} and {phrases[1]}" + else: + body = ", ".join(phrases[:-1]) + f", and {phrases[-1]}" + if total_actions > len(rendered_facts): + body += f", plus {total_actions - len(rendered_facts)} more parsed steps" + return body[0].upper() + body[1:] + ".", _metrics( + nodes, [], False, rendered_facts, semantic, literal, total_actions + ) + + +class PowerShellAstBackend: + """Persistent PowerShell parser process; performs no model inference.""" + + name = "parser:powershell" + source = "parser" + + def __init__(self, executable: str | None = None): + self.executable = executable or shutil.which("pwsh") or shutil.which("powershell") + if not self.executable: + raise RuntimeError("PowerShell is required for parser:powershell") + self.host = Path(__file__).with_name("powershell_ast_host.ps1") + self._process: subprocess.Popen[str] | None = None + self._lock = threading.Lock() + self._request_id = 0 + + def _start(self) -> None: + if self._process and self._process.poll() is None: + return + self._process = subprocess.Popen( + [self.executable, "-NoLogo", "-NoProfile", "-NonInteractive", + "-File", str(self.host)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + text=True, encoding="utf-8", bufsize=1, + ) + + def parse(self, command: str) -> dict[str, Any]: + with self._lock: + self._start() + assert self._process and self._process.stdin and self._process.stdout + self._request_id += 1 + self._process.stdin.write(json.dumps( + {"id": self._request_id, "command": command}, ensure_ascii=False + ) + "\n") + self._process.stdin.flush() + line = self._process.stdout.readline() + if not line: + raise RuntimeError( + f"PowerShell AST host exited unexpectedly ({self._process.poll()})" + ) + response = json.loads(line) + if response.get("id") != self._request_id: + raise RuntimeError("PowerShell AST host returned a mismatched response") + return response + + def generate(self, command: str) -> tuple[str, dict[str, Any]]: + started = time.perf_counter() + status, metrics = render(self.parse(command)) + metrics["wall_s"] = time.perf_counter() - started + metrics["backend"] = "parser:powershell" + return status, metrics + + def warm(self) -> None: + self.parse("Get-Location") + + def close(self) -> None: + process, self._process = self._process, None + if not process: + return + if process.stdin: + process.stdin.close() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.terminate() + process.wait(timeout=2) + if process.stdout: + process.stdout.close() + + def __del__(self) -> None: + try: + self.close() + except Exception: + pass diff --git a/live-status/inference/powershell_ast_host.ps1 b/live-status/inference/powershell_ast_host.ps1 new file mode 100644 index 0000000..c32eb19 --- /dev/null +++ b/live-status/inference/powershell_ast_host.ps1 @@ -0,0 +1,45 @@ +param() + +$ErrorActionPreference = 'Stop' +while (($line = [Console]::In.ReadLine()) -ne $null) { + if ([string]::IsNullOrWhiteSpace($line)) { continue } + $request = $null + try { + $request = $line | ConvertFrom-Json + $tokens = $null + $parseErrors = $null + $ast = [System.Management.Automation.Language.Parser]::ParseInput( + [string]$request.command, [ref]$tokens, [ref]$parseErrors + ) + $nodes = @() + foreach ($commandAst in $ast.FindAll( + { param($node) $node -is [System.Management.Automation.Language.CommandAst] }, $true + )) { + $name = $commandAst.GetCommandName() + $nodes += [ordered]@{ + kind = 'command' + name = $name + dynamic = ($null -eq $name) + elements = @($commandAst.CommandElements | ForEach-Object { $_.Extent.Text }) + evidence = $commandAst.Extent.Text + start = $commandAst.Extent.StartOffset + } + } + $response = [ordered]@{ + id = $request.id + ok = ($parseErrors.Count -eq 0) + errors = @($parseErrors | ForEach-Object { $_.Message }) + nodes = @($nodes | Sort-Object start) + } + } + catch { + $response = [ordered]@{ + id = if ($null -ne $request) { $request.id } else { $null } + ok = $false + errors = @($_.Exception.Message) + nodes = @() + } + } + [Console]::Out.WriteLine(($response | ConvertTo-Json -Compress -Depth 8)) + [Console]::Out.Flush() +} diff --git a/live-status/research/2026-09-20-model-sweep.md b/live-status/research/2026-09-20-model-sweep.md index 4a55379..a055918 100644 --- a/live-status/research/2026-09-20-model-sweep.md +++ b/live-status/research/2026-09-20-model-sweep.md @@ -124,5 +124,33 @@ Correction: the judge request was sent to CLIProxyAPI with model `claude-opus-5` DPO passed 16 rows that current failed, while current passed 14 that DPO failed; both passed 41 and both failed 49. DPO's paired difference is +1.7 percentage points (bootstrap 95% interval -7.5 to +10.8; exact McNemar p=0.856), so the new run does not establish a winner between them. Current remains the deployment choice because it invents slightly less, passes the deterministic validator more often, and won the older blind comparison. DPO deserves targeted work: it was stronger on simple commands, long PowerShell, conditionals, nested quoting, and loops, but weaker on pipelines. The drop from the old 75% screen to 45.8% here is real distribution-shift evidence, dominated by omissions on long multi-action commands; neither model is accurate enough for an unqualified trusted UI. +### Parser-first PowerShell prototype + +The frozen recent-session sample is entirely PowerShell, so a CPU-only prototype now uses +PowerShell's own AST instead of asking a small model to rediscover shell structure. A +persistent parser process extracts command names and their exact source extents. The renderer +maps known commands, repeats unknown executable names without inferring their purpose, and +returns "Running a PowerShell command" for parse errors or dynamic invocation. Long cells +show three distinct actions and the exact number of remaining parsed steps. + +| Frozen recent-session sample (120 rows) | Result | +|---|---:| +| AST parse success | 97.5% | +| Fully mapped output | 84.2% | +| Literal executable fallback | 13.3% | +| Abstention | 2.5% | +| Output facts with AST evidence | 97.5% | +| Deterministic validator pass | 100.0% | +| CPU latency p50 / p90 / max | 3.25 / 5.05 / 13.63 ms | + +These are coverage and grounding measurements, not a claim of 97.5% semantic accuracy. A +30-row stratified manual audit found no unsupported rendered action; 26 rows were useful as a +live status, while three deliberate abstentions and one plumbing-dominated summary were safe +but unhelpful. That 86.7% usefulness observation is an exploratory audit on one workload, not +a promotion result. The prototype is worth continuing as a PowerShell path because it is +roughly forty times faster than the current model's 140 ms p50, consumes no GPU, and removes +free-form invention. It does not solve Bash, infer the purpose of unknown tools, or reliably +identify the user's primary intent in long orchestration cells. + Weight pruning comes after semantic parity. Removing generic-domain weights without retraining can destroy useful syntax and language behavior, and zeroed weights do not guarantee lower latency in Ollama/llama.cpp. Quantizing the smaller dense base already produced the useful size and latency gain. If semantic grading passes and further compression is needed, distill the structured task into a smaller student and compare quantization-aware training before structured pruning. Relevant pruning evidence: [Iterative Structured Pruning with Multi-Domain Calibration](https://arxiv.org/abs/2601.02674) argues for hardware-friendly structured removal and mixed-domain calibration; [GPrune-LLM](https://arxiv.org/abs/2603.13418) shows that single-domain calibration can bias neuron importance; [Pruning as a Domain-specific LLM Extractor](https://arxiv.org/abs/2405.06275) supports task-calibrated pruning but does not establish that arbitrary out-of-domain weights can be safely deleted from a sub-1B model. diff --git a/live-status/tests/test_parser_first.py b/live-status/tests/test_parser_first.py new file mode 100644 index 0000000..4f6f329 --- /dev/null +++ b/live-status/tests/test_parser_first.py @@ -0,0 +1,90 @@ +from __future__ import annotations + +import shutil +import unittest + +from inference.backends import from_spec +from inference.parser_first import ABSTENTION, PowerShellAstBackend, render + + +def command(name: str, *elements: str): + return { + "kind": "command", "name": name, "dynamic": False, + "elements": [name, *elements], "evidence": " ".join([name, *elements]), + "start": 0, + } + + +class ParserFirstTests(unittest.TestCase): + def test_git_switch_does_not_invent_branch_creation(self): + status, metrics = render({ + "ok": True, "errors": [], "nodes": [command("git", "switch", "topic")] + }) + self.assertEqual(status, "Switching Git branches.") + self.assertTrue(metrics["fully_mapped"]) + + def test_parse_errors_and_dynamic_invocation_abstain(self): + status, metrics = render({ + "ok": False, "errors": ["Incomplete input"], "nodes": [] + }) + self.assertEqual(status, ABSTENTION) + self.assertTrue(metrics["abstained"]) + dynamic = command("", "&", "$tool") + dynamic["dynamic"] = True + status, metrics = render({ + "ok": True, "errors": [], "nodes": [dynamic] + }) + self.assertEqual(status, ABSTENTION) + self.assertTrue(metrics["abstained"]) + + def test_unknown_executable_is_literal_fallback(self): + status, metrics = render({ + "ok": True, "errors": [], "nodes": [command("widgetctl", "deploy")] + }) + self.assertEqual(status, "Running widgetctl.") + self.assertTrue(metrics["literal_fallback"]) + self.assertFalse(metrics["fully_mapped"]) + + def test_long_sequence_is_bounded_by_parsed_step_count(self): + nodes = [ + command("Get-Content"), command("Select-String"), + command("git", "status"), command("Write-Output"), + command("Test-Path"), + ] + status, metrics = render({"ok": True, "errors": [], "nodes": nodes}) + self.assertEqual( + status, + "Reading a file, searching text, and checking Git status, " + "plus 2 more parsed steps.", + ) + self.assertEqual(metrics["total_actions"], 5) + self.assertEqual(len(metrics["facts"]), 3) + + @unittest.skipUnless( + shutil.which("pwsh") or shutil.which("powershell"), + "PowerShell unavailable", + ) + def test_persistent_host_parses_pipeline_and_sequence(self): + backend = PowerShellAstBackend() + try: + status, metrics = backend.generate( + "Get-Content README.md | Select-String TODO; git status" + ) + finally: + backend.close() + self.assertEqual( + status, "Reading a file, searching text, and checking Git status." + ) + self.assertEqual(metrics["mapped_actions"], 3) + self.assertFalse(metrics["abstained"]) + + def test_public_backend_spec(self): + backend = from_spec("parser:powershell") + try: + self.assertIsInstance(backend, PowerShellAstBackend) + finally: + backend.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/live-status/tests/test_service.py b/live-status/tests/test_service.py index d843d01..abd4e91 100644 --- a/live-status/tests/test_service.py +++ b/live-status/tests/test_service.py @@ -72,6 +72,14 @@ def test_invalid_or_failed_model_output_falls_back(self): good = server.Service(FakeBackend("Building the incremental search index.")).summarize(cmd, "bash", None) self.assertEqual((good["status"], good["source"]), ("Building the incremental search index.", "model")) + def test_deterministic_backend_keeps_source_and_skips_best_of(self): + backend = FakeBackend("Checking Git status.") + backend.source = "parser" + service = server.Service(backend, cache_size=0, best_of=4) + with mock.patch.object(service, "_best_of", side_effect=AssertionError("model-only")): + out = service.summarize("git status", "powershell", None) + self.assertEqual((out["status"], out["source"]), ("Checking Git status.", "parser")) + if __name__ == "__main__": unittest.main() diff --git a/live-status/tools/benchmark_parser_first.py b/live-status/tools/benchmark_parser_first.py new file mode 100644 index 0000000..f79e63a --- /dev/null +++ b/live-status/tools/benchmark_parser_first.py @@ -0,0 +1,107 @@ +"""Benchmark parser-first status generation without model or GPU calls.""" + +from __future__ import annotations + +import argparse +import json +import statistics +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from inference.parser_first import PowerShellAstBackend + + +def percentile(values: list[float], fraction: float) -> float: + ordered = sorted(values) + if not ordered: + return 0.0 + position = (len(ordered) - 1) * fraction + lower = int(position) + upper = min(lower + 1, len(ordered) - 1) + weight = position - lower + return ordered[lower] * (1 - weight) + ordered[upper] * weight + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--sample", type=Path, required=True) + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args() + rows = [ + json.loads(line) + for line in args.sample.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + args.out.mkdir(parents=True, exist_ok=True) + backend = PowerShellAstBackend() + outputs = [] + try: + backend.warm() + for row in rows: + status, metrics = backend.generate(row["command"]) + outputs.append({ + "id": row.get("id"), "session": row.get("session"), + "complexity": row.get("complexity"), "tags": row.get("tags"), + "command": row["command"], "status": status, "metrics": metrics, + }) + finally: + backend.close() + + count = len(outputs) + latencies = [item["metrics"]["wall_s"] * 1000 for item in outputs] + parse_success = sum(not item["metrics"]["parse_errors"] for item in outputs) + fully_mapped = sum(item["metrics"]["fully_mapped"] for item in outputs) + literal = sum(item["metrics"]["literal_fallback"] for item in outputs) + abstained = sum(item["metrics"]["abstained"] for item in outputs) + mapped = sum(item["metrics"]["mapped_actions"] > 0 for item in outputs) + evidence = sum( + bool(item["metrics"]["facts"]) + and all(fact["evidence"] for fact in item["metrics"]["facts"]) + for item in outputs + ) + validator = sum( + item["status"].endswith(".") and "\n" not in item["status"] + and all( + fact["text"].lower() in item["status"].lower() + for fact in item["metrics"]["facts"] + ) + for item in outputs + ) + + def pct(value: int) -> float: + return round(100 * value / count, 1) if count else 0.0 + + summary = { + "backend": "parser:powershell", "rows": count, + "ast_parse_success_pct": pct(parse_success), + "fully_mapped_coverage_pct": pct(fully_mapped), + "literal_fallback_pct": pct(literal), + "abstention_pct": pct(abstained), + "mapped_action_coverage_pct": pct(mapped), + "evidence_grounded_pct": pct(evidence), + "validator_pass_pct": pct(validator), + "latency_ms": { + "mean": round(statistics.fmean(latencies), 3) if latencies else 0.0, + "p50": round(percentile(latencies, 0.5), 3), + "p90": round(percentile(latencies, 0.9), 3), + "max": round(max(latencies), 3) if latencies else 0.0, + }, + "limitations": [ + "Mapped coverage and grounding are mechanical measurements, not human-rated semantic accuracy.", + "The prototype handles PowerShell only.", + "Literal fallback repeats an AST-proven executable name without claiming its purpose.", + ], + } + with (args.out / "outputs.jsonl").open("w", encoding="utf-8") as handle: + for output in outputs: + handle.write(json.dumps(output, ensure_ascii=False) + "\n") + (args.out / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + print(json.dumps(summary, indent=2)) + + +if __name__ == "__main__": + main() From 2106f94b05c2213d2eceb084e1637ab4a9ebed0d Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:05:15 -0400 Subject: [PATCH 2/4] Resolve command purpose from project context --- live-status/ARCHITECTURE.md | 8 +- live-status/README.md | 14 +- live-status/api/server.py | 7 +- live-status/inference/linguistic_map.json | 4 + live-status/inference/parser_first.py | 230 +++++++++++++++++- live-status/linguistic_training.py | 156 ++++++++++++ .../research/2026-09-20-model-sweep.md | 26 +- live-status/tests/test_linguistic_training.py | 45 ++++ live-status/tests/test_parser_first.py | 64 ++++- live-status/tests/test_service.py | 14 ++ live-status/tools/benchmark_parser_first.py | 7 +- 11 files changed, 546 insertions(+), 29 deletions(-) create mode 100644 live-status/inference/linguistic_map.json create mode 100644 live-status/linguistic_training.py create mode 100644 live-status/tests/test_linguistic_training.py diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index 182ba20..a2dc10b 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -41,8 +41,12 @@ - **Parser-first PowerShell prototype.** `parser:powershell` keeps one local PowerShell process warm, extracts commands with PowerShell's real AST, and renders only mapped facts. - Parse errors and dynamic invocation abstain; unknown executables remain literal. It uses - no model or GPU. Coverage and grounding are reported separately from human usefulness. + With a working directory, a semantic resolver can ground a test action in its declared test + name. Parse errors and dynamic invocation abstain; unknown executables remain literal. It + uses no model or GPU at runtime. A development-only linguistic trainer places every mapping + in simulated combinations and asks an explicitly selected LLM for proposals; only reviewed, + accepted phrases enter `linguistic_map.json`. Coverage and grounding are reported separately + from human usefulness. - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. diff --git a/live-status/README.md b/live-status/README.md index b63471a..c20efab 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -24,16 +24,26 @@ python live-status/api/client.py "git fetch origin && git status -sb" `parser:powershell` is the CPU-only prototype. It uses PowerShell's own AST, maps only known commands, repeats unknown executable names literally, and abstains on malformed or -dynamic invocation. It does not load a model or use the GPU. The recent-session result and +dynamic invocation. When `cwd` is supplied, it can resolve a named test file and report its +declared test purpose instead of restating the filename. It does not load a model or use the GPU. +The recent-session result and its limitations are recorded in [the model sweep](research/2026-09-20-model-sweep.md#parser-first-powershell-prototype). +An LLM may improve the linguistic mapping during development without joining the runtime path. +`python live-status/linguistic_training.py simulate --output ` creates stable single, +pair, and triple phrase combinations. `propose --route --model + --output ` asks that explicitly selected CLIProxyAPI route and +model for grounded wording proposals. Every proposal +defaults to `accepted: false`; after review, `apply --input ` writes only accepted entries +to `inference/linguistic_map.json`. + `--best-of 4` samples several candidates and lets jev pick the best: +8 points of quality for ~1 s per request instead of ~0.1 s. `POST /v1/summarize-command` with `{"command": "...", "shell": "powershell", "cwd": "optional"}` returns `{"status": "..."}` (add `"debug": true` for source and latency). The service redacts -before inference, caches by normalised command, bounds concurrency, falls back to a +before inference, caches by shell, working directory, and normalised command, bounds concurrency, falls back to a deterministic describer on timeout or invalid output, and never logs commands. For remote use set `LIVE_STATUS_API_TOKEN` and pass `--tls-cert/--tls-key`; a non-loopback bind without a token is refused. `api/client.py` works unchanged against `https://my-server.example`. diff --git a/live-status/api/server.py b/live-status/api/server.py index b95319b..2b7e91d 100644 --- a/live-status/api/server.py +++ b/live-status/api/server.py @@ -59,7 +59,7 @@ def _log(self, row: dict) -> None: def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: t0 = time.perf_counter() red = redact(command[:MAX_COMMAND]) - key = sha(f"{shell}|{normalize_ws(red)}") + key = sha(f"{shell}|{cwd}|{normalize_ws(red)}") with self.lock: hit = self.cache.get(key) if hit is not None: @@ -74,7 +74,10 @@ def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: if self.slots.acquire(timeout=self.queue_wait): try: model_in = red if len(red) <= MODEL_INPUT_CHARS else red[:MODEL_INPUT_CHARS] + " …" - status, _ = self.backend.generate(model_in) + if getattr(self.backend, "accepts_cwd", False): + status, _ = self.backend.generate(model_in, cwd=cwd) + else: + status, _ = self.backend.generate(model_in) if self.best_of > 1 and source == "model": picked, source = self._best_of(model_in, status) status = picked or status diff --git a/live-status/inference/linguistic_map.json b/live-status/inference/linguistic_map.json new file mode 100644 index 0000000..5da9efe --- /dev/null +++ b/live-status/inference/linguistic_map.json @@ -0,0 +1,4 @@ +{ + "version": 1, + "action_overrides": {} +} diff --git a/live-status/inference/parser_first.py b/live-status/inference/parser_first.py index a42881d..7230083 100644 --- a/live-status/inference/parser_first.py +++ b/live-status/inference/parser_first.py @@ -94,6 +94,30 @@ "python.exe": "running Python", "python3": "running Python", "pytest": "running Python tests", "uv": "running uv", "yarn": "running Yarn", } +_LINGUISTIC_MAP_PATH = Path(__file__).with_name("linguistic_map.json") + + +def _load_linguistic_overrides(path: Path = _LINGUISTIC_MAP_PATH) -> dict[str, str]: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeError, ValueError): + return {} + values = data.get("action_overrides", {}) if isinstance(data, dict) else {} + if not isinstance(values, dict): + return {} + return { + str(key): str(value).strip() + for key, value in values.items() + if isinstance(value, str) and value.strip() + } + + +_LINGUISTIC_OVERRIDES = _load_linguistic_overrides() + + +def mapped_phrase(key: str, fallback: str) -> str: + """Return reviewed development-time wording, never model output at runtime.""" + return _LINGUISTIC_OVERRIDES.get(key, fallback) def _clean(value: str) -> str: @@ -103,6 +127,161 @@ def _clean(value: str) -> str: return value +def _human_name(value: str) -> str | None: + value = _clean(value).replace("\\", "/").rstrip("/") + if not value or value.startswith(("$", "@", "{", "(")): + return None + name = value.rsplit("/", 1)[-1] + for suffix in ( + ".test.ts", ".test.tsx", ".test.js", ".test.jsx", + ".spec.ts", ".spec.tsx", ".spec.js", ".spec.jsx", + ".py", ".ts", ".tsx", ".js", ".jsx", + ): + if name.lower().endswith(suffix): + name = name[:-len(suffix)] + break + name = re.sub(r"[-_]+", " ", name).strip() + return name or None + + +def _file_name(value: str) -> str | None: + value = _clean(value).replace("\\", "/").rstrip("/") + if not value or value.startswith(("$", "@", "{", "(")): + return None + return value.rsplit("/", 1)[-1] or None + + +def _option_values(args: list[str], options: set[str]) -> list[str]: + values = [] + skip = False + for raw in args: + value = _clean(raw) + if skip: + skip = False + continue + if value.lower() in options: + skip = True + continue + if value.startswith("-"): + continue + values.append(value) + return values + + +def _join_words(values: list[str]) -> str: + if len(values) == 1: + return values[0] + if len(values) == 2: + return f"{values[0]} and {values[1]}" + return ", ".join(values[:-1]) + f", and {values[-1]}" + + +def _resolved_file(cwd: str | None, target: str) -> Path | None: + if not cwd or target.startswith(("$", "@", "{", "(")): + return None + try: + root = Path(cwd).resolve() + candidate = (root / _clean(target)).resolve() + except (OSError, RuntimeError): + return None + if not root.is_dir(): + return None + try: + candidate.relative_to(root) + except ValueError: + return None + return candidate if candidate.is_file() else None + + +def _test_intent(cwd: str | None, target: str) -> tuple[str, str] | None: + path = _resolved_file(cwd, target) + if not path: + return None + try: + source = path.read_text(encoding="utf-8") + except (OSError, UnicodeError): + return None + names = [] + for match in re.finditer( + r"\b(?:test|it)\s*\(\s*(['\"])(.*?)\1", source + ): + name = normalize_ws(match.group(2)).rstrip(".") + if name and name not in names: + names.append(name) + if len(names) == 1: + return f"testing that {names[0]}", f"{path}:{names[0]}" + suites = [] + for match in re.finditer( + r"\bdescribe\s*\(\s*(['\"])(.*?)\1", source + ): + name = normalize_ws(match.group(2)).rstrip(".") + if name and name not in suites: + suites.append(name) + if len(suites) == 1: + return f"testing {suites[0]}", f"{path}:{suites[0]}" + core = re.search(r"@core-prevents\s+([^\r\n*]+)", source) + if core: + purpose = normalize_ws(core.group(1)).rstrip(".") + return f"testing protection against {purpose}", f"{path}:{purpose}" + return None + + +def normalize_ws(value: str) -> str: + return " ".join(value.split()) + + +def _context_description( + node: dict[str, Any], cwd: str | None +) -> tuple[str, str] | None: + raw_name = node.get("name") + if not raw_name: + return None + name = Path(str(raw_name)).name.lower() + if name not in ("bun", "bun.exe"): + return None + elements = [str(item) for item in node.get("elements", [])] + args = [_clean(item) for item in elements[1:]] + if not args or args[0].lower() != "test": + return None + targets = _option_values( + args[1:], {"--timeout", "--filter", "--preload", "--rerun-each"} + ) + intents = [intent for target in targets if (intent := _test_intent(cwd, target))] + if len(intents) == 1: + return intents[0] + return None + + +def _describe_runtime(name: str, elements: list[str]) -> tuple[str, bool]: + args = [_clean(item) for item in elements[1:]] + if name in ("bun", "bun.exe"): + if args and args[0].lower() == "test": + targets = _option_values( + args[1:], {"--timeout", "--filter", "--preload", "--rerun-each"} + ) + labels = [label for item in targets if (label := _human_name(item))] + if labels: + return f"running the {_join_words(labels)} tests", True + return "running tests with Bun", True + if len(args) >= 2 and args[0].lower() == "run": + return f"running {args[1]} with Bun", True + if args and (label := _human_name(args[0])): + return f"running {label} with Bun", True + if name == "bunx": + targets = _option_values(args, {"--package"}) + if targets: + return f"running {targets[0]} with Bun", True + if name in ("python", "python.exe", "python3"): + if len(args) >= 2 and args[0] == "-m": + return f"running {args[1]} with Python", True + targets = _option_values(args, {"-W", "-X"}) + if targets and targets[0] not in ("-c", "-"): + label = _file_name(targets[0]) + if label: + return f"running {label} with Python", True + return mapped_phrase(f"runtime:{name}", _RUNTIME_ACTIONS[name]), True + + def _subcommands(elements: list[str]) -> list[str]: values: list[str] = [] skip_value = False @@ -130,24 +309,38 @@ def _describe(node: dict[str, Any]) -> tuple[str | None, bool]: if name == "git": args = _subcommands(elements) if args: - return _GIT_ACTIONS.get(args[0], f"running git {args[0]}"), args[0] in _GIT_ACTIONS + fallback = _GIT_ACTIONS.get(args[0], f"running git {args[0]}") + return mapped_phrase(f"git:{args[0]}", fallback), args[0] in _GIT_ACTIONS return "running Git", False if name == "gh": args = _subcommands(elements) pair = tuple(args[:2]) if pair in _GH_ACTIONS: - return _GH_ACTIONS[pair], True + key = " ".join(pair) + return mapped_phrase(f"gh:{key}", _GH_ACTIONS[pair]), True return (f"running gh {args[0]}" if args else "running GitHub CLI"), False if name in ("rg", "ripgrep"): - return "searching text", True + return mapped_phrase("command:rg", "searching text"), True if name == "curl": - return "making an HTTP request", True + return mapped_phrase("command:curl", "making an HTTP request"), True + if name == "get-content": + args = elements[1:] + targets = _option_values( + args, + { + "-credential", "-delimiter", "-encoding", "-filter", + "-readcount", "-stream", "-tail", "-totalcount", + }, + ) + labels = [label for item in targets if (label := _file_name(item))] + if labels: + return f"reading {_join_words(labels)}", True if name in _COMMAND_ACTIONS: - return _COMMAND_ACTIONS[name], True + return mapped_phrase(f"command:{name}", _COMMAND_ACTIONS[name]), True if name == "get-location": return "reading the current directory", True if name in _RUNTIME_ACTIONS: - return _RUNTIME_ACTIONS[name], True + return _describe_runtime(name, elements) if _SIMPLE_NAME.fullmatch(name): return f"running {name}", False return None, False @@ -161,10 +354,13 @@ def _metrics(nodes, errors, abstained, facts, semantic, literal, total_actions=0 "mapped_actions": semantic, "literal_actions": literal, "facts": facts, "parse_errors": errors, "ast_nodes": len(nodes), "total_actions": total_actions, + "context_actions": sum("context_evidence" in fact for fact in facts), } -def render(parsed: dict[str, Any]) -> tuple[str, dict[str, Any]]: +def render( + parsed: dict[str, Any], *, cwd: str | None = None +) -> tuple[str, dict[str, Any]]: nodes = list(parsed.get("nodes") or []) errors = list(parsed.get("errors") or []) if not parsed.get("ok") or errors or any(node.get("dynamic") for node in nodes): @@ -175,11 +371,20 @@ def render(parsed: dict[str, Any]) -> tuple[str, dict[str, Any]]: for node in nodes: if node.get("kind") != "command": continue - action, known = _describe(node) + contextual = _context_description(node, cwd) + if contextual: + action, context_evidence = contextual + known = True + else: + action, known = _describe(node) + context_evidence = None if not action or action in seen: continue seen.add(action) - facts.append({"text": action, "evidence": str(node.get("evidence", ""))}) + fact = {"text": action, "evidence": str(node.get("evidence", ""))} + if context_evidence: + fact["context_evidence"] = context_evidence + facts.append(fact) semantic += int(known) literal += int(not known) if not facts: @@ -205,6 +410,7 @@ class PowerShellAstBackend: name = "parser:powershell" source = "parser" + accepts_cwd = True def __init__(self, executable: str | None = None): self.executable = executable or shutil.which("pwsh") or shutil.which("powershell") @@ -244,9 +450,11 @@ def parse(self, command: str) -> dict[str, Any]: raise RuntimeError("PowerShell AST host returned a mismatched response") return response - def generate(self, command: str) -> tuple[str, dict[str, Any]]: + def generate( + self, command: str, *, cwd: str | None = None + ) -> tuple[str, dict[str, Any]]: started = time.perf_counter() - status, metrics = render(self.parse(command)) + status, metrics = render(self.parse(command), cwd=cwd) metrics["wall_s"] = time.perf_counter() - started metrics["backend"] = "parser:powershell" return status, metrics diff --git a/live-status/linguistic_training.py b/live-status/linguistic_training.py new file mode 100644 index 0000000..cef0456 --- /dev/null +++ b/live-status/linguistic_training.py @@ -0,0 +1,156 @@ +"""Development-only LLM proposals for the deterministic linguistic map.""" +from __future__ import annotations + +import argparse +import json +import os +import random +import re +from pathlib import Path + +from inference.parser_first import ( + _COMMAND_ACTIONS, + _GH_ACTIONS, + _GIT_ACTIONS, + _LINGUISTIC_MAP_PATH, + _RUNTIME_ACTIONS, +) + + +def action_catalog() -> dict[str, str]: + catalog = {f"command:{key}": value for key, value in _COMMAND_ACTIONS.items()} + catalog.update({f"git:{key}": value for key, value in _GIT_ACTIONS.items()}) + catalog.update({f"gh:{' '.join(key)}": value for key, value in _GH_ACTIONS.items()}) + catalog.update({f"runtime:{key}": value for key, value in _RUNTIME_ACTIONS.items()}) + catalog["command:rg"] = "searching text" + catalog["command:curl"] = "making an HTTP request" + return dict(sorted(catalog.items())) + + +def simulate(seed: int = 20260920) -> list[dict]: + """Place every atomic phrase in stable single, pair, and triple contexts.""" + rng = random.Random(seed) + catalog = action_catalog() + keys = list(catalog) + rows = [] + for key, phrase in catalog.items(): + others = [candidate for candidate in keys if candidate != key] + pair_key, third_key = rng.sample(others, 2) + pair = [phrase, catalog[pair_key]] + triple = [catalog[third_key], phrase, catalog[pair_key]] + rows.append({ + "key": key, + "current_phrase": phrase, + "simulated_sentences": [ + phrase.capitalize() + ".", + f"{pair[0].capitalize()} and {pair[1]}.", + f"{triple[0].capitalize()}, {triple[1]}, and {triple[2]}.", + ], + }) + return rows + + +SYSTEM = """You improve an approved deterministic shell-status phrase map during development. +Each item contains a semantic key, its grounded atomic phrase, and simulated combinations. +Suggest an override only when it improves natural English while preserving exactly the same +meaning. Atomic phrases must start with a present participle, contain no period, command, +flag, invented purpose, or result. Return JSON only: {\"results\":[{\"key\":\"...\", +\"suggested_phrase\":\"...\",\"reason\":\"...\"}]}. Omit keys that should stay unchanged.""" + + +def validate_proposals(raw: list[dict], catalog: dict[str, str] | None = None) -> list[dict]: + catalog = catalog or action_catalog() + accepted = [] + for item in raw: + if not isinstance(item, dict): + continue + key = item.get("key") + phrase = item.get("suggested_phrase") + if key not in catalog or not isinstance(phrase, str): + continue + phrase = " ".join(phrase.split()).strip() + if not phrase or phrase.endswith(".") or any(char in phrase for char in "|;&`\r\n"): + continue + if not re.match(r"^[A-Za-z]+ing\b", phrase, re.IGNORECASE): + continue + accepted.append({ + "key": key, + "current_phrase": catalog[key], + "suggested_phrase": phrase, + "reason": str(item.get("reason", "")).strip(), + "accepted": False, + }) + return accepted + + +def propose(route: str, model: str, output: Path, seed: int = 20260920) -> dict: + from labeling.llm import chat, parse_json + + rows = simulate(seed) + text, usage = chat([ + {"role": "system", "content": SYSTEM}, + {"role": "user", "content": json.dumps(rows, ensure_ascii=False)}, + ], model=model) + parsed = parse_json(text) + raw = parsed.get("results", []) if isinstance(parsed, dict) else [] + record = { + "version": 1, + "route": route, + "model": model, + "seed": seed, + "usage": usage, + "proposals": validate_proposals(raw), + } + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(record, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + return record + + +def apply_reviewed(proposals: Path, destination: Path = _LINGUISTIC_MAP_PATH) -> dict: + source = json.loads(proposals.read_text(encoding="utf-8")) + current = json.loads(destination.read_text(encoding="utf-8")) + overrides = dict(current.get("action_overrides", {})) + catalog = action_catalog() + applied = [] + for item in source.get("proposals", []): + checked = validate_proposals([item], catalog) + if item.get("accepted") is True and checked: + proposal = checked[0] + overrides[proposal["key"]] = proposal["suggested_phrase"] + applied.append(proposal["key"]) + updated = {"version": 1, "action_overrides": dict(sorted(overrides.items()))} + temporary = destination.with_suffix(destination.suffix + ".tmp") + temporary.write_text(json.dumps(updated, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + os.replace(temporary, destination) + return {"applied": applied, "destination": str(destination)} + + +def main(argv=None) -> None: + parser = argparse.ArgumentParser(prog="linguistic-training") + sub = parser.add_subparsers(dest="command", required=True) + simulation = sub.add_parser("simulate") + simulation.add_argument("--output", type=Path, required=True) + simulation.add_argument("--seed", type=int, default=20260920) + proposal = sub.add_parser("propose") + proposal.add_argument("--route", required=True, help="user-confirmed provider route identity") + proposal.add_argument("--model", required=True, help="exact user-selected model on that route") + proposal.add_argument("--output", type=Path, required=True) + proposal.add_argument("--seed", type=int, default=20260920) + apply = sub.add_parser("apply") + apply.add_argument("--input", type=Path, required=True) + apply.add_argument("--destination", type=Path, default=_LINGUISTIC_MAP_PATH) + args = parser.parse_args(argv) + if args.command == "simulate": + rows = simulate(args.seed) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(rows, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + result = {"simulations": len(rows), "output": str(args.output)} + elif args.command == "propose": + result = propose(args.route, args.model, args.output, args.seed) + else: + result = apply_reviewed(args.input, args.destination) + print(json.dumps(result, ensure_ascii=False, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/live-status/research/2026-09-20-model-sweep.md b/live-status/research/2026-09-20-model-sweep.md index a055918..e920a00 100644 --- a/live-status/research/2026-09-20-model-sweep.md +++ b/live-status/research/2026-09-20-model-sweep.md @@ -139,18 +139,24 @@ show three distinct actions and the exact number of remaining parsed steps. | Fully mapped output | 84.2% | | Literal executable fallback | 13.3% | | Abstention | 2.5% | +| Rows retaining working directory / context resolved | 0 / 0 | | Output facts with AST evidence | 97.5% | | Deterministic validator pass | 100.0% | -| CPU latency p50 / p90 / max | 3.25 / 5.05 / 13.63 ms | - -These are coverage and grounding measurements, not a claim of 97.5% semantic accuracy. A -30-row stratified manual audit found no unsupported rendered action; 26 rows were useful as a -live status, while three deliberate abstentions and one plumbing-dominated summary were safe -but unhelpful. That 86.7% usefulness observation is an exploratory audit on one workload, not -a promotion result. The prototype is worth continuing as a PowerShell path because it is -roughly forty times faster than the current model's 140 ms p50, consumes no GPU, and removes -free-form invention. It does not solve Bash, infer the purpose of unknown tools, or reliably -identify the user's primary intent in long orchestration cells. +| CPU latency p50 / p90 / max | 2.75 / 4.49 / 10.13 ms | + +These are coverage and grounding measurements, not semantic accuracy. The earlier 30-row audit +that called 26 outputs useful is invalid: its rubric counted executable or filename restatements +as useful even when they did not explain the operation. The frozen sample also omitted working +directories, so it cannot measure the new context resolver. Human usefulness is therefore +unmeasured until a held-out set captures command, working directory, and expected purpose. + +The context resolver now follows a Bun test target inside the supplied working directory and +uses a unique declared test name as evidence. For `bun test --timeout 90000 +test/codex-quest-dev-installer.test.ts`, the grounded output is “Testing that every development +install advances, seals and adds a new cache version without removing the old one.” The runtime +still does not solve Bash, infer opaque tools, or reliably identify the primary intent in long +orchestration cells. Linguistic improvements are trained outside runtime: an explicitly selected +LLM reviews simulated mapping combinations, and only human-accepted proposals change the map. Weight pruning comes after semantic parity. Removing generic-domain weights without retraining can destroy useful syntax and language behavior, and zeroed weights do not guarantee lower latency in Ollama/llama.cpp. Quantizing the smaller dense base already produced the useful size and latency gain. If semantic grading passes and further compression is needed, distill the structured task into a smaller student and compare quantization-aware training before structured pruning. Relevant pruning evidence: [Iterative Structured Pruning with Multi-Domain Calibration](https://arxiv.org/abs/2601.02674) argues for hardware-friendly structured removal and mixed-domain calibration; [GPrune-LLM](https://arxiv.org/abs/2603.13418) shows that single-domain calibration can bias neuron importance; [Pruning as a Domain-specific LLM Extractor](https://arxiv.org/abs/2405.06275) supports task-calibrated pruning but does not establish that arbitrary out-of-domain weights can be safely deleted from a sub-1B model. diff --git a/live-status/tests/test_linguistic_training.py b/live-status/tests/test_linguistic_training.py new file mode 100644 index 0000000..dc807d8 --- /dev/null +++ b/live-status/tests/test_linguistic_training.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +import json +import tempfile +import unittest +from pathlib import Path + +from linguistic_training import action_catalog, apply_reviewed, simulate, validate_proposals + + +class LinguisticTrainingTests(unittest.TestCase): + def test_simulates_each_mapping_in_combinations(self): + rows = simulate(seed=7) + self.assertEqual(len(rows), len(action_catalog())) + self.assertTrue(all(len(row["simulated_sentences"]) == 3 for row in rows)) + self.assertTrue(any(row["key"] == "git:status" for row in rows)) + + def test_invalid_or_unknown_proposals_are_rejected(self): + proposals = validate_proposals([ + {"key": "git:status", "suggested_phrase": "reviewing Git status"}, + {"key": "git:status", "suggested_phrase": "review Git status"}, + {"key": "git:status", "suggested_phrase": "git status; delete everything"}, + {"key": "unknown:key", "suggested_phrase": "doing something"}, + ]) + self.assertEqual([item["suggested_phrase"] for item in proposals], ["reviewing Git status"]) + self.assertFalse(proposals[0]["accepted"]) + + def test_only_human_accepted_proposals_change_mapping(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + destination = root / "map.json" + destination.write_text('{"version":1,"action_overrides":{}}', encoding="utf-8") + proposals = root / "proposals.json" + proposals.write_text(json.dumps({"proposals": [ + {"key": "git:status", "suggested_phrase": "reviewing Git status", "accepted": True}, + {"key": "git:diff", "suggested_phrase": "reviewing Git changes", "accepted": False}, + ]}), encoding="utf-8") + result = apply_reviewed(proposals, destination) + saved = json.loads(destination.read_text(encoding="utf-8")) + self.assertEqual(result["applied"], ["git:status"]) + self.assertEqual(saved["action_overrides"], {"git:status": "reviewing Git status"}) + + +if __name__ == "__main__": + unittest.main() diff --git a/live-status/tests/test_parser_first.py b/live-status/tests/test_parser_first.py index 4f6f329..0fdee34 100644 --- a/live-status/tests/test_parser_first.py +++ b/live-status/tests/test_parser_first.py @@ -1,7 +1,9 @@ from __future__ import annotations import shutil +import tempfile import unittest +from pathlib import Path from inference.backends import from_spec from inference.parser_first import ABSTENTION, PowerShellAstBackend, render @@ -45,6 +47,66 @@ def test_unknown_executable_is_literal_fallback(self): self.assertTrue(metrics["literal_fallback"]) self.assertFalse(metrics["fully_mapped"]) + def test_bun_test_names_the_test_instead_of_the_runtime(self): + status, metrics = render({ + "ok": True, + "errors": [], + "nodes": [command( + "bun", "test", "--timeout", "90000", + "test/codex-quest-dev-installer.test.ts", + )], + }) + self.assertEqual( + status, "Running the codex quest dev installer tests." + ) + self.assertTrue(metrics["fully_mapped"]) + + def test_bun_test_resolves_grounded_intent_from_cwd(self): + with tempfile.TemporaryDirectory() as directory: + target = Path(directory) / "installer.test.ts" + target.write_text( + "test('every install creates a sealed version without deleting " + "the old one', async () => {})\n", + encoding="utf-8", + ) + status, metrics = render( + { + "ok": True, + "errors": [], + "nodes": [command("bun", "test", "installer.test.ts")], + }, + cwd=directory, + ) + self.assertEqual( + status, + "Testing that every install creates a sealed version without " + "deleting the old one.", + ) + self.assertIn("context_evidence", metrics["facts"][0]) + self.assertEqual(metrics["context_actions"], 1) + + def test_invalid_cwd_falls_back_without_failing(self): + status, metrics = render({ + "ok": True, "errors": [], + "nodes": [command("bun", "test", "installer.test.ts")], + }, cwd="Z:/a-directory-that-does-not-exist") + self.assertEqual(status, "Running the installer tests.") + self.assertEqual(metrics["context_actions"], 0) + + def test_file_and_python_commands_keep_grounded_targets(self): + status, _ = render({ + "ok": True, "errors": [], + "nodes": [command( + "Get-Content", "-Raw", "-LiteralPath", "local-session.json" + )], + }) + self.assertEqual(status, "Reading local-session.json.") + status, _ = render({ + "ok": True, "errors": [], + "nodes": [command("python", "tools/audit_failures.py", "--help")], + }) + self.assertEqual(status, "Running audit_failures.py with Python.") + def test_long_sequence_is_bounded_by_parsed_step_count(self): nodes = [ command("Get-Content"), command("Select-String"), @@ -73,7 +135,7 @@ def test_persistent_host_parses_pipeline_and_sequence(self): finally: backend.close() self.assertEqual( - status, "Reading a file, searching text, and checking Git status." + status, "Reading README.md, searching text, and checking Git status." ) self.assertEqual(metrics["mapped_actions"], 3) self.assertFalse(metrics["abstained"]) diff --git a/live-status/tests/test_service.py b/live-status/tests/test_service.py index abd4e91..6342eed 100644 --- a/live-status/tests/test_service.py +++ b/live-status/tests/test_service.py @@ -29,6 +29,14 @@ def generate(self, command): return self.reply, {} +class CwdBackend(FakeBackend): + source = "parser" + accepts_cwd = True + + def generate(self, command, *, cwd=None): + return f"Using {cwd}.", {} + + def post(url, body, token=None): headers = {"Content-Type": "application/json"} if token: @@ -80,6 +88,12 @@ def test_deterministic_backend_keeps_source_and_skips_best_of(self): out = service.summarize("git status", "powershell", None) self.assertEqual((out["status"], out["source"]), ("Checking Git status.", "parser")) + def test_context_backend_receives_cwd(self): + out = server.Service(CwdBackend("unused"), cache_size=0).summarize( + "git status", "powershell", "C:/repo" + ) + self.assertEqual((out["status"], out["source"]), ("Using C:/repo.", "parser")) + if __name__ == "__main__": unittest.main() diff --git a/live-status/tools/benchmark_parser_first.py b/live-status/tools/benchmark_parser_first.py index f79e63a..ad0663c 100644 --- a/live-status/tools/benchmark_parser_first.py +++ b/live-status/tools/benchmark_parser_first.py @@ -40,7 +40,7 @@ def main() -> None: try: backend.warm() for row in rows: - status, metrics = backend.generate(row["command"]) + status, metrics = backend.generate(row["command"], cwd=row.get("cwd")) outputs.append({ "id": row.get("id"), "session": row.get("session"), "complexity": row.get("complexity"), "tags": row.get("tags"), @@ -56,6 +56,8 @@ def main() -> None: literal = sum(item["metrics"]["literal_fallback"] for item in outputs) abstained = sum(item["metrics"]["abstained"] for item in outputs) mapped = sum(item["metrics"]["mapped_actions"] > 0 for item in outputs) + contextual = sum(item["metrics"]["context_actions"] > 0 for item in outputs) + rows_with_cwd = sum(bool(row.get("cwd")) for row in rows) evidence = sum( bool(item["metrics"]["facts"]) and all(fact["evidence"] for fact in item["metrics"]["facts"]) @@ -80,6 +82,8 @@ def pct(value: int) -> float: "literal_fallback_pct": pct(literal), "abstention_pct": pct(abstained), "mapped_action_coverage_pct": pct(mapped), + "context_resolved_pct": pct(contextual), + "rows_with_cwd": rows_with_cwd, "evidence_grounded_pct": pct(evidence), "validator_pass_pct": pct(validator), "latency_ms": { @@ -92,6 +96,7 @@ def pct(value: int) -> float: "Mapped coverage and grounding are mechanical measurements, not human-rated semantic accuracy.", "The prototype handles PowerShell only.", "Literal fallback repeats an AST-proven executable name without claiming its purpose.", + "Context resolution can only be measured on rows that retain their working directory.", ], } with (args.out / "outputs.jsonl").open("w", encoding="utf-8") as handle: From 557d1c69a24e25162817dffc1664b1736babc53c Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:14:03 -0400 Subject: [PATCH 3/4] Include Bun timeout in grounded test status --- live-status/inference/parser_first.py | 40 +++++++++++++++++-- .../research/2026-09-20-model-sweep.md | 5 ++- live-status/tests/test_parser_first.py | 11 +++-- 3 files changed, 47 insertions(+), 9 deletions(-) diff --git a/live-status/inference/parser_first.py b/live-status/inference/parser_first.py index 7230083..0f50219 100644 --- a/live-status/inference/parser_first.py +++ b/live-status/inference/parser_first.py @@ -168,6 +168,29 @@ def _option_values(args: list[str], options: set[str]) -> list[str]: return values +def _option_value(args: list[str], option: str) -> str | None: + option = option.lower() + for index, raw in enumerate(args): + value = _clean(raw) + lowered = value.lower() + if lowered == option and index + 1 < len(args): + return _clean(args[index + 1]) + if lowered.startswith(option + "="): + return value.split("=", 1)[1] + return None + + +def _milliseconds_phrase(value: str | None) -> str | None: + if value is None or not value.isdigit(): + return None + milliseconds = int(value) + if milliseconds and milliseconds % 60_000 == 0: + return f"{milliseconds // 60_000}-minute" + if milliseconds and milliseconds % 1_000 == 0: + return f"{milliseconds // 1_000}-second" + return f"{milliseconds}-millisecond" + + def _join_words(values: list[str]) -> str: if len(values) == 1: return values[0] @@ -248,7 +271,16 @@ def _context_description( ) intents = [intent for target in targets if (intent := _test_intent(cwd, target))] if len(intents) == 1: - return intents[0] + intent, evidence = intents[0] + timeout = _milliseconds_phrase(_option_value(args[1:], "--timeout")) + timeout_text = f" with a {timeout} timeout" if timeout else "" + if intent.startswith("testing that "): + action = f"running a Bun test{timeout_text} to verify that {intent[13:]}" + elif intent.startswith("testing protection against "): + action = f"running a Bun test{timeout_text} to verify protection against {intent[27:]}" + else: + action = f"running a Bun test{timeout_text} for {intent.removeprefix('testing ')}" + return action, evidence return None @@ -256,13 +288,15 @@ def _describe_runtime(name: str, elements: list[str]) -> tuple[str, bool]: args = [_clean(item) for item in elements[1:]] if name in ("bun", "bun.exe"): if args and args[0].lower() == "test": + timeout = _milliseconds_phrase(_option_value(args[1:], "--timeout")) + timeout_text = f" with a {timeout} timeout" if timeout else "" targets = _option_values( args[1:], {"--timeout", "--filter", "--preload", "--rerun-each"} ) labels = [label for item in targets if (label := _human_name(item))] if labels: - return f"running the {_join_words(labels)} tests", True - return "running tests with Bun", True + return f"running the {_join_words(labels)} tests{timeout_text}", True + return f"running tests with Bun{timeout_text}", True if len(args) >= 2 and args[0].lower() == "run": return f"running {args[1]} with Bun", True if args and (label := _human_name(args[0])): diff --git a/live-status/research/2026-09-20-model-sweep.md b/live-status/research/2026-09-20-model-sweep.md index e920a00..6ffea32 100644 --- a/live-status/research/2026-09-20-model-sweep.md +++ b/live-status/research/2026-09-20-model-sweep.md @@ -152,8 +152,9 @@ unmeasured until a held-out set captures command, working directory, and expecte The context resolver now follows a Bun test target inside the supplied working directory and uses a unique declared test name as evidence. For `bun test --timeout 90000 -test/codex-quest-dev-installer.test.ts`, the grounded output is “Testing that every development -install advances, seals and adds a new cache version without removing the old one.” The runtime +test/codex-quest-dev-installer.test.ts`, the grounded output is “Running a Bun test with a +90-second timeout to verify that every development install advances, seals and adds a new cache +version without removing the old one.” The runtime still does not solve Bash, infer opaque tools, or reliably identify the primary intent in long orchestration cells. Linguistic improvements are trained outside runtime: an explicitly selected LLM reviews simulated mapping combinations, and only human-accepted proposals change the map. diff --git a/live-status/tests/test_parser_first.py b/live-status/tests/test_parser_first.py index 0fdee34..71bbcbd 100644 --- a/live-status/tests/test_parser_first.py +++ b/live-status/tests/test_parser_first.py @@ -57,7 +57,8 @@ def test_bun_test_names_the_test_instead_of_the_runtime(self): )], }) self.assertEqual( - status, "Running the codex quest dev installer tests." + status, + "Running the codex quest dev installer tests with a 90-second timeout." ) self.assertTrue(metrics["fully_mapped"]) @@ -73,14 +74,16 @@ def test_bun_test_resolves_grounded_intent_from_cwd(self): { "ok": True, "errors": [], - "nodes": [command("bun", "test", "installer.test.ts")], + "nodes": [command( + "bun", "test", "--timeout=90000", "installer.test.ts" + )], }, cwd=directory, ) self.assertEqual( status, - "Testing that every install creates a sealed version without " - "deleting the old one.", + "Running a Bun test with a 90-second timeout to verify that every " + "install creates a sealed version without deleting the old one.", ) self.assertIn("context_evidence", metrics["facts"][0]) self.assertEqual(metrics["context_actions"], 1) From 7189ae066f11f287b313c6255b8589e2a2bd5734 Mon Sep 17 00:00:00 2001 From: Noisemaker111 <139656120+Noisemaker111@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:17:11 -0400 Subject: [PATCH 4/4] Preserve targets in agent command statuses --- live-status/ARCHITECTURE.md | 4 +++ live-status/inference/parser_first.py | 42 ++++++++++++++++++++++++++ live-status/tests/test_parser_first.py | 40 ++++++++++++++++++++++++ 3 files changed, 86 insertions(+) diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index a2dc10b..cf06b7f 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -47,6 +47,10 @@ in simulated combinations and asks an explicitly selected LLM for proposals; only reviewed, accepted phrases enter `linguistic_map.json`. Coverage and grounding are reported separately from human usefulness. + The product target is the actual command cells agents execute: test runners, Python modules + and scripts, file searches, Git/GitHub operations, and their meaningful targets and settings. + A passing mapping must retain those details; naming only the executable is a fallback, not a + useful result. Bash needs its own parser and is not covered by this PowerShell prototype. - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. diff --git a/live-status/inference/parser_first.py b/live-status/inference/parser_first.py index 0f50219..6450925 100644 --- a/live-status/inference/parser_first.py +++ b/live-status/inference/parser_first.py @@ -199,6 +199,10 @@ def _join_words(values: list[str]) -> str: return ", ".join(values[:-1]) + f", and {values[-1]}" +def _positional_values(args: list[str], options_with_values: set[str]) -> list[str]: + return _option_values(args, {value.lower() for value in options_with_values}) + + def _resolved_file(cwd: str | None, target: str) -> Path | None: if not cwd or target.startswith(("$", "@", "{", "(")): return None @@ -307,6 +311,17 @@ def _describe_runtime(name: str, elements: list[str]) -> tuple[str, bool]: return f"running {targets[0]} with Bun", True if name in ("python", "python.exe", "python3"): if len(args) >= 2 and args[0] == "-m": + module = args[1].lower() + module_args = args[2:] + if module == "unittest": + search = _option_value(module_args, "-s") + location = f" discovered in {search}" if search else "" + verbose = " with verbose output" if "-v" in module_args or "--verbose" in module_args else "" + return f"running Python unit tests{location}{verbose}", True + if module == "compileall": + targets = _positional_values(module_args, {"-d", "-ddir", "-j", "-o", "-p", "-s"}) + if targets: + return f"compiling Python files in {_join_words(targets)}", True return f"running {args[1]} with Python", True targets = _option_values(args, {"-W", "-X"}) if targets and targets[0] not in ("-c", "-"): @@ -343,12 +358,35 @@ def _describe(node: dict[str, Any]) -> tuple[str | None, bool]: if name == "git": args = _subcommands(elements) if args: + raw_args = [_clean(item) for item in elements[2:]] + if args[0] == "status" and any(item in raw_args for item in ("-b", "--branch")): + return "checking Git status and branch", True + if args[0] == "add": + targets = [value for value in raw_args if value != "--" and not value.startswith("-")] + labels = [label for value in targets if (label := _file_name(value))] + if labels: + return f"staging {_join_words(labels)}", True + if args[0] == "commit": + message = _option_value(raw_args, "-m") or _option_value(raw_args, "--message") + if message: + return f"committing the staged changes as {message}", True + if args[0] == "push": + targets = [value for value in raw_args if not value.startswith("-")] + if len(targets) >= 2: + return f"pushing branch {targets[1]} to {targets[0]}", True fallback = _GIT_ACTIONS.get(args[0], f"running git {args[0]}") return mapped_phrase(f"git:{args[0]}", fallback), args[0] in _GIT_ACTIONS return "running Git", False if name == "gh": args = _subcommands(elements) pair = tuple(args[:2]) + if pair == ("run", "watch"): + raw_args = [_clean(item) for item in elements[3:]] + run_id = next((item for item in raw_args if not item.startswith("-")), None) + interval = _option_value(raw_args, "--interval") + if run_id: + timing = f" every {interval} seconds" if interval and interval.isdigit() else "" + return f"watching GitHub Actions run {run_id}{timing}", True if pair in _GH_ACTIONS: key = " ".join(pair) return mapped_phrase(f"gh:{key}", _GH_ACTIONS[pair]), True @@ -357,6 +395,10 @@ def _describe(node: dict[str, Any]) -> tuple[str | None, bool]: return mapped_phrase("command:rg", "searching text"), True if name == "curl": return mapped_phrase("command:curl", "making an HTTP request"), True + if name == "select-string": + pattern = _option_value(elements[1:], "-pattern") + if pattern: + return f"searching text for {pattern}", True if name == "get-content": args = elements[1:] targets = _option_values( diff --git a/live-status/tests/test_parser_first.py b/live-status/tests/test_parser_first.py index 71bbcbd..c106ef9 100644 --- a/live-status/tests/test_parser_first.py +++ b/live-status/tests/test_parser_first.py @@ -110,6 +110,46 @@ def test_file_and_python_commands_keep_grounded_targets(self): }) self.assertEqual(status, "Running audit_failures.py with Python.") + def test_actual_agent_commands_keep_operational_targets(self): + cases = [ + ( + [command("python", "-m", "unittest", "discover", "-s", "live-status/tests", "-t", "live-status", "-v")], + "Running Python unit tests discovered in live-status/tests with verbose output.", + ), + ( + [command("python", "-m", "compileall", "-q", "live-status", "scripts", "skills", "experiments/command_model")], + "Compiling Python files in live-status, scripts, skills, and experiments/command_model.", + ), + ( + [command("gh", "run", "watch", "35556852505", "--interval", "10", "--exit-status")], + "Watching GitHub Actions run 35556852505 every 10 seconds.", + ), + ( + [command("Get-Content", "-LiteralPath", "README.md"), command("Select-String", "-Pattern", "parser")], + "Reading README.md and searching text for parser.", + ), + ] + for nodes, expected in cases: + with self.subTest(expected=expected): + status, metrics = render({"ok": True, "errors": [], "nodes": nodes}) + self.assertEqual(status, expected) + self.assertTrue(metrics["fully_mapped"]) + + def test_git_sequence_keeps_files_message_remote_and_branch(self): + status, _ = render({ + "ok": True, + "errors": [], + "nodes": [ + command("git", "add", "--", "live-status/inference/parser_first.py", "live-status/tests/test_parser_first.py"), + command("git", "commit", "-m", "Include Bun timeout"), + command("git", "push", "origin", "feature/parser-first-prototype-20260920"), + ], + }) + self.assertEqual( + status, + "Staging parser_first.py and test_parser_first.py, committing the staged changes as Include Bun timeout, and pushing branch feature/parser-first-prototype-20260920 to origin.", + ) + def test_long_sequence_is_bounded_by_parsed_step_count(self): nodes = [ command("Get-Content"), command("Select-String"),