diff --git a/live-status/ARCHITECTURE.md b/live-status/ARCHITECTURE.md index 744cebe..cf06b7f 100644 --- a/live-status/ARCHITECTURE.md +++ b/live-status/ARCHITECTURE.md @@ -21,7 +21,9 @@ evaluation/evaluate.py ──► validators + jev/Opus grading + latency/memory ──► evaluation/registry.json evaluation/active.py ──► student failures on unlabeled commands ──► teacher/judge ──► prefs.jsonl (DPO) - client ──► api/server.py ──► redact ─► cache ─► model (Ollama) ─► validate ─► heuristic fallback + client ──► api/server.py ──► redact ─► cache ─┬► PowerShell AST ─► mapped renderer ─┐ + └► model (Ollama) ───────────────────┤ + validate ─► heuristic fallback ``` ## Boundaries @@ -37,6 +39,18 @@ ## Why these choices +- **Parser-first PowerShell prototype.** `parser:powershell` keeps one local PowerShell + process warm, extracts commands with PowerShell's real AST, and renders only mapped facts. + With a working directory, a semantic resolver can ground a test action in its declared test + name. Parse errors and dynamic invocation abstain; unknown executables remain literal. It + uses no model or GPU at runtime. A development-only linguistic trainer places every mapping + in simulated combinations and asks an explicitly selected LLM for proposals; only reviewed, + accepted phrases enter `linguistic_map.json`. Coverage and grounding are reported separately + from human usefulness. + The product target is the actual command cells agents execute: test runners, Python modules + and scripts, file searches, Git/GitHub operations, and their meaningful targets and settings. + A passing mapping must retain those details; naming only the executable is a fallback, not a + useful result. Bash needs its own parser and is not covered by this PowerShell prototype. - **Heuristic as fallback, not fast path.** The deterministic describer answers in under a millisecond, but its confidence ≥ 0.9 outputs cover only 8.7% of executions and the judge rated just 19% of them ≥ 80 (they are correct but generic: "Reviewing the Git diff." for `git diff --stat`). The service therefore always asks the model and uses the heuristic when the model fails, times out or produces an invalid sentence; `--fast-path` re-enables the shortcut. - **Plain completion format.** The student learns `Command:\n…\n\nStatus: ` with no system prompt, so each request costs only the command's tokens. - **Ollama/llama.cpp for serving.** It already runs on this machine, serves GGUF at every quantization level, and keeps models warm. `llama-server` is supported by the same backend interface. diff --git a/live-status/README.md b/live-status/README.md index 7ab4204..c20efab 100644 --- a/live-status/README.md +++ b/live-status/README.md @@ -17,16 +17,33 @@ Details: [TRAINING.md](TRAINING.md), [EVALUATION.md](EVALUATION.md), ## Use it ```powershell +python live-status/cli.py serve --backend parser:powershell --port 8765 python live-status/cli.py serve --backend ollama:live-status-v1-qwen3-06b-lora-q4_k_m --port 8765 python live-status/api/client.py "git fetch origin && git status -sb" ``` +`parser:powershell` is the CPU-only prototype. It uses PowerShell's own AST, maps only +known commands, repeats unknown executable names literally, and abstains on malformed or +dynamic invocation. When `cwd` is supplied, it can resolve a named test file and report its +declared test purpose instead of restating the filename. It does not load a model or use the GPU. +The recent-session result and +its limitations are recorded in +[the model sweep](research/2026-09-20-model-sweep.md#parser-first-powershell-prototype). + +An LLM may improve the linguistic mapping during development without joining the runtime path. +`python live-status/linguistic_training.py simulate --output ` creates stable single, +pair, and triple phrase combinations. `propose --route --model + --output ` asks that explicitly selected CLIProxyAPI route and +model for grounded wording proposals. Every proposal +defaults to `accepted: false`; after review, `apply --input ` writes only accepted entries +to `inference/linguistic_map.json`. + `--best-of 4` samples several candidates and lets jev pick the best: +8 points of quality for ~1 s per request instead of ~0.1 s. `POST /v1/summarize-command` with `{"command": "...", "shell": "powershell", "cwd": "optional"}` returns `{"status": "..."}` (add `"debug": true` for source and latency). The service redacts -before inference, caches by normalised command, bounds concurrency, falls back to a +before inference, caches by shell, working directory, and normalised command, bounds concurrency, falls back to a deterministic describer on timeout or invalid output, and never logs commands. For remote use set `LIVE_STATUS_API_TOKEN` and pass `--tls-cert/--tls-key`; a non-loopback bind without a token is refused. `api/client.py` works unchanged against `https://my-server.example`. diff --git a/live-status/api/server.py b/live-status/api/server.py index 0f28f34..2b7e91d 100644 --- a/live-status/api/server.py +++ b/live-status/api/server.py @@ -59,7 +59,7 @@ def _log(self, row: dict) -> None: def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: t0 = time.perf_counter() red = redact(command[:MAX_COMMAND]) - key = sha(f"{shell}|{normalize_ws(red)}") + key = sha(f"{shell}|{cwd}|{normalize_ws(red)}") with self.lock: hit = self.cache.get(key) if hit is not None: @@ -69,12 +69,16 @@ def summarize(self, command: str, shell: str | None, cwd: str | None) -> dict: h_text, h_conf = describe(red, shell) if (self.fast_path and h_conf >= FAST_PATH) or self.backend is None: return self._done(h_text, "heuristic" if h_conf >= FAST_PATH else "fallback", key, t0, cache=h_conf >= FAST_PATH) - status, source = None, "model" + status = None + source = getattr(self.backend, "source", "model") if self.slots.acquire(timeout=self.queue_wait): try: model_in = red if len(red) <= MODEL_INPUT_CHARS else red[:MODEL_INPUT_CHARS] + " …" - status, _ = self.backend.generate(model_in) - if self.best_of > 1: + if getattr(self.backend, "accepts_cwd", False): + status, _ = self.backend.generate(model_in, cwd=cwd) + else: + status, _ = self.backend.generate(model_in) + if self.best_of > 1 and source == "model": picked, source = self._best_of(model_in, status) status = picked or status except Exception as exc: # timeouts, backend down @@ -197,7 +201,7 @@ def do_POST(self): def main(argv=None): p = argparse.ArgumentParser(prog="serve") p.add_argument("--backend", default=os.environ.get("LIVE_STATUS_BACKEND", "ollama:live-status"), - help="ollama: | llama-server: | hf:@ | none") + help="parser:powershell | ollama: | llama-server: | hf:@ | none") p.add_argument("--host", default="127.0.0.1") p.add_argument("--port", type=int, default=8765) p.add_argument("--timeout", type=float, default=8.0) diff --git a/live-status/inference/backends.py b/live-status/inference/backends.py index e169439..9433046 100644 --- a/live-status/inference/backends.py +++ b/live-status/inference/backends.py @@ -150,8 +150,12 @@ def _stop_ids(self) -> list[int]: def from_spec(spec: str): - """ollama:[:mode][:cpu] | llama-server: | hf:[@][:structured]""" + """Create a configured inference or deterministic parser backend.""" kind, _, rest = spec.partition(":") + if kind == "parser" and rest == "powershell": + from inference.parser_first import PowerShellAstBackend + + return PowerShellAstBackend() if kind == "ollama": mode, cpu = "plain", False if rest.endswith(":cpu"): diff --git a/live-status/inference/linguistic_map.json b/live-status/inference/linguistic_map.json new file mode 100644 index 0000000..5da9efe --- /dev/null +++ b/live-status/inference/linguistic_map.json @@ -0,0 +1,4 @@ +{ + "version": 1, + "action_overrides": {} +} diff --git a/live-status/inference/parser_first.py b/live-status/inference/parser_first.py new file mode 100644 index 0000000..6450925 --- /dev/null +++ b/live-status/inference/parser_first.py @@ -0,0 +1,559 @@ +"""CPU-only PowerShell AST backend with conservative deterministic rendering.""" + +from __future__ import annotations + +import json +import re +import shutil +import subprocess +import threading +import time +from pathlib import Path +from typing import Any + +ABSTENTION = "Running a PowerShell command." +_SIMPLE_NAME = re.compile(r"^[A-Za-z0-9_.+-]+$") +_COMMAND_ACTIONS = { + "add-content": "appending to a file", "compare-object": "comparing objects", + "compress-archive": "creating an archive", "convertfrom-json": "parsing JSON", + "convertto-json": "formatting JSON", "copy-item": "copying an item", + "expand-archive": "extracting an archive", "findstr": "searching text", + "foreach-object": "processing pipeline items", + "format-list": "formatting output as a list", + "format-table": "formatting output as a table", + "get-ciminstance": "reading system information", + "get-childitem": "listing directory contents", "get-content": "reading a file", + "get-command": "inspecting available commands", + "get-date": "reading the current date and time", + "get-filehash": "calculating a file hash", + "get-item": "inspecting an item", "get-process": "listing processes", + "get-nettcpconnection": "listing network connections", + "get-service": "listing services", "group-object": "grouping objects", + "invoke-restmethod": "calling a REST endpoint", + "invoke-webrequest": "making an HTTP request", + "join-path": "building a path", + "measure-object": "measuring objects", "move-item": "moving an item", + "new-item": "creating an item", "out-file": "writing a file", + "out-null": "discarding output", "pop-location": "restoring the previous directory", + "push-location": "saving and changing directories", + "remove-item": "removing an item", "rename-item": "renaming an item", + "resolve-path": "resolving a path", "select-object": "selecting object properties", + "select-string": "searching text", "set-content": "writing a file", + "set-location": "changing directories", "sort-object": "sorting objects", + "start-process": "starting a process", "stop-process": "stopping a process", + "start-sleep": "waiting", + "split-path": "extracting part of a path", + "tee-object": "copying pipeline output", "test-path": "checking whether a path exists", + "where-object": "filtering pipeline items", "write-error": "writing an error", + "write-output": "writing output", "write-warning": "writing a warning", +} +_ALIASES = { + "%": "foreach-object", "?": "where-object", "cat": "get-content", + "cd": "set-location", "cp": "copy-item", "del": "remove-item", + "dir": "get-childitem", "echo": "write-output", "gc": "get-content", + "gci": "get-childitem", "gi": "get-item", "ls": "get-childitem", + "mi": "move-item", "mv": "move-item", "ni": "new-item", + "pwd": "get-location", "ren": "rename-item", "rg.exe": "rg", + "ri": "remove-item", "rm": "remove-item", "sls": "select-string", + "select": "select-object", "type": "get-content", "curl.exe": "curl", +} +_GIT_ACTIONS = { + "add": "staging Git changes", "branch": "inspecting Git branches", + "checkout": "switching Git revisions", "clean": "cleaning the Git worktree", + "clone": "cloning a Git repository", "commit": "committing Git changes", + "diff": "inspecting Git changes", "fetch": "fetching Git updates", + "log": "reading Git history", "merge": "merging Git changes", + "ls-files": "listing tracked Git files", + "pull": "pulling Git updates", "push": "pushing Git changes", + "rebase": "rebasing Git changes", "remote": "inspecting Git remotes", + "reset": "resetting Git state", "restore": "restoring Git files", + "rev-parse": "inspecting Git revision data", "show": "showing a Git revision", + "status": "checking Git status", "switch": "switching Git branches", + "tag": "inspecting Git tags", "worktree": "managing Git worktrees", +} +_GH_ACTIONS = { + ("pr", "checks"): "checking pull request status", + ("pr", "create"): "creating a pull request", + ("pr", "diff"): "inspecting a pull request diff", + ("pr", "list"): "listing pull requests", + ("pr", "merge"): "merging a pull request", + ("pr", "view"): "viewing a pull request", + ("run", "view"): "viewing a workflow run", + ("run", "watch"): "watching a workflow run", +} +_RUNTIME_ACTIONS = { + "bun": "running Bun", "bun.exe": "running Bun", "bunx": "running Bun", + "cargo": "running Cargo", + "cmd": "running Command Prompt", "cmd.exe": "running Command Prompt", + "deno": "running Deno", "dotnet": "running .NET", "go": "running Go", + "java": "running Java", "make": "running Make", "node": "running Node.js", + "node.exe": "running Node.js", "npm": "running npm", "npx": "running npx", + "pnpm": "running pnpm", "pwsh": "running PowerShell", + "pwsh.exe": "running PowerShell", "powershell": "running Windows PowerShell", + "powershell.exe": "running Windows PowerShell", "python": "running Python", + "python.exe": "running Python", "python3": "running Python", + "pytest": "running Python tests", "uv": "running uv", "yarn": "running Yarn", +} +_LINGUISTIC_MAP_PATH = Path(__file__).with_name("linguistic_map.json") + + +def _load_linguistic_overrides(path: Path = _LINGUISTIC_MAP_PATH) -> dict[str, str]: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeError, ValueError): + return {} + values = data.get("action_overrides", {}) if isinstance(data, dict) else {} + if not isinstance(values, dict): + return {} + return { + str(key): str(value).strip() + for key, value in values.items() + if isinstance(value, str) and value.strip() + } + + +_LINGUISTIC_OVERRIDES = _load_linguistic_overrides() + + +def mapped_phrase(key: str, fallback: str) -> str: + """Return reviewed development-time wording, never model output at runtime.""" + return _LINGUISTIC_OVERRIDES.get(key, fallback) + + +def _clean(value: str) -> str: + value = value.strip() + if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": + return value[1:-1] + return value + + +def _human_name(value: str) -> str | None: + value = _clean(value).replace("\\", "/").rstrip("/") + if not value or value.startswith(("$", "@", "{", "(")): + return None + name = value.rsplit("/", 1)[-1] + for suffix in ( + ".test.ts", ".test.tsx", ".test.js", ".test.jsx", + ".spec.ts", ".spec.tsx", ".spec.js", ".spec.jsx", + ".py", ".ts", ".tsx", ".js", ".jsx", + ): + if name.lower().endswith(suffix): + name = name[:-len(suffix)] + break + name = re.sub(r"[-_]+", " ", name).strip() + return name or None + + +def _file_name(value: str) -> str | None: + value = _clean(value).replace("\\", "/").rstrip("/") + if not value or value.startswith(("$", "@", "{", "(")): + return None + return value.rsplit("/", 1)[-1] or None + + +def _option_values(args: list[str], options: set[str]) -> list[str]: + values = [] + skip = False + for raw in args: + value = _clean(raw) + if skip: + skip = False + continue + if value.lower() in options: + skip = True + continue + if value.startswith("-"): + continue + values.append(value) + return values + + +def _option_value(args: list[str], option: str) -> str | None: + option = option.lower() + for index, raw in enumerate(args): + value = _clean(raw) + lowered = value.lower() + if lowered == option and index + 1 < len(args): + return _clean(args[index + 1]) + if lowered.startswith(option + "="): + return value.split("=", 1)[1] + return None + + +def _milliseconds_phrase(value: str | None) -> str | None: + if value is None or not value.isdigit(): + return None + milliseconds = int(value) + if milliseconds and milliseconds % 60_000 == 0: + return f"{milliseconds // 60_000}-minute" + if milliseconds and milliseconds % 1_000 == 0: + return f"{milliseconds // 1_000}-second" + return f"{milliseconds}-millisecond" + + +def _join_words(values: list[str]) -> str: + if len(values) == 1: + return values[0] + if len(values) == 2: + return f"{values[0]} and {values[1]}" + return ", ".join(values[:-1]) + f", and {values[-1]}" + + +def _positional_values(args: list[str], options_with_values: set[str]) -> list[str]: + return _option_values(args, {value.lower() for value in options_with_values}) + + +def _resolved_file(cwd: str | None, target: str) -> Path | None: + if not cwd or target.startswith(("$", "@", "{", "(")): + return None + try: + root = Path(cwd).resolve() + candidate = (root / _clean(target)).resolve() + except (OSError, RuntimeError): + return None + if not root.is_dir(): + return None + try: + candidate.relative_to(root) + except ValueError: + return None + return candidate if candidate.is_file() else None + + +def _test_intent(cwd: str | None, target: str) -> tuple[str, str] | None: + path = _resolved_file(cwd, target) + if not path: + return None + try: + source = path.read_text(encoding="utf-8") + except (OSError, UnicodeError): + return None + names = [] + for match in re.finditer( + r"\b(?:test|it)\s*\(\s*(['\"])(.*?)\1", source + ): + name = normalize_ws(match.group(2)).rstrip(".") + if name and name not in names: + names.append(name) + if len(names) == 1: + return f"testing that {names[0]}", f"{path}:{names[0]}" + suites = [] + for match in re.finditer( + r"\bdescribe\s*\(\s*(['\"])(.*?)\1", source + ): + name = normalize_ws(match.group(2)).rstrip(".") + if name and name not in suites: + suites.append(name) + if len(suites) == 1: + return f"testing {suites[0]}", f"{path}:{suites[0]}" + core = re.search(r"@core-prevents\s+([^\r\n*]+)", source) + if core: + purpose = normalize_ws(core.group(1)).rstrip(".") + return f"testing protection against {purpose}", f"{path}:{purpose}" + return None + + +def normalize_ws(value: str) -> str: + return " ".join(value.split()) + + +def _context_description( + node: dict[str, Any], cwd: str | None +) -> tuple[str, str] | None: + raw_name = node.get("name") + if not raw_name: + return None + name = Path(str(raw_name)).name.lower() + if name not in ("bun", "bun.exe"): + return None + elements = [str(item) for item in node.get("elements", [])] + args = [_clean(item) for item in elements[1:]] + if not args or args[0].lower() != "test": + return None + targets = _option_values( + args[1:], {"--timeout", "--filter", "--preload", "--rerun-each"} + ) + intents = [intent for target in targets if (intent := _test_intent(cwd, target))] + if len(intents) == 1: + intent, evidence = intents[0] + timeout = _milliseconds_phrase(_option_value(args[1:], "--timeout")) + timeout_text = f" with a {timeout} timeout" if timeout else "" + if intent.startswith("testing that "): + action = f"running a Bun test{timeout_text} to verify that {intent[13:]}" + elif intent.startswith("testing protection against "): + action = f"running a Bun test{timeout_text} to verify protection against {intent[27:]}" + else: + action = f"running a Bun test{timeout_text} for {intent.removeprefix('testing ')}" + return action, evidence + return None + + +def _describe_runtime(name: str, elements: list[str]) -> tuple[str, bool]: + args = [_clean(item) for item in elements[1:]] + if name in ("bun", "bun.exe"): + if args and args[0].lower() == "test": + timeout = _milliseconds_phrase(_option_value(args[1:], "--timeout")) + timeout_text = f" with a {timeout} timeout" if timeout else "" + targets = _option_values( + args[1:], {"--timeout", "--filter", "--preload", "--rerun-each"} + ) + labels = [label for item in targets if (label := _human_name(item))] + if labels: + return f"running the {_join_words(labels)} tests{timeout_text}", True + return f"running tests with Bun{timeout_text}", True + if len(args) >= 2 and args[0].lower() == "run": + return f"running {args[1]} with Bun", True + if args and (label := _human_name(args[0])): + return f"running {label} with Bun", True + if name == "bunx": + targets = _option_values(args, {"--package"}) + if targets: + return f"running {targets[0]} with Bun", True + if name in ("python", "python.exe", "python3"): + if len(args) >= 2 and args[0] == "-m": + module = args[1].lower() + module_args = args[2:] + if module == "unittest": + search = _option_value(module_args, "-s") + location = f" discovered in {search}" if search else "" + verbose = " with verbose output" if "-v" in module_args or "--verbose" in module_args else "" + return f"running Python unit tests{location}{verbose}", True + if module == "compileall": + targets = _positional_values(module_args, {"-d", "-ddir", "-j", "-o", "-p", "-s"}) + if targets: + return f"compiling Python files in {_join_words(targets)}", True + return f"running {args[1]} with Python", True + targets = _option_values(args, {"-W", "-X"}) + if targets and targets[0] not in ("-c", "-"): + label = _file_name(targets[0]) + if label: + return f"running {label} with Python", True + return mapped_phrase(f"runtime:{name}", _RUNTIME_ACTIONS[name]), True + + +def _subcommands(elements: list[str]) -> list[str]: + values: list[str] = [] + skip_value = False + for raw in elements[1:]: + value = _clean(raw) + if skip_value: + skip_value = False + continue + if value in ("-C", "--git-dir", "--work-tree"): + skip_value = True + continue + if value.startswith("-"): + continue + values.append(value.lower()) + return values + + +def _describe(node: dict[str, Any]) -> tuple[str | None, bool]: + raw_name = node.get("name") + if not raw_name: + return None, False + basename = Path(str(raw_name)).name.lower() + name = _ALIASES.get(basename, basename) + elements = [str(item) for item in node.get("elements", [])] + if name == "git": + args = _subcommands(elements) + if args: + raw_args = [_clean(item) for item in elements[2:]] + if args[0] == "status" and any(item in raw_args for item in ("-b", "--branch")): + return "checking Git status and branch", True + if args[0] == "add": + targets = [value for value in raw_args if value != "--" and not value.startswith("-")] + labels = [label for value in targets if (label := _file_name(value))] + if labels: + return f"staging {_join_words(labels)}", True + if args[0] == "commit": + message = _option_value(raw_args, "-m") or _option_value(raw_args, "--message") + if message: + return f"committing the staged changes as {message}", True + if args[0] == "push": + targets = [value for value in raw_args if not value.startswith("-")] + if len(targets) >= 2: + return f"pushing branch {targets[1]} to {targets[0]}", True + fallback = _GIT_ACTIONS.get(args[0], f"running git {args[0]}") + return mapped_phrase(f"git:{args[0]}", fallback), args[0] in _GIT_ACTIONS + return "running Git", False + if name == "gh": + args = _subcommands(elements) + pair = tuple(args[:2]) + if pair == ("run", "watch"): + raw_args = [_clean(item) for item in elements[3:]] + run_id = next((item for item in raw_args if not item.startswith("-")), None) + interval = _option_value(raw_args, "--interval") + if run_id: + timing = f" every {interval} seconds" if interval and interval.isdigit() else "" + return f"watching GitHub Actions run {run_id}{timing}", True + if pair in _GH_ACTIONS: + key = " ".join(pair) + return mapped_phrase(f"gh:{key}", _GH_ACTIONS[pair]), True + return (f"running gh {args[0]}" if args else "running GitHub CLI"), False + if name in ("rg", "ripgrep"): + return mapped_phrase("command:rg", "searching text"), True + if name == "curl": + return mapped_phrase("command:curl", "making an HTTP request"), True + if name == "select-string": + pattern = _option_value(elements[1:], "-pattern") + if pattern: + return f"searching text for {pattern}", True + if name == "get-content": + args = elements[1:] + targets = _option_values( + args, + { + "-credential", "-delimiter", "-encoding", "-filter", + "-readcount", "-stream", "-tail", "-totalcount", + }, + ) + labels = [label for item in targets if (label := _file_name(item))] + if labels: + return f"reading {_join_words(labels)}", True + if name in _COMMAND_ACTIONS: + return mapped_phrase(f"command:{name}", _COMMAND_ACTIONS[name]), True + if name == "get-location": + return "reading the current directory", True + if name in _RUNTIME_ACTIONS: + return _describe_runtime(name, elements) + if _SIMPLE_NAME.fullmatch(name): + return f"running {name}", False + return None, False + + +def _metrics(nodes, errors, abstained, facts, semantic, literal, total_actions=0): + return { + "abstained": abstained, + "fully_mapped": not abstained and literal == 0 and semantic > 0, + "literal_fallback": literal > 0, + "mapped_actions": semantic, "literal_actions": literal, + "facts": facts, "parse_errors": errors, "ast_nodes": len(nodes), + "total_actions": total_actions, + "context_actions": sum("context_evidence" in fact for fact in facts), + } + + +def render( + parsed: dict[str, Any], *, cwd: str | None = None +) -> tuple[str, dict[str, Any]]: + nodes = list(parsed.get("nodes") or []) + errors = list(parsed.get("errors") or []) + if not parsed.get("ok") or errors or any(node.get("dynamic") for node in nodes): + return ABSTENTION, _metrics(nodes, errors, True, [], 0, 0) + facts: list[dict[str, str]] = [] + semantic = literal = 0 + seen: set[str] = set() + for node in nodes: + if node.get("kind") != "command": + continue + contextual = _context_description(node, cwd) + if contextual: + action, context_evidence = contextual + known = True + else: + action, known = _describe(node) + context_evidence = None + if not action or action in seen: + continue + seen.add(action) + fact = {"text": action, "evidence": str(node.get("evidence", ""))} + if context_evidence: + fact["context_evidence"] = context_evidence + facts.append(fact) + semantic += int(known) + literal += int(not known) + if not facts: + return ABSTENTION, _metrics(nodes, [], True, [], 0, 0) + total_actions = len(facts) + rendered_facts = facts[:3] + phrases = [fact["text"] for fact in rendered_facts] + if len(phrases) == 1: + body = phrases[0] + elif len(phrases) == 2: + body = f"{phrases[0]} and {phrases[1]}" + else: + body = ", ".join(phrases[:-1]) + f", and {phrases[-1]}" + if total_actions > len(rendered_facts): + body += f", plus {total_actions - len(rendered_facts)} more parsed steps" + return body[0].upper() + body[1:] + ".", _metrics( + nodes, [], False, rendered_facts, semantic, literal, total_actions + ) + + +class PowerShellAstBackend: + """Persistent PowerShell parser process; performs no model inference.""" + + name = "parser:powershell" + source = "parser" + accepts_cwd = True + + def __init__(self, executable: str | None = None): + self.executable = executable or shutil.which("pwsh") or shutil.which("powershell") + if not self.executable: + raise RuntimeError("PowerShell is required for parser:powershell") + self.host = Path(__file__).with_name("powershell_ast_host.ps1") + self._process: subprocess.Popen[str] | None = None + self._lock = threading.Lock() + self._request_id = 0 + + def _start(self) -> None: + if self._process and self._process.poll() is None: + return + self._process = subprocess.Popen( + [self.executable, "-NoLogo", "-NoProfile", "-NonInteractive", + "-File", str(self.host)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + text=True, encoding="utf-8", bufsize=1, + ) + + def parse(self, command: str) -> dict[str, Any]: + with self._lock: + self._start() + assert self._process and self._process.stdin and self._process.stdout + self._request_id += 1 + self._process.stdin.write(json.dumps( + {"id": self._request_id, "command": command}, ensure_ascii=False + ) + "\n") + self._process.stdin.flush() + line = self._process.stdout.readline() + if not line: + raise RuntimeError( + f"PowerShell AST host exited unexpectedly ({self._process.poll()})" + ) + response = json.loads(line) + if response.get("id") != self._request_id: + raise RuntimeError("PowerShell AST host returned a mismatched response") + return response + + def generate( + self, command: str, *, cwd: str | None = None + ) -> tuple[str, dict[str, Any]]: + started = time.perf_counter() + status, metrics = render(self.parse(command), cwd=cwd) + metrics["wall_s"] = time.perf_counter() - started + metrics["backend"] = "parser:powershell" + return status, metrics + + def warm(self) -> None: + self.parse("Get-Location") + + def close(self) -> None: + process, self._process = self._process, None + if not process: + return + if process.stdin: + process.stdin.close() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.terminate() + process.wait(timeout=2) + if process.stdout: + process.stdout.close() + + def __del__(self) -> None: + try: + self.close() + except Exception: + pass diff --git a/live-status/inference/powershell_ast_host.ps1 b/live-status/inference/powershell_ast_host.ps1 new file mode 100644 index 0000000..c32eb19 --- /dev/null +++ b/live-status/inference/powershell_ast_host.ps1 @@ -0,0 +1,45 @@ +param() + +$ErrorActionPreference = 'Stop' +while (($line = [Console]::In.ReadLine()) -ne $null) { + if ([string]::IsNullOrWhiteSpace($line)) { continue } + $request = $null + try { + $request = $line | ConvertFrom-Json + $tokens = $null + $parseErrors = $null + $ast = [System.Management.Automation.Language.Parser]::ParseInput( + [string]$request.command, [ref]$tokens, [ref]$parseErrors + ) + $nodes = @() + foreach ($commandAst in $ast.FindAll( + { param($node) $node -is [System.Management.Automation.Language.CommandAst] }, $true + )) { + $name = $commandAst.GetCommandName() + $nodes += [ordered]@{ + kind = 'command' + name = $name + dynamic = ($null -eq $name) + elements = @($commandAst.CommandElements | ForEach-Object { $_.Extent.Text }) + evidence = $commandAst.Extent.Text + start = $commandAst.Extent.StartOffset + } + } + $response = [ordered]@{ + id = $request.id + ok = ($parseErrors.Count -eq 0) + errors = @($parseErrors | ForEach-Object { $_.Message }) + nodes = @($nodes | Sort-Object start) + } + } + catch { + $response = [ordered]@{ + id = if ($null -ne $request) { $request.id } else { $null } + ok = $false + errors = @($_.Exception.Message) + nodes = @() + } + } + [Console]::Out.WriteLine(($response | ConvertTo-Json -Compress -Depth 8)) + [Console]::Out.Flush() +} diff --git a/live-status/linguistic_training.py b/live-status/linguistic_training.py new file mode 100644 index 0000000..cef0456 --- /dev/null +++ b/live-status/linguistic_training.py @@ -0,0 +1,156 @@ +"""Development-only LLM proposals for the deterministic linguistic map.""" +from __future__ import annotations + +import argparse +import json +import os +import random +import re +from pathlib import Path + +from inference.parser_first import ( + _COMMAND_ACTIONS, + _GH_ACTIONS, + _GIT_ACTIONS, + _LINGUISTIC_MAP_PATH, + _RUNTIME_ACTIONS, +) + + +def action_catalog() -> dict[str, str]: + catalog = {f"command:{key}": value for key, value in _COMMAND_ACTIONS.items()} + catalog.update({f"git:{key}": value for key, value in _GIT_ACTIONS.items()}) + catalog.update({f"gh:{' '.join(key)}": value for key, value in _GH_ACTIONS.items()}) + catalog.update({f"runtime:{key}": value for key, value in _RUNTIME_ACTIONS.items()}) + catalog["command:rg"] = "searching text" + catalog["command:curl"] = "making an HTTP request" + return dict(sorted(catalog.items())) + + +def simulate(seed: int = 20260920) -> list[dict]: + """Place every atomic phrase in stable single, pair, and triple contexts.""" + rng = random.Random(seed) + catalog = action_catalog() + keys = list(catalog) + rows = [] + for key, phrase in catalog.items(): + others = [candidate for candidate in keys if candidate != key] + pair_key, third_key = rng.sample(others, 2) + pair = [phrase, catalog[pair_key]] + triple = [catalog[third_key], phrase, catalog[pair_key]] + rows.append({ + "key": key, + "current_phrase": phrase, + "simulated_sentences": [ + phrase.capitalize() + ".", + f"{pair[0].capitalize()} and {pair[1]}.", + f"{triple[0].capitalize()}, {triple[1]}, and {triple[2]}.", + ], + }) + return rows + + +SYSTEM = """You improve an approved deterministic shell-status phrase map during development. +Each item contains a semantic key, its grounded atomic phrase, and simulated combinations. +Suggest an override only when it improves natural English while preserving exactly the same +meaning. Atomic phrases must start with a present participle, contain no period, command, +flag, invented purpose, or result. Return JSON only: {\"results\":[{\"key\":\"...\", +\"suggested_phrase\":\"...\",\"reason\":\"...\"}]}. Omit keys that should stay unchanged.""" + + +def validate_proposals(raw: list[dict], catalog: dict[str, str] | None = None) -> list[dict]: + catalog = catalog or action_catalog() + accepted = [] + for item in raw: + if not isinstance(item, dict): + continue + key = item.get("key") + phrase = item.get("suggested_phrase") + if key not in catalog or not isinstance(phrase, str): + continue + phrase = " ".join(phrase.split()).strip() + if not phrase or phrase.endswith(".") or any(char in phrase for char in "|;&`\r\n"): + continue + if not re.match(r"^[A-Za-z]+ing\b", phrase, re.IGNORECASE): + continue + accepted.append({ + "key": key, + "current_phrase": catalog[key], + "suggested_phrase": phrase, + "reason": str(item.get("reason", "")).strip(), + "accepted": False, + }) + return accepted + + +def propose(route: str, model: str, output: Path, seed: int = 20260920) -> dict: + from labeling.llm import chat, parse_json + + rows = simulate(seed) + text, usage = chat([ + {"role": "system", "content": SYSTEM}, + {"role": "user", "content": json.dumps(rows, ensure_ascii=False)}, + ], model=model) + parsed = parse_json(text) + raw = parsed.get("results", []) if isinstance(parsed, dict) else [] + record = { + "version": 1, + "route": route, + "model": model, + "seed": seed, + "usage": usage, + "proposals": validate_proposals(raw), + } + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(record, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + return record + + +def apply_reviewed(proposals: Path, destination: Path = _LINGUISTIC_MAP_PATH) -> dict: + source = json.loads(proposals.read_text(encoding="utf-8")) + current = json.loads(destination.read_text(encoding="utf-8")) + overrides = dict(current.get("action_overrides", {})) + catalog = action_catalog() + applied = [] + for item in source.get("proposals", []): + checked = validate_proposals([item], catalog) + if item.get("accepted") is True and checked: + proposal = checked[0] + overrides[proposal["key"]] = proposal["suggested_phrase"] + applied.append(proposal["key"]) + updated = {"version": 1, "action_overrides": dict(sorted(overrides.items()))} + temporary = destination.with_suffix(destination.suffix + ".tmp") + temporary.write_text(json.dumps(updated, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + os.replace(temporary, destination) + return {"applied": applied, "destination": str(destination)} + + +def main(argv=None) -> None: + parser = argparse.ArgumentParser(prog="linguistic-training") + sub = parser.add_subparsers(dest="command", required=True) + simulation = sub.add_parser("simulate") + simulation.add_argument("--output", type=Path, required=True) + simulation.add_argument("--seed", type=int, default=20260920) + proposal = sub.add_parser("propose") + proposal.add_argument("--route", required=True, help="user-confirmed provider route identity") + proposal.add_argument("--model", required=True, help="exact user-selected model on that route") + proposal.add_argument("--output", type=Path, required=True) + proposal.add_argument("--seed", type=int, default=20260920) + apply = sub.add_parser("apply") + apply.add_argument("--input", type=Path, required=True) + apply.add_argument("--destination", type=Path, default=_LINGUISTIC_MAP_PATH) + args = parser.parse_args(argv) + if args.command == "simulate": + rows = simulate(args.seed) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(rows, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + result = {"simulations": len(rows), "output": str(args.output)} + elif args.command == "propose": + result = propose(args.route, args.model, args.output, args.seed) + else: + result = apply_reviewed(args.input, args.destination) + print(json.dumps(result, ensure_ascii=False, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/live-status/research/2026-09-20-model-sweep.md b/live-status/research/2026-09-20-model-sweep.md index 4a55379..6ffea32 100644 --- a/live-status/research/2026-09-20-model-sweep.md +++ b/live-status/research/2026-09-20-model-sweep.md @@ -124,5 +124,40 @@ Correction: the judge request was sent to CLIProxyAPI with model `claude-opus-5` DPO passed 16 rows that current failed, while current passed 14 that DPO failed; both passed 41 and both failed 49. DPO's paired difference is +1.7 percentage points (bootstrap 95% interval -7.5 to +10.8; exact McNemar p=0.856), so the new run does not establish a winner between them. Current remains the deployment choice because it invents slightly less, passes the deterministic validator more often, and won the older blind comparison. DPO deserves targeted work: it was stronger on simple commands, long PowerShell, conditionals, nested quoting, and loops, but weaker on pipelines. The drop from the old 75% screen to 45.8% here is real distribution-shift evidence, dominated by omissions on long multi-action commands; neither model is accurate enough for an unqualified trusted UI. +### Parser-first PowerShell prototype + +The frozen recent-session sample is entirely PowerShell, so a CPU-only prototype now uses +PowerShell's own AST instead of asking a small model to rediscover shell structure. A +persistent parser process extracts command names and their exact source extents. The renderer +maps known commands, repeats unknown executable names without inferring their purpose, and +returns "Running a PowerShell command" for parse errors or dynamic invocation. Long cells +show three distinct actions and the exact number of remaining parsed steps. + +| Frozen recent-session sample (120 rows) | Result | +|---|---:| +| AST parse success | 97.5% | +| Fully mapped output | 84.2% | +| Literal executable fallback | 13.3% | +| Abstention | 2.5% | +| Rows retaining working directory / context resolved | 0 / 0 | +| Output facts with AST evidence | 97.5% | +| Deterministic validator pass | 100.0% | +| CPU latency p50 / p90 / max | 2.75 / 4.49 / 10.13 ms | + +These are coverage and grounding measurements, not semantic accuracy. The earlier 30-row audit +that called 26 outputs useful is invalid: its rubric counted executable or filename restatements +as useful even when they did not explain the operation. The frozen sample also omitted working +directories, so it cannot measure the new context resolver. Human usefulness is therefore +unmeasured until a held-out set captures command, working directory, and expected purpose. + +The context resolver now follows a Bun test target inside the supplied working directory and +uses a unique declared test name as evidence. For `bun test --timeout 90000 +test/codex-quest-dev-installer.test.ts`, the grounded output is “Running a Bun test with a +90-second timeout to verify that every development install advances, seals and adds a new cache +version without removing the old one.” The runtime +still does not solve Bash, infer opaque tools, or reliably identify the primary intent in long +orchestration cells. Linguistic improvements are trained outside runtime: an explicitly selected +LLM reviews simulated mapping combinations, and only human-accepted proposals change the map. + Weight pruning comes after semantic parity. Removing generic-domain weights without retraining can destroy useful syntax and language behavior, and zeroed weights do not guarantee lower latency in Ollama/llama.cpp. Quantizing the smaller dense base already produced the useful size and latency gain. If semantic grading passes and further compression is needed, distill the structured task into a smaller student and compare quantization-aware training before structured pruning. Relevant pruning evidence: [Iterative Structured Pruning with Multi-Domain Calibration](https://arxiv.org/abs/2601.02674) argues for hardware-friendly structured removal and mixed-domain calibration; [GPrune-LLM](https://arxiv.org/abs/2603.13418) shows that single-domain calibration can bias neuron importance; [Pruning as a Domain-specific LLM Extractor](https://arxiv.org/abs/2405.06275) supports task-calibrated pruning but does not establish that arbitrary out-of-domain weights can be safely deleted from a sub-1B model. diff --git a/live-status/tests/test_linguistic_training.py b/live-status/tests/test_linguistic_training.py new file mode 100644 index 0000000..dc807d8 --- /dev/null +++ b/live-status/tests/test_linguistic_training.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +import json +import tempfile +import unittest +from pathlib import Path + +from linguistic_training import action_catalog, apply_reviewed, simulate, validate_proposals + + +class LinguisticTrainingTests(unittest.TestCase): + def test_simulates_each_mapping_in_combinations(self): + rows = simulate(seed=7) + self.assertEqual(len(rows), len(action_catalog())) + self.assertTrue(all(len(row["simulated_sentences"]) == 3 for row in rows)) + self.assertTrue(any(row["key"] == "git:status" for row in rows)) + + def test_invalid_or_unknown_proposals_are_rejected(self): + proposals = validate_proposals([ + {"key": "git:status", "suggested_phrase": "reviewing Git status"}, + {"key": "git:status", "suggested_phrase": "review Git status"}, + {"key": "git:status", "suggested_phrase": "git status; delete everything"}, + {"key": "unknown:key", "suggested_phrase": "doing something"}, + ]) + self.assertEqual([item["suggested_phrase"] for item in proposals], ["reviewing Git status"]) + self.assertFalse(proposals[0]["accepted"]) + + def test_only_human_accepted_proposals_change_mapping(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + destination = root / "map.json" + destination.write_text('{"version":1,"action_overrides":{}}', encoding="utf-8") + proposals = root / "proposals.json" + proposals.write_text(json.dumps({"proposals": [ + {"key": "git:status", "suggested_phrase": "reviewing Git status", "accepted": True}, + {"key": "git:diff", "suggested_phrase": "reviewing Git changes", "accepted": False}, + ]}), encoding="utf-8") + result = apply_reviewed(proposals, destination) + saved = json.loads(destination.read_text(encoding="utf-8")) + self.assertEqual(result["applied"], ["git:status"]) + self.assertEqual(saved["action_overrides"], {"git:status": "reviewing Git status"}) + + +if __name__ == "__main__": + unittest.main() diff --git a/live-status/tests/test_parser_first.py b/live-status/tests/test_parser_first.py new file mode 100644 index 0000000..c106ef9 --- /dev/null +++ b/live-status/tests/test_parser_first.py @@ -0,0 +1,195 @@ +from __future__ import annotations + +import shutil +import tempfile +import unittest +from pathlib import Path + +from inference.backends import from_spec +from inference.parser_first import ABSTENTION, PowerShellAstBackend, render + + +def command(name: str, *elements: str): + return { + "kind": "command", "name": name, "dynamic": False, + "elements": [name, *elements], "evidence": " ".join([name, *elements]), + "start": 0, + } + + +class ParserFirstTests(unittest.TestCase): + def test_git_switch_does_not_invent_branch_creation(self): + status, metrics = render({ + "ok": True, "errors": [], "nodes": [command("git", "switch", "topic")] + }) + self.assertEqual(status, "Switching Git branches.") + self.assertTrue(metrics["fully_mapped"]) + + def test_parse_errors_and_dynamic_invocation_abstain(self): + status, metrics = render({ + "ok": False, "errors": ["Incomplete input"], "nodes": [] + }) + self.assertEqual(status, ABSTENTION) + self.assertTrue(metrics["abstained"]) + dynamic = command("", "&", "$tool") + dynamic["dynamic"] = True + status, metrics = render({ + "ok": True, "errors": [], "nodes": [dynamic] + }) + self.assertEqual(status, ABSTENTION) + self.assertTrue(metrics["abstained"]) + + def test_unknown_executable_is_literal_fallback(self): + status, metrics = render({ + "ok": True, "errors": [], "nodes": [command("widgetctl", "deploy")] + }) + self.assertEqual(status, "Running widgetctl.") + self.assertTrue(metrics["literal_fallback"]) + self.assertFalse(metrics["fully_mapped"]) + + def test_bun_test_names_the_test_instead_of_the_runtime(self): + status, metrics = render({ + "ok": True, + "errors": [], + "nodes": [command( + "bun", "test", "--timeout", "90000", + "test/codex-quest-dev-installer.test.ts", + )], + }) + self.assertEqual( + status, + "Running the codex quest dev installer tests with a 90-second timeout." + ) + self.assertTrue(metrics["fully_mapped"]) + + def test_bun_test_resolves_grounded_intent_from_cwd(self): + with tempfile.TemporaryDirectory() as directory: + target = Path(directory) / "installer.test.ts" + target.write_text( + "test('every install creates a sealed version without deleting " + "the old one', async () => {})\n", + encoding="utf-8", + ) + status, metrics = render( + { + "ok": True, + "errors": [], + "nodes": [command( + "bun", "test", "--timeout=90000", "installer.test.ts" + )], + }, + cwd=directory, + ) + self.assertEqual( + status, + "Running a Bun test with a 90-second timeout to verify that every " + "install creates a sealed version without deleting the old one.", + ) + self.assertIn("context_evidence", metrics["facts"][0]) + self.assertEqual(metrics["context_actions"], 1) + + def test_invalid_cwd_falls_back_without_failing(self): + status, metrics = render({ + "ok": True, "errors": [], + "nodes": [command("bun", "test", "installer.test.ts")], + }, cwd="Z:/a-directory-that-does-not-exist") + self.assertEqual(status, "Running the installer tests.") + self.assertEqual(metrics["context_actions"], 0) + + def test_file_and_python_commands_keep_grounded_targets(self): + status, _ = render({ + "ok": True, "errors": [], + "nodes": [command( + "Get-Content", "-Raw", "-LiteralPath", "local-session.json" + )], + }) + self.assertEqual(status, "Reading local-session.json.") + status, _ = render({ + "ok": True, "errors": [], + "nodes": [command("python", "tools/audit_failures.py", "--help")], + }) + self.assertEqual(status, "Running audit_failures.py with Python.") + + def test_actual_agent_commands_keep_operational_targets(self): + cases = [ + ( + [command("python", "-m", "unittest", "discover", "-s", "live-status/tests", "-t", "live-status", "-v")], + "Running Python unit tests discovered in live-status/tests with verbose output.", + ), + ( + [command("python", "-m", "compileall", "-q", "live-status", "scripts", "skills", "experiments/command_model")], + "Compiling Python files in live-status, scripts, skills, and experiments/command_model.", + ), + ( + [command("gh", "run", "watch", "35556852505", "--interval", "10", "--exit-status")], + "Watching GitHub Actions run 35556852505 every 10 seconds.", + ), + ( + [command("Get-Content", "-LiteralPath", "README.md"), command("Select-String", "-Pattern", "parser")], + "Reading README.md and searching text for parser.", + ), + ] + for nodes, expected in cases: + with self.subTest(expected=expected): + status, metrics = render({"ok": True, "errors": [], "nodes": nodes}) + self.assertEqual(status, expected) + self.assertTrue(metrics["fully_mapped"]) + + def test_git_sequence_keeps_files_message_remote_and_branch(self): + status, _ = render({ + "ok": True, + "errors": [], + "nodes": [ + command("git", "add", "--", "live-status/inference/parser_first.py", "live-status/tests/test_parser_first.py"), + command("git", "commit", "-m", "Include Bun timeout"), + command("git", "push", "origin", "feature/parser-first-prototype-20260920"), + ], + }) + self.assertEqual( + status, + "Staging parser_first.py and test_parser_first.py, committing the staged changes as Include Bun timeout, and pushing branch feature/parser-first-prototype-20260920 to origin.", + ) + + def test_long_sequence_is_bounded_by_parsed_step_count(self): + nodes = [ + command("Get-Content"), command("Select-String"), + command("git", "status"), command("Write-Output"), + command("Test-Path"), + ] + status, metrics = render({"ok": True, "errors": [], "nodes": nodes}) + self.assertEqual( + status, + "Reading a file, searching text, and checking Git status, " + "plus 2 more parsed steps.", + ) + self.assertEqual(metrics["total_actions"], 5) + self.assertEqual(len(metrics["facts"]), 3) + + @unittest.skipUnless( + shutil.which("pwsh") or shutil.which("powershell"), + "PowerShell unavailable", + ) + def test_persistent_host_parses_pipeline_and_sequence(self): + backend = PowerShellAstBackend() + try: + status, metrics = backend.generate( + "Get-Content README.md | Select-String TODO; git status" + ) + finally: + backend.close() + self.assertEqual( + status, "Reading README.md, searching text, and checking Git status." + ) + self.assertEqual(metrics["mapped_actions"], 3) + self.assertFalse(metrics["abstained"]) + + def test_public_backend_spec(self): + backend = from_spec("parser:powershell") + try: + self.assertIsInstance(backend, PowerShellAstBackend) + finally: + backend.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/live-status/tests/test_service.py b/live-status/tests/test_service.py index d843d01..6342eed 100644 --- a/live-status/tests/test_service.py +++ b/live-status/tests/test_service.py @@ -29,6 +29,14 @@ def generate(self, command): return self.reply, {} +class CwdBackend(FakeBackend): + source = "parser" + accepts_cwd = True + + def generate(self, command, *, cwd=None): + return f"Using {cwd}.", {} + + def post(url, body, token=None): headers = {"Content-Type": "application/json"} if token: @@ -72,6 +80,20 @@ def test_invalid_or_failed_model_output_falls_back(self): good = server.Service(FakeBackend("Building the incremental search index.")).summarize(cmd, "bash", None) self.assertEqual((good["status"], good["source"]), ("Building the incremental search index.", "model")) + def test_deterministic_backend_keeps_source_and_skips_best_of(self): + backend = FakeBackend("Checking Git status.") + backend.source = "parser" + service = server.Service(backend, cache_size=0, best_of=4) + with mock.patch.object(service, "_best_of", side_effect=AssertionError("model-only")): + out = service.summarize("git status", "powershell", None) + self.assertEqual((out["status"], out["source"]), ("Checking Git status.", "parser")) + + def test_context_backend_receives_cwd(self): + out = server.Service(CwdBackend("unused"), cache_size=0).summarize( + "git status", "powershell", "C:/repo" + ) + self.assertEqual((out["status"], out["source"]), ("Using C:/repo.", "parser")) + if __name__ == "__main__": unittest.main() diff --git a/live-status/tools/benchmark_parser_first.py b/live-status/tools/benchmark_parser_first.py new file mode 100644 index 0000000..ad0663c --- /dev/null +++ b/live-status/tools/benchmark_parser_first.py @@ -0,0 +1,112 @@ +"""Benchmark parser-first status generation without model or GPU calls.""" + +from __future__ import annotations + +import argparse +import json +import statistics +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from inference.parser_first import PowerShellAstBackend + + +def percentile(values: list[float], fraction: float) -> float: + ordered = sorted(values) + if not ordered: + return 0.0 + position = (len(ordered) - 1) * fraction + lower = int(position) + upper = min(lower + 1, len(ordered) - 1) + weight = position - lower + return ordered[lower] * (1 - weight) + ordered[upper] * weight + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--sample", type=Path, required=True) + parser.add_argument("--out", type=Path, required=True) + args = parser.parse_args() + rows = [ + json.loads(line) + for line in args.sample.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + args.out.mkdir(parents=True, exist_ok=True) + backend = PowerShellAstBackend() + outputs = [] + try: + backend.warm() + for row in rows: + status, metrics = backend.generate(row["command"], cwd=row.get("cwd")) + outputs.append({ + "id": row.get("id"), "session": row.get("session"), + "complexity": row.get("complexity"), "tags": row.get("tags"), + "command": row["command"], "status": status, "metrics": metrics, + }) + finally: + backend.close() + + count = len(outputs) + latencies = [item["metrics"]["wall_s"] * 1000 for item in outputs] + parse_success = sum(not item["metrics"]["parse_errors"] for item in outputs) + fully_mapped = sum(item["metrics"]["fully_mapped"] for item in outputs) + literal = sum(item["metrics"]["literal_fallback"] for item in outputs) + abstained = sum(item["metrics"]["abstained"] for item in outputs) + mapped = sum(item["metrics"]["mapped_actions"] > 0 for item in outputs) + contextual = sum(item["metrics"]["context_actions"] > 0 for item in outputs) + rows_with_cwd = sum(bool(row.get("cwd")) for row in rows) + evidence = sum( + bool(item["metrics"]["facts"]) + and all(fact["evidence"] for fact in item["metrics"]["facts"]) + for item in outputs + ) + validator = sum( + item["status"].endswith(".") and "\n" not in item["status"] + and all( + fact["text"].lower() in item["status"].lower() + for fact in item["metrics"]["facts"] + ) + for item in outputs + ) + + def pct(value: int) -> float: + return round(100 * value / count, 1) if count else 0.0 + + summary = { + "backend": "parser:powershell", "rows": count, + "ast_parse_success_pct": pct(parse_success), + "fully_mapped_coverage_pct": pct(fully_mapped), + "literal_fallback_pct": pct(literal), + "abstention_pct": pct(abstained), + "mapped_action_coverage_pct": pct(mapped), + "context_resolved_pct": pct(contextual), + "rows_with_cwd": rows_with_cwd, + "evidence_grounded_pct": pct(evidence), + "validator_pass_pct": pct(validator), + "latency_ms": { + "mean": round(statistics.fmean(latencies), 3) if latencies else 0.0, + "p50": round(percentile(latencies, 0.5), 3), + "p90": round(percentile(latencies, 0.9), 3), + "max": round(max(latencies), 3) if latencies else 0.0, + }, + "limitations": [ + "Mapped coverage and grounding are mechanical measurements, not human-rated semantic accuracy.", + "The prototype handles PowerShell only.", + "Literal fallback repeats an AST-proven executable name without claiming its purpose.", + "Context resolution can only be measured on rows that retain their working directory.", + ], + } + with (args.out / "outputs.jsonl").open("w", encoding="utf-8") as handle: + for output in outputs: + handle.write(json.dumps(output, ensure_ascii=False) + "\n") + (args.out / "summary.json").write_text( + json.dumps(summary, indent=2) + "\n", encoding="utf-8" + ) + print(json.dumps(summary, indent=2)) + + +if __name__ == "__main__": + main()