From d705cc4b5fbeb69ee3d98c22c5b88414d8f10c65 Mon Sep 17 00:00:00 2001 From: mjwsolo <19599997+mjwsolo@users.noreply.github.com> Date: Sun, 27 Sep 2026 22:31:16 +0100 Subject: [PATCH] docs: one interface; release notes and GitHub releases from the changelog README and the site no longer describe a second interface: the 0.3 docs snapshot and its version switcher are removed, the picker is described as it behaves since 0.4.4 (prompt first, picker on the first message or /models), and the plugin section covers the workspace snapshot and the current stop behaviour. publish.yml gains a GitHub-release job: on every tag it creates or updates the release with the CHANGELOG section for that version (scripts/release_notes.py) and attaches the wheel and sdist. --- .github/workflows/publish.yml | 38 ++++++ .gitignore | 1 - README.md | 4 +- scripts/release_notes.py | 53 ++++++++ website/astro.config.mjs | 10 -- website/package-lock.json | 116 +----------------- website/package.json | 3 +- website/src/components/Header.astro | 6 - .../content/docs/0.3/concepts/architecture.md | 45 ------- .../docs/0.3/concepts/network-boundary.md | 56 --------- .../docs/0.3/concepts/unified-memory.md | 35 ------ website/src/content/docs/0.3/contributing.md | 21 ---- website/src/content/docs/0.3/guides/mcp.md | 45 ------- .../src/content/docs/0.3/guides/offline.md | 39 ------ .../docs/0.3/guides/skills-and-hooks.md | 27 ---- website/src/content/docs/0.3/reference/cli.md | 51 -------- .../docs/0.3/reference/configuration.md | 61 --------- .../content/docs/0.3/reference/error-codes.md | 25 ---- .../docs/0.3/reference/jsonl-events.md | 74 ----------- .../docs/0.3/reference/slash-commands.md | 33 ----- .../docs/0.3/start-here/choose-a-model.md | 47 ------- .../docs/0.3/start-here/first-change.md | 75 ----------- .../docs/0.3/start-here/permissions.md | 17 --- .../src/content/docs/concepts/architecture.md | 7 +- .../content/docs/concepts/network-boundary.md | 7 -- .../content/docs/guides/skills-and-hooks.md | 2 +- website/src/content/docs/reference/cli.md | 10 +- .../content/docs/reference/configuration.md | 7 +- .../src/content/docs/reference/error-codes.md | 2 +- .../content/docs/start-here/choose-a-model.md | 2 +- .../content/docs/start-here/first-change.md | 4 +- .../content/docs/start-here/permissions.md | 2 +- website/src/pages/index.astro | 7 +- website/src/styles/docs.css | 2 +- 34 files changed, 110 insertions(+), 824 deletions(-) create mode 100755 scripts/release_notes.py delete mode 100644 website/src/content/docs/0.3/concepts/architecture.md delete mode 100644 website/src/content/docs/0.3/concepts/network-boundary.md delete mode 100644 website/src/content/docs/0.3/concepts/unified-memory.md delete mode 100644 website/src/content/docs/0.3/contributing.md delete mode 100644 website/src/content/docs/0.3/guides/mcp.md delete mode 100644 website/src/content/docs/0.3/guides/offline.md delete mode 100644 website/src/content/docs/0.3/guides/skills-and-hooks.md delete mode 100644 website/src/content/docs/0.3/reference/cli.md delete mode 100644 website/src/content/docs/0.3/reference/configuration.md delete mode 100644 website/src/content/docs/0.3/reference/error-codes.md delete mode 100644 website/src/content/docs/0.3/reference/jsonl-events.md delete mode 100644 website/src/content/docs/0.3/reference/slash-commands.md delete mode 100644 website/src/content/docs/0.3/start-here/choose-a-model.md delete mode 100644 website/src/content/docs/0.3/start-here/first-change.md delete mode 100644 website/src/content/docs/0.3/start-here/permissions.md diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index eff8633c..2b9fc71a 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -101,3 +101,41 @@ jobs: uses: pypa/gh-action-pypi-publish@release/v1 with: skip-existing: true + + # The GitHub release is created from the same tag, with the notes taken from + # the CHANGELOG section for that version (scripts/release_notes.py), and the + # wheel and sdist attached. Only on a tag push: creating the release fires a + # `release: published` event that re-runs this workflow, and that run must not + # try to create it again. + github-release: + name: GitHub release + needs: [publish] + if: github.event_name == 'push' && startsWith(github.ref, 'refs/tags/') + runs-on: ubuntu-latest + permissions: + contents: write + steps: + - uses: actions/checkout@v4 + with: + fetch-depth: 1 + - uses: actions/download-artifact@v4 + with: + name: dist + path: dist/ + - name: Release notes from CHANGELOG.md + run: | + python3 scripts/release_notes.py "${GITHUB_REF_NAME#v}" --check > notes.md + cat notes.md + - name: Create or update the release + env: + GH_TOKEN: ${{ github.token }} + run: | + tag="$GITHUB_REF_NAME" + pre="" + if [[ "${tag#v}" =~ [a-zA-Z] ]]; then pre="--prerelease"; fi + if gh release view "$tag" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then + gh release edit "$tag" --repo "$GITHUB_REPOSITORY" --title "$tag" --notes-file notes.md $pre + gh release upload "$tag" dist/* --repo "$GITHUB_REPOSITORY" --clobber + else + gh release create "$tag" dist/* --repo "$GITHUB_REPOSITORY" --title "$tag" --notes-file notes.md $pre --verify-tag + fi diff --git a/.gitignore b/.gitignore index e5079ac1..6739ecc3 100644 --- a/.gitignore +++ b/.gitignore @@ -68,7 +68,6 @@ learn-*/ *-learning-app*/ /logo/ todo-tui*/ -textual-todo-app/ prime-numbers/ prime_numbers/ primes/ diff --git a/README.md b/README.md index 2c00614c..d714406c 100644 --- a/README.md +++ b/README.md @@ -48,8 +48,6 @@ start typing. `/models` switches later. > Implement the retry decorator in retry.py so every test in test_retry.py passes. Then run: pytest -q ``` -`localcode --classic` opens the previous (0.3) interface. - Docs: [mjwsolo.github.io/localcode](https://mjwsolo.github.io/localcode/) ## What it does @@ -86,7 +84,7 @@ localcode recommends a model by your Mac's memory and marks it with a star. You Min RAM is the memory at which localcode will recommend the model. You can pick a heavier one by hand. DiffusionGemma is a research model that is never recommended automatically. -Measured on a MacBook Pro (M5 Max, 128 GB) with Qwen 3.6 35B-A3B UD-IQ2_M at a 131072-token context: about 89 tokens/s generation, about 1174 tokens/s prompt processing, and 12 to 15 seconds for a typical four-tool-call task. +Measured on a 128 GB Apple Silicon Mac with Qwen 3.6 35B-A3B UD-IQ2_M at a 131072-token context: about 89 tokens/s generation, about 1174 tokens/s prompt processing, and 12 to 15 seconds for a typical four-tool-call task. ## Network diff --git a/scripts/release_notes.py b/scripts/release_notes.py new file mode 100755 index 00000000..7cd27ddb --- /dev/null +++ b/scripts/release_notes.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python3 +"""Print the CHANGELOG.md section for one version, for the GitHub release body. + + scripts/release_notes.py 0.4.8 -> the "## 0.4.8 — ..." section body + scripts/release_notes.py 0.4.8 --check -> exit 1 when the section is missing or empty + +Headings are "## — " (or "## — unreleased"); the body +runs until the next "## " heading. The publish workflow calls this on every tag, +so a release always carries the notes that were written for it. +""" +from __future__ import annotations + +import re +import sys +from pathlib import Path + + +def section(text: str, version: str) -> str | None: + lines = text.splitlines() + start = None + for i, line in enumerate(lines): + if re.match(rf"^## {re.escape(version)}(\s|$)", line): + start = i + 1 + break + if start is None: + return None + body = [] + for line in lines[start:]: + if line.startswith("## "): + break + body.append(line) + return "\n".join(body).strip() + "\n" + + +def main(argv: list[str]) -> int: + if not argv or argv[0].startswith("-"): + print(__doc__, file=sys.stderr) + return 2 + version = argv[0].lstrip("v") + check = "--check" in argv[1:] + text = Path(__file__).resolve().parents[1].joinpath("CHANGELOG.md").read_text() + body = section(text, version) + if body is None or not body.strip(): + if check: + print(f"CHANGELOG.md has no notes for {version}", file=sys.stderr) + return 1 + body = f"See CHANGELOG.md for {version}.\n" + sys.stdout.write(body) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/website/astro.config.mjs b/website/astro.config.mjs index 84c2e88e..c625d8e7 100644 --- a/website/astro.config.mjs +++ b/website/astro.config.mjs @@ -1,7 +1,6 @@ // @ts-check import { defineConfig } from 'astro/config'; import starlight from '@astrojs/starlight'; -import starlightVersions from 'starlight-versions'; // Preview docs site for localcode. // @@ -15,15 +14,6 @@ export default defineConfig({ integrations: [ starlight({ title: 'localcode', - plugins: [ - // Versioned docs. The current tree documents the default interface - // (0.4+); `0.3` is a snapshot of the docs for the previous interface, - // still reachable with `localcode --classic`. - starlightVersions({ - current: { label: '0.4' }, - versions: [{ slug: '0.3', label: '0.3 · classic interface', redirect: 'root' }], - }), - ], description: 'An open-source coding agent that runs local models on Apple Silicon. No API key, and no remote inference unless you point it at one.', components: { diff --git a/website/package-lock.json b/website/package-lock.json index 16cbcb35..4dd1b11b 100644 --- a/website/package-lock.json +++ b/website/package-lock.json @@ -15,8 +15,7 @@ "@fontsource/inter": "^5.3.0", "@fontsource/martian-mono": "^5.3.0", "astro": "^7.2.3", - "sharp": "^0.35.3", - "starlight-versions": "^0.10.1" + "sharp": "^0.35.3" }, "devDependencies": { "@astrojs/check": "^0.9.4", @@ -3648,19 +3647,6 @@ "fast-string-width": "^3.0.2" } }, - "node_modules/fault": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/fault/-/fault-2.0.1.tgz", - "integrity": "sha512-WtySTkS4OKev5JtpHXnib4Gxiurzh5NCGvWrFaZ34m6JehfTUhKZvn9njTfw48t6JumVQOmrKqpmGcdwxnhqBQ==", - "license": "MIT", - "dependencies": { - "format": "^0.2.0" - }, - "funding": { - "type": "github", - "url": "https://github.com/sponsors/wooorm" - } - }, "node_modules/fdir": { "version": "6.5.0", "resolved": "https://registry.npmjs.org/fdir/-/fdir-6.5.0.tgz", @@ -3722,14 +3708,6 @@ "node": ">=20" } }, - "node_modules/format": { - "version": "0.2.2", - "resolved": "https://registry.npmjs.org/format/-/format-0.2.2.tgz", - "integrity": "sha512-wzsgA6WOq+09wrU1tsJ09udeR/YZRaeArL9e1wPbFg3GG2yDnC2ldKpxs4xunpFF9DgqCqOIra3bc1HWrJ37Ww==", - "engines": { - "node": ">=0.4.x" - } - }, "node_modules/fsevents": { "version": "2.3.3", "resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.3.tgz", @@ -4778,24 +4756,6 @@ "url": "https://opencollective.com/unified" } }, - "node_modules/mdast-util-frontmatter": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/mdast-util-frontmatter/-/mdast-util-frontmatter-2.0.1.tgz", - "integrity": "sha512-LRqI9+wdgC25P0URIJY9vwocIzCcksduHQ9OF2joxQoyTNVduwLAFUzjoopuRJbJAReaKrNQKAZKL3uCMugWJA==", - "license": "MIT", - "dependencies": { - "@types/mdast": "^4.0.0", - "devlop": "^1.0.0", - "escape-string-regexp": "^5.0.0", - "mdast-util-from-markdown": "^2.0.0", - "mdast-util-to-markdown": "^2.0.0", - "micromark-extension-frontmatter": "^2.0.0" - }, - "funding": { - "type": "opencollective", - "url": "https://opencollective.com/unified" - } - }, "node_modules/mdast-util-gfm": { "version": "3.1.0", "resolved": "https://registry.npmjs.org/mdast-util-gfm/-/mdast-util-gfm-3.1.0.tgz", @@ -5137,22 +5097,6 @@ "url": "https://opencollective.com/unified" } }, - "node_modules/micromark-extension-frontmatter": { - "version": "2.0.0", - "resolved": "https://registry.npmjs.org/micromark-extension-frontmatter/-/micromark-extension-frontmatter-2.0.0.tgz", - "integrity": "sha512-C4AkuM3dA58cgZha7zVnuVxBhDsbttIMiytjgsM2XbHAB2faRVaHRle40558FBN+DJcrLNCoqG5mlrpdU4cRtg==", - "license": "MIT", - "dependencies": { - "fault": "^2.0.0", - "micromark-util-character": "^2.0.0", - "micromark-util-symbol": "^2.0.0", - "micromark-util-types": "^2.0.0" - }, - "funding": { - "type": "opencollective", - "url": "https://opencollective.com/unified" - } - }, "node_modules/micromark-extension-gfm": { "version": "3.0.0", "resolved": "https://registry.npmjs.org/micromark-extension-gfm/-/micromark-extension-gfm-3.0.0.tgz", @@ -6463,22 +6407,6 @@ "url": "https://opencollective.com/unified" } }, - "node_modules/remark": { - "version": "15.0.1", - "resolved": "https://registry.npmjs.org/remark/-/remark-15.0.1.tgz", - "integrity": "sha512-Eht5w30ruCXgFmxVUSlNWQ9iiimq07URKeFS3hNc8cUWy1llX4KDWfyEDZRycMc+znsN9Ux5/tJ/BFdgdOwA3A==", - "license": "MIT", - "dependencies": { - "@types/mdast": "^4.0.0", - "remark-parse": "^11.0.0", - "remark-stringify": "^11.0.0", - "unified": "^11.0.0" - }, - "funding": { - "type": "opencollective", - "url": "https://opencollective.com/unified" - } - }, "node_modules/remark-directive": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/remark-directive/-/remark-directive-4.0.0.tgz", @@ -6495,22 +6423,6 @@ "url": "https://opencollective.com/unified" } }, - "node_modules/remark-frontmatter": { - "version": "5.0.0", - "resolved": "https://registry.npmjs.org/remark-frontmatter/-/remark-frontmatter-5.0.0.tgz", - "integrity": "sha512-XTFYvNASMe5iPN0719nPrdItC9aU0ssC4v14mH1BCi1u0n1gAocqcujWUrByftZTbLhRtiKRyjYTSIOcr69UVQ==", - "license": "MIT", - "dependencies": { - "@types/mdast": "^4.0.0", - "mdast-util-frontmatter": "^2.0.0", - "micromark-extension-frontmatter": "^2.0.0", - "unified": "^11.0.0" - }, - "funding": { - "type": "opencollective", - "url": "https://opencollective.com/unified" - } - }, "node_modules/remark-gfm": { "version": "4.0.1", "resolved": "https://registry.npmjs.org/remark-gfm/-/remark-gfm-4.0.1.tgz", @@ -6902,31 +6814,6 @@ "url": "https://github.com/sponsors/wooorm" } }, - "node_modules/starlight-versions": { - "version": "0.10.1", - "resolved": "https://registry.npmjs.org/starlight-versions/-/starlight-versions-0.10.1.tgz", - "integrity": "sha512-v0YJBd1eyAHWmCa9uiy2jYxULjcRLR1BYKq2taydTR+Ai2bcxJEi5hbqQKDwwAdnN0E6F0NxUCJSyBtZ0p/8qA==", - "license": "MIT", - "dependencies": { - "@pagefind/default-ui": "^1.3.0", - "github-slugger": "^2.0.0", - "mdast-util-mdx-jsx": "^3.2.0", - "mdast-util-mdxjs-esm": "^2.0.1", - "remark": "^15.0.1", - "remark-directive": "^4.0.0", - "remark-frontmatter": "^5.0.0", - "remark-mdx": "^3.1.1", - "unist-util-visit": "^5.1.0", - "vfile": "^6.0.3", - "yaml": "^2.8.2" - }, - "engines": { - "node": ">=22.12.0" - }, - "peerDependencies": { - "@astrojs/starlight": ">=0.39.0" - } - }, "node_modules/stream-replace-string": { "version": "2.0.0", "resolved": "https://registry.npmjs.org/stream-replace-string/-/stream-replace-string-2.0.0.tgz", @@ -7983,6 +7870,7 @@ "version": "2.9.0", "resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.0.tgz", "integrity": "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==", + "devOptional": true, "license": "ISC", "bin": { "yaml": "bin.mjs" diff --git a/website/package.json b/website/package.json index 70088fef..c81c4106 100644 --- a/website/package.json +++ b/website/package.json @@ -26,8 +26,7 @@ "@fontsource/inter": "^5.3.0", "@fontsource/martian-mono": "^5.3.0", "astro": "^7.2.3", - "sharp": "^0.35.3", - "starlight-versions": "^0.10.1" + "sharp": "^0.35.3" }, "devDependencies": { "@astrojs/check": "^0.9.4", diff --git a/website/src/components/Header.astro b/website/src/components/Header.astro index 59572523..10acbbe4 100644 --- a/website/src/components/Header.astro +++ b/website/src/components/Header.astro @@ -11,23 +11,17 @@ import Search from '@astrojs/starlight/components/Search.astro'; import MobileMenuToggle from '@astrojs/starlight/components/MobileMenuToggle.astro'; import SiteNav from './SiteNav.astro'; -// The docs version switcher (starlight-versions). The plugin normally mounts it -// through Starlight's ThemeSelect slot, which this site overrides with nothing, -// so it is rendered here next to search instead. -import VersionSelect from 'starlight-versions/components/VersionSelect.astro'; ---
-
diff --git a/website/src/content/docs/0.3/concepts/architecture.md b/website/src/content/docs/0.3/concepts/architecture.md deleted file mode 100644 index 37deb67c..00000000 --- a/website/src/content/docs/0.3/concepts/architecture.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -title: Architecture -description: The pieces localcode is made of, from the TUI down to the inference server. -slug: 0.3/concepts/architecture ---- - -## The stack - -This is the default setup. The rest of this page describes it: - -
-
- Textual TUI - → - Agent loop - → - Tools (read / edit / bash / search / MCP) -
- - ↓ -
runtime.base\_url (default http://localhost:8081)
- ↓ -
llama-server started by localcode
llama.cpp fork + TurboQuant KV compression
- ↓ -
GGUF weights on disk
-
- -* **TUI** - the main product interface. Setup, mode choice, the model picker, and chat are all screens in one Textual app. -* **Agent loop** - the model creates tool calls, the tools run, and the results go back to the model. Turn state, todos, and goal context continue across user messages. -* **Tools** - file reading and editing, glob/grep, shell commands, project checks, syntax checks, code navigation and symbol inspection, notebook editing, app launching, the two network tools, and any MCP tools you have configured. -* **Inference server** - by default, localcode starts its own `llama-server` (the binary included in the wheel) at `localhost:8081`. - -## Built specifically for small models - -localcode is designed specifically to enable high-performance agentic coding with local models on consumer hardware. The prompts, the agent loop, and the model server are all tuned for small quantised models rather than a frontier model: - -* **Prompts tuned for small models** - the system prompt runs a plan-then-execute loop: lay out the steps, keep exactly one in progress, and require evidence before a task counts as done, instead of assuming the model self-organises. -* **Finishes the whole task** - the loop keeps working until the goal is actually done, so the model does not stop mid-task and call it finished. -* **Tool-call repair** - malformed JSON arguments and extra spaces in tool names are fixed instead of failing the round. -* **Recovery modes** - separate paths handle cut-off tool calls and reasoning loops, each with its own exit reason in the event stream. -* **Long context on 16 GB** - the llama.cpp fork compresses the KV cache with TurboQuant (about 3.8x smaller than f16), so long contexts fit on small machines. -* **Fast multi-turn** - the server snapshots its state at turn boundaries, so the next turn reuses the prefix instead of re-reading it. -* **Speculative decoding** - an optional draft model speeds up generation without changing the output. -* **Hidden reasoning is off by default** - turn it on per model with `/thinking`; models without a reasoning channel say so instead of silently ignoring it. -* **Syntax checks before shell runs** - tree-sitter catches broken edits before they run. diff --git a/website/src/content/docs/0.3/concepts/network-boundary.md b/website/src/content/docs/0.3/concepts/network-boundary.md deleted file mode 100644 index 3e3df44a..00000000 --- a/website/src/content/docs/0.3/concepts/network-boundary.md +++ /dev/null @@ -1,56 +0,0 @@ ---- -title: Network Boundary -description: The network paths localcode can use, what triggers them, and what - stays on your Mac by default. -slug: 0.3/concepts/network-boundary ---- - -**With the default setup, inference runs on your Mac. It needs no API key or -model provider.** A few features do use the network, and one setting can move -inference off your Mac. This page lists each one. - -## What stays on your Mac by default - -* **Inference.** Generation sends requests to `/v1/chat/completions`. - By default that is a `llama-server` process at `http://localhost:8081`, using - the binary included in the wheel. -* **Your files, prompts and edits.** The agent reads and writes your working - tree directly. -* **Session and event logs.** `/.localcode/events.jsonl` is a local, - append-only record of tool calls, turn boundaries and server lifecycle. It is - never uploaded. Set `LOCALCODE_TELEMETRY=0` to turn off the UI turn-trace - records inside it. - -There is no analytics endpoint, usage reporting or version check. - -## Inference endpoint - -`runtime.base_url` in `~/.localcode/config.toml`, or the `LOCALCODE_BASE_URL` -environment variable, controls where chat completions are sent. **It accepts -any URL and is not limited to localhost.** If you point it at a remote server, -localcode sends every prompt it builds to that server: your message, the file -contents gathered for context, tool results and the model's replies. Check the -current value with `/status`. - -## Where localcode uses the network - -| # | What | Where it goes | When | -| --- | --- | --- | --- | -| 1 | **Connectivity probe** | TCP connect to `1.1.1.1:443` | Automatic, at most once per turn (cached 30 s). No payload - it opens and closes the connection to check for internet before a download | -| 2 | **Model download** | `huggingface.co` | On first launch, and whenever you choose a model you do not have yet | -| 3 | **Quant browsing** | Hugging Face repo tree API | Only when you browse other quantisations in the model picker. Cached | -| 4 | **`llama-server` fallback binary** | `github.com/mjwsolo/localcode` Releases | Only when the included binary cannot be used. Refused if TLS verification fails | -| 5 | **Voice model** (optional `voice` extra) | `huggingface.co/ggerganov/whisper.cpp` | The first time you enable voice input | -| 6 | **Voice output voices** (optional `voice` extra) | `huggingface.co/rhasspy/piper-voices` | The first time a speech voice is used | -| 7 | **`web_search` tool** | DuckDuckGo | Whenever the model calls it | -| 8 | **`web_fetch` tool** | The URL named in the call | Whenever the model calls it | -| 9 | **Skill install from a URL** | That URL | Only when you install one that way | -| 10 | **MCP servers** | Wherever you pointed them | Whenever the model calls one of their tools | -| 11 | **Shell commands** | Wherever the command goes | Whenever a `bash` or `background_process` call runs | -| 12 | **Custom inference endpoint** | Whatever `base_url` names | Every turn, if you changed it from the default | - -The network tools (`web_search`, `web_fetch`, and MCP tools) run without a -confirmation prompt at every autonomy level. `localcode run` (headless) forces -`full_auto`, so a headless run can make network requests without asking. See -[Permissions](/localcode/0.3/start-here/permissions) for what does and does not -prompt. diff --git a/website/src/content/docs/0.3/concepts/unified-memory.md b/website/src/content/docs/0.3/concepts/unified-memory.md deleted file mode 100644 index c4af1273..00000000 --- a/website/src/content/docs/0.3/concepts/unified-memory.md +++ /dev/null @@ -1,35 +0,0 @@ ---- -title: Unified Memory -description: Why RAM, not GPU class, decides which model you can run. -slug: 0.3/concepts/unified-memory ---- - -On Apple Silicon, the CPU and GPU use the same memory pool. There is no separate VRAM. Model weights, the KV cache, your editor, your browser, and macOS all use this shared memory. This is why localcode bases its model recommendation only on memory size. - -## The memory budget - -localcode uses about **55% of unified memory** for model weights. The remaining memory must cover: - -* **KV cache** - It grows as the context gets longer and can use a lot of laptop memory. -* **Activations** - These are temporary, but they still use memory. -* **macOS and everything else you have open.** - -localcode recommends the most capable production-ready model whose weights fit within this budget. See [Choose a Model](/localcode/0.3/start-here/choose-a-model) for the recommended models. - -## Keeping the KV cache small - -localcode uses a llama.cpp fork with **TurboQuant KV cache compression**. It uses asymmetric `q8_0`-K and `turbo4`-V quantisation. According to the fork, this combination is about 3.8× smaller than `f16`. This figure describes the quantisation method, not your machine's performance. - -Compressing the cache leaves more memory for a longer context. This is why you can configure the K/V cache types with `kv_cache_type_k` and `kv_cache_type_v`. - -## Why Mixture-of-Experts models suit mid-range machines - -Memory bandwidth limits decode speed on Apple Silicon. This means speed depends on how many bytes the system must read for each token. - -An MoE model uses only a few billion of its parameters for each token. It therefore reads much less data per token than a dense model of the same total size. This is the trade-off behind the mid-range recommendations. The actual tokens per second depend on your chip. localcode does not publish throughput figures. - -## Memory headroom in practice - -* Close memory-heavy apps before a long session, including browsers, IDEs, and Docker. -* A smaller quantisation leaves more memory for a longer context. -* If there is not enough memory to launch, localcode raises `E1010` instead of thrashing. diff --git a/website/src/content/docs/0.3/contributing.md b/website/src/content/docs/0.3/contributing.md deleted file mode 100644 index 5ddace53..00000000 --- a/website/src/content/docs/0.3/contributing.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -title: Contributing -description: How to contribute to localcode. -slug: 0.3/contributing ---- - -localcode is Apache-2.0 and contributions are welcome. Keep changes focused on the local-first workflow: easier setup, better reliability and clarity, and no broad complexity unless it clearly helps users. - -The contributor guides live in the repository: [`CONTRIBUTING.md`](https://github.com/mjwsolo/localcode/blob/main/CONTRIBUTING.md) (development setup, tests, and PR steps) and [`SECURITY.md`](https://github.com/mjwsolo/localcode/blob/main/SECURITY.md). - -## Working on these docs - -This site is in `website/` (Astro + Starlight): - -```sh -cd website -npm install -npm run dev -``` - -See `website/README.md` for full preview instructions. diff --git a/website/src/content/docs/0.3/guides/mcp.md b/website/src/content/docs/0.3/guides/mcp.md deleted file mode 100644 index e2e60f7f..00000000 --- a/website/src/content/docs/0.3/guides/mcp.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -title: MCP -description: Connect Model Context Protocol servers to localcode. -slug: 0.3/guides/mcp ---- - -localcode is an MCP client. The servers you add give the agent access to their tools alongside the built-in tools. - -## Add a server - -Add servers to the `mcpServers` key in `~/.localcode/mcp.json`, then run `/mcp reload` in the TUI. This is the same file format other MCP clients use: - -```json -{ - "mcpServers": { - "filesystem": { - "command": "npx", - "args": ["-y", "@modelcontextprotocol/server-filesystem", "/Users/you/work"] - } - } -} -``` - -A remote server that needs a token takes it as a header: - -```json -{ - "mcpServers": { - "github": { - "transport": "http", - "url": "https://api.githubcopilot.com/mcp/", - "headers": { "Authorization": "Bearer YOUR_TOKEN" } - } - } -} -``` - -Set `LOCALCODE_HOME` to read the file from another location. The MCP SDK supports stdio, HTTP, and SSE connections. - -## From the TUI - -```text -/mcp # list configured servers and their tools -/mcp reload # re-read mcp.json and reconnect after an edit -``` diff --git a/website/src/content/docs/0.3/guides/offline.md b/website/src/content/docs/0.3/guides/offline.md deleted file mode 100644 index cd1213ed..00000000 --- a/website/src/content/docs/0.3/guides/offline.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -title: Offline -description: What works with no network, and what doesn't. -slug: 0.3/guides/offline ---- - -localcode is local-first. After a model is on disk, the things you use most work -with no connection. A few features still need the network. - -## Works offline - -* Model inference, while `runtime.base_url` points at the local server. -* Reading, editing and writing files. -* `grep`, `glob`, `list_files`, code navigation and symbol inspection. -* Shell commands that do not reach the network. -* Syntax checks and your repo's own test and lint commands. -* Session resume and the project event log. - -## Needs the network - -* The first model download. This is the one blocking step - do it before going - offline. -* Browsing other quantisations in the model picker (cached results still show). -* `web_search` and `web_fetch`. -* Remote MCP servers, and any shell command that downloads something. -* Enabling voice for the first time (it downloads speech models). - -## Preparing for an offline session - -1. Start localcode once while connected and let the recommended model finish - downloading. -2. Run `/status` to confirm the server is healthy and the model is loaded. -3. Install any project dependencies first - a `pip install` or `npm install` - fails offline like any other network command. - -When offline, `web_search` and `web_fetch` return errors. The agent uses the -failed result as context and keeps going. See -[Network Boundary](/localcode/0.3/concepts/network-boundary) for every path that -uses the network. diff --git a/website/src/content/docs/0.3/guides/skills-and-hooks.md b/website/src/content/docs/0.3/guides/skills-and-hooks.md deleted file mode 100644 index 5db6cf8e..00000000 --- a/website/src/content/docs/0.3/guides/skills-and-hooks.md +++ /dev/null @@ -1,27 +0,0 @@ ---- -title: Skills -description: Reusable prompt templates the model can load into a task. -slug: 0.3/guides/skills-and-hooks ---- - -## Skills - -A skill is a Markdown file. It is a reusable prompt template that the model can use. Frontmatter is optional. If a file has no frontmatter, its full body is used. - -localcode finds skills in several places. This means it can use skills you wrote for another agent: - -```text -~/.localcode/skills/ -/.localcode/skills/ -/.agents/skills/ ~/.agents/skills/ -/.claude/skills/ ~/.claude/skills/ -/.opencode/skills/ ~/.config/opencode/skills/ -``` - -In the TUI: - -```text -/skills # list loaded skills and where each came from -``` - -You can also install skills from a URL. This fetches data from the network. See [Network Boundary](/localcode/0.3/concepts/network-boundary). diff --git a/website/src/content/docs/0.3/reference/cli.md b/website/src/content/docs/0.3/reference/cli.md deleted file mode 100644 index f2a01018..00000000 --- a/website/src/content/docs/0.3/reference/cli.md +++ /dev/null @@ -1,51 +0,0 @@ ---- -title: CLI -description: Every flag and subcommand localcode accepts. -slug: 0.3/reference/cli ---- - -```text -localcode [--profile P] [--model TAG] [--resume SESSION_ID] [-c DIR] - [--preview-screen SCREEN] -localcode run --goal "..." [options] -``` - -Run `localcode` by itself to start the TUI. The TUI is the product. It includes first-run setup, configuration, and model management. **There is no `localcode setup` subcommand.** There is also no benchmark subcommand or screen. The speeds in the model picker are [calculated estimates](/localcode/0.3/start-here/choose-a-model#the-toks-numbers-in-the-model-picker), not measurements. - -## Global options - -| Flag | Description | -| --- | --- | -| `--profile P` | Gemma 4 profile: `e2b`, `e4b`, `26b-laptop`, `26b-moe`, `31b` | -| `--model TAG` | Exact local runtime model tag | -| `--resume SESSION_ID` | Continue an earlier session. Use `--resume last` for the most recent session. Session IDs appear when you exit | -| `-c`, `--cwd DIR` | Project working directory. The default is the current directory | -| `--preview-screen SCREEN` | Test one screen visually with mock data: `setup`, `mode-picker`, `model-picker`, `chat`. This does not start a server or model | - -## `localcode run` - -Run one coding goal without the TUI, then exit. Use this for scripts, CI, and evaluation. Approvals always use full-auto because no person is available to answer prompts. - -| Flag | Description | -| --- | --- | -| `--goal TEXT` | **Required.** The task the agent must complete | -| `--binary PATH` | Path to a `llama-server` binary. For example, use stock llama.cpp on Linux CI with `LOCALCODE_SERVER_FLAVOR=vanilla` | -| `--timeout N` | Stop after N seconds (`0` = no limit) | -| `--max-rounds N` | Maximum number of model/tool rounds (`0` = unlimited) | -| `--thinking off\|auto\|on` | Hidden-reasoning setting for this run | -| `--thinking-budget N` | Reasoning-token limit (`0` = model default, negative disables) | -| `--quiet` | Hide streamed output and print only the final answer | -| `--json` | Write the event stream to stdout as JSON Lines | - -Exit codes: `0` ok · `1` error · `124` timeout · `130` interrupted. - -See [JSONL Events](/localcode/0.3/reference/jsonl-events). - -## Environment variables - -| Variable | Effect | -| --- | --- | -| `LOCALCODE_AUTONOMY` | `suggest`, `auto_edit` (default), or `full_auto` | -| `LOCALCODE_HOME` | Use a location other than `~/.localcode` | -| `LOCALCODE_SERVER_FLAVOR` | Use `vanilla` with a stock llama.cpp binary | -| `LOCALCODE_ALLOW_DEBUGGER` | Set to `1` to skip macOS anti-debugger hardening | diff --git a/website/src/content/docs/0.3/reference/configuration.md b/website/src/content/docs/0.3/reference/configuration.md deleted file mode 100644 index 43b1902b..00000000 --- a/website/src/content/docs/0.3/reference/configuration.md +++ /dev/null @@ -1,61 +0,0 @@ ---- -title: Configuration -description: Config file locations, layering, and the sections you are most likely to touch. -slug: 0.3/reference/configuration ---- - -## Where the config is stored - -| Path | Scope | -| --- | --- | -| `~/.localcode/config.toml` | Global, for the whole machine | -| `/.localcode/config.toml` | Project - added on top of the global config | - -Set `LOCALCODE_HOME` to replace `~/.localcode` with another location. - -You can change most settings in the TUI (`/model`, `/thinking`, `/permissions`, `/sounds`, …). Changes are saved to the global file automatically. Edit the TOML by hand only for settings that are not available in the UI. - -## Sections - -### `[runtime]` - -Settings for the model server and text generation. - -| Key | Default | Notes | -| --- | --- | --- | -| `provider` | `llama_cpp` | Inference backend | -| `base_url` | `http://localhost:8081` | The URL that receives chat completions. **Any URL is allowed. It is not checked or limited to localhost.** You can override it with `LOCALCODE_BASE_URL` | -| `model` | *(per-Mac recommendation)* | Model tag | -| `mode` | `fast` | `fast` or focused more on reasoning | -| `internal_thinking_mode` | `off` | Hidden reasoning: `off` or `auto`. Off by default | -| `thinking_budget_tokens` | `0` | `0` = catalogue default; a negative value disables it | -| `max_rounds` | `0` | `0` = unlimited interactive loop | -| `kv_cache_type_k` | `q8_0` | K cache quantisation | -| `kv_cache_type_v` | `turbo4` | V cache quantisation (TurboQuant) | -| `model_dir` | *(empty)* | Where GGUFs are downloaded; blank → `~/.local/share/localcode/models` | -| `vision_enabled` | `false` | Allows image input | -| `request_timeout_seconds` | `600` | Timeout for each request | - -### `[safety]` - -| Key | Default | -| --- | --- | -| `confirm_destructive` | `true` | -| `confirm_installs` | `true` | -| `show_diff_before_apply` | `true` | -| `jail_to_project` | `true` | -| `auto_approve_agent` | `false` | - -### `[ui]` and `[logging]` - -The settings are `ui.show_debug`, `ui.sounds_enabled`, `logging.enabled`, `logging.log_prompts`, `logging.log_responses`, and `logging.max_days` (30). - -## Other files in `~/.localcode/` - -| File | Purpose | -| --- | --- | -| `mcp.json` | MCP server definitions (key: `mcpServers`) | -| `hooks.toml` | Global lifecycle hooks | -| `skills/` | Global skills + `registry.json` | - -UI turn traces are **no longer** written to `~/.localcode/telemetry/turns.jsonl`. They are now stored in each project's `/.localcode/events.jsonl` file as `ui_turn_end` records. Set `LOCALCODE_TELEMETRY=0` to stop creating them. In either case, localcode does not upload the records. diff --git a/website/src/content/docs/0.3/reference/error-codes.md b/website/src/content/docs/0.3/reference/error-codes.md deleted file mode 100644 index 0faea2fd..00000000 --- a/website/src/content/docs/0.3/reference/error-codes.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -title: Error Codes -description: Stable Eccc codes for every user-facing error. -slug: 0.3/reference/error-codes ---- - -Every error shown to localcode users has a stable `Eccc` code. This makes it easy to search for and refer to an error across versions. - -| Range | Area | -| --- | --- | -| `E1xxx` | Setup and startup - starting the server, model files, and memory | -| `E2xxx` | Tool handling - unknown tools, invalid arguments, and permission or hook denials | -| `E3xxx` | Runtime and model | - -The full table is **generated from the code**. It is not written by hand. The source of truth is `src/localcode/errors.py`. The generated table is in [`docs/ERRORS.md`](https://github.com/mjwsolo/localcode/blob/main/docs/ERRORS.md) in the repository. Regenerate it with: - -```sh -python -m localcode.errors --emit-docs > docs/ERRORS.md -``` - -Detailed technical information about the latest project error is written to `/.localcode/last_error.log`. - -:::note[There is no `localcode setup` command] -Setup runs inside the TUI on first launch. An older generated copy of the error table still tells users to run `localcode setup` for `E1001`, `E1002`, and `E1003`. Ignore that: relaunch localcode and let the TUI handle setup and the model download. -::: diff --git a/website/src/content/docs/0.3/reference/jsonl-events.md b/website/src/content/docs/0.3/reference/jsonl-events.md deleted file mode 100644 index ba1e96fd..00000000 --- a/website/src/content/docs/0.3/reference/jsonl-events.md +++ /dev/null @@ -1,74 +0,0 @@ ---- -title: JSONL Events -description: The event stream emitted on stdout by localcode run --json. -slug: 0.3/reference/jsonl-events ---- - -`localcode run --goal "..." --json` outputs the agent's event stream as JSON Lines on stdout. Each line contains one JSON object. This lets editors and CI control localcode with code. - -When `--json` is active, stdout stays clean. Rich output is turned off. The agent's raw ANSI output goes to `/dev/null`. The JSONL is written to a private copy of the original stdout file descriptor. - -## Stream format - -Every line is one event. The last line is always a `result` summary. Each line combines the event's `type` with its payload: - -```json -{"type": "tool_start", "name": "read_file", "args": "src/timeutil.py", "index": "0"} -``` - -## Event types - -| `type` | Payload | Meaning | -| --- | --- | --- | -| `thinking_start` | `reset` (`"true"`/`"false"`) | The model started thinking | -| `thinking_chunk` | `chunk` | Part of the hidden reasoning (limited to 2000 chars) | -| `thinking_peek` | `text` | Short preview of the current reasoning (120 chars) | -| `thinking_done` | `text` | The reasoning finished (limited to 8000 chars) | -| `stage` | `stage` | A named work stage that the UI can show | -| `stream_start` | - | The model started streaming its answer | -| `content` | `chunk`, `chars` | Part of the assistant output (limited to 2000 chars) | -| `tool_preview` | `name`, `chars`, `snippet` | A tool call is being built during streaming; its args are still growing | -| `tool_start` | `name`, `args`, `index` | A tool call was sent | -| `tool_result` | `name`, `args`, `index`, `result`, `error` | The call returned. `error` is `"true"`/`"false"`; `result` is limited to 4000 chars | -| `turn_tokens` | `prompt_tokens`, `completion_tokens`, `total_tokens` | Token use for one round. Values are **strings** | -| `notice` | `text` | A notice for the user, such as why a turn ended | -| `error` | `message` | An error that ends the turn (240 chars) | -| `done` | - | The turn finished | -| `result` | see below | Final event, always the last line | - -Fields that look numeric (`index`, `chars`, and the three token counts) are output as **strings**. Convert them before doing arithmetic. - -## The final `result` line - -```json -{ - "type": "result", - "status": "ok", - "exit_code": 0, - "reason": "completed", - "final_text": "…", - "tokens": {"prompt": 1200, "completion": 340, "total": 1540} -} -``` - -Here, the token counts *are* integers. The emitter adds them up from the `turn_tokens` events it received. - -* `status` is `ok` when the turn's completion status is `completed`. Any other non-empty status (blocked, interrupted, stopped\_early, error, incomplete) is reported as `incomplete`. -* `exit_code` is `0` for `ok`, `1` for `incomplete`, `124` for a timeout, and `130` for an interrupt. -* `reason` is the loop's blocked reason or exit reason, such as `completed`. - -Read this line instead of joining the `content` chunks: - -```sh -localcode run --goal "add a test for parse_config" --json | tail -1 | jq . -``` - -## Not included on stdout - -The project audit log at `/.localcode/events.jsonl` is a **separate** stream with a **different set of events**. It includes `turn_start`, `turn_end`, `round_start`, `round_end`, `auto_nudge`, server lifecycle records, and more. These events are written only to the file and never appear on stdout. - -The project log is written whether or not you use `--json`. It is append-only. localcode never uploads it because no code path reads it for transmission. You can follow it during a run: - -```sh -tail -f .localcode/events.jsonl | jq . -``` diff --git a/website/src/content/docs/0.3/reference/slash-commands.md b/website/src/content/docs/0.3/reference/slash-commands.md deleted file mode 100644 index 48eeeb9d..00000000 --- a/website/src/content/docs/0.3/reference/slash-commands.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -title: Slash Commands -description: Commands available inside the localcode TUI. -slug: 0.3/reference/slash-commands ---- - -Type `/` in the chat box to open the command palette. Text that starts with `/` is only a command if the first word is a known command. A path like `/Users/you/project` is sent to the model as a normal message. - -Start a line with `!` to run a shell command, for example `!git status`. The output appears in the chat log and the model is not involved. - -| Command | What it does | -| --- | --- | -| `/permissions` | Turn command approvals on or off | -| `/status` | Show the server health, current model, and performance settings | -| `/restart` | Restart the model server when `/status` shows "unreachable" | -| `/mcp` | List MCP servers and their tools, or `/mcp reload` after editing `~/.localcode/mcp.json` | -| `/skills` | List loaded skills and their sources | -| `/model` | List or switch models, for example `/model qwen` | -| `/delete` | Delete a downloaded model to free disk space after asking first | -| `/thinking` | Show or set the hidden-reasoning policy to `off` or `auto` | -| `/sounds` | Turn completion and approval sounds on or off | -| `/voice` | Turn voice mode on or off for push-to-talk dictation in the input box | -| `/audio` | Turn audio output on or off so macOS `say` can read replies aloud | -| `/vision` | Turn vision mode on or off so the model can see images | -| `/search` | Turn the conversation search bar on or off, like `Ctrl+F` | -| `/clear` | Clear the conversation history | -| `/exit` | Exit localcode | - -`/search` is a valid command, but it does not appear in the `/` palette. `Ctrl+F` is the main way to open it. Typing `/search` opens or closes the same search bar. - -You can also use `/quit` instead of `/exit`, and `/copy` to copy the last reply. - -`/thinking` does nothing for models without a hidden-reasoning channel. localcode tells you this instead of acting as if the setting worked. diff --git a/website/src/content/docs/0.3/start-here/choose-a-model.md b/website/src/content/docs/0.3/start-here/choose-a-model.md deleted file mode 100644 index a1b07827..00000000 --- a/website/src/content/docs/0.3/start-here/choose-a-model.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -title: Models -description: How localcode picks a model for your Mac, how to override it, and - what determines speed. -slug: 0.3/start-here/choose-a-model ---- - -localcode has **no fixed default model**. When it starts, it checks your Mac's unified memory. It then recommends the most capable production-ready model whose weights fit in the available memory. - -## The rule - -Model weights must use about **55% of unified memory** or less. The rest is for the KV cache, activations, and macOS. localcode recommends the most capable model that fits. It never recommends experimental architectures automatically, but you can choose them yourself. - -## Recommendations for each Mac - -| Unified memory | Recommended | Quant | Weights | -| ---: | --- | --- | ---: | -| 16 GB | Gemma 4 12B | UD-Q4\_K\_XL | 7.37 GB | -| 24–48 GB | Qwen 3.6 35B-A3B | Q2 | 10.7 GB | -| 64 GB | Gemma 4 26B-A4B | Q8 | 28.0 GB | -| 96 GB+ | Qwen 3.6 35B-A3B | UD-Q8\_K\_XL | 38.5 GB | - -## What determines speed - -Three things matter, roughly in this order: - -1. **Active parameters per token.** On Apple Silicon, memory bandwidth limits decoding speed. A Mixture-of-Experts model uses only a few billion parameters per token. It reads far fewer bytes per token than a dense model with the same total size. -2. **Memory bandwidth.** This varies much more between chip tiers than the number of cores. It directly affects decoding speed. -3. **KV cache size.** TurboQuant compression (`q8_0`-K + `turbo4`-V) keeps the cache small enough for long contexts to remain practical. See [Unified Memory](/localcode/0.3/concepts/unified-memory). - -## The tok/s numbers in the model picker - -The model picker shows an estimated decoding speed next to each quantisation. This number is **calculated, not measured**. localcode does not run a benchmark on your machine. There is no benchmark command or benchmark screen. - -The estimate uses an analytic model. It divides the bytes read per token by an assumed share of your chip's rated memory bandwidth. It then adds a fixed compute time for each token. The bytes per token come from the quant's size and its active-parameter fraction. This means MoE models count only their active experts. The model is calibrated using a small number of maintainer measurements from one machine. - -Use the estimate to compare options. It can show that one quant will be slower than another on your hardware. Do not treat it as a prediction of your actual throughput. Real speed also depends on context length, thermal state, and other running tasks. - -## Switching models - -Type `/model` in the TUI to open the picker and choose another model. `/delete` removes a downloaded model to free disk (it asks first). - -If you switch to a model you do not have, localcode downloads it first. You only need to download each model once. - -## Next - -* [Unified Memory](/localcode/0.3/concepts/unified-memory) - explains the memory budget. diff --git a/website/src/content/docs/0.3/start-here/first-change.md b/website/src/content/docs/0.3/start-here/first-change.md deleted file mode 100644 index 0f0518f6..00000000 --- a/website/src/content/docs/0.3/start-here/first-change.md +++ /dev/null @@ -1,75 +0,0 @@ ---- -title: Install -description: Install localcode, open a repo, pick a model, make one change. -slug: 0.3/start-here/first-change ---- - -```sh -pip install -U localcode -``` - -| | | -| --- | --- | -| Machine | Mac with Apple Silicon | -| Unified memory | At least 16 GB | -| Python | 3.10 or newer | -| Disk | Space for one model - the smallest recommended GGUF is about 7.4 GB | - -Apple Silicon is the supported platform. Metal-accelerated inference works only on Mac. localcode also installs and runs on Linux in CI for development, but Linux is not the product platform. - -## Open your repo - -```sh -cd ~/work/some-project -localcode -``` - -Choose a project whose tests already pass. localcode uses your repo's own checks as proof. - -## Choose a model - -The model picker opens on first launch. localcode checks your Mac's unified memory and marks a recommended model with a star. Use the arrow keys to choose one, then press Enter. - -![The localcode model picker: seven models, moving down the list and choosing one](/localcode/demo/0.3/step-2-choose-model.gif?v=a0c3cc9d) - -localcode downloads the model's GGUF from Hugging Face - the only step that needs the network, and only once per model. Then it starts the included `llama-server` at `http://localhost:8081` and connects the agent to it. - -Learn more in [Models](/localcode/0.3/start-here/choose-a-model). - -## Start building - -Enter your request in the chat screen. Include the file name and the check to run. - -![Entering a goal in the localcode chat screen and pressing Enter](/localcode/demo/0.3/step-3-ask.gif?v=0523fe0c) - -```text -> Implement the retry decorator in retry.py so every test in test_retry.py - passes. Do not modify test_retry.py. Then run: pytest -q -``` - -## Watch it verify - -The model reads the stub and tests. It writes the code, runs `pytest -q`, and reports what it checked. - -![localcode reading files, editing them, and then showing 5 passed in pytest](/localcode/demo/0.3/step-4-verify.gif?v=92c546ff) - -Qwen3.6-35B-A3B (IQ2\_M) runs locally on `127.0.0.1:8081`. It uses four tool -calls, takes 11.5 s, and uses 276 tokens. The repository's tests fail before the turn and -pass after it. - -## Key commands - -| Command | What it does | -| --- | --- | -| `/status` | Shows server health, the current model, and performance settings | -| `/model` | Lists models or switches models (`/model qwen`) | -| `/permissions` | Turns command approvals on or off | -| `/clear` | Clears the conversation history | -| `/exit` | Quits | - -See the full list: [Slash Commands](/localcode/0.3/reference/slash-commands). - -## Next - -* [Models](/localcode/0.3/start-here/choose-a-model) - find the best model that fits on your Mac. -* [Network Boundary](/localcode/0.3/concepts/network-boundary) - learn what leaves your machine and when. diff --git a/website/src/content/docs/0.3/start-here/permissions.md b/website/src/content/docs/0.3/start-here/permissions.md deleted file mode 100644 index 673b9300..00000000 --- a/website/src/content/docs/0.3/start-here/permissions.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: Permissions -description: What the agent can do on its own, what it asks about, and what is - never allowed. -slug: 0.3/start-here/permissions ---- - -localcode has three autonomy levels. Set one at startup or toggle approvals with `/permissions`: - -* **suggest** - asks before every shell command and every file write. -* **auto\_edit** (the interactive default) - edits files without asking, but confirms risky or destructive shell commands such as `rm -rf`, `git push`, `pip install`, `npm install`, and `curl ... | sh`. -* **full\_auto** - nothing prompts. - -Two rules hold at every level: - -* **Network tools never prompt.** `web_search`, `web_fetch`, and MCP tools run without asking, even in `suggest`. See [Network Boundary](/localcode/0.3/concepts/network-boundary). -* **A hard safety block cannot be turned off.** Catastrophic operations - `rm -rf /`, `mkfs`, `dd` to a disk device, or writing credential files like `~/.ssh/id_rsa` - are refused in every mode, including `full_auto`. It is a guard against mistakes, not a security boundary. diff --git a/website/src/content/docs/concepts/architecture.md b/website/src/content/docs/concepts/architecture.md index 0cf8d7fa..4c00c922 100644 --- a/website/src/content/docs/concepts/architecture.md +++ b/website/src/content/docs/concepts/architecture.md @@ -30,18 +30,17 @@ description: The pieces localcode is made of, from the launcher down to the infe One model server runs per user. Opening `localcode` in another terminal starts a separate interface session attached to that server. Close either window without ending the other session; use `/models` to switch the shared model. -The previous interface (0.3) is still in the package as `localcode --classic`. It runs its own Python agent loop against the same `llama-server`. `localcode run`, the headless mode, uses that loop too. - ## The discipline plugin Small local models finish a task when the loop makes them. The plugin adds localcode's completion discipline to the runtime: -- **Plan first** - the model lays out the steps in the todo list before editing. +- **Orient from a snapshot** - the first message of a session carries a snapshot of the project's layout, and later messages report files that changed outside the session, so the model does not spend steps listing directories. +- **Plan first** - for work of three or more steps the model lays out the steps in the todo list before editing, and marks each one done as its last edit lands. - **Keep going** - a turn does not end while todo items are still open. - **Prove it** - after edits, the plugin runs the project's own typecheck and tests and feeds failures back to the model. - **Audit stubs** - placeholder code and `TODO` bodies are flagged before the task counts as done. - **No foreground servers** - a command that would block the session, such as a dev server, is refused with a hint to run it in the background. -- **Stop cleanly** - when no progress is being made across rounds, the plugin asks the model to wrap up and then stops, instead of looping. +- **Stop cleanly** - when the model only repeats itself across rounds, the plugin asks it to wrap up; if that does not help it refuses further tool calls with the reason, so the model reports what it has and the turn ends. A task that keeps covering new ground is never stopped. ## Built specifically for small models diff --git a/website/src/content/docs/concepts/network-boundary.md b/website/src/content/docs/concepts/network-boundary.md index a3ed5e23..fa69e982 100644 --- a/website/src/content/docs/concepts/network-boundary.md +++ b/website/src/content/docs/concepts/network-boundary.md @@ -42,10 +42,3 @@ that provider. The web tools and MCP tools run without a permission prompt. Shell commands prompt; see [Permissions](/localcode/start-here/permissions). - -## The classic interface - -`localcode --classic` and the headless `localcode run` use the 0.3 agent loop. -It has a configurable inference endpoint and a connectivity probe that the -default interface does not have. See the 0.3 docs through the version switcher -in the header. diff --git a/website/src/content/docs/guides/skills-and-hooks.md b/website/src/content/docs/guides/skills-and-hooks.md index 5369c39d..b1aec78e 100644 --- a/website/src/content/docs/guides/skills-and-hooks.md +++ b/website/src/content/docs/guides/skills-and-hooks.md @@ -24,4 +24,4 @@ Skills are read from disk only. localcode does not fetch skills from a URL. ## Hooks -Lifecycle hooks are a feature of the 0.3 classic interface (`~/.localcode/hooks.toml`) and are not part of the 0.4 interface. In 0.4, the discipline plugin runs the project's own checks after edits; see [Architecture](/localcode/concepts/architecture). For the classic hook format, use the 0.3 docs through the version switcher in the header. +localcode has no user-defined lifecycle hooks. The discipline plugin runs the project's own checks after edits and feeds failures back to the model; see [Architecture](/localcode/concepts/architecture). To run something on every commit, use your repository's own pre-commit hooks; the agent's `git commit` goes through them like yours does. diff --git a/website/src/content/docs/reference/cli.md b/website/src/content/docs/reference/cli.md index f6959776..c6c6abf4 100644 --- a/website/src/content/docs/reference/cli.md +++ b/website/src/content/docs/reference/cli.md @@ -4,7 +4,7 @@ description: Every flag and subcommand localcode accepts. --- ```text -localcode [-c DIR] [--model TAG] [--classic] +localcode [-c DIR] [--model TAG] localcode --version localcode run --goal "..." [options] localcode unstick @@ -18,16 +18,13 @@ Run `localcode` by itself to open the interface. Setup, the model picker, and se | --- | --- | | `-c`, `--cwd DIR` | Project directory. The default is the current directory | | `--model TAG` | Start with an already-downloaded model alias instead of the picker | -| `--classic` | Open the previous 0.3 interface. `LOCALCODE_FRONTEND=classic` does the same | | `--version` | Print the version and exit | -`--profile`, `--resume`, and `--preview-screen` belong to the classic interface and imply `--classic`. - A second `localcode` opens another interface session attached to the running model server. The windows share that model; closing the second window leaves the first session running. Use `/models` to switch the shared model. ## `localcode run` -Run one coding goal without the interface, then exit. Use this for scripts, CI, and evaluation. This is the headless agent from 0.3 and is unchanged. Approvals always use full-auto because no person is available to answer prompts; writes outside the project directory are rejected. +Run one coding goal without the interface, then exit. Use this for scripts, CI, and evaluation. Approvals always use full-auto because no person is available to answer prompts; writes outside the project directory are rejected. | Flag | Description | | --- | --- | @@ -52,7 +49,6 @@ Recovers from a stuck `llama-server` without a reboot. It runs `memory_pressure` | Variable | Effect | | --- | --- | -| `LOCALCODE_FRONTEND` | `ui` (default) or `classic` | | `LOCALCODE_MODEL_DIR` | Where GGUFs live. Default `~/.local/share/localcode/models` | | `LOCALCODE_MODELS_DIR` | Older name for the models directory override; `LOCALCODE_MODEL_DIR` takes precedence | | `LOCALCODE_AGENT_RUN_DIR` | Supervisor logs and the voice runtime. Default `~/.local/share/localcode-agent/run` | @@ -64,4 +60,4 @@ Recovers from a stuck `llama-server` without a reboot. It runs `memory_pressure` | `LOCALCODE_UI_BIN` | Developer override: path to the interface binary | | `LOCALCODE_LLAMA_SERVER` | Developer override: path to a `llama-server` binary | -The full list, with the classic-only variables, is in [Configuration](/localcode/reference/configuration). +The full list is in [Configuration](/localcode/reference/configuration). diff --git a/website/src/content/docs/reference/configuration.md b/website/src/content/docs/reference/configuration.md index c5621356..287fb25b 100644 --- a/website/src/content/docs/reference/configuration.md +++ b/website/src/content/docs/reference/configuration.md @@ -1,6 +1,6 @@ --- title: Configuration -description: Environment variables, the run directory, localcode.json, LOCALCODE.md, and the classic interface. +description: Environment variables, the run directory, localcode.json, and LOCALCODE.md. --- There is nothing to configure before the first run. Most day-to-day settings live in the interface: `/models`, `/settings`, `/themes`, `/permissions`, `/mcps`. This page covers what lives outside it. @@ -9,7 +9,6 @@ There is nothing to configure before the first run. Most day-to-day settings liv | Variable | Effect | | --- | --- | -| `LOCALCODE_FRONTEND` | `ui` (default) or `classic` for the previous 0.3 interface | | `LOCALCODE_MODEL_DIR` | Where GGUFs live. Default `~/.local/share/localcode/models`. The picker's **Models folder** entry changes the same setting | | `LOCALCODE_MODELS_DIR` | Older override name; `LOCALCODE_MODEL_DIR` takes precedence | | `LOCALCODE_AGENT_RUN_DIR` | Supervisor logs, the per-session runtime config, and the voice runtime. Default `~/.local/share/localcode-agent/run` | @@ -60,7 +59,3 @@ See [MCP](/localcode/guides/mcp) for the server shapes. The model provider secti ## Skills Skill folders are read from `/.localcode-agent/skills/` and `~/.config/localcode-agent/skills/`. See [Skills](/localcode/guides/skills-and-hooks). - -## The classic interface - -`localcode --classic` (or `LOCALCODE_FRONTEND=classic`) opens the 0.3 interface. It keeps its own configuration in `~/.localcode/config.toml` and `/.localcode/`, and its own variables such as `LOCALCODE_AUTONOMY` and `LOCALCODE_HOME`. The headless `localcode run` uses the same files. Those are documented in the 0.3 docs, reachable from the version switcher in the header. diff --git a/website/src/content/docs/reference/error-codes.md b/website/src/content/docs/reference/error-codes.md index 26a2dc50..fd8d5b27 100644 --- a/website/src/content/docs/reference/error-codes.md +++ b/website/src/content/docs/reference/error-codes.md @@ -17,7 +17,7 @@ The full table is **generated from the code**. It is not written by hand. The so python -m localcode.errors --emit-docs > docs/ERRORS.md ``` -When the model server fails to load a model, the detail is in `server.log` under the run directory (`~/.local/share/localcode-agent/run` by default). The classic interface and `localcode run` write `/.localcode/last_error.log`. +When the model server fails to load a model, the detail is in `server.log` under the run directory (`~/.local/share/localcode-agent/run` by default). The headless `localcode run` writes `/.localcode/last_error.log`. :::note[`dyld: Library not loaded` on launch] If `localcode` fails right away with a `dyld` error such as "Library not loaded", the Mac is running a macOS older than 13. The bundled binaries are built for macOS 13 and newer on Apple Silicon. Update macOS; there is no build for older versions. diff --git a/website/src/content/docs/start-here/choose-a-model.md b/website/src/content/docs/start-here/choose-a-model.md index 1cdd3ed2..ce8a9ed1 100644 --- a/website/src/content/docs/start-here/choose-a-model.md +++ b/website/src/content/docs/start-here/choose-a-model.md @@ -20,7 +20,7 @@ Model weights must use about **55% of unified memory** or less. The rest is for ## The model picker -The picker opens on first launch when no model is loaded. Type `/models` to open it at any time. Esc goes back one level. +With no model loaded, the picker opens when you send your first message. Type `/models` to open it at any time. Esc goes back one level. **Level 1: models.** Every model in the catalog, shown as display name and maker. A count such as "2 on disk" tells you how many of its quants are already downloaded. The star marks the model recommended for this Mac's memory. Press Enter to open a model. diff --git a/website/src/content/docs/start-here/first-change.md b/website/src/content/docs/start-here/first-change.md index 8a9cc484..c7dc3e8e 100644 --- a/website/src/content/docs/start-here/first-change.md +++ b/website/src/content/docs/start-here/first-change.md @@ -28,11 +28,11 @@ localcode Choose a project with a runnable test suite. localcode uses your repo's own checks as proof. -`localcode` opens the home screen. The previous 0.3 interface is still there: run `localcode --classic`, and read its docs through the version switcher in the header. +`localcode` opens the prompt. ## Choose a model -With no model loaded, the model picker opens on top of the home screen. The first level lists every model in the catalog with its maker, how many of its quants are already on disk, and a star on the model recommended for your Mac's unified memory. Press Enter on a model to see every quant its Hugging Face repo ships, with the size in GB and whether it fits in memory. Enter on a row marked **Download** starts the download and shows a live percentage. Nothing downloads until you choose it and see its size. +With no model loaded, the model picker opens when you send your first message, or when you type `/models`. The first level lists every model in the catalog with its maker, how many of its quants are already on disk, and a star on the model recommended for your Mac's unified memory. Press Enter on a model to see every quant its Hugging Face repo ships, with the size in GB and whether it fits in memory. Enter on a row marked **Download** starts the download and shows a live percentage. Nothing downloads until you choose it and see its size. ![The localcode model picker: seven models and the quant options for one model](/localcode/demo/step-2-choose-model.gif?v=3f833e97) diff --git a/website/src/content/docs/start-here/permissions.md b/website/src/content/docs/start-here/permissions.md index 9c5e70d4..78b79a70 100644 --- a/website/src/content/docs/start-here/permissions.md +++ b/website/src/content/docs/start-here/permissions.md @@ -14,7 +14,7 @@ Reading files does not prompt. `/permissions` shows the current rules for the se ## Rules that always hold - **Writes outside the project directory are refused.** The agent can only edit files under the directory you opened. This holds in the interface and in headless runs. -- **Headless runs cannot answer prompts.** `localcode run` has nobody to ask, so it runs the previous 0.3 agent loop in full-auto; out-of-workspace writes are auto-rejected. See [CLI](/localcode/reference/cli). +- **Headless runs cannot answer prompts.** `localcode run` has nobody to ask, so it runs in full-auto; out-of-workspace writes are auto-rejected. See [CLI](/localcode/reference/cli). - **Network tools do not prompt.** `websearch`, `webfetch`, and MCP tools run when the model calls them. See [Network Boundary](/localcode/concepts/network-boundary). ## Project rules diff --git a/website/src/pages/index.astro b/website/src/pages/index.astro index 3db4f035..75b1589f 100644 --- a/website/src/pages/index.astro +++ b/website/src/pages/index.astro @@ -184,7 +184,7 @@ localcode
  • 2 Choose a model

    - The picker opens on first launch. The star shows the largest + The picker opens when you send your first message. The star shows the largest model your Mac can run comfortably. Press Enter, pick a quant, and see its size before anything downloads.

    @@ -211,11 +211,6 @@ localcode -

    - Upgrading from 0.3? The previous interface is still there as - localcode --classic. -

    - Read more diff --git a/website/src/styles/docs.css b/website/src/styles/docs.css index 289361ab..81207d06 100644 --- a/website/src/styles/docs.css +++ b/website/src/styles/docs.css @@ -87,7 +87,7 @@ } /* ── Demo recordings ───────────────────────────────────────────── - The Textual frames are 898px wide natively. Never upscale them: + The terminal captures are 898px wide natively. Never upscale them: stretched terminal type is the fastest way to make a real capture look fake. */ .sl-markdown-content img[src*='/demo/'] {