diff --git a/CHANGELOG.md b/CHANGELOG.md index 65bb5dbc..3e5b556c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,26 @@ All notable changes to LocalCode will be documented here. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). +## 0.4.5 — 2026-09-26 + +### Fixed + +- **Every turn after the first is much faster.** llama-server reuses its cache + only for an identical prompt prefix, and the system prompt was changing + between steps: the workspace block was added after the first tool call of a + turn and dropped again on the next, the planning rule followed the same + switch, and the open-todo list was rewritten into the system prompt on every + update. Each change made the server re-read the whole conversation, twice + per turn (15-23 s at 30k tokens on an M5). The system prompt is now constant + for a session, and todos travel with the user turn. Measured on Qwen3.8 27B: + requests after the first tool call re-read 23-310 tokens (0.3-0.8 s) + instead of 6-7k (10-15 s). +- Title generation and other short side requests no longer land on the + conversation's server slot when the server has several. +- The code-intelligence panel said "Starts automatically for your project's + language", which was never true: localcode does not download language + servers on its own. It now says "Not running · /lsp to set up". + ## 0.4.4 — 2026-09-26 ### Fixed diff --git a/pyproject.toml b/pyproject.toml index 77123bba..cbc0e938 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "localcode" -version = "0.4.4" +version = "0.4.5" description = "High-performance AI coding on consumer hardware." readme = "README.md" requires-python = ">=3.10" diff --git a/src/localcode/__init__.py b/src/localcode/__init__.py index 901e335d..5f818996 100644 --- a/src/localcode/__init__.py +++ b/src/localcode/__init__.py @@ -8,4 +8,4 @@ try: __version__ = _pkg_version("localcode") except PackageNotFoundError: # not installed (e.g. raw source checkout) - __version__ = "0.4.4" + __version__ = "0.4.5" diff --git a/src/localcode/bin/localcode-ui b/src/localcode/bin/localcode-ui index 70408cce..6762d4d8 100755 Binary files a/src/localcode/bin/localcode-ui and b/src/localcode/bin/localcode-ui differ diff --git a/src/localcode/ui/FORK_COMMIT b/src/localcode/ui/FORK_COMMIT index 0fc0497a..e588fcef 100644 --- a/src/localcode/ui/FORK_COMMIT +++ b/src/localcode/ui/FORK_COMMIT @@ -1 +1 @@ -b88aa9395bb0cbc3aa9bf44a04e36ed0dea0f0bc +f65ef33d195169f749bf406b50cdde8c819253f8 diff --git a/src/localcode/ui/plugin/localcode.ts b/src/localcode/ui/plugin/localcode.ts index 622bed18..cdfe6d26 100644 --- a/src/localcode/ui/plugin/localcode.ts +++ b/src/localcode/ui/plugin/localcode.ts @@ -437,10 +437,8 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => { try { const r = await fetch(`${control}/status`, { signal: AbortSignal.timeout(1_000) }); const j = (await r.json()) as { current?: string; state?: string }; - loadedName = { name: j.state === "ready" && j.current ? j.current : loadedName.name, at: Date.now() }; - } catch { - loadedName = { ...loadedName, at: Date.now() }; - } + if (j.state === "ready" && j.current) loadedName = { name: j.current, at: Date.now() }; + } catch {} return loadedName.name; } @@ -509,10 +507,15 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => { if (output.system[i].includes("__pending__")) output.system[i] = output.system[i].replaceAll("__pending__", loaded); } } - if (!workspaceActive || (input.agent && input.agent !== "build")) return; + if (input.agent && input.agent !== "build") return; + // The system prompt must be IDENTICAL for every request of a session: it + // is the cached prefix. This rule used to be added only once + // workspaceActive flipped (after the first tool call of each turn), so the + // prompt changed mid-turn and the server re-read the whole conversation + // once per turn (15-23 s at 30k tokens). It is inert in plain chat (the + // gates check workspaceActive in code), so always add it. The open-todo + // list is not put here either: it rides on the user turn (chat.message). output.system.push(PLANNING_RULE); - const open = renderTodos(todos); - if (open) output.system.push(open); }, "chat.message": async (input, output) => { @@ -532,6 +535,10 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => { turnStartedAt = Date.now() - 1000; plateau = new PlateauTracker(directory); plateauStopped = false; seenSteps.clear(); } + // Open todos travel with the turn (user message or nudge), never in the + // system prompt, so the cached prefix stays stable across todo updates. + const open = renderTodos(todos); + if (open) output.parts.push({ type: "text", text: open, synthetic: true } as any); }, "tool.execute.after": async (input, output) => { diff --git a/tests/ui/prompt-scope.test.ts b/tests/ui/prompt-scope.test.ts index 04116e50..e9611e83 100644 --- a/tests/ui/prompt-scope.test.ts +++ b/tests/ui/prompt-scope.test.ts @@ -4,7 +4,11 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import Plugin from "../../src/localcode/ui/plugin/localcode"; -test("completion rules stay out of compaction and other helper prompts", async () => { +// The system prompt is llama-server's cached prefix: it must be identical for +// every request of a session, so the completion rules are present for the build +// agent from the first request on (never toggled by workspace activity) and the +// open-todo list travels on the user turn, not in the system prompt. +test("completion rules stay out of helper prompts and are constant for the build agent", async () => { const directory = mkdtempSync(join(tmpdir(), "prompt-scope-")); try { const hooks = await (Plugin as any)({ directory, client: {} }); @@ -13,19 +17,17 @@ test("completion rules stay out of compaction and other helper prompts", async ( await hooks["experimental.chat.system.transform"]({ agent }, output); expect(output.system).toEqual(["Summarize only."]); } - const initial = { system: [] as string[] }; - await hooks["experimental.chat.system.transform"]({ agent: "build" }, initial); - expect(initial.system).toEqual([]); - await hooks["tool.execute.before"]({ tool: "websearch" }, { args: { query: "general research" } }); - await hooks["experimental.chat.system.transform"]({ agent: "build" }, initial); - expect(initial.system).toEqual([]); + const first = { system: [] as string[] }; + await hooks["experimental.chat.system.transform"]({ agent: "build" }, first); + expect(first.system.join("\n")).toContain("requested multi-step workspace changes"); await hooks["tool.execute.before"]({ tool: "read" }, { args: { filePath: "." } }); - const output = { system: [] as string[] }; - await hooks["experimental.chat.system.transform"]({ agent: "build" }, output); - expect(output.system.join("\n")).toContain("requested multi-step workspace changes"); + const afterTool = { system: [] as string[] }; + await hooks["experimental.chat.system.transform"]({ agent: "build" }, afterTool); + expect(afterTool.system).toEqual(first.system); await hooks["chat.message"]({ sessionID: "test", agent: "build" }, { parts: [{ type: "text", text: "a different topic" }] }); - const next = { system: [] as string[] }; - await hooks["experimental.chat.system.transform"]({ agent: "build" }, next); - expect(next.system).toEqual([]); + const nextTurn = { system: [] as string[] }; + await hooks["experimental.chat.system.transform"]({ agent: "build" }, nextTurn); + expect(nextTurn.system).toEqual(first.system); + expect(nextTurn.system.join("\n")).not.toContain("YOUR OPEN TODOS"); } finally { rmSync(directory, { recursive: true, force: true }); } });