Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,26 @@
All notable changes to LocalCode will be documented here. The format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/).

## 0.4.5 — 2026-09-26

### Fixed

- **Every turn after the first is much faster.** llama-server reuses its cache
only for an identical prompt prefix, and the system prompt was changing
between steps: the workspace block was added after the first tool call of a
turn and dropped again on the next, the planning rule followed the same
switch, and the open-todo list was rewritten into the system prompt on every
update. Each change made the server re-read the whole conversation, twice
per turn (15-23 s at 30k tokens on an M5). The system prompt is now constant
for a session, and todos travel with the user turn. Measured on Qwen3.8 27B:
requests after the first tool call re-read 23-310 tokens (0.3-0.8 s)
instead of 6-7k (10-15 s).
- Title generation and other short side requests no longer land on the
conversation's server slot when the server has several.
- The code-intelligence panel said "Starts automatically for your project's
language", which was never true: localcode does not download language
servers on its own. It now says "Not running · /lsp to set up".

## 0.4.4 — 2026-09-26

### Fixed
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

[project]
name = "localcode"
version = "0.4.4"
version = "0.4.5"
description = "High-performance AI coding on consumer hardware."
readme = "README.md"
requires-python = ">=3.10"
Expand Down
2 changes: 1 addition & 1 deletion src/localcode/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,4 +8,4 @@
try:
__version__ = _pkg_version("localcode")
except PackageNotFoundError: # not installed (e.g. raw source checkout)
__version__ = "0.4.4"
__version__ = "0.4.5"
Binary file modified src/localcode/bin/localcode-ui
Binary file not shown.
2 changes: 1 addition & 1 deletion src/localcode/ui/FORK_COMMIT
Original file line number Diff line number Diff line change
@@ -1 +1 @@
b88aa9395bb0cbc3aa9bf44a04e36ed0dea0f0bc
f65ef33d195169f749bf406b50cdde8c819253f8
21 changes: 14 additions & 7 deletions src/localcode/ui/plugin/localcode.ts
Original file line number Diff line number Diff line change
Expand Up @@ -437,10 +437,8 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => {
try {
const r = await fetch(`${control}/status`, { signal: AbortSignal.timeout(1_000) });
const j = (await r.json()) as { current?: string; state?: string };
loadedName = { name: j.state === "ready" && j.current ? j.current : loadedName.name, at: Date.now() };
} catch {
loadedName = { ...loadedName, at: Date.now() };
}
if (j.state === "ready" && j.current) loadedName = { name: j.current, at: Date.now() };
} catch {}
return loadedName.name;
}

Expand Down Expand Up @@ -509,10 +507,15 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => {
if (output.system[i].includes("__pending__")) output.system[i] = output.system[i].replaceAll("__pending__", loaded);
}
}
if (!workspaceActive || (input.agent && input.agent !== "build")) return;
if (input.agent && input.agent !== "build") return;
// The system prompt must be IDENTICAL for every request of a session: it
// is the cached prefix. This rule used to be added only once
// workspaceActive flipped (after the first tool call of each turn), so the
// prompt changed mid-turn and the server re-read the whole conversation
// once per turn (15-23 s at 30k tokens). It is inert in plain chat (the
// gates check workspaceActive in code), so always add it. The open-todo
// list is not put here either: it rides on the user turn (chat.message).
output.system.push(PLANNING_RULE);
const open = renderTodos(todos);
if (open) output.system.push(open);
},

"chat.message": async (input, output) => {
Expand All @@ -532,6 +535,10 @@ const LocalcodePlugin: Plugin = async ({ client, directory }) => {
turnStartedAt = Date.now() - 1000;
plateau = new PlateauTracker(directory); plateauStopped = false; seenSteps.clear();
}
// Open todos travel with the turn (user message or nudge), never in the
// system prompt, so the cached prefix stays stable across todo updates.
const open = renderTodos(todos);
if (open) output.parts.push({ type: "text", text: open, synthetic: true } as any);
},

"tool.execute.after": async (input, output) => {
Expand Down
28 changes: 15 additions & 13 deletions tests/ui/prompt-scope.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,11 @@ import { tmpdir } from "node:os";
import { join } from "node:path";
import Plugin from "../../src/localcode/ui/plugin/localcode";

test("completion rules stay out of compaction and other helper prompts", async () => {
// The system prompt is llama-server's cached prefix: it must be identical for
// every request of a session, so the completion rules are present for the build
// agent from the first request on (never toggled by workspace activity) and the
// open-todo list travels on the user turn, not in the system prompt.
test("completion rules stay out of helper prompts and are constant for the build agent", async () => {
const directory = mkdtempSync(join(tmpdir(), "prompt-scope-"));
try {
const hooks = await (Plugin as any)({ directory, client: {} });
Expand All @@ -13,19 +17,17 @@ test("completion rules stay out of compaction and other helper prompts", async (
await hooks["experimental.chat.system.transform"]({ agent }, output);
expect(output.system).toEqual(["Summarize only."]);
}
const initial = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, initial);
expect(initial.system).toEqual([]);
await hooks["tool.execute.before"]({ tool: "websearch" }, { args: { query: "general research" } });
await hooks["experimental.chat.system.transform"]({ agent: "build" }, initial);
expect(initial.system).toEqual([]);
const first = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, first);
expect(first.system.join("\n")).toContain("requested multi-step workspace changes");
await hooks["tool.execute.before"]({ tool: "read" }, { args: { filePath: "." } });
const output = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, output);
expect(output.system.join("\n")).toContain("requested multi-step workspace changes");
const afterTool = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, afterTool);
expect(afterTool.system).toEqual(first.system);
await hooks["chat.message"]({ sessionID: "test", agent: "build" }, { parts: [{ type: "text", text: "a different topic" }] });
const next = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, next);
expect(next.system).toEqual([]);
const nextTurn = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, nextTurn);
expect(nextTurn.system).toEqual(first.system);
expect(nextTurn.system.join("\n")).not.toContain("YOUR OPEN TODOS");
} finally { rmSync(directory, { recursive: true, force: true }); }
});
Loading