Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 27 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,33 @@
All notable changes to LocalCode will be documented here. The format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/).

## 0.4.8 — 2026-09-27

### Fixed

- **No more "Interrupted" on research and investigation tasks.** The loop breaker
measured progress only by edits, passing checks and completed plan items, so a turn
that was legitimately reading, fetching and trying new commands without editing
anything was stopped after 14 rounds as "no progress". A turn that has delivered
nothing yet now counts a round with a never-seen tool call as progress; only
repeating the same calls stalls it. And when the breaker does stop a task it no
longer aborts the session, which killed whatever command was running and showed a
bare "Interrupted": it refuses further tool calls with the reason, so the model
writes up what it has and the turn ends normally. Abort remains only as a last
resort if the model ignores three refusals.

### Changed

- **About a quarter fewer tool calls per task.** On the regression evals (16 tasks,
Qwen3.8 27B) the agent now takes a median of 4 steps where it took 7, with the same
pass rate. Three causes: it no longer lists directories or globs to orient itself,
because the session's first turn carries a snapshot of the workspace and later turns
report files changed outside the session; it runs the project check once after an
edit instead of before and after; and it updates its plan once per step instead of
around every edit. The plan itself is unchanged and still gates completion.
- When a gate sends the model back to open items or placeholders, the UI now says so
("Continuing: 2 todos open") instead of pausing silently.

## 0.4.7 — 2026-09-26

### Fixed
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

[project]
name = "localcode"
version = "0.4.7"
version = "0.4.8"
description = "High-performance AI coding on consumer hardware."
readme = "README.md"
requires-python = ">=3.10"
Expand Down
2 changes: 1 addition & 1 deletion src/localcode/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,4 +8,4 @@
try:
__version__ = _pkg_version("localcode")
except PackageNotFoundError: # not installed (e.g. raw source checkout)
__version__ = "0.4.7"
__version__ = "0.4.8"
151 changes: 141 additions & 10 deletions src/localcode/ui/plugin/localcode.ts

Large diffs are not rendered by default.

63 changes: 63 additions & 0 deletions tests/ui/layout.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
import { expect, test } from "bun:test";
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import Plugin from "../../src/localcode/ui/plugin/localcode";

// The snapshot rides on the session's FIRST user turn (history, cached, labelled as a
// snapshot); later turns get a change report; the system prompt carries neither.
test("snapshot on the first turn, change report later, system prompt untouched", async () => {
const dir = mkdtempSync(join(tmpdir(), "layout-"));
try {
mkdirSync(join(dir, "app")); writeFileSync(join(dir, "app", "calc.py"), "");
mkdirSync(join(dir, "tests")); writeFileSync(join(dir, "tests", "test_calc.py"), "");
mkdirSync(join(dir, "node_modules", "x"), { recursive: true }); writeFileSync(join(dir, "node_modules", "x", "i.js"), "");
const hooks = await (Plugin as any)({ directory: dir, client: {} });
const sys = { system: [] as string[] }; await hooks["experimental.chat.system.transform"]({ agent: "build" }, sys);
expect(sys.system.join("\n")).not.toContain("WORKSPACE SNAPSHOT");
const t1 = { parts: [{ type: "text", text: "hello" }] as any[] };
await hooks["chat.message"]({ sessionID: "s1", agent: "build", messageID: "m1" }, t1);
const snap = t1.parts.find((p) => String(p.text).startsWith("WORKSPACE SNAPSHOT"));
expect(snap).toBeDefined();
expect(snap.text).toContain("app/"); expect(snap.text).toContain(" calc.py"); expect(snap.text).not.toContain("node_modules");
// the user edits a file outside the session; the agent edits another via a tool
writeFileSync(join(dir, "app", "calc.py"), "changed", { flush: true });
writeFileSync(join(dir, "NEW.md"), "x");
await hooks["tool.execute.after"]({ tool: "write", args: { filePath: join(dir, "tests", "agent.py") }, sessionID: "s1" }, { output: "ok", metadata: {} });
writeFileSync(join(dir, "tests", "agent.py"), "by agent");
const t2 = { parts: [{ type: "text", text: "next" }] as any[] };
await hooks["chat.message"]({ sessionID: "s1", agent: "build", messageID: "m2" }, t2);
expect(t2.parts.some((p) => String(p.text).startsWith("WORKSPACE SNAPSHOT"))).toBe(false);
const rep = t2.parts.find((p) => String(p.text).startsWith("WORKSPACE CHANGED"));
expect(rep).toBeDefined();
expect(rep.text).toContain("added: NEW.md"); expect(rep.text).toContain("modified: app/calc.py"); expect(rep.text).not.toContain("agent.py");
// nothing changed since -> no report
const t3 = { parts: [{ type: "text", text: "again" }] as any[] };
await hooks["chat.message"]({ sessionID: "s1", agent: "build", messageID: "m3" }, t3);
expect(t3.parts.length).toBe(1);
} finally { rmSync(dir, { recursive: true, force: true }); }
});


// A second plugin instance (new process, same session) must not re-send the snapshot,
// must see the user's external change, and must not report the first instance's own edit.
test("snapshot state survives a process restart", async () => {
const dir = mkdtempSync(join(tmpdir(), "layout2-"));
try {
mkdirSync(join(dir, "app")); writeFileSync(join(dir, "app", "calc.py"), "");
const h1 = await (Plugin as any)({ directory: dir, client: {} });
const t1 = { parts: [{ type: "text", text: "hello" }] as any[] };
await h1["chat.message"]({ sessionID: "s9", agent: "build", messageID: "m1" }, t1);
expect(t1.parts.some((p) => String(p.text).startsWith("WORKSPACE SNAPSHOT"))).toBe(true);
await h1["tool.execute.after"]({ tool: "edit", args: { filePath: join(dir, "app", "calc.py") }, sessionID: "s9" }, { output: "ok", metadata: {} });
writeFileSync(join(dir, "app", "calc.py"), "edited by agent", { flush: true });
writeFileSync(join(dir, "USER.md"), "by user");
const h2 = await (Plugin as any)({ directory: dir, client: {} }); // new process
const t2 = { parts: [{ type: "text", text: "next" }] as any[] };
await h2["chat.message"]({ sessionID: "s9", agent: "build", messageID: "m2" }, t2);
expect(t2.parts.some((p) => String(p.text).startsWith("WORKSPACE SNAPSHOT"))).toBe(false);
const rep = t2.parts.find((p) => String(p.text).startsWith("WORKSPACE CHANGED"));
expect(rep).toBeDefined();
expect(rep.text).toContain("added: USER.md"); expect(rep.text).not.toContain("calc.py");
} finally { rmSync(dir, { recursive: true, force: true }); }
});
89 changes: 76 additions & 13 deletions tests/ui/plateau.test.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import { describe, expect, test } from "bun:test";
import LocalcodePluginDefault from "../../src/localcode/ui/plugin/localcode";
type ToolEvent = { tool: string; args: any; output?: string; metadata?: any };
const { PLATEAU_MIN_ROUND, PLATEAU_NUDGE_AFTER, PLATEAU_STOP_AFTER, PLATEAU_PASS_GRACE, PlateauTracker, checkPassed, editedPaths, failureSignatures, isCheckCommand, isTempPath, newProgressMemory, plateauNudgeText, progressOf } = LocalcodePluginDefault as any;
const { PLATEAU_MIN_ROUND, PLATEAU_NUDGE_AFTER, PLATEAU_STOP_AFTER, PLATEAU_PASS_GRACE, PLATEAU_STOP_DENIALS, PlateauTracker, checkPassed, editedPaths, failureSignatures, isCheckCommand, isTempPath, newProgressMemory, plateauNudgeText, progressOf } = LocalcodePluginDefault as any;

const edit = (filePath: string): ToolEvent => ({ tool: "edit", args: { filePath, oldString: "a", newString: "b" } });
const write = (filePath: string): ToolEvent => ({ tool: "write", args: { filePath, content: "x" } });
Expand Down Expand Up @@ -125,6 +125,12 @@ describe("progressOf", () => {
});

/** Run `n` no-progress tool rounds (reads, or re-edits of a known path), returning decisions. */
/** A tracker that has already seen the calls these tests repeat (grep, read, the check): the one novel round is spent. */
function primed(path = "/p/a.ts") {
const t = new PlateauTracker();
t.observe(grep()); t.observe(read(path)); t.observe(check("ok", 0)); t.endRound();
return t;
}
function stall(t: PlateauTracker, n: number, path = "/p/a.ts") {
const out: string[] = [];
for (let i = 0; i < n; i++) {
Expand Down Expand Up @@ -158,13 +164,13 @@ describe("PlateauTracker thresholds", () => {
expect(t.round).toBe(1 + PLATEAU_NUDGE_AFTER);
expect(t.round).toBeGreaterThanOrEqual(PLATEAU_MIN_ROUND);
});
test("a fresh tracker that stalls from round 1 still waits for the 6th no-progress round (>= round 4)", () => {
const t = new PlateauTracker();
test("a tracker that only repeats itself still waits for the 6th no-progress round (>= round 4)", () => {
const t = primed();
const d = stall(t, PLATEAU_NUDGE_AFTER);
expect(d.indexOf("nudge")).toBe(PLATEAU_NUDGE_AFTER - 1);
});
test("after the nudge, 8 further no-progress rounds stop; progress in between resets", () => {
const t = new PlateauTracker();
const t = primed();
stall(t, PLATEAU_NUDGE_AFTER);
expect(t.nudged).toBe(true);
stall(t, 3);
Expand All @@ -178,21 +184,22 @@ describe("PlateauTracker thresholds", () => {
t.observe(edit("/p/a.ts")); expect(t.endRound()).toBe("none");
});
test("no second nudge: after the first nudge the only outcome is stop", () => {
const t = new PlateauTracker();
const t = primed();
stall(t, PLATEAU_NUDGE_AFTER);
const d = stall(t, PLATEAU_STOP_AFTER);
expect(d.filter((x) => x === "nudge")).toHaveLength(0);
expect(d.at(-1)).toBe("stop");
});
test("a check that passed within the last 2 rounds converts the stop into a finish-now nudge and 3 more rounds", () => {
const t = new PlateauTracker();
t.observe(grep()); t.observe(read("/p/a.ts")); t.endRound(); // primed, but no check yet
stall(t, PLATEAU_NUDGE_AFTER);
stall(t, PLATEAU_STOP_AFTER - 2);
// round with a passing check: first pass this session -> progress, resets the counter
t.observe(check("ok", 0)); expect(t.endRound()).toBe("none");
expect(t.noProgress).toBe(0);
// now stall again: 8 rounds, but the pass is old by then -> stop
const t2 = new PlateauTracker();
const t2 = primed();
stall(t2, PLATEAU_NUDGE_AFTER);
stall(t2, PLATEAU_STOP_AFTER - 1);
// a repeat pass (not progress) in the 8th round: passed within 2 rounds -> pass-nudge instead of stop
Expand All @@ -205,7 +212,7 @@ describe("PlateauTracker thresholds", () => {
expect(d.at(-1)).toBe("stop");
});
test("the pass-nudge is granted only once", () => {
const t = new PlateauTracker();
const t = primed();
stall(t, PLATEAU_NUDGE_AFTER);
stall(t, PLATEAU_STOP_AFTER - 1);
t.mem.lastCheckPassed = true;
Expand Down Expand Up @@ -257,7 +264,8 @@ async function boot() {
const userMessage = async (text: string) => hooks["chat.message"]({ sessionID: sid }, { message: {}, parts: [{ type: "text", text }] });
const idle = async () => hooks.event({ event: { type: "session.idle", properties: { sessionID: sid } } });
const setTodos = async (todos: any[]) => hooks.event({ event: { type: "todo.updated", properties: { todos } } });
return { dir, hooks, prompts, aborts, sid, toolRound, userMessage, idle, setTodos };
const before = async (ev: ToolEvent) => hooks["tool.execute.before"]({ tool: ev.tool, sessionID: sid, callID: `b${++n}` }, { args: ev.args });
return { dir, hooks, prompts, aborts, sid, toolRound, userMessage, idle, setTodos, before };
}
const tick = () => new Promise((r) => setTimeout(r, 0));

Expand Down Expand Up @@ -299,7 +307,7 @@ describe("plugin wiring", () => {
await h.idle();
expect(h.prompts).toHaveLength(1);
});
test("nudges once mid-loop after 6 stalled rounds, with open items; then aborts after 8 more and silences the todo gate", async () => {
test("nudges once mid-loop after 6 stalled rounds, with open items; then stops after 8 more by refusing tools, and silences the todo gate", async () => {
const h = await boot();
await h.userMessage("build me a thing");
await h.setTodos([{ content: "Ship README", status: "pending" }, { content: "Done part", status: "completed" }]);
Expand All @@ -311,9 +319,16 @@ describe("plugin wiring", () => {
expect(h.prompts[0]).toContain("- Ship README");
expect(h.prompts[0]).not.toContain("Done part");
for (let i = 0; i < PLATEAU_STOP_AFTER - 1; i++) await h.toolRound(edit("/p/app.ts"));
expect(h.aborts).toHaveLength(0);
await expect(h.before(read("/p/app.ts"))).resolves.toBeUndefined();
await h.toolRound(edit("/p/app.ts"));
expect(h.aborts).toEqual([h.sid]);
// Stopped: the session is NOT aborted (that killed running tools and showed a bare
// "Interrupted"); the next tool calls are refused with the reason instead.
expect(h.aborts).toHaveLength(0);
for (let i = 0; i < PLATEAU_STOP_DENIALS; i++) await expect(h.before(read("/p/app.ts"))).rejects.toThrow("LocalCode stopped this task");
expect(h.aborts).toHaveLength(0);
await expect(h.before(read("/p/app.ts"))).rejects.toThrow("stopped");
await tick();
expect(h.aborts).toEqual([h.sid]); // last resort after 3 ignored refusals
// the todo gate would normally re-prompt here (one todo still open) - it must not
await h.idle();
await tick();
Expand All @@ -336,7 +351,8 @@ describe("plugin wiring", () => {
const part = { id: "dup", type: "step-finish", sessionID: h.sid, messageID: "m", reason: "tool-calls" };
await h.hooks.event({ event: { type: "message.part.updated", properties: { part } } });
await h.hooks.event({ event: { type: "message.part.updated", properties: { part } } });
for (let i = 0; i < PLATEAU_NUDGE_AFTER - 2; i++) await h.toolRound(read("/p/a"));
await h.toolRound(read("/p/b")); // second novel read: still progress (nothing delivered)
for (let i = 0; i < PLATEAU_NUDGE_AFTER - 1; i++) await h.toolRound(read("/p/a"));
await tick();
expect(h.prompts).toHaveLength(0);
await h.userMessage("SYSTEM: some nudge"); // not a genuine user message
Expand All @@ -348,6 +364,7 @@ describe("plugin wiring", () => {
const h = await boot();
await h.userMessage("go");
await h.toolRound(check("ok", 0)); // first pass: progress
await h.toolRound(read("/p/a")); // first sighting of the read: novel, progress
for (let i = 0; i < PLATEAU_NUDGE_AFTER; i++) await h.toolRound(read("/p/a"));
for (let i = 0; i < PLATEAU_STOP_AFTER - 1; i++) await h.toolRound(read("/p/a"));
await h.toolRound(check("ok", 0)); // repeat pass, no progress, 8th round
Expand All @@ -356,7 +373,53 @@ describe("plugin wiring", () => {
expect(h.prompts).toHaveLength(2);
expect(h.prompts[1]).toContain("Your check passed — finish now");
for (let i = 0; i < PLATEAU_PASS_GRACE; i++) await h.toolRound(read("/p/a"));
expect(h.aborts).toEqual([h.sid]);
expect(h.aborts).toHaveLength(0);
await expect(h.before(read("/p/a"))).rejects.toThrow("LocalCode stopped this task");
});
test("a new user message after a stop lifts the refusal", async () => {
const h = await boot();
await h.userMessage("go");
await h.toolRound(edit("/p/a"));
for (let i = 0; i < PLATEAU_NUDGE_AFTER + PLATEAU_STOP_AFTER; i++) await h.toolRound(edit("/p/a"));
await expect(h.before(read("/p/a"))).rejects.toThrow("stopped");
await h.userMessage("ok try again");
await expect(h.before(read("/p/a"))).resolves.toBeUndefined();
expect(h.aborts).toHaveLength(0);
});
});

describe("exploration (nothing delivered yet)", () => {
test("rounds that cover new ground are progress; repeating the same calls is not", () => {
const t = new PlateauTracker();
for (let i = 0; i < 30; i++) { t.observe(read(`/p/f${i}.ts`)); expect(t.endRound()).toBe("none"); }
expect(t.noProgress).toBe(0);
for (let i = 0; i < 30; i++) { t.observe({ tool: "bash", args: { command: `curl -s https://x.test/${i}` } }); expect(t.endRound()).toBe("none"); }
expect(t.noProgress).toBe(0);
const d: string[] = [];
for (let i = 0; i < PLATEAU_NUDGE_AFTER; i++) { t.observe(read("/p/f0.ts")); d.push(t.endRound()); } // already seen
expect(d.at(-1)).toBe("nudge");
});
test("whitespace-only differences in a command are the same call", () => {
const t = new PlateauTracker();
t.observe({ tool: "bash", args: { command: "cat a.txt" } }); t.endRound();
for (let i = 0; i < PLATEAU_NUDGE_AFTER; i++) { t.observe({ tool: "bash", args: { command: "cat a.txt " } }); t.endRound(); }
expect(t.noProgress === 0 && t.nudged).toBe(true);
});
test("once something is delivered, only deliverables count", () => {
const t = new PlateauTracker();
t.observe(edit("/p/a.ts")); expect(t.endRound()).toBe("none");
const d: string[] = [];
for (let i = 0; i < PLATEAU_NUDGE_AFTER; i++) { t.observe(read(`/p/new${i}.ts`)); d.push(t.endRound()); }
expect(d.at(-1)).toBe("nudge");
});
test("a research turn is never stopped while it keeps trying new things", async () => {
const h = await boot();
await h.userMessage("find what londoners earn from that article");
for (let i = 0; i < 40; i++) await h.toolRound({ tool: "bash", args: { command: `curl -sL https://example.test/${i}` }, output: "..." });
await tick();
expect(h.prompts).toHaveLength(0);
expect(h.aborts).toHaveLength(0);
await expect(h.before(read("/p/a"))).resolves.toBeUndefined();
});
});

Expand Down
2 changes: 1 addition & 1 deletion tests/ui/prompt-scope.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,7 @@ test("completion rules stay out of helper prompts and are constant for the build
}
const first = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, first);
expect(first.system.join("\n")).toContain("requested multi-step workspace changes");
expect(first.system.join("\n")).toContain("WORKSPACE TASK COMPLETION");
await hooks["tool.execute.before"]({ tool: "read" }, { args: { filePath: "." } });
const afterTool = { system: [] as string[] };
await hooks["experimental.chat.system.transform"]({ agent: "build" }, afterTool);
Expand Down
Loading