diff --git a/CHANGELOG.md b/CHANGELOG.md index 21906fce..74a1de35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,27 @@ All notable changes to LocalCode will be documented here. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). +## 0.4.9 — 2026-09-27 + +### Changed + +- **Sessions compact at the point where this Mac stays fast, not at the edge of the + context window.** The server is loaded with the largest context that fits in + memory, and the runtime used to fill almost all of it before compacting; on the + same machine a model's decode speed halves and its prompt processing drops four to + six times between a 16k and a 64k prompt, so late in a long session every step + crawled and any cache miss took minutes. The supervisor now measures prompt- + processing and decode speed from the server's own timings as the session runs + and sets the compaction point where a full re-read still finishes within a minute + and decode keeps at least half its short-prompt speed. Nothing is a fixed token + count: the budget is relative to the context the server actually loaded, per + model, per machine, and it is recomputed as measurements arrive. +- **The runtime's idea of the context window is the server's.** The supervisor reads + the per-slot context from the server after it loads (with parallel slots the + window is split between them), and the launcher no longer falls back to a fixed + 32768. This is the class of failure behind "request exceeds the available context + size" in earlier 0.4 builds. + ## 0.4.8 — 2026-09-27 ### Fixed @@ -791,7 +812,7 @@ Small releases on the way to the docs site launch, listed together. is reserved exclusively for terminal, turn-ending failures. ### Docs -- README tested-hardware table adds the M5 (M5 Max, 128 GB) primary-dev row and +- README tested-hardware table adds the top-memory laptop row and notes that Linux is CI/dev-only while Apple Silicon (Metal) is the supported target. diff --git a/README.md b/README.md index d714406c..1421aa3a 100644 --- a/README.md +++ b/README.md @@ -84,7 +84,7 @@ localcode recommends a model by your Mac's memory and marks it with a star. You Min RAM is the memory at which localcode will recommend the model. You can pick a heavier one by hand. DiffusionGemma is a research model that is never recommended automatically. -Measured on a 128 GB Apple Silicon Mac with Qwen 3.6 35B-A3B UD-IQ2_M at a 131072-token context: about 89 tokens/s generation, about 1174 tokens/s prompt processing, and 12 to 15 seconds for a typical four-tool-call task. +Measured on a top-memory Apple Silicon laptop with Qwen 3.6 35B-A3B UD-IQ2_M at a 131072-token context: about 89 tokens/s generation, about 1174 tokens/s prompt processing, and 12 to 15 seconds for a typical four-tool-call task. ## Network diff --git a/pyproject.toml b/pyproject.toml index cb7bf6a0..b4e23452 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "localcode" -version = "0.4.8" +version = "0.4.9" description = "High-performance AI coding on consumer hardware." readme = "README.md" requires-python = ">=3.10" diff --git a/src/localcode/__init__.py b/src/localcode/__init__.py index f45d8c61..2b1525db 100644 --- a/src/localcode/__init__.py +++ b/src/localcode/__init__.py @@ -8,4 +8,4 @@ try: __version__ = _pkg_version("localcode") except PackageNotFoundError: # not installed (e.g. raw source checkout) - __version__ = "0.4.8" + __version__ = "0.4.9" diff --git a/src/localcode/bin/localcode-ui b/src/localcode/bin/localcode-ui index e2536768..be935176 100755 Binary files a/src/localcode/bin/localcode-ui and b/src/localcode/bin/localcode-ui differ diff --git a/src/localcode/ui/FORK_COMMIT b/src/localcode/ui/FORK_COMMIT index 817e891c..3d7cf236 100644 --- a/src/localcode/ui/FORK_COMMIT +++ b/src/localcode/ui/FORK_COMMIT @@ -1 +1 @@ -32c4172bc967b9e0c6be111dd7ba27f3ace7e2d5 +86d2d97133a017491e210e0bd00da194ea08c817 diff --git a/src/localcode/ui/context_budget.py b/src/localcode/ui/context_budget.py new file mode 100644 index 00000000..18a05944 --- /dev/null +++ b/src/localcode/ui/context_budget.py @@ -0,0 +1,189 @@ +"""How much context a session should use before compacting, measured on this machine. + +The server is launched with the largest context that fits in memory (the RAM +ladder in runtime.py). That is the right ceiling for *fitting*, not for *speed*: +on the same Mac a model's decode speed halves and its prefill speed drops four +to six times between a 16k and a 64k prompt, so a session that fills 120k +tokens spends minutes on every cache miss and crawls on every step. + +The budget is derived from the server's own timing lines (llama-server prints +one per request into server.log, which the supervisor owns): + + slot print_timing: id 0 | task 42 | prompt eval time = 727 ms / 650 tokens (...) + slot print_timing: id 0 | task 42 | eval time = 16471 ms / 852 tokens (...) + slot release: id 0 | task 42 | stop processing: n_tokens = 31577, truncated = 0 + +Each request gives one sample: how long the prompt was, how fast prefill ran +at that depth, how fast decode ran at that depth. The budget is the largest +prompt length at which a full re-read still finishes within REREAD_MAX_S and +decode keeps at least DECODE_FLOOR of its short-prompt speed. Both constants +are latencies a person feels, never token counts: every token number here is +relative to the context the server actually loaded. +""" +from __future__ import annotations + +import re +from statistics import median +from typing import NamedTuple + +REREAD_MAX_S = 60.0 # a cache miss (compaction, restart, model switch) may cost this long +DECODE_FLOOR = 0.5 # keep at least half of the short-prompt decode speed +PP_MIN_CHUNK = 256 # prefill chunks smaller than this measure overhead, not speed +TG_MIN_TOKENS = 16 # decode timings on fewer tokens are noise +PRIOR_DEPTH_FACTOR = 0.6 # prefill at the budget depth is slower than the warm-up's cold prefill +GRID = 16 # candidate budgets: n_ctx/16 steps +FLOOR_DIV = 4 # never budget below n_ctx/4: the fixed prefix must fit with room to work + +_TIMING = re.compile(r"task +(\d+) \| +(prompt eval time|eval time) = +([\d.]+) ms / +(\d+) tokens") +_RELEASE = re.compile(r"task +(\d+) \| stop processing: n_tokens = (\d+)") + + +class Sample(NamedTuple): + n_tokens: int # prompt length when the request finished + pp_tps: float | None # prefill tokens/s for this request (None: no prefill line) + pp_n: int # tokens prefilled + tg_tps: float | None # decode tokens/s + tg_n: int # tokens decoded + + +class TimingTable: + """Accumulates samples from server.log text fed in any chunking.""" + + def __init__(self) -> None: + self._pending: dict[int, dict] = {} + self._carry = "" + self.samples: list[Sample] = [] + + def feed(self, text: str) -> int: + text = self._carry + text + lines = text.split("\n") + self._carry = lines.pop() # partial last line + added = 0 + for line in lines: + m = _TIMING.search(line) + if m: + task = int(m.group(1)) + ms, n = float(m.group(3)), int(m.group(4)) + tps = n / (ms / 1000.0) if ms > 0 else None + d = self._pending.setdefault(task, {}) + if m.group(2).startswith("prompt"): + d["pp_tps"], d["pp_n"] = tps, n + else: + d["tg_tps"], d["tg_n"] = tps, n + continue + m = _RELEASE.search(line) + if m: + task, n_tokens = int(m.group(1)), int(m.group(2)) + d = self._pending.pop(task, None) + if d is None: + continue + self.samples.append(Sample(n_tokens, d.get("pp_tps"), int(d.get("pp_n", 0)), + d.get("tg_tps"), int(d.get("tg_n", 0)))) + added += 1 + if len(self._pending) > 512: # a request whose release line never came + for k in sorted(self._pending)[:-256]: + self._pending.pop(k, None) + return added + + def clear(self) -> None: + self._pending.clear(); self._carry = ""; self.samples.clear() + + +def _near(samples: list[Sample], c: int, step: int, key: str, min_n: str, min_val: int) -> float | None: + """Median speed measured near prompt length c. With nothing near c, the closest + measurement below c decides (speed never improves with depth); with nothing + below either, the shallowest measurement stands in (optimistic).""" + usable = [s for s in samples if getattr(s, key) is not None and getattr(s, min_n) >= min_val] + if not usable: + return None + radius = max(step, c // 4) + near = [getattr(s, key) for s in usable if abs(s.n_tokens - c) <= radius] + if near: + return float(median(near)) + below = [s for s in usable if s.n_tokens < c] + pool = below if below else usable + anchor = max(pool, key=lambda s: s.n_tokens) if below else min(pool, key=lambda s: s.n_tokens) + band = [getattr(s, key) for s in pool if abs(s.n_tokens - anchor.n_tokens) <= radius] + return float(median(band)) + + +def budget(n_ctx: int, reserve: int, samples: list[Sample], prior_pp_tps: float | None = None, + *, reread_max_s: float = REREAD_MAX_S, decode_floor: float = DECODE_FLOOR) -> tuple[int, dict]: + """Prompt tokens a session may reach before compacting, and how that was decided. + + Everything is relative to n_ctx, the context the server actually loaded + (per slot). `reserve` is the headroom one step may add (output plus tool + results); the budget never exceeds n_ctx - reserve and never drops below + n_ctx / FLOOR_DIV.""" + n_ctx = max(1, int(n_ctx)) + step = max(1, n_ctx // GRID) + cap = max(1, n_ctx - max(0, int(reserve))) + # The floor is the larger of a quarter of the context and twice the fixed + # prefix (system prompt, tool schemas, first message). Below the prefix a + # compaction cannot bring the prompt under the budget, so the runtime would + # compact on every step; measured live at 16k ctx before this rule existed. + prefix = prefix_tokens(samples) + floor = max(1, n_ctx // FLOOR_DIV, 2 * prefix if prefix else 0) + if floor > cap: + floor = cap + info: dict = {"n_ctx": n_ctx, "floor": floor, "cap": cap, "prefix": prefix, "samples": 0, "basis": "cap"} + + measured = [s for s in samples if s.pp_tps is not None and s.pp_n >= PP_MIN_CHUNK] + if not measured: + if prior_pp_tps and prior_pp_tps > 0: + c = int(PRIOR_DEPTH_FACTOR * prior_pp_tps * reread_max_s) + info.update(basis="prior", pp_prior=round(prior_pp_tps, 1)) + return _snap(min(cap, max(floor, c)), step, floor, cap), info + return cap, info + + info["samples"] = len(samples) + short = [s.tg_tps for s in samples if s.tg_tps is not None and s.tg_n >= TG_MIN_TOKENS and s.n_tokens < n_ctx // 8] + if not short: + deep = sorted((s for s in samples if s.tg_tps is not None and s.tg_n >= TG_MIN_TOKENS), key=lambda s: s.n_tokens) + short = [s.tg_tps for s in deep[: max(1, len(deep) // 4)]] if deep else [] + tg_short = float(median(short)) if short else None + info["tg_short"] = round(tg_short, 1) if tg_short else None + + best = floor + for k in range(1, GRID + 1): + c = step * k + if c > cap: + break + pp = _near(samples, c, step, "pp_tps", "pp_n", PP_MIN_CHUNK) + reread = (c / pp) if pp else 0.0 + if reread > reread_max_s: + info.update(limit="reread", at=c, reread_s=round(reread, 1), pp_tps=round(pp or 0, 1)) + break + if tg_short: + tg = _near(samples, c, step, "tg_tps", "tg_n", TG_MIN_TOKENS) + if tg is not None and tg < decode_floor * tg_short: + info.update(limit="decode", at=c, tg_tps=round(tg, 1)) + break + best = c + info["basis"] = "measured" + return _snap(best, step, floor, cap), info + + +PREFIX_FIRST_N = 3 # the first main request is among the first few large prefills after load + + +def prefix_tokens(samples: list[Sample]) -> int: + """Size of the fixed prompt prefix (system prompt, tool schemas, first + message). The first main request of a session prefills it cold; the other + early requests are small side requests (title) or the warm-up replay of the + same prefix, so the largest of the first few large prefills is the estimate. + Later prefills cannot be used: after a compaction they are short. 0 until a + large prefill has been observed.""" + cold = [s.pp_n for s in samples if s.pp_n >= PP_MIN_CHUNK and s.pp_tps is not None][:PREFIX_FIRST_N] + return max(cold) if cold else 0 + + +def _snap(c: int, step: int, floor: int, cap: int) -> int: + c = (c // step) * step + return min(cap, max(floor, c)) + + +def step_reserve(n_ctx: int, output: int, tool_output_bytes: int) -> int: + """Tokens one model step can add: its reply plus a couple of large tool + results. Bounded to half the context so tiny contexts keep room to work.""" + return min(max(1, n_ctx // 2), max(0, output) + 2 * (max(0, tool_output_bytes) // 4)) diff --git a/src/localcode/ui/launch.py b/src/localcode/ui/launch.py index 5b93f83b..ece8d0af 100644 --- a/src/localcode/ui/launch.py +++ b/src/localcode/ui/launch.py @@ -70,7 +70,13 @@ def _context_size(models_dir: Path, ctrl: int) -> int: from localcode.ui.server_cmd import context_size return context_size(str(models_dir / "x.gguf")) except Exception: # noqa: BLE001 - return 32768 + # No server and no runtime gateway to ask: the same policy knob the + # gateway starts from, never a literal token count. + try: + from localcode.config import load_config + return max(2048, int(load_config().max_context_chars) // 4) + except Exception: # noqa: BLE001 + return 2048 def write_config(path: Path, *, port: int, ctx: int, alias: str | None) -> None: @@ -97,6 +103,10 @@ def write_config(path: Path, *, port: int, ctx: int, alias: str | None) -> None: } }, "enabled_providers": ["localcode"], + # The supervisor's /status budget already includes the headroom one step + # can add (ui/context_budget.step_reserve), and the provider sets it as + # the model's input limit; a second reserve here would double-count. + "compaction": {"reserved": 0}, "tools": {"task": False}, "agent": {"plan": {"disable": True}}, "lsp": True, @@ -116,9 +126,11 @@ def attach(ctrl: int, status: dict, ui_bin: Path, project_dir: Path, alias: str """Open another UI window on the model server a running session owns.""" port = int(status["port"]) try: - ctx = int(status.get("ctx") or 0) or 32768 + ctx = int(status.get("ctx") or 0) except (TypeError, ValueError): - ctx = 32768 + ctx = 0 + if not ctx: + ctx = _context_size(_models_dir(), ctrl) current = status.get("current") or None if alias and current and alias != current: print(f"localcode: attaching to the running session's model {current}; use /models there to switch.", diff --git a/src/localcode/ui/supervisor.py b/src/localcode/ui/supervisor.py index 1f1de4e9..ed6b326c 100644 --- a/src/localcode/ui/supervisor.py +++ b/src/localcode/ui/supervisor.py @@ -92,6 +92,16 @@ def __init__(self, server_bin: str, port: int, models_dir: Path, ctx: int) -> No self.ram_gb = _system_ram_gb() self.bandwidth = _bandwidth() self.log = open(HERE / "server.log", "ab", buffering=0) # noqa: SIM115 (lives with the server) + # Context budget (ui/context_budget.py): the per-slot context the server + # really loaded, and how much of it this machine can use at speed. + self.ctx_total = ctx + self.slots = 1 + self.pp_prior: float | None = None # prefill tokens/s from the warm-up replay + self._timing = None # context_budget.TimingTable for the loaded model + self._log_offset = 0 # server.log bytes already fed to it + self._budget: int | None = None + self._budget_info: dict = {} + self._budget_lock = threading.Lock() self._warm_stop = threading.Event() self._warm_replay = None # warmup.Replay while the prefix is being pre-read @@ -189,8 +199,16 @@ def start(self, alias: str, wait_s: int = 240) -> bool: from localcode.ui.server_cmd import server_command cmd = server_command(str(gguf), self.port, alias) cmd[0] = self.server_bin - self.ctx = int(cmd[cmd.index("--ctx-size") + 1]) + self.ctx = self.ctx_total = int(cmd[cmd.index("--ctx-size") + 1]) + self.slots = int(cmd[cmd.index("--parallel") + 1]) if "--parallel" in cmd else 1 self.log.write(f"\n=== {time.ctime()} {' '.join(cmd)}\n".encode()) + with self._budget_lock: + from localcode.ui.context_budget import TimingTable + self._timing = TimingTable() + self._log_offset = self.log.tell() + self.pp_prior = None + self._budget = None + self._budget_info = {} self.proc = subprocess.Popen(cmd, stdout=self.log, stderr=subprocess.STDOUT, start_new_session=True, pass_fds=(self.lease_fd,) if getattr(self, "lease_fd", None) is not None else ()) log(f"supervisor: started llama-server pid {self.proc.pid} for {alias}") @@ -199,6 +217,7 @@ def start(self, alias: str, wait_s: int = 240) -> bool: return False if self.healthy(): self.current = alias + self._probe_props() self._start_warmup(alias) return True time.sleep(1) @@ -226,6 +245,12 @@ def _warmup_worker(self, alias: str, stop: threading.Event) -> None: tm = rp.run(toks) log(f"warmup: replayed {len(toks)} tokens for {alias} in {time.time() - t0:.1f}s " f"(server prompt_ms={tm.get('prompt_ms')}, cache_n={tm.get('cache_n')})") + try: + ms = float(tm.get("prompt_ms") or 0) + if ms > 0 and len(toks) >= 512: + self.pp_prior = len(toks) / (ms / 1000.0) + except (TypeError, ValueError): + pass except Exception as e: # noqa: BLE001 if rp.cancelled or stop.is_set(): log(f"warmup: replay cancelled after {time.time() - t0:.1f}s (user turn started)") @@ -279,6 +304,61 @@ def stop(self) -> None: self.proc.kill() self.proc = None + def _probe_props(self) -> None: + """The context the server really gives each request. With --parallel N the + flag's --ctx-size is split N ways; /props reports the per-slot figure.""" + try: + with urllib.request.urlopen(f"http://127.0.0.1:{self.port}/props", timeout=3) as r: + props = json.loads(r.read().decode()) + n_ctx = int((props.get("default_generation_settings") or {}).get("n_ctx") or 0) + slots = int(props.get("total_slots") or 0) + except Exception: # noqa: BLE001 + n_ctx, slots = 0, 0 + if slots > 0: + self.slots = slots + if n_ctx > 0: + self.ctx = n_ctx + elif self.slots > 1: + self.ctx = max(1, self.ctx_total // self.slots) + log(f"supervisor: context per slot {self.ctx} (total {self.ctx_total}, slots {self.slots})") + + def context_info(self) -> dict: + """ctx (per slot), the speed-derived budget, and the measurements behind it. + Reads whatever llama-server has appended to server.log since the last call.""" + from localcode.ui import context_budget as cb + with self._budget_lock: + table = self._timing + if table is not None: + try: + with open(HERE / "server.log", "rb") as f: + f.seek(self._log_offset) + chunk = f.read() + self._log_offset += len(chunk) + if chunk: + table.feed(chunk.decode("utf-8", "replace")) + except OSError: + pass + samples = table.samples if table is not None else [] + reserve = cb.step_reserve(self.ctx, min(8192, max(1, self.ctx // 4)), 50 * 1024) + # LOCALCODE_CONTEXT_BUDGET=off publishes no budget (the runtime then + # compacts at ctx - output as before); LOCALCODE_REREAD_MAX_S tunes + # the one latency constant. Both are escape hatches, not settings. + if os.environ.get("LOCALCODE_CONTEXT_BUDGET", "").strip().lower() in {"0", "off", "false", "no"}: + return {"ctx": self.ctx, "ctx_total": self.ctx_total, "slots": self.slots} + try: + reread_s = float(os.environ.get("LOCALCODE_REREAD_MAX_S") or cb.REREAD_MAX_S) + except ValueError: + reread_s = cb.REREAD_MAX_S + new, info = cb.budget(self.ctx, reserve, samples, self.pp_prior, reread_max_s=reread_s) + step = max(1, self.ctx // cb.GRID) + if self._budget is None or abs(new - self._budget) >= step: + if self._budget is not None: + log(f"supervisor: context budget {self._budget} -> {new} ({info.get('basis')}, {info.get('limit', 'no limit hit')})") + self._budget = new + self._budget_info = info + return {"ctx": self.ctx, "ctx_total": self.ctx_total, "slots": self.slots, + "budget": self._budget, "budget_info": info} + def healthy(self) -> bool: try: with urllib.request.urlopen(f"http://127.0.0.1:{self.port}/health", timeout=1) as r: @@ -684,9 +764,9 @@ def do_GET(self): key = parse_qs(u.query).get("group", [""])[0] return self._json(sup.quants(key)) if u.path == "/status": - return self._json(dict(sup.state, current=sup.current, port=sup.port, ctx=sup.ctx, + return self._json(dict(sup.state, current=sup.current, port=sup.port, group=sup.active.get("group"), filename=sup.active.get("filename"), - **sup.vision_info())) + **sup.context_info(), **sup.vision_info())) if u.path == "/models_dir": return self._json(sup.models_dir_info()) if u.path == "/voice/status": diff --git a/tests/test_ui_context_budget.py b/tests/test_ui_context_budget.py new file mode 100644 index 00000000..f7772523 --- /dev/null +++ b/tests/test_ui_context_budget.py @@ -0,0 +1,137 @@ +"""The context budget is measured on this machine, never a token literal.""" +from __future__ import annotations + +import json +import types + +import pytest + +from localcode.ui import context_budget as cb +from localcode.ui.launch import write_config + +LOG = """0.01 I slot print_timing: id 0 | task 1 | prompt eval time = 1000.00 ms / 2000 tokens ( 0.5 ms per token, 2000.00 tokens per second) +0.01 I slot print_timing: id 0 | task 1 | eval time = 1000.00 ms / 80 tokens (12.5 ms per token, 80.00 tokens per second) +0.01 I slot release: id 0 | task 1 | stop processing: n_tokens = 4000, truncated = 0 +""" + + +def sample(n_tokens, pp, tg, pp_n=1000, tg_n=100): + return cb.Sample(n_tokens, pp, pp_n, tg, tg_n) + + +def test_parses_timing_lines_in_any_chunking(): + t = cb.TimingTable() + for i in range(0, len(LOG), 7): # feed in 7-byte pieces + t.feed(LOG[i:i + 7]) + assert len(t.samples) == 1 + s = t.samples[0] + assert s.n_tokens == 4000 and s.pp_n == 2000 and s.tg_n == 80 + assert s.pp_tps == pytest.approx(2000) and s.tg_tps == pytest.approx(80) + + +def test_release_without_timing_is_ignored(): + t = cb.TimingTable() + t.feed("x | task 9 | stop processing: n_tokens = 100, truncated = 0\n") + assert t.samples == [] + + +def test_no_measurements_means_the_full_context_minus_headroom(): + b, info = cb.budget(65536, 4096, []) + assert b == 65536 - 4096 and info["basis"] == "cap" + + +def test_prior_from_warmup_prefill_speed(): + # 300 tok/s cold prefill: 0.6 * 300 * 60 s = 10800 tokens, snapped down to the n_ctx/16 grid (2048) = 10240 + b, info = cb.budget(32768, 2048, [], prior_pp_tps=300) + assert info["basis"] == "prior" and b == 10240 + b2, _ = cb.budget(131072, 8192, [], prior_pp_tps=1800) # 64800 -> grid of 8192 -> 57344 + assert b2 == 57344 + + +def test_reread_time_bounds_the_budget(): + n = 131072 + # prefill 2000 t/s at short prompts, 400 t/s from 40k on: 48k/400 = 120 s > 60 s; 40k/400 = 100 s; 32k/2000 ok + samples = [sample(4000, 2000, 80)] * 5 + [sample(45000, 400, 60)] * 5 + [sample(70000, 300, 45)] * 5 + b, info = cb.budget(n, 8192, samples) + assert info["basis"] == "measured" and info["limit"] == "reread" + assert b == 32768 + + +def test_decode_collapse_bounds_the_budget_even_when_prefill_is_fast(): + n = 65536 + samples = [sample(3000, 5000, 80)] * 5 + [sample(30000, 5000, 30)] * 5 # decode drops to 37% at 30k + b, info = cb.budget(n, 4096, samples) + assert info["limit"] == "decode" and b < 30000 and b >= n // 4 + + +def test_fast_machine_keeps_the_cap(): + n = 65536 + samples = [sample(3000, 5000, 80)] * 5 + [sample(50000, 4000, 70)] * 5 + b, info = cb.budget(n, 4096, samples) + assert b == n - 4096 and info["basis"] == "measured" and "limit" not in info + + +def test_budget_is_relative_to_the_loaded_context(): + small_samples = [sample(1000, 300, 12)] * 5 + [sample(6000, 100, 9)] * 5 + b16, _ = cb.budget(16384, 2048, small_samples) + assert 16384 // 4 <= b16 <= 16384 - 2048 + b8, _ = cb.budget(8192, 1024, small_samples) + assert 8192 // 4 <= b8 <= 8192 - 1024 + + +def test_floor_never_drops_below_twice_the_prompt_prefix(): + # 16k context, 4.8k prefix, very slow prefill: the re-read rule wants the floor, + # but a budget under the prefix would compact on every step. + # title request (631), first main request (4800), then post-compaction prefills (~900): the prefix is 4800 + samples = [cb.Sample(640, 800.0, 631, 20.0, 40), cb.Sample(5000, 100.0, 4800, 20.0, 100)] + [cb.Sample(6000, 50.0, 900, 15.0, 100)] * 6 + b, info = cb.budget(16384, 8192, samples, reread_max_s=1.0) + assert info["prefix"] == 4800 + assert info["floor"] == 8192 and b == 8192 # 2 x 4800 exceeds the cap: the cap is the floor + b1, info1 = cb.budget(32768, 8192, samples, reread_max_s=1.0) + assert info1["floor"] == 9600 and b1 == 9600 # 2 x prefix beats n_ctx/4 (8192); the re-read rule wanted less + b2, info2 = cb.budget(65536, 8192, [cb.Sample(5000, 100.0, 4800, 20.0, 100)] * 3, reread_max_s=1.0) + assert b2 >= 2 * 4800 and info2["floor"] == max(65536 // 4, 9600) + + +def test_step_reserve_scales_with_context(): + assert cb.step_reserve(131072, 8192, 50 * 1024) == 8192 + 2 * (50 * 1024 // 4) + assert cb.step_reserve(8192, 2048, 50 * 1024) == 4096 # capped at half the context + + +def test_launch_config_leaves_the_reserve_to_the_supervisor(tmp_path): + p = tmp_path / "c.json" + write_config(p, port=8123, ctx=40000, alias="m") + cfg = json.loads(p.read_text()) + assert cfg["compaction"] == {"reserved": 0} + assert cfg["provider"]["localcode"]["models"]["m"]["limit"]["context"] == 40000 + + +def test_supervisor_status_reports_per_slot_context_and_budget(tmp_path, monkeypatch): + from localcode.ui import supervisor as sv + monkeypatch.setattr(sv, "HERE", tmp_path) + (tmp_path / "server.log").write_bytes(b"") + sup = sv.Supervisor.__new__(sv.Supervisor) + sup.port, sup.ctx, sup.ctx_total, sup.slots = 8123, 131072, 131072, 1 + sup.pp_prior = None + sup._timing = cb.TimingTable(); sup._log_offset = 0; sup._budget = None; sup._budget_info = {} + import threading + sup._budget_lock = threading.Lock() + # /props says two slots of 65536 each + class R: + def __init__(self, body): self.body = body + def read(self): return json.dumps(self.body).encode() + def __enter__(self): return self + def __exit__(self, *a): return False + monkeypatch.setattr(sv.urllib.request, "urlopen", lambda url, timeout=0: R({"default_generation_settings": {"n_ctx": 65536}, "total_slots": 2})) + sup._probe_props() + assert sup.ctx == 65536 and sup.slots == 2 and sup.ctx_total == 131072 + info = sup.context_info() + assert info["ctx"] == 65536 and info["budget"] == 65536 - cb.step_reserve(65536, 8192, 50 * 1024) + # the server logs a slow deep request: the budget comes down on the next poll + with open(tmp_path / "server.log", "ab") as f: + for i in range(5): + f.write(f"x task {i} | prompt eval time = 1000.00 ms / 2000 tokens (x)\nx task {i} | eval time = 1000.00 ms / 100 tokens (x)\nx task {i} | stop processing: n_tokens = 3000, truncated = 0\n".encode()) + for i in range(5, 10): + f.write(f"x task {i} | prompt eval time = 10000.00 ms / 2000 tokens (x)\nx task {i} | eval time = 1000.00 ms / 100 tokens (x)\nx task {i} | stop processing: n_tokens = 40000, truncated = 0\n".encode()) + info2 = sup.context_info() + assert info2["budget"] < info["budget"] and info2["budget_info"]["limit"] == "reread"