diff --git a/README.md b/README.md index eb73ff3..e20cf6f 100644 --- a/README.md +++ b/README.md @@ -169,6 +169,8 @@ npm run harness -- status npm run harness -- status ``` +Resume continues an active sprint even when it has reached `maxSprints`. Completed evaluation rounds are reused only after their canonical results, verdicts, and frozen evidence are validated. Incomplete rounds are replayed, and failed rounds retain their repair instructions and generator session. + ## Providers Harness CLI is provider-agnostic. You choose which AI backend builds your app — and you can use different providers for different roles. diff --git a/src/core/harness.ts b/src/core/harness.ts index 8d785ca..8bb85df 100644 --- a/src/core/harness.ts +++ b/src/core/harness.ts @@ -89,6 +89,7 @@ interface SprintProgress { nextRound: number; latestEvalPath: string | null; latestEvalParsed: Record | null; + latestRepairDirectivePath: string | null; allEvalPaths: string[]; allFrozenEvidenceDirs: string[]; } @@ -450,7 +451,7 @@ export class HarnessRunner { runState.status = 'running'; await this.saveRunState(runState); - while (runState.sprint < this.config.maxSprints) { + while (runState.sprint < this.config.maxSprints || this.getActiveFeature(backlog, runState)) { const feature = await this.getCurrentOrNextFeature(runState, backlog); if (!feature) { const backlogComplete = backlog.features.every((candidate) => candidate.status === 'done'); @@ -475,7 +476,7 @@ export class HarnessRunner { const progress = await this.loadSprintProgress(runState, evalCriteria); let latestEvalPath = progress.latestEvalPath; let latestEvalParsed = progress.latestEvalParsed; - let latestRepairDirectivePath: string | null = null; + let latestRepairDirectivePath = progress.latestRepairDirectivePath; const allEvalPaths = [...progress.allEvalPaths]; let latestFrozenEvidenceDir = progress.allFrozenEvidenceDirs.at(-1) || null; const allFrozenEvidenceDirs = [...progress.allFrozenEvidenceDirs]; @@ -812,26 +813,62 @@ export class HarnessRunner { const allFrozenEvidenceDirs: string[] = []; let latestEvalPath: string | null = null; let latestEvalParsed: Record | null = null; + let latestRepairDirectivePath: string | null = null; + let passed = false; + // Only a contiguous sequence of finalized evaluations consumes repair rounds. + // Provider-written output alone may have survived an interrupted task. + runState.currentVerdictPath = null; for (let round = 0; round <= this.config.maxRepairRounds; round += 1) { const evalPath = this.evalPath(runState.sprint, round, runState.runDir); - if (!(await fileExists(evalPath))) continue; + const evalJsonPath = this.evalJsonPath(runState.sprint, round, runState.runDir); + const verdictPath = this.verdictPath(runState.sprint, round, runState.runDir); + if (!(await fileExists(evalPath)) || !(await fileExists(evalJsonPath)) || !(await fileExists(verdictPath))) break; - allEvalPaths.push(evalPath); - latestEvalPath = evalPath; - latestEvalParsed = await this.readParsedTaskLog(this.evaluatorLogName(runState.sprint, round), runState); + const verdict = await readJson(verdictPath, null); + if (!verdict || verdict.version !== 1 || verdict.sprint !== runState.sprint + || verdict.evaluationRound !== round || verdict.featureId !== runState.currentFeatureId) break; + const canonicalEval = await this.readCanonicalEvaluation(evalJsonPath); + if (canonicalEval.sprint !== runState.sprint || canonicalEval.evaluationRound !== round + || canonicalEval.feature.id !== runState.currentFeatureId) break; const frozenEvidenceDir = this.frozenEvidenceDir(runState.sprint, round, runState.runDir); - if (await fileExists(frozenEvidenceDir)) { + const hasFrozenEvidence = await fileExists(frozenEvidenceDir); + if ((verdict.reason !== 'smoke_failure' || (await fileExists(this.evidenceDir(runState.sprint, round, runState.runDir)))) + && !(await fileExists(this.frozenEvidenceManifestPath(frozenEvidenceDir)))) break; + if (hasFrozenEvidence) { + await this.assertFrozenEvidenceIntact(runState, frozenEvidenceDir); allFrozenEvidenceDirs.push(frozenEvidenceDir); } + + allEvalPaths.push(evalPath); + latestEvalPath = evalPath; + latestEvalParsed = canonicalEval as unknown as Record; + runState.currentEvalPath = evalPath; + runState.currentEvalJsonPath = evalJsonPath; + runState.currentVerdictPath = verdictPath; + passed = verdict.passed && resolvePass(latestEvalParsed, evalCriteria, this.getEffectivePassBarOverrides(runState)); + if (passed) break; + // Also recover an interruption between writing the verdict and directive. + latestRepairDirectivePath = await this.writeRepairDirective( + runState, round, verdict, evalCriteria, hasFrozenEvidence ? frozenEvidenceDir : null, + ); + } + + if (!passed) { + // A replay must not inherit a verdict from an interrupted attempt, including + // later files after a gap in the completed-round sequence. + for (let round = allEvalPaths.length; round <= this.config.maxRepairRounds; round += 1) { + await fs.rm(this.verdictPath(runState.sprint, round, runState.runDir), { force: true }); + } } return { - passed: resolvePass(latestEvalParsed, evalCriteria, runState.currentNegotiation?.passBarOverrides ?? {}), + passed, nextRound: allEvalPaths.length, latestEvalPath, latestEvalParsed, + latestRepairDirectivePath, allEvalPaths, allFrozenEvidenceDirs, }; diff --git a/src/core/providers/claude-sdk.ts b/src/core/providers/claude-sdk.ts index 921bb04..7e8b8d4 100644 --- a/src/core/providers/claude-sdk.ts +++ b/src/core/providers/claude-sdk.ts @@ -36,74 +36,9 @@ export class ClaudeSdkProvider implements ProviderRuntime { async runTask(task: TaskDefinition): Promise { const options = this.resolveOptionsForTask(task); - let assistantText = ''; - let resultText = ''; - let structuredOutput: Record | null = null; - let failureMessage: string | null = null; - let claudeCodeVersion: string | undefined; - const responseModels = new Set(); - let sessionId: string | undefined; - - try { - // eslint-disable-next-line @typescript-eslint/no-explicit-any - for await (const message of query({ prompt: task.prompt, options } as any)) { - const msg = message as Record; - - if (msg.type === 'system' && msg.subtype === 'init') { - sessionId = msg.session_id as string; - claudeCodeVersion = msg.claude_code_version as string; - } - - if (msg.type === 'assistant') { - const assistant = msg.message as Record; - if (typeof assistant?.model === 'string') responseModels.add(assistant.model); - const content = assistant?.content; - if (Array.isArray(content)) { - for (const block of content) { - const b = block as Record; - if (b.type === 'text') { - assistantText += b.text as string; - } - if (b.type === 'tool_use') { - this.onUpdate({ sessionUpdate: 'tool_call', title: b.name as string }, task); - } - } - } - } - - if (msg.type === 'result' && msg.subtype === 'success') { - resultText = (msg.result as string) || ''; - if (isPlainObject(msg.structured_output)) { - structuredOutput = msg.structured_output; - } - } - - failureMessage = claudeResultError(msg) || failureMessage; - } - } catch (error) { - if (failureMessage) throw new Error(`Claude Agent SDK: ${failureMessage}`); - // SDK process crashed — use whatever we collected so far. - const partialText = (resultText || assistantText).trim(); - if (partialText) { - const parsed = extractJsonObject(partialText); - if (parsed) { - return { rawText: partialText, parsed, meta: { sessionId, crashed: true } }; - } - } - throw error; - } - - if (failureMessage) { - throw new Error(`Claude Agent SDK: ${failureMessage}`); - } - - const rawText = (resultText || assistantText).trim(); - - return { - rawText, - parsed: structuredOutput || extractJsonObject(rawText), - meta: { sessionId, structuredOutput: !!structuredOutput, claudeCodeVersion, responseModels: [...responseModels] }, - }; + // eslint-disable-next-line @typescript-eslint/no-explicit-any + return collectClaudeTaskResult(query({ prompt: task.prompt, options } as any), + (update) => this.onUpdate(update, task)); } /** @@ -168,6 +103,77 @@ export class ClaudeSdkProvider implements ProviderRuntime { } } +/** Consume the SDK stream; only a terminal successful result completes a task. */ +export async function collectClaudeTaskResult( + messages: AsyncIterable, + onUpdate: (update: Record) => void = () => {}, +): Promise { + let assistantText = ''; + let resultText = ''; + let structuredOutput: Record | null = null; + let failureMessage: string | null = null; + let completed = false; + let claudeCodeVersion: string | undefined; + const responseModels = new Set(); + let sessionId: string | undefined; + + try { + for await (const message of messages) { + const msg = message as Record; + + if (msg.type === 'system' && msg.subtype === 'init') { + sessionId = msg.session_id as string; + claudeCodeVersion = msg.claude_code_version as string; + } + + if (msg.type === 'assistant') { + const assistant = msg.message as Record; + if (typeof assistant?.model === 'string') responseModels.add(assistant.model); + const content = assistant?.content; + if (Array.isArray(content)) { + for (const block of content) { + const b = block as Record; + if (b.type === 'text') { + assistantText += b.text as string; + } + if (b.type === 'tool_use') { + onUpdate({ sessionUpdate: 'tool_call', title: b.name as string }); + } + } + } + } + + if (msg.type === 'result' && msg.subtype === 'success') { + completed = !msg.is_error; + resultText = (msg.result as string) || ''; + if (isPlainObject(msg.structured_output)) { + structuredOutput = msg.structured_output; + } + } + + failureMessage = claudeResultError(msg) || failureMessage; + } + } catch (error) { + if (failureMessage) throw new Error(`Claude Agent SDK: ${failureMessage}`); + // Partial assistant JSON is not a successful terminal SDK result. + throw error; + } + + if (failureMessage) { + throw new Error(`Claude Agent SDK: ${failureMessage}`); + } + + if (!completed) throw new Error('Claude Agent SDK: stream ended without a successful result'); + + const rawText = (resultText || assistantText).trim(); + + return { + rawText, + parsed: structuredOutput || extractJsonObject(rawText), + meta: { sessionId, structuredOutput: !!structuredOutput, claudeCodeVersion, responseModels: [...responseModels] }, + }; +} + // SDK failures have specific subtypes (for example error_max_turns), not just "error". export function claudeResultError(message: Record): string | null { if (message.type !== 'result') return null; diff --git a/src/core/providers/codex-client.ts b/src/core/providers/codex-client.ts index 327081a..a298317 100644 --- a/src/core/providers/codex-client.ts +++ b/src/core/providers/codex-client.ts @@ -69,6 +69,8 @@ export class CodexAppServerClient { private rl: readline.Interface | null = null; private started = false; private activeThreadId: string | null = null; + private transportError: Error | null = null; + private failureListeners = new Set<(error: Error) => void>(); constructor(options: CodexAppServerClientOptions) { this.command = options.command; @@ -88,11 +90,13 @@ export class CodexAppServerClient { async start(): Promise { if (this.started) return; - this.child = spawn(this.command, this.args, { + this.transportError = null; + const child = spawn(this.command, this.args, { cwd: this.cwd, env: { ...process.env, ...this.env }, stdio: ['pipe', 'pipe', 'pipe'], }); + this.child = child; this.child.stdout?.setEncoding('utf8'); this.child.stderr?.setEncoding('utf8'); @@ -108,14 +112,13 @@ export class CodexAppServerClient { this.onStdErr(String(chunk)); }); - this.child.on('error', (error: Error) => { - this.rejectAllPending(error); - }); - - this.child.on('close', (code: number | null) => { - if (code !== 0 && this.pending.size > 0) { - this.rejectAllPending(new Error(`Codex app-server exited with code ${code}`)); - } + const fail = (error: Error): void => { + if (this.child === child) this.failTransport(error); + }; + child.on('error', fail); + child.stdin?.on('error', fail); + child.on('close', (code: number | null, signal: NodeJS.Signals | null) => { + fail(new Error(`Codex app-server exited with ${signal ? `signal ${signal}` : `code ${code}`}`)); }); this.started = true; @@ -136,30 +139,40 @@ export class CodexAppServerClient { } async close(): Promise { - if (!this.started || !this.child) return; + const child = this.child; + if (!child) return; - if (this.activeThreadId) { + // Unsubscribe is best effort: a wedged server must not prevent task cleanup. + if (this.activeThreadId && !this.transportError) { + let timeout: ReturnType | undefined; try { - await this.request('thread/unsubscribe', { threadId: this.activeThreadId }); + await Promise.race([ + this.request('thread/unsubscribe', { threadId: this.activeThreadId }), + new Promise((resolve) => { timeout = setTimeout(resolve, 250); }), + ]); } catch { - // Ignore best-effort cleanup failures. + // Transport failures are already delivered to active requests/turns. + } finally { + clearTimeout(timeout); } } - try { - this.child.stdin?.end(); - } catch { - // ignore - } - - if (!this.child.killed) { - this.child.kill('SIGTERM'); + this.failTransport(new Error('Codex app-server client closed')); + child.stdin?.end(); + if (child.exitCode === null && child.signalCode === null) { + child.kill('SIGTERM'); + const forceKill = setTimeout(() => { + if (child.exitCode === null && child.signalCode === null) child.kill('SIGKILL'); + }, 1000); + forceKill.unref(); + child.once('close', () => clearTimeout(forceKill)); } this.started = false; this.activeThreadId = null; this.rl?.close(); this.rl = null; + this.child = null; } request(method: string, params: unknown): Promise { @@ -167,8 +180,17 @@ export class CodexAppServerClient { const payload = { id, method, params }; return new Promise((resolve, reject) => { + if (this.transportError) { + reject(this.transportError); + return; + } this.pending.set(id, { resolve, reject, method, createdAt: nowIso() }); - this.send(payload); + try { + this.send(payload); + } catch (error) { + this.pending.delete(id); + reject(error instanceof Error ? error : new Error(String(error))); + } }); } @@ -234,6 +256,7 @@ export class CodexAppServerClient { const cleanup = (): void => { clearTimeout(timeout); this.notificationListeners.delete(listener); + this.failureListeners.delete(onFailure); }; const listener: NotificationListener = (message) => { @@ -265,7 +288,10 @@ export class CodexAppServerClient { reject(new Error(`Codex ChatGPT login failed: ${errorText}`)); }; + const onFailure = (error: Error): void => { cleanup(); reject(error); }; this.notificationListeners.add(listener); + this.failureListeners.add(onFailure); + if (this.transportError) onFailure(this.transportError); }); } @@ -309,6 +335,8 @@ export class CodexAppServerClient { const method = asString(message.method); if (!method) return; const params = asRecord(message.params); + if (params?.threadId && params.threadId !== threadId) return; + if (params?.turnId && turnId && params.turnId !== turnId) return; if (method === 'turn/plan/updated') { const entries = Array.isArray(params?.plan) @@ -420,6 +448,7 @@ export class CodexAppServerClient { } while (!completed) { + if (this.transportError) throw this.transportError; await sleep(25); } @@ -555,6 +584,12 @@ export class CodexAppServerClient { }); } + private failTransport(error: Error): void { + this.transportError ||= error; + this.rejectAllPending(this.transportError); + for (const listener of this.failureListeners) listener(this.transportError); + } + private rejectAllPending(error: Error): void { for (const [, pending] of this.pending) { pending.reject(error); diff --git a/test/provider-streams.test.mjs b/test/provider-streams.test.mjs new file mode 100644 index 0000000..a902fed --- /dev/null +++ b/test/provider-streams.test.mjs @@ -0,0 +1,93 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { CodexAppServerClient } from '../dist/core/providers/codex-client.js'; +import { collectClaudeTaskResult } from '../dist/core/providers/claude-sdk.js'; + +const turn = { prompt: 'test', cwd: process.cwd(), sandboxMode: 'workspaceWrite', writableRoots: [], networkAccess: false, approvalMode: 'never' }; +function client(mode, t) { + const script = ` + const readline = require('node:readline'); + const send = value => process.stdout.write(JSON.stringify(value) + '\\n'); + const mode = ${JSON.stringify(mode)}; + if (mode === 'stubborn') { process.on('SIGTERM', () => {}); setInterval(() => {}, 1000); } + readline.createInterface({input:process.stdin}).on('line', line => { + const m = JSON.parse(line); + if (!m.id) return; + if (mode === 'initialize-exit') process.exit(0); + if (m.method === 'thread/unsubscribe') return; + let result = {}; + if (m.method === 'account/read') result = mode === 'login-exit' ? {requiresOpenaiAuth:true} : {account:{type:'chatgpt'}}; + if (m.method === 'account/login/start') result = {loginId:'login',authUrl:'https://example.invalid'}; + if (m.method === 'thread/start' || m.method === 'thread/resume') result = {thread:{id:'thread'}}; + if (m.method === 'turn/start') result = {turn:{id:'turn'}}; + send({id:m.id,result}); + if (m.method === 'account/login/start') setTimeout(()=>process.exit(3),20); + if (m.method === 'turn/start') { + if (mode === 'turn-exit') setTimeout(()=>process.exit(7),20); + else if (mode === 'complete') { + send({method:'item/completed',params:{threadId:'unrelated',turnId:'turn',item:{type:'agentMessage',phase:'final_answer',text:'wrong'}}}); + send({method:'item/completed',params:{threadId:'thread',turnId:'turn',item:{type:'agentMessage',phase:'final_answer',text:'correct'}}}); + send({method:'item/completed',params:{threadId:'thread',turnId:'other',item:{type:'agentMessage',phase:'final_answer',text:'wrong'}}}); + send({method:'turn/completed',params:{threadId:'thread',turn:{id:'turn',status:'completed'}}}); + } + } + });`; + const c = new CodexAppServerClient({command:process.execPath,args:['-e',script]}); + // Keep regressions from leaving fixture subprocesses alive after a test timeout. + t.after(() => c.child?.kill('SIGKILL')); + return c; +} +for (const [mode, code] of [['initialize-exit',0],['turn-exit',7],['login-exit',3]]) { + test(`Codex rejects ${mode} without hanging`, {timeout:3000}, async (t) => { + const c = client(mode, t); + try { await assert.rejects(c.runTurn(turn), new RegExp(`exited with code ${code}`)); } + finally { await c.close(); } + }); +} +test('Codex isolates turn notifications and bounds unresponsive unsubscribe', {timeout:3000}, async (t) => { + const c = client('complete', t); + try { assert.equal((await c.runTurn(turn)).text, 'correct'); } + finally { await c.close(); } + assert.equal(c.pending.size, 0); +}); +test('Codex close interrupts an active turn and permits restarting the client', {timeout:3000}, async (t) => { + const c = client('waiting', t); + const running = c.runTurn(turn); + const rejected = assert.rejects(running, /client closed/); + while (!c.activeThreadId) await new Promise(resolve=>setTimeout(resolve,10)); + await c.close(); + await rejected; + await c.start(); + await c.close(); +}); + +test('Codex reports a missing executable and cleans up failed initialization', {timeout:3000}, async () => { + const c = new CodexAppServerClient({command:'/definitely-missing-harness-codex'}); + try { await assert.rejects(c.start(), /ENOENT/); } + finally { await c.close(); } + assert.equal(c.pending.size, 0); +}); +test('Codex forces cleanup when a child ignores SIGTERM', {timeout:4000}, async (t) => { + const c = client('stubborn', t); + await c.start(); + const child = c.child; + const closed = new Promise(resolve => child.once('close', (code, signal) => resolve({code,signal}))); + await c.close(); + assert.deepEqual(await closed, {code:null,signal:'SIGKILL'}); +}); + +async function* stream(messages, error) { yield* messages; if (error) throw error; } +const partial = {type:'assistant',message:{model:'claude-opus-5-5',content:[{type:'text',text:'{"summary":"partial","status":"done"}'}]}}; +test('Claude rejects partial JSON after a transport failure or truncated stream', async () => { + await assert.rejects(collectClaudeTaskResult(stream([partial],new Error('connection lost'))), /connection lost/); + await assert.rejects(collectClaudeTaskResult(stream([partial])), /without a successful result/); +}); +test('Claude preserves terminal failure over partial JSON and transport errors', async () => { + await assert.rejects(collectClaudeTaskResult(stream([partial,{type:'result',subtype:'error_max_turns',errors:['turn limit']}],new Error('closed'))), /turn limit/); +}); +test('Claude uses successful structured output and retains provider metadata', async () => { + const parsed = {summary:'verified',status:'done'}; + const result = await collectClaudeTaskResult(stream([{type:'system',subtype:'init',session_id:'session',claude_code_version:'2.1.287'},partial,{type:'result',subtype:'success',result:'done',structured_output:parsed}])); + assert.deepEqual(result.parsed,parsed); + assert.deepEqual(result.meta,{sessionId:'session',structuredOutput:true,claudeCodeVersion:'2.1.287',responseModels:['claude-opus-5-5']}); +}); diff --git a/test/resume-runtime.test.mjs b/test/resume-runtime.test.mjs new file mode 100644 index 0000000..b7560a7 --- /dev/null +++ b/test/resume-runtime.test.mjs @@ -0,0 +1,218 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; + +import { loadConfig } from '../dist/core/config.js'; +import { HarnessRunner } from '../dist/core/harness.js'; + +const ALL_ROLES = ['researcher', 'planner', 'generator', 'evaluator']; +const PASSING_SCORES = { + conceptAlignment: 5, + completeness: 5, + craft: 5, + intentionality: 5, + artifactCompatibility: 5, + verificationEvidence: 5, +}; +const FAILING_SCORES = Object.fromEntries(Object.keys(PASSING_SCORES).map((key) => [key, 1])); + +const silentOutput = { log() {} }; + +async function setupMinimalRun(extraConfig = {}) { + const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), 'harness-minimal-run-')); + const configPath = path.join(tempRoot, 'harness.config.json'); + await fs.writeFile( + configPath, + `${JSON.stringify({ workspace: 'workspace', runRoot: 'run-root', ...extraConfig }, null, 2)}\n`, + ); + const { config } = await loadConfig(configPath, {}, { profile: 'minimal' }); + return { tempRoot, config }; +} + +function createFakeRegistry(options) { + const labels = []; + return { + labels, + getRouting() { + return Object.fromEntries(ALL_ROLES.map((role) => [role, 'claude-sdk'])); + }, + getProviderName() { + return 'claude-sdk'; + }, + getTaskCapabilities(role) { + return { role, provider: 'claude-sdk', hasBrowserQa: false, supportsSessionResume: true }; + }, + async runTask(task) { + labels.push(task.label); + await options.beforeTask?.(task); + + if (task.kind === 'generator') { + for (const artifactPath of Object.values(task.artifacts)) { + await fs.mkdir(path.dirname(artifactPath), { recursive: true }); + await fs.writeFile(artifactPath, 'updated by fake generator\n'); + } + return { + rawText: '{"status":"ok","summary":"implemented"}', + parsed: { status: 'ok', summary: 'implemented', filesTouched: [], commandsRun: [] }, + meta: { sessionId: 'fake-session' }, + }; + } + + if (task.kind === 'evaluator') { + const scores = options.scoresForRound(task.evaluationRound ?? 0); + const canonicalEval = { + version: 1, + sprint: task.sprintNumber ?? 1, + evaluationRound: task.evaluationRound ?? 0, + feature: { id: task.feature.id, title: task.feature.title }, + confidence: 'high', + evidenceQuality: 'strong', + summary: 'fake evaluation', + scores, + contractCriteria: [], + projectPrinciples: [], + bugs: [], + suggestedRepairPlan: [], + notes: [], + sourceMarkdownPath: task.artifacts.eval, + devSmoke: { required: false, ok: true, logPath: null, url: null }, + }; + await fs.mkdir(path.dirname(task.artifacts.eval), { recursive: true }); + await fs.writeFile(task.artifacts.eval, '# Fake evaluation\n'); + await fs.writeFile(task.artifacts.evalJson, JSON.stringify(canonicalEval, null, 2)); + await options.afterEvaluation?.(task, canonicalEval); + return { + rawText: JSON.stringify(canonicalEval), + parsed: { summary: 'fake evaluation', scores }, + meta: {}, + }; + } + + throw new Error(`Unexpected ${task.kind} task in minimal mode: ${task.label}`); + }, + }; +} + + +async function interruptedRun(config, registry) { + const runner = new HarnessRunner(config, registry, silentOutput); + await assert.rejects(() => runner.runNew('Build a tiny tool'), /disconnect/); + const [runId] = await fs.readdir(path.join(config.runRoot, 'runs')); + return { runner, runId, runDir: path.join(config.runRoot, 'runs', runId) }; +} + +test('resume continues an interrupted active sprint at maxSprints', async (t) => { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1 }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + let first = true; + const registry = createFakeRegistry({ scoresForRound: () => PASSING_SCORES, beforeTask() { + if (first) { first = false; throw new Error('disconnect'); } + } }); + const { runner, runId } = await interruptedRun(config, registry); + const state = await runner.resume(runId); + assert.equal(state.status, 'completed'); + assert.equal(state.sprint, 1); + assert.deepEqual(registry.labels, ['generator-s1-r0', 'generator-s1-r0', 'evaluator-s1-r0']); +}); + +test('resume replays provider-written evaluation artifacts without a committed verdict', async (t) => { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1, maxRepairRounds: 0 }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + let first = true; + const registry = createFakeRegistry({ scoresForRound: () => PASSING_SCORES, async afterEvaluation(task, evaluation) { + if (first) { + first = false; + const runDir = path.dirname(path.dirname(task.artifacts.eval)); + await fs.writeFile(path.join(runDir, 'logs', 'evaluator-s01-r00.parsed.json'), JSON.stringify(evaluation)); + throw new Error('disconnect'); + } + } }); + const { runner, runId } = await interruptedRun(config, registry); + assert.equal((await runner.resume(runId)).status, 'completed'); + assert.deepEqual(registry.labels, ['generator-s1-r0', 'evaluator-s1-r0', 'generator-s1-r0', 'evaluator-s1-r0']); +}); + +test('resume restores failed evaluation and its repair directive without consuming another round', async (t) => { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1, maxRepairRounds: 1 }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + let firstRepair = true; + let resumedPrompt; + const registry = createFakeRegistry({ scoresForRound: (round) => round === 0 ? FAILING_SCORES : PASSING_SCORES, + beforeTask(task) { + if (task.label === 'generator-s1-r1') { + if (firstRepair) { firstRepair = false; throw new Error('disconnect'); } + resumedPrompt = task.prompt; + assert.equal(task.resumeSessionId, 'fake-session'); + } + }, + }); + const { runner, runId, runDir } = await interruptedRun(config, registry); + await fs.rm(path.join(runDir, 'repair-directives', 'repair-s01-r00.json')); + assert.equal((await runner.resume(runId)).status, 'completed'); + assert.match(resumedPrompt, /repair-s01-r00.json/); + assert.deepEqual(registry.labels, ['generator-s1-r0', 'evaluator-s1-r0', 'generator-s1-r1', 'generator-s1-r1', 'evaluator-s1-r1']); +}); + +test('resume trusts finalized evaluation but rejects altered frozen evidence', async (t) => { + for (const tamper of [false, true]) { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1 }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + const registry = createFakeRegistry({ scoresForRound: () => PASSING_SCORES }); + const runner = new HarnessRunner(config, registry, silentOutput); + const originalMarkDone = runner.markFeatureDone; + runner.markFeatureDone = () => { throw new Error('disconnect'); }; + await assert.rejects(() => runner.runNew('Build a tiny tool'), /disconnect/); + const [runId] = await fs.readdir(path.join(config.runRoot, 'runs')); + runner.markFeatureDone = originalMarkDone; + if (tamper) { + const state = await runner.status(runId); + const frozenDir = runner.frozenEvidenceDir(1, 0, state.runDir); + const manifestPath = path.join(frozenDir, 'manifest.json'); + const manifest = JSON.parse(await fs.readFile(manifestPath, 'utf8')); + manifest.files.push({ path: 'missing-evidence.txt', sha256: 'unavailable' }); + await fs.writeFile(manifestPath, JSON.stringify(manifest)); + await assert.rejects(() => runner.resume(runId), /Frozen evaluator evidence was modified/); + } else { + assert.equal((await runner.resume(runId)).status, 'completed'); + } + assert.deepEqual(registry.labels, ['generator-s1-r0', 'evaluator-s1-r0']); + } +}); + + +test('resume replays a round if its frozen evidence was not completed', async (t) => { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1, maxRepairRounds: 0 }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + const registry = createFakeRegistry({ scoresForRound: () => PASSING_SCORES }); + const runner = new HarnessRunner(config, registry, silentOutput); + const originalMarkDone = runner.markFeatureDone; + runner.markFeatureDone = () => { throw new Error('disconnect'); }; + await assert.rejects(() => runner.runNew('Build a tiny tool'), /disconnect/); + const [runId] = await fs.readdir(path.join(config.runRoot, 'runs')); + const state = await runner.status(runId); + await fs.rm(runner.frozenEvidenceDir(1, 0, state.runDir), { recursive: true, force: true }); + runner.markFeatureDone = originalMarkDone; + assert.equal((await runner.resume(runId)).status, 'completed'); + assert.deepEqual(registry.labels, ['generator-s1-r0', 'evaluator-s1-r0', 'generator-s1-r0', 'evaluator-s1-r0']); +}); + +test('resume retains a synthetic smoke-failure verdict without evaluator evidence', async (t) => { + const { tempRoot, config } = await setupMinimalRun({ maxSprints: 1, maxRepairRounds: 1, smoke: { test: 'exit 1' } }); + t.after(() => fs.rm(tempRoot, { recursive: true, force: true })); + let firstRepair = true; + const registry = createFakeRegistry({ scoresForRound: () => PASSING_SCORES, beforeTask(task) { + if (task.label === 'generator-s1-r1' && firstRepair) { + firstRepair = false; + throw new Error('disconnect'); + } + } }); + const { runner, runId, runDir } = await interruptedRun(config, registry); + const verdict = JSON.parse(await fs.readFile(path.join(runDir, 'verdicts', 'verdict-01-r00.json'), 'utf8')); + assert.equal(verdict.reason, 'smoke_failure'); + await assert.rejects(fs.access(runner.frozenEvidenceDir(1, 0, runDir)), { code: 'ENOENT' }); + config.smoke.test = 'exit 0'; + assert.equal((await runner.resume(runId)).status, 'completed'); + assert.deepEqual(registry.labels, ['generator-s1-r0', 'generator-s1-r1', 'generator-s1-r1', 'evaluator-s1-r1']); +});