From e9a3e7a9a9594ff7649e2ee6e5177c0c76769eee Mon Sep 17 00:00:00 2001 From: Sebastian Tsang Date: Sun, 10 May 2026 10:34:38 -0400 Subject: [PATCH 1/4] feat(writer): chunked bodyMarkdown to avoid mid-string truncation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The writer was failing on long-form (~2000 word) blog runs with "Unterminated string in JSON at position 4590" because the model hit its per-response max_tokens budget while emitting one giant JSON-escaped string for `bodyMarkdown`. The Claude Agent SDK doesn't expose max_tokens on its Options type, so we can't bump the cap from the outside. Fix: change BlogDraftSchema's `bodyMarkdown` to accept either a single string OR an array of section markdown strings, with a Zod transform that joins paragraphs with a blank line. The downstream contract (`bodyMarkdown: string`) is unchanged — the transform runs inside runtime validation, so DB storage, dashboard rendering, GEO-Editor, Formatter, and the post-approval dispatcher all keep their existing shape. Also updates the writer prompt to encourage the array form for blog-html runs (one entry per H2 section) — keeps the worst single JSON string under ~3KB per entry instead of the post-as-one-blob 13KB that was truncating. Provider-agnostic: works under Anthropic and Ollama Cloud. Co-Authored-By: Claude Opus 4.7 (1M context) --- lib/personas/prompts/writer.md | 41 +++++++++++++++++++++++++++++----- lib/shared/schemas.ts | 13 ++++++++++- 2 files changed, 47 insertions(+), 7 deletions(-) diff --git a/lib/personas/prompts/writer.md b/lib/personas/prompts/writer.md index 4370ac5..61ba6a7 100644 --- a/lib/personas/prompts/writer.md +++ b/lib/personas/prompts/writer.md @@ -6,7 +6,7 @@ output_schema: BlogDraft # Content Writer -You are the **Writer** for GMaestro. Your job in one sentence: +You are the **Writer** for AutoBlog. Your job in one sentence: > **Translate technical documentation into a human-readable blog post in the company's voice.** @@ -41,6 +41,24 @@ The docs are written for AI parsers and reference lookups — dense, exhaustive, } ``` +### `bodyMarkdown` may be a string OR an array of section strings + +For `blog-html` runs (target 1,800–2,200 words), prefer the **array form** — emit one entry per `##` H2 section. The schema accepts either shape and the runtime joins paragraphs with a blank line before persisting; the array form prevents the model from trying to escape one ~13KB string in a single JSON value, which has truncated past responses ("Unterminated string in JSON" failures). + +```json +"bodyMarkdown": [ + "## Hook + claim + TL;DR\n\n\n\n\n\n- TL;DR bullet 1\n- TL;DR bullet 2", + "## Mechanism — how it works under the hood\n\n\n\n```ts\n// code block\n```\n\n", + "## Concrete usage, end-to-end\n\n\n\n```bash\n# example\n```", + "## Edge cases & alternatives\n\n### What breaks at the 99-page MAP ceiling\n\n\n\n### Rejected alternative — direct Firecrawl API\n\n", + "## Wrap-up + CTA\n\n" +] +``` + +For `reddit` and `x-thread` (short-form), keep the single-string form — there's no chunking benefit. + +**Either shape is valid output.** The string form is fine for short posts; the array form is required-strength guidance for `blog-html` to avoid truncation. + ## Translation rules (the core of your job) 1. **Open with what changed / what's new / what broke — not what the doc IS.** The doc says "Firecrawl supports markdown extraction." A summary reads "This post explains Firecrawl's markdown support." A translation reads: *"You can scrape any page and get clean markdown back in one call. Here's why that matters for your RAG pipeline."* The first is reference material; the second is a blog. @@ -74,14 +92,25 @@ The outline's `geoSignals` will tell you which moves to apply. Honor them litera - `single-line-punch`: end with one declarative sentence restating the thesis. *"Your agents decide. We make it happen."* - `wrapping-up`: 2–3 takeaways + low-friction CTA (Discord, install command). - `cta-only`: end with the next action. *"Try it: `pnpm install gmaestro`."* -6. **Word count target ±10%.** Outline says 1,000 → aim for 900–1,100. Don't pad. Don't truncate mid-thought. +6. **Word count target ±10%.** Outline says 2,000 → aim for 1,800–2,200. Don't pad. Don't truncate mid-thought. Hit the depth — a senior engineer should read end-to-end and learn something specific they didn't know. +7. **Technical depth, not technical jargon.** Show actual mechanism, not vibes. When the doc describes a flow, render it: an ASCII diagram, a numbered step-by-step, a config snippet with the relevant fields highlighted. When the doc names a parameter, explain *why* that parameter exists and what breaks without it. **Every H2 averages ~400 words, every H3 has ≥1 full paragraph (3–5 sentences) of real substance** — never a heading followed by a single sentence stub. +8. **At least one runnable code block per major section that warrants one.** Use ` ``` ` fenced blocks with the language tag (`bash`, `ts`, `py`, `json`, `yaml`, etc.). Prefer real, copy-pasteable snippets pulled from the doc over hand-waved pseudocode. +9. **Edge cases get their own paragraph or callout.** "What if X is null." "What about rate limits." "How to debug when this fails." If the doc lists errors / caveats / limits, name at least 2 by the actual error string or limit value. +10. **Alternatives must name names.** "We considered X" — say what X is, link to it, and give the one specific reason it didn't fit. No "various other approaches" hedging. ## Per-destination overrides -### `blog-html` (900–1,100 words) -- Full markdown post per outline. -- `##` H2s, `###` H3s if needed. NEVER `#`. -- Code blocks: ` ``` ` fenced with language tag. +### `blog-html` (1,800–2,200 words) +- Full markdown post per outline. **Exactly 5 `##` H2s** — match the strategist's 5-section arc. Use `###` H3s INSIDE an H2 when it covers 2–4 sub-ideas (this is encouraged — it's how you fill the section). NEVER `#` (title is separate). NEVER add a 6th H2. +- Each H2 section is **~400 words on average** — substantial bodies, not paragraphs. Don't drop into a single sentence under a heading; if you don't have enough to say in a section, pull material from the doc to fill it. Empty sub-bodies under H3 sub-headings are the most common failure mode — every H3 needs at least 1 full paragraph (3–5 sentences) of substance, not just a sentence stub. +- Code blocks: ` ``` ` fenced with language tag (`bash`, `ts`, `py`, `json`, `yaml`). 2–6 blocks total. Each block load-bearing — no example-for-example's-sake. Surround each block with 1–3 paragraphs of narration explaining *why each line exists* and *what happens at runtime*. +- ASCII diagrams welcome for flows, request lifecycles, retry/queue topology. Wrap in a fenced ` ``` ` block (no language). +- The 5 H2s in order: + 1. **Hook + claim + TL;DR** — anomaly/contrarian/stat opening, the thesis, then a 3–5-bullet TL;DR of what the post proves. (~350 words) + 2. **Mechanism — how it works under the hood** — the actual moving parts, components, phases. 2–3 H3s by component. (~450 words) + 3. **Concrete usage, end-to-end** — 2+ code blocks with full narration. Walk through runtime behaviour. (~500 words) + 4. **Edge cases & alternatives** — actual error strings / limits from the doc, how to detect each, 1–2 alternatives by name. 2–3 H3s. (~450 words) + 5. **Wrap-up + CTA** — restate the thesis + next step, closing per `closingPattern`. (~250 words) ### `reddit` (~250 words body) - No `#` heading (Reddit titles are separate). diff --git a/lib/shared/schemas.ts b/lib/shared/schemas.ts index 0a9dd0c..680cf68 100644 --- a/lib/shared/schemas.ts +++ b/lib/shared/schemas.ts @@ -141,6 +141,7 @@ export const CitationSourceSchema = z.enum([ "twitter", "linkedin", "blog", + "docs", "perplexity", "hackernews", "other", @@ -212,7 +213,17 @@ export const BlogDraftSchema = z.object({ title: z.string(), slug: z.string(), excerpt: z.string(), - bodyMarkdown: z.string(), + // Accepts either a single markdown string OR an array of markdown sections + // that the Writer can emit when long-form output (~2000 words) would + // otherwise blow past the model's per-response max_tokens budget. The + // transform joins sections with a blank line so downstream consumers (DB, + // dashboard, GEO-Editor, Formatter) keep their `bodyMarkdown: string` + // contract unchanged. Background: the previous failure mode was + // "Unterminated string in JSON" mid-body when Kimi K2.6 hit its response + // budget while emitting a single ~13KB JSON-escaped string. + bodyMarkdown: z + .union([z.string(), z.array(z.string())]) + .transform((v) => (Array.isArray(v) ? v.join("\n\n") : v)), tags: z.array(z.string()).default([]), citations: z.array(SourceCitationSchema).default([]), geoNotes: z.array(z.string()).optional(), From 2c937b78cce481f51ed541bba0e70931e9b00b9d Mon Sep 17 00:00:00 2001 From: Sebastian Tsang Date: Sun, 10 May 2026 10:59:45 -0400 Subject: [PATCH 2/4] =?UTF-8?q?fix(runtime):=20bump=20writer=20timeout=203?= =?UTF-8?q?00s=E2=86=92600s=20+=20pin=20to=20Sonnet=204.6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Symptoms: 2,000-word blog runs ended in workflow state "done" with no actual draft produced. Root cause: writer hit `single-task invocation exceeded 300s` mid-generation, GEO-Editor + Formatter cascade-skipped, and pipeline-reporter + slack-digest still ran (triggerRule "all_done"), which made the run appear successful. Two changes: - SINGLE_TIMEOUT_MS: 300_000 → 600_000. Sonnet 4.6 on a deep-technical 2K-word draft routinely lands at 4–6 min wall-clock; 300s clipped legitimate completions. 600s gives long-form generation real room while still failing fast on a genuinely hung model. - Per-persona model pin (PERSONA_MODEL_PINS): writer now hard-pinned to "claude-sonnet-4-6" regardless of `getModelForTier(tier)` and regardless of `GMAESTRO_LLM_PROVIDER`. Flipping the global provider env to ollama no longer silently downgrades the writer to Kimi K2.6, which can't produce coherent 2K-word drafts. Caveat: the SDK still reads global ANTHROPIC_BASE_URL, so this pin assumes the user is on `GMAESTRO_LLM_PROVIDER=anthropic` (current setup). A future change to env.ts can preserve Anthropic credentials for pinned-model calls when the global provider is ollama. Co-Authored-By: Claude Opus 4.7 (1M context) --- lib/personas/runtime.ts | 47 ++++++++++++++++++++++++++++++++--------- 1 file changed, 37 insertions(+), 10 deletions(-) diff --git a/lib/personas/runtime.ts b/lib/personas/runtime.ts index ea1e8ae..1375ad2 100644 --- a/lib/personas/runtime.ts +++ b/lib/personas/runtime.ts @@ -91,7 +91,7 @@ export async function runPersona( query({ prompt: buildUserPrompt(personaId, parsedInput, founderObjective), options: { - model: getModelForTier(persona.modelTier), + model: resolveModelForPersona(personaId, persona.modelTier), systemPrompt: promptBody, mcpServers: { composio: mcpConfig }, allowedTools: getAllowedToolsForPersona(personaId), @@ -175,15 +175,42 @@ const BATCH_CHUNK_SIZE_ON_RETRY = 10; const BATCH_TIMEOUT_MS = 90_000; /** * Hard ceiling for single-task fanout personas. Bumped from 120s → 300s - * 2026-05-10: the content pivot's writer + geo-editor + formatter each - * generate / edit 1,800–2,200 word blog posts. Sonnet 4.6 routinely lands - * those at 90–150s for the writer alone, plus 60–120s for geo-editing and - * 30–90s for formatter. The previous 120s budget caused all three to - * silently time out under triggerRule: "all_done" so the workflow reported - * "done" with no draft produced. 300s gives long-form generation real - * room while still failing fast on a hung model. + * 2026-05-10 (content pivot), then 300s → 600s same day after a 2,000-word + * Composio Firecrawl blog timed out at exactly 300s on Sonnet 4.6 with + * GEO-Editor + Formatter cascade-skipping behind it. Sonnet 4.6 on a + * deep-technical 2K-word draft routinely lands at 4–6 min — 300s clipped + * legitimate completions and the workflow's `triggerRule: "all_done"` + * tail (pipeline-reporter, slack-digest) made the run report "done" with + * no actual draft produced. 600s gives long-form generation real room + * while still failing fast on a genuinely hung model. */ -const SINGLE_TIMEOUT_MS = 300_000; +const SINGLE_TIMEOUT_MS = 600_000; + +/** + * Per-persona hard model pins. When a persona is in this map, we use the + * pinned model regardless of `getModelForTier(persona.modelTier)` AND + * regardless of `GMAESTRO_LLM_PROVIDER`. Used today to keep the writer on + * Anthropic Sonnet 4.6 (the only model that produces coherent 2,000-word + * drafts at this latency budget), so flipping the global provider env + * doesn't silently downgrade content quality. + * + * Caveat: the SDK still reads the global ANTHROPIC_BASE_URL / auth env vars, + * so if the user is on `GMAESTRO_LLM_PROVIDER=ollama` (which force-rewrites + * those vars), this pin will route the model name through the Ollama + * endpoint — which 404s. Either run with `GMAESTRO_LLM_PROVIDER=anthropic` + * (current setup) or extend env.ts to preserve the Anthropic credentials + * for pinned-model calls. + */ +const PERSONA_MODEL_PINS: Partial> = { + writer: "claude-sonnet-4-6", +}; + +function resolveModelForPersona( + personaId: PersonaId, + tier: Parameters[0], +): string { + return PERSONA_MODEL_PINS[personaId] ?? getModelForTier(tier); +} /** * Run a persona in BATCH mode: one LLM call processes all items at once. @@ -331,7 +358,7 @@ async function runBatchAttempt( query({ prompt: userPrompt, options: { - model: getModelForTier(persona.modelTier), + model: resolveModelForPersona(personaId, persona.modelTier), systemPrompt: promptBody, mcpServers: { composio: mcpConfig }, allowedTools: getAllowedToolsForPersona(personaId), From 4c7e51892ce5dd3d8cc56e2a4a500fb2abee7121 Mon Sep 17 00:00:00 2001 From: Sebastian Tsang Date: Sun, 10 May 2026 12:01:58 -0400 Subject: [PATCH 3/4] feat(dispatch): add direct GitHub-PR provider for BlogDraft MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The BlogDraft approval card's "Send via" picker only listed Reddit, because PROVIDERS_BY_ARTIFACT.BlogDraft only had a reddit entry. An internal blog post intended for the company's own static-site repo had no in-app path to dispatch. Adds a github entry as the FIRST provider in BlogDraft (so it wins the auto-default when both reddit and github are connected). It mirrors the existing ChannelVariant.github 2-step flow (commit file → open PR), pulling content directly from the BlogDraft's bodyMarkdown / title / slug. Repo + branch + path defaults target the Anvil marketing site; override per-draft via proposed_action.metadata.{repo, branch, path, prTitle, prBody}. Verified: approvals page picker now shows "GitHub PR" + "Reddit", GitHub is auto-selected, approve button reads "Approve & send via GitHub PR". Note: this conflicts with a WIP variant of providers.ts (a different worktree session had set BlogDraft: [] and routed via Formatter+ChannelVariant). Coordinate with that session before merging if both designs land at the same time. Co-Authored-By: Claude Opus 4.7 (1M context) --- lib/dispatch/providers.ts | 81 +++++++++++++++++++++++++++++++++++---- 1 file changed, 73 insertions(+), 8 deletions(-) diff --git a/lib/dispatch/providers.ts b/lib/dispatch/providers.ts index d1f6614..8daffb9 100644 --- a/lib/dispatch/providers.ts +++ b/lib/dispatch/providers.ts @@ -45,15 +45,80 @@ function asObject(v: unknown): Record { */ export const PROVIDERS_BY_ARTIFACT: Record = { /** - * BlogDraft approvals carry `targets: ToolkitId[]` set by the founder via - * the channels picker. The dispatcher fans out one publish per target by - * looking up the matching ChannelVariant entry below — there's no single - * "BlogDraft provider" call. We expose all 7 target toolkits here so the - * approval card can render the channels picker; the dispatcher itself - * doesn't invoke these directly for BlogDraft (it routes through Formatter - * + ChannelVariant approvals). + * BlogDraft approvals — direct one-click Reddit publish. Skips the + * Formatter/ChannelVariant fanout: the founder reviews the draft once and + * "Approve & send via Reddit" posts the markdown body verbatim to the + * configured subreddit. Default destination is the founder's profile + * subreddit (`u_`); override per-draft by setting + * `proposed.subreddit` upstream. */ - BlogDraft: [], + BlogDraft: [ + /** + * GitHub PR — opens a pull request against a static-site repo with the + * draft's markdown body as a new content file. The dispatcher recognizes + * this provider and runs a 2-step flow (commit then PR) using the same + * choreography as the ChannelVariant.github path. Default repo + + * frontmatter target the Anvil marketing site; override per-draft via + * `proposed_action.metadata.{repo, branch, path, prTitle, prBody}`. + * + * Listed FIRST so an "internal" BlogDraft with `targets: ["github"]` + * auto-picks GitHub as the default destination — Reddit ranks lower + * because it's a public publish. + */ + { + toolkit: "github", + action: "GITHUB_CREATE_PULL_REQUEST", + label: "GitHub PR", + buildArgs: (p) => { + const metadata = asObject(p.metadata); + const repo = asString(metadata.repo) ?? "anvil-co/anvil-site"; + const [owner, repoName] = repo.split("/"); + const slug = asString(p.slug) ?? "post"; + const title = asString(p.title) ?? "(untitled)"; + const body = + asString(p.bodyMarkdown) ?? + asString(p.content) ?? + asString(p.excerpt) ?? + ""; + return { + owner, + repo: repoName, + title: asString(metadata.prTitle) ?? `Add post: ${title}`, + head: asString(metadata.branch) ?? `content/${slug}`, + base: "main", + body: + asString(metadata.prBody) ?? + `Adds new blog post: **${title}**\n\n_Drafted by GMaestro and approved by the founder._`, + // The dispatcher's pre-step calls GITHUB_COMMIT_MULTIPLE_FILES with + // these values to land the markdown file before opening the PR. + _commitFile: { + path: asString(metadata.path) ?? `content/blog/${slug}.md`, + content: body, + }, + }; + }, + }, + { + toolkit: "reddit", + action: "REDDIT_CREATE_REDDIT_POST", + label: "Reddit", + buildArgs: (p) => { + const subreddit = asString(p.subreddit) ?? "u_Pale-Taste2766"; + const title = asString(p.title) ?? "(untitled)"; + const body = + asString(p.bodyMarkdown) ?? + asString(p.content) ?? + asString(p.excerpt) ?? + ""; + return { + subreddit, + kind: "self", + title, + text: body, + }; + }, + }, + ], /** * ChannelVariant — one provider per target. The dispatcher reads the From d09cc6b91f0793aa6640510645a5e9a2ea0197ab Mon Sep 17 00:00:00 2001 From: Sebastian Tsang Date: Sun, 10 May 2026 12:09:44 -0400 Subject: [PATCH 4/4] fix(dag-view): finalize stuck-running nodes on workflow_done MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Conductor and Manager nodes were stuck rendering "RUNNING" forever on workflows that had clearly terminated, because aggregate() returns "running" if ANY child is still "running" — and at least one child specialist routinely lacks a persona_completed event: 1. The persona failed at exec/parse and the workflow function recorded the failure to workflow_nodes but never emitted a persona_completed event to the activity bus (e.g. writer timing out at 600s). 2. Skip-cascade dropped the started→completed pair for a mid-chain node whose upstream errored. Adds a sweep to deriveNodeStatuses: when workflow_done arrives, flip any remaining "running" nodes to "done". The detailed per-node error state is still surfaced via workflow_nodes / the popover; the DAG overview just needed to reflect "this workflow is finished." Verified on two real runs: - 99eb6c9a (Reddit thread): conductor + content-mgr were "running" because geo-editor never finished. Now: all "done" except the correctly-skipped Formatter. - f4276704 (deep blog): conductor + content-mgr were "running" because writer timed out at 600s. Now: all "done" except the correctly-skipped GEO-Editor + Formatter. Co-Authored-By: Claude Opus 4.7 (1M context) --- lib/ui/components/dag-view.tsx | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/lib/ui/components/dag-view.tsx b/lib/ui/components/dag-view.tsx index e540f2e..0216611 100644 --- a/lib/ui/components/dag-view.tsx +++ b/lib/ui/components/dag-view.tsx @@ -180,6 +180,21 @@ function deriveNodeStatuses(events: WireEvent[]): Map { break; } case "workflow_done": + // Finalize any nodes still showing "running" once the workflow has + // terminated. Two known causes leave a stuck-running node behind: + // 1. The persona failed at exec/parse and the workflow function + // recorded the failure in workflow_nodes but never emitted a + // persona_completed event to the activity bus. + // 2. Skip-cascade dropped the started→completed pair for a + // mid-chain node whose upstream errored. + // In both cases the workflow is over — leaving the node painted + // "running" forever is misleading. We bias to "done" since the + // detailed per-node error (when there is one) is already surfaced + // via the workflow_nodes status badge inside the popover; the DAG + // overview just needs to reflect "this workflow is finished." + for (const [id, s] of statuses) { + if (s === "running") apply(id, "done"); + } break; } }