diff --git a/evals/README.md b/evals/README.md index cb85a52..42644ea 100644 --- a/evals/README.md +++ b/evals/README.md @@ -15,7 +15,7 @@ fixture and produces a scored JSONL record with the full S0–S11 stage funnel. | `evals/runner/` | Runner + scorer (`run.ts` CLI, driver interfaces, stage gates) | | `evals/runner/drivers/` | Driver implementations — Tier-0 local/static driver; authoring contract in `drivers/README.md` | | `evals/scenarios/` | Scenario definitions (`tier1-directory.json`, `tier1-directory-guide-only.json`, `tier1-directory-full.json`, `pre1-directory-proceed.json`, `pre1-noiam-park.json`) | -| `evals/skills-bundle/` | Skill-bundle mount point (v0.3.0 manifest — seven skills in `skills/`) | +| `evals/skills-bundle/` | Skill-bundle mount point (v0.4.0 manifest — ten skills in `skills/`) | | `evals/results/` | JSONL run records (gitignored; `.gitkeep` committed) | ## How to run @@ -242,7 +242,6 @@ real-tenant driver and the tool surface are available. ## Non-goals -- The remaining three skills (verify-connector-output, update-and-rollback, diagnose-authoring-failure) — later PRs. - Tier-2 real sandbox providers and the qualitative LLM-judge tier. - Operator-side activation E2E leg (redeeming the approval token) — those two fields are `skipped_human_boundary`. diff --git a/evals/runner/agent.test.ts b/evals/runner/agent.test.ts index 3665551..7fceeb8 100644 --- a/evals/runner/agent.test.ts +++ b/evals/runner/agent.test.ts @@ -78,7 +78,7 @@ test("buildPrompt pre1 branch names the pre1Path, both skills, and the output co basicAuth: {username: "connector@example.com", password: "fixture-token"}, bearerToken: "fixture-token", }, - skillBundle: {mode: "full", version: "0.3.0"}, + skillBundle: {mode: "full", version: "0.4.0"}, model: "together/deepseek-ai/DeepSeek-V4-Flash-0731", reasoningEffort: "high", kind: "pre1", diff --git a/evals/runner/scenario.test.ts b/evals/runner/scenario.test.ts index 66d14c9..c5f741e 100644 --- a/evals/runner/scenario.test.ts +++ b/evals/runner/scenario.test.ts @@ -100,7 +100,7 @@ test("loadScenario loads the pre1-directory-proceed scenario (kind pre1, proceed assert.equal(s.seed, undefined) assert.equal(s.requiredSourceFiles, undefined) assert.equal(s.skillBundle.mode, "full") - assert.equal(s.skillBundle.version, "0.3.0") + assert.equal(s.skillBundle.version, "0.4.0") }) test("loadScenario loads the pre1-noiam-park scenario (kind pre1, park)", () => { diff --git a/evals/runner/skills_bundle.test.ts b/evals/runner/skills_bundle.test.ts index 735bb16..02a3093 100644 --- a/evals/runner/skills_bundle.test.ts +++ b/evals/runner/skills_bundle.test.ts @@ -14,8 +14,8 @@ import {loadScenario} from "./scenario.ts" const execFileAsync = promisify(execFile) const RUN = "evals/runner/run.ts" const BUNDLE = "evals/skills-bundle/bundle.json" -const SKILLS = ["author-in-app-connector", "read-authoring-contract", "write-connector-source", "build-and-test", "deploy-and-activate", "design-access-model", "source-openapi-spec"] -const VERSION = "0.3.0" +const SKILLS = ["author-in-app-connector", "read-authoring-contract", "write-connector-source", "build-and-test", "deploy-and-activate", "design-access-model", "source-openapi-spec", "verify-connector-output", "update-and-rollback", "diagnose-authoring-failure"] +const VERSION = "0.4.0" function readBundle(): {version: string; skills: {name: string; version: string; path: string}[]} { return JSON.parse(readFileSync(BUNDLE, "utf8")) as {version: string; skills: {name: string; version: string; path: string}[]} @@ -56,7 +56,7 @@ test("(a) each SKILL.md exists with the locked frontmatter contract", () => { test("(b) every bundle.json path resolves to an existing file", () => { const bundle = readBundle() assert.equal(bundle.version, VERSION) - assert.equal(bundle.skills.length, 7) + assert.equal(bundle.skills.length, 10) assert.deepEqual(bundle.skills.map((s) => s.name), SKILLS) const skillsRoot = resolve("skills") for (const skill of bundle.skills) { @@ -133,6 +133,52 @@ const SKILL_LITERALS: Record = { "Parking with evidence", "authority ladder", ], + "verify-connector-output": [ + "never invent data to make a demo appear complete", + "SYNC_STATUS_DONE", + "SYNC_STATUS_RUNNING", + "terminal status after ~10 polls, STOP and report", + "data-anomaly auto-pause", + "significant drop in sync data", + "sync_disabled", + "every grant principal references an emitted resource", + "entitlement ID references an emitted entitlement", + "ID stability", + "/admin/connector/", + ], + "update-and-rollback": [ + "serve image does not match the revision-pinned runtime image", + "/api/v1/connector-authoring/rollbacks", + "target_revision_id", + "approval_token_id", + "activation_epoch", + "image digest", + "Do not redeem the approval token", + "Do not clear runtime fields, call the provisioner directly, or mutate the", + "evidence is unsatisfied", + "if no ACTIVE row after ~10 polls, STOP and report", + "no terminal status after ~10 polls, STOP and report", + "SYNC_STATUS_ERROR", + "SYNC_STATUS_DISABLED", + "Do not print, log, or otherwise expose the OWNER bearer token value", + ], + "diagnose-authoring-failure": [ + "262144 byte compile limit", + "1048576 byte limit", + "is_secret", + "credential re-entry required", + "missing type", + "unregistered transport", + "ticketing.enabled must be true when ticketing is configured", + "activation evidence is unsatisfied", + "Invalid token provided", + "ConnectionOK", + "HostCallOK", + "Poll with backoff", + "row after ~10 polls, stop and report", + "c1_connector_service_get", + "status.lastError", + ], } test("(c) each SKILL.md carries the locked section markers, content literals, ASCII-only bodies, and stays <= 200 lines", () => { @@ -152,7 +198,7 @@ test("(c) each SKILL.md carries the locked section markers, content literals, AS test("(d) the full-mode scenario parses with mode full and the two pinned scenarios keep their locked modes", () => { const full = loadScenario("evals/scenarios/tier1-directory-full.json") assert.equal(full.skillBundle.mode, "full") - assert.equal(full.skillBundle.version, "0.3.0") + assert.equal(full.skillBundle.version, "0.4.0") assert.equal(full.id, "tier1-directory-full") const none = loadScenario("evals/scenarios/tier1-directory.json") assert.equal(none.skillBundle.mode, "none") @@ -201,8 +247,25 @@ test("(e) CLI end-to-end: full-mode Tier-0 run exits 0 and the record meta carri const lines = readFileSync(join(dir, records[0]), "utf8").trim().split("\n") const meta = JSON.parse(lines[0]) as Record assert.equal(meta.skill_bundle_mode, "full") - assert.equal(meta.skill_bundle_version, "0.3.0") + assert.equal(meta.skill_bundle_version, "0.4.0") } finally { rmSync(dir, {recursive: true, force: true}) } }) + +test("(f) every skill ships a non-empty SOURCES.md naming its pinned sources", () => { + const skillsRoot = resolve("skills") + for (const name of SKILLS) { + const sourcesPath = join(skillsRoot, name, "SOURCES.md") + assert.ok(existsSync(sourcesPath), `missing SOURCES.md: ${name}`) + const content = readFileSync(sourcesPath, "utf8") + assert.ok(content.length > 0, `empty SOURCES.md: ${name}`) + } + // The three new skills must name the c1 pin (decision 5: nothing written + // from model memory) so a dropped or truncated pin fails the gate. + const c1Pin = "2e5f53eb441a93087d9754085ca17a5061e125ea" + for (const name of ["verify-connector-output", "update-and-rollback", "diagnose-authoring-failure"]) { + const content = readFileSync(join(skillsRoot, name, "SOURCES.md"), "utf8") + assert.ok(content.includes(c1Pin), `${name}: SOURCES.md does not name the c1 pin`) + } +}) diff --git a/evals/scenarios/pre1-directory-proceed.json b/evals/scenarios/pre1-directory-proceed.json index 52fa477..e7e522a 100644 --- a/evals/scenarios/pre1-directory-proceed.json +++ b/evals/scenarios/pre1-directory-proceed.json @@ -29,7 +29,7 @@ }, "skillBundle": { "mode": "full", - "version": "0.3.0" + "version": "0.4.0" }, "model": "together/deepseek-ai/DeepSeek-V4-Flash-0731", "reasoningEffort": "high" diff --git a/evals/scenarios/pre1-noiam-park.json b/evals/scenarios/pre1-noiam-park.json index c90bada..2bc6f20 100644 --- a/evals/scenarios/pre1-noiam-park.json +++ b/evals/scenarios/pre1-noiam-park.json @@ -27,7 +27,7 @@ }, "skillBundle": { "mode": "full", - "version": "0.3.0" + "version": "0.4.0" }, "model": "together/deepseek-ai/DeepSeek-V4-Flash-0731", "reasoningEffort": "high" diff --git a/evals/scenarios/tier1-directory-full.json b/evals/scenarios/tier1-directory-full.json index f5f111e..5b4b6a2 100644 --- a/evals/scenarios/tier1-directory-full.json +++ b/evals/scenarios/tier1-directory-full.json @@ -27,7 +27,7 @@ }, "skillBundle": { "mode": "full", - "version": "0.3.0" + "version": "0.4.0" }, "model": "together/deepseek-ai/DeepSeek-V4-Flash-0731", "reasoningEffort": "high", diff --git a/evals/skills-bundle/README.md b/evals/skills-bundle/README.md index 0fe1f6b..83a9f1d 100644 --- a/evals/skills-bundle/README.md +++ b/evals/skills-bundle/README.md @@ -2,6 +2,24 @@ This directory is the skill-bundle mount point for the eval harness. +## v0.4.0 - the ten skills + +The bundle ships ten skills: the seven prior skills plus three post-funnel +and cross-cutting skills at `0.1.0`: + +- `verify-connector-output` - post-11: post-sync verification (counts, + grant wiring, ID stability across re-sync, UI spot-check). +- `update-and-rollback` - post-activation: same-catalog update flow with + the image-digest reuse rule, the rotation STOP/escalate limitation, and + REST-only OWNER-gated rollback. +- `diagnose-authoring-failure` - cross-cutting: symptom -> cause -> fix + router over the common-failures table, the draft-test FAIL reading, and + where logs live. + +The seven prior skills are unchanged; the bundle version is `0.4.0`. The +three new skills pin the c1 contract sources at `2e5f53eb…` (see each +skill's SOURCES.md). + ## v0.3.0 — the seven skills The bundle ships seven skills: the five funnel skills plus two new pre-1 @@ -43,4 +61,4 @@ names a skill and a `path` relative to this directory pointing into `skills/` (the canonical home). Private drivers mount per the manifest — the manifest indirection is the contract, not a directory copy. The scenario file selects the bundle via `skillBundle.mode` (`full`) and pins -`skillBundle.version` (`0.2.0`); the runner records both in every run record. +`skillBundle.version` (`0.4.0`); the runner records both in every run record. diff --git a/evals/skills-bundle/bundle.json b/evals/skills-bundle/bundle.json index 1815d6b..b312544 100644 --- a/evals/skills-bundle/bundle.json +++ b/evals/skills-bundle/bundle.json @@ -1,5 +1,5 @@ { - "version": "0.3.0", + "version": "0.4.0", "skills": [ {"name": "author-in-app-connector", "version": "0.2.1", "path": "../../skills/author-in-app-connector/SKILL.md"}, {"name": "read-authoring-contract", "version": "0.2.0", "path": "../../skills/read-authoring-contract/SKILL.md"}, @@ -7,6 +7,9 @@ {"name": "build-and-test", "version": "0.2.0", "path": "../../skills/build-and-test/SKILL.md"}, {"name": "deploy-and-activate", "version": "0.2.0", "path": "../../skills/deploy-and-activate/SKILL.md"}, {"name": "design-access-model", "version": "0.1.0", "path": "../../skills/design-access-model/SKILL.md"}, - {"name": "source-openapi-spec", "version": "0.1.0", "path": "../../skills/source-openapi-spec/SKILL.md"} + {"name": "source-openapi-spec", "version": "0.1.0", "path": "../../skills/source-openapi-spec/SKILL.md"}, + {"name": "verify-connector-output", "version": "0.1.0", "path": "../../skills/verify-connector-output/SKILL.md"}, + {"name": "update-and-rollback", "version": "0.1.0", "path": "../../skills/update-and-rollback/SKILL.md"}, + {"name": "diagnose-authoring-failure", "version": "0.1.0", "path": "../../skills/diagnose-authoring-failure/SKILL.md"} ] } diff --git a/skills/README.md b/skills/README.md index e58368c..aca394a 100644 --- a/skills/README.md +++ b/skills/README.md @@ -1,8 +1,12 @@ # Agent skills -The seven skills shipped in this batch, authored against -the v0.0.26 DSL contract and the 23-tool tenant MCP surface. Each skill's -`SOURCES.md` names the pinned sources with their SHAs. +The ten skills shipped in this batch: the five funnel skills and two +pre-1 skills authored against the v0.0.26 DSL contract and the 23-tool +tenant MCP surface, plus three post-funnel and cross-cutting skills +(`verify-connector-output`, `update-and-rollback`, +`diagnose-authoring-failure`) over the tenant connector and authoring +tool surface. Each skill's `SOURCES.md` names the pinned sources with +their SHAs. | Skill | Stage coverage | Source basis | |---|---|---| @@ -13,6 +17,9 @@ the v0.0.26 DSL contract and the 23-tool tenant MCP surface. Each skill's | `deploy-and-activate` | Stages 6–8, 11 — app, provision, configure, deploy, mint, handoff | MCP-served guide, `authoring.proto`, lifecycle doc | | `design-access-model` | Pre-1 — access-model design for net-new providers | baton-admin `design-baton-access-model` @ `6fe6886f…` | | `source-openapi-spec` | Pre-1 — OpenAPI spec sourcing + IAM go/no-go | claude-marketplace `source-openapi-spec` @ `0cc5ac2a…` | +| `verify-connector-output` | Post-11 - post-sync verification (counts, grant wiring, ID stability, UI spot-check) | MCP-served guide, lifecycle doc, `probe-contracts.md` @ `0cc5ac2a…` | +| `update-and-rollback` | Post-activation - same-catalog update + REST rollback | MCP-served guide, `authoring.proto`, lifecycle doc | +| `diagnose-authoring-failure` | Cross-cutting - symptom -> cause -> fix router | lifecycle doc, baton-admin `diagnose-connector-failure` @ `6fe6886f…` | The eval bundle (`evals/skills-bundle/bundle.json`) is a manifest pointing into this directory; the skill bodies live here as the single source of diff --git a/skills/diagnose-authoring-failure/SKILL.md b/skills/diagnose-authoring-failure/SKILL.md new file mode 100644 index 0000000..5f44bee --- /dev/null +++ b/skills/diagnose-authoring-failure/SKILL.md @@ -0,0 +1,91 @@ +--- +name: diagnose-authoring-failure +description: Use when a build, draft test, activation, or production sync fails and you need the symptom-to-cause-to-fix route. Do not use when the connector is healthy and you are verifying sync output - use verify-connector-output; do not use when updating or rolling back a healthy live connector - use update-and-rollback. +version: 0.1.0 +--- + +# diagnose-authoring-failure + +Symptom -> cause -> fix router for authored-connector failures. Built on the +lifecycle doc's common-failures table, the draft-test evidence reading, and +where logs live. Ports the taxonomy approach of baton-admin +`diagnose-connector-failure`; the baton-admin skill's repo-local CLI +tooling is replaced by the tenant MCP tools named below. + +## Workflow + +1. Start from the symptom: the exact error text, the failing step (build, + draft test, activation, production sync), and the IDs at hand. +2. Route the symptom through the table below; collect the smallest evidence + needed for that cause. +3. Apply the fix, then re-run the smallest verification step that failed. +4. Output a chat diagnostic summary: failure summary, classified cause, + evidence used, likely owner (source, schema, runtime, credentials, or + platform), one concrete next fix, and the smallest verification command. + +## Symptom -> cause -> fix + +| Symptom | Cause / fix | +|---------|-------------| +| Build rejected: `connector source exceeds the 262144 byte compile limit` | The esbuild-bundled source is over 256 KiB. Trim the source set; large OpenAPI-derived spec assets are the usual weight. | +| Build rejected: `connector bundle with embedded runtime specs is bytes, above the 1048576 byte limit` | The bundled `connector.js` plus its embedded runtime specs is over 1 MiB - a separate cap from the 256 KiB source limit, hit after bundling. Trim the source or the generated spec assets it embeds. | +| Build rejected: `credential-class config field must be marked is_secret` | A token/secret/password-named field in `runtime-schema.json` lacks `is_secret: true`. The `secret:` spelling is not read. | +| Draft test: `credential re-entry required: ` | Configure the instance credentials before the draft test. | +| Draft test: `connector config field X is missing type` | Add `"type": "string"` (etc.) to the field in `runtime.config_schema`. | +| Sync error mentions an `unregistered transport` | A `node` or `reuse` references a transport missing from `connector({ transports: ... })`. Register that same transport object, rebuild, and retest. | +| Draft test: `ticketing.enabled must be true when ticketing is configured` | The runtime-schema carries a `ticketing` block the connector code does not back. Remove the block (and any `actions` / `policy_surface` entries referencing dropped code); `enabled: false` is not a valid off-switch. | +| Activation reports `activation evidence is unsatisfied` | No PASS test-sync evidence binds this revision. Run the draft test (with credentials set) and confirm it passed before asking the OWNER to review a fresh approval URL. | +| Production sync `Invalid token provided` | The API token is wrong or truncated. Re-configure with the full token, then re-run. | + +## Draft-test FAIL reading + +The evidence row is authoritative: poll +`c1_connector_authoring_get_test_run_evidence` and read the `result` (PASS +or FAIL) and the `error` field. Poll with backoff (e.g. every 5-10s); if no +row after ~10 polls, stop and report `NotFound`/pending. The FAIL reason +lives on the evidence row; it is not always logged when the read activity +succeeds but the outcome evaluation returns FAIL. PASS requires all of: + +- `ConnectionOK` - Validate succeeded +- `HostCallOK` - GetMetadata succeeded +- No read error and no write attempt +- The config version handle matches the candidate revision +- The runtime image digest matches the revision-pinned image + +A FAIL row (or no row yet) means activation stays `evidence is unsatisfied` +until a new draft test writes PASS. + +## Where logs live + +Use the product's connector activity and sync logs - the same surface you +use to view any connector in your tenant. There is no separate +operator-only log surface. The connector's status row +(`c1_connector_service_get` -> `status.status`, `status.lastError`) is the +authoritative outcome for a sync. + +## Exit criteria + +- Every one of the nine common-failures rows routes to its documented fix. +- A draft-test FAIL is read from the evidence row, not from completion. +- The logs location and the status row are named. +- The body contains the literals `262144 byte compile limit`, + `1048576 byte limit`, `is_secret`, `credential re-entry required`, + `missing type`, `unregistered transport`, + `ticketing.enabled must be true when ticketing is configured`, + `activation evidence is unsatisfied`, `Invalid token provided`, + `ConnectionOK`, `HostCallOK`, `c1_connector_service_get`, and + `status.lastError`. + +## Anti-patterns + +- Do not start with broad full-suite runs when a small probe can isolate + the issue. +- Do not hide unresolved drift with waivers or mock shaping. +- Do not expose auth material, tokens, or customer data in diagnostics. +- Do not guess a fix without reading the evidence row first. + +## Blocker protocol + +If the same validation or runtime error remains unchanged after 2 failed +fix cycles on the same error, stop and report the exact error text instead +of guessing further. diff --git a/skills/diagnose-authoring-failure/SOURCES.md b/skills/diagnose-authoring-failure/SOURCES.md new file mode 100644 index 0000000..8871c61 --- /dev/null +++ b/skills/diagnose-authoring-failure/SOURCES.md @@ -0,0 +1,13 @@ +# Sources - diagnose-authoring-failure + +Authored against the pinned sources below (decision 5: nothing written from +model memory). Source-of-truth precedence: (a) MCP-served guide, (b) +`authoring.proto`, (c) lifecycle doc, (e) baton-admin +`diagnose-connector-failure`. + +| Source | Pin / SHA | What this skill quotes | +|---|---|---| +| MCP-served guide | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The Caps table (262144-byte compile limit, 1048576-byte bundle limit) and the Evidence and credentials contract (`credential re-entry required`; activation fails closed until a PASS evidence row binds the revision digests). | +| Authoring proto | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | `GetTestRunEvidence` (poll `(catalog_id, revision_id, test_run_id)`; `result` PASS/FAIL + `error` on the evidence row). | +| Lifecycle doc | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The Debugging section: the common-failures table (all nine rows), the draft-test FAIL reading (evidence row authoritative; PASS requires `ConnectionOK`, `HostCallOK`, no read error and no write attempt, config version handle match, runtime image digest match), and where logs live (product connector activity and sync logs; the status row `c1_connector_service_get` -> `status.status`, `status.lastError`). | +| baton-admin `diagnose-connector-failure` | `6fe6886f607ed0d2e48a616c30e7ce4bffc32489` | The taxonomy approach: start from the symptom, classify the failure surface, collect the smallest evidence, keep the diagnosis focused on symptom/evidence/owner/next-fix/rerun-target, and the output contract (failure summary, classified surface, evidence used, likely owner, one concrete next fix, smallest verification command). | diff --git a/skills/update-and-rollback/SKILL.md b/skills/update-and-rollback/SKILL.md new file mode 100644 index 0000000..8d713cc --- /dev/null +++ b/skills/update-and-rollback/SKILL.md @@ -0,0 +1,117 @@ +--- +name: update-and-rollback +description: Use when shipping a change to an activated connector (same-catalog rerun) or rolling back to a previously activated revision. Do not use when verifying a healthy connector's sync output - use verify-connector-output; do not use when diagnosing a failed build, draft test, or sync - use diagnose-authoring-failure. +version: 0.1.0 +--- + +# update-and-rollback + +Update and rollback for an activated authored connector. Runs in a +post-activation session - never during the funnel run. Tool names below are +the exact tenant MCP titles. + +## Update flow (same-catalog rerun) + +1. Reuse the existing `catalog_id` and `draft_id`; update the draft source + (lifecycle step 2) and build a new revision (steps 4-5). +2. Reuse the existing managed runtime instance ONLY when its image digest + matches the target revision's pinned runtime image digest. GATE: image + digest match. STOP if the digests differ - see the rotation limitation + below. +3. Pass a fresh draft test with the instance credentials (steps 9-10). + GATE: durable PASS evidence binds the new revision. +4. Mint a new approval URL with `c1_connector_authoring_mint_approval_token` + (`expires_in_seconds` 1-14400). GATE: non-empty `activation_url`. +5. HARD STOP at the human boundary: present the URL to a human tenant OWNER + and stop. Do not redeem the approval token for activation. +6. After the OWNER activates, poll + `c1_connector_authoring_list_revision_summaries` until the target + revision is `REVISION_STATUS_ACTIVE`; record its `activation_epoch`. + GATE: ACTIVE. STOP if not ACTIVE - if approval reports `evidence is unsatisfied`, + return to the draft-test step and confirm a fresh PASS row binds this + revision before minting a new approval URL. Poll with backoff (e.g. every + 5-10s); if no ACTIVE row after ~10 polls, STOP and report. +7. Call `c1_connector_service_force_sync`; poll `c1_connector_service_get` + with backoff (e.g. every 5-10s) until `status.status` is + `SYNC_STATUS_DONE`, `SYNC_STATUS_ERROR`, or `SYNC_STATUS_DISABLED`; if + no terminal status after ~10 polls, STOP and report. If the status row + reports `SYNC_STATUS_ERROR`, read `status.lastError` and route through + diagnose-authoring-failure; if `SYNC_STATUS_DISABLED`, read the + connector's `sync_disabled_reason` and route through + diagnose-authoring-failure unless the reason indicates a deliberate + pause (customer opt-out or ops). + +## Rotation STOP (known limitation) + +An active update can fail with `serve image does not match the revision-pinned runtime image`. The deployed instance's image digest must +match the target revision's pinned runtime image digest. There is no +supported customer, MCP, or Support Dashboard recovery today: deploy and +teardown are pre-activation-only, and rollback enforces the same image +binding. + +**STOP.** Do not clear runtime fields, call the provisioner directly, or +mutate the deployment or AWS resources. Record the tenant, catalog, app, +connector, and target revision IDs, then escalate to Connector Authoring / +managed-runtime engineering. + +## Rollback (REST-only, OWNER-gated) + +To roll back to a previously activated revision, mint an approval token for +the rollback target revision and redeem it against the rollback endpoint - +unlike activation, rollback is REST-only and OWNER-gated. The agent executes +the POST with an OWNER bearer token read from the environment or secret +store - never ask a human to paste the token into chat. Use the product +base URL and OWNER bearer token: + +``` +POST /api/v1/connector-authoring/rollbacks +{ + "catalog_id": "", + "target_revision_id": "", + "instance_app_id": "", + "instance_connector_id": "", + "approval_token_id": "" +} +``` + +The rollback re-points the published and instance pointers at the target +revision under a strictly greater activation epoch. The rolled-back-from +revision's serve state is untouched - the pointer move alone stops it +serving. + +If the rollback call fails closed (e.g. a precondition error), read the +error text and re-check the request fields; if it does not resolve after 2 +fix cycles, stop and report the exact error text (see Blocker protocol). + +## Exit criteria + +- Same-catalog rerun: new revision ACTIVE with a recorded `activation_epoch`, + fresh PASS evidence, and a completed force sync. +- The rotation STOP fired verbatim (`serve image does not match the revision-pinned runtime image`) with the escalation record: tenant, + catalog, app, connector, and target revision IDs. +- Rollback: `POST /api/v1/connector-authoring/rollbacks` accepted with + `target_revision_id`, `instance_app_id`, `instance_connector_id`, and + `approval_token_id`; the target revision is ACTIVE under a strictly + greater activation epoch. +- The body contains the literals `serve image does not match the revision-pinned runtime image`, `/api/v1/connector-authoring/rollbacks`, + `target_revision_id`, `approval_token_id`, `activation_epoch`, and + `image digest`. + +## Anti-patterns + +- Do not redeem the approval token for activation - activation is a human OWNER step. +- Do not reuse the managed runtime instance when the image digest does not + match the target revision's pinned runtime image digest. +- Do not clear runtime fields, call the provisioner directly, or mutate the + deployment or AWS resources on the rotation STOP. +- Do not roll back via the activation endpoint - rollback is REST-only. +- Do not force-sync before the target revision is ACTIVE. +- Do not print, log, or otherwise expose the OWNER bearer token value - + reference it only by variable or placeholder in any command or + diagnostic output. + +## Blocker protocol + +If the same validation or runtime error remains unchanged after 2 failed +fix cycles on the same error, stop and report the exact error text instead +of guessing further. diff --git a/skills/update-and-rollback/SOURCES.md b/skills/update-and-rollback/SOURCES.md new file mode 100644 index 0000000..24a4c42 --- /dev/null +++ b/skills/update-and-rollback/SOURCES.md @@ -0,0 +1,12 @@ +# Sources - update-and-rollback + +Authored against the pinned sources below (decision 5: nothing written from +model memory). Source-of-truth precedence: (a) MCP-served guide, (b) +`authoring.proto`, (c) lifecycle doc. + +| Source | Pin / SHA | What this skill quotes | +|---|---|---| +| MCP-served guide | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The "Updating an active connector" contract: reuse the existing managed runtime instance, update the draft source, build a new revision, fresh PASS evidence, mint a new approval URL, human OWNER activates, poll `list_revision_summaries` until ACTIVE, record `activation_epoch`, force sync. | +| Authoring proto | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | `MintApprovalToken` (`expires_in_seconds` 1-14400, `activation_url`); `ListRevisionSummaries` + `RevisionStatus` enum (`REVISION_STATUS_ACTIVE` = 1, `activation_epoch` on the ACTIVE row). The proto exposes no rollback RPC - its service comment notes rollback is the REST-only endpoint; the rollback contract lives in the lifecycle doc row below. | +| Lifecycle doc | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The "Updating and rolling back a live connector" section: the image-digest reuse rule; the rotation STOP verbatim (`serve image does not match the revision-pinned runtime image`; do not clear runtime fields, call the provisioner directly, or mutate the deployment or AWS resources; record tenant/catalog/app/connector/target-revision IDs; escalate to Connector Authoring / managed-runtime engineering); the REST-only OWNER-gated rollback body and the strictly-greater-activation-epoch pointer move. | +| c1 Go source (same pin) | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The `SYNC_STATUS_ERROR` / `SYNC_STATUS_DISABLED` terminal-state semantics: `ConnectorStatusToAPI` derives DISABLED only from an ERROR-classified sync; `sync_disabled_reason` distinguishes the data-anomaly auto-pause from deliberate pauses. | diff --git a/skills/verify-connector-output/SKILL.md b/skills/verify-connector-output/SKILL.md new file mode 100644 index 0000000..2770d52 --- /dev/null +++ b/skills/verify-connector-output/SKILL.md @@ -0,0 +1,77 @@ +--- +name: verify-connector-output +description: Use when verifying a post-activation connector's sync output: resource/entitlement/grant counts, grant wiring, ID stability across re-sync, and the UI spot-check. Do not use when updating or rolling back a live connector - use update-and-rollback; do not use when diagnosing a failed build, draft test, or sync - use diagnose-authoring-failure. +version: 0.1.0 +--- + +# verify-connector-output + +Post-sync verification for an activated authored connector. Runs in a +post-activation session - never during the funnel run. Tool names below are +the exact tenant MCP titles. + +## Checklist + +1. Read the connector: call `c1_connector_service_get`; confirm + `status.status` is `SYNC_STATUS_DONE` (a subsequent `sync_disabled` is + normal for authored connectors). GATE: DONE. STOP if the status row + reports an error - read `status.lastError` and route through + diagnose-authoring-failure. If `status.status` is `SYNC_STATUS_RUNNING`, + poll with backoff (e.g. every 5-10s) until DONE, ERROR, or DISABLED; if + no terminal status after ~10 polls, STOP and report. + `SYNC_STATUS_DISABLED` is + an error-classified outcome: read the connector's `sync_disabled_reason`. + A data-anomaly auto-pause (reason starting `Sync paused due to + significant drop in sync data`) means the sync was paused after repeated + data drops - investigate the counts below and the drop reason; a + deliberate pause (customer opt-out or ops) is normal; an empty or + unexpected `sync_disabled_reason` routes through + diagnose-authoring-failure. Any other unexpected status value routes + through diagnose-authoring-failure. +2. Counts: inspect the resource, entitlement, and grant counts for the + intended scope. Assert count parity against the live tenant API, not + against a seed list - the live product may own extra rows (default + users, bootstrap objects). GATE: counts match the intended scope. +3. Grant wiring: verify that every grant principal references an emitted resource + and every grant entitlement ID references an emitted entitlement. GATE: no + dangling principal or entitlement ID. +4. ID stability: call `c1_connector_service_force_sync`, then poll + `c1_connector_service_get` with backoff (e.g. every 5-10s) until + `status.status` is `SYNC_STATUS_DONE`, `SYNC_STATUS_ERROR`, or + `SYNC_STATUS_DISABLED`; if no terminal status after ~10 polls, STOP and + report. Confirm the emitted resource and entitlement IDs are stable + across the re-sync - no churn in identity fields. GATE: IDs unchanged. +5. UI spot-check: open `/admin/connector///` + on the product base URL and confirm the connector's resources and grants + appear. GATE: resources and grants visible. + +Investigate empty or unexpected results; never invent data to make a demo appear complete. + +## Exit criteria + +- `c1_connector_service_get` reports `SYNC_STATUS_DONE` (a subsequent + `sync_disabled` is normal). +- Resource, entitlement, and grant counts match the intended scope. +- Every grant principal references an emitted resource; every grant + entitlement ID references an emitted entitlement. +- IDs are stable across a re-sync. +- The UI spot-check at `/admin/connector///` + shows the connector's resources and grants. +- The body contains the literals `SYNC_STATUS_DONE`, `sync_disabled`, + `ID stability`, and `/admin/connector/`. + +## Anti-patterns + +- Never invent data to make a demo appear complete - investigate empty or + unexpected results instead. +- Do not assert fixture counts as the expected counts; query the live API + for count parity. +- Do not run this during the funnel run - it is a post-activation session + skill. +- Do not skip the grant-wiring check because counts look right. + +## Blocker protocol + +If the same validation or runtime error remains unchanged after 2 failed +fix cycles on the same error, stop and report the exact error text instead +of guessing further. diff --git a/skills/verify-connector-output/SOURCES.md b/skills/verify-connector-output/SOURCES.md new file mode 100644 index 0000000..1dc5aec --- /dev/null +++ b/skills/verify-connector-output/SOURCES.md @@ -0,0 +1,13 @@ +# Sources - verify-connector-output + +Authored against the pinned sources below (decision 5: nothing written from +model memory). Source-of-truth precedence: (a) MCP-served guide, (b) +`authoring.proto`, (c) lifecycle doc, (d) marketplace `probe-contracts.md`. + +| Source | Pin / SHA | What this skill quotes | +|---|---|---| +| MCP-served guide | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The Production-sync verification contract verbatim: inspect resource/entitlement/grant counts for the intended scope; every grant principal references an emitted resource; every grant entitlement ID references an emitted entitlement; "Investigate empty or unexpected results; never invent data to make a demo appear complete". | +| Authoring proto | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The authoring RPC surface the post-activation session must not call during verification (the funnel tools; verification uses the tenant connector tools). | +| Lifecycle doc | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | "What success looks like": `SYNC_STATUS_DONE` via `c1_connector_service_get` (a subsequent `sync_disabled` is normal); the UI spot-check path `/admin/connector///`. | +| Marketplace `probe-contracts.md` | claude-marketplace `0cc5ac2a2dbe60b430444c59e53016da2c72b3d1` | The assertion-inventory adaptations: per-resource-type sync counts; count parity against the live API, not the seed list; ID stability across re-sync. | +| c1 Go source (same pin) | c1 `2e5f53eb441a93087d9754085ca17a5061e125ea` | The `SYNC_STATUS_DISABLED` semantics: `ConnectorStatusToAPI` derives DISABLED only from an ERROR-classified sync; `sync_disabled_reason` distinguishes the data-anomaly auto-pause (`Sync paused due to significant drop in sync data` prefix) from deliberate pauses (`system`, `system-customer-opt-out`). |