diff --git a/corpus/verdicts/superdesigndev-treg.json b/corpus/verdicts/superdesigndev-treg.json new file mode 100644 index 0000000..47c2cbf --- /dev/null +++ b/corpus/verdicts/superdesigndev-treg.json @@ -0,0 +1,297 @@ +{ + "repository": "C:\\Users\\Elfrost\\AppData\\Local\\Temp\\ai-patchlab-clone-eb9ea3mu\\repo", + "generated_at": "2026-09-24T13:14:23.423309+00:00", + "total_dismissed": 99, + "by_reason": { + "ignore-pattern": 1, + "below-min-severity": 3, + "by-design": 45, + "credited-defense": 6, + "sql-identifier-fp": 17, + "not-reachable": 19, + "active-harm-fp": 4, + "domain-noun-collision": 3, + "dependency-currency": 1 + }, + "records": [ + { + "source": "scanner", + "reason_code": "ignore-pattern", + "tool": "semgrep", + "rule": "generic.unicode.security.bidi.contains-bidirectional-characters", + "count": 1, + "verdict": "", + "detail": "Suppressed by an ignore pattern." + }, + { + "source": "scanner", + "reason_code": "below-min-severity", + "tool": "semgrep", + "rule": "package_managers.dependabot.dependabot-missing-cooldown.dependabot-missing-cooldown", + "count": 2, + "verdict": "", + "detail": "Below the --min-severity floor (medium)." + }, + { + "source": "scanner", + "reason_code": "below-min-severity", + "tool": "semgrep", + "rule": "package_managers.uv.uv-missing-dependency-cooldown.uv-missing-dependency-cooldown", + "count": 1, + "verdict": "", + "detail": "Below the --min-severity floor (medium)." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "javascript.vue.security.audit.xss.templates.avoid-v-html", + "count": 16, + "verdict": "false-positive", + "detail": "HelpPage.vue x16: v-html renders window.TREG_TUTORIAL, the project's own static tutorial (web/tutorial.js) - first-party content, no user or upstream data." + }, + { + "source": "curation", + "reason_code": "credited-defense", + "tool": "semgrep", + "rule": "javascript.vue.security.audit.xss.templates.avoid-v-html", + "count": 2, + "verdict": "false-positive", + "detail": "App.vue:217 + CopyToolDialog.vue:16: snippet HTML from state/snippets.js + state/skills.js; every interpolated tool field (name, host, path, example, method, org) passes esc(&<>) inside element text before wrapping, and a source comment says why. Attribute values are static literals." + }, + { + "source": "curation", + "reason_code": "sql-identifier-fp", + "tool": "semgrep", + "rule": "python.sqlalchemy.security.audit.avoid-sqlalchemy-text.avoid-sqlalchemy-text", + "count": 17, + "verdict": "false-positive", + "detail": "16 in alembic/versions (SET lock_timeout/statement_timeout from module constants + CONCURRENTLY index DDL); 1 in infra/db.py:330, a test-reset TRUNCATE over ORM-metadata table names quoted by the dialect's identifier_preparer. No value is ever interpolated." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "yaml.github-actions.security.github-actions-mutable-action-tag", + "count": 14, + "verdict": "hardening", + "detail": "Same-rule flood across the workflow files: SHA-pinning third-party actions is supply-chain hardening, not a vulnerability." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "html.security.audit.missing-integrity.missing-integrity", + "count": 6, + "verdict": "false-positive", + "detail": "6 of 8 hits are /preconnect tags (treg.to self-links, pbs.twimg.com) - not resource loads, SRI does not apply." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "html.security.audit.missing-integrity.missing-integrity", + "count": 2, + "verdict": "hardening", + "detail": "landing.html:583/1367 load lenis@1.3.26 CSS+JS from cdn.jsdelivr.net without integrity=; the landing page shares the treg.to origin with the /app dashboard and there is no site-wide CSP, so an integrity hash is worthwhile supply-chain hardening (it only matters after a CDN/npm compromise)." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "semgrep", + "rule": "python.lang.security.audit.dynamic-urllib-use-detected.dynamic-urllib-use-detected", + "count": 9, + "verdict": "false-positive", + "detail": "Operator/dev scripts only (scripts/catalog_drift.py, catalog_fx_update.py, indexnow_submit.py x3, usage_report.py x2, e2e_check.py, one .agents skill script) with operator-configured URLs; none is on a request path. The 2 usage_report.py hits came from the targeted re-run that recovered 5 files lost to a semgrep temp-file error." + }, + { + "source": "curation", + "reason_code": "active-harm-fp", + "tool": "semgrep", + "rule": "python.lang.security.audit.insecure-file-permissions.insecure-file-permissions", + "count": 4, + "verdict": "false-positive", + "detail": "os.chmod(, 0o700) x4 (cli.py:2836 fsjail dir, localproxy.py:257, shell.py:272/275 session+shim dirs) - the project TIGHTENING permissions; the rule's looser suggestion would weaken them." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "python.lang.security.audit.insecure-file-permissions.insecure-file-permissions", + "count": 3, + "verdict": "false-positive", + "detail": "0o755 on executables: cli.py:3049 egress loader + cli.py:3140 runner (root-owned; per the source comment the member cannot modify it), shell.py:148 PATH shims (tool name/route/binary path, no secret)." + }, + { + "source": "curation", + "reason_code": "credited-defense", + "tool": "semgrep", + "rule": "generic.unicode.security.bidi.contains-bidirectional-characters", + "count": 2, + "verdict": "false-positive", + "detail": "scripts/catalog_ingest.py:103 is the project's own ZERO_WIDTH regex that STRIPS zero-width/bidi characters from ingested catalog text - the rule fired on the sanitizer." + }, + { + "source": "curation", + "reason_code": "domain-noun-collision", + "tool": "semgrep", + "rule": "python.lang.security.audit.logging.logger-credential-leak.python-logger-credential-disclosure", + "count": 2, + "verdict": "false-positive", + "detail": "application/auth.py:952 logs a refresh-token FAMILY id + revoked count (reuse detection), not a token; application/call/evidence.py:160 logs the exception from the fail-closed credential-masking layer (_secret_renderings), which then replaces the evidence wholesale." + }, + { + "source": "curation", + "reason_code": "credited-defense", + "tool": "semgrep", + "rule": "python.flask.security.audit.directly-returned-format-string.directly-returned-format-string", + "count": 1, + "verdict": "false-positive", + "detail": "routers/auth.py:180 _login_callback_base returns https://{host} only when the parsed Host is in PUBLIC_HOST_ALIASES (treg.to, treg.superdesign.dev), built from the parsed value not the raw header; a URL string, not an HTML response." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "python.flask.security.audit.directly-returned-format-string.directly-returned-format-string", + "count": 1, + "verdict": "hardening", + "detail": "routers/billing.py:131 _return_base builds the Stripe return URL from the raw Host header (documented intent: preview/local servers return to themselves). Same-user only - the URL goes back to the requesting admin's own Checkout/portal session - but its login sibling already allowlists Host; reusing PUBLIC_HOST_ALIASES/public_url would align them." + }, + { + "source": "curation", + "reason_code": "credited-defense", + "tool": "semgrep", + "rule": "python.django.security.injection.tainted-url-host.tainted-url-host", + "count": 1, + "verdict": "false-positive", + "detail": "routers/auth.py:180 - same site as above: Host allowlisted against PUBLIC_HOST_ALIASES before use." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "semgrep", + "rule": "javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp", + "count": 2, + "verdict": "false-positive", + "detail": "web/tutorial.js:514 (+ its dashboard-legacy copy): RegExp compiled from the project's own static highlighter RULES table; no attacker input reaches the pattern." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "semgrep", + "rule": "javascript.lang.security.insecure-object-assign.insecure-object-assign", + "count": 1, + "verdict": "false-positive", + "detail": "frontend/src/state/tools.js:19 deep-copies the org's own tool CLI profile into fresh local objects for an edit form; no shared/global prototype is reachable." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "python.lang.security.insecure-hash-algorithms.insecure-hash-algorithm-sha1", + "count": 1, + "verdict": "false-positive", + "detail": "cli.py:1570 sha1(base_url)[:10] names a local catalog cache file per server - non-security identifier." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "python.lang.correctness.len-all-count.len-all-count", + "count": 1, + "verdict": "not-applicable", + "detail": "Correctness/perf lint, not a security rule." + }, + { + "source": "curation", + "reason_code": "by-design", + "tool": "semgrep", + "rule": "python.lang.compatibility.python37.python37-compatibility-importlib2", + "count": 1, + "verdict": "not-applicable", + "detail": "Python 3.7 compatibility lint; the project targets modern Python." + }, + { + "source": "curation", + "reason_code": "dependency-currency", + "tool": "trivy", + "rule": "CVE-2026-63374", + "count": 1, + "verdict": "hardening", + "detail": "anyio 4.14.1 -> 4.14.2 (uv.lock; the deploy example installs with --locked). IDNA 2003/2008 hostname confusion in connect_tcp/TLSStream.wrap. On the path: treg's async httpx calls go through httpcore's anyio backend. Exploitation needs a network-positioned attacker AND a non-ASCII upstream host whose IDNA 2003/2008 mappings differ. Lock refresh." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-63349", + "count": 1, + "verdict": "not-applicable", + "detail": "anyio extra_groups bug in run_process/open_process: treg never calls either (only anyio.to_thread in infra/stripe.py)." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-64847", + "count": 1, + "verdict": "not-applicable", + "detail": "anyio process-pool stderr DoS: treg never uses anyio.to_process." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-84382", + "count": 1, + "verdict": "not-applicable", + "detail": "httpx2 client-side decompression DoS: httpx2 is a transitive dep of the mcp server extra; treg never imports it and mcp.py uses server-side SDK modules only." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-84379", + "count": 1, + "verdict": "not-applicable", + "detail": "httpx2 client-side multipart header injection: same - no httpx2 client code runs in treg." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-84380", + "count": 1, + "verdict": "not-applicable", + "detail": "httpx2 client-side request smuggling: same - no httpx2 client code runs in treg." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-41205", + "count": 1, + "verdict": "not-applicable", + "detail": "mako via alembic: template lookup is used only when generating migration scripts (dev time), never on a request path." + }, + { + "source": "curation", + "reason_code": "not-reachable", + "tool": "trivy", + "rule": "CVE-2026-44307", + "count": 1, + "verdict": "not-applicable", + "detail": "mako via alembic: same as CVE-2026-41205." + }, + { + "source": "curation", + "reason_code": "domain-noun-collision", + "tool": "gitleaks", + "rule": "generic-api-key", + "count": 1, + "verdict": "false-positive", + "detail": "catalog/anyapi.extended.yaml:6722 is a YouTube API continuationToken - a public pagination cursor in a catalog example request, not a credential." + } + ] +} \ No newline at end of file diff --git a/docs/index.md b/docs/index.md index 942dae1..9c2ec4a 100644 --- a/docs/index.md +++ b/docs/index.md @@ -1,7 +1,7 @@ --- layout: default title: AI PatchLab Scans -description: "110 curated security scans of open-source AI agents, MCP servers and LLM apps - 25 confirmed fixes, run local-first with Semgrep, Gitleaks, Trivy and pip-audit." +description: "111 curated security scans of open-source AI agents, MCP servers and LLM apps - 25 confirmed fixes, run local-first with Semgrep, Gitleaks, Trivy and pip-audit." --- # AI PatchLab Scans @@ -20,7 +20,7 @@ remediation and confidence rules to normalize the findings. > **Want this run privately against your own codebase?** I do independent > security review of AI agents, MCP servers, and LLM apps — -> [**work with me →**]({{ '/work-with-me' | relative_url }}). 110 scans, 25 confirmed fixes, methodology in the open. +> [**work with me →**]({{ '/work-with-me' | relative_url }}). 111 scans, 25 confirmed fixes, methodology in the open. > **OpenAI just launched [Daybreak](https://openai.com/index/daybreak-securing-the-world/) and Patch the Planet.** > Same remediation loop, opposite trade-off: their path is a cloud frontier model; @@ -103,7 +103,7 @@ login and static assets. Fifty-two flagged, none reported. ## All scans -110 scans, newest first. **Findings** is the raw count the tools produced; +111 scans, newest first. **Findings** is the raw count the tools produced; **Real** is what survived curation. The gap between those two columns is the entire job. @@ -118,6 +118,7 @@ filed, which is the usual outcome of a clean scan. | Date | Repository | Findings | Real | Outcome | | --- | --- | ---: | --- | --- | +| 2026-09-24 | [superdesigndev/treg](scans/superdesigndev-treg.html) | 96 | 1 real — withheld | private | | 2026-09-23 | [can4hou6joeng4/boss-agent-cli](scans/can4hou6joeng4-boss-agent-cli.html) | 26 | 1 real — withheld | private | | 2026-09-22 | [overwirehq/claude-code-telegram](scans/overwirehq-claude-code-telegram.html) | 54 | 0 first-party — dependency | — | | 2026-09-21 | [HarnessRouter/harnessrouter](scans/harnessrouter-harnessrouter.html) | 117 | 1 real — dependency | — | diff --git a/docs/scan-log.md b/docs/scan-log.md index 7feb3db..108d98b 100644 --- a/docs/scan-log.md +++ b/docs/scan-log.md @@ -6,9 +6,10 @@ description: "The complete AI PatchLab scan log: every public repository scanned # Full scan log -Every scan in the series, newest first, with the summary written on the day of the scan. 110 scans. For the compact index, see the [scan log home]({{ '/' | relative_url }}). +Every scan in the series, newest first, with the summary written on the day of the scan. 111 scans. For the compact index, see the [scan log home]({{ '/' | relative_url }}). -- **2026-09-23** — [can4hou6joeng4/boss-agent-cli](scans/can4hou6joeng4-boss-agent-cli.html) — 26 findings at `medium+` (7 high, 16 medium, 3 meta), **1 real — withheld, filed privately** — a **CLI for the BOSS Zhipin recruitment platform** (2k★, MIT; a human drives it from a terminal wizard, an agent drives the same workflows over a JSON-envelope API, an ~77-tool MCP server, or a Python SDK) that authenticates by reusing the operator's own logged-in browser session. Responsive maintainer — 67 merged PRs from 7 human authors in 60 days — which is why it cleared the pre-check. **Zero of the 26 scanner findings is the vulnerability**, which is the whole point of the entry: the real item is the *absence* of a check on an otherwise-working local channel, the shape no rule can match, and it is the [desktop-loopback-inversion](scans/liaohch3-claude-tap.html) class — a `127.0.0.1` control surface that binds neither *who* may call it nor which host the request claims, driving privileged browser actions against a site where the operator is authenticated. Confirmed with a **server-side exploit primitive against the shipped daemon** (a foreign `Host` is served, a cross-origin "simple" request reaches the command channel with no preflight, a hostile `Origin` is not rejected) plus a read of the extension source for the browser half — the [run-the-primitive discipline](scans/observal-observal.html), because reading alone could not separate "the channel is drivable" from "the channel is drivable *and* blind does not blunt it." Graded **High not Critical** on an honest precondition (the bridge component must be actively running — a bounded window, not an always-on service) and the sensitive scope is the one site the tool is built around; **the fix was written before it was named**, narrow and verified not to break the tool's own two legitimate clients ([write the patch first](scans/roflcoopter-viseron.html)). The 26: eight SQL hits (five high) are the [#1 identifier FP](scans/aurelio-labs-semantic-router.html) — every interpolation is a hardcoded table/column name with values bound as `?`, settled by reading the call sites; eight of sixteen mediums are one `github-actions-mutable-action-tag` flood; the rest are two `sha1()[:16]` dedup-id digests, two `urllib` calls to a local DevTools endpoint and a build script, a React rule on a loopback health-check `fetch`, and a Python-3.7 compatibility rule that is not a security check. Coverage `partial` and stated so — Semgrep did not cover every file and the dependency scan left `uv.lock` unaudited (Trivy read it, the [luck that hides the bug](scans/roflcoopter-viseron.html)), neither touching the finding. **Strict-norm** (`SECURITY.md` forbids public vulnerability issues, gives an email and a PVR link) · PVR-enabled, filed **privately, accepted into triage first try** ([GHSA-xp54-2hxc-w78q](https://github.com/can4hou6joeng4/boss-agent-cli/security/advisories/GHSA-xp54-2hxc-w78q)) **with the `vulnerabilities` array** — the [viseron rule](scans/roflcoopter-viseron.html) holds again · post-only, finding withheld at class level under embargo +- **2026-09-24** — [superdesigndev/treg](scans/superdesigndev-treg.html) — 96 findings at `medium+` (1 critical, 23 high, 69 medium, 3 meta), **1 real — withheld, reported privately by email** — **"OpenRouter for agent tools"** (3k★, source-available; hosted at treg.to and self-hostable), a multi-tenant server holding a team's provider credentials Fernet-encrypted and proxying calls to 3,000+ upstream endpoints with the credential injected server-side so it never reaches the caller. Its `SECURITY.md` makes two load-bearing promises: the proxy never hands a key to the caller, and outbound requests are *"re-resolved at call time and internal/metadata addresses are refused."* **Zero of the 96 scanner findings is the vulnerability** — the real item came from tabulating the server's outbound request paths and asking which the call-time host guard actually covers. It does have a proper guard, `host_is_public` in `infra/upstream/ssrf.py`, and its address classification is genuinely thorough (every loopback/link-local/CGNAT/NAT64 and decimal/hex/octal/short-form IPv4 literal I tried was refused). The finding is that the guard is wired into the main proxy path and **not** into a sibling outbound path a plain member can drive; the registration-time check that does run there is, by its own docstring, a static check that allows any DNS name, so a name resolving to an internal address passes it and nothing re-resolves before the request goes out — a member-authenticated, semi-blind [SSRF on the unguarded sibling of a guarded transport](scans/liaohch3-claude-tap.html). Confirmed with the [run-the-primitive](scans/observal-observal.html) discipline: treg's own two guard functions run side by side on a host that resolves internally return "allowed" (registration) and "refused" (call-time), and the sibling path calls only the first — one differential, using the project's code not my reasoning. Graded **Medium**: the injected credential on the affected path is the member's own, so it is internal + cloud-metadata reach, not cross-tenant theft; the org boundary itself held everywhere ([the probe could return "no" on isolation and did](scans/tracecathq-tracecat.html)). **The fix is one guard applied at the outbound path** and changes nothing for a legitimate public target. The 96: seventeen `sqlalchemy-text` are the [#1 identifier FP](scans/aurelio-labs-semantic-router.html) (Alembic `SET`/`CREATE INDEX CONCURRENTLY` from module constants + a test-reset `TRUNCATE` over dialect-quoted table names, no value interpolated); eighteen `v-html` render the project's own static tutorial/snippet HTML with an `esc(&<>)` helper on every interpolated field; four `insecure-file-permissions` are the project [tightening to `0o700`](scans/realiti4-claude-swap.html) (the rule's suggestion would loosen them); one Trivy critical is `anyio` 4.14.1→4.14.2 IDNA hostname-confusion, a worthwhile lockfile bump reachable only by a network-positioned attacker against a non-ASCII upstream host. Coverage `partial` and stated so — Semgrep hit a Windows per-file temp error on five large modules (recovered by a targeted re-run of just those five), and pip-audit left `uv.lock` unaudited while Trivy read it; neither touches the finding. **Strict-norm** (`SECURITY.md` forbids public vuln issues, PVR disabled — checked `{"enabled":false}` — names an email) · reported **privately by email to the maintainer**; post-only here, finding withheld at class level, **the send is the operator's step** +- **2026-09-23** — [can4hou6joeng4/boss-agent-cli](scans/can4hou6joeng4-boss-agent-cli.html) — 26 findings at `medium+` (7 high, 16 medium, 3 meta), **1 real — withheld, filed privately** — a **CLI for the BOSS Zhipin recruitment platform** (2k★, MIT; a human drives it from a terminal wizard, an agent drives the same workflows over a JSON-envelope API, an ~77-tool MCP server, or a Python SDK) that authenticates by reusing the operator's own logged-in browser session. Responsive maintainer — 67 merged PRs from 7 human authors in 60 days — which is why it cleared the pre-check. **Zero of the 26 scanner findings is the vulnerability**, which is the whole point of the entry: the real item is the *absence* of a check on an otherwise-working local channel, the shape no rule can match, and it is the [desktop-loopback-inversion](scans/liaohch3-claude-tap.html) class — a `127.0.0.1` control surface that binds neither *who* may call it nor which host the request claims, driving privileged browser actions against a site where the operator is authenticated. Confirmed with a **server-side exploit primitive against the shipped daemon** (a foreign `Host` is served, a cross-origin "simple" request reaches the command channel with no preflight, a hostile `Origin` is not rejected) plus a read of the extension source for the browser half — the [run-the-primitive discipline](scans/observal-observal.html), because reading alone could not separate "the channel is drivable" from "the channel is drivable *and* blind does not blunt it." Graded **High not Critical** on an honest precondition (the bridge component must be running, and the user has to start it deliberately; once started it stays up while the browser side is connected — *corrected 2026-09-24: first published as "a bounded window", the one claim reasoned about rather than checked, and wrong both ways; the private report was corrected the same day*) and the sensitive scope is the one site the tool is built around; **the fix was written before it was named**, narrow and verified not to break the tool's own two legitimate clients ([write the patch first](scans/roflcoopter-viseron.html)). The 26: eight SQL hits (five high) are the [#1 identifier FP](scans/aurelio-labs-semantic-router.html) — every interpolation is a hardcoded table/column name with values bound as `?`, settled by reading the call sites; eight of sixteen mediums are one `github-actions-mutable-action-tag` flood; the rest are two `sha1()[:16]` dedup-id digests, two `urllib` calls to a local DevTools endpoint and a build script, a React rule on a loopback health-check `fetch`, and a Python-3.7 compatibility rule that is not a security check. Coverage `partial` and stated so — Semgrep did not cover every file and the dependency scan left `uv.lock` unaudited (Trivy read it, the [luck that hides the bug](scans/roflcoopter-viseron.html)), neither touching the finding. **Strict-norm** (`SECURITY.md` forbids public vulnerability issues, gives an email and a PVR link) · PVR-enabled, filed **privately, accepted into triage first try** ([GHSA-xp54-2hxc-w78q](https://github.com/can4hou6joeng4/boss-agent-cli/security/advisories/GHSA-xp54-2hxc-w78q)) **with the `vulnerabilities` array** — the [viseron rule](scans/roflcoopter-viseron.html) holds again · post-only, finding withheld at class level under embargo - **2026-09-22** — [overwirehq/claude-code-telegram](scans/overwirehq-claude-code-telegram.html) — 54 findings at `medium+` (1 critical, 21 high, 30 medium), **0 first-party defects — dependency currency, post-only** — a **Telegram bot that gives authorised users remote access to Claude Code**, i.e. it runs file and Bash tools on a host machine by design, so the whole security surface *is* the boundary around that execution (2k★, MIT, org-backed; active — 15 merged PRs from 7 authors and 9 closed issues in 60 days; strict-norm: real `SECURITY.md`, PVR enabled). Picked with the manual queue at zero. **The story is what precise security documentation looks like.** The real boundary is the `can_use_tool` callback that stops Claude — when steered off-course by content it reads mid-task — from acting outside the approved directory, and the maintainers know and *state* exactly what it covers: path validation is scoped in the docs to precisely six tools (`Read`, `Write`, `Edit`, `MultiEdit`, `NotebookEdit`, `NotebookRead`), while `Grep`/`Glob`/`LS` are deliberately left to the OS sandbox. Reading that as "Read is guarded but Grep is not, so the boundary is incomplete" is the plausible-but-wrong finding this site exists to resist — the boundary is [advertised, not accidental](scans/tracecathq-tracecat.html), and `ROADMAP-v2.md` even records the SDK's own `CanUseToolShadowedWarning` naming every tool the callback will never see, filed as tracking item #221. Two more choices earn credit: guarded tools are deliberately *stripped from* the SDK `allowed_tools` list (a pre-approved tool never produces a `can_use_tool` request, so leaving them in would render the checks silently inert — [issue #219](https://github.com/overwirehq/claude-code-telegram/issues/219)), and `autoAllowBashIfSandboxed` is disabled whenever the boundary checks must run, closing a second bypass. That is a maintainer who traced how the framework resolves permissions rather than trusting that a config allow-list is an enforcement boundary — it isn't, and the code says so. **When the machine is built and documented this carefully, the finding moves to the dependency manifest.** The 30 Trivy advisories in `poetry.lock` split cleanly by reachability: **17 are opt-in-only and not reachable on a default install** — `starlette` (6) and `python-multipart` (3) need the FastAPI webhook server (`ENABLE_API_SERVER=false` default); `pyjwt` (5) needs token auth (`ENABLE_TOKEN_AUTH=false` default, and the SECURITY.md documents token auth as non-functional, [#58](https://github.com/overwirehq/claude-code-telegram/issues/58)); the MCP SDK highs (3) need MCP enabled — while **13 are unconditionally installed** (`anyio` incl. the critical IDNA/TLS advisory, the `cryptography`/OpenSSL cluster, `urllib3`, `idna`, `requests`, `pydantic-settings`, `python-dotenv`) and are the genuine currency gap. The direct deps are current; the drift is transitive, which the repo's existing `.github/dependabot.yml` (configured for *version* updates, not *security* updates) will not raise on its own. **No advisory filed** — there is no first-party vulnerability. Of the other 24: 16 `github-actions-mutable-action-tag` (SHA-pin hardening); one `pull_request_target` checkout wrapped in a 40-line threat-model header (read-only tool allowlist, secrets scrubbed, "worst case is a prompt-injected review comment" — [the trigger decides severity](scans/lightseekorg-tokenspeed.html)); two `sqlalchemy-execute-raw-query` [identifier FPs](scans/aurelio-labs-semantic-router.html) (only a generated run of `?` placeholders is interpolated, values bound via `execute(query, params)`); and three gitleaks hits that are a `your-api-secret` placeholder plus two fake tokens in a test asserting the project's own `_redact_secrets()` helper scrubs them ([credited defence](scans/realiti4-claude-swap.html)). Coverage `partial` and honestly so: pip-audit resolved **no dependencies** from a `pyproject.toml` declaring `dynamic = ["dependencies"]` — reading identically to "clean" — and Trivy's `poetry.lock` read carried the run, the [same dependency blind spot](scans/harnessrouter-harnessrouter.html) as yesterday reached by a different manifest shape. - **2026-09-21** — [HarnessRouter/harnessrouter](scans/harnessrouter-harnessrouter.html) — 117 findings (2 critical, 42 high, 70 medium), **1 real — dependency currency, post-only** — a **self-hosted control plane that turns agent CLIs (Codex, Claude Code, Hermes, and a dozen more) into a single OpenAI-compatible API** (1.7k★, Apache-2.0, org-backed with a hosted "Cloud" edition; very active — 91 merged PRs from 5 authors in 60 days). Picked with the manual queue at zero: strict-norm (real `SECURITY.md`, PVR enabled, commercial backing), which is a fair target when the backlog is clear. **The story here is a well-built authorization layer that survived the sweep the series usually breaks projects on, so the one finding moved to the dependency manifest.** The route inventory — 113 gateway routes — comes back with exactly five answering without a credential once the project's own identity idioms (`_owned_session`, `_pub_org_member`, `_principal`) are resolved: `/v1/uhp`, `/healthz`, `/readyz`, `/version`, and `/share/{token}` where the unguessable token *is* the credential. That is the [name-matched sweep](scans/mai-with-u-maibot.html) returning empty for the right reason. Two design choices earn explicit credit: the self-hosted BFF stamps its internal trust key onto a gateway call **only for a request carrying a valid session cookie** — an earlier build stamped it unconditionally, and the code comments document catching and fixing exactly that [conditional-verification](scans/sentelabsai-openexecutive.html) class before I arrived — and the gateway **binds loopback with only the UI port published**, so the [DNS-rebinding shape](scans/liaohch3-claude-tap.html) that has caught several desktop apps here does not apply (the published surface is the session-gated Next.js app with `X-Frame-Options: DENY`). **The actionable finding is dependency currency, and it only surfaced because the two dependency tools disagreed about whether there was anything to scan.** pip-audit reported **no manifest** — it scans the repo root, and the requirements live in `gateway/requirements.txt` and `runner/requirements.txt` one level down — which on a monorepo reads identically to "clean". Trivy's whole-tree walk read both and found the pinned `next` **15.5.23** is one patch behind **15.5.24**, which fixes two criticals: [CVE-2026-75604](https://github.com/advisories) (Windows-hosted RCE — **dropped, the image is Linux**) and [GHSA-2xp9-vwfh-vxw4](https://github.com/advisories) (AVIF image-optimizer RCE — the optimizer runs at its default-enabled setting and the middleware matcher **excludes `_next/image`**, so the surface is reachable unauthenticated; default-empty `remotePatterns` constrains full exploitation, and with no Docker Linux engine available I could not run the primitive, so I claim the reachable surface and the currency gap, not a demonstrated RCE). `gateway/requirements.txt` also carries `PyJWT` 2.10.1 (CVE-2026-48526, auth-bypass) and `cryptography`/`aiohttp`/`python-multipart` CVEs — low reachability self-hosted (gateway loopback, `HR_IDENTITY_MODE=off`) but the hosted build shares the code. **No advisory filed**: the actionable item is a published upstream CVE with a one-line fix (`next >=15.5.24` + a `dependabot.yml` covering npm *and* pip, of which there is none today), not a first-party defect, and the "run the exploit primitive before filing" rule forbids an RCE-shaped advisory I can't demonstrate. Of the other 116: 29 `github-actions-mutable-action-tag` (SHA-pin hardening); 11 workflow shell-injection all on `workflow_dispatch`/`push:tags` triggers ([trigger decides severity](scans/lightseekorg-tokenspeed.html) — write access already required); 13 subprocess-audit hits in `runner/`, which runs agent CLIs as a per-session uid ([running code is the product](scans/realiti4-claude-swap.html)); a `ws://` `detect-insecure-websocket` false positive (matched a `.replace()` scheme-transform string); a test-fixture key in `gateway/tests/`. Coverage `partial` and honestly so — Semgrep's errors are non-Python config/data files, no first-party module skipped. The backlog item is real and is the day's [tooling note](scans/whiteguo233-openbiliclaw.html): `scan_dependency` is root-only, so on a monorepo it silently disagrees with Trivy — it should descend into subdirectory manifests or emit a louder meta-finding. diff --git a/docs/scans/can4hou6joeng4-boss-agent-cli.md b/docs/scans/can4hou6joeng4-boss-agent-cli.md index 44d27f6..a6f0cb7 100644 --- a/docs/scans/can4hou6joeng4-boss-agent-cli.md +++ b/docs/scans/can4hou6joeng4-boss-agent-cli.md @@ -89,10 +89,22 @@ shipped code rather than reasoning about them: one, which exposes the credential material for the logged-in session directly. The precondition is honest and worth stating plainly: this is only exploitable -while the bridge component is actively running, which is a bounded window, not an -always-on service — and the sensitive scope is the one site the tool is built -around. That is why it is a High and not a Critical, and I graded it that way in -the report rather than inviting the maintainer to discover the scope themselves. +while the bridge component is running, and the user has to start it deliberately — +nothing in the package starts it for them. Once started, though, it stays up for as +long as the browser side stays connected, so the exposure is the whole browsing +session rather than a short window — and the sensitive scope is the one site the +tool is built around. That is why it is a High and not a Critical, and I graded it +that way in the report rather than inviting the maintainer to discover the scope +themselves. + +*Correction, 2026-09-24:* the first version of this paragraph, and of the private +report, called the exposure "a bounded window". That was the one claim in the +report I reasoned about instead of checking, and it was wrong in both directions. I +had described the bridge as starting on its own: a start routine exists, but +nothing calls it, and the maintainer has since removed it as dead code. I had also +described an idle timeout that, read in full, does not fire while the browser side +is connected. The private report was corrected the same day; the finding, the fix +and the severity are unchanged. **No rule described it.** Every scanner here looks for a dangerous *thing that is present* — a tainted sink, a bad call, a known-vulnerable version. This is the @@ -161,7 +173,8 @@ python scanner/run_scan.py --repo /tmp/scan-target --reports-dir ./reports/can4h ## More from this series +- **Next scan:** [superdesigndev/treg](superdesigndev-treg.html) — 2026-09-24, 1 real — withheld - **Previous scan:** [overwirehq/claude-code-telegram](overwirehq-claude-code-telegram.html) — 2026-09-22, 0 first-party -- [Every scan in the series]({{ '/' | relative_url }}) — 110 repositories, newest first +- [Every scan in the series]({{ '/' | relative_url }}) — 111 repositories, newest first - [Where a maintainer shipped a fix]({{ '/fixed' | relative_url }}) — the 25 that resolved - [Scans that found nothing]({{ '/clean' | relative_url }}) — published as they were diff --git a/docs/scans/superdesigndev-treg.md b/docs/scans/superdesigndev-treg.md new file mode 100644 index 0000000..2c007c4 --- /dev/null +++ b/docs/scans/superdesigndev-treg.md @@ -0,0 +1,178 @@ +--- +layout: default +title: "superdesigndev/treg: security scan" +description: "Security scan of superdesigndev/treg: 96 findings at medium+, 1 real — withheld, reported privately. Local-first curated review: Semgrep, Gitleaks, Trivy, pip-audit." +date: 2026-09-24 +--- + +# superdesigndev/treg — security scan + +**Repository:** [superdesigndev/treg](https://github.com/superdesigndev/treg) +**Commit scanned:** `240c595` +**Scan date:** 2026-09-24 +**Disclosure status:** withheld — one real finding reported privately by email, per the +project's `SECURITY.md` (which asks that vulnerabilities not be opened as public +issues; GitHub private vulnerability reporting is disabled on the repo, and the +policy names an email address, so email is the channel). Detail below is kept at +class level until the maintainer has had a chance to respond. **The private send +is the operator's step and had not gone out when this page was published.** + +## Summary + +| Severity | Count | +| --- | ---: | +| Critical | 1 | +| High | 23 | +| Medium | 69 | +| Low | — | +| Info | 3 (scanner meta) | + +**Total findings:** 96 at `--min-severity medium` (1 real after curation — withheld). +The one real finding is **not among the 96** — no rule reported it. It came from a +structural question about the server's outbound requests, not a pattern match. + +treg (3.0k★, source-available) is **"OpenRouter for agent tools"**: a multi-tenant +server that holds a team's provider credentials (Fernet-encrypted, server-side) and +proxies tool calls to 3,000+ upstream endpoints, injecting the credential so it never +reaches the caller. It runs hosted at treg.to and is self-hostable. A security model +like that lives and dies on two promises its own `SECURITY.md` makes plainly: the +proxy never hands a key to the caller, and outbound requests are re-resolved at call +time so they cannot be aimed at internal or cloud-metadata hosts. + +Maintenance is responsive: hundreds of merged PRs from several human authors in the +last 60 days, with real back-and-forth on closed issues. That responsiveness is why +it cleared the pre-check. + +## Scan coverage + +| Tool | Status | Detail | +| --- | --- | --- | +| `semgrep` | `partial` | did not cover every file it was pointed at (`semgrep-partial-coverage`) | +| `gitleaks` | `ran` | no coverage problem reported | +| `trivy` | `ran` | no coverage problem reported | +| `dependency-scan` | `partial` | a shipped lockfile was not covered (`dependency-scan-unaudited-lockfile`) | +| `ai-security-review` | `not_run` | disabled by default (ADR-010) | + +**Coverage complete:** no. Semgrep hit a per-file error on five large modules — a +Windows temp-file permission error mid-scan, not a parse failure — so I re-ran those +five files on their own and folded the two recovered findings (both dev-script +`urllib` calls) into the curation below. The dependency scan left the shipped +`uv.lock` unaudited by pip-audit; Trivy read it in the same run, which is what +carried the dependency picture. Neither gap touches the real finding, which is in +first-party code the scanners did read — they simply had no rule for its shape. + +## The one real finding, at class level + +**Severity: Medium. Class: a server-side outbound request path, reachable by any +member, that skips the call-time host check the main proxy enforces — an SSRF on the +unguarded sibling of a guarded transport.** + +The interesting thing about treg's SSRF posture is that it is *mostly right*. It has a +proper call-time guard that resolves a target host and refuses internal, loopback, +link-local, CGNAT, NAT64 and cloud-metadata addresses — I tried the whole zoo of +obfuscated IPv4 literals (decimal, hex, octal, short-form) and the NAT64 prefix, and +every one was refused. That guard is wired into the main proxy path, exactly where the +threat model expects it. + +The finding is that the same server makes outbound requests on **more than one** path, +and the call-time guard is applied on one of them and not on a sibling that an +ordinary member can drive. The registration-time check that *does* run on the sibling +is, by its own documented design, a static check that deliberately allows any DNS +name — so a name that resolves to an internal address passes it, and nothing +re-resolves-and-checks before the request goes out. The result is a member-authenticated +server-side request reachable from the product's own egress, with enough of a +status/error signal returned to make it useful. + +I confirmed the gap the way this series always tries to: not by reading that a check +was missing, but by running the project's *own* two guard functions side by side on a +host that resolves internally. One returns "allowed" (the registration check); the +other returns "refused" (the call-time check the sibling path never calls). That one +differential is the whole finding, and it is decisive because it uses treg's code, not +my reasoning about it. + +**Why Medium, and stated plainly:** the credential injected on the affected path is the +member's *own*, so this is not a route to another tenant's key — it is internal +reconnaissance and metadata reach, not cross-tenant theft. It needs an authenticated +member, which on a self-service product is a low bar but not zero. I graded it Medium +in the private report rather than inviting the maintainer to discover the scope +themselves. + +**The fix is one guard, applied at the outbound path, and it changes nothing for a +legitimate public target** — it only refuses the hosts the project's own documentation +already says are refused at call time. The private report gives the exact call sites +and the one-line change at each, plus a few clearly-labelled lower-confidence +authorization observations for the maintainer to assess. The org boundary itself held +everywhere I looked — I found no way for one tenant to reach another's data, and I say +so because a probe that can only ever return "yes" is not worth much; this one could +have returned "no" on the isolation question and did. + +**No rule described it.** Every scanner here looks for a dangerous *thing that is +present* — a tainted sink, a bad call, a known-vulnerable version. This is a check that +is present and correct on one path and simply *absent* on its sibling, which is the +shape no per-file pattern matcher can see. It is the same class as several earlier +findings in this series where a control guarded one transport and not the one beside +it. + +## Why the rest were dismissed + +| Reason | Count | | +| --- | ---: | --- | +| `by-design` | 45 | non-security, intended behaviour, or supply-chain hardening | +| `not-reachable` | 19 | dev/operator scripts, or a transitive CVE not on a request path | +| `sql-identifier-fp` | 17 | identifier-only interpolation, values bound | +| `credited-defense` | 6 | the rule fired on the project's own guard | +| `active-harm-fp` | 4 | the suggested change would *loosen* a `0o700` tightening | +| `domain-noun-collision` | 3 | a "token"/"key" that is a pagination cursor or a family id | +| `dependency-currency` | 1 | one lockfile bump worth making, not a first-party defect | + +The two biggest survivors are the recurring ones. Seventeen `sqlalchemy-text` hits are +the **#1 identifier false positive**: sixteen are Alembic migrations setting +`lock_timeout` from module constants and building `CREATE INDEX CONCURRENTLY` DDL, and +the last is a test-reset `TRUNCATE` over ORM-metadata table names quoted by the +dialect's own identifier preparer — no value is ever interpolated. Eighteen `v-html` +hits render the project's *own* static tutorial and syntax-highlighted snippet HTML, +where every interpolated tool field is passed through an `esc(&<>)` helper before +wrapping and a source comment says why. Four `insecure-file-permissions` hits are the +project **tightening** to `0o700` on its sandbox directories — the rule's looser +suggestion would weaken them, the [active-harm false-positive](realiti4-claude-swap.html) +shape. And the one Trivy critical (`anyio` 4.14.1 → 4.14.2, an IDNA hostname-confusion +issue) is a real lockfile bump worth making, reachable only by a network-positioned +attacker against a non-ASCII upstream host — hardening, not a first-party vulnerability. + +## Notes on the tool + +- **The real finding was invisible to all four scanners, and that is the point.** 96 + findings at `medium+`, zero of them the vulnerability. It came from tabulating the + server's outbound request paths and asking which ones the call-time host guard + actually covers — a question no per-file rule poses. +- **Coverage was partial in two ways, and the report says so.** Semgrep hit a + Windows-side per-file temp error on five large modules (recovered by a targeted + re-run), and pip-audit left `uv.lock` unaudited while Trivy read it. Both are in the + coverage block above; neither touches the finding. +- **Same-rule and same-shape flooding, again.** The 96 collapse to a handful of + families — eighteen `v-html`, seventeen `sqlalchemy-text`, fourteen unpinned-action + hits. Collapsing repeated same-rule hits into one entry with a count remains a + standing backlog item across the series. + +--- + +*Scanned with [AI PatchLab](https://github.com/elfrost/ai-patchlab). Findings are +curated by hand; scanner output alone is not a vulnerability report. This page will be +updated with full technical detail once the maintainer has responded and a fix has +shipped.* + +## Reproduce + +```bash +git clone https://github.com/superdesigndev/treg /tmp/scan-target +python scanner/run_scan.py --repo /tmp/scan-target --reports-dir ./reports/superdesigndev-treg --min-severity medium --ignore-samples +``` + +--- + +## More from this series + +- **Previous scan:** [can4hou6joeng4/boss-agent-cli](can4hou6joeng4-boss-agent-cli.html) — 2026-09-23, 1 real — withheld +- [Every scan in the series]({{ '/' | relative_url }}) — 111 repositories, newest first +- [Where a maintainer shipped a fix]({{ '/fixed' | relative_url }}) — the 25 that resolved +- [Scans that found nothing]({{ '/clean' | relative_url }}) — published as they were