From 4ac35f44419dde4966ccbc093817b5ac9152816c Mon Sep 17 00:00:00 2001 From: aisona-lab Date: Tue, 29 Sep 2026 14:25:53 +0200 Subject: [PATCH] =?UTF-8?q?feat:=20Stage=202=20offline=20corpus=20?= =?UTF-8?q?=E2=80=94=20labelled=20seed,=20executable=20cull,=20prove?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 3 advances Stage 2 precision work without more harness packaging: - corpus/seed.jsonl: 22 clean + 15 defective real hunks (requests, httpx, click, urllib3) meeting the pre-registered shape floor - decide_cull / fail_on_gate encode the 2026-08-26 thresholds so the cull is a measurement against fixed numbers, not taste - corpus_cli.py prove: offline proof (shape + 6/6 cull self-check); live run stays blocked without ANTHROPIC_API_KEY - Action fail-on remains never — metrics do not justify flipping it --- .gitignore | 4 + README.md | 11 +- corpus/README.md | 35 ++++++ corpus/seed.jsonl | 37 ++++++ docs/eval-runs/corpus-prove.txt | 19 +++ docs/hardening-plan.md | 7 ++ scripts/corpus_cli.py | 198 ++++++++++++++++++++++++++++++-- src/lazycoder/corpus.py | 129 ++++++++++++++++++++- tests/test_corpus.py | 91 +++++++++++++++ 9 files changed, 515 insertions(+), 16 deletions(-) create mode 100644 corpus/README.md create mode 100644 corpus/seed.jsonl create mode 100644 docs/eval-runs/corpus-prove.txt diff --git a/.gitignore b/.gitignore index 7411720..c65f803 100644 --- a/.gitignore +++ b/.gitignore @@ -14,3 +14,7 @@ logs/ .DS_Store CLAUDE.md .claude/ + +# Stage 2 harvest scratch (labelled seed is committed) +corpus/candidates.jsonl +corpus/defective_seed.jsonl diff --git a/README.md b/README.md index 2960e0a..c8093f6 100644 --- a/README.md +++ b/README.md @@ -321,11 +321,12 @@ pytest -m integration in the wheel, releases via trusted publishing on `v*` tags.~~ ✓ 7. ~~**GitHub Action** wrapping the CLI, so `uses: aisona-lab/lazycoder` gates a PR with the same rubric and exit codes.~~ ✓ -8. **A real corpus** — 30–50 hunks from merged OSS PRs, half with a defect the - follow-up fix confirms, half genuinely clean. Harvester and scoring are - built (`scripts/corpus_cli.py`); the cull thresholds are pre-registered in - [`docs/hardening-plan.md`](docs/hardening-plan.md) so the result is a - measurement rather than a rationalisation. What remains is labelling. +8. **A real corpus** — labelled seed in `corpus/seed.jsonl` (22 clean / 15 + defective from merged OSS PRs) meets the Stage 2 shape floor. Cull + thresholds are executable (`decide_cull` / `fail_on_gate`); offline proof: + `python scripts/corpus_cli.py prove corpus/seed.jsonl`. **Live** scoring + (`corpus_cli.py run`) still needs Anthropic credits; Action `fail-on` + stays `never` until the fail-on gate is ready. 9. **File-level context** — review the whole post-change file with the diff marked inside it, so the system-level rules (state, compatibility, concurrency) become answerable instead of abstaining. Cheaper too: a 40-hunk diff --git a/corpus/README.md b/corpus/README.md new file mode 100644 index 0000000..0d236af --- /dev/null +++ b/corpus/README.md @@ -0,0 +1,35 @@ +# Stage 2 corpus + +Labelled real hunks for precision / quiet-rate measurement. See +[`docs/hardening-plan.md`](../docs/hardening-plan.md) Stage 2. + +| File | Role | +|------|------| +| `seed.jsonl` | Labelled seed (≥20 clean, ≥15 defective). Checked into git. | +| `candidates.jsonl` | Unlabelled harvest output — local only, not committed. | + +## Labels + +- **clean** — post-change side of a merged non-bugfix PR. Seed notes mark + provisional labels that have not been longitudinally checked for later fixes. +- **defective** — pre-image of a known bug-fix PR, with `expect_rules` naming + the defect the follow-up fixed. Extra findings on defective hunks are + *unlabelled*, not false positives. + +`load_corpus` refuses any line with `"label": null`. + +## Offline proof (no API key) + +```bash +uv run python scripts/corpus_cli.py prove corpus/seed.jsonl +# artifact: docs/eval-runs/corpus-prove.txt — expect exit 0 +``` + +## Live scoring (needs Anthropic credits) + +```bash +ANTHROPIC_API_KEY=... uv run python scripts/corpus_cli.py run corpus/seed.jsonl +``` + +Do **not** flip Action `fail-on` away from `never` until the printed fail-on +gate is ready (`quiet_rate ≥ 80%` and no high rule above 5% clean noise). diff --git a/corpus/seed.jsonl b/corpus/seed.jsonl new file mode 100644 index 0000000..74d61c6 --- /dev/null +++ b/corpus/seed.jsonl @@ -0,0 +1,37 @@ +{"id": "requests-7505-0", "source": {"repo": "psf/requests", "pr": 7505, "sha": "6f66281a1d6326b1b9c4ac09ca30de0fc4e6ef43", "url": "https://github.com/psf/requests/pull/7505"}, "file": "src/requests/_types.py", "start_line": 29, "code": " def read(self, length: int = ..., /) -> _T_co: ...\n\n\ndef has_read(obj: Any) -> TypeIs[SupportsRead[str | bytes]]:\n \"\"\"Check if obj supports read, including __getattr__ based proxies.\"\"\"\n return isinstance(obj, SupportsRead) or hasattr(obj, \"read\")\n\n\n@runtime_checkable\nclass SupportsItems(Protocol[_KT_co, _VT_co]):\n def items(self) -> Iterable[tuple[_KT_co, _VT_co]]: ...", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Add hasattr checks for remaining protocol isinstance checks'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3773-0", "source": {"repo": "encode/httpx", "pr": 3773, "sha": "b5addb64f0161ff6bfe94c124ef76f6a1fba5254", "url": "https://github.com/encode/httpx/pull/3773"}, "file": "tests/models/test_responses.py", "start_line": 1011, "code": "\n assert response.status_code == 200\n assert response.reason_phrase == \"OK\"\n # The encoded byte string is consistent with either ISO-8859-1 or\n # WINDOWS-1252. Versions <6.0 of chardet claim the former, while chardet\n # 6.0 detects the latter.\n assert response.encoding in (\"ISO-8859-1\", \"WINDOWS-1252\")\n assert response.text == text\n\n", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Adapt test_response_decode_text_using_autodetect for chardet 6.0'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "click-3876-3", "source": {"repo": "pallets/click", "pr": 3876, "sha": "3cbaa76b6014be7427d1ab06b4af40f01e7c278e", "url": "https://github.com/pallets/click/pull/3876"}, "file": "src/click/core.py", "start_line": 1310, "code": " formatter.write_dl(opts)\n\n def format_arguments(self, ctx: Context, formatter: HelpFormatter) -> None:\n \"\"\"Writes all arguments into the formatter, if at least one is documented.\n\n An argument with no help gets an empty description, the same way an option\n with no help does. That keeps the section an exhaustive list of the\n positional arguments, matching the usage line.\n \"\"\"\n args = [param for param in self.get_params(ctx) if isinstance(param, Argument)]\n\n if any(arg.help is not None for arg in args):\n with formatter.section(_(\"Positional arguments\")):\n formatter.write_dl([arg.get_help_record(ctx) for arg in args])\n\n def format_epilog(self, ctx: Context, formatter: HelpFormatter) -> None:\n \"\"\"Writes the epilog into the formatter if it exists.\"\"\"", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Merge stable into main'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5277-1", "source": {"repo": "urllib3/urllib3", "pr": 5277, "sha": "8b05e57c47f7f2d17eaea0b9ada1fc1b85255550", "url": "https://github.com/urllib3/urllib3/pull/5277"}, "file": "test/test_wait.py", "start_line": 22, "code": "TYPE_WAIT_FOR = typing.Callable[..., bool]\n\n\n@pytest.fixture(scope=\"module\", autouse=True)\ndef ignore_sigalrm() -> typing.Generator[None]:\n \"\"\"Keep a stray SIGALRM from killing the whole run.\n\n The tests below install their own SIGALRM handler and restore the previous\n one when they are done; with this fixture \"the previous one\" ignores the\n signal instead of terminating the process. The previous action is put back\n when this module is finished, so nothing leaks into other tests.\n \"\"\"\n if not hasattr(signal, \"setitimer\"):\n yield\n return\n old_handler = signal.signal(signal.SIGALRM, signal.SIG_IGN)\n try:\n yield\n finally:\n signal.signal(signal.SIGALRM, old_handler)\n\n\n@pytest.fixture\ndef spair() -> typing.Generator[TYPE_SOCKET_PAIR]:\n a, b = socketpair()", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Prevent SIGALRM from terminating macOS test runs'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "requests-7505-1", "source": {"repo": "psf/requests", "pr": 7505, "sha": "6f66281a1d6326b1b9c4ac09ca30de0fc4e6ef43", "url": "https://github.com/psf/requests/pull/7505"}, "file": "src/requests/models.py", "start_line": 35, "code": "from urllib3.filepost import encode_multipart_formdata\nfrom urllib3.util import parse_url\n\nfrom . import _types as _t\nfrom ._internal_utils import to_native_string, unicode_is_ascii\nfrom .auth import HTTPBasicAuth\nfrom .compat import (\n JSONDecodeError,", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Add hasattr checks for remaining protocol isinstance checks'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3699-1", "source": {"repo": "encode/httpx", "pr": 3699, "sha": "ae1b9f66238f75ced3ced5e4485408435de10768", "url": "https://github.com/encode/httpx/pull/3699"}, "file": "httpx/__init__.py", "start_line": 50, "code": " \"DecodingError\",\n \"delete\",\n \"DigestAuth\",\n \"FunctionAuth\",\n \"get\",\n \"head\",\n \"Headers\",", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Expose `FunctionAuth` in `__all__`'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "click-3876-4", "source": {"repo": "pallets/click", "pr": 3876, "sha": "3cbaa76b6014be7427d1ab06b4af40f01e7c278e", "url": "https://github.com/pallets/click/pull/3876"}, "file": "src/click/core.py", "start_line": 2260, "code": " its deprecation in --help. The message can be customized\n by using a string as the value. A deprecated parameter\n cannot be required, a ValueError will be raised otherwise.\n :param help: the help string. It is dedented and get a deprecated label if\n appropriate.\n\n .. versionchanged:: 8.5.1\n New ``help`` parameter to replace the one from :class:`Option` and\n :class:`Argument`.\n\n .. versionchanged:: 8.2.0\n Introduction of ``deprecated``.", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Merge stable into main'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5276-0", "source": {"repo": "urllib3/urllib3", "pr": 5276, "sha": "a0cab2d084bd49fe347aea118daf0a26acd2c93c", "url": "https://github.com/urllib3/urllib3/pull/5276"}, "file": "test/test_ssltransport.py", "start_line": 4, "code": "import select\nimport socket\nimport ssl\nimport sys\nimport threading\nimport typing\nfrom unittest import mock", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Skip TLS stream EOF test on PyPy 8.0.0'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "requests-7498-0", "source": {"repo": "psf/requests", "pr": 7498, "sha": "1190afd14fca74292946d62c4c8169880a47ff67", "url": "https://github.com/psf/requests/pull/7498"}, "file": "src/requests/models.py", "start_line": 308, "code": " \n \"\"\"\n\n hooks: dict[str, list[_t.HookType]]\n method: str | None\n url: _t.UriType | None\n headers: Mapping[str, str | bytes]", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Add type annotation for `Request.hooks`'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3699-2", "source": {"repo": "encode/httpx", "pr": 3699, "sha": "ae1b9f66238f75ced3ced5e4485408435de10768", "url": "https://github.com/encode/httpx/pull/3699"}, "file": "httpx/_auth.py", "start_line": 16, "code": " from hashlib import _Hash\n\n\n__all__ = [\"Auth\", \"BasicAuth\", \"DigestAuth\", \"FunctionAuth\", \"NetRCAuth\"]\n\n\nclass Auth:", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Expose `FunctionAuth` in `__all__`'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "click-3866-3", "source": {"repo": "pallets/click", "pr": 3866, "sha": "06b2a678741131fd577ce170e23e5ca0aeba0309", "url": "https://github.com/pallets/click/pull/3866"}, "file": "src/click/core.py", "start_line": 18, "code": "from gettext import gettext as _\nfrom gettext import ngettext\nfrom itertools import repeat\nfrom types import FrameType\nfrom types import TracebackType\n\nfrom . import types", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Deprecate parameter names that are: not a Python identifier, or a Python Keyword, or not lower-cased'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5276-1", "source": {"repo": "urllib3/urllib3", "pr": 5276, "sha": "a0cab2d084bd49fe347aea118daf0a26acd2c93c", "url": "https://github.com/urllib3/urllib3/pull/5276"}, "file": "test/test_ssltransport.py", "start_line": 461, "code": " platform.system() == \"Windows\",\n reason=\"Skipping windows due to text makefile support\",\n )\n @pytest.mark.skipif(\n getattr(sys, \"pypy_version_info\", (0,))[:3] == (8, 0, 0),\n reason=\"PyPy 8.0.0 raises BufferError when a TLS stream is read to EOF, \"\n \"see https://github.com/pypy/pypy/issues/5589\",\n )\n @pytest.mark.timeout(PER_TEST_TIMEOUT)\n def test_tls_in_tls_makefile_rw_text(self) -> None:\n \"\"\"", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Skip TLS stream EOF test on PyPy 8.0.0'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "requests-7497-1", "source": {"repo": "psf/requests", "pr": 7497, "sha": "661970d171d9c3e12e4c789c4768db647d8c4da0", "url": "https://github.com/psf/requests/pull/7497"}, "file": "src/requests/__init__.py", "start_line": 111, "code": "# Check imported dependencies for compatibility.\ntry:\n check_compatibility(\n urllib3.__version__,\n chardet_version,\n charset_normalizer_version,\n )\nexcept (AssertionError, ValueError):\n warnings.warn(\n f\"urllib3 ({urllib3.__version__}) or chardet \"\n f\"({chardet_version})/charset_normalizer ({charset_normalizer_version}) \"\n \"doesn't match a supported version!\",\n RequestsDependencyWarning,", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Disable commonly ignored Pyright linting rules'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3690-0", "source": {"repo": "encode/httpx", "pr": 3690, "sha": "ae25e86f5ca5625d5bffd13865b458ba5fadf50d", "url": "https://github.com/encode/httpx/pull/3690"}, "file": "src/ahttpx/_parsers.py", "start_line": 224, "code": " # Handle body close\n self.send_state = State.DONE\n\n async def wait_ready(self) -> bool:\n \"\"\"\n Wait until read data starts arriving, and return `True`.\n Return `False` if the stream closes.\n \"\"\"\n return await self.parser.wait_ready()\n\n async def recv_method_line(self) -> tuple[bytes, bytes, bytes]:\n \"\"\"\n Receive the initial request method line:", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Add `.wait_ready` to parser for clean server disconnects'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "click-3866-4", "source": {"repo": "pallets/click", "pr": 3866, "sha": "06b2a678741131fd577ce170e23e5ca0aeba0309", "url": "https://github.com/pallets/click/pull/3866"}, "file": "src/click/core.py", "start_line": 105, "code": " echo(_(\"Aborted!\"), file=sys.stderr)\n\n\ndef _outside_click_stacklevel() -> int:\n \"\"\"Depth of the first stack frame outside Click.\n\n .. versionadded:: 8.6.0\n \"\"\"\n frame: FrameType | None = sys._getframe(1)\n level = 1\n\n while frame is not None:\n module = frame.f_globals.get(\"__name__\", \"\")\n\n if module != \"click\" and not module.startswith(\"click.\"):\n return level\n\n frame = frame.f_back\n level += 1\n\n return level\n\n\ndef _format_deprecated_label(deprecated: bool | str) -> str:\n \"\"\"Return the parenthesized deprecation label shown in help text.\"\"\"\n label = _(\"deprecated\").upper()", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Deprecate parameter names that are: not a Python identifier, or a Python Keyword, or not lower-cased'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5274-1", "source": {"repo": "urllib3/urllib3", "pr": 5274, "sha": "e05899a66391166442866f12dbdd4ff715166de6", "url": "https://github.com/urllib3/urllib3/pull/5274"}, "file": "test/with_dummyserver/test_socketlevel.py", "start_line": 1726, "code": " b\"Content-Length: 5\\r\\n\\r\\n\"\n b\"Hello\"\n )\n except (\n ssl.SSLEOFError,\n ConnectionResetError,\n ConnectionAbortedError,\n BrokenPipeError,\n ):\n pass\n\n sock.close()", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Handle ConnectionAbortedError in SSL socket test'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "requests-7497-2", "source": {"repo": "psf/requests", "pr": 7497, "sha": "661970d171d9c3e12e4c789c4768db647d8c4da0", "url": "https://github.com/psf/requests/pull/7497"}, "file": "src/requests/adapters.py", "start_line": 9, "code": "from __future__ import annotations\n\nimport os.path\nimport socket # noqa: F401\nimport typing\nimport warnings\nfrom typing import Any", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Disable commonly ignored Pyright linting rules'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3690-1", "source": {"repo": "encode/httpx", "pr": 3690, "sha": "ae25e86f5ca5625d5bffd13865b458ba5fadf50d", "url": "https://github.com/encode/httpx/pull/3690"}, "file": "src/ahttpx/_parsers.py", "start_line": 460, "code": " assert self._buffer == b''\n self._buffer = buffer\n\n async def wait_ready(self) -> bool:\n \"\"\"\n Attempt a read, and return True if read succeeds or False if the\n stream is closed. The data remains in the read buffer.\n \"\"\"\n data = await self._read_some()\n self._push_back(data)\n return data != b''\n\n async def read(self, size: int) -> bytes:\n \"\"\"\n Read and return up to 'size' bytes from the stream, with I/O buffering provided.", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Add `.wait_ready` to parser for clean server disconnects'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "click-3861-1", "source": {"repo": "pallets/click", "pr": 3861, "sha": "6d5bac5c712195cbfa9666ba0ecbf0573cdea356", "url": "https://github.com/pallets/click/pull/3861"}, "file": "src/click/core.py", "start_line": 2259, "code": " its deprecation in --help. The message can be customized\n by using a string as the value. A deprecated parameter\n cannot be required, a ValueError will be raised otherwise.\n :param help: the help string. It is dedented and get a deprecated label if\n appropriate.\n\n .. versionchanged:: 8.5.1\n New ``help`` parameter to replace the one from :class:`Option` and\n :class:`Argument`.\n\n .. versionchanged:: 8.2.0\n Introduction of ``deprecated``.", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Move common `help` argument from `Option` and `Argument` to `Parameter`'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5269-3", "source": {"repo": "urllib3/urllib3", "pr": 5269, "sha": "911bc94d227519483c82e516e2bf10860396afd5", "url": "https://github.com/urllib3/urllib3/pull/5269"}, "file": "src/urllib3/connection.py", "start_line": 119, "code": " Accepted parameters include:\n\n - ``source_address``: Set the source address for the current connection.\n - ``blocksize``: Read size for file-like request bodies, in bytes. Defaults to\n 16384 (16 KiB). Does not split or resize elements of iterable bodies.\n Larger values may improve upload throughput but use more memory per active\n upload. See :ref:`upload_buffering` for tuning guidance.\n - ``socket_options``: Set specific options on the underlying socket. If not specified, then\n defaults are loaded from ``HTTPConnection.default_socket_options`` which includes disabling\n Nagle's algorithm (sets TCP_NODELAY to 1) unless the connection is behind a proxy.", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Document upload buffering and `blocksize` tuning'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "requests-7496-1", "source": {"repo": "psf/requests", "pr": 7496, "sha": "e50e5945294f79ee9ab4ec69de24d14f9b26a7ae", "url": "https://github.com/psf/requests/pull/7496"}, "file": "src/requests/__init__.py", "start_line": 52, "code": " charset_normalizer_version = None\n\ntry:\n from chardet import __version__ as chardet_version\nexcept ImportError:\n chardet_version = None\n", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Improve static typing of 3rd party imports '); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "httpx-3673-0", "source": {"repo": "encode/httpx", "pr": 3673, "sha": "20380490fd0bd13c68704080754c1385bff96d69", "url": "https://github.com/encode/httpx/pull/3673"}, "file": "src/ahttpx/_parsers.py", "start_line": 375, "code": " self.recv_state = State.DONE\n return body\n\n async def reset(self) -> bool:\n is_fully_complete = self.send_state == State.DONE and self.recv_state == State.DONE\n is_keepalive = self.send_keep_alive and self.recv_keep_alive\n\n if not (is_fully_complete and is_keepalive):\n await self.close()\n return False\n\n if self.mode == Mode.CLIENT:\n self.send_state = State.SEND_METHOD_LINE", "label": "clean", "expect_rules": [], "note": "seed clean: post-change side of merged non-bugfix PR ('Connection resets'); provisional \u2014 not longitudinally checked for later fix commits"} +{"id": "urllib3-5029-old-0", "source": {"repo": "urllib3/urllib3", "pr": 5029, "sha": "c2a56d9e524344c715ecbb891fcd25f962a67405", "url": "https://github.com/urllib3/urllib3/pull/5029"}, "file": "src/urllib3/util/url.py", "start_line": 55, "code": "_REG_NAME_PAT = r\"(?:[^\\[\\]%:/?#]|%[a-fA-F0-9]{2})*\"\n_TARGET_RE = re.compile(r\"^(/[^?#]*)(?:\\?([^#]*))?(?:#.*)?$\")\n\n_IPV4_RE = re.compile(\"^\" + _IPV4_PAT + \"$\")\n_IPV6_RE = re.compile(\"^\" + _IPV6_PAT + \"$\")\n_IPV6_ADDRZ_RE = re.compile(\"^\" + _IPV6_ADDRZ_PAT + \"$\")\n_BRACELESS_IPV6_ADDRZ_RE = re.compile(\"^\" + _IPV6_ADDRZ_PAT[2:-2] + \"$\")", "label": "defective", "expect_rules": ["R7"], "note": "CVE-ish: is_ipaddress missed non-standard IPv4 forms (SSRF/bypass class) | PR title: Make `is_ipaddress` detect non-standard forms of IPv4 addresses"} +{"id": "urllib3-5029-old-1", "source": {"repo": "urllib3/urllib3", "pr": 5029, "sha": "c2a56d9e524344c715ecbb891fcd25f962a67405", "url": "https://github.com/urllib3/urllib3/pull/5029"}, "file": "test/test_ssl.py", "start_line": 23, "code": " \"127.0.0.1\",\n \"8.8.8.8\",\n b\"127.0.0.1\",\n # IPv6 w/ Zone IDs\n \"FE80::8939:7684:D84b:a5A4%251\",\n b\"FE80::8939:7684:D84b:a5A4%251\",", "label": "defective", "expect_rules": ["R7"], "note": "CVE-ish: is_ipaddress missed non-standard IPv4 forms (SSRF/bypass class) | PR title: Make `is_ipaddress` detect non-standard forms of IPv4 addresses"} +{"id": "urllib3-5255-old-0", "source": {"repo": "urllib3/urllib3", "pr": 5255, "sha": "0716e31534345dc1599ea95d903c79f276239bd8", "url": "https://github.com/urllib3/urllib3/pull/5255"}, "file": "src/urllib3/contrib/pyopenssl.py", "start_line": 512, "code": " # versions because set_passwd_cb() became deprecated in 26.3.0.\n if int(OpenSSL.__version__.split(\".\")[0]) >= 26:\n with open(keyfile or certfile, \"rb\") as key_file:\n private_key = load_pem_private_key(key_file.read(), password)\n # cryptography's loader returns a wider private-key union\n # than pyOpenSSL accepts, so we add `type: ignore` here.\n self._ctx.use_privatekey(private_key) # type: ignore[arg-type]", "label": "defective", "expect_rules": ["R4"], "note": "Loading unencrypted client keys with a password failed silently / mishandled | PR title: Fix loading unencrypted client keys with a password in pyOpenSSL"} +{"id": "urllib3-5255-old-1", "source": {"repo": "urllib3/urllib3", "pr": 5255, "sha": "0716e31534345dc1599ea95d903c79f276239bd8", "url": "https://github.com/urllib3/urllib3/pull/5255"}, "file": "test/with_dummyserver/test_socketlevel.py", "start_line": 438, "code": "\n assert len(client_certs) == 1\n\n def test_load_keyfile_with_invalid_password(self) -> None:\n assert ssl_.SSLContext is not None\n context = ssl_.SSLContext(ssl_.PROTOCOL_SSLv23)\n with pytest.raises(ssl.SSLError):\n context.load_cert_chain(\n certfile=self.cert_path,\n keyfile=self.password_key_path,\n password=b\"letmei\",\n )\n\n def test_load_invalid_cert_file(self) -> None:", "label": "defective", "expect_rules": ["R4"], "note": "Loading unencrypted client keys with a password failed silently / mishandled | PR title: Fix loading unencrypted client keys with a password in pyOpenSSL"} +{"id": "urllib3-5260-old-0", "source": {"repo": "urllib3/urllib3", "pr": 5260, "sha": "ed0ed075c6b93f7c515ebd3abe9a7248507ef8c5", "url": "https://github.com/urllib3/urllib3/pull/5260"}, "file": "src/urllib3/connectionpool.py", "start_line": 90, "code": " # to avoid removing square braces around IPv6 addresses.\n # This value is sent to `HTTPConnection.set_tunnel()` if called\n # because square braces are required for HTTP CONNECT tunneling.\n self._tunnel_host = normalize_host(host, scheme=self.scheme).lower()\n\n def __str__(self) -> str:\n return f\"{type(self).__name__}(host={self.host!r}, port={self.port!r})\"", "label": "defective", "expect_rules": ["R4"], "note": "Several IPv6 / zone-ID parsing failure modes | PR title: Fix several issues with IPv6 and IPv6 zone IDs"} +{"id": "urllib3-5260-old-1", "source": {"repo": "urllib3/urllib3", "pr": 5260, "sha": "ed0ed075c6b93f7c515ebd3abe9a7248507ef8c5", "url": "https://github.com/urllib3/urllib3/pull/5260"}, "file": "src/urllib3/connectionpool.py", "start_line": 708, "code": " redirect. Typically this won't need to be set because urllib3 will\n auto-populate the value when needed.\n \"\"\"\n # Ensure that the URL we're connecting to is properly encoded\n if url.startswith(\"/\"):\n # URLs starting with / are inherently schemeless.", "label": "defective", "expect_rules": ["R4"], "note": "Several IPv6 / zone-ID parsing failure modes | PR title: Fix several issues with IPv6 and IPv6 zone IDs"} +{"id": "requests-7308-old-0", "source": {"repo": "psf/requests", "pr": 7308, "sha": "bc7dd0fc4d56e808bcdd85ac2d797b3107c89259", "url": "https://github.com/psf/requests/pull/7308"}, "file": "src/requests/_internal_utils.py", "start_line": 10, "code": "\nfrom .compat import builtin_str\n\n_VALID_HEADER_NAME_RE_BYTE = re.compile(rb\"^[^:\\s][^:\\r\\n]*$\")\n_VALID_HEADER_NAME_RE_STR = re.compile(r\"^[^:\\s][^:\\r\\n]*$\")\n_VALID_HEADER_VALUE_RE_BYTE = re.compile(rb\"^\\S[^\\r\\n]*$|^$\")\n_VALID_HEADER_VALUE_RE_STR = re.compile(r\"^\\S[^\\r\\n]*$|^$\")\n\n_HEADER_VALIDATORS_STR = (_VALID_HEADER_NAME_RE_STR, _VALID_HEADER_VALUE_RE_STR)\n_HEADER_VALIDATORS_BYTE = (_VALID_HEADER_NAME_RE_BYTE, _VALID_HEADER_VALUE_RE_BYTE)", "label": "defective", "expect_rules": ["R4"], "note": "Cosmetic but real: header validity regex accepted invalid forms | PR title: Fix cosmetic header validity parsing regex"} +{"id": "requests-7308-old-1", "source": {"repo": "psf/requests", "pr": 7308, "sha": "bc7dd0fc4d56e808bcdd85ac2d797b3107c89259", "url": "https://github.com/psf/requests/pull/7308"}, "file": "tests/test_requests.py", "start_line": 1791, "code": " {\"fo\\r\\no\": \"bar\"},\n {\"fo\\n\\ro\": \"bar\"},\n {\"fo\\no\": \"bar\"},\n ),\n )\n def test_header_no_return_chars(self, httpbin, invalid_header):", "label": "defective", "expect_rules": ["R4"], "note": "Cosmetic but real: header validity regex accepted invalid forms | PR title: Fix cosmetic header validity parsing regex"} +{"id": "httpx-3442-old-0", "source": {"repo": "encode/httpx", "pr": 3442, "sha": "89599a9541af14bcf906fc4ed58ccbdf403802ba", "url": "https://github.com/encode/httpx/pull/3442"}, "file": "httpx/_config.py", "start_line": 39, "code": " # Default case...\n ctx = ssl.create_default_context(cafile=certifi.where())\n elif verify is False:\n ssl_context = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)\n ssl_context.check_hostname = False\n ssl_context.verify_mode = ssl.CERT_NONE\n return ssl_context\n elif isinstance(verify, str): # pragma: nocover\n message = (\n \"`verify=` is deprecated. \"", "label": "defective", "expect_rules": ["R7"], "note": "verify=False with cert= broke TLS posture / trust boundary | PR title: Fix `verify=False`, `cert=...` case."} +{"id": "requests-7310-old-0", "source": {"repo": "psf/requests", "pr": 7310, "sha": "a044b020dea43230585126901684a0f30ec635a8", "url": "https://github.com/psf/requests/pull/7310"}, "file": "src/requests/auth.py", "start_line": 145, "code": " def md5_utf8(x):\n if isinstance(x, str):\n x = x.encode(\"utf-8\")\n return hashlib.md5(x).hexdigest()\n\n hash_utf8 = md5_utf8\n elif _algorithm == \"SHA\":", "label": "defective", "expect_rules": ["R7"], "note": "DigestAuth hashes used without usedforsecurity=False (crypto misuse) | PR title: Move DigestAuth hash algorithms to use usedforsecurity=False"} +{"id": "requests-7310-old-1", "source": {"repo": "psf/requests", "pr": 7310, "sha": "a044b020dea43230585126901684a0f30ec635a8", "url": "https://github.com/psf/requests/pull/7310"}, "file": "src/requests/auth.py", "start_line": 153, "code": " def sha_utf8(x):\n if isinstance(x, str):\n x = x.encode(\"utf-8\")\n return hashlib.sha1(x).hexdigest()\n\n hash_utf8 = sha_utf8\n elif _algorithm == \"SHA-256\":", "label": "defective", "expect_rules": ["R7"], "note": "DigestAuth hashes used without usedforsecurity=False (crypto misuse) | PR title: Move DigestAuth hash algorithms to use usedforsecurity=False"} +{"id": "httpx-3380-old-0", "source": {"repo": "encode/httpx", "pr": 3380, "sha": "a33c87852b8a0dddc65e5f739af1e0a6fca4b91f", "url": "https://github.com/encode/httpx/pull/3380"}, "file": "httpx/_models.py", "start_line": 398, "code": " self.method = method.upper()\n self.url = URL(url) if params is None else URL(url, params=params)\n self.headers = Headers(headers)\n self.extensions = {} if extensions is None else extensions\n\n if cookies:\n Cookies(cookies).set_cookie_header(self)", "label": "defective", "expect_rules": ["R3"], "note": "Wrong extensions type annotation \u2014 contract lie at the boundary | PR title: Fix `extensions` type annotation."} +{"id": "httpx-3380-old-1", "source": {"repo": "encode/httpx", "pr": 3380, "sha": "a33c87852b8a0dddc65e5f739af1e0a6fca4b91f", "url": "https://github.com/encode/httpx/pull/3380"}, "file": "httpx/_models.py", "start_line": 537, "code": " # the client will set `response.next_request`.\n self.next_request: Request | None = None\n\n self.extensions: ResponseExtensions = {} if extensions is None else extensions\n self.history = [] if history is None else list(history)\n\n self.is_closed = False", "label": "defective", "expect_rules": ["R3"], "note": "Wrong extensions type annotation \u2014 contract lie at the boundary | PR title: Fix `extensions` type annotation."} +{"id": "urllib3-5271-old-0", "source": {"repo": "urllib3/urllib3", "pr": 5271, "sha": "48113e7a5fa3b98a61467afa6bcbb0f9b7ca7207", "url": "https://github.com/urllib3/urllib3/pull/5271"}, "file": "src/urllib3/response.py", "start_line": 205, "code": " return bool(self._unconsumed_tail)\n\n def flush(self) -> bytes:\n return self._obj.flush()\n\n", "label": "defective", "expect_rules": ["R4"], "note": "Swallowed gzip trailing garbage not flushed \u2014 incomplete decode edge | PR title: Avoid flushing swallowed gzip trailing garbage"} +{"id": "urllib3-5271-old-1", "source": {"repo": "urllib3/urllib3", "pr": 5271, "sha": "48113e7a5fa3b98a61467afa6bcbb0f9b7ca7207", "url": "https://github.com/urllib3/urllib3/pull/5271"}, "file": "test/test_response.py", "start_line": 30, "code": " _MAX_CHUNK_LINE_LENGTH,\n BaseHTTPResponse,\n BytesQueueBuffer,\n HTTPResponse,\n brotli,\n)", "label": "defective", "expect_rules": ["R4"], "note": "Swallowed gzip trailing garbage not flushed \u2014 incomplete decode edge | PR title: Avoid flushing swallowed gzip trailing garbage"} diff --git a/docs/eval-runs/corpus-prove.txt b/docs/eval-runs/corpus-prove.txt new file mode 100644 index 0000000..45689a6 --- /dev/null +++ b/docs/eval-runs/corpus-prove.txt @@ -0,0 +1,19 @@ +lazycoder Stage 2 — corpus prove (offline) +date: 2026-09-29 14:25 CEST (PT) +corpus: corpus/seed.jsonl + +Pre-registered thresholds (hardening-plan.md, 2026-08-26): + delete if clean_noise > 25% and caught == 0 + demote if clean_noise > 10% and severity high + stage3a if abstention > 50% and clean_noise < 10% + fail-on ready if quiet_rate ≥ 80% and no high rule > 5% clean noise + shape: ≥20 clean, ≥15 defective + +shape: 22 clean (≥20), 15 defective (≥15) — shape OK +repos: encode/httpx, pallets/click, psf/requests, urllib3/urllib3 +hunks: 37 +cull self-check: PASS (6/6 decision-table cases) +live scoring: BLOCKED — ANTHROPIC_API_KEY unset (no credits/key in this environment; offline half only) +fail-on: stays never until live corpus metrics satisfy the gate (no live scores in this prove run — gate not claimed ready) + +OFFLINE_PROVE: PASS LIVE_SCORING: BLOCKED FAIL_ON: never diff --git a/docs/hardening-plan.md b/docs/hardening-plan.md index e551e96..e73dd74 100644 --- a/docs/hardening-plan.md +++ b/docs/hardening-plan.md @@ -52,6 +52,13 @@ decision log fully determine the verdict. ## Stage 2 — the number +**Phase 3 (offline) landed:** `corpus/seed.jsonl` meets the shape floor +(≥20 clean / ≥15 defective); `decide_cull` / `fail_on_gate` encode the +pre-registered thresholds; `scripts/corpus_cli.py prove` is the offline +proof. Live `run` + rule cull from real scores remain gated on Anthropic +credits — do not invent precision numbers and do not flip `fail-on`. + + Corpus of 30–50 hunks from merged OSS PRs: half with a defect the follow-up fix confirms, half genuinely clean. Not toy snippets. diff --git a/scripts/corpus_cli.py b/scripts/corpus_cli.py index 618cc2b..8ee3a0b 100644 --- a/scripts/corpus_cli.py +++ b/scripts/corpus_cli.py @@ -1,4 +1,4 @@ -"""Harvest and score a corpus of real diff hunks. +"""Harvest, prove, and score a corpus of real diff hunks. # pull candidate hunks from merged PRs (needs gh) python scripts/corpus_cli.py harvest --repo psf/requests --prs 8 \ @@ -6,34 +6,55 @@ # label each line by hand: "label": "clean" | "defective" (+ expect_rules) - # score the labelled corpus against the live model - ANTHROPIC_API_KEY=... python scripts/corpus_cli.py run corpus/corpus.jsonl + # offline Stage 2 proof (no Anthropic key): shape + cull thresholds + python scripts/corpus_cli.py prove corpus/seed.jsonl \ + --out docs/eval-runs/corpus-prove.txt + + # score the labelled corpus against the live model (needs ANTHROPIC_API_KEY) + ANTHROPIC_API_KEY=... python scripts/corpus_cli.py run corpus/seed.jsonl Harvest emits the post-change side of each hunk, which is what a *clean* -sample is. Defective samples come from bug-fix PRs and are labelled by hand -for now; a --side old extractor is only worth building if that becomes the -bottleneck. +sample is. Defective samples come from bug-fix PRs (pre-image) and are +labelled by hand. """ from __future__ import annotations import argparse import json +import os import subprocess import sys +from datetime import datetime from pathlib import Path +from zoneinfo import ZoneInfo sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src")) from lazycoder.config import load_all_configs # noqa: E402 from lazycoder.corpus import ( # noqa: E402 + DELETE_NOISE_RATE, + DEMOTE_HIGH_NOISE_RATE, + FAIL_ON_MAX_HIGH_NOISE_RATE, + FAIL_ON_MIN_QUIET_RATE, + MIN_CLEAN_HUNKS, + MIN_DEFECTIVE_HUNKS, + STAGE3A_ABSTENTION_RATE, + STAGE3A_MAX_NOISE_RATE, CorpusLoadError, + CullAction, HunkResult, Label, + RuleVerdict, + corpus_shape, + cull_plan, + decide_cull, + fail_on_gate, load_corpus, report, score_hunk, ) +from lazycoder.domain import RuleId, Severity # noqa: E402 from lazycoder.llm.anthropic_client import AnthropicClient # noqa: E402 from lazycoder.orchestrator import parse_diff # noqa: E402 from lazycoder.reviewers import SingleRuleReviewer # noqa: E402 @@ -43,9 +64,8 @@ # Most recently merged PRs on a popular repo are dependency bumps. Scanning # them finds nothing and makes an empty harvest look like a broken one. BOT_AUTHORS = {"dependabot", "pre-commit-ci", "renovate", "github-actions"} -# Most recently merged PRs on a popular repo are dependency bumps. Scanning -# them finds nothing and makes an empty harvest look like a broken one. -BOT_AUTHORS = {"dependabot", "pre-commit-ci", "renovate", "github-actions"} + +USER_TZ = ZoneInfo("Europe/Prague") def _gh(*args: str) -> str: @@ -159,6 +179,14 @@ def _ids(rules: frozenset) -> str: def run(path: Path) -> int: + if not os.environ.get("ANTHROPIC_API_KEY"): + print( + "error: ANTHROPIC_API_KEY unset — live corpus scoring is blocked.\n" + " Offline half: python scripts/corpus_cli.py prove " + str(path), + file=sys.stderr, + ) + return 2 + config = load_all_configs() rubric = config.review_rules hunks = load_corpus(path) @@ -185,13 +213,143 @@ def run(path: Path) -> int: f"recall {summary.recall:.0%}" f" ({summary.caught} caught, {summary.missed} missed)" ) - print(f"\n{'rule':6}{'noise%':>8}{'caught':>8}{'missed':>8}{'abstain':>9}") + print( + f"\n{'rule':6}{'noise%':>8}{'caught':>8}{'missed':>8}{'abstain':>9}{'cull':>14}" + ) + plan = cull_plan(summary, rubric) for rule_id, verdict in summary.per_rule.items(): print( f"{rule_id.value:6}{verdict.clean_noise_rate:>7.0%}" f"{verdict.caught:>8}{verdict.missed:>8}" f"{verdict.abstention_rate:>8.0%}" + f"{plan[rule_id].value:>14}" + ) + gate = fail_on_gate(summary, rubric) + print(f"\nfail-on gate ready: {gate.ready} ({gate.reason})") + if not gate.ready: + print("Action fail-on stays never — metrics do not justify a blocking gate.") + return 0 + + +def _cull_self_check() -> list[str]: + """Measurable offline proof that the encoded thresholds match the plan table.""" + cases: list[tuple[str, RuleVerdict, Severity, CullAction]] = [ + ( + "delete: noise>25% and never right", + RuleVerdict(RuleId.R12, 20, 6, 0, 0, 0, 0, 20), + Severity.MEDIUM, + CullAction.DELETE, + ), + ( + "demote: high rule noise>10%", + RuleVerdict(RuleId.R4, 20, 3, 1, 0, 0, 0, 20), + Severity.HIGH, + CullAction.DEMOTE, + ), + ( + "stage3a: abstains a lot, quiet on clean", + RuleVerdict(RuleId.R14, 20, 1, 0, 0, 0, 12, 20), + Severity.MEDIUM, + CullAction.KEEP_STAGE_3A, + ), + ( + "keep: modest noise, sometimes right", + RuleVerdict(RuleId.R7, 20, 2, 3, 1, 0, 2, 20), + Severity.HIGH, + CullAction.KEEP, + ), + ( + "delete beats demote when never right", + RuleVerdict(RuleId.R4, 20, 6, 0, 0, 0, 0, 20), + Severity.HIGH, + CullAction.DELETE, + ), + ( + "keep: noisy medium with catches (not delete, not demote)", + RuleVerdict(RuleId.R3, 20, 6, 2, 0, 0, 0, 20), + Severity.MEDIUM, + CullAction.KEEP, + ), + ] + failures: list[str] = [] + for name, verdict, severity, expected in cases: + got = decide_cull(verdict, severity) + if got is not expected: + failures.append(f"{name}: expected {expected.value}, got {got.value}") + return failures + + +def prove(path: Path, out: Path | None) -> int: + """Offline Stage 2 proof: labelled corpus shape + cull threshold self-check. + + Does not call the model. Live scoring remains a separate `run` step. + """ + now = datetime.now(tz=USER_TZ) + lines: list[str] = [ + "lazycoder Stage 2 — corpus prove (offline)", + f"date: {now.strftime('%Y-%m-%d %H:%M %Z')} (PT)", + f"corpus: {path}", + "", + "Pre-registered thresholds (hardening-plan.md, 2026-08-26):", + f" delete if clean_noise > {DELETE_NOISE_RATE:.0%} and caught == 0", + f" demote if clean_noise > {DEMOTE_HIGH_NOISE_RATE:.0%} and severity high", + f" stage3a if abstention > {STAGE3A_ABSTENTION_RATE:.0%}" + f" and clean_noise < {STAGE3A_MAX_NOISE_RATE:.0%}", + f" fail-on ready if quiet_rate ≥ {FAIL_ON_MIN_QUIET_RATE:.0%}" + f" and no high rule > {FAIL_ON_MAX_HIGH_NOISE_RATE:.0%} clean noise", + f" shape: ≥{MIN_CLEAN_HUNKS} clean, ≥{MIN_DEFECTIVE_HUNKS} defective", + "", + ] + + hunks = load_corpus(path) + shape = corpus_shape(hunks) + lines.append(f"shape: {shape.summary}") + repos = sorted({h.source.repo for h in hunks}) + lines.append(f"repos: {', '.join(repos)}") + lines.append(f"hunks: {len(hunks)}") + + cull_failures = _cull_self_check() + if cull_failures: + lines.append("cull self-check: FAIL") + lines.extend(f" - {f}" for f in cull_failures) + else: + lines.append("cull self-check: PASS (6/6 decision-table cases)") + + key_present = bool(os.environ.get("ANTHROPIC_API_KEY")) + if key_present: + lines.append( + "live scoring: API key present — run `corpus_cli.py run` separately" + ) + live_status = "KEY_PRESENT" + else: + lines.append( + "live scoring: BLOCKED — ANTHROPIC_API_KEY unset " + "(no credits/key in this environment; offline half only)" ) + live_status = "BLOCKED" + + lines.append( + "fail-on: stays never until live corpus metrics satisfy the gate " + "(no live scores in this prove run — gate not claimed ready)" + ) + lines.append("") + + offline_ok = shape.ok and not cull_failures + lines.append( + f"OFFLINE_PROVE: {'PASS' if offline_ok else 'FAIL'} " + f"LIVE_SCORING: {live_status} FAIL_ON: never" + ) + + text = "\n".join(lines) + "\n" + sys.stdout.write(text) + + if out is not None: + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(text, encoding="utf-8") + print(f"artifact: {out}", file=sys.stderr) + + if not offline_ok: + return 1 return 0 @@ -209,10 +367,30 @@ def main() -> int: r = sub.add_parser("run", help="score a labelled corpus against the live model") r.add_argument("path", type=Path) + p = sub.add_parser( + "prove", + help="offline Stage 2 proof: corpus shape + cull threshold self-check", + ) + p.add_argument("path", type=Path, help="labelled corpus JSONL") + p.add_argument( + "--out", + type=Path, + default=Path("docs/eval-runs/corpus-prove.txt"), + help="where to write the prove artifact", + ) + p.add_argument( + "--no-artifact", + action="store_true", + help="print only; do not write --out", + ) + args = parser.parse_args() try: if args.command == "harvest": return harvest(args.repo, args.prs, args.per_pr, args.out, bots=args.bots) + if args.command == "prove": + out = None if args.no_artifact else args.out + return prove(args.path, out) return run(args.path) except (CorpusLoadError, RuntimeError, OSError) as exc: print(f"error: {exc}", file=sys.stderr) diff --git a/src/lazycoder/corpus.py b/src/lazycoder/corpus.py index 2e2dcca..e51bfc7 100644 --- a/src/lazycoder/corpus.py +++ b/src/lazycoder/corpus.py @@ -21,7 +21,7 @@ from pydantic import BaseModel, ConfigDict, Field, ValidationError from lazycoder.config.models import ReviewRulesConfig -from lazycoder.domain import CodeBlock, RuleId, Verdict +from lazycoder.domain import CodeBlock, RuleId, Severity, Verdict from lazycoder.reviewers import SingleRuleReviewer @@ -249,3 +249,130 @@ def report(results: list[HunkResult], rubric: ReviewRulesConfig) -> CorpusReport missed=sum(missed.values()), per_rule=per_rule, ) + + +# --------------------------------------------------------------------------- +# Stage 2 pre-registered cull thresholds (docs/hardening-plan.md, 2026-08-26). +# Encoded here so the cull is a measurement against fixed numbers, not taste. +# --------------------------------------------------------------------------- + +MIN_CLEAN_HUNKS = 20 +MIN_DEFECTIVE_HUNKS = 15 + +# Per-rule cull +DELETE_NOISE_RATE = 0.25 # and caught == 0 → delete +DEMOTE_HIGH_NOISE_RATE = 0.10 # and severity high → demote to medium +STAGE3A_ABSTENTION_RATE = 0.50 # and clean_noise < 10% → keep, route to 3a +STAGE3A_MAX_NOISE_RATE = 0.10 + +# Action fail-on gate (do not flip fail-on away from never until both hold) +FAIL_ON_MIN_QUIET_RATE = 0.80 +FAIL_ON_MAX_HIGH_NOISE_RATE = 0.05 + + +class CullAction(StrEnum): + """What the pre-registered thresholds say to do with one rule.""" + + DELETE = "delete" + DEMOTE = "demote" + KEEP_STAGE_3A = "keep_stage_3a" + KEEP = "keep" + + +@dataclass(frozen=True) +class CorpusShape: + clean: int + defective: int + + @property + def ok(self) -> bool: + return self.clean >= MIN_CLEAN_HUNKS and self.defective >= MIN_DEFECTIVE_HUNKS + + @property + def summary(self) -> str: + return ( + f"{self.clean} clean (≥{MIN_CLEAN_HUNKS}), " + f"{self.defective} defective (≥{MIN_DEFECTIVE_HUNKS})" + + (" — shape OK" if self.ok else " — shape SHORT") + ) + + +def corpus_shape(hunks: list[CorpusHunk]) -> CorpusShape: + return CorpusShape( + clean=sum(1 for h in hunks if h.label is Label.CLEAN), + defective=sum(1 for h in hunks if h.label is Label.DEFECTIVE), + ) + + +def decide_cull(verdict: RuleVerdict, severity: Severity) -> CullAction: + """Apply the Stage 2 pre-registered decision table to one rule's numbers. + + Order matches the table in docs/hardening-plan.md: delete first, then + demote, then route-to-3a, else keep. Changing a threshold requires a + commit that says so — do not tweak while staring at live results. + """ + if verdict.clean_noise_rate > DELETE_NOISE_RATE and verdict.caught == 0: + return CullAction.DELETE + if verdict.clean_noise_rate > DEMOTE_HIGH_NOISE_RATE and severity is Severity.HIGH: + return CullAction.DEMOTE + if ( + verdict.abstention_rate > STAGE3A_ABSTENTION_RATE + and verdict.clean_noise_rate < STAGE3A_MAX_NOISE_RATE + ): + return CullAction.KEEP_STAGE_3A + return CullAction.KEEP + + +@dataclass(frozen=True) +class FailOnGate: + """Whether metrics justify flipping Action fail-on off `never`.""" + + quiet_rate: float + high_rules_over_noise_cap: tuple[RuleId, ...] + ready: bool + reason: str + + +def fail_on_gate(summary: CorpusReport, rubric: ReviewRulesConfig) -> FailOnGate: + """Gate from the hardening plan: quiet_rate ≥ 80% and no high rule > 5% noise.""" + high_noisy = tuple( + rule.id + for rule in rubric.rules + if rule.severity_if_unjustified is Severity.HIGH + and summary.per_rule[rule.id].clean_noise_rate > FAIL_ON_MAX_HIGH_NOISE_RATE + ) + quiet_ok = summary.quiet_rate >= FAIL_ON_MIN_QUIET_RATE + ready = quiet_ok and not high_noisy + if ready: + reason = ( + f"quiet_rate {summary.quiet_rate:.0%} ≥ {FAIL_ON_MIN_QUIET_RATE:.0%} " + f"and no high rule above {FAIL_ON_MAX_HIGH_NOISE_RATE:.0%} clean noise" + ) + else: + parts: list[str] = [] + if not quiet_ok: + parts.append( + f"quiet_rate {summary.quiet_rate:.0%} < {FAIL_ON_MIN_QUIET_RATE:.0%}" + ) + if high_noisy: + parts.append( + "high-noise " + + ",".join(r.value for r in high_noisy) + + f" > {FAIL_ON_MAX_HIGH_NOISE_RATE:.0%}" + ) + reason = "; ".join(parts) + return FailOnGate( + quiet_rate=summary.quiet_rate, + high_rules_over_noise_cap=high_noisy, + ready=ready, + reason=reason, + ) + + +def cull_plan( + summary: CorpusReport, rubric: ReviewRulesConfig +) -> dict[RuleId, CullAction]: + return { + rule.id: decide_cull(summary.per_rule[rule.id], rule.severity_if_unjustified) + for rule in rubric.rules + } diff --git a/tests/test_corpus.py b/tests/test_corpus.py index 349a277..3220ed2 100644 --- a/tests/test_corpus.py +++ b/tests/test_corpus.py @@ -1,5 +1,7 @@ from __future__ import annotations +from pathlib import Path + import pytest from pydantic import ValidationError @@ -204,3 +206,92 @@ def test_load_corpus_keeps_code_verbatim_and_skips_blank_lines(tmp_path) -> None assert len(hunks) == 1 assert hunks[0].code == INDENTED assert hunks[0].source.repo == "acme/widgets" + + +def test_decide_cull_matches_preregistered_table() -> None: + from lazycoder.corpus import CullAction, RuleVerdict, decide_cull + from lazycoder.domain import Severity + + # delete: interrupts constantly, never been right + assert ( + decide_cull(RuleVerdict(RuleId.R12, 20, 6, 0, 0, 0, 0, 20), Severity.MEDIUM) + is CullAction.DELETE + ) + # demote: high severity whose noise must not block + assert ( + decide_cull(RuleVerdict(RuleId.R4, 20, 3, 1, 0, 0, 0, 20), Severity.HIGH) + is CullAction.DEMOTE + ) + # stage 3a: silent is not the same failure as noisy + assert ( + decide_cull(RuleVerdict(RuleId.R14, 20, 1, 0, 0, 0, 12, 20), Severity.MEDIUM) + is CullAction.KEEP_STAGE_3A + ) + # keep + assert ( + decide_cull(RuleVerdict(RuleId.R7, 20, 2, 3, 1, 0, 2, 20), Severity.HIGH) + is CullAction.KEEP + ) + # delete wins over demote when caught == 0 + assert ( + decide_cull(RuleVerdict(RuleId.R4, 20, 6, 0, 0, 0, 0, 20), Severity.HIGH) + is CullAction.DELETE + ) + + +def test_fail_on_gate_requires_quiet_rate_and_calm_high_rules( + rubric: ReviewRulesConfig, +) -> None: + from lazycoder.corpus import ( + HunkResult, + Label, + fail_on_gate, + report, + ) + + # Three quiet clean hunks → quiet_rate 100%, no high noise → ready + quiet = HunkResult( + "C1", + Label.CLEAN, + frozenset(), + frozenset(), + frozenset(), + Verdict.APPROVE, + frozenset(), + ) + summary = report([quiet, quiet, quiet], rubric) + gate = fail_on_gate(summary, rubric) + assert gate.ready is True + + # One high-severity finding on clean collapses the gate + noisy = HunkResult( + "C2", + Label.CLEAN, + frozenset(), + frozenset({RuleId.R4}), + frozenset(), + Verdict.BLOCK, + frozenset(), + ) + summary = report([quiet, quiet, noisy], rubric) + gate = fail_on_gate(summary, rubric) + assert gate.ready is False + assert RuleId.R4 in gate.high_rules_over_noise_cap + + +def test_corpus_shape_targets_stage2_minimums(tmp_path) -> None: + from lazycoder.corpus import ( + MIN_CLEAN_HUNKS, + MIN_DEFECTIVE_HUNKS, + corpus_shape, + load_corpus, + ) + + # The committed seed must meet the pre-registered shape, or prove fails. + seed = Path(__file__).resolve().parent.parent / "corpus" / "seed.jsonl" + if not seed.exists(): + pytest.skip("corpus/seed.jsonl not present") + shape = corpus_shape(load_corpus(seed)) + assert shape.clean >= MIN_CLEAN_HUNKS + assert shape.defective >= MIN_DEFECTIVE_HUNKS + assert shape.ok is True