Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 30 additions & 0 deletions loopx/capabilities/pr_review_queue/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -299,6 +299,36 @@ scan, so the latency improvement does not change queue ordering or freshness
semantics. The scan does not use a stale cache: rerunning the command always
re-reads the requested GitHub window.

When GitHub's paginated file API is capped at 3000 entries, a PR declaring
more files can recover its inventory from already available exact Git objects
in the caller's checkout. The `origin` must identify the requested GitHub
repository. LoopX compares the unique merge base to the exact head, folds only
API-confirmed rename pairs, and requires the resulting file count and whole-diff
addition/deletion totals to match fresh GitHub metadata. Head, base and totals
are fenced before pagination and after recovery. This is source completeness,
not review evidence or approval of the recovered PR.

The source provider's recovered file rows identify `source` as `github` or
`git`. Known API per-file statistics are retained, including 0/0 for omitted
or generated diffs. Unobserved Git binary rows retain unknown per-file counts
(`null`); Git's textual totals exclude binary lines. Mixed display values need
not sum to the independently verified whole-diff totals. Ordinary complete GraphQL/REST reads keep their
existing shape. No object fetch, clone, checkout, external diff or textconv is
performed. A wrong repository, missing objects, ambiguous rename or
malformed statistics, total mismatch or remote version change remains an
incomplete source. Callers without a versioned snapshot (including approval
closeout) retain their complete-API requirement; inventory recovery grants no
additional closeout, publication or merge authority.

GitHub 文件 API 达到 3000 项上限时,调用方当前 checkout 中已有的精确 Git 对象
可用于恢复更大 PR 的文件清单。必须核对 origin 仓库、唯一 merge base 与精确 head,
仅折叠 API 明确确认的 rename,并让文件数和全 diff 增删量与远端 metadata 一致。
分页前及恢复后校验 head、base 和总量。保留 API 已知行的统计值;恢复行的 Git 来源
不等于已完成 review。Git 二进制行不贡献文本行数,未被 API 观测的逐行统计保留 null,
不会假定为 API 的 0/0。不会自动抓取对象、切分支或执行外部 diff/textconv;缺对象、
统计格式损坏、rename 歧义、总量不符或版本变化继续明确返回 incomplete。没有精确快照
的 approval closeout 仍要求完整 API 读回,不扩大撤回 review 或合并权限。

For an autonomous maintainer monitor, request the complete open queue while
persisting its compact cursor in an ignored local checkpoint:

Expand Down
163 changes: 138 additions & 25 deletions loopx/capabilities/pr_review_queue/github_source.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,8 @@
from __future__ import annotations

import json
import os
import re
import subprocess
from concurrent.futures import ThreadPoolExecutor
from collections.abc import Sequence
Expand Down Expand Up @@ -62,52 +64,162 @@ def run_gh_json(args: list[str], *, cwd: Path | None = None) -> Any:
return json.loads(proc.stdout or "null")


def _read_git(args: list[str], *, cwd: Path) -> bytes:
# Local objects only. Lazy fetching, external diff and textconv must not turn
# a read-only inventory into provider execution or a network operation.
return subprocess.run(
["git", "--no-replace-objects", *args], cwd=cwd, check=True,
env={**os.environ, "GIT_NO_LAZY_FETCH": "1", "GIT_OPTIONAL_LOCKS": "0"},
stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=30,
).stdout


def _recover_git_pr_files(
*, repository: str, snapshot: dict[str, Any], observed: list[dict[str, Any]],
cwd: Path | None,
) -> list[dict[str, Any]] | None:
"""Reconcile local immutable objects with API-confirmed rename pairs."""
cwd = cwd if cwd is not None else Path.cwd()
head, base = snapshot.get("headRefOid"), snapshot.get("baseRefOid")
if any(not isinstance(oid, str) or not re.fullmatch(r"[0-9a-f]{40}", oid)
for oid in (head, base)):
return None
try:
remote = _read_git(["remote", "get-url", "origin"], cwd=cwd).decode().strip()
match = re.fullmatch(
r"(?:https://github\.com/|ssh://git@github\.com/|git@github\.com:)"
r"([^/]+/[^/]+?)(?:\.git)?/?", remote,
)
if not match or match[1].casefold() != repository.casefold():
return None
for oid in (head, base):
_read_git(["cat-file", "-e", f"{oid}^{{commit}}"], cwd=cwd)
bases = _read_git(["merge-base", "--all", base, head], cwd=cwd).splitlines()
if len(bases) != 1:
return None
merge_base = bases[0].decode("ascii")
args = ["diff", "--no-ext-diff", "--no-textconv", "--no-renames"]
stats = _read_git([*args, "--numstat", "-z", merge_base, head, "--"], cwd=cwd)
names = _read_git([*args, "--name-status", "-z", merge_base, head, "--"], cwd=cwd)
if not stats.endswith(b"\0") or not names.endswith(b"\0"):
return None
inventory: dict[str, dict[str, Any]] = {}
for entry in stats[:-1].split(b"\0"):
added, deleted, raw_path = entry.split(b"\t", 2)
binary = added == deleted == b"-"
if not binary and (not added.isdigit() or not deleted.isdigit()):
return None
path = raw_path.decode("utf-8", errors="strict")
if not path or path in inventory:
return None
inventory[path] = {"path": path, "additions": None if binary else int(added),
"deletions": None if binary else int(deleted), "source": "git"}
tokens = names[:-1].split(b"\0")
if len(tokens) % 2:
return None
statuses = {tokens[i + 1].decode("utf-8", errors="strict"):
tokens[i].decode("ascii") for i in range(0, len(tokens), 2)}
if len(statuses) != len(inventory) or statuses.keys() != inventory.keys():
return None
api_paths: set[str] = set()
rename_endpoints: set[str] = set()
for item in observed:
path = item["path"]
if path in api_paths:
return None
api_paths.add(path)
if item.get("status") != "renamed":
continue
old = item.get("previous_filename")
if (not isinstance(old, str) or old == path
or {old, path} & rename_endpoints
or statuses.get(old) != "D" or statuses.get(path) != "A"):
return None
rename_endpoints.update((old, path))
del inventory[old]
inventory[path] = {"path": path, "additions": item["additions"],
"deletions": item["deletions"], "source": "github"}
if not api_paths <= inventory.keys():
return None
# Verify Git whole-diff totals with only API-confirmed rename folding.
# Ordinary API rows may report 0/0 for generated/omitted diffs: replacing
# their Git statistics before this check would conceal disagreements.
# Git's valid -/- binary rows contribute no textual lines; retain None
# for unknown per-file counts, rather than inventing API 0/0 values.
if (len(inventory) != snapshot["changedFiles"]
or sum(row["additions"] or 0 for row in inventory.values()) != snapshot["additions"]
or sum(row["deletions"] or 0 for row in inventory.values()) != snapshot["deletions"]):
return None
for item in observed:
inventory[item["path"]] = {key: item[key] for key in
("path", "additions", "deletions")} | {"source": "github"}
return [inventory[path] for path in sorted(inventory)]
except (OSError, subprocess.SubprocessError, UnicodeError, ValueError, KeyError):
return None


def _fetch_complete_pr_files(
*,
repository: str,
number: str,
expected_count: int,
cwd: Path | None,
run_gh_json: GitHubJsonRunner,
snapshot: dict[str, Any] | None = None,
) -> list[dict[str, Any]] | None:
# The ordinary REST path and closeout callers retain their existing contract.
# Only a declared API-cap-sized PR with a versioned snapshot may use Git.
recover = snapshot is not None and expected_count > 3000
version_fields = ("headRefOid", "baseRefOid", "changedFiles", "additions", "deletions")

def same_snapshot() -> bool:
current = run_gh_json(["pr", "view", number, "--json", ",".join(version_fields),
"--repo", repository], cwd=cwd)
return isinstance(current, dict) and all(
current.get(key) == snapshot.get(key) for key in version_fields
)

try:
if recover and (any(type(snapshot.get(key)) is not int or snapshot[key] < 0
for key in version_fields[2:]) or not same_snapshot()):
return None
payload = run_gh_json(
[
"api",
"--paginate",
"--slurp",
f"repos/{repository}/pulls/{number}/files?per_page=100",
],
cwd=cwd,
["api", "--paginate", "--slurp",
f"repos/{repository}/pulls/{number}/files?per_page=100"], cwd=cwd,
)
except Exception:
return None
if not isinstance(payload, list):
return None
pages = payload if all(isinstance(page, list) for page in payload) else [payload]
files: list[dict[str, Any]] = []
try:
if not isinstance(payload, list):
return None
pages = payload if all(isinstance(page, list) for page in payload) else [payload]
files: list[dict[str, Any]] = []
observed: list[dict[str, Any]] = []
for page in pages:
for item in page:
if not isinstance(item, dict):
return None
path = str(item.get("filename") or item.get("path") or "").strip()
if not path:
path = item.get("filename") or item.get("path")
if not isinstance(path, str) or not path:
return None
if recover and any(type(item.get(key)) is not int for key in ("additions", "deletions")):
return None
additions = int(item.get("additions") or 0)
deletions = int(item.get("deletions") or 0)
if additions < 0 or deletions < 0:
return None
files.append(
{
"path": path,
"additions": additions,
"deletions": deletions,
}
)
except (TypeError, ValueError):
row = {"path": path, "additions": additions, "deletions": deletions}
files.append(row)
observed.append(row | {"status": item.get("status"),
"previous_filename": item.get("previous_filename")})
if len(files) == expected_count:
return files if not recover or same_snapshot() else None
if not recover or not files or len(files) > expected_count:
return None
recovered = _recover_git_pr_files(
repository=repository, snapshot=snapshot, observed=observed, cwd=cwd,
)
return recovered if recovered is not None and same_snapshot() else None
except Exception:
# This provider's failed read remains an explicit incomplete source.
return None
return files if len(files) == expected_count else None


def attach_pr_review_details(
Expand Down Expand Up @@ -163,6 +275,7 @@ def attach_pr_review_details(
repository=repository,
number=number,
expected_count=expected_file_count,
snapshot=row,
cwd=cwd,
run_gh_json=run_gh_json,
)
Expand Down
Loading
Loading