Skip to content

Commit fb305bf

Browse files
fix: harden evidence provenance
1 parent 1c068fe commit fb305bf

4 files changed

Lines changed: 33 additions & 1 deletion

File tree

‎eval/benchmark.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -256,7 +256,9 @@ def _redact_url(value: str) -> str:
256256
try:
257257
parsed = urlsplit(value)
258258
except ValueError:
259-
return value
259+
# An invalid authority can still contain credentials. Without a trustworthy parse,
260+
# preserve neither the authority nor the rest of the URL in public evidence.
261+
return "<redacted>"
260262
if not parsed.scheme or not parsed.netloc:
261263
return value
262264

‎eval/longmemeval_v2_evidence.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -171,6 +171,10 @@ def build_evidence_report(
171171
per_question = Path(per_question_path)
172172
if re.fullmatch(r"[0-9a-f]{40}", reader_revision) is None:
173173
raise ValueError("reader_revision must be an immutable lowercase 40-character commit")
174+
if bool(evaluator_model) != bool(evaluator_revision):
175+
raise ValueError("evaluator_model and evaluator_revision must be used together")
176+
if evaluator_revision and re.fullmatch(r"[0-9a-f]{40}", evaluator_revision) is None:
177+
raise ValueError("evaluator_revision must be an immutable lowercase 40-character commit")
174178
source_paths = [
175179
per_question,
176180
Path(haystack_path),

‎tests/test_benchmark_evidence.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -442,6 +442,10 @@ def test_command_provenance_redacts_userinfo_when_a_url_port_is_malformed():
442442
]
443443

444444

445+
def test_command_provenance_fails_closed_when_url_splitting_rejects_userinfo():
446+
assert redact_command(["https://user:password@[invalid/path"]) == ["<redacted>"]
447+
448+
445449
def test_canonical_profile_validator_and_immutable_artifact_writer(tmp_path):
446450
dataset = tmp_path / "fixture.jsonl"
447451
dataset.write_text('{"id":"one"}\n', encoding="utf-8")

‎tests/test_longmemeval_v2_evidence.py‎

Lines changed: 22 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -106,6 +106,28 @@ def test_official_v2_evidence_export_rejects_unpinned_reader_metadata(tmp_path):
106106
)
107107

108108

109+
@pytest.mark.parametrize(
110+
("evaluator_model", "evaluator_revision", "message"),
111+
[
112+
("example/evaluator", None, "must be used together"),
113+
("example/evaluator", "main", "immutable lowercase 40-character commit"),
114+
],
115+
)
116+
def test_official_v2_evidence_export_requires_a_pinned_evaluator(
117+
evaluator_model, evaluator_revision, message,
118+
):
119+
with pytest.raises(ValueError, match=message):
120+
build_evidence_report(
121+
per_question_path="unused.jsonl",
122+
questions_path="unused-questions.json",
123+
haystack_path="unused-haystack.json",
124+
trajectories_path="unused-trajectories.json",
125+
memory_config_path="unused-memory.json",
126+
evaluator_model=evaluator_model,
127+
evaluator_revision=evaluator_revision,
128+
)
129+
130+
109131
def test_official_v2_evidence_export_rejects_malformed_measured_fields(tmp_path):
110132
source_paths = [
111133
_write_json(tmp_path / name, {} if name != "questions.json" else [])

0 commit comments

Comments
 (0)