diff --git a/.github/workflows/perf-hosted-calibration.yml b/.github/workflows/perf-hosted-calibration.yml new file mode 100644 index 00000000..a74c46b8 --- /dev/null +++ b/.github/workflows/perf-hosted-calibration.yml @@ -0,0 +1,343 @@ +name: Hosted PR performance calibration + +on: + workflow_dispatch: + inputs: + pr_number: + description: Open pull request number to measure against its base + required: true + type: number + +permissions: + contents: read + pull-requests: read + +jobs: + resolve-identity: + name: Freeze PR identity + if: github.repository == 'semantic-reasoning/wirelog' && github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + outputs: + pr_number: ${{ steps.identity.outputs.pr_number }} + base_sha: ${{ steps.identity.outputs.base_sha }} + head_sha: ${{ steps.identity.outputs.head_sha }} + merge_sha: ${{ steps.identity.outputs.merge_sha }} + steps: + - name: Prepare HOME-local TMPDIR + run: | + mkdir -p "$HOME/.tmp" + echo "TMPDIR=$HOME/.tmp" >> "$GITHUB_ENV" + - name: Resolve and verify PR identity + id: identity + env: + GH_TOKEN: ${{ github.token }} + PR_NUMBER: ${{ inputs.pr_number }} + shell: bash + run: | + set -euo pipefail + [[ "$PR_NUMBER" =~ ^[1-9][0-9]*$ ]] + response=$(gh api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}") + state=$(jq -r '.state' <<<"$response") + base_repo=$(jq -r '.base.repo.full_name // ""' <<<"$response") + base_ref=$(jq -r '.base.ref // ""' <<<"$response") + base_sha=$(jq -r '.base.sha // ""' <<<"$response") + head_sha=$(jq -r '.head.sha // ""' <<<"$response") + merge_sha=$(jq -r '.merge_commit_sha // ""' <<<"$response") + [[ "$state" == open && "$base_repo" == "$GITHUB_REPOSITORY" && "$base_ref" == main ]] + [[ "$base_sha" =~ ^[0-9a-f]{40}$ && "$head_sha" =~ ^[0-9a-f]{40}$ && "$merge_sha" =~ ^[0-9a-f]{40}$ ]] + merge_ref=$(git ls-remote "https://github.com/${GITHUB_REPOSITORY}.git" \ + "refs/pull/${PR_NUMBER}/merge" | awk 'NR == 1 { print $1 }') + [[ "$merge_ref" == "$merge_sha" ]] + { + printf 'pr_number=%s\n' "$PR_NUMBER" + printf 'base_sha=%s\n' "$base_sha" + printf 'head_sha=%s\n' "$head_sha" + printf 'merge_sha=%s\n' "$merge_sha" + } >> "$GITHUB_OUTPUT" + mkdir -p evidence + jq -n --arg pr "$PR_NUMBER" --arg base "$base_sha" \ + --arg head "$head_sha" --arg merge "$merge_sha" \ + --arg state "$state" --arg repo "$base_repo" --arg ref "$base_ref" \ + '{pr_number:$pr,base_sha:$base,head_sha:$head,merge_sha:$merge,state:$state,base_repo:$repo,base_ref:$ref}' \ + > evidence/identity.json + - name: Upload frozen identity + if: always() + continue-on-error: true + uses: actions/upload-artifact@v7 + with: + name: hosted-calibration-${{ github.run_id }}-identity + path: evidence/identity.json + if-no-files-found: warn + retention-days: 30 + + calibration-arm: + name: Arm ${{ matrix.ordinal }} / ${{ matrix.first }} to ${{ matrix.second }} + needs: [resolve-identity] + if: github.repository == 'semantic-reasoning/wirelog' && github.ref == 'refs/heads/main' && needs.resolve-identity.result == 'success' + runs-on: ubuntu-latest + timeout-minutes: 90 + strategy: + fail-fast: false + max-parallel: 1 + matrix: + include: + - ordinal: '01' + first: base + second: head + - ordinal: '02' + first: head + second: base + - ordinal: '03' + first: base + second: head + - ordinal: '04' + first: head + second: base + - ordinal: '05' + first: base + second: base + - ordinal: '06' + first: head + second: head + steps: + - name: Prepare isolated evidence directory and TMPDIR + shell: bash + run: | + set -euo pipefail + mkdir -p "$HOME/.tmp" "evidence/${{ matrix.ordinal }}" + echo "TMPDIR=$HOME/.tmp" >> "$GITHUB_ENV" + + - name: Checkout trusted workflow tools + uses: actions/checkout@v5 + with: + ref: ${{ github.workflow_sha }} + path: trusted + fetch-depth: 1 + persist-credentials: false + + - name: Freeze arm identity + env: + PR_NUMBER: ${{ needs.resolve-identity.outputs.pr_number }} + BASE_SHA: ${{ needs.resolve-identity.outputs.base_sha }} + HEAD_SHA: ${{ needs.resolve-identity.outputs.head_sha }} + MERGE_SHA: ${{ needs.resolve-identity.outputs.merge_sha }} + shell: bash + run: | + set -euo pipefail + jq -n --arg pr "$PR_NUMBER" --arg base "$BASE_SHA" \ + --arg head "$HEAD_SHA" --arg merge "$MERGE_SHA" \ + '{pr_number:$pr,base_sha:$base,head_sha:$head,merge_sha:$merge,state:"open",base_repo:"semantic-reasoning/wirelog",base_ref:"main"}' \ + > "evidence/${{ matrix.ordinal }}/identity.json" + + - name: Checkout immutable first source + uses: actions/checkout@v5 + with: + ref: ${{ matrix.first == 'base' && needs.resolve-identity.outputs.base_sha || needs.resolve-identity.outputs.merge_sha }} + path: source/run-01 + fetch-depth: 1 + persist-credentials: false + + - name: Checkout immutable second source + uses: actions/checkout@v5 + with: + ref: ${{ matrix.second == 'base' && needs.resolve-identity.outputs.base_sha || needs.resolve-identity.outputs.merge_sha }} + path: source/run-02 + fetch-depth: 1 + persist-credentials: false + + - name: Install dependencies and pinned Meson + run: | + sudo apt-get update + sudo apt-get install -y ninja-build linux-tools-common linux-tools-generic + shell: bash + - name: Set up Meson 1.12.0 + id: meson + uses: ./trusted/.github/actions/setup-meson + + - name: Request performance governor (best effort) + continue-on-error: true + shell: bash + run: | + sudo cpupower frequency-set -g performance || true + for cpu in /sys/devices/system/cpu/cpu[0-9]*/cpufreq/scaling_governor; do + echo "governor($cpu) = $(cat "$cpu" 2>/dev/null || echo unavailable)" + done + + - name: Build isolated sources and run paired actual gates + working-directory: trusted + env: + CC: gcc + shell: bash + run: | + set -euo pipefail + for run_no in 01 02; do + source="$GITHUB_WORKSPACE/source/run-$run_no" + build="$GITHUB_WORKSPACE/build-${{ matrix.ordinal }}-$run_no" + evidence="$GITHUB_WORKSPACE/evidence/${{ matrix.ordinal }}/run-$run_no" + mkdir -p "$evidence" + git -C "$source" status --porcelain > "$evidence/source-status.txt" + git -C "$source" rev-parse HEAD > "$evidence/source-sha.txt" + git -C "$source" rev-parse 'HEAD^{tree}' > "$evidence/source-tree.txt" + set +e + meson setup "$build" "$source" --buildtype=release \ + -Dwirelog_log_max_level=trace -Dtests=true -DmbedTLS=disabled \ + > "$evidence/meson-setup.stdout.log" \ + 2> "$evidence/meson-setup.stderr.log" + status=$? + set -e + printf '%s\n' "$status" > "$evidence/meson-setup.exit" + if [ "$status" -ne 0 ]; then + exit "$status" + fi + set +e + meson compile -C "$build" test_crdt_perf_gate test_cspa_perf_gate \ + > "$evidence/meson-compile.stdout.log" \ + 2> "$evidence/meson-compile.stderr.log" + status=$? + set -e + printf '%s\n' "$status" > "$evidence/meson-compile.exit" + if [ "$status" -ne 0 ]; then + exit "$status" + fi + sha256sum "$build/tests/test_crdt_perf_gate" \ + "$build/tests/test_cspa_perf_gate" > "$evidence/binary-sha256.txt" + for gate in crdt cspa; do + case "$gate" in + crdt) test_name=crdt_perf_gate ;; + cspa) test_name=cspa_w1_gate ;; + esac + gate_dir="$evidence/$gate" + mkdir -p "$gate_dir" + taskset -c 0 env TMPDIR="$HOME/.tmp" \ + python scripts/perf/hosted_perf_calibration.py capture-host \ + --output "$gate_dir/host-before.json" + set +e + taskset -c 0 env WIRELOG_PERF_GATE=1 \ + meson test -C "$build" "$test_name" --print-errorlogs \ + --verbose --num-processes 1 \ + > "$gate_dir/stdout.log" 2> "$gate_dir/stderr.log" + status=$? + set -e + printf '%s\n' "$status" > "$gate_dir/exit" + cp "$build/meson-logs/testlog.txt" \ + "$gate_dir/meson-testlog.txt" + taskset -c 0 env TMPDIR="$HOME/.tmp" \ + python scripts/perf/hosted_perf_calibration.py capture-host \ + --output "$gate_dir/host-after.json" || true + echo "$run_no $gate Meson exit status: $status" + # Target misses, skips, and correctness failures remain evidence. + done + done + + - name: Capture host state and parsed arm evidence + if: always() && steps.meson.outcome == 'success' + working-directory: trusted + env: + ORDINAL: ${{ matrix.ordinal }} + shell: bash + run: | + mkdir -p "$HOME/.tmp" "evidence/$ORDINAL" + TMPDIR="$HOME/.tmp" python scripts/perf/hosted_perf_calibration.py capture-ordinal \ + --ordinal "$ORDINAL" \ + --identity "$GITHUB_WORKSPACE/evidence/$ORDINAL/identity.json" \ + --source "$GITHUB_WORKSPACE/source" \ + --evidence "$GITHUB_WORKSPACE/evidence/$ORDINAL" \ + --output "$GITHUB_WORKSPACE/evidence/$ORDINAL/$ORDINAL.json" || true + + - name: Upload arm evidence, including partial failures + if: always() + uses: actions/upload-artifact@v7 + with: + name: hosted-calibration-${{ github.run_id }}-${{ matrix.ordinal }} + path: evidence/${{ matrix.ordinal }}/ + if-no-files-found: warn + retention-days: 30 + + aggregate: + name: Aggregate calibration evidence + needs: [resolve-identity, calibration-arm] + if: always() && github.repository == 'semantic-reasoning/wirelog' && github.ref == 'refs/heads/main' && needs.resolve-identity.result == 'success' + runs-on: ubuntu-latest + steps: + - name: Prepare aggregate workspace and TMPDIR + run: | + mkdir -p "$HOME/.tmp" evidence/downloaded + echo "TMPDIR=$HOME/.tmp" >> "$GITHUB_ENV" + - name: Checkout trusted workflow tools + uses: actions/checkout@v5 + with: + ref: ${{ github.workflow_sha }} + path: trusted + fetch-depth: 1 + persist-credentials: false + - name: Download every available arm artifact + id: download + continue-on-error: true + uses: actions/download-artifact@v6 + with: + pattern: hosted-calibration-${{ github.run_id }}-0* + path: evidence/downloaded + merge-multiple: false + - name: Recheck PR and merge identity + env: + GH_TOKEN: ${{ github.token }} + PR_NUMBER: ${{ needs.resolve-identity.outputs.pr_number }} + BASE_SHA: ${{ needs.resolve-identity.outputs.base_sha }} + HEAD_SHA: ${{ needs.resolve-identity.outputs.head_sha }} + MERGE_SHA: ${{ needs.resolve-identity.outputs.merge_sha }} + shell: bash + run: | + set -euo pipefail + response=$(gh api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}") || response='{}' + state=$(jq -r '.state // ""' <<<"$response") + base_repo=$(jq -r '.base.repo.full_name // ""' <<<"$response") + base_ref=$(jq -r '.base.ref // ""' <<<"$response") + base_sha=$(jq -r '.base.sha // ""' <<<"$response") + head_sha=$(jq -r '.head.sha // ""' <<<"$response") + merge_sha=$(jq -r '.merge_commit_sha // ""' <<<"$response") + merge_ref=$(git ls-remote "https://github.com/${GITHUB_REPOSITORY}.git" \ + "refs/pull/${PR_NUMBER}/merge" | awk 'NR == 1 { print $1 }' || true) + if [[ "$merge_ref" != "$MERGE_SHA" ]]; then + merge_sha=stale + fi + jq -n --arg pr "$PR_NUMBER" --arg base "$base_sha" \ + --arg head "$head_sha" --arg merge "$merge_sha" \ + --arg state "$state" --arg repo "$base_repo" --arg ref "$base_ref" \ + '{pr_number:$pr,base_sha:$base,head_sha:$head,merge_sha:$merge,state:$state,base_repo:$repo,base_ref:$ref}' \ + > evidence/current-identity.json + jq -n --arg pr "$PR_NUMBER" --arg base "$BASE_SHA" \ + --arg head "$HEAD_SHA" --arg merge "$MERGE_SHA" \ + '{pr_number:$pr,base_sha:$base,head_sha:$head,merge_sha:$merge,state:"open",base_repo:"semantic-reasoning/wirelog",base_ref:"main"}' \ + > evidence/expected-identity.json + - name: Classify complete or partial evidence + if: always() + working-directory: trusted + run: | + set -euo pipefail + mkdir -p "$HOME/.tmp" + TMPDIR="$HOME/.tmp" python scripts/perf/hosted_perf_calibration.py aggregate \ + --identity "$GITHUB_WORKSPACE/evidence/expected-identity.json" \ + --current-identity "$GITHUB_WORKSPACE/evidence/current-identity.json" \ + --evidence "$GITHUB_WORKSPACE/evidence/downloaded" \ + --output "$GITHUB_WORKSPACE/evidence/campaign-result.json" \ + | tee "$GITHUB_WORKSPACE/evidence/campaign-status.txt" + - name: Publish campaign status + if: always() + run: | + echo '## Hosted performance calibration' >> "$GITHUB_STEP_SUMMARY" + if [ -f evidence/campaign-status.txt ]; then + echo '```json' >> "$GITHUB_STEP_SUMMARY" + cat evidence/campaign-status.txt >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + else + echo 'Campaign result is incomplete; no final classification was produced.' \ + >> "$GITHUB_STEP_SUMMARY" + fi + - name: Upload aggregate result and logs + if: always() + uses: actions/upload-artifact@v7 + with: + name: hosted-calibration-${{ github.run_id }}-aggregate + path: evidence/ + if-no-files-found: warn + retention-days: 30 diff --git a/scripts/perf/hosted_perf_calibration.py b/scripts/perf/hosted_perf_calibration.py new file mode 100644 index 00000000..68462dd9 --- /dev/null +++ b/scripts/perf/hosted_perf_calibration.py @@ -0,0 +1,474 @@ +#!/usr/bin/env python3 +"""Capture and classify hosted, read-only PR performance evidence.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import platform +import re +import subprocess + + +ORDINALS = { + "01": ("base", "head"), "02": ("head", "base"), + "03": ("base", "head"), "04": ("head", "base"), + "05": ("base", "base"), "06": ("head", "head"), +} +GATES = ("crdt", "cspa") +TRIALS = 9 +PROFILE = { + "cc": "gcc", + "meson": "1.12.0", + "meson_args": ["--buildtype=release", "-Dwirelog_log_max_level=trace", + "-Dtests=true", "-DmbedTLS=disabled"], + "gates": {"crdt": "crdt_perf_gate", "cspa": "cspa_w1_gate"}, + "trials": TRIALS, + "affinity_cpu": 0, + "governor": "best-effort performance request; gate may skip if unavailable", +} +PROFILE_SHA256 = hashlib.sha256( + json.dumps(PROFILE, sort_keys=True, separators=(",", ":")).encode() +).hexdigest() +RAW_RE = re.compile( + r"test_(crdt|cspa)_perf_gate: raw_ms =((?:\s+[0-9]+(?:\.[0-9]+)?){1,})") +TARGET_RE = re.compile(r"median_ms\s*=\s*[0-9]+(?:\.[0-9]+)?\s+\(target\s+(\d+)\)") +MEDIAN_RE = re.compile(r"median_ms\s*=\s*([0-9]+(?:\.[0-9]+)?)") +COV_RE = re.compile(r"CoV\s+([0-9]+(?:\.[0-9]+)?)%") + + +def run_text(command: list[str], *, timeout: int = 10) -> str | None: + try: + result = subprocess.run(command, text=True, capture_output=True, + check=False, timeout=timeout) + except (OSError, subprocess.TimeoutExpired): + return None + if result.returncode != 0: + return None + return (result.stdout + result.stderr).strip() + + +def read_text(path: Path) -> str | None: + try: + return path.read_text(encoding="utf-8").strip() + except OSError: + return None + + +def cpu_model() -> str | None: + text = read_text(Path("/proc/cpuinfo")) + if not text: + return None + for line in text.splitlines(): + if line.startswith(("model name", "Hardware")) and ":" in line: + return line.split(":", 1)[1].strip() + return None + + +def capture_host() -> dict: + cpu0 = Path("/sys/devices/system/cpu/cpu0/cpufreq") + cgroup_text = read_text(Path("/proc/self/cgroup")) + cgroup_dir = Path("/sys/fs/cgroup") + if cgroup_text: + for line in cgroup_text.splitlines(): + fields = line.split(":", 2) + if len(fields) == 3 and fields[0] == "0" and not fields[1]: + cgroup_dir = cgroup_dir / fields[2].lstrip("/") + break + cgroup: dict[str, str | None] = { + "membership": cgroup_text, + "resolved_directory": str(cgroup_dir), + } + for name in ("cpu.max", "cpu.stat", "memory.max", "memory.current", + "memory.events"): + cgroup[name] = read_text(cgroup_dir / name) + return { + "runner_image": "/".join(filter(None, (os.getenv("ImageOS"), + os.getenv("ImageVersion")))) or None, + "os": platform.platform(), + "architecture": platform.machine(), + "cpu_model": cpu_model(), + "cpu_count": os.cpu_count(), + "kernel": platform.release(), + "gcc": run_text(["gcc", "--version"]), + "meson": run_text(["meson", "--version"]), + "affinity": run_text(["taskset", "-pc", str(os.getpid())]), + "requested_affinity_cpu": 0, + "governor": read_text(cpu0 / "scaling_governor"), + "frequency_khz": read_text(cpu0 / "scaling_cur_freq"), + "frequency_min_khz": read_text(cpu0 / "scaling_min_freq"), + "frequency_max_khz": read_text(cpu0 / "scaling_max_freq"), + "loadavg": read_text(Path("/proc/loadavg")), + "psi_cpu": read_text(Path("/proc/pressure/cpu")), + "psi_memory": read_text(Path("/proc/pressure/memory")), + "psi_io": read_text(Path("/proc/pressure/io")), + "cgroup": cgroup, + } + + +def parse_gate(gate: str, exit_code: int, output: str) -> dict: + if gate not in GATES: + raise ValueError(f"unknown gate: {gate}") + test_name = f"test_{gate}_perf_gate" + raw_match = RAW_RE.search(output) + raw_values = ([float(value) for value in raw_match[2].split()] + if raw_match and raw_match[1] == gate else []) + raw_complete = len(raw_values) == TRIALS and all(value > 0 for value in raw_values) + target_match = TARGET_RE.search(output) + target_ms = int(target_match[1]) if target_match else None + trials_match = re.search(rf"{re.escape(test_name)}: trials=(\d+) workers=1", output) + exit_match = re.search(r"(?m)^result:\s+exit status (\d+)$", output) + test_exit_code = int(exit_match[1]) if exit_match else exit_code + if gate == "crdt": + correctness = re.search( + r"result\s*=\s*(\d+)\s+\(expected\s+(\d+)\)", output) + correctness_values = ([int(correctness[1]), int(correctness[2])] + if correctness else None) + correctness_ok = bool(correctness_values == [104851, 104851]) + else: + tuples = re.search(r"tuples\s*=\s*(\d+)/20,381", output) + iterations = re.search(r"iterations\s*=\s*(\d+)/6", output) + correctness_values = ([int(tuples[1]), 20381, int(iterations[1]), 6] + if tuples and iterations else None) + correctness_ok = bool(tuples and iterations + and tuples[1] == "20381" and iterations[1] == "6") + target_miss = (test_exit_code == 1 and raw_complete and correctness_ok + and f"{test_name}: FAIL: median " in output + and " exceeds target " in output) + if test_exit_code == 77 or "SKIP:" in output: + status = "no_measurement" + elif target_miss: + status = "target_miss" + elif (exit_code == 0 and raw_complete and correctness_ok + and f"{test_name} OK" in output): + status = "pass" + elif "FAIL:" in output and not target_miss: + status = "correctness_failed" + else: + status = "no_measurement" + return { + "gate": gate, + "status": status, + "exit_code": test_exit_code, + "meson_exit_code": exit_code, + "raw_ms": raw_values, + "raw_trial_count": len(raw_values), + "trials": int(trials_match[1]) if trials_match else None, + "target_ms": target_ms, + "correctness_ok": correctness_ok, + "correctness_values": correctness_values, + "median_ms": (float(MEDIAN_RE.search(output)[1]) + if MEDIAN_RE.search(output) else None), + "cov_percent": [float(value) for value in COV_RE.findall(output)], + } + + +def classify_campaign(expected: dict, current: dict, + arms: dict[str, dict | None]) -> dict: + missing = [ordinal for ordinal in ORDINALS + if arms.get(ordinal) is None] + reasons: list[str] = [] + if missing: + stale = current != expected + if stale: + reasons.append("PR base/head/merge identity changed during campaign") + for ordinal, arm in arms.items(): + if arm is None: + continue + if (not isinstance(arm, dict) or arm.get("ordinal") != ordinal + or arm.get("identity") != expected): + stale = True + reasons.append(f"present arm {ordinal} identity is stale or malformed") + continue + runs = arm.get("runs") + if not isinstance(runs, list) or len(runs) != 2: + stale = True + reasons.append(f"present arm {ordinal} run structure is malformed") + continue + for run_index, run in enumerate(runs): + if not isinstance(run, dict): + stale = True + reasons.append(f"present arm {ordinal} run {run_index + 1} is malformed") + continue + side = ORDINALS[ordinal][run_index] + expected_sha = expected.get( + "base_sha" if side == "base" else "merge_sha") + if (run.get("side") != side + or run.get("source_sha") != expected_sha): + stale = True + reasons.append( + f"present arm {ordinal} run {run_index + 1} SHA is stale") + return { + "classification": "incomplete", "complete": False, + "eligible": False, "stale": stale, + "reasons": sorted(set(reasons + [ + "missing arm artifacts: " + ",".join(missing)])), + "arms": arms, + } + + stale = current != expected + profiles: set[str | None] = set() + host_profiles: set[tuple | None] = set() + targets: dict[str, set[int | None]] = {gate: set() for gate in GATES} + eligible = True + complete = True + for ordinal, arm in arms.items(): + if not isinstance(arm, dict): + complete = False + reasons.append(f"arm {ordinal} artifact is malformed") + continue + runs = arm.get("runs") + if (arm.get("ordinal") != ordinal or not isinstance(runs, list) + or len(runs) != 2 or arm.get("identity") != expected): + stale = True + complete = False + reasons.append(f"arm {ordinal} identity or two-run structure is invalid") + continue + for run_index, run in enumerate(runs): + if not isinstance(run, dict): + complete = False + stale = True + reasons.append(f"arm {ordinal} run {run_index + 1} is malformed") + continue + side = ORDINALS[ordinal][run_index] + expected_revision = expected.get( + "base_sha" if side == "base" else "merge_sha") + if (run.get("side") != side + or run.get("source_sha") != expected_revision): + stale = True + reasons.append(f"arm {ordinal} run {run_index + 1} SHA is stale") + if run.get("source_status") != "": + eligible = False + reasons.append(f"arm {ordinal} run {run_index + 1} source checkout is dirty") + host = run.get("host") + if not isinstance(host, dict): + profiles.add(None) + host_profiles.add(None) + eligible = False + reasons.append(f"arm {ordinal} run {run_index + 1} host profile is missing") + else: + profile_sha = host.get("profile_sha256") + profiles.add(profile_sha if isinstance(profile_sha, str) else None) + host_profile = host.get("host_profile") + if (isinstance(host_profile, (list, tuple)) + and len(host_profile) == 6 + and all(isinstance(value, str) and value + for value in host_profile)): + host_profiles.add(tuple(host_profile)) + else: + host_profiles.add(None) + eligible = False + reasons.append( + f"arm {ordinal} run {run_index + 1} host profile is malformed") + gates = run.get("gates") + if not isinstance(gates, dict) or any(gate not in gates for gate in GATES): + eligible = False + complete = False + reasons.append(f"arm {ordinal} run {run_index + 1} lacks gate evidence") + continue + for gate in GATES: + result = gates[gate] + if not isinstance(result, dict): + eligible = False + complete = False + reasons.append( + f"arm {ordinal} run {run_index + 1} {gate} evidence is malformed") + continue + targets[gate].add(result.get("target_ms")) + if not result.get("evidence_complete"): + complete = False + reasons.append( + f"arm {ordinal} run {run_index + 1} {gate} logs or host samples missing") + if result.get("status") not in ("pass", "target_miss"): + eligible = False + reasons.append( + f"arm {ordinal} run {run_index + 1} {gate} status is " + f"{result.get('status')}") + if (result.get("raw_trial_count") != TRIALS + or result.get("trials") != TRIALS + or not isinstance(result.get("raw_ms"), list) + or len(result["raw_ms"]) != TRIALS + or not result.get("correctness_ok") + or result.get("median_ms") is None + or not result.get("cov_percent")): + eligible = False + reasons.append( + f"arm {ordinal} run {run_index + 1} {gate} lacks nine correct raw trials") + + if profiles != {PROFILE_SHA256}: + eligible = False + reasons.append("build profile hash differs or is missing across arms") + if len(host_profiles) != 1 or None in host_profiles: + eligible = False + reasons.append("host image or CPU profile differs or is missing across runs") + for gate, values in targets.items(): + if len(values) != 1 or None in values: + eligible = False + reasons.append(f"{gate} target differs or is missing across arms") + + if stale: + eligible = False + reasons.append("PR base/head/merge identity changed during campaign") + if not complete: + eligible = False + return { + "classification": ("incomplete" if not complete else + "eligible" if eligible else "ineligible"), + "complete": complete, + "eligible": eligible, + "stale": stale, + "reasons": sorted(set(reasons)), + "identity": expected, + "profile_sha256": next(iter(profiles)) if len(profiles) == 1 else None, + "arms": arms, + } + + +def write_json(path: Path, value: dict) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n", + encoding="utf-8") + + +def command_capture_host(args: argparse.Namespace) -> None: + host = capture_host() + host["profile"] = PROFILE + host["profile_sha256"] = ( + PROFILE_SHA256 if host["meson"] == PROFILE["meson"] + and host["gcc"] and host["gcc"].startswith("gcc ") else "toolchain-mismatch") + write_json(args.output, host) + + +def parse_run(evidence: Path, source: Path, side: str, identity: dict, + run_number: int) -> dict: + source_sha = run_text(["git", "-C", str(source), "rev-parse", "HEAD"]) + source_tree = run_text(["git", "-C", str(source), "rev-parse", "HEAD^{tree}"]) + source_status = read_text(evidence / "source-status.txt") + gates = {} + profile_hashes = set() + runner_images = set() + host_profiles = set() + for gate in GATES: + folder = evidence / gate + stdout = read_text(folder / "stdout.log") or "" + stderr = read_text(folder / "stderr.log") or "" + meson_log = read_text(folder / "meson-testlog.txt") or "" + try: + exit_code = int((folder / "exit").read_text(encoding="ascii")) + except (OSError, ValueError): + exit_code = 255 + parsed = parse_gate(gate, exit_code, + stdout + "\n" + stderr + "\n" + meson_log) + for snapshot in ("host-before.json", "host-after.json"): + try: + host_sample = json.loads( + (folder / snapshot).read_text(encoding="utf-8")) + parsed[snapshot.removesuffix(".json")] = host_sample + profile_hashes.add(host_sample.get("profile_sha256")) + runner_images.add(host_sample.get("runner_image")) + host_profiles.add((host_sample.get("runner_image"), + host_sample.get("architecture"), + host_sample.get("cpu_model"), + host_sample.get("gcc"), + host_sample.get("meson"), + host_sample.get("kernel"))) + except (OSError, json.JSONDecodeError): + parsed[snapshot.removesuffix(".json")] = None + parsed["evidence_complete"] = all( + (folder / name).is_file() for name in + ("stdout.log", "stderr.log", "meson-testlog.txt", "exit", + "host-before.json", "host-after.json")) + gates[gate] = parsed + host_profile = (next(iter(host_profiles)) if len(host_profiles) == 1 else None) + if host_profile and any(not value for value in host_profile): + host_profile = None + return { + "run_number": run_number, + "side": side, + "source_ref": "base" if side == "base" else "pull-merge", + "source_sha": source_sha, + "source_tree": source_tree, + "source_status": source_status, + "identity": identity, + "host": { + "profile": PROFILE, + "profile_sha256": (next(iter(profile_hashes)) + if len(profile_hashes) == 1 else None), + "runner_image": (next(iter(runner_images)) + if len(runner_images) == 1 else None), + "host_profile": host_profile, + }, + "binary_sha256": read_text(evidence / "binary-sha256.txt"), + "gates": gates, + } + + +def command_capture_ordinal(args: argparse.Namespace) -> None: + identity = json.loads(args.identity.read_text(encoding="utf-8")) + sides = ORDINALS[args.ordinal] + runs = [parse_run(args.evidence / f"run-{number:02d}", + args.source / f"run-{number:02d}", side, identity, number) + for number, side in enumerate(sides, start=1)] + write_json(args.output, {"ordinal": args.ordinal, "identity": identity, + "runs": runs}) + + +def load_arms(evidence: Path) -> dict[str, dict | None]: + arms: dict[str, dict | None] = {} + for ordinal in ORDINALS: + candidates = list(evidence.rglob(f"{ordinal}.json")) + if len(candidates) != 1: + arms[ordinal] = None + continue + try: + arms[ordinal] = json.loads(candidates[0].read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + arms[ordinal] = None + return arms + + +def command_aggregate(args: argparse.Namespace) -> int: + expected = json.loads(args.identity.read_text(encoding="utf-8")) + current = json.loads(args.current_identity.read_text(encoding="utf-8")) + arms = load_arms(args.evidence) + result = classify_campaign(expected, current, arms) + write_json(args.output, result) + print(json.dumps({key: result[key] for key in + ("classification", "complete", "eligible", "stale", "reasons")}, + sort_keys=True)) + return 0 if result["classification"] == "eligible" else 1 + + +def make_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser() + commands = parser.add_subparsers(dest="command", required=True) + host = commands.add_parser("capture-host") + host.add_argument("--output", type=Path, required=True) + host.set_defaults(function=command_capture_host) + arm = commands.add_parser("capture-ordinal") + arm.add_argument("--ordinal", choices=tuple(ORDINALS), required=True) + arm.add_argument("--identity", type=Path, required=True) + arm.add_argument("--source", type=Path, required=True) + arm.add_argument("--evidence", type=Path, required=True) + arm.add_argument("--output", type=Path, required=True) + arm.set_defaults(function=command_capture_ordinal) + aggregate = commands.add_parser("aggregate") + aggregate.add_argument("--identity", type=Path, required=True) + aggregate.add_argument("--current-identity", type=Path, required=True) + aggregate.add_argument("--evidence", type=Path, required=True) + aggregate.add_argument("--output", type=Path, required=True) + aggregate.set_defaults(function=command_aggregate) + return parser + + +def main(argv: list[str] | None = None) -> int: + args = make_parser().parse_args(argv) + result = args.function(args) + return result if isinstance(result, int) else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/perf/test-hosted-perf-calibration.py b/scripts/perf/test-hosted-perf-calibration.py new file mode 100644 index 00000000..0f992d0e --- /dev/null +++ b/scripts/perf/test-hosted-perf-calibration.py @@ -0,0 +1,363 @@ +#!/usr/bin/env python3 +"""Contract and evidence-classification tests for hosted PR calibration.""" + +from pathlib import Path +from contextlib import redirect_stdout +import io +import json +import re +import tempfile +import unittest + +import hosted_perf_calibration as calibration + + +ROOT = Path(__file__).resolve().parents[2] +WORKFLOW = ROOT / ".github/workflows/perf-hosted-calibration.yml" +EXPECTED = { + "pr_number": "2085", + "base_sha": "b" * 40, + "head_sha": "h" * 40, + "merge_sha": "m" * 40, + "state": "open", + "base_repo": "semantic-reasoning/wirelog", + "base_ref": "main", +} +ORDINAL_SIDES = { + "01": ("base", "head"), "02": ("head", "base"), + "03": ("base", "head"), "04": ("head", "base"), + "05": ("base", "base"), "06": ("head", "head"), +} + + +def gate_log(gate: str, *, miss: bool = False, wrong: bool = False) -> str: + test_name = f"test_{gate}_perf_gate" + if wrong: + return (f"{test_name}: FAIL: trial 1 correctness sentinel mismatch\n" + "result: exit status 1\n") + if gate == "crdt": + result = 0 if wrong else 104851 + correctness = (f"{test_name}: trials=9 workers=1\n" + f" result = {result} (expected 104851)\n" + " iterations = 6\n") + target = 38120 + else: + tuples = 0 if wrong else 20381 + correctness = (f"{test_name}: trials=9 workers=1\n" + f" tuples = {tuples}/20,381\n" + " iterations = 6/6\n") + target = 2050 + start = target + 20 if miss else target - 100 + trials = [start + index * 10 for index in range(9)] + raw = " ".join(f"{value:.1f}" for value in trials) + median = trials[len(trials) // 2] + mean = sum(trials) / len(trials) + ending = (f"{test_name}: FAIL: median 110.0 ms exceeds target {target} ms " + "(regression)\n" if miss else f"{test_name} OK\n") + if miss: + ending = (f"{test_name}: FAIL: median {median:.1f} ms exceeds target " + f"{target} ms (regression)\n") + status = 1 if miss else 0 + return (f"{test_name}: raw_ms = {raw}\n" + f" mean_ms = {mean:.1f}\n stdev_ms = 3.0 (CoV 0.100%)\n" + f" median_ms = {median:.1f} (target {target})\n" + f"{correctness}{ending}result: exit status {status}\n") + + +def result_for(gate: str, *, status: str = "pass", target: int | None = None) -> dict: + output = gate_log(gate, miss=status == "target_miss") + exit_code = 1 if status == "target_miss" else 0 + if status == "skip": + output = (f"test_{gate}_perf_gate: SKIP: governor unavailable\n" + "result: exit status 77\n") + exit_code = 0 + elif status == "correctness_failed": + output, exit_code = gate_log(gate, wrong=True), 1 + parsed = calibration.parse_gate(gate, exit_code, output) + parsed["evidence_complete"] = True + parsed["host-before"] = {"profile_sha256": calibration.PROFILE_SHA256} + parsed["host-after"] = {"profile_sha256": calibration.PROFILE_SHA256} + if target is not None: + parsed["target_ms"] = target + return parsed + + +def full_arms() -> dict[str, dict]: + arms = {} + for ordinal, sides in ORDINAL_SIDES.items(): + runs = [] + for index, side in enumerate(sides, start=1): + sha = EXPECTED["base_sha" if side == "base" else "merge_sha"] + runs.append({ + "run_number": index, + "side": side, + "source_ref": "base" if side == "base" else "pull-merge", + "source_sha": sha, + "source_status": "", + "identity": EXPECTED, + "host": { + "profile_sha256": calibration.PROFILE_SHA256, + "host_profile": ("ubuntu-24.04/20260101", "x86_64", "Hosted CPU", + "gcc 14.1", "1.12.0", "6.8.0"), + }, + "gates": {gate: result_for(gate) for gate in calibration.GATES}, + }) + arms[ordinal] = {"ordinal": ordinal, "identity": EXPECTED, "runs": runs} + return arms + + +class HostedPerfWorkflowContract(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.workflow = WORKFLOW.read_text(encoding="utf-8") + + def test_manual_read_only_hosted_contract(self): + triggers = self.workflow.split("permissions:", 1)[0] + self.assertIn("workflow_dispatch:", triggers) + for forbidden in ("pull_request:", "push:", "schedule:"): + self.assertNotIn(forbidden, triggers) + self.assertIn("contents: read", self.workflow) + self.assertIn("pull-requests: read", self.workflow) + self.assertNotIn("secrets.", self.workflow) + self.assertNotRegex(self.workflow, r"(?m)^\s+(?:id-token|contents|pull-requests):\s+write$") + self.assertNotIn("self-hosted", self.workflow) + self.assertNotRegex(self.workflow, r"(?m)^\s+.*perf\s*\]$") + for name in ("resolve-identity", "calibration-arm", "aggregate"): + block = re.search( + rf"(?m)^ {re.escape(name)}:\n(.*?)(?=^ [\w-]+:|\Z)", + self.workflow, re.S) + self.assertIsNotNone(block, name) + self.assertIn("runs-on: ubuntu-latest", block[1]) + self.assertIn("github.repository == 'semantic-reasoning/wirelog'", block[1]) + self.assertIn("github.ref == 'refs/heads/main'", block[1]) + + def test_pr_and_merge_sha_are_frozen_and_rechecked(self): + self.assertIn("gh api \"repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}\"", self.workflow) + self.assertIn('refs/pull/${PR_NUMBER}/merge', self.workflow) + self.assertIn("[[ \"$merge_ref\" == \"$merge_sha\" ]]", self.workflow) + self.assertIn("ref: ${{ matrix.first == 'base' && needs.resolve-identity.outputs.base_sha || needs.resolve-identity.outputs.merge_sha }}", self.workflow) + self.assertIn("ref: ${{ matrix.second == 'base' && needs.resolve-identity.outputs.base_sha || needs.resolve-identity.outputs.merge_sha }}", self.workflow) + self.assertIn("current-identity.json", self.workflow) + self.assertIn("github.workflow_sha", self.workflow) + self.assertIn("persist-credentials: false", self.workflow) + + def test_six_ordered_serialized_pairs_and_partial_artifacts(self): + matrix = self.workflow.split(" include:\n", 1)[1].split(" steps:", 1)[0] + entries = re.findall( + r"- ordinal: '([0-9]{2})'\n\s+first: (base|head)\n\s+second: (base|head)", + matrix) + self.assertEqual([(ordinal, first, second) + for ordinal, first, second in entries], + [(key, *sides) for key, sides in ORDINAL_SIDES.items()]) + self.assertIn("fail-fast: false", self.workflow) + self.assertIn("max-parallel: 1", self.workflow) + self.assertGreaterEqual(self.workflow.count("name: Upload arm evidence"), 1) + self.assertIn("if: always()", self.workflow) + self.assertIn("stdout.log", self.workflow) + self.assertIn("stderr.log", self.workflow) + self.assertIn("capture-host", self.workflow) + + def test_profile_and_actual_serial_gate_contract(self): + self.assertIn("CC: gcc", self.workflow) + for token in ("--buildtype=release", "-Dwirelog_log_max_level=trace", + "-Dtests=true", "-DmbedTLS=disabled", + "test_crdt_perf_gate", "test_cspa_perf_gate", + "WIRELOG_PERF_GATE=1", "taskset -c 0", "TMPDIR=$HOME/.tmp"): + self.assertIn(token, self.workflow) + required = (ROOT / ".github/workflows/perf-suite-required.yml").read_text( + encoding="utf-8") + self.assertIn("runs-on: ubuntu-latest", required) + self.assertNotIn("self-hosted", required) + + def test_ineligible_aggregate_fails_but_upload_still_runs(self): + aggregate = self.workflow.split("\n aggregate:\n", 1)[1] + classify = aggregate.split("- name: Classify complete or partial evidence", 1)[1] + self.assertIn("set -euo pipefail", classify) + upload = aggregate.split("- name: Upload aggregate result and logs", 1)[1] + self.assertIn("if: always()", upload) + + +class HostedPerfParserTests(unittest.TestCase): + def test_timed_crdt_pass_and_target_miss_use_summary_correctness(self): + passed = calibration.parse_gate("crdt", 0, gate_log("crdt")) + miss = calibration.parse_gate("crdt", 1, gate_log("crdt", miss=True)) + self.assertEqual(passed["status"], "pass") + self.assertEqual(len(passed["raw_ms"]), 9) + self.assertEqual(passed["median_ms"], 38060.0) + self.assertEqual(passed["cov_percent"], [0.1]) + self.assertTrue(passed["correctness_ok"]) + self.assertEqual(miss["status"], "target_miss") + wrong_gold = gate_log("crdt").replace( + "result = 104851 (expected 104851)", "result = 1 (expected 1)") + self.assertFalse(calibration.parse_gate("crdt", 0, wrong_gold)["correctness_ok"]) + + def test_timed_cspa_summary_and_correctness_failure(self): + passed = calibration.parse_gate("cspa", 0, gate_log("cspa")) + wrong = calibration.parse_gate("cspa", 1, gate_log("cspa", miss=True, wrong=True)) + self.assertEqual(passed["status"], "pass") + self.assertTrue(passed["correctness_ok"]) + self.assertEqual(wrong["status"], "correctness_failed") + self.assertFalse(wrong["correctness_ok"]) + + def test_skip_or_missing_raw_sample_is_no_measurement(self): + skipped = calibration.parse_gate( + "crdt", 0, "test_crdt_perf_gate: SKIP: governor unavailable\n" + "result: exit status 77\n") + missing = calibration.parse_gate("cspa", 0, "no raw output") + self.assertEqual(skipped["status"], "no_measurement") + self.assertEqual(skipped["exit_code"], 77) + self.assertEqual(missing["status"], "no_measurement") + + +class HostedPerfCampaignClassificationTests(unittest.TestCase): + def run_aggregate_cli(self, root: Path, arms: dict, + current: dict = EXPECTED) -> tuple[int, dict]: + root.mkdir(parents=True, exist_ok=True) + identity_path = root / "identity.json" + current_path = root / "current.json" + artifact_root = root / "downloaded" + identity_path.write_text(json.dumps(EXPECTED), encoding="utf-8") + current_path.write_text(json.dumps(current), encoding="utf-8") + artifact_root.mkdir() + for ordinal, arm in arms.items(): + folder = artifact_root / f"hosted-calibration-{ordinal}" + folder.mkdir() + (folder / f"{ordinal}.json").write_text( + json.dumps(arm), encoding="utf-8") + output_path = root / "result.json" + with redirect_stdout(io.StringIO()): + exit_code = calibration.main([ + "aggregate", "--identity", str(identity_path), + "--current-identity", str(current_path), + "--evidence", str(artifact_root), "--output", str(output_path), + ]) + return exit_code, json.loads(output_path.read_text(encoding="utf-8")) + + def test_downloaded_artifact_layout_loads_six_two_run_ordinals(self): + cache = Path.home() / ".cache" + cache.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix="wirelog-hosted-calibration-", + dir=cache) as temporary: + root = Path(temporary) + for ordinal, arm in full_arms().items(): + folder = root / f"hosted-calibration-run-{ordinal}" + folder.mkdir() + (folder / f"{ordinal}.json").write_text( + calibration.json.dumps(arm), encoding="utf-8") + loaded = calibration.load_arms(root) + self.assertEqual(set(loaded), set(ORDINAL_SIDES)) + self.assertEqual(len(loaded["01"]["runs"]), 2) + self.assertEqual( + calibration.classify_campaign(EXPECTED, EXPECTED, loaded)["classification"], + "eligible") + + def test_two_run_pairs_and_target_miss_are_eligible(self): + arms = full_arms() + arms["01"]["runs"][0]["gates"]["crdt"] = result_for("crdt", status="target_miss") + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "eligible") + self.assertEqual(len(result["arms"]["01"]["runs"]), 2) + + def test_missing_ordinal_is_incomplete(self): + arms = full_arms() + del arms["06"] + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "incomplete") + self.assertFalse(result["complete"]) + + def test_missing_ordinal_preserves_stale_identity(self): + arms = full_arms() + del arms["06"] + changed = dict(EXPECTED, head_sha="x" * 40) + result = calibration.classify_campaign(EXPECTED, changed, arms) + self.assertEqual(result["classification"], "incomplete") + self.assertTrue(result["stale"]) + arms = full_arms() + del arms["06"] + arms["01"]["runs"][0]["source_sha"] = "x" * 40 + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "incomplete") + self.assertTrue(result["stale"]) + + def test_missing_gate_log_is_incomplete(self): + arms = full_arms() + arms["01"]["runs"][0]["gates"]["crdt"]["evidence_complete"] = False + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "incomplete") + + def test_skip_and_correctness_failure_are_ineligible(self): + arms = full_arms() + arms["01"]["runs"][0]["gates"]["cspa"] = result_for("cspa", status="skip") + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + self.assertFalse(result["eligible"]) + arms = full_arms() + arms["02"]["runs"][1]["gates"]["crdt"] = result_for( + "crdt", status="correctness_failed") + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + + def test_profile_mismatch_and_stale_identity_are_ineligible(self): + arms = full_arms() + arms["03"]["runs"][0]["host"]["profile_sha256"] = "different" + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + arms = full_arms() + arms["04"]["runs"][0]["host"]["host_profile"] = ( + "ubuntu-24.04/20260101", "x86_64", "Different CPU", + "gcc 14.1", "1.12.0", "6.8.0") + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + arms = full_arms() + changed = dict(EXPECTED, head_sha="x" * 40) + result = calibration.classify_campaign(EXPECTED, changed, arms) + self.assertEqual(result["classification"], "ineligible") + self.assertTrue(result["stale"]) + + def test_missing_none_and_malformed_host_profiles_are_safe_and_ineligible(self): + for malformed in (None, "ubuntu/CPU", ["only", "five", "fields"], 42): + with self.subTest(profile=malformed): + arms = full_arms() + arms["01"]["runs"][0]["host"]["host_profile"] = malformed + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + arms = full_arms() + del arms["01"]["runs"][0]["host"] + result = calibration.classify_campaign(EXPECTED, EXPECTED, arms) + self.assertEqual(result["classification"], "ineligible") + + def test_aggregate_command_exit_matches_eligibility(self): + cache = Path.home() / ".cache" + cache.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix="wirelog-hosted-aggregate-", + dir=cache) as temporary: + root = Path(temporary) + arms = full_arms() + arms["01"]["runs"][0]["gates"]["crdt"] = result_for( + "crdt", status="target_miss") + code, result = self.run_aggregate_cli(root / "eligible", arms) + self.assertEqual(code, 0) + self.assertEqual(result["classification"], "eligible") + + with tempfile.TemporaryDirectory(prefix="wirelog-hosted-aggregate-", + dir=cache) as temporary: + root = Path(temporary) + arms = full_arms() + arms["01"]["runs"][0]["gates"]["cspa"] = result_for( + "cspa", status="skip") + code, result = self.run_aggregate_cli(root / "ineligible", arms) + self.assertEqual(code, 1) + self.assertEqual(result["classification"], "ineligible") + + with tempfile.TemporaryDirectory(prefix="wirelog-hosted-aggregate-", + dir=cache) as temporary: + root = Path(temporary) + arms = full_arms() + del arms["06"] + code, result = self.run_aggregate_cli(root / "incomplete", arms) + self.assertEqual(code, 1) + self.assertEqual(result["classification"], "incomplete") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/meson.build b/tests/meson.build index 06b85511..c8821578 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -953,6 +953,10 @@ test('perf_nightly_governor_policy', py3, args: [meson.project_source_root() / 'scripts/ci/test-perf-nightly-governor-policy.py'], suite: 'abi') +test('hosted_perf_calibration_contract', py3, + args: [meson.project_source_root() / 'scripts/perf/test-hosted-perf-calibration.py'], + suite: 'abi', timeout: 60) + if host_machine.system() == 'linux' test('perf_batch_append_plan_contract', py3, args: [meson.project_source_root() / 'scripts/perf/test-prepare-batch-append-campaign.py'],