diff --git a/.github/workflows/heart-health.yml b/.github/workflows/heart-health.yml index e1caa10..ed71884 100644 --- a/.github/workflows/heart-health.yml +++ b/.github/workflows/heart-health.yml @@ -94,7 +94,7 @@ jobs: - name: Install PyYAML run: pip install --quiet pyyaml - - name: Run cloud-safe checks (ci_status, open_prs, ci_timing, no_run_census, smoke_timings, unit_timings) + - name: Run cloud-safe checks (CI, timing, smoke and release evidence) env: # The repo-scoped GITHUB_TOKEN is enough for every check here except # the artifact /zip downloads smoke_timings and unit_timings make @@ -109,7 +109,7 @@ jobs: # the auto-provisioned token and Heart's own artifacts still ingest. GH_TOKEN: ${{ secrets.HEART_TIMINGS_TOKEN || secrets.GITHUB_TOKEN }} run: | - # Six API-only checks: CI conclusions and open PRs (health), then CI + # API-only checks: CI conclusions and open PRs (health), then CI # wall-clock, the NO_RUN census, the per-script smoke timings and the # per-test unit timings (the ⏱ performance surface). ci_timing's and # smoke_timings' aggregate steps re-read the board.json published by @@ -125,6 +125,9 @@ jobs: bash heart/checks/no_run_census.sh bash heart/checks/smoke_timings.sh bash heart/checks/unit_timings.sh + # Read validation artifacts before readiness; never dispatch a run. + PYTHONPATH="$PWD" python -m heart.checks.test_run + PYTHONPATH="$PWD" python -m heart.checks.cloud_validation exit 0 - name: Append today's observations to the timing record diff --git a/config/repos.yaml b/config/repos.yaml index 665df62..51d3ce5 100644 --- a/config/repos.yaml +++ b/config/repos.yaml @@ -286,3 +286,7 @@ thresholds: worktree_drift: orphan_red: true # orphan WT with uncommitted work → red orphan_yellow: true # orphan WT (clean) → yellow + +# Read-only cloud evidence producer (identity from the body map). +release_evidence: + rehearsal_repo: PyAutoLabs/PyAutoHands diff --git a/docs/internals.md b/docs/internals.md index a675205..464436c 100644 --- a/docs/internals.md +++ b/docs/internals.md @@ -112,3 +112,20 @@ NUMBA_CACHE_DIR=/tmp/numba_cache MPLCONFIGDIR=/tmp/matplotlib \ The never-rewrite-history rules live in [`AGENTS.md`](../AGENTS.md) and apply here as everywhere. + +## Cloud validation evidence + +The daily board runs the existing smoke-result reader and the read-only +`heart.checks.cloud_validation` collector before aggregation. The latter reads +only the newest main integration run and searches at most 20 main rehearsal +runs in the configured `release_evidence.rehearsal_repo`. It requires exact +version, run ID, attempt, producer SHA and chronology agreement. Missing or +expired rehearsal artifacts leave validation incomplete. Failed producers +remain adverse even when their report claims success; no older integration +pass is substituted. No build is dispatched. + +The canonical validator receives the artifacts and their original producer +time (the earlier stage start), so a new cloud runner cannot rejuvenate an +old pass. Installation checks retain their own timestamps and source/index. +Readiness still checks release fidelity, current library SHAs and evidence age. +This bounded search is daily-only and does not add work to the fast local tick. diff --git a/heart/checks/cloud_validation.py b/heart/checks/cloud_validation.py new file mode 100644 index 0000000..2d90f40 --- /dev/null +++ b/heart/checks/cloud_validation.py @@ -0,0 +1,187 @@ +"""Read completed validation artifacts for the ephemeral cloud board. + +No dispatch, build or publication. This bounded daily collector is deliberately +outside tick: a fresh runner needs both stages, not the local integrate cache. +The validator remains the only fold, and readiness remains the only verdict. +""" + +from __future__ import annotations + +import json +import subprocess +import tempfile +from pathlib import Path +from typing import Any + +import yaml + +from heart import state, validate +from heart.checks import release_run + + +def api(path: str) -> dict: + try: + result = subprocess.run( + ["gh", "api", path], capture_output=True, text=True, timeout=30 + ) + data = json.loads(result.stdout) if result.returncode == 0 else {} + return data if isinstance(data, dict) else {} + except (OSError, ValueError, subprocess.TimeoutExpired): + return {} + + +def download(repo: str, run: dict, name: str, filename: str, dest: Path) -> dict | None: + try: + result = subprocess.run( + [ + "gh", + "run", + "download", + str(run["id"]), + "--repo", + repo, + "--name", + name, + "--dir", + str(dest), + ], + capture_output=True, + text=True, + timeout=60, + ) + if result.returncode: + return None + data = json.loads((dest / filename).read_text()) + return data if isinstance(data, dict) else None + except (OSError, ValueError, KeyError, subprocess.TimeoutExpired): + return None + + +def matching_rehearsal( + stage: dict, integration: dict, artifact: dict, run: dict +) -> bool: + """Require actual producer identity and an exact wheel version, not proximity.""" + start = release_run._parse_ts(integration.get("created_at")) + produced = release_run._parse_ts(run.get("updated_at")) + return bool( + stage.get("version") + and artifact.get("version") == stage["version"] + and artifact.get("mode") == "rehearsal" + and artifact.get("index") == "testpypi" + and artifact.get("run_id") + and artifact.get("run_attempt") + and run.get("run_attempt") + and str(artifact.get("run_id")) == str(run.get("id")) + and str(artifact.get("run_attempt")) == str(run.get("run_attempt")) + and artifact.get("build_sha") + and artifact["build_sha"] == run.get("head_sha") + and run.get("status") == "completed" + and start + and produced + and produced <= start + ) + + +def collect(config: dict, fetch=api, get_artifact=download) -> dict[str, Any]: + """Observe the newest integration only; never fall back to an older pass. + + Callables are injectable for offline provenance and freshness tests. + All temporary and persisted files live inside HEART_STATE_DIR. + """ + source = config["release_evidence"] + repo = release_run.RELEASE_REPO + runs = fetch( + f"repos/{repo}/actions/workflows/release-integrate.yml/runs?branch=main&per_page=1" + ).get("workflow_runs", []) + if not runs: + return {"action": "unavailable"} + integration = runs[0] + if integration.get("status") != "completed": + return {"action": "in-progress", "run_url": integration.get("html_url")} + observed = release_run._parse_ts(integration.get("created_at")) + if observed is None: + return {"action": "missing-producer-time"} + adverse = integration.get("conclusion") != "success" + state.HEART_STATE_DIR.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(dir=state.HEART_STATE_DIR) as tmp: + root = Path(tmp) + stage = get_artifact( + repo, + integration, + "release-stage-report", + "stage_report.json", + root / "integration", + ) + if ( + not stage + or stage.get("stage") != "integrate" + or stage.get("run_url") != integration.get("html_url") + ): + if not adverse: + return {"action": "artifact-unavailable"} + # A failed producer without an artifact is still adverse evidence. + stage = { + "stage": "integrate", + "status": "fail", + "run_url": integration.get("html_url"), + } + sources = [] + stage_path = root / "stage_report.json" + state.atomic_write_json(stage_path, stage) + sources.append(stage_path) + matched = False + if stage.get("version"): + producer_repo = source["rehearsal_repo"] + candidates = fetch( + f"repos/{producer_repo}/actions/workflows/release.yml/runs?branch=main&per_page=20" + ).get("workflow_runs", []) + for candidate in candidates[:20]: + created = release_run._parse_ts(candidate.get("created_at")) + if ( + not created + or created > observed + or candidate.get("status") != "completed" + ): + continue + artifact = get_artifact( + producer_repo, + candidate, + "testpypi-rehearsal-version", + "rehearsal.json", + root / str(candidate.get("id")), + ) + if not artifact or not matching_rehearsal( + stage, integration, artifact, candidate + ): + continue + # A matching unsuccessful producer cannot be laundered by its + # artifact or replaced by another, older producer. + adverse = adverse or candidate.get("conclusion") != "success" + observed = min(observed, created) + path = root / "rehearsal.json" + state.atomic_write_json(path, artifact) + sources.append(path) + matched = True + break + report = validate.run(sources, now=observed, force_fail=adverse) + return { + "action": "ingested", + "rehearsal_matched": matched, + "run_url": integration.get("html_url"), + "validation_outcome": report["validation_outcome"], + "ts": report["ts"], + } + + +def main() -> int: + config = yaml.safe_load((release_run.HEART_HOME / "config/repos.yaml").read_text()) + result = collect(config) + state.atomic_write_json(state.HEART_STATE_DIR / "cloud_validation.json", result) + from heart.heart_color import c_info + + print(c_info("cloud_validation: " + json.dumps(result, sort_keys=True))) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_cloud_validation.py b/tests/test_cloud_validation.py new file mode 100644 index 0000000..d4f0796 --- /dev/null +++ b/tests/test_cloud_validation.py @@ -0,0 +1,239 @@ +"""Fresh cloud state must reconstruct evidence, never manufacture readiness.""" + +import json +import pytest +from heart import readiness, state, validate +from heart.checks import cloud_validation as cloud + + +@pytest.fixture +def evidence(tmp_path, monkeypatch): + monkeypatch.setattr(state, "HEART_STATE_DIR", tmp_path) + monkeypatch.setattr( + validate, "VALIDATION_REPORT_FILE", tmp_path / "validation_report.json" + ) + monkeypatch.setattr( + validate, "VERIFY_INSTALL_FILE", tmp_path / "verify_install.json" + ) + monkeypatch.setattr(validate, "VALIDATION_HISTORY_DIR", tmp_path / "history") + integration = dict( + id=20, + status="completed", + conclusion="success", + created_at="2026-10-01T12:00:00Z", + html_url="https://example.test/runs/20", + ) + producer = dict( + id=10, + status="completed", + conclusion="success", + created_at="2026-10-01T10:00:00Z", + updated_at="2026-10-01T11:00:00Z", + run_attempt=1, + head_sha="abcdef123", + ) + artifact = dict( + version="2026.10.1.1.dev101", + mode="rehearsal", + index="testpypi", + run_id="10", + run_attempt="1", + build_sha="abcdef123", + ) + stage = dict( + stage="integrate", + status="pass", + profile="release", + version=artifact["version"], + run_url=integration["html_url"], + summary=dict(passed=719, failed=0, timeout=0, skipped=82), + commit_shas={n: "abcdef123" for n in readiness._GATE_SHA_LIBS}, + verify_install=dict( + ready=True, + ts="2026-10-01T12:30:00Z", + version=artifact["version"], + index="testpypi", + checks=[dict(check="A", status="PASS")], + ), + ) + return dict( + integration=integration, + producer=producer, + artifact=artifact, + stage=stage, + tmp=tmp_path, + ) + + +def run(e): + def fetch(path): + return { + "workflow_runs": ( + [e["integration"]] + if "release-integrate.yml" in path + else [e["producer"]] + ) + } + + def download(repo, record, name, filename, dest): + return e["stage"] if name == "release-stage-report" else e["artifact"] + + return cloud.collect( + {"release_evidence": {"rehearsal_repo": "example/build"}}, fetch, download + ) + + +def report(e): + return json.loads((e["tmp"] / "validation_report.json").read_text()) + + +def test_complete_evidence_does_not_rejuvenate(evidence): + e = evidence + assert run(e)["validation_outcome"] == "pass" + first = report(e) + assert first["totals"]["passed"] == 719 + assert first["ts"].startswith("2026-10-01T10:00:00") + run(e) + assert report(e) == first + vi = json.loads((e["tmp"] / "verify_install.json").read_text()) + assert vi["ts"] == e["stage"]["verify_install"]["ts"] + assert vi["index"] == "testpypi" + + +@pytest.mark.parametrize( + "field,value", + [ + ("version", "wrong"), + ("run_id", "99"), + ("run_attempt", "2"), + ("build_sha", "wrong"), + ("index", "pypi"), + ("mode", "live"), + ], +) +def test_mismatched_provenance_is_incomplete(evidence, field, value): + evidence["artifact"][field] = value + assert run(evidence)["validation_outcome"] == "incomplete" + assert "rehearse" not in report(evidence)["stages"] + + +@pytest.mark.parametrize("status", ["in_progress", "queued"]) +def test_pending_rehearsal_is_not_proof(evidence, status): + evidence["producer"]["status"] = status + assert run(evidence)["validation_outcome"] == "incomplete" + + +def test_future_rehearsal_is_not_proof(evidence): + evidence["producer"]["updated_at"] = "2026-10-02T00:00:00Z" + assert run(evidence)["validation_outcome"] == "incomplete" + + +@pytest.mark.parametrize("which", ["integration", "producer"]) +@pytest.mark.parametrize("conclusion", ["failure", "cancelled", "timed_out"]) +def test_failed_producers_override_pass(evidence, which, conclusion): + evidence[which]["conclusion"] = conclusion + assert run(evidence)["validation_outcome"] == "fail" + + +def test_missing_rehearsal_is_incomplete(evidence): + evidence["artifact"] = None + assert run(evidence)["validation_outcome"] == "incomplete" + + +def test_failed_integration_without_artifact_is_adverse(evidence): + evidence["stage"] = None + evidence["integration"]["conclusion"] = "failure" + assert run(evidence)["validation_outcome"] == "fail" + + +def test_missing_success_artifact_does_not_invent_report(evidence): + evidence["stage"] = None + assert run(evidence)["action"] == "artifact-unavailable" + assert not (evidence["tmp"] / "validation_report.json").exists() + + +def test_pending_integration_does_not_use_old_run(evidence): + evidence["integration"]["status"] = "in_progress" + assert run(evidence)["action"] == "in-progress" + assert not (evidence["tmp"] / "validation_report.json").exists() + + +def test_wrong_integration_identity_is_not_ingested(evidence): + evidence["stage"]["run_url"] = "https://example.test/runs/19" + assert run(evidence)["action"] == "artifact-unavailable" + + +def test_missing_timestamp_cannot_create_fresh_success(evidence): + evidence["integration"]["created_at"] = None + assert run(evidence)["action"] == "missing-producer-time" + + +def test_bounded_search(evidence): + calls = [] + + def fetch(path): + if "release-integrate" in path: + return { + "workflow_runs": [ + evidence["integration"], + dict(evidence["integration"], id=19), + ] + } + return {"workflow_runs": [dict(evidence["producer"], id=n) for n in range(100)]} + + def download(repo, record, name, filename, dest): + calls.append((record["id"], name)) + return evidence["stage"] if name == "release-stage-report" else None + + result = cloud.collect( + {"release_evidence": {"rehearsal_repo": "example/build"}}, fetch, download + ) + assert result["validation_outcome"] == "incomplete" + assert len(calls) == 21 + assert (19, "release-stage-report") not in calls + + +def cloud_verdict(e, report_override=None): + from tests.test_readiness import make_snapshot, LIBS + + value = report_override or report(e) + snap = make_snapshot( + ts="2026-10-01T13:00:00Z", + validation_report=value, + verify_install=json.loads((e["tmp"] / "verify_install.json").read_text()), + ) + for lib in LIBS: + snap["repos"][lib]["ci_status"]["head_sha"] = "abcdef123" + return readiness.compute(snap, libraries=LIBS) + + +def test_collected_proof_is_green_only_for_current_source(evidence): + run(evidence) + assert cloud_verdict(evidence)["verdict"] == "green", cloud_verdict(evidence) + value = report(evidence) + value["commit_shas"][next(iter(readiness._GATE_SHA_LIBS))] = "moved" + verdict = cloud_verdict(evidence, value) + assert verdict["verdict"] == "stale" + assert any("source moved" in s for s in verdict["stale_reasons"]) + + +def test_missing_sha_and_old_evidence_stay_stale(evidence): + run(evidence) + value = report(evidence) + value["commit_shas"] = {} + assert cloud_verdict(evidence, value)["verdict"] == "stale" + value = report(evidence) + value["ts"] = "2026-01-01T00:00:00Z" + assert cloud_verdict(evidence, value)["verdict"] == "stale" + + +def test_missing_attempt_cannot_match(evidence): + evidence["artifact"].pop("run_attempt") + evidence["producer"].pop("run_attempt") + assert run(evidence)["validation_outcome"] == "incomplete" + + +def test_non_release_profile_is_not_release_evidence(evidence): + evidence["stage"]["profile"] = "smoke" + run(evidence) + assert cloud_verdict(evidence)["verdict"] == "stale" diff --git a/tests/test_heart_health_wiring.py b/tests/test_heart_health_wiring.py index b7628b4..190c39c 100644 --- a/tests/test_heart_health_wiring.py +++ b/tests/test_heart_health_wiring.py @@ -142,3 +142,11 @@ def test_the_snapshot_folds_in_the_unit_timings_rollup(): state_py = (Path(__file__).resolve().parent.parent / "heart" / "state.py").read_text() assert '"unit_timings": _read_json_or_default(' in state_py assert '"unit_timings.json"' in state_py + + +def test_cloud_reads_validation_before_aggregation(): + _, job = _job() + body = _cloud_step(job)['run'] + assert 'python -m heart.checks.test_run' in body + assert 'python -m heart.checks.cloud_validation' in body + assert _step_index(job, 'Run cloud-safe checks') < _step_index(job, 'Aggregate snapshot')