Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions docs/release_validation.md
Original file line number Diff line number Diff line change
Expand Up @@ -137,6 +137,18 @@ gaps are WARNs carried in the report. Until 2026-09-15 Check F installed the
stack **with** dependencies before running the cell, so the bootstrap's misses
(`corner`, `optax`, `xxhash`, `blackjax`) were invisible to it and shipped.

In a rehearsal (`--version`) that cell necessarily bootstraps the **released**
stack: it is injected verbatim, and neither its own `pip install autonerves` nor
the released `setup_colab.setup()` it calls is pinned, so pip cannot select the
candidate's dev pre-release. Audited as-is the gate would measure the release
and never the wheels about to ship — and a broken released bootstrap would hold
Heart RED over the release carrying its fix. So since 2026-09-17 Check F audits
the released bootstrap first and reports it as an advisory **`WARN`** row that
never fails the run or moves `ready`, then re-pins the venv to the candidate
(all five PyAuto packages at the rehearsal version, `--no-deps`, `setup_colab`
reloaded and its own package list reinstalled) and gates on that. A continuous
run without `--version` is unchanged: one audit, one verdict.

Check B then requires the unpinned install to be refused as well. That is a
separate guarantee, and it was not met until 2026-08-19: `pip install autolens`
on 3.11 backtracked to `2026.7.29.1` and installed a stale, JAX-less stack
Expand Down
2 changes: 1 addition & 1 deletion health_agent/capabilities.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -96,7 +96,7 @@ deep_checks:
- id: verify_install
impl: heart/checks/verify_install.sh
cli: "pyauto-heart verify_install"
measures: "pip, conda, and Colab install-path checks A-F; Check B exact release succeeds on Python 3.12/3.13 and rejects on 3.11; Check E installs historical 2026.2.26.4 on Python 3.12 because its stack has no Python 3.13 dependency wheels; Check F builds a python3.12 venv holding Google's Colab package set (googlecolab/backend-info manifest), runs the injected setup cell verbatim, and fails on any unguarded import or imported-but-declared dependency the --no-deps bootstrap leaves unmet"
measures: "pip, conda, and Colab install-path checks A-F; Check B exact release succeeds on Python 3.12/3.13 and rejects on 3.11; Check E installs historical 2026.2.26.4 on Python 3.12 because its stack has no Python 3.13 dependency wheels; Check F builds a python3.12 venv holding Google's Colab package set (googlecolab/backend-info manifest), runs the injected setup cell verbatim, and fails on any unguarded import or imported-but-declared dependency the --no-deps bootstrap leaves unmet; in a --version rehearsal check F re-pins to the candidate and gates on it, reporting the released bootstrap as an advisory WARN row"
gate_role: "RED if last run ready==false; STALE if find-links-only, older than 14d, or never run"
- id: url_check
impl: heart/checks/url_check.sh
Expand Down
271 changes: 228 additions & 43 deletions heart/checks/verify_install.sh

Large diffs are not rendered by default.

19 changes: 19 additions & 0 deletions heart/dashboard.py
Original file line number Diff line number Diff line change
Expand Up @@ -961,6 +961,25 @@ def build_board(
f"development-only (find-links; last run {vi.get('ts', '?')})",
[],
))
elif any(str(c.get("status")).upper() == "WARN"
for c in (vi.get("checks") or []) if isinstance(c, dict)):
# Advisory rows: the run passed, but a check reported something a
# human should see (check F's released-bootstrap facet in a
# rehearsal). Verdict-neutral — readiness never reads these — so the
# dashboard is the only place they surface.
warns = [c for c in (vi.get("checks") or [])
if isinstance(c, dict) and str(c.get("status")).upper() == "WARN"]
letters = list(dict.fromkeys(str(c.get("check")) for c in warns))
details = [f"{c.get('check')}: {str(c.get('detail') or '')[:200]}"
for c in warns]
sections.append(Section(
"verify_install",
"Install verify",
WARN,
f"passed with warnings ({index}; {', '.join(letters)}) "
f"({vi.get('ts', '?')})",
details,
))
else:
sections.append(Section("verify_install", "Install verify", OK,
f"passed ({index}; last run {vi.get('ts', '?')})", []))
Expand Down
8 changes: 8 additions & 0 deletions heart/readiness.py
Original file line number Diff line number Diff line change
Expand Up @@ -508,6 +508,14 @@ def scope_local(msg: str, key: str) -> None:
# cannot satisfy a release gate: a pass stays STALE until PyPI/TestPyPI is
# exercised. A failure remains RED because an exact local artifact failing
# its install contract is still actionable evidence.
#
# A WARN check row is verdict-neutral, by decision 2026-09-17. The only
# producer today is check F's released-bootstrap facet in a --version
# rehearsal: it reports that the Colab bootstrap a reader gets from the
# CURRENT release is broken, which is not evidence against shipping the
# candidate — the release is the remedy, and grading it YELLOW would block
# the very release that fixes it. Only `ready is False` (i.e. a FAIL row)
# is RED; WARN rows travel in the sidecar and render on the dashboard.
vi = snapshot.get("verify_install")
if isinstance(vi, dict) and "ready" in vi:
index = str(vi.get("index") or "index unknown")
Expand Down
25 changes: 21 additions & 4 deletions skills/verify_install/verify_install.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,16 +47,33 @@ It does **not** cover:
always names the manifest's source (`live`, `cache` or the vendored
snapshot) and date so the evidence can be dated.

**In a `--version` rehearsal the gate audits the candidate, not the release.**
The injected setup cell is verbatim, and verbatim means unpinned: an unpinned
`pip install autonerves` can never select a dev pre-release, and the released
`setup_colab.setup()` then reinstalls the whole stack `--no-deps` unpinned too,
so the cell pulls the venv back down to the current PyPI release whatever the
seed step pinned. Check F therefore audits that first and reports it as an
advisory **`WARN`** row — a broken released bootstrap is not evidence against
shipping the candidate, because the release is the remedy — then **re-pins** the
venv to the candidate (`autonerves`, `autofit`, `autoarray`, `autogalaxy`,
`autolens` all `==<version>` from the rehearsal index, `--no-deps`, then
`setup_colab` reloaded and its own package list reinstalled exactly as
`_colab_setup` does) and gates on that. A `WARN` row never changes `ready`, so
it never moves the Heart verdict; it prints in the table, travels into the
sidecar and renders on the dashboard. A continuous run without `--version` is
unchanged: the released bootstrap *is* the candidate, so there is one audit.

The manifest is fetched live, cached at `$HEART_STATE_DIR/colab_pip_freeze.txt`,
and falls back to `heart/checks/colab_pip_freeze.snapshot.txt` when both are
unavailable. Deliberate exemptions live in `heart/config/colab_gate.yaml`
(`accepted_missing`), each with a written reason that travels into the report.

**`COLAB_GATE_AUTONERVES_SRC`** (development / witness runs only) installs a
path or requirement `--no-deps` over the released `autonerves` immediately
after the setup cell's own bootstrap install. It exists because the package
list the gate measures lives in `autonerves/setup_colab.py`, so a fix to it
cannot otherwise be rehearsed until it is on PyPI:
path or requirement `--no-deps` over the installed `autonerves` — after the
candidate re-pin in a `--version` rehearsal, and before the single audit in a
continuous run. It exists because the package list the gate measures lives in
`autonerves/setup_colab.py`, so a fix to it cannot otherwise be rehearsed until
it is on PyPI:

```bash
COLAB_GATE_AUTONERVES_SRC=/path/to/PyAutoNerves pyauto-heart verify_install F
Expand Down
43 changes: 43 additions & 0 deletions tests/test_dashboard.py
Original file line number Diff line number Diff line change
Expand Up @@ -210,6 +210,49 @@ def test_test_run_failing_scripts_listed_in_details():
for d in section.details)


def test_install_warn_row_renders_warn_with_detail():
"""A passing run carrying an advisory row renders WARN, not OK.

Readiness is deliberately blind to WARN rows (decision 2026-09-17), so the
dashboard is the only place check F's released-bootstrap facet surfaces.
"""
detail = (
"released Colab bootstrap (autonerves=2026.9.15.1) broken for readers: "
"colab gate: corner (autofit/plot.py:95); candidate 2026.9.17.1.dev1 passes"
)
snap = make_snapshot(verify_install={
"ready": True,
"index": "testpypi",
"ts": TS,
"checks": [
{"check": "F", "status": "WARN", "detail": detail},
{"check": "F", "status": "PASS", "detail": "Colab manifest live"},
],
})

board = dashboard.build_board(snap, make_verdict(), now=FRESH_NOW)
section = next(s for s in board.sections if s.key == "verify_install")

assert section.state == dashboard.WARN
assert "passed with warnings (testpypi; F)" in section.summary
assert any("released Colab bootstrap" in d for d in section.details)


def test_install_fail_row_still_renders_fail():
snap = make_snapshot(verify_install={
"ready": False,
"index": "testpypi",
"ts": TS,
"checks": [{"check": "F", "status": "FAIL", "detail": "colab gate: corner"}],
})

board = dashboard.build_board(snap, make_verdict("red", 60), now=FRESH_NOW)
section = next(s for s in board.sections if s.key == "verify_install")

assert section.state == dashboard.FAIL
assert section.summary.startswith("FAILED")


def test_release_install_pass_names_the_index():
snap = make_snapshot(verify_install={
"ready": True,
Expand Down
41 changes: 41 additions & 0 deletions tests/test_readiness.py
Original file line number Diff line number Diff line change
Expand Up @@ -227,6 +227,47 @@ def test_install_verification_failed_is_red():
assert v["score"] == 60


def test_install_verification_warn_row_is_verdict_neutral():
"""Check F's released-bootstrap facet (decision 2026-09-17).

In a rehearsal check F reports the Colab bootstrap a reader gets from the
CURRENT release as a WARN row. A broken release is not evidence against
shipping the candidate that fixes it, so the row moves nothing.
"""
snap = make_snapshot(verify_install={
"ready": True,
"ts": "2026-06-01T00:00:00+00:00",
"index": "testpypi",
"checks": [
{"check": "F", "status": "WARN",
"detail": "released Colab bootstrap (autonerves=2026.9.15.1) broken "
"for readers: colab gate: corner (autofit/plot.py:95)"},
{"check": "F", "status": "PASS"},
],
})
v = compute(snap)
assert v["verdict"] == "green"
assert not any("install" in r for r in v["reasons"])


def test_install_verification_fail_beside_warn_is_still_red():
"""The WARN row is advisory; a FAIL row beside it still blocks."""
snap = make_snapshot(verify_install={
"ready": False,
"ts": "2026-06-01T00:00:00+00:00",
"index": "testpypi",
"checks": [
{"check": "F", "status": "WARN", "detail": "released Colab bootstrap ..."},
{"check": "B", "status": "FAIL", "detail": "pip install failed"},
],
})
v = compute(snap)
assert v["verdict"] == "red"
assert any("install verification FAILED" in r and "B" in r for r in v["red_reasons"])
# The advisory row is not mistaken for a failure: only B is named.
assert any(r.endswith("checks B)") for r in v["red_reasons"])


def test_install_verification_stale_is_stale_tier():
snap = make_snapshot(verify_install={
"ready": True, "ts": "2026-05-01T00:00:00+00:00", # ~31d before snapshot ts
Expand Down
Loading
Loading