From 4f7e04e93ffc6cc1703124c46db1ceac5e6b58d2 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 17 Sep 2026 13:24:30 +0000 Subject: [PATCH] verify_install: check F gates on the candidate in a rehearsal, reports the released bootstrap as WARN In a TestPyPI rehearsal (--version) the injected setup cell is verbatim and therefore unpinned: `pip install autonerves --no-deps` and the released `setup_colab.setup()` it imports pull the whole stack back down to the current PyPI release, so check F audited the RELEASED bootstrap and never the candidate. A broken released bootstrap then held Heart RED over the very release carrying its fix (run 35195111347; PyAutoNerves #168/#169 merged but untagged). With --version check F now runs the audit twice: the released bootstrap first, advisory, reported as an `F|WARN|...` row that never changes `ready`; then the venv is re-pinned to the candidate (all five PyAuto packages ==VERSION from the rehearsal index, --no-deps, setup_colab reloaded and its own package list reinstalled as _colab_setup does) and the gate audits that with today's FAIL semantics. The continuous run without --version is unchanged. The COLAB_GATE_AUTONERVES_SRC overlay moves out of the verbatim cell to after the re-pin (or before the single audit). Results table counts n_warn (n_fail still counts only FAIL, so `ready` stays true on WARN); the sidecar folds the released report in as colab_gate.verify_released. readiness stays verdict-neutral on WARN (decision 2026-09-17, recorded in a comment); the dashboard renders a WARN "passed with warnings" section with the detail. Docs, skill text and capabilities.yaml updated; tests for the script text, the sidecar writer, readiness and the dashboard. Refs #229, #228. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01ATSR1eVsUBBK49nfLhb7JQ --- docs/release_validation.md | 12 ++ health_agent/capabilities.yaml | 2 +- heart/checks/verify_install.sh | 271 ++++++++++++++++++++---- heart/dashboard.py | 19 ++ heart/readiness.py | 8 + skills/verify_install/verify_install.md | 25 ++- tests/test_dashboard.py | 43 ++++ tests/test_readiness.py | 41 ++++ tests/test_verify_install_script.py | 184 +++++++++++++++- 9 files changed, 554 insertions(+), 51 deletions(-) diff --git a/docs/release_validation.md b/docs/release_validation.md index e77f05e..ab600f5 100644 --- a/docs/release_validation.md +++ b/docs/release_validation.md @@ -137,6 +137,18 @@ gaps are WARNs carried in the report. Until 2026-09-15 Check F installed the stack **with** dependencies before running the cell, so the bootstrap's misses (`corner`, `optax`, `xxhash`, `blackjax`) were invisible to it and shipped. +In a rehearsal (`--version`) that cell necessarily bootstraps the **released** +stack: it is injected verbatim, and neither its own `pip install autonerves` nor +the released `setup_colab.setup()` it calls is pinned, so pip cannot select the +candidate's dev pre-release. Audited as-is the gate would measure the release +and never the wheels about to ship — and a broken released bootstrap would hold +Heart RED over the release carrying its fix. So since 2026-09-17 Check F audits +the released bootstrap first and reports it as an advisory **`WARN`** row that +never fails the run or moves `ready`, then re-pins the venv to the candidate +(all five PyAuto packages at the rehearsal version, `--no-deps`, `setup_colab` +reloaded and its own package list reinstalled) and gates on that. A continuous +run without `--version` is unchanged: one audit, one verdict. + Check B then requires the unpinned install to be refused as well. That is a separate guarantee, and it was not met until 2026-08-19: `pip install autolens` on 3.11 backtracked to `2026.7.29.1` and installed a stale, JAX-less stack diff --git a/health_agent/capabilities.yaml b/health_agent/capabilities.yaml index 0df16d7..362685a 100644 --- a/health_agent/capabilities.yaml +++ b/health_agent/capabilities.yaml @@ -96,7 +96,7 @@ deep_checks: - id: verify_install impl: heart/checks/verify_install.sh cli: "pyauto-heart verify_install" - measures: "pip, conda, and Colab install-path checks A-F; Check B exact release succeeds on Python 3.12/3.13 and rejects on 3.11; Check E installs historical 2026.2.26.4 on Python 3.12 because its stack has no Python 3.13 dependency wheels; Check F builds a python3.12 venv holding Google's Colab package set (googlecolab/backend-info manifest), runs the injected setup cell verbatim, and fails on any unguarded import or imported-but-declared dependency the --no-deps bootstrap leaves unmet" + measures: "pip, conda, and Colab install-path checks A-F; Check B exact release succeeds on Python 3.12/3.13 and rejects on 3.11; Check E installs historical 2026.2.26.4 on Python 3.12 because its stack has no Python 3.13 dependency wheels; Check F builds a python3.12 venv holding Google's Colab package set (googlecolab/backend-info manifest), runs the injected setup cell verbatim, and fails on any unguarded import or imported-but-declared dependency the --no-deps bootstrap leaves unmet; in a --version rehearsal check F re-pins to the candidate and gates on it, reporting the released bootstrap as an advisory WARN row" gate_role: "RED if last run ready==false; STALE if find-links-only, older than 14d, or never run" - id: url_check impl: heart/checks/url_check.sh diff --git a/heart/checks/verify_install.sh b/heart/checks/verify_install.sh index 85d5e11..a4f8f58 100755 --- a/heart/checks/verify_install.sh +++ b/heart/checks/verify_install.sh @@ -90,13 +90,24 @@ Options: testpypi, or find-links). -h, --help Show this help. +Check F in a rehearsal (--version): + An unpinned `pip install autonerves` can never select a dev pre-release, so + the injected setup cell necessarily bootstraps the RELEASED stack. Check F + therefore audits the released bootstrap first and reports it as an advisory + WARN row, which never changes `ready` and never fails the run; it then + re-pins the venv to the candidate (all five PyAuto packages at VERSION from + the rehearsal index, --no-deps, setup_colab reloaded and its own package list + reinstalled) and gates on that. A continuous run (no --version) is unchanged: + one audit, one verdict. + Environment: COLAB_GATE_AUTONERVES_SRC Check F only, dev/witness use. A path or requirement - installed --no-deps over the released `autonerves` right - after the setup cell's own bootstrap install, so an + installed --no-deps over the released `autonerves`, so an UNRELEASED autonerves/setup_colab.py package list can be - rehearsed against the gate. Never set this in CI: a + rehearsed against the gate. It is applied after the + candidate re-pin in a --version rehearsal, and before the + single audit in a continuous run. Never set this in CI: a release gate must read the wheels about to ship. USAGE } @@ -234,7 +245,11 @@ CONDA_ENVS=() # conda env names to remove at end HEART_STATE_DIR="${HEART_STATE_DIR:-$HOME/.pyauto-heart}" COLAB_MANIFEST_CACHE="$HEART_STATE_DIR/colab_pip_freeze.txt" F_GATE_SEED_JSON="" # colab_gate seed report, folded into the sidecar -F_GATE_VERIFY_JSON="" # colab_gate verify report, folded into the sidecar +F_GATE_VERIFY_JSON="" # colab_gate verify report (the gated facet) +# The advisory RELEASED-bootstrap report, written only on the rehearsal path. +# Declared here so the sidecar writer can reference it even when check F never +# ran (the writer skips a path that does not exist). +F_GATE_VERIFY_RELEASED_JSON="" PIP_INSTALL_TARGET="autolens" PIP_INSTALL_OPTIONAL="autolens[optional]" @@ -726,11 +741,74 @@ check_e() { # The interpreter is python3.12 because Colab is python3.12 — check B's # "missing required interpreter is FAIL" rule applies here too. # +# --- two facets in a rehearsal (--version), one continuously --------------- +# +# The injected setup cell is verbatim, and verbatim means UNPINNED: it runs +# `pip install autonerves --no-deps`, and the released `setup_colab.setup()` it +# then imports installs `autolens autogalaxy autofit autoarray autonerves ... +# --no-deps` unpinned too. pip cannot select a dev pre-release for either, so in +# a TestPyPI rehearsal the cell pulls the whole stack back down to the current +# PyPI release — whatever the seed step pinned. Audited as-is, the gate would +# therefore measure the RELEASED bootstrap and never the candidate, and a +# released bootstrap that is broken would hold Heart RED over the very release +# that carries its fix (chicken-and-egg, 2026-09-17). +# +# So with --version check F runs the audit twice: +# +# RELEASED advisory. What a reader who opens the notebook today actually +# gets. Reported as a WARN row: it never fails the run, because +# the release IS the remedy and grading it YELLOW/RED would block +# it. +# all five PyAuto packages pinned to the candidate from the +# rehearsal index (--no-deps), then `setup_colab` reloaded and its +# own package list reinstalled exactly as `_colab_setup` does — the +# candidate's bootstrap, run for real. +# CANDIDATE the gate. FAIL here is FAIL as it has always been. +# +# Without --version (the continuous run against PyPI) the released bootstrap IS +# the candidate, so there is one audit and nothing changes. +# # COLAB_GATE_AUTONERVES_SRC (dev/witness only): a path or requirement installed -# `--no-deps` immediately after the setup cell's verbatim `pip install -# autonerves --no-deps`. It exists to rehearse an UNRELEASED setup_colab.py — -# the bootstrap package list is the thing this check gates, and until it is on -# PyPI there is no other way to run the gate against a fix. Never set in CI. +# `--no-deps` over the released `autonerves` — after the candidate re-pin in a +# rehearsal, before the single audit in a continuous run. It exists to rehearse +# an UNRELEASED setup_colab.py — the bootstrap package list is the thing this +# check gates, and until it is on PyPI there is no other way to run the gate +# against a fix. Never set in CI. + +# f_report_str : print a string field out of a colab_gate +# report (nested keys walk down), or "" when it is missing or unreadable. +f_report_str() { + python3 -c ' +import json, sys +try: + data = json.load(open(sys.argv[1])) +except Exception: + print("") +else: + for key in sys.argv[2:]: + data = data.get(key) if isinstance(data, dict) else None + print(data if isinstance(data, str) else "") +' "$@" 2>/dev/null +} + +# f_overlay_autonerves_src: the COLAB_GATE_AUTONERVES_SRC overlay for the +# CONTINUOUS path (the rehearsal path does the same install inside the re-pin +# driver, where it has to sit between the candidate pins and the setup_colab +# reload). No-op when the variable is unset, which is every CI run. +f_overlay_autonerves_src() { + if [ -z "${COLAB_GATE_AUTONERVES_SRC:-}" ]; then + return 0 + fi + COLAB_GATE_AUTONERVES_SRC="$COLAB_GATE_AUTONERVES_SRC" python -c ' +import os, subprocess, sys + +_autonerves_src = os.environ["COLAB_GATE_AUTONERVES_SRC"] +print(f"COLAB_GATE_AUTONERVES_SRC set — overlaying {_autonerves_src}") +subprocess.check_call( + [sys.executable, "-m", "pip", "install", "--no-deps", _autonerves_src] +) +' +} check_f() { echo @@ -740,11 +818,15 @@ check_f() { local ws_dir="/tmp/colab_sim_workspace_F_$TS" local seed_json="/tmp/F_gate_seed_$TS.json" local verify_json="/tmp/F_gate_verify_$TS.json" - # The two gate reports are read by the sidecar writer, which runs before + local verify_released_json="/tmp/F_gate_verify_released_$TS.json" + # The gate reports are read by the sidecar writer, which runs before # cleanup — so they can be swept with everything else (and kept by --keep). - ARTEFACTS+=("$venv" "$ws_dir" "$seed_json" "$verify_json") + # The released report only exists on the rehearsal path; the writer skips a + # path that is not a file. + ARTEFACTS+=("$venv" "$ws_dir" "$seed_json" "$verify_json" "$verify_released_json") F_GATE_SEED_JSON="$seed_json" F_GATE_VERIFY_JSON="$verify_json" + F_GATE_VERIFY_RELEASED_JSON="$verify_released_json" # Colab runs Python 3.12. A different interpreter would seed Colab's pins # against the wrong wheels, so a missing python3.12 is FAIL, not SKIP — @@ -833,21 +915,9 @@ else: ) _setup_colab = importlib.import_module("autonerves.setup_colab") -# --- dev/witness override: rehearse an unreleased setup_colab.py --- -# -# The bootstrap package list this check gates lives in autonerves.setup_colab. -# A fix to it cannot be exercised until it is on PyPI unless the local source -# can be laid over the released wheel here, which is what this does. Not set in -# CI: a release gate must read the wheels that are about to ship. -_autonerves_src = os.environ.get("COLAB_GATE_AUTONERVES_SRC") -if _autonerves_src: - print(f"COLAB_GATE_AUTONERVES_SRC set — overlaying {_autonerves_src}") - subprocess.check_call( - [sys.executable, "-m", "pip", "install", "--no-deps", _autonerves_src] - ) - import importlib - _setup_colab = importlib.import_module("autonerves.setup_colab") - _setup_colab = importlib.reload(_setup_colab) +# NB: nothing is laid over the bootstrap here — the cell is the cell. The +# COLAB_GATE_AUTONERVES_SRC overlay and, in a rehearsal, the candidate re-pin +# happen after this driver, so what the cell installs is measurable on its own. if not hasattr(_setup_colab, "setup"): print( @@ -881,25 +951,121 @@ PYEOF fi # --- the gate: what did the --no-deps bootstrap actually leave behind? --- - # Run from inside the cloned workspace so autonerves resolves its config the - # way a notebook cell does (conf reads the cwd). - step "auditing the bootstrapped environment against Colab's package set" + # Every audit runs from inside the cloned workspace so autonerves resolves + # its config the way a notebook cell does (conf reads the cwd). local gate_rc=0 + local gate_detail="" + local released_rc=0 + local released_detail="" + local released_ver="?" + + if [ -n "$TARGET_VERSION" ]; then + # --- facet 1: the RELEASED bootstrap, advisory --- + # This is what the verbatim cell just installed, and what a reader who + # opens the notebook today gets. It is never a FAIL: see the header. + step "auditing the RELEASED bootstrap (advisory — what a reader gets today)" + (cd "$ws_dir" && "$venv/bin/python" "$VERIFY_INSTALL_DIR/colab_gate.py" verify \ + --manifest-cache "$COLAB_MANIFEST_CACHE" \ + --seed-report "$seed_json" \ + --report-json "$verify_released_json" \ + --index-args "${PIP_INDEX_ARGS[@]}") |& tee /tmp/F_gate_released.log + released_rc=${PIPESTATUS[0]} + released_detail=$(f_report_str "$verify_released_json" detail) + released_ver=$(f_report_str "$verify_released_json" packages autonerves) + [ -n "$released_ver" ] || released_ver="?" + if [ -z "$released_detail" ]; then + released_detail="verify could not run (rc=$released_rc)" + fi + + # --- re-pin the venv to the candidate, then re-run its bootstrap --- + step "re-pinning the venv to the candidate $TARGET_VERSION" + cat > /tmp/F_driver_repin.py <<'PYEOF' +"""Re-pin the simulated Colab venv from the release to the candidate. + +The verbatim setup cell has already run: the workspace is cloned and the cwd is +inside it, so `setup()` is NOT called again. What is redone is the part a +rehearsal needs pinned — the five PyAuto wheels, and then the candidate's own +bootstrap package list, installed exactly as `autonerves.setup_colab._colab_setup` +installs it. +""" + +import importlib +import os +import subprocess +import sys + +index_args = sys.argv[1:] # the rehearsal's pip index args, if any +version = os.environ["COLAB_GATE_TARGET_VERSION"] + +pins = [ + f"autonerves=={version}", + f"autofit=={version}", + f"autoarray=={version}", + f"autogalaxy=={version}", + f"autolens=={version}", +] +print(f"re-pinning the PyAuto stack to the candidate {version}") +subprocess.check_call( + [sys.executable, "-m", "pip", "install", "--no-deps", *index_args, *pins] +) + +# --- dev/witness override: rehearse an unreleased setup_colab.py --- +# +# After the pins, so the local source wins over the candidate wheel; before the +# reload, so the package list read below is the one being rehearsed. +_autonerves_src = os.environ.get("COLAB_GATE_AUTONERVES_SRC") +if _autonerves_src: + print(f"COLAB_GATE_AUTONERVES_SRC set — overlaying {_autonerves_src}") + subprocess.check_call( + [sys.executable, "-m", "pip", "install", "--no-deps", _autonerves_src] + ) + +# The wheels landed after this interpreter started, so the import system's +# directory caches predate them. +importlib.invalidate_caches() +import autonerves.setup_colab as sc + +sc = importlib.reload(sc) +if not isinstance(getattr(sc, "_PROJECTS", None), dict) or "autolens" not in sc._PROJECTS: + print( + "ERROR: candidate autonerves.setup_colab exposes no _PROJECTS['autolens'] " + "entry — its bootstrap package list cannot be mirrored" + ) + sys.exit(4) + +packages = sc._PROJECTS["autolens"]["packages"] +subprocess.check_call([sys.executable, "-m", "pip", "install", *packages, "--no-deps"]) +print( + f"re-pin OK: candidate {version} bootstrap package list installed " + f"({len(packages)} packages)" +) +PYEOF + local repin_rc=0 + COLAB_GATE_TARGET_VERSION="$TARGET_VERSION" \ + python /tmp/F_driver_repin.py "${PIP_INDEX_ARGS[@]}" |& tee /tmp/F_repin.log + repin_rc=${PIPESTATUS[0]} + if [ "$repin_rc" -ne 0 ]; then + RESULTS+=("F|FAIL|candidate re-pin rc=$repin_rc") + tail_log "Check F candidate re-pin output" "$(cat /tmp/F_repin.log 2>/dev/null)" + deactivate + return + fi + + step "auditing the CANDIDATE bootstrap (the gate)" + else + # Continuous run: the released bootstrap IS the candidate, so there is + # one audit and the dev/witness overlay (if any) goes in front of it. + f_overlay_autonerves_src + step "auditing the bootstrapped environment against Colab's package set" + fi + (cd "$ws_dir" && "$venv/bin/python" "$VERIFY_INSTALL_DIR/colab_gate.py" verify \ --manifest-cache "$COLAB_MANIFEST_CACHE" \ --seed-report "$seed_json" \ --report-json "$verify_json" \ --index-args "${PIP_INDEX_ARGS[@]}") |& tee /tmp/F_gate.log gate_rc=${PIPESTATUS[0]} - - local gate_detail="" - gate_detail=$(python3 -c ' -import json, sys -try: - print(json.load(open(sys.argv[1])).get("detail", "") or "") -except Exception: - print("") -' "$verify_json" 2>/dev/null) + gate_detail=$(f_report_str "$verify_json" detail) if [ "$gate_rc" -eq 2 ] || { [ "$gate_rc" -ne 0 ] && [ -z "$gate_detail" ]; }; then RESULTS+=("F|FAIL|colab gate: verify could not run (rc=$gate_rc)") @@ -914,6 +1080,13 @@ except Exception: return fi + # The candidate passed. If the bootstrap a reader gets today did not, say + # so — as a WARN row, which prints in the table and travels into the + # sidecar but leaves `ready` (and therefore the Heart verdict) alone. + if [ -n "$TARGET_VERSION" ] && [ "$released_rc" -ne 0 ]; then + RESULTS+=("F|WARN|released Colab bootstrap (autonerves=$released_ver) broken for readers: $released_detail; candidate $TARGET_VERSION passes") + fi + # --- driver part 2: one real notebook cell (the top of imaging/start_here) --- # # Load the SAME dataset the current imaging/start_here.py loads: the bundled @@ -976,18 +1149,24 @@ printf '%-5s %-6s %s\n' "-----" "------" "------" n_fail=0 n_skip=0 +# WARN is advisory and deliberately NOT counted in n_fail: check F's +# released-bootstrap facet reports what a reader gets today, and a broken +# release is not evidence against shipping the candidate that fixes it +# (human decision, 2026-09-17). Only FAIL moves `ready`. +n_warn=0 for row in "${RESULTS[@]}"; do IFS='|' read -r letter status detail <<< "$row" printf '%-5s %-6s %s\n' "$letter" "$status" "$detail" [ "$status" = "FAIL" ] && n_fail=$((n_fail + 1)) [ "$status" = "SKIP" ] && n_skip=$((n_skip + 1)) + [ "$status" = "WARN" ] && n_warn=$((n_warn + 1)) done echo if [ "$n_fail" -eq 0 ]; then - echo "Overall: PASS ($n_skip skipped)" + echo "Overall: PASS ($n_skip skipped, $n_warn warning(s))" else - echo "Overall: FAIL ($n_fail failure(s), $n_skip skipped)" + echo "Overall: FAIL ($n_fail failure(s), $n_skip skipped, $n_warn warning(s))" fi if [ -n "$RESULTS_LOG" ]; then @@ -1016,6 +1195,7 @@ if [ -n "$REPORT_JSON" ]; then VI_REPORT_JSON="$REPORT_JSON" \ VI_INDEX="$vi_index" \ VI_F_GATE_SEED="$F_GATE_SEED_JSON" VI_F_GATE_VERIFY="$F_GATE_VERIFY_JSON" \ + VI_F_GATE_VERIFY_RELEASED="$F_GATE_VERIFY_RELEASED_JSON" \ python3 -c ' import datetime, json, os, sys checks = [] @@ -1027,9 +1207,12 @@ for line in sys.stdin: checks.append({"check": parts[0], "status": parts[1], "detail": parts[2]}) # Check F carries the Colab gate report as a nested "colab_gate" key on its own -# row. Additive only: every existing key and the shape of "checks" (a list of -# {check,status,detail}) are untouched, because heart/readiness.py and -# heart/validate.py parse this file and must keep working unchanged. +# row: "seed", "verify" (the gated facet) and, on a rehearsal, "verify_released" +# (the advisory audit of the bootstrap a reader gets today). It is attached to +# every F row, the WARN one included. Additive only: every existing key and the +# shape of "checks" (a list of {check,status,detail}) are untouched, because +# heart/readiness.py and heart/validate.py parse this file and must keep +# working unchanged. def _read(var): path = os.environ.get(var) or "" if not path or not os.path.isfile(path): @@ -1041,7 +1224,9 @@ def _read(var): return None gate = {k: v for k, v in (("seed", _read("VI_F_GATE_SEED")), - ("verify", _read("VI_F_GATE_VERIFY"))) if v is not None} + ("verify", _read("VI_F_GATE_VERIFY")), + ("verify_released", _read("VI_F_GATE_VERIFY_RELEASED"))) + if v is not None} if gate: for entry in checks: if entry["check"] == "F": diff --git a/heart/dashboard.py b/heart/dashboard.py index 92610bc..0957f73 100644 --- a/heart/dashboard.py +++ b/heart/dashboard.py @@ -961,6 +961,25 @@ def build_board( f"development-only (find-links; last run {vi.get('ts', '?')})", [], )) + elif any(str(c.get("status")).upper() == "WARN" + for c in (vi.get("checks") or []) if isinstance(c, dict)): + # Advisory rows: the run passed, but a check reported something a + # human should see (check F's released-bootstrap facet in a + # rehearsal). Verdict-neutral — readiness never reads these — so the + # dashboard is the only place they surface. + warns = [c for c in (vi.get("checks") or []) + if isinstance(c, dict) and str(c.get("status")).upper() == "WARN"] + letters = list(dict.fromkeys(str(c.get("check")) for c in warns)) + details = [f"{c.get('check')}: {str(c.get('detail') or '')[:200]}" + for c in warns] + sections.append(Section( + "verify_install", + "Install verify", + WARN, + f"passed with warnings ({index}; {', '.join(letters)}) " + f"({vi.get('ts', '?')})", + details, + )) else: sections.append(Section("verify_install", "Install verify", OK, f"passed ({index}; last run {vi.get('ts', '?')})", [])) diff --git a/heart/readiness.py b/heart/readiness.py index 25fe9fc..9ca8c19 100644 --- a/heart/readiness.py +++ b/heart/readiness.py @@ -508,6 +508,14 @@ def scope_local(msg: str, key: str) -> None: # cannot satisfy a release gate: a pass stays STALE until PyPI/TestPyPI is # exercised. A failure remains RED because an exact local artifact failing # its install contract is still actionable evidence. + # + # A WARN check row is verdict-neutral, by decision 2026-09-17. The only + # producer today is check F's released-bootstrap facet in a --version + # rehearsal: it reports that the Colab bootstrap a reader gets from the + # CURRENT release is broken, which is not evidence against shipping the + # candidate — the release is the remedy, and grading it YELLOW would block + # the very release that fixes it. Only `ready is False` (i.e. a FAIL row) + # is RED; WARN rows travel in the sidecar and render on the dashboard. vi = snapshot.get("verify_install") if isinstance(vi, dict) and "ready" in vi: index = str(vi.get("index") or "index unknown") diff --git a/skills/verify_install/verify_install.md b/skills/verify_install/verify_install.md index 66bb80d..cf4a373 100644 --- a/skills/verify_install/verify_install.md +++ b/skills/verify_install/verify_install.md @@ -47,16 +47,33 @@ It does **not** cover: always names the manifest's source (`live`, `cache` or the vendored snapshot) and date so the evidence can be dated. +**In a `--version` rehearsal the gate audits the candidate, not the release.** +The injected setup cell is verbatim, and verbatim means unpinned: an unpinned +`pip install autonerves` can never select a dev pre-release, and the released +`setup_colab.setup()` then reinstalls the whole stack `--no-deps` unpinned too, +so the cell pulls the venv back down to the current PyPI release whatever the +seed step pinned. Check F therefore audits that first and reports it as an +advisory **`WARN`** row — a broken released bootstrap is not evidence against +shipping the candidate, because the release is the remedy — then **re-pins** the +venv to the candidate (`autonerves`, `autofit`, `autoarray`, `autogalaxy`, +`autolens` all `==` from the rehearsal index, `--no-deps`, then +`setup_colab` reloaded and its own package list reinstalled exactly as +`_colab_setup` does) and gates on that. A `WARN` row never changes `ready`, so +it never moves the Heart verdict; it prints in the table, travels into the +sidecar and renders on the dashboard. A continuous run without `--version` is +unchanged: the released bootstrap *is* the candidate, so there is one audit. + The manifest is fetched live, cached at `$HEART_STATE_DIR/colab_pip_freeze.txt`, and falls back to `heart/checks/colab_pip_freeze.snapshot.txt` when both are unavailable. Deliberate exemptions live in `heart/config/colab_gate.yaml` (`accepted_missing`), each with a written reason that travels into the report. **`COLAB_GATE_AUTONERVES_SRC`** (development / witness runs only) installs a -path or requirement `--no-deps` over the released `autonerves` immediately -after the setup cell's own bootstrap install. It exists because the package -list the gate measures lives in `autonerves/setup_colab.py`, so a fix to it -cannot otherwise be rehearsed until it is on PyPI: +path or requirement `--no-deps` over the installed `autonerves` — after the +candidate re-pin in a `--version` rehearsal, and before the single audit in a +continuous run. It exists because the package list the gate measures lives in +`autonerves/setup_colab.py`, so a fix to it cannot otherwise be rehearsed until +it is on PyPI: ```bash COLAB_GATE_AUTONERVES_SRC=/path/to/PyAutoNerves pyauto-heart verify_install F diff --git a/tests/test_dashboard.py b/tests/test_dashboard.py index 5ae6fbb..a9d0f7b 100644 --- a/tests/test_dashboard.py +++ b/tests/test_dashboard.py @@ -210,6 +210,49 @@ def test_test_run_failing_scripts_listed_in_details(): for d in section.details) +def test_install_warn_row_renders_warn_with_detail(): + """A passing run carrying an advisory row renders WARN, not OK. + + Readiness is deliberately blind to WARN rows (decision 2026-09-17), so the + dashboard is the only place check F's released-bootstrap facet surfaces. + """ + detail = ( + "released Colab bootstrap (autonerves=2026.9.15.1) broken for readers: " + "colab gate: corner (autofit/plot.py:95); candidate 2026.9.17.1.dev1 passes" + ) + snap = make_snapshot(verify_install={ + "ready": True, + "index": "testpypi", + "ts": TS, + "checks": [ + {"check": "F", "status": "WARN", "detail": detail}, + {"check": "F", "status": "PASS", "detail": "Colab manifest live"}, + ], + }) + + board = dashboard.build_board(snap, make_verdict(), now=FRESH_NOW) + section = next(s for s in board.sections if s.key == "verify_install") + + assert section.state == dashboard.WARN + assert "passed with warnings (testpypi; F)" in section.summary + assert any("released Colab bootstrap" in d for d in section.details) + + +def test_install_fail_row_still_renders_fail(): + snap = make_snapshot(verify_install={ + "ready": False, + "index": "testpypi", + "ts": TS, + "checks": [{"check": "F", "status": "FAIL", "detail": "colab gate: corner"}], + }) + + board = dashboard.build_board(snap, make_verdict("red", 60), now=FRESH_NOW) + section = next(s for s in board.sections if s.key == "verify_install") + + assert section.state == dashboard.FAIL + assert section.summary.startswith("FAILED") + + def test_release_install_pass_names_the_index(): snap = make_snapshot(verify_install={ "ready": True, diff --git a/tests/test_readiness.py b/tests/test_readiness.py index fe1353b..07fc3b4 100644 --- a/tests/test_readiness.py +++ b/tests/test_readiness.py @@ -227,6 +227,47 @@ def test_install_verification_failed_is_red(): assert v["score"] == 60 +def test_install_verification_warn_row_is_verdict_neutral(): + """Check F's released-bootstrap facet (decision 2026-09-17). + + In a rehearsal check F reports the Colab bootstrap a reader gets from the + CURRENT release as a WARN row. A broken release is not evidence against + shipping the candidate that fixes it, so the row moves nothing. + """ + snap = make_snapshot(verify_install={ + "ready": True, + "ts": "2026-06-01T00:00:00+00:00", + "index": "testpypi", + "checks": [ + {"check": "F", "status": "WARN", + "detail": "released Colab bootstrap (autonerves=2026.9.15.1) broken " + "for readers: colab gate: corner (autofit/plot.py:95)"}, + {"check": "F", "status": "PASS"}, + ], + }) + v = compute(snap) + assert v["verdict"] == "green" + assert not any("install" in r for r in v["reasons"]) + + +def test_install_verification_fail_beside_warn_is_still_red(): + """The WARN row is advisory; a FAIL row beside it still blocks.""" + snap = make_snapshot(verify_install={ + "ready": False, + "ts": "2026-06-01T00:00:00+00:00", + "index": "testpypi", + "checks": [ + {"check": "F", "status": "WARN", "detail": "released Colab bootstrap ..."}, + {"check": "B", "status": "FAIL", "detail": "pip install failed"}, + ], + }) + v = compute(snap) + assert v["verdict"] == "red" + assert any("install verification FAILED" in r and "B" in r for r in v["red_reasons"]) + # The advisory row is not mistaken for a failure: only B is named. + assert any(r.endswith("checks B)") for r in v["red_reasons"]) + + def test_install_verification_stale_is_stale_tier(): snap = make_snapshot(verify_install={ "ready": True, "ts": "2026-05-01T00:00:00+00:00", # ~31d before snapshot ts diff --git a/tests/test_verify_install_script.py b/tests/test_verify_install_script.py index e375ce4..308344f 100644 --- a/tests/test_verify_install_script.py +++ b/tests/test_verify_install_script.py @@ -340,8 +340,10 @@ def test_check_f_seeds_and_verifies_through_colab_gate(): assert '"$VERIFY_INSTALL_DIR/colab_gate.py" seed' in body assert '"$VERIFY_INSTALL_DIR/colab_gate.py" verify' in body # Run with the simulated venv's interpreter, or importlib.metadata and the - # import probe would see the host environment instead. - assert body.count('"$venv/bin/python" "$VERIFY_INSTALL_DIR/colab_gate.py"') == 2 + # import probe would see the host environment instead. Three invocations: + # seed, the advisory released audit (rehearsal only), and the gated audit + # the continuous and candidate paths share. + assert body.count('"$venv/bin/python" "$VERIFY_INSTALL_DIR/colab_gate.py"') == 3 assert '--manifest-cache "$COLAB_MANIFEST_CACHE"' in body assert (ROOT / "heart/checks/colab_gate.py").is_file() assert (ROOT / "heart/checks/colab_pip_freeze.snapshot.txt").is_file() @@ -379,17 +381,120 @@ def test_check_f_preserves_the_skip_exit_code(): def test_autonerves_source_override_is_wired_and_documented(): + text = SCRIPT.read_text() body = check_f_body() help_result = run("--help") + # Read on both paths: inside the re-pin driver (rehearsal — it has to sit + # between the candidate pins and the setup_colab reload) and in the shared + # f_overlay_autonerves_src step the continuous path runs in front of its + # single audit. assert 'os.environ.get("COLAB_GATE_AUTONERVES_SRC")' in body - assert '"--no-deps", _autonerves_src' in body + assert "f_overlay_autonerves_src() {" in text + assert 'os.environ["COLAB_GATE_AUTONERVES_SRC"]' in text + assert text.count('"--no-deps", _autonerves_src') == 2 + assert " f_overlay_autonerves_src\n" in body + # It is no longer laid over the verbatim setup cell's own bootstrap, so + # what that cell installs stays measurable on its own. + setup_driver = body[body.index("F_driver_setup.py") : body.index("setup_rc=0")] + assert '"--no-deps", _autonerves_src' not in setup_driver + assert 'os.environ.get("COLAB_GATE_AUTONERVES_SRC")' not in setup_driver assert "COLAB_GATE_AUTONERVES_SRC" in help_result.stdout assert "COLAB_GATE_AUTONERVES_SRC" in ( ROOT / "skills/verify_install/verify_install.md" ).read_text() +def test_help_documents_the_rehearsal_re_pin_and_warn_row(): + """A reader of --help must learn that a rehearsal gates on the candidate.""" + result = run("--help") + + assert result.returncode == 0 + assert "WARN" in result.stdout + assert "re-pin" in result.stdout + + +def test_check_f_rehearsal_re_pins_to_the_candidate_and_audits_both_facets(): + """The chicken-and-egg fix (2026-09-17). + + The injected setup cell is verbatim and therefore unpinned, so in a + rehearsal it bootstraps the RELEASED stack. Check F audits that advisorily, + re-pins to the candidate, and gates on the candidate. + """ + body = check_f_body() + + # The advisory facet exists, writes its own report, and is rehearsal-only. + driver = body.index("F_driver_setup.py") + guard = body.index('if [ -n "$TARGET_VERSION" ]; then', driver) + released = body.index("$verify_released_json", guard) + assert driver < guard < released + assert 'local verify_released_json="/tmp/F_gate_verify_released_$TS.json"' in body + assert 'F_GATE_VERIFY_RELEASED_JSON="$verify_released_json"' in body + assert '--report-json "$verify_released_json"' in body + + # The re-pin: all five PyAuto packages at the candidate, then the + # candidate's own bootstrap package list, mirroring _colab_setup. + for package in ("autonerves", "autofit", "autoarray", "autogalaxy", "autolens"): + assert f'f"{package}=={{version}}"' in body + assert '"pip", "install", "--no-deps", *index_args, *pins' in body + assert "importlib.reload(sc)" in body + assert 'sc._PROJECTS["autolens"]["packages"]' in body + assert '"pip", "install", *packages, "--no-deps"' in body + assert "re-pin OK: candidate " in body + # setup() is NOT called again — the workspace is cloned and cwd has moved. + assert body.count("_setup_colab.setup(") == 1 + assert 'RESULTS+=("F|FAIL|candidate re-pin rc=$repin_rc")' in body + + # The released facet reports a WARN row, never a FAIL. + assert 'RESULTS+=("F|WARN|released Colab bootstrap ' in body + assert "F|FAIL|released" not in body + + +def test_check_f_continuous_path_runs_one_verify(): + """No --version: the released bootstrap IS the candidate — one audit.""" + body = check_f_body() + + # One gated audit, shared by the continuous and candidate paths, plus the + # rehearsal-only advisory one. + assert body.count('colab_gate.py" verify') == 2 + assert body.count('--report-json "$verify_json"') == 1 + assert body.count('--report-json "$verify_released_json"') == 1 + # The continuous branch is the `else` of the rehearsal guard, and applies + # the dev/witness overlay in front of its single audit. + guard = body.index('if [ -n "$TARGET_VERSION" ]; then', body.index("F_driver_setup.py")) + otherwise = body.index(" else", guard) + gate = body.index('--report-json "$verify_json"', otherwise) + assert guard < body.index("f_overlay_autonerves_src\n", otherwise) < gate + # Today's FAIL semantics are untouched. + assert 'RESULTS+=("F|FAIL|colab gate: verify could not run (rc=$gate_rc)")' in body + assert 'RESULTS+=("F|FAIL|$gate_detail")' in body + assert 'RESULTS+=("F|PASS|$gate_detail")' in body + + +def test_warn_rows_are_counted_but_never_fail_the_run(): + text = SCRIPT.read_text() + + assert 'n_warn=$((n_warn + 1))' in text + assert '[ "$status" = "WARN" ] && n_warn=$((n_warn + 1))' in text + assert 'echo "Overall: PASS ($n_skip skipped, $n_warn warning(s))"' in text + assert ( + 'echo "Overall: FAIL ($n_fail failure(s), $n_skip skipped, ' + '$n_warn warning(s))"' in text + ) + # ready is n_fail-driven, so a WARN row leaves it true. + assert 'if [ "$n_fail" -eq 0 ]; then ready_bool=true; else ready_bool=false; fi' in text + + +def test_sidecar_nests_the_released_gate_report(): + text = SCRIPT.read_text() + + assert 'VI_F_GATE_VERIFY_RELEASED="$F_GATE_VERIFY_RELEASED_JSON"' in text + assert '"verify_released"' in text + assert '_read("VI_F_GATE_VERIFY_RELEASED")' in text + # Declared globally, so the writer is safe when check F never ran. + assert 'F_GATE_VERIFY_RELEASED_JSON=""' in text + + def test_sidecar_nests_the_gate_report_under_check_f_without_changing_its_shape(): text = SCRIPT.read_text() @@ -490,3 +595,76 @@ def test_sidecar_still_parses_through_readiness_with_the_gate_report(tmp_path): "install verification FAILED" in reason and "F" in reason for reason in result["red_reasons"] ) + + +def test_sidecar_warn_row_keeps_ready_true_and_readiness_not_red(tmp_path): + """A rehearsal's two F rows: the advisory WARN, then the gated PASS. + + The WARN row must travel end to end — writer, sidecar, readiness — without + moving `ready` or the verdict (decision 2026-09-17). + """ + import json + import os + + from heart import readiness + + seed = tmp_path / "seed.json" + verify = tmp_path / "verify.json" + released = tmp_path / "verify_released.json" + seed.write_text(json.dumps({"phase": "seed", "manifest_source": "live"})) + verify.write_text(json.dumps({ + "phase": "verify", + "ok": True, + "detail": "Colab manifest live 2026-09-17; 61 Colab-provided, 9 extras", + "packages": {"autonerves": "2026.9.17.1.dev1"}, + })) + released.write_text(json.dumps({ + "phase": "verify", + "ok": False, + "detail": "colab gate: corner (autofit/plot.py:95)", + "packages": {"autonerves": "2026.9.15.1"}, + })) + out = tmp_path / "verify_install.json" + + env = dict(os.environ) + env.update({ + "VI_READY": "true", + "VI_VERSION": "2026.9.17.1.dev1", + "VI_CHECK_B_VERSION": "2026.9.17.1.dev1", + "VI_REPORT_JSON": str(out), + "VI_INDEX": "testpypi", + "VI_F_GATE_SEED": str(seed), + "VI_F_GATE_VERIFY": str(verify), + "VI_F_GATE_VERIFY_RELEASED": str(released), + }) + rows = ( + "A|PASS|ok\n" + "F|WARN|released Colab bootstrap (autonerves=2026.9.15.1) broken for " + "readers: colab gate: corner (autofit/plot.py:95); candidate " + "2026.9.17.1.dev1 passes\n" + "F|PASS|Colab manifest live 2026-09-17\n" + ) + result = subprocess.run( + ["python3", "-c", sidecar_writer_source()], + input=rows, capture_output=True, text=True, env=env, + ) + assert result.returncode == 0, result.stderr + + sidecar = json.loads(out.read_text()) + assert [c["status"] for c in sidecar["checks"]] == ["PASS", "WARN", "PASS"] + assert sidecar["ready"] is True + + # The gate report rides on both F rows, both facets present. + for row in sidecar["checks"][1:]: + assert row["check"] == "F" + assert row["colab_gate"]["verify"]["ok"] is True + assert row["colab_gate"]["verify_released"]["ok"] is False + assert row["colab_gate"]["verify_released"]["packages"]["autonerves"] == ( + "2026.9.15.1" + ) + assert "colab_gate" not in sidecar["checks"][0] + + # And the verdict is untouched by the WARN row. + verdict = readiness.compute({"ts": sidecar["ts"], "verify_install": sidecar}) + assert not any("install" in reason for reason in verdict["red_reasons"]) + assert not any("install" in reason for reason in verdict["yellow_reasons"])