From 079b1824bc76f0de07e9cafc1d925db5c8289114 Mon Sep 17 00:00:00 2001 From: Jammy2211 Date: Sat, 26 Sep 2026 22:00:45 +0100 Subject: [PATCH] Nerves board: flag library config keys no library code reads (#174) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit scripts/board.py scans every library package's .py (and autonerves/) with an AST walk for config lookups — literal conf.instance[...] chains (multi-line, .get("k")), non-literal subscripts as wildcards, sections bound to a local and indexed later (closures included) and should_output("name") — and classes every library settings key used / section-read / unused against the reads of the whole stack (priors exempt; skipped if any library is missing). Rendered as an unused chip on the key and per-file counts on the repo page, a "Possibly unused config keys" index section grouped by library with GitHub links, the class in board.json and the key index, and one info item per library in state.json — never yellow. SOURCES gains each library's package dir; the workflow's sparse checkout also fetches /**/*.py. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01SbKQQHRRgm2b69aT9t7771 --- .github/workflows/nerves_board.yml | 13 +- AGENTS.md | 22 +- README.md | 20 +- scripts/board.py | 431 +++++++++++++++++++++++++++-- test_autonerves/test_board.py | 183 +++++++++++- 5 files changed, 638 insertions(+), 31 deletions(-) diff --git a/.github/workflows/nerves_board.yml b/.github/workflows/nerves_board.yml index 4566592..a1d2cb3 100644 --- a/.github/workflows/nerves_board.yml +++ b/.github/workflows/nerves_board.yml @@ -13,8 +13,9 @@ name: Nerves Board # * dashboard.md (the markdown mirror). # # Read-only: the board edits nothing. Truth is a sparse checkout of each -# source's config folder (blobless clone, only `/**` checked out) -# taken fresh on every run; slugs come from the Mind's body map. +# source's config folder (blobless clone, only `/**` checked out; +# for a library also `/**/*.py`, scanned for the config keys its +# code reads) taken fresh on every run; slugs come from the Mind's body map. on: schedule: @@ -63,7 +64,7 @@ jobs: # and the board can never disagree about what is collected. A source # that fails to clone is simply absent — the board reports it under # "unavailable this render" rather than pretending it was read. - - name: Sparse-clone every config folder + - name: Sparse-clone every config folder (and library package code) run: | python3 - <<'PY' import importlib.util, subprocess, yaml @@ -74,6 +75,10 @@ jobs: repos = yaml.safe_load(Path("PyAutoMind/repos.yaml").read_text())["repos"] for src in board.SOURCES: name, cfg = src["repo"], src["config"] + # a library's package .py files too: the board scans them for + # the config keys the code reads ("possibly unused keys") + patterns = [f"{cfg}/**"] + ( + [f"{src['package']}/**/*.py"] if src.get("package") else []) home = (repos.get(name) or {}).get("github") if not home: print(f"::warning::{name}: no github slug in repos.yaml") @@ -85,7 +90,7 @@ jobs: f"https://github.com/{home}", str(dest)], check=True, timeout=300) subprocess.run(["git", "-C", str(dest), "sparse-checkout", - "set", "--no-cone", f"{cfg}/**"], + "set", "--no-cone", *patterns], check=True, timeout=60) subprocess.run(["git", "-C", str(dest), "checkout", "--quiet"], check=True, timeout=300) diff --git a/AGENTS.md b/AGENTS.md index 1ed8fc6..91eeee5 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -84,10 +84,24 @@ NUMBA_CACHE_DIR=/tmp/numba_cache MPLCONFIGDIR=/tmp/matplotlib python -m pytest t (): every YAML file under each library's `/config/` and each workspace's `config/`, with keys, comments, source, prior tables, the workspace → library override map, the -`PYAUTO_*` env vars, and a `state.json` cockpit feed (contract owned by -`PyAutoBrain/board/_state.py`). The config sources are the `SOURCES` table at -the top of the script (repo, config dir, kind, library lookup stack) — add a -new library or workspace there. It is stdlib + PyYAML, is **not** packaged +`PYAUTO_*` env vars, the "possibly unused config keys" scan, and a +`state.json` cockpit feed (contract owned by `PyAutoBrain/board/_state.py`). +The config sources are the `SOURCES` table at the top of the script (repo, +config dir, kind, library package dir, library lookup stack) — add a new +library or workspace there. + +**Unused-key classification** (`scan_lookups` → `classify_files`): an AST +scan of every library package's `.py` (and `autonerves/`) collects the config +paths the code reads — literal `conf.instance[...]` / `instance[...]` chains +(`.get("k")` counts; a non-literal subscript ends the chain as a wildcard), +sections bound to a local and indexed later in the function or a closure, +and `should_output("name")` (`HELPER_READS`); `logging.yaml` is loaded whole +(`WHOLESALE_FILES`). Each library settings key (priors exempt) is `used`, +`section-read` or `unused` against the reads of the *whole* stack. It is +skipped entirely if any library's config or package dir is missing (a partial +scan would call that library's reads unused). It is static and untrusted: +unused keys are one `info` item per library in `state.json`, never yellow — +add a new lookup style to `lookups_from` rather than colouring the feed. It is stdlib + PyYAML, is **not** packaged (`scripts/` is excluded in `pyproject.toml` / `MANIFEST.in`), and is tested by `test_autonerves/test_board.py`. Published by `.github/workflows/nerves_board.yml` (daily + dispatch). Local run: diff --git a/README.md b/README.md index 3ee5ab3..f117f0e 100644 --- a/README.md +++ b/README.md @@ -79,13 +79,31 @@ workspace's `config/` that overrides them. array → fit). Keys whose value differs, keys no library defines (*orphans*) and keys that fall through to the libraries are listed per file. `build/*.yaml` is workspace tooling and is grouped apart. +- **Possibly unused config keys**: each library's Python (and autonerves' + own) is scanned for `conf.instance[...]` lookups, and every key of a + library settings file is classed *used* (a lookup reads it, or reads a key + under it — PyAutoLens reading a PyAutoFit key counts, since autonerves + merges every layer), *section-read* (a lookup reads an ancestor section + whole, or a non-literal subscript such as `["plots"][section][name]` ends + the chain above it) or *unused* (nothing references it). Unused keys get a + chip on the repo page, per-file counts and an index section grouped by + library with GitHub links. Prior files are exempt (looked up by class name). + **Limits:** the scan is static — it follows literal subscript chains + (including multi-line ones and `.get("k")`), sections bound to a local and + indexed later in the same function or a closure, and the `should_output` + helper; a key read any other way (built key strings, `getattr`, a section + passed to another function and indexed there) shows as section-read at + best and may show as unused. Treat the list as candidates, not a verdict. - **Environment variables**: the `PYAUTO_*` switches autonerves reads. - **Cockpit feed** (`state.json`): green when every file parses and no workspace key is orphaned; yellow with one item per unparseable file or orphan-carrying workspace file; grey when nothing was collected. Never red. + Possibly unused keys add one *info* item per library — never yellow, until + the scan is trusted. Rendered daily by [`.github/workflows/nerves_board.yml`](.github/workflows/nerves_board.yml) -from sparse checkouts of the config folders; the renderer is +from sparse checkouts of the config folders (plus each library's package +`.py` files, for the lookup scan); the renderer is [`scripts/board.py`](scripts/board.py) (not part of the `autonerves` package). Nothing on the board edits config. diff --git a/scripts/board.py b/scripts/board.py index 680f021..017b45a 100644 --- a/scripts/board.py +++ b/scripts/board.py @@ -26,13 +26,19 @@ differs, keys only the workspace sets (**orphans** — no library defines them), keys only the stack sets. ``build/*.yaml`` is workspace tooling (CI/build lists, not library settings) and is grouped apart; +* **possibly unused library keys**: every library package's ``.py`` (and + ``autonerves/``) is scanned for config lookups (``scan_lookups``) and each + key of a library settings file is classed ``used`` / ``section-read`` / + ``unused`` against the reads of the whole stack (``classify_files``) — + static and untrusted, so it only ever adds ``info`` feed items; * the ``PYAUTO_*`` environment variables read by ``autonerves/test_mode.py``, ``workspace.py`` and ``__init__.py`` — the non-YAML options the Nerves also own — each with the comment above it (or its function's docstring). **Where it reads from.** Locally, each source resolves through the body map (``PyAutoMind/repos.yaml`` ``path:``) under ``--root``; in the workflow, -``--sources DIR`` points at sparse clones laid out ``DIR//``. +``--sources DIR`` points at sparse clones laid out ``DIR//`` +(and ``DIR//`` for a library's code). GitHub slugs come from the body map's ``github:``; the Nerves' own owner comes from ``git remote`` (the tenant firewall: no owner is written here). @@ -59,6 +65,7 @@ import re import subprocess import sys +import warnings from pathlib import Path import yaml @@ -71,17 +78,19 @@ # The config sources, in one table. `stack` is the lookup order autonerves # uses behind a workspace (the last-imported library first, PyAutoFit last), # so a workspace file is compared against what the libraries would give. +# `package` is a library's source dir, scanned for the config keys its code +# reads (the "possibly unused keys" flag). SOURCES = ( {"repo": "PyAutoFit", "config": "autofit/config", "kind": "library", - "stack": ()}, + "package": "autofit", "stack": ()}, {"repo": "PyAutoArray", "config": "autoarray/config", "kind": "library", - "stack": ()}, + "package": "autoarray", "stack": ()}, {"repo": "PyAutoGalaxy", "config": "autogalaxy/config", "kind": "library", - "stack": ()}, + "package": "autogalaxy", "stack": ()}, {"repo": "PyAutoLens", "config": "autolens/config", "kind": "library", - "stack": ()}, + "package": "autolens", "stack": ()}, {"repo": "PyAutoCTI", "config": "autocti/config", "kind": "library", - "stack": ()}, + "package": "autocti", "stack": ()}, {"repo": "autofit_workspace", "config": "config", "kind": "workspace", "stack": ("PyAutoFit",)}, {"repo": "autogalaxy_workspace", "config": "config", "kind": "workspace", @@ -420,16 +429,274 @@ def env_vars_from(texts: dict) -> list[dict]: return sorted(seen.values(), key=lambda e: e["name"]) +# --- config lookups in library code (the "not in use anymore" scan) ------------ +# autonerves resolves ``conf.instance["general"]["output"]["remove_files"]`` as +# file ``general.yaml`` → section ``output`` → key ``remove_files`` (keys +# lowercased, every layer merged), so a read is a dotted path whose head is the +# config file's path without its suffix. +# +# Helpers that take a key name and read it under a fixed prefix — a literal +# first argument is a read of ``prefix.``. +HELPER_READS = {"should_output": ("output",)} +# Files autonerves loads whole outside the subscript API +# (``Config.logging_config`` opens ``logging.yaml`` and hands it to +# ``logging.config.dictConfig``) — every key in them is a section read. +WHOLESALE_FILES = ("logging",) +# Config lookups are only counted once every library's package was scanned: +# a partial scan would call every key of the missing library unused. +LOOKUP_CLASSES = ("used", "section-read", "unused") +# autonerves' own package — a config reader too (swappable for tests). +NERVES_PACKAGE = NERVES_HOME / "autonerves" + + +def _is_conf_root(node, names: set) -> bool: + """``conf.instance`` / ``x.conf.instance`` / ``conf.instance.dict`` / + a bare ``instance`` imported from ``autonerves.conf``.""" + if isinstance(node, ast.Attribute) and node.attr == "dict": + node = node.value + if isinstance(node, ast.Attribute) and node.attr == "instance": + v = node.value + return (isinstance(v, ast.Name) and v.id == "conf") or \ + (isinstance(v, ast.Attribute) and v.attr == "conf") + return isinstance(node, ast.Name) and node.id in names + + +def _literal(node): + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return node.value.lower() + return None + + +def _chain(node, roots: set, local: dict): + """Unwind a subscript chain ending at ``node``: ``(keys, open)`` where + ``keys`` is the literal path from the root and ``open`` says a non-literal + subscript ended it (everything under ``keys`` may be read); ``None`` when + the chain does not start at the config.""" + steps = [] + cur = node + while True: + if isinstance(cur, ast.Subscript): + steps.append(cur.slice) + cur = cur.value + elif (isinstance(cur, ast.Call) and isinstance(cur.func, ast.Attribute) + and cur.func.attr == "get" and cur.args): + steps.append(cur.args[0]) + cur = cur.func.value + else: + break + if isinstance(cur, ast.Name) and cur.id in local: + base = list(local[cur.id]) + elif _is_conf_root(cur, roots): + base = [] + else: + return None + keys = base + for s in reversed(steps): + lit = _literal(s) + if lit is None: + return keys, True + keys.append(lit) + return keys, False + + +def _parents(tree) -> dict: + par = {} + for node in ast.walk(tree): + for child in ast.iter_child_nodes(node): + par[child] = node + return par + + +def _is_chain_link(node, parent) -> bool: + """Whether ``node`` continues into ``parent`` as part of a longer chain.""" + if isinstance(parent, ast.Subscript) and parent.value is node: + return True + if isinstance(parent, ast.Attribute) and parent.value is node and \ + parent.attr in ("get", "dict"): + return True + return False + + +def lookups_from(text: str) -> tuple[set, set, set]: + """``(reads, wildcards, bound)`` — the dotted config paths one module + reads. + + * a literal subscript chain from ``conf.instance`` (``.get("k")`` counts as + a subscript) is a read of that path; + * a non-literal subscript ends the chain: its literal prefix becomes a + wildcard (anything under it may be read); + * ``x = conf.instance["a"]["b"]`` then ``x["c"]`` is followed within the + function (or module, or a closure inside it) — the assignment reads + ``a.b`` itself (``bound``: the path is used, its keys are not covered), + each use of ``x`` reads ``a.b`` + its own literal keys, and a bare use + of ``x`` (passed on, iterated) reads ``a.b`` whole; + * ``HELPER_READS`` calls with a literal first argument. + + Raises ``SyntaxError`` on a module that does not parse.""" + tree = ast.parse(text) + roots = set() + for node in ast.walk(tree): + if isinstance(node, ast.ImportFrom) and (node.module or "").endswith( + "conf") and (node.module or "").split(".")[0] in ( + "autonerves", "autoconf"): + roots |= {a.asname or a.name for a in node.names + if a.name == "instance"} + par = _parents(tree) + reads, wild, bound = set(), set(), set() + funcs = (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda) + scopes = [n for n in ast.walk(tree) if isinstance(n, funcs)] + scopes.append(tree) + owned_by, own_locals, skip = {}, {}, set() + for scope in scopes: + # nodes owned by this scope (a nested def is its own scope) + owned, stack = [], list(ast.iter_child_nodes(scope)) + while stack: + n = stack.pop() + if isinstance(n, funcs): + continue + owned.append(n) + stack.extend(ast.iter_child_nodes(n)) + owned_by[scope] = owned + local: dict = {} + for n in owned: + if isinstance(n, ast.Assign) and len(n.targets) == 1 and \ + isinstance(n.targets[0], ast.Name): + got = _chain(n.value, roots, {}) + if got and not got[1] and got[0]: + local[n.targets[0].id] = tuple(got[0]) + bound.add(".".join(got[0])) + skip.add(n.value) + own_locals[scope] = local + seen: set = set() + for scope in scopes: + # a closure sees the names bound in the functions around it + local, up = {}, par.get(scope) + chain_up = [] + while up is not None: + if isinstance(up, funcs): + chain_up.append(up) + up = par.get(up) + for outer in reversed(chain_up): + local.update(own_locals[outer]) + local.update(own_locals[scope]) + owned = owned_by[scope] + for n in owned: + if n in seen or n in skip: + continue + if isinstance(n, ast.Call) and isinstance(n.func, ast.Name) and \ + n.func.id in HELPER_READS and n.args: + lit = _literal(n.args[0]) + if lit is not None: + reads.add(".".join(HELPER_READS[n.func.id] + (lit,))) + continue + if not isinstance(n, (ast.Subscript, ast.Call, ast.Name, + ast.Attribute)): + continue + if isinstance(n, ast.Name) and not isinstance(n.ctx, ast.Load): + continue + if _is_chain_link(n, par.get(n)): + continue + got = _chain(n, roots, local) + if not got: + continue + seen.add(n) + keys, open_ = got + if not keys: + if open_: + wild.add("") + continue + (wild if open_ else reads).add(".".join(keys)) + return reads, wild, bound + + +def scan_lookups(pkg_dir) -> dict: + """Every config read in a package's ``.py`` files: + ``{files, reads, wildcards, bound, unparsed}`` (sorted lists).""" + reads, wild, bound, unparsed, n = set(), set(), set(), [], 0 + for path in sorted(Path(pkg_dir).rglob("*.py")): + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + unparsed.append(path.relative_to(pkg_dir).as_posix()) + continue + if "instance" not in text and not any(h in text for h in HELPER_READS): + n += 1 + continue + try: + with warnings.catch_warnings(): + warnings.simplefilter("ignore") # old escapes in library code + r, w, b = lookups_from(text) + except SyntaxError: + unparsed.append(path.relative_to(pkg_dir).as_posix()) + continue + n += 1 + reads |= r + wild |= w + bound |= b + return {"files": n, "reads": sorted(reads), "wildcards": sorted(wild), + "bound": sorted(bound), "unparsed": unparsed} + + +def config_path(rel: str, key: str) -> str: + """A file key's full lookup path: ``visualize/general.yaml`` + + ``general.backend`` → ``visualize.general.general.backend``.""" + stem = rel.rsplit(".", 1)[0].replace("/", ".") + return f"{stem}.{key}".lower() if key else stem.lower() + + +def classify_key(path: str, reads: set, prefixes: set) -> str: + """``used`` — the path is read (``reads``: exact reads, wildcard stems and + bound locals), or a read runs through it (a section one of whose keys is + read); ``section-read`` — a whole-section read or wildcard (``prefixes``) + covers one of its ancestors; ``unused`` — nothing references it.""" + if path in reads: + return "used" + dotted = path + "." + if any(r.startswith(dotted) for r in reads): + return "used" + parts = path.split(".") + for i in range(len(parts) - 1, -1, -1): + if ".".join(parts[:i]) in prefixes: + return "section-read" + return "unused" + + +def classify_files(files: list[dict], lookups: dict) -> None: + """Stamp ``use`` on every key of every library settings file (in place), + against the reads of the whole stack — autonerves merges every layer, so a + PyAutoFit key read by PyAutoLens is used.""" + reads, prefixes = set(), set(WHOLESALE_FILES) + for lk in lookups.values(): + # a chain that stops on a section reads it whole; a wildcard stem + # ``a.b[x]`` reads ``a.b`` and may read anything under it + prefixes |= set(lk["reads"]) | set(lk["wildcards"]) + reads |= set(lk["reads"]) | set(lk["wildcards"]) | \ + set(lk.get("bound") or ()) + for f in files: + if not f.get("_library") or f.get("prior") or f.get("error") or \ + f.get("tooling"): + continue + counts = dict.fromkeys(LOOKUP_CLASSES, 0) + for k in f["keys"]: + k["use"] = classify_key(config_path(f["path"], k["k"]), reads, + prefixes) + counts[k["use"]] += 1 + f["use_counts"] = counts + + # --- collection (the only I/O) -------------------------------------------------- -def _source_dir(src: dict, root: Path, body: dict, sources: Path | None - ) -> Path | None: +def _source_dir(src: dict, root: Path, body: dict, sources: Path | None, + field: str = "config") -> Path | None: + sub = src.get(field) + if not sub: + return None if sources is not None: - cand = sources / src["repo"] / src["config"] + cand = sources / src["repo"] / sub return cand if cand.is_dir() else None rel = (body.get(src["repo"]) or {}).get("path") for base in ((root / rel) if rel else None, root / src["repo"]): - if base and (base / src["config"]).is_dir(): - return base / src["config"] + if base and (base / sub).is_dir(): + return base / sub return None @@ -440,8 +707,13 @@ def _now_z() -> str: def build_snapshot(files: list[dict], sources: list[dict], env: list[dict], errors: list[str], owner: str = "", repo: str = "", - generated: str | None = None) -> dict: - """Assemble the snapshot from read file records (pure).""" + generated: str | None = None, + lookups: dict | None = None) -> dict: + """Assemble the snapshot from read file records (pure). ``lookups`` + (``{reader: scan_lookups(…)}``) classifies every library settings key — + only when every library's package was scanned (``None`` skips it).""" + if lookups is not None: + classify_files(files, lookups) by_repo: dict = {} for f in files: by_repo.setdefault(f["repo"], {})[f["path"]] = f @@ -465,13 +737,17 @@ def build_snapshot(files: list[dict], sources: list[dict], env: list[dict], src["keys"] = sum(len(r["keys"]) for r in recs) src["priors"] = sum(1 for r in recs if r["prior"]) src["errors"] = sum(1 for r in recs if r["error"]) + if src["kind"] == "library" and lookups is not None: + src["unused"] = sum((r.get("use_counts") or {}).get("unused", 0) + for r in recs) clean = [{k: v for k, v in f.items() if not k.startswith("_")} for f in files] return {"schema_version": SCHEMA_VERSION, "generated": generated or _now_z(), "owner": owner, "repo": repo or "PyAutoNerves", "sources": sources, "files": clean, "overrides": overrides, - "env_vars": env, "errors": errors} + "env_vars": env, "errors": errors, + "lookups": lookups or {}} def collect(root=None, brain=None, mind=None, sources_dir=None, @@ -486,6 +762,7 @@ def collect(root=None, brain=None, mind=None, sources_dir=None, errors.append("PyAutoMind/repos.yaml not found — GitHub links use this " "repo's owner and local paths fall back to /") srcs, files = [], [] + scans: dict | None = {} for src in SOURCES: meta = body.get(src["repo"]) or {} github = meta.get("github") or (f"{owner}/{src['repo']}" if owner else "") @@ -496,6 +773,8 @@ def collect(root=None, brain=None, mind=None, sources_dir=None, if where is None: errors.append(f"{src['repo']}: config dir {src['config']}/ not found") srcs.append(entry) + if src["kind"] == "library": + scans = None # its code's reads are unknown: classify nothing continue entry["found"] = True for path in sorted(where.rglob("*")): @@ -507,16 +786,33 @@ def collect(root=None, brain=None, mind=None, sources_dir=None, except (OSError, UnicodeDecodeError) as e: errors.append(f"{src['repo']}/{rel}: unreadable ({e})") continue - files.append(read_file(src["repo"], rel, text, src["kind"])) + rec = read_file(src["repo"], rel, text, src["kind"]) + rec["_library"] = src["kind"] == "library" + files.append(rec) srcs.append(entry) + if src["kind"] == "library": + pkg = _source_dir(src, root, body, sdir, "package") + if pkg is None: + errors.append(f"{src['repo']}: package {src.get('package')}/ " + "not found — config keys are not classified") + scans = None + elif scans is not None: + scans[src["repo"]] = scan_lookups(pkg) + for rel in scans[src["repo"]]["unparsed"]: + errors.append(f"{src['repo']}: {src['package']}/{rel} did " + "not parse — its config reads are not counted") texts = {} for rel in ENV_MODULES: try: texts[rel] = (NERVES_HOME / rel).read_text(encoding="utf-8") except OSError as e: errors.append(f"env vars: {rel} unreadable ({e})") + if scans is not None: + # autonerves reads config itself (should_output, the backend) — its + # own package is a reader too, whatever repo it runs from. + scans["PyAutoNerves"] = scan_lookups(NERVES_PACKAGE) return build_snapshot(files, srcs, env_vars_from(texts), errors, owner, - repo, generated) + repo, generated, scans) # --- derived views ---------------------------------------------------------------- @@ -546,6 +842,10 @@ def slug(rel: str) -> str: return "f-" + re.sub(r"[^A-Za-z0-9]+", "-", rel).strip("-").lower() +def unused_anchor(lib: str) -> str: + return "unused-" + re.sub(r"[^A-Za-z0-9]+", "-", lib).strip("-").lower() + + def orphans(snap: dict) -> list[dict]: return [o for o in snap.get("overrides") or [] if o["workspace_only"]] @@ -554,6 +854,18 @@ def parse_errors(snap: dict) -> list[dict]: return [f for f in snap.get("files") or [] if f.get("error")] +def unused_keys(snap: dict) -> dict: + """``{library: [{path, k, line}]}`` — the library settings keys no code + in the stack reads, in source order (only classified snapshots).""" + out: dict = {} + for f in snap.get("files") or []: + for k in f.get("keys") or []: + if k.get("use") == "unused": + out.setdefault(f["repo"], []).append( + {"path": f["path"], "k": k["k"], "line": k["line"]}) + return out + + def status(snap: dict) -> str: if not snap.get("files"): return "grey" @@ -571,6 +883,9 @@ def _summary(snap: dict) -> str: bits.append(f"{len(parse_errors(snap))} unparseable") if orphans(snap): bits.append(f"{len(orphans(snap))} with orphan keys") + unused = sum(len(v) for v in unused_keys(snap).values()) + if unused: + bits.append(f"{unused} possibly unused library keys") return " · ".join(bits) @@ -579,6 +894,14 @@ def _keys_text(keys: list[str], limit: int = 6) -> str: return shown + (f" (+{len(keys) - limit} more)" if len(keys) > limit else "") +UNUSED_CAVEAT = ( + "A static scan of every library's (and autonerves') Python for " + "`conf.instance[...]` lookups: a key is *unused* when no literal lookup " + "reads it, no read runs through it and no section read or non-literal " + "subscript covers it. Keys read dynamically some other way are false " + "positives — check before deleting.") + + # --- markdown ----------------------------------------------------------------------- def _render_md(snap: dict) -> str: st = status(snap) @@ -612,6 +935,15 @@ def _render_md(snap: dict) -> str: out += ["", "## Unparseable files", ""] for f in parse_errors(snap): out.append(f"- `{f['repo']}/{f['path']}`: {f['error']}") + unused = unused_keys(snap) + if unused: + out += ["", "## Possibly unused config keys (no library code reads " + "them)", "", UNUSED_CAVEAT, ""] + for lib, keys in unused.items(): + out.append(f"- **{lib}** ({len(keys)}):") + for u in keys: + out.append(f" - [`{u['path']}` `{u['k']}`]" + f"({file_url(snap, lib, u['path'], u['line'])})") if snap.get("env_vars"): out += ["", "## Environment variables", ""] for e in snap["env_vars"]: @@ -662,7 +994,8 @@ def to_state(snap: dict) -> dict: """The organ-cockpit feed (contract v1, owned by PyAutoBrain ``board/_state.py``): one yellow item per unparseable file, then one per workspace file carrying orphan keys, then collection errors (info); - capped at 20. Never red — the board is a map, not a gate.""" + one info item per library with possibly unused keys; capped at 20. Never + red — the board is a map, not a gate.""" st = status(snap) items = [] for f in parse_errors(snap): @@ -687,6 +1020,22 @@ def to_state(snap: dict) -> dict: f"); check {' → '.join(o['stack'])} and either add " "them to the library config or drop them from the " "workspace.", 600)}) + # Info only, never yellow: the scan is static and untrusted until a human + # has reviewed its false positives. + for lib, keys in unused_keys(snap).items(): + items.append({"severity": "info", + "text": _clip(f"{lib}: {len(keys)} possibly unused " + f"config key{'s' if len(keys) != 1 else ''}" + f" (no library code reads " + f"{'them' if len(keys) != 1 else 'it'})"), + "url": (pages_url(snap) + "#" + unused_anchor(lib) + if pages_url(snap) else None), + "prompt": _clip( + f"Review the {lib} config keys the Nerves board " + f"flags as possibly unused (" + f"{', '.join(u['path'] + ':' + u['k'] for u in keys[:15])}" + "); confirm no code reads them, then remove them " + "from the library config and the workspaces.", 600)}) items += [{"severity": "info", "text": _clip(f"unavailable this render: {e}"), "url": None, "prompt": None} for e in snap.get("errors") or []] if st == "grey": @@ -727,6 +1076,7 @@ def _esc(v) -> str: border:1px solid var(--line);border-radius:999px;background:var(--btn)} .chip.y{border-color:var(--warn);color:var(--warn)} .chip.r{border-color:var(--bad);color:var(--bad)} +.chip.u{border-style:dashed;color:var(--muted)} pre.src{font:.8em/1.45 ui-monospace,SFMono-Regular,Menlo,monospace; background:var(--btn);border:1px solid var(--line);border-radius:8px; padding:.5rem 0;margin:.4rem 0} @@ -754,6 +1104,7 @@ def _esc(v) -> str: out.innerHTML=hits.map(function(e){var f=IDX.f[e[0]],r=IDX.r[f[0]]; var href='repos/'+encodeURIComponent(r)+'.html#'+f[2]+(e[2]?'-L'+e[2]:''); return '
  • '+esc(e[1]||f[1])+' '+ + (e[4]==='unused'?'unused ':'')+ ''+esc(r+'/'+f[1])+(e[2]?':'+e[2]:'')+''+ (e[3]?'
    '+esc(e[3])+'':'')+'
  • ';}).join(''); info.textContent=total?(total+' match'+(total==1?'':'es')+ @@ -806,7 +1157,8 @@ def _page(snap: dict, title: str, body: str, js: str, depth: int = 0) -> str: def key_index(snap: dict) -> dict: """The compact search index the index page embeds: repos ``r``, files ``f = [repoIdx, path, anchor]`` and entries ``k = [fileIdx, key, line, - comment]`` (a file itself is an entry with an empty key). Prior files + comment, use]`` (a file itself is an entry with an empty key; ``use`` is + the lookup class of a library settings key, else ``""``). Prior files index their ``Class`` and ``Class.param`` paths, not every leaf field.""" repos = [s["repo"] for s in snap.get("sources") or []] ri = {r: i for i, r in enumerate(repos)} @@ -817,11 +1169,12 @@ def key_index(snap: dict) -> dict: repos.append(f["repo"]) fi = len(files) files.append([ri[f["repo"]], f["path"], slug(f["path"])]) - entries.append([fi, "", 0, ""]) + entries.append([fi, "", 0, "", ""]) for k in f.get("keys") or []: if f.get("prior") and k["k"].count(".") > 1: continue - entries.append([fi, k["k"], k["line"], _clip(k["c"], 120)]) + entries.append([fi, k["k"], k["line"], _clip(k["c"], 120), + k.get("use", "")]) return {"r": repos, "f": files, "k": entries} @@ -835,6 +1188,8 @@ def _render_html_index(snap: dict) -> str: stats = t_.stats((len(files), "files"), (len(found), "sources"), (sum(len(f["keys"]) for f in files), "keys"), (len(ov), "overrides"), (len(orphans(snap)), "orphaned"), + (sum(len(v) for v in unused_keys(snap).values()), + "possibly unused"), (len(snap.get("env_vars") or []), "env vars")) \ if hasattr(t_, "stats") else "" rows = [] @@ -912,6 +1267,28 @@ def _render_html_index(snap: dict) -> str: f"{_esc(f['repo'])}/{_esc(f['path'])} — " f"{_esc(f['error'])}" for f in parse_errors(snap)) + "") + unused_html = "" + unused = unused_keys(snap) + if unused: + groups = [] + for lib, keys in unused.items(): + lis = "".join( + f"
  • {_esc(u['k'])} {_esc(u['path'])}:{u['line']}" + + (f" · GitHub" if file_url(snap, lib, u['path']) else "") + + "
  • " for u in keys) + groups.append(f"
    " + f"{_esc(lib)} — {len(keys)} key" + f"{'s' if len(keys) != 1 else ''}" + f"
      {lis}
    ") + unused_html = ("

    Possibly unused config keys

    " + + _esc(UNUSED_CAVEAT).replace("`", "").replace("*", "") + "

    " + + "".join(groups)) + elif snap.get("lookups"): + unused_html = ("

    Possibly unused config keys

    " + "none — every library settings key is read.

    ") errors = "" if snap.get("errors"): errors = ("

    unavailable this " @@ -937,6 +1314,7 @@ def _render_html_index(snap: dict) -> str:

      {overview} {override} +{unused_html} {problems} {env_html} {tool_html} @@ -947,11 +1325,14 @@ def _render_html_index(snap: dict) -> str: def _source_html(f: dict, sid: str) -> str: err_line = f.get("error_line") if f.get("error") else None + unused = {k["line"] for k in f.get("keys") or [] if k.get("use") == "unused"} out = [] for n, line in enumerate(f.get("text", "").splitlines(), start=1): code, comment = split_comment(line) cls = " errline" if n == err_line else "" inner = _esc(code) + (f"{_esc(comment)}" if comment else "") + if n in unused: + inner += " unused" out.append(f'{n}{inner}') return '
      ' + "".join(out) + "
      " @@ -975,6 +1356,10 @@ def _file_html(snap: dict, f: dict, ov: dict | None) -> str: kind = ("tooling" if f.get("tooling") else "prior file" if f.get("prior") else "settings") meta = [f"{f['lines']} lines", f"{len(f['keys'])} keys", kind] + uc = f.get("use_counts") + if uc: + meta.append(f"{uc['used']} used · {uc['section-read']} section-read · " + f"{uc['unused']} unused") notes = [] if f.get("error"): notes.append(f"

      does not parse: {_esc(f['error'])}

      ") @@ -995,6 +1380,12 @@ def _file_html(snap: dict, f: dict, ov: dict | None) -> str: f"{_esc(k)}" for k in ov["workspace_only"]) + "".join( f"" f"{_esc(k)}" for k in ov["differs"]) + "
      " + unused = [k for k in f.get("keys") or [] if k.get("use") == "unused"] + if unused: + diff_chips += "
      " + "".join( + f"unused: {_esc(k['k'])}" + f"" for k in unused) + "
      " if f.get("prior") and f.get("priors"): body = (f"
      priors ({len(f['priors'])} params)" f"{_prior_html(f, sid)}
      ") diff --git a/test_autonerves/test_board.py b/test_autonerves/test_board.py index e9e64d6..e0ec5b6 100644 --- a/test_autonerves/test_board.py +++ b/test_autonerves/test_board.py @@ -29,9 +29,9 @@ def _load_board(): FAKE_SOURCES = ( {"repo": "LibLow", "config": "liblow/config", "kind": "library", - "stack": ()}, + "package": "liblow", "stack": ()}, {"repo": "LibHigh", "config": "libhigh/config", "kind": "library", - "stack": ()}, + "package": "libhigh", "stack": ()}, {"repo": "ws_demo", "config": "config", "kind": "workspace", "stack": ("LibHigh", "LibLow")}, ) @@ -110,6 +110,62 @@ def _load_board(): BROKEN = "key: [unclosed\nother: 1\n" +# The fake libraries' code — every config lookup style the scan follows. +# Style 1: a literal chain from ``conf.instance`` (split over lines), and a +# chain ended by a non-literal subscript (a wildcard over ``fit``). +LOW_CODE = """\ +from autonerves import conf + + +def log_level(): + return conf.instance["general"]["output"][ + "log_level" + ] + + +def fit_plot(name): + return conf.instance["visualize"]["plots"]["fit"][name] +""" + +# Style 2: a bare ``instance`` imported from autonerves.conf, read with +# ``.get`` — a cross-library read: ``hpc.hpc_mode`` lives only in LibLow. +# Style 3: a section bound to a local, then indexed (inside a closure). +HIGH_CODE = """\ +from autonerves.conf import instance + + +def hpc_mode(): + return instance["general"]["hpc"].get("hpc_mode", False) + + +def dataset_image(): + plots = instance["visualize"]["plots"] + + def inner(): + return plots["dataset"]["image"] + + return inner() +""" + +# autonerves' own reads (a helper with a literal name). +NERVES_CODE = """\ +from autonerves.output import should_output + + +def go(): + return should_output("samples") +""" + +LOW_PLOTS = """\ +fit: + subplot: true + data: false +dataset: + image: true +retired: + old_flag: true # nothing reads this any more +""" + FAKE_MODULE = '''\ import os @@ -147,6 +203,9 @@ def make_tree(tmp_path): _write(ws / "priors" / "blob.yaml", WS_PRIOR) _write(ws / "broken.yaml", BROKEN) _write(ws / "build" / "no_run.yaml", "# skip list\n- slow_script\n") + _write(root / "libs" / "LibLow" / "liblow" / "plot.py", LOW_CODE) + _write(root / "libs" / "LibHigh" / "libhigh" / "util.py", HIGH_CODE) + _write(root / "nerves_pkg" / "helper.py", NERVES_CODE) return root @@ -163,6 +222,7 @@ def make_tree(tmp_path): @pytest.fixture(name="snap") def make_snap(tree, monkeypatch): monkeypatch.setattr(board, "SOURCES", FAKE_SOURCES) + monkeypatch.setattr(board, "NERVES_PACKAGE", tree / "nerves_pkg") monkeypatch.setattr(board, "theme", lambda: FAKE_THEME) monkeypatch.setenv("GITHUB_REPOSITORY", "SomeOrg/PyAutoNerves") monkeypatch.delenv("PYAUTO_MIND", raising=False) @@ -353,6 +413,7 @@ def test_site_writes_every_surface(snap, tmp_path): def test_sources_dir_overrides_body_map_resolution(tree, tmp_path, monkeypatch): monkeypatch.setattr(board, "SOURCES", FAKE_SOURCES) + monkeypatch.setattr(board, "NERVES_PACKAGE", tree / "nerves_pkg") monkeypatch.setenv("GITHUB_REPOSITORY", "SomeOrg/PyAutoNerves") sources = tmp_path / "sources" _write(sources / "LibLow" / "liblow" / "config" / "general.yaml", LOW_GENERAL) @@ -361,3 +422,121 @@ def test_sources_dir_overrides_body_map_resolution(tree, tmp_path, monkeypatch): assert any("LibHigh" in e for e in snap["errors"]) # slugs still come from the body map assert snap["sources"][0]["github"] == "SomeOrg/LibLow" + # a library whose code was not fetched: nothing is classified + assert snap["lookups"] == {} + assert not any("use" in k for f in snap["files"] for k in f["keys"]) + + +# --- the "possibly unused keys" scan ------------------------------------------- +def test_lookups_from_every_style(): + reads, wild, bound = board.lookups_from(LOW_CODE + HIGH_CODE) + assert reads == {"general.output.log_level", "general.hpc.hpc_mode", + "visualize.plots.dataset.image"} + assert wild == {"visualize.plots.fit"} + # the local's own path is used, but its keys are not blanket-covered + assert bound == {"visualize.plots"} + r, w, b = board.lookups_from( + "from autonerves import conf\n" + "cfg = conf.instance['output']\n" + "for k in cfg:\n pass\n" + "x = conf.instance[name]['a']\n" + "conf.instance.output_path\n") + # a bare use of a bound section reads it whole; a non-literal root is an + # everything-wildcard; attribute access on the config is not a key read + assert r == {"output"} and w == {""} and b == {"output"} + r, _, _ = board.lookups_from(NERVES_CODE) + assert r == {"output.samples"} + + +def test_classify_key(): + reads = {"general.output.log_level"} + prefixes = {"visualize.plots.fit", "general.output.log_level"} + assert board.classify_key("general.output.log_level", reads, prefixes) \ + == "used" + assert board.classify_key("general.output", reads, prefixes) == "used" + assert board.classify_key("visualize.plots.fit.subplot", reads, + prefixes) == "section-read" + assert board.classify_key("general.hpc.hpc_mode", reads, prefixes) \ + == "unused" + assert board.config_path("visualize/plots.yaml", "fit.data") == \ + "visualize.plots.fit.data" + + +@pytest.fixture(name="usnap") +def make_usnap(tree, monkeypatch): + _write(tree / "libs" / "LibLow" / "liblow" / "config" / "visualize" + / "plots.yaml", LOW_PLOTS) + monkeypatch.setattr(board, "SOURCES", FAKE_SOURCES) + monkeypatch.setattr(board, "NERVES_PACKAGE", tree / "nerves_pkg") + monkeypatch.setattr(board, "theme", lambda: FAKE_THEME) + monkeypatch.setenv("GITHUB_REPOSITORY", "SomeOrg/PyAutoNerves") + monkeypatch.delenv("PYAUTO_MIND", raising=False) + return board.collect(root=tree, generated="2026-01-02T03:04:05Z") + + +def _uses(snap, repo, path): + return {k["k"]: k.get("use") for k in _file(snap, repo, path)["keys"]} + + +def test_library_keys_are_classified_across_the_stack(usnap): + assert set(usnap["lookups"]) == {"LibLow", "LibHigh", "PyAutoNerves"} + g = _uses(usnap, "LibLow", "general.yaml") + assert g["output.log_level"] == "used" # literal chain + assert g["hpc.hpc_mode"] == "used" # read by LibHigh's code + assert g["output"] == "used" and g["hpc"] == "used" + assert g["output.remove_files"] == "unused" + p = _uses(usnap, "LibLow", "visualize/plots.yaml") + assert p["fit"] == "used" # the wildcard's stem + assert p["fit.subplot"] == p["fit.data"] == "section-read" + assert p["dataset.image"] == "used" # via the bound local + assert p["dataset"] == "used" + assert p["retired"] == p["retired.old_flag"] == "unused" + assert _file(usnap, "LibLow", "visualize/plots.yaml")["use_counts"] == \ + {"used": 3, "section-read": 2, "unused": 2} + # workspace files and priors are never classified + for repo, path in (("ws_demo", "general.yaml"), + ("LibHigh", "priors/blob.yaml")): + assert not any("use" in k for k in _file(usnap, repo, path)["keys"]) + assert board.unused_keys(usnap) == {"LibLow": [ + {"path": "general.yaml", "k": "output.remove_files", "line": 4}, + {"path": "visualize/plots.yaml", "k": "retired", "line": 6}, + {"path": "visualize/plots.yaml", "k": "retired.old_flag", "line": 7}]} + + +def test_unused_keys_render_on_every_surface(usnap): + page = board.render(usnap, "html-index") + assert "Possibly unused config keys" in page + assert "id='unused-liblow'" in page + assert ("https://github.com/SomeOrg/LibLow/blob/main/liblow/config/" + "visualize/plots.yaml#L7") in page + section = page.split("Possibly unused config keys", 1)[1] \ + .split('id="idx"', 1)[0] + assert "retired.old_flag" in section + assert "dataset.image" not in section and "hpc.hpc_mode" not in section + raw = re.search(r'id="idx">(.*?)', page, re.S).group(1) + idx = json.loads(raw.replace("<\\/", "unused" in repo + md = board.render(usnap, "md") + assert "## Possibly unused config keys" in md and "`retired.old_flag`" in md + j = json.loads(board.render(usnap, "json")) + assert any(k.get("use") == "unused" for f in j["files"] for k in f["keys"]) + + +def test_unused_keys_are_one_info_item_per_library_never_yellow(usnap): + s = board.to_state(usnap) + _state_ok(s) + items = [i for i in s["items"] if "possibly unused" in i["text"]] + assert len(items) == 1 and items[0]["severity"] == "info" + assert items[0]["text"].startswith("LibLow: 3 possibly unused config keys") + assert items[0]["url"] == \ + "https://someorg.github.io/PyAutoNerves/#unused-liblow" + # the unused keys alone never colour the board + clean = dict(usnap, files=[f for f in usnap["files"] if not f["error"]], + overrides=[dict(o, workspace_only=[]) + for o in usnap["overrides"]]) + assert board.to_state(clean)["status"] == "green"