From a0e97092cce26fa6dda49032b461fec4bdb30ca6 Mon Sep 17 00:00:00 2001 From: wshallwshall Date: Thu, 3 Sep 2026 17:38:33 -0500 Subject: [PATCH] feat(ledger): screen open items whose subject already exists on main (BACKLOG #1426) Every existing ledger gate reads the ledger. This one reads the code: for each OPEN item it extracts commit shas, merged pull requests, file paths and symbol names, then asks git whether they are already on origin/main. It reports candidates and flips nothing. A wrongly-closed item is invisible forever, so closing a row stays a person's act. #1229 and #1040 are wired as controls: both were dispatched as builds on 2026-09-03 and both were already complete. A structural control over the probes runs first and exits 2, so a broken screen cannot read as a clean ledger. Ancestry uses merge-base --is-ancestor, not git log presence. A bare "#N" is never read. This clone is shallow, so a false ancestry answer reports unknown. First run at 46ea10a78: 275 open items, 3781 subjects, 80 candidates. Unread. Also records #1422 and #1425 as allocation holes in the Ledger erratum. Co-Authored-By: Claude Opus 5 --- .github/workflows/ci.yml | 1 + docs/BACKLOG.md | 70 ++ scripts/docs/subject_exists_screen.py | 1069 +++++++++++++++++++++++++ tests/test_subject_exists_screen.py | 531 ++++++++++++ tests/tooling_manifest.txt | 1 + 5 files changed, 1672 insertions(+) create mode 100644 scripts/docs/subject_exists_screen.py create mode 100644 tests/test_subject_exists_screen.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ad60a6bdc..811253165 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -248,6 +248,7 @@ jobs: tests/test_claude_section_citations.py tests/test_write_share_denominator.py tests/test_cutover_slug_rot.py tests/test_backlog_citation_check.py tests/test_dangling_citation_check.py + tests/test_subject_exists_screen.py tests/test_docs_cite_no_refused_config_keys.py" # Every named module must EXIST. A path typo would otherwise make pytest error on an unknown # file, or — worse under a future -k/--ignore form — silently scan nothing and read as a pass. diff --git a/docs/BACKLOG.md b/docs/BACKLOG.md index a19e6b606..2d3f09e0d 100644 --- a/docs/BACKLOG.md +++ b/docs/BACKLOG.md @@ -54,6 +54,19 @@ worktree") and the item was re-filed at **#1298**. The gate did its job; the les records the claim against **whatever tree it runs in**, so run it from the worktree that will commit. Always allocate with `scripts/coord/alloc.ps1`; never pick a number by reading this file. +**At least #1422 and #1425 are holes of the #1297 kind with one variation, and the variation is what +makes it recur: both were allocated by a COORDINATING session on a BUILDER's behalf.** Recorded +2026-09-03. The allocating shell was in the right worktree for itself and the wrong one for the seat +that would commit, so the claim went to the coordinator's tree while the commit came from the +Builder's. The gate keys ownership on the allocating WORKTREE first and falls back to the BRANCH only +once that worktree is gone, so with the coordinator's worktree still live BOTH keys miss and the +commit is refused — correctly, and far from the cause. #1425 was re-filed at **#1426** by the Builder, +allocating in its own worktree. **The rule that follows is narrower than "run it from the right +directory": a number must be allocated by the seat that will commit it, and a coordinator that +allocates ahead for someone else spends the number without being able to use it.** Only these two are +verified holes of this shape; the coordinator's other 2026-09-03 allocations (#1401 and up) were filed +from that same worktree and are ordinary items. + **If you allocated a backlog number before 2026-07-31T00:31Z, re-check it — the trigger is the timestamp, not the value.** That is when the floor fix landed. Any number issued before it came from a floor that could not see most of the namespace, so it is suspect **regardless of how low or high it @@ -19940,3 +19953,60 @@ That is the same `self._lock` the staged-pipeline handoffs take. On a first depl **PARTLY CLOSED ALREADY, AND THE CLOSURE SITS IN THE WRONG ARTIFACT.** The full record -- both questions, all eight options, both answers quoted -- is [comment 5515263760 on PR 749](https://github.com/MEFORORG/MessageFoundry/pull/749#issuecomment-5515263760), written 2026-09-02. A pull-request comment is a real improvement on a session transcript, which does not survive its session. It is still not the ADR, and the ADR is what a reader consults. **This limb differs from the first two in shape:** closing it needs no decision about the engine, only the record moved into the artifact people actually read. **THE GENERAL PROBLEM, stated once so it is not re-derived per incident.** A decision recorded as an outcome plus a delegation is not reviewable. The inputs -- the question, the options, the answer -- are what let a later reader tell a considered call from an arbitrary one, and they are exactly the part that lives in the least durable place. + +## 1426. Nothing reads the code to ask whether an open item's subject already exists on main, so a re-score that predates the landing keeps the row open forever + +> 🚧 **Filed 2026-09-03 -- the screen is BUILT and reports candidates; nothing yet flips a banner off it.** Every existing ledger gate reads the LEDGER. `scripts/docs/subject_exists_screen.py` is the first that reads the CODE: for each OPEN item it extracts the concrete code-side subjects the row names -- commit shas, merged pull requests, file paths and distinctive symbol names -- and asks git whether they are already on `origin/main`. **It reports candidates and flips nothing.** A wrongly-closed item is invisible forever, so the closing act stays a person reading each row. +> +> **The remaining work is the reading pass, not the tool.** First run at `46ea10a78`: 275 open items, 3781 subjects, **80 candidates and 114 weak-candidates**. Nobody has read that list. Two of the 80 were already known-true and are wired as controls; the other 78 are unread. +> Verdict: build +> Closing-act: code + +**Cluster:** Ledger hygiene / dispatch. **Priority:** P2 -- filed, not separately scored. +**Severity:** no deployment axis (section 0). Zero instances run. The cost is spent Builder sessions and a ledger that misdescribes build state, not anything an operator would meet. + +### The defect, measured twice on one day + +On 2026-09-03 five items were dispatched as builds. **Two were already complete, and a Builder was spent on each discovering it.** + +| item | what had already landed | why the ledger did not say so | +|---|---|---| +| **#1040** | all three commits from its branch were ancestors of `main`, and its cited pull request was on `main` | a note said the banner was "left open for the archive pass", while an OLDER re-score beneath it still described the landed work as outstanding | +| **#1229** | its backslash-escape limb shipped 2026-08-22 in `3c5cb9885` | that commit's subject names **BACKLOG #1268**, not #1229, so a search keyed on the item number never finds it. The re-score calling the limb unbuilt is dated **2026-08-20** -- two days BEFORE the merge | + +**The common shape is one sentence: a re-score dated before the landing, and nothing afterward reads the code.** The #1234 amendment already states why no ledger-side check can close it -- *"Does the subject exist on main is the only check that reads the CODE, and it is the one that decides startability."* + +### What was built + +`scripts/docs/subject_exists_screen.py`, with `tests/test_subject_exists_screen.py` (51 tests, on the tooling manifest and in `ci.yml`'s `DOC_GUARDS`). Six signals, ranked, so the strongest rows sort first: + +| strength | signal | what fires it | +|---|---|---| +| strong | `sha-ancestor-landing` | a cited sha that IS an ancestor of the ref, on a line worded as a landing | +| strong | `pr-merged-landing` | an explicit `PR #N` whose squash commit is on the ref, worded as a landing | +| strong | `path-added-after` | a cited path first ADDED to the ref after the row's newest date -- it did not exist when the row was last read | +| medium | `sha-ancestor`, `pr-merged` | the same two, cited as a base ref rather than a landing | +| medium | `sha-unverifiable-shallow` | ancestry could not be settled under a shallow clone (below) | +| weak | `path-changed-after`, `symbol-on-main` | the cited file moved since the row was read; the cited identifier exists | + +### Four properties that are not incidental + +1. **It reports, it never flips.** There is no `--fix` and there must not be one. +2. **Over-firing is the tolerable direction.** A false candidate costs one read; a missed one costs a Builder. 80 candidates from 275 rows is deliberate, and where a probe cannot answer the answer is a signal rather than silence. +3. **It prints what it scanned** -- items, subjects by kind, probes run, probes skipped by the cap. An empty scan and a clean scan must not render alike. +4. **It runs a control that must fire, in both directions, before it reports anything.** A structural control over the probes (a known commit resolves and a nonsense one does not; a tracked path resolves and an invented one does not) exits **2** on failure and says the SCREEN is broken. A ledger control over #1229 and #1040 exits **1** if either stops firing while still open -- and RETIRES BY NAME when one is closed, so a control that stopped applying can never read like a control that passed. + +**`#N` is never read bare.** It spells a pull request and a ledger item identically, and a security record entry reading "the build is #156" once resolved to a pull request while backlog #156 was unrelated work. Only `BACKLOG #N` (a cross-reference, never a subject) and an explicit `PR #N` are read. + +**A sha is evidence only under `git merge-base --is-ancestor`.** Presence in `git log` output answers a different question. + +### The shallow-clone trap is live here, and it is why one signal exists + +Measured 2026-09-03: this repository reports `--is-shallow-repository` **true**, with **16 graft points** over 931 commits reachable from `origin/main`. **Under a graft the two ancestry answers are not equally sound.** A TRUE is reliable -- the walk found the commit. A FALSE may only mean the walk stopped at a boundary, and an unresolvable sha may merely sit beyond it. Rendering either as "not on main" is a confident wrong answer, so both become `sha-unverifiable-shallow`, which surfaces the item instead of silently dropping it. **That is not a rare corner: it fired 63 times in the first run.** + +### What is open + +- **Nobody has read the 80 candidates.** Two are the controls. The other 78 are unread, and reading them is the act that closes rows. +- Two rows already look like repeats of the #1229 shape and are named here so the reading pass starts somewhere rather than at the top: **#1255**, whose `tests/test_conftest_name_collision_guard.py` was added to the ref on 2026-08-26 against a row last dated 2026-08-25; and **#1276**, against which `docs/adr/0172-the-engine-always-serves-tls-minting-a-self-signed-certificate-on-first-run.md` was added on 2026-09-02 against a row last dated 2026-08-25. **Named as candidates, not as findings** -- neither has been read, and this row does not close them. +- The screen runs on demand and is wired to no schedule. Whether it should run on a cron, or at dispatch time, is unanswered. +- The date proxy is the newest date anywhere in the row, because `parse_items` returns status and fields but not the banner block's text. Its error runs toward under-firing, which is why the two strongest signals are date-free. diff --git a/scripts/docs/subject_exists_screen.py b/scripts/docs/subject_exists_screen.py new file mode 100644 index 000000000..ace042d17 --- /dev/null +++ b/scripts/docs/subject_exists_screen.py @@ -0,0 +1,1069 @@ +#!/usr/bin/env python3 +# SPDX-License-Identifier: AGPL-3.0-or-later +# Copyright (C) 2026 MessageFoundry Organization and contributors +"""Report OPEN backlog items whose code-side subject already exists on `origin/main` (BACKLOG #1426). + +WHAT EVERY OTHER LEDGER SCREEN READS, AND WHY THAT IS NOT ENOUGH. `backlog_status_check.py` asks +whether an item declares a status. `backlog_citation_check.py` asks whether a citation names the file +its item lives in. `dangling_citation_check.py` asks whether a cited number names anything at all. +Every one of them reads the LEDGER. None reads the CODE, so none can notice the failure below, and +the #1234 amendment states the gap in terms: "Does the subject exist on main is the only check that +reads the CODE, and it is the one that decides startability." + +THE FAILURE, MEASURED TWICE ON 2026-09-03. Five items were dispatched as builds and two were already +complete, each discovered only after a Builder had been spent on it. + + * #1040 -- all three commits from its branch were ancestors of `main` and its cited PR was on + `main`. The row stayed open behind a note saying the banner was left open for an archive pass, + while an older re-score beneath it still described the landed work as outstanding. + * #1229 -- its backslash-escape limb shipped 2026-08-22 in `3c5cb9885`, whose subject names + BACKLOG #1268, NOT #1229. The re-score calling the limb unbuilt is dated 2026-08-20, two days + BEFORE the merge. Searching on the item number does not find the commit that implemented it. + +The common shape is a re-score dated before the landing, with nothing afterward reading the code. So +this tool extracts each open item's concrete code-side subjects -- commit shas, merged pull requests, +file paths and distinctive symbol names -- and asks git whether they are already on `origin/main`. + +FOUR CONSTRAINTS, EACH LOAD-BEARING. + +1. IT REPORTS CANDIDATES AND NEVER FLIPS A BANNER. A wrongly-closed item is invisible forever, so + the output is a list for a person to read, item by item. There is no `--fix` and there must not + be one. +2. OVER-FIRING IS THE TOLERABLE DIRECTION. A false candidate costs one read; a missed one costs a + whole Builder, which is what the two cases above each cost. Where a probe cannot answer, the + answer is a signal rather than silence -- see the shallow-clone handling below. +3. IT PRINTS WHAT IT SCANNED. Items examined, subjects extracted per kind, subjects resolved, probes + skipped by a cap. An empty scan and a clean scan must not render alike; that is this project's + named failure shape and it has fired repeatedly. +4. IT RUNS A CONTROL THAT MUST FIRE, in both directions, before it reports anything. `--self-test` + alone runs it and stops. Two layers: + * A STRUCTURAL control over the probes themselves, with a positive and a negative arm each: + a known commit resolves and a nonsense one does not, a tracked path resolves and an invented + one does not. A probe validated on one input is not validated. + * A LEDGER control over #1229 and #1040, the two known-true cases. While either is OPEN it MUST + come out a candidate. When one is closed the control RETIRES and says so by name rather than + passing silently -- a control that stops applying must not read like a control that passed. + A structural failure exits 2 and says the SCREEN is broken. A clean report exits 0. The two must + never be confused, which is the whole reason the exit codes differ. + +`#N` IS AMBIGUOUS IN THIS REPOSITORY AND IS NEVER READ BARE. A bare `#N` spells a pull request just +as well as a backlog item, and a security record entry reading "the build is #156" once resolved to a +pull request while backlog #156 was unrelated work. So only the literal forms are read: `BACKLOG #N` +is a cross-reference and is NEVER treated as a subject, and only an explicit `PR #N` / `pull request +#N` is resolved against the merge history. A bare `#N` is ignored by both. + +A SHA IS EVIDENCE ONLY IF ANCESTRY IS TESTED. Appearing in `git log` output answers a different +question -- every branch's commits appear there. The probe is `git merge-base --is-ancestor +origin/main`. + +THE SHALLOW-CLONE TRAP IS LIVE HERE, NOT HYPOTHETICAL. Measured 2026-09-03: this repository reports +`--is-shallow-repository` true with 16 graft points over 931 commits reachable from `origin/main`. +Under a graft the two ancestry answers are NOT equally sound. A TRUE is reliable -- the walk found +the commit. A FALSE may only mean the walk hit a boundary and stopped, and an unresolvable sha may +merely be beyond it. Reporting either as "not on main" is the confident wrong answer constraint 7 of +the brief forbids, so both become an explicit `unverifiable-shallow` signal that still surfaces the +item. + +WHY THE ITEM'S DATE IS "the newest date written anywhere in the row" and not the banner's. The banner +block's extent is defined by `parse_items`, which this module imports rather than re-deriving +(CLAUDE.md section 11), and that reader returns status and fields but not text. The newest date over +the whole row is a coarser proxy for "when did a person last read this", and its error runs one way: +a stray later date makes the date-based signals fire LESS. That is the under-firing direction, so the +two signals that caught both known cases -- sha ancestry and merged pull requests -- are deliberately +date-free, and an item carrying NO date at all is surfaced rather than skipped. + +Usage:: + + python scripts/docs/subject_exists_screen.py # full screen, text report + python scripts/docs/subject_exists_screen.py --self-test # controls only, then stop + python scripts/docs/subject_exists_screen.py --item 1229 # one item, with every probe shown + python scripts/docs/subject_exists_screen.py --json # machine-readable + +Exit 0 report produced (candidates or not); 1 a ledger control failed to fire; 2 the screen is broken. +""" + +from __future__ import annotations + +import argparse +import importlib.util +import json +import re +import subprocess +import sys +from collections.abc import Sequence +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Protocol + +_HERE = Path(__file__).resolve() +_ROOT = _HERE.parents[2] + +# --------------------------------------------------------------------------------------------- +# Reading the ledger. `parse_items` DEFINES what an item and its status are; a hand-rolled scan is a +# second, silently different definition (CLAUDE.md section 11), so it is imported by path rather than +# reimplemented. Loaded this way because `scripts/` is not an importable package. +# --------------------------------------------------------------------------------------------- + + +def _load_status_check() -> Any: + spec = importlib.util.spec_from_file_location( + "_backlog_status_check", _HERE.parent / "backlog_status_check.py" + ) + if spec is None or spec.loader is None: # pragma: no cover - a broken checkout + raise RuntimeError("cannot load scripts/docs/backlog_status_check.py") + module = importlib.util.module_from_spec(spec) + # Registered BEFORE execution. Harmless today and load-bearing the moment that module grows a + # `@dataclass`: dataclass processing resolves `sys.modules[cls.__module__]` mid-class-body, and + # an unregistered module turns that into an AttributeError naming neither file. + sys.modules[spec.name] = module + spec.loader.exec_module(module) + return module + + +# --------------------------------------------------------------------------------------------- +# Extraction -- the pure layer. Everything here is a function of text alone, so it is testable with +# no git, no network and no live ledger. +# --------------------------------------------------------------------------------------------- + +#: A cross-reference to another ledger item. Matched ONLY in its literal form. Never a subject. +_BACKLOG_REF = re.compile(r"BACKLOG\s+#(\d+)", re.IGNORECASE) + +#: A pull request, in its literal forms only. A bare `#N` is deliberately NOT matched: it spells a +#: pull request and a backlog item identically, and guessing has already resolved one to the wrong +#: subject in this repository. +_PR_REF = re.compile(r"\b(?:PRs?|pull\s+requests?)\s+#?(\d+)", re.IGNORECASE) + +#: An abbreviated or full commit sha. A run of hex 7-40 long, not touching a word character or `#` on +#: either side. Pure-decimal runs are rejected below: they are overwhelmingly numbers, and a real sha +#: that happens to be all digits simply fails to resolve, which costs recall on roughly one sha in +#: ten million. +_SHA = re.compile(r"(? str: + """How a line frames the reference it carries. + + Landing wins a line carrying both. A sentence such as "measured after it landed" is a landing + claim first; ranking it as a base ref would drop the strongest signal this screen has. + """ + if _LANDING_WORDS.search(line): + return "landing" + if _MEASUREMENT_WORDS.search(line): + return "measurement" + return "neutral" + + +def _normalise_path(raw: str) -> str: + """Strip the markdown-link `../` prefixes and any `:line` / `:line-line` suffix.""" + path = raw.strip() + while path.startswith("../") or path.startswith("./"): + path = path.split("/", 1)[1] + return path.rstrip(".,;:)]") + + +def crossrefs_in(body: str) -> list[int]: + """Ledger item numbers this row cites, in the literal `BACKLOG #N` form only. + + Returned so a report can show them as CONTEXT. They are never subjects: an item citing another + item says nothing about whether its own subject exists. + """ + return sorted({int(n) for n in _BACKLOG_REF.findall(body)}) + + +def subjects_in(body: str) -> list[Subject]: + """Every code-side subject an item names, deduplicated on (kind, value). + + Deduplication keeps the FIRST occurrence, and a later landing-worded occurrence upgrades it. A + sha named once as a base ref and again as the commit that landed the work is the second thing. + """ + found: dict[tuple[str, str], Subject] = {} + + def add(kind: str, value: str, lineno: int, wording: str) -> None: + key = (kind, value) + prior = found.get(key) + if prior is None: + found[key] = Subject(kind, value, lineno, wording) + elif prior.wording != "landing" and wording == "landing": + found[key] = Subject(kind, value, prior.lineno, "landing") + + for offset, line in enumerate(body.splitlines(), start=1): + wording = _wording(line) + + # A heading line names the item's own number, never a subject. + stripped = line.lstrip("> ").strip() + is_heading = stripped.startswith("## ") + + for sha in _SHA.findall(line): + if sha.isdigit(): + continue + add("sha", sha, offset, wording) + + for num in _PR_REF.findall(line): + add("pr", num, offset, wording) + + seen_paths: set[str] = set() + for raw in _PATH.findall(line): + path = _normalise_path(raw) + if path: + seen_paths.add(path) + add("path", path, offset, wording) + if not is_heading: + for raw in _BARE_FILE.findall(line): + name = _normalise_path(raw) + # A bare name already covered by a full path on the same line is not a second + # subject -- `messagefoundry/store/store.py` must not also yield `store.py`. + if name and not any(p.endswith("/" + name) for p in seen_paths): + add("path", name, offset, wording) + + for span in _TICKED.findall(line): + symbol = _symbol_from_span(span) + if symbol is not None: + add("symbol", symbol, offset, wording) + + return sorted(found.values(), key=lambda s: (s.kind, s.value)) + + +def _symbol_from_span(span: str) -> str | None: + """A distinctive identifier, or None. + + Rejects anything with whitespace (a command), anything path-shaped (the path extractor owns it), + anything under 8 characters, and the stoplist. Trailing `()` is stripped so `parse_items()` and + `parse_items` are one subject. + """ + token = span.strip() + if not token or " " in token or "\t" in token: + return None + if token.endswith("()"): + token = token[:-2] + token = token.strip("`.,;:") + if "/" in token or "\\" in token or "." in token: + return None + if len(token) < 8 or token in _SYMBOL_STOPLIST: + return None + if _SNAKE.match(token) or _VERB_NOUN.match(token) or _PASCAL.match(token): + return token + return None + + +def newest_date_in(body: str) -> str | None: + """The newest ISO date written anywhere in the row, as a proxy for when a person last read it. + + See the module docstring: coarser than the banner block's own date on purpose, and its error runs + in the under-firing direction, which is why the two strongest signals do not depend on it. + """ + dates = _ISO_DATE.findall(body) + return max(dates) if dates else None + + +# --------------------------------------------------------------------------------------------- +# The git layer, behind a protocol so every test runs against a fake and no test depends on the live +# history of this clone. +# --------------------------------------------------------------------------------------------- + + +class RepoReader(Protocol): + """Everything this screen needs to ask about `origin/main`.""" + + def is_shallow(self) -> bool: ... + + def is_commit(self, sha: str) -> bool: ... + + def is_ancestor(self, sha: str) -> bool: ... + + def path_on_main(self, path: str) -> str | None: + """The resolved tree path, or None. A bare filename resolves only if it is unambiguous.""" + + def path_added(self, path: str) -> str | None: ... + + def path_last_changed(self, path: str) -> str | None: ... + + def pr_merged(self, number: str) -> tuple[str, str] | None: + """`(sha, iso_date)` of the squash commit whose subject ends `(#N)`, or None.""" + + def symbol_on_main(self, symbol: str) -> bool: ... + + +class GitRepo: + """`RepoReader` over a real checkout. + + Three whole-history reads happen ONCE and everything else is a dict lookup: the tree file list, + the merge-subject map, and a `--name-status` walk giving every path's add and last-change dates. + Probing those per subject would be hundreds of git invocations for answers one walk already has. + """ + + def __init__(self, root: Path, ref: str = "origin/main") -> None: + self.root = root + self.ref = ref + self._shallow: bool | None = None + self._tree: set[str] | None = None + self._by_basename: dict[str, list[str]] | None = None + self._prs: dict[str, tuple[str, str]] | None = None + self._added: dict[str, str] = {} + self._changed: dict[str, str] = {} + self._dates_loaded = False + self._symbol_probes = 0 + + # -- plumbing ------------------------------------------------------------------------------ + + def _git(self, *args: str) -> subprocess.CompletedProcess[str]: + # Every argument is a fixed verb or a value taken from the ledger, and all of it reaches git + # as argv rather than a shell string, so nothing here is word-split or expanded. All five + # verbs used are read-only: rev-parse, merge-base, ls-tree, log, grep. + return subprocess.run( # nosec B603 B607 - fixed argv, no shell; read-only git + ["git", *args], + cwd=self.root, + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + check=False, + ) + + @property + def symbol_probes(self) -> int: + return self._symbol_probes + + def resolve(self, ref: str) -> str: + """The sha `ref` names, or an empty string. Public so the driver need not reach into _git.""" + return self._git("rev-parse", ref).stdout.strip() + + # -- probes -------------------------------------------------------------------------------- + + def is_shallow(self) -> bool: + if self._shallow is None: + self._shallow = ( + self._git("rev-parse", "--is-shallow-repository").stdout.strip() == "true" + ) + return self._shallow + + def is_commit(self, sha: str) -> bool: + return self._git("rev-parse", "--quiet", "--verify", f"{sha}^{{commit}}").returncode == 0 + + def is_ancestor(self, sha: str) -> bool: + return self._git("merge-base", "--is-ancestor", sha, self.ref).returncode == 0 + + def _load_tree(self) -> None: + if self._tree is not None: + return + out = self._git("ls-tree", "-r", "--name-only", self.ref).stdout + self._tree = {line.strip() for line in out.splitlines() if line.strip()} + basenames: dict[str, list[str]] = {} + for path in self._tree: + basenames.setdefault(path.rsplit("/", 1)[-1], []).append(path) + self._by_basename = basenames + + def path_on_main(self, path: str) -> str | None: + self._load_tree() + assert self._tree is not None and self._by_basename is not None + if path in self._tree: + return path + if "/" in path: + return None + # A bare filename is admitted only when it names exactly one file. Two matches is a real + # ambiguity and resolving it by picking one would attach a date to the wrong file. + matches = self._by_basename.get(path, []) + return matches[0] if len(matches) == 1 else None + + def _load_dates(self) -> None: + """One `--name-status` walk of the ref, newest first. + + `--no-renames` is deliberate: a renamed file should read as ARRIVING at its new path, which + is the question "does this subject exist on main" actually asks. + """ + if self._dates_loaded: + return + self._dates_loaded = True + out = self._git( + "log", + self.ref, + "--no-renames", + "--diff-filter=AM", + "--name-status", + "--format=%x00%cI", + ).stdout + date = "" + for line in out.splitlines(): + if line.startswith("\x00"): + date = line[1:].strip() + continue + if not line.strip() or "\t" not in line: + continue + status, _, path = line.partition("\t") + path = path.strip() + if not path: + continue + # Newest first, so the first sighting of a path is its last change and the LAST `A` + # seen is its earliest add. + self._changed.setdefault(path, date) + if status.startswith("A"): + self._added[path] = date + + def path_added(self, path: str) -> str | None: + self._load_dates() + return self._added.get(path) + + def path_last_changed(self, path: str) -> str | None: + self._load_dates() + return self._changed.get(path) + + def _load_prs(self) -> None: + if self._prs is not None: + return + out = self._git("log", self.ref, "--format=%H%x09%cI%x09%s").stdout + prs: dict[str, tuple[str, str]] = {} + subject_pr = re.compile(r"\(#(\d+)\)\s*$") + for line in out.splitlines(): + parts = line.split("\t", 2) + if len(parts) != 3: + continue + sha, date, subject = parts + m = subject_pr.search(subject) + if m: + prs.setdefault(m.group(1), (sha[:9], date)) + self._prs = prs + + def pr_merged(self, number: str) -> tuple[str, str] | None: + self._load_prs() + assert self._prs is not None + return self._prs.get(number) + + def symbol_on_main(self, symbol: str) -> bool: + self._symbol_probes += 1 + return self._git("grep", "--quiet", "-I", "-F", "-e", symbol, self.ref).returncode == 0 + + +# --------------------------------------------------------------------------------------------- +# Screening +# --------------------------------------------------------------------------------------------- + +_STRENGTH_ORDER = {"strong": 0, "medium": 1, "weak": 2} + + +@dataclass(frozen=True) +class Signal: + strength: str # "strong" | "medium" | "weak" + code: str + detail: str + + +@dataclass +class ItemReport: + num: int + heading: str + source: str + last_read: str | None + crossrefs: list[int] + subject_counts: dict[str, int] + signals: list[Signal] = field(default_factory=list) + unresolved: list[str] = field(default_factory=list) + notes: list[str] = field(default_factory=list) + + @property + def verdict(self) -> str: + if any(s.strength == "strong" for s in self.signals): + return "candidate" + if any(s.strength == "medium" for s in self.signals): + return "weak-candidate" + return "no-signal" + + @property + def rank(self) -> tuple[int, int, int]: + best = min((_STRENGTH_ORDER[s.strength] for s in self.signals), default=3) + strong = sum(1 for s in self.signals if s.strength == "strong") + return (best, -strong, self.num) + + +def screen_item( + num: int, + heading: str, + body: str, + repo: RepoReader, + *, + source: str = "docs/BACKLOG.md", + probe_symbols: bool = True, +) -> ItemReport: + """Resolve one open item's subjects against the ref and grade what came back.""" + subjects = subjects_in(body) + last_read = newest_date_in(body) + counts = {kind: sum(1 for s in subjects if s.kind == kind) for kind in _SUBJECT_KINDS} + report = ItemReport( + num=num, + heading=heading, + source=source, + last_read=last_read, + crossrefs=crossrefs_in(body), + subject_counts=counts, + ) + shallow = repo.is_shallow() + # Two subjects can name ONE file -- `scripts/hooks/worktree_gate.ps1` and the bare + # `worktree_gate.ps1:654` both appear in the #1229 row and both resolve to the same path. + # Deduplicating on the raw text would report that file's date twice, which reads as two + # independent pieces of evidence for the same fact. + screened_paths: set[str] = set() + if last_read is None: + # No date means the date-based comparisons cannot run. Surfacing the item is the + # over-firing direction and therefore the correct one. + report.signals.append( + Signal( + "medium", "no-date-anchor", "the row carries no date, so nothing dates its subjects" + ) + ) + + for subject in subjects: + if subject.kind == "sha": + _screen_sha(report, subject, repo, shallow) + elif subject.kind == "pr": + _screen_pr(report, subject, repo) + elif subject.kind == "path": + _screen_path(report, subject, repo, last_read, shallow, screened_paths) + elif subject.kind == "symbol" and probe_symbols: + if repo.symbol_on_main(subject.value): + report.signals.append( + Signal( + "weak", + "symbol-on-main", + f"`{subject.value}` (line {subject.lineno}) is present on the ref", + ) + ) + else: + report.unresolved.append(f"symbol {subject.value} not found on the ref") + return report + + +def _screen_sha(report: ItemReport, subject: Subject, repo: RepoReader, shallow: bool) -> None: + sha = subject.value + if not repo.is_commit(sha): + if shallow: + report.signals.append( + Signal( + "medium", + "sha-unverifiable-shallow", + f"{sha} (line {subject.lineno}) is not a commit in THIS clone, which is " + f"shallow -- it may sit beyond a graft boundary, so this is not a " + f"'not on main' answer", + ) + ) + else: + report.unresolved.append(f"sha {sha} is not a commit in this clone") + return + if repo.is_ancestor(sha): + strength = "strong" if subject.wording == "landing" else "medium" + code = "sha-ancestor-landing" if subject.wording == "landing" else "sha-ancestor" + report.signals.append( + Signal( + strength, + code, + f"{sha} (line {subject.lineno}, worded as {subject.wording}) IS an ancestor of the ref", + ) + ) + return + if shallow: + report.signals.append( + Signal( + "medium", + "sha-unverifiable-shallow", + f"{sha} (line {subject.lineno}) resolves but ancestry returned false under a " + f"SHALLOW clone -- the walk may have stopped at a graft, so the answer is unknown", + ) + ) + else: + report.unresolved.append(f"sha {sha} is not an ancestor of the ref") + + +def _screen_pr(report: ItemReport, subject: Subject, repo: RepoReader) -> None: + hit = repo.pr_merged(subject.value) + if hit is None: + report.unresolved.append(f"PR #{subject.value} has no merge commit on the ref") + return + sha, date = hit + strength = "strong" if subject.wording == "landing" else "medium" + code = "pr-merged-landing" if subject.wording == "landing" else "pr-merged" + report.signals.append( + Signal( + strength, + code, + f"PR #{subject.value} (line {subject.lineno}, worded as {subject.wording}) merged as " + f"{sha} on {date}", + ) + ) + + +def _screen_path( + report: ItemReport, + subject: Subject, + repo: RepoReader, + last_read: str | None, + shallow: bool, + screened: set[str], +) -> None: + resolved = repo.path_on_main(subject.value) + if resolved is None: + report.unresolved.append(f"path {subject.value} is absent from the ref (or ambiguous)") + return + if resolved in screened: + return + screened.add(resolved) + added = repo.path_added(resolved) + changed = repo.path_last_changed(resolved) + if added is None and shallow: + report.notes.append( + f"{resolved}: no add commit on the ref -- under a shallow clone that usually means the " + f"file predates the graft boundary, not that it was never added" + ) + if last_read is None: + return + if added is not None and added[:10] > last_read: + report.signals.append( + Signal( + "strong", + "path-added-after", + f"{resolved} was ADDED to the ref on {added[:10]}, after the row's newest date " + f"{last_read} -- it did not exist when this row was last read", + ) + ) + elif changed is not None and changed[:10] > last_read: + report.signals.append( + Signal( + "weak" if subject.wording != "landing" else "medium", + "path-changed-after", + f"{resolved} last changed on the ref on {changed[:10]}, after the row's newest " + f"date {last_read}", + ) + ) + + +# --------------------------------------------------------------------------------------------- +# Controls +# --------------------------------------------------------------------------------------------- + +#: A synthetic row modelled on #1229 and #1040. It exercises the EXTRACTOR with no git and no live +#: ledger, so a change that quietly stops matching one subject kind reds here rather than showing up +#: as a smaller candidate list nobody can tell from a cleaner ledger. +CONTROL_BODY = """## 4242. a synthetic control row + +> Re-scored 2026-08-20 -> P2. The ordering limb landed in c7f0e308 naming BACKLOG #1229, +> and `Remove-QuotedSpans` is called from worktree_gate.ps1:654. That LANDED on 2026-08-23 +> in `889dd9409` (PR #547). Measured at efe061a3f on this branch. +> The test is tests/test_worktree_gate_quote_straddle.py and the store is +> [`store.py`](../messagefoundry/store/store.py). See also #999 and BACKLOG #1268. +""" + +#: What CONTROL_BODY must yield. Written out rather than computed so the assertion cannot drift with +#: the code it checks. `#999` appears in the body and must be in NO list: a bare `#N` is never read. +CONTROL_EXPECTED: dict[str, set[str]] = { + "sha": {"c7f0e308", "889dd9409", "efe061a3f"}, + "pr": {"547"}, + "path": { + "worktree_gate.ps1", + "tests/test_worktree_gate_quote_straddle.py", + "messagefoundry/store/store.py", + }, + "symbol": {"Remove-QuotedSpans"}, +} +CONTROL_CROSSREFS = [1229, 1268] +CONTROL_DATE = "2026-08-23" + +#: The two measured cases. While one is OPEN this screen MUST call it a candidate. +LEDGER_CONTROLS = (1229, 1040) + + +def extractor_control() -> list[str]: + """Failures of the pure-layer control. Empty means the extractor still sees every subject kind.""" + failures: list[str] = [] + got: dict[str, set[str]] = {kind: set() for kind in _SUBJECT_KINDS} + for subject in subjects_in(CONTROL_BODY): + got[subject.kind].add(subject.value) + for kind, expected in CONTROL_EXPECTED.items(): + missing = expected - got[kind] + if missing: + failures.append(f"extractor lost {kind} subject(s): {sorted(missing)}") + if "999" in got["pr"]: + failures.append("extractor read a bare `#999` as a pull request -- the forms are ambiguous") + if "1229" in got["pr"] or "1268" in got["pr"]: + failures.append("extractor read a `BACKLOG #N` cross-reference as a pull request") + if crossrefs_in(CONTROL_BODY) != CONTROL_CROSSREFS: + failures.append( + f"cross-references came out {crossrefs_in(CONTROL_BODY)}, expected {CONTROL_CROSSREFS}" + ) + if newest_date_in(CONTROL_BODY) != CONTROL_DATE: + failures.append( + f"newest date came out {newest_date_in(CONTROL_BODY)!r}, expected {CONTROL_DATE!r}" + ) + # The negative arm. A probe validated on one input is not validated. + if subjects_in("nothing here but prose and a bare #12 reference"): + failures.append("extractor invented a subject from a row that names none") + return failures + + +def probe_control(repo: RepoReader, ref_sha: str, *, probe_symbols: bool) -> list[str]: + """Failures of the structural control over the git probes, positive AND negative arm each.""" + failures: list[str] = [] + if not repo.is_commit(ref_sha): + failures.append(f"is_commit said the ref's own commit {ref_sha} is not a commit") + if not repo.is_ancestor(ref_sha): + failures.append(f"is_ancestor said the ref's own commit {ref_sha} is not an ancestor") + absent_sha = "0" * 40 + if repo.is_commit(absent_sha): + failures.append("is_commit resolved an all-zero sha -- the probe cannot say no") + if repo.path_on_main("docs/BACKLOG.md") is None: + failures.append("path_on_main cannot find docs/BACKLOG.md") + if repo.path_on_main("docs/no-such-file-9d3f2b.md") is not None: + failures.append("path_on_main resolved an invented path -- the probe cannot say no") + if repo.path_last_changed("docs/BACKLOG.md") is None: + failures.append("path_last_changed has no date for docs/BACKLOG.md") + if probe_symbols: + if not repo.symbol_on_main("parse_items"): + failures.append("symbol_on_main cannot find `parse_items`") + if repo.symbol_on_main("zzq_no_such_symbol_9d3f2b"): + failures.append("symbol_on_main found an invented symbol -- the probe cannot say no") + return failures + + +# --------------------------------------------------------------------------------------------- +# Driver +# --------------------------------------------------------------------------------------------- + + +@dataclass +class OpenItem: + num: int + heading: str + body: str + source: str + + +def open_items(sources: Sequence[tuple[str, str]], status_check: Any) -> list[OpenItem]: + """Every OPEN item with its body text, using `parse_items` for both the split and the status.""" + parse_items = status_check.parse_items + out: list[OpenItem] = [] + for label, text in sources: + lines = text.splitlines() + items = parse_items(text) + for index, item in enumerate(items): + if not item.is_open: + continue + start = item.line - 1 + end = items[index + 1].line - 1 if index + 1 < len(items) else len(lines) + heading = lines[start].removeprefix("## ").strip() if start < len(lines) else "" + out.append(OpenItem(item.num, heading, "\n".join(lines[start:end]), label)) + return out + + +def _render(reports: Sequence[ItemReport], *, include_weak: bool, verbose: bool) -> list[str]: + lines: list[str] = [] + for report in reports: + if report.verdict == "no-signal" and not verbose: + continue + if report.verdict == "weak-candidate" and not include_weak and not verbose: + continue + lines.append("") + lines.append(f"#{report.num} [{report.verdict.upper()}] {report.heading[:110]}") + counted = ", ".join(f"{k}={report.subject_counts[k]}" for k in _SUBJECT_KINDS) + lines.append( + f" source {report.source} | newest date in row: {report.last_read or '(none)'} " + f"| subjects {counted}" + ) + if report.crossrefs: + lines.append(f" cites BACKLOG {', '.join('#' + str(n) for n in report.crossrefs)}") + for signal in sorted(report.signals, key=lambda s: _STRENGTH_ORDER[s.strength]): + if signal.strength == "weak" and not (include_weak or verbose): + continue + lines.append(f" [{signal.strength:6}] {signal.code}: {signal.detail}") + for note in report.notes if verbose else []: + lines.append(f" [note ] {note}") + for item in report.unresolved if verbose else []: + lines.append(f" [absent] {item}") + return lines + + +def main(argv: list[str] | None = None) -> int: + # This module prints ledger headings verbatim, and docs/BACKLOG.md is a sanctioned holdout for + # characters cp1252 cannot represent. Without this a stock Windows console aborts mid-report. + if hasattr(sys.stdout, "reconfigure"): + sys.stdout.reconfigure(encoding="utf-8", errors="replace") + + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + ap.add_argument( + "--root", type=Path, default=_ROOT, help="repository to read (default: this one)" + ) + ap.add_argument("--ref", default="origin/main", help="the ref a subject must exist on") + ap.add_argument( + "--backlog", + type=Path, + action="append", + dest="backlogs", + metavar="PATH", + help="a file holding numbered items; repeatable. Defaults to backlog_status_check's sources.", + ) + ap.add_argument("--item", type=int, action="append", help="screen only these item numbers") + ap.add_argument("--include-weak", action="store_true", help="also list weak-candidate rows") + ap.add_argument( + "--verbose", action="store_true", help="show every row, note and absent subject" + ) + ap.add_argument("--no-symbols", action="store_true", help="skip the per-symbol git grep probes") + ap.add_argument( + "--max-symbol-probes", + type=int, + default=1500, + metavar="N", + help="cap on git-grep symbol probes; the number skipped is always reported", + ) + ap.add_argument("--self-test", action="store_true", help="run the controls only, then stop") + ap.add_argument("--json", action="store_true", dest="as_json") + args = ap.parse_args(argv) + + root: Path = args.root.resolve() + repo = GitRepo(root, args.ref) + probe_symbols = not args.no_symbols + + ref_sha = repo.resolve(args.ref) + if not ref_sha: + print(f"ERROR: cannot resolve {args.ref} in {root}", file=sys.stderr) + return 2 + + # THE CONTROLS RUN FIRST AND UNCONDITIONALLY. A report produced by a broken screen is worse than + # no report, because it reads as a clean ledger. + control_failures = extractor_control() + probe_control( + repo, ref_sha, probe_symbols=probe_symbols + ) + print(f"subject-exists screen -- ref {args.ref} @ {ref_sha[:9]}, root {root}") + print(f" shallow clone: {repo.is_shallow()}") + if control_failures: + print("CONTROL FAILED -- THE SCREEN IS BROKEN, NOT THE LEDGER CLEAN:", file=sys.stderr) + for failure in control_failures: + print(f" {failure}", file=sys.stderr) + return 2 + print( + f" controls: extractor OK ({sum(len(v) for v in CONTROL_EXPECTED.values())} subjects " + f"across {len(CONTROL_EXPECTED)} kinds, both arms), probes OK " + f"(positive and negative arm each{'' if probe_symbols else ', symbols skipped'})" + ) + if args.self_test: + return 0 + # The control runs symbol probes of its own. Counting them in with the screen's would inflate + # "what this run looked at" by a constant, and a probe total that is never zero cannot show that + # symbol probing was switched off. + control_probes = repo.symbol_probes + + status_check = _load_status_check() + default_sources = status_check.DEFAULT_SOURCES + paths: list[Path] = args.backlogs if args.backlogs else [root / p for p in default_sources] + sources: list[tuple[str, str]] = [] + for path in paths: + if path.exists(): + label = path.resolve().relative_to(root).as_posix() if path.is_absolute() else str(path) + sources.append((label, path.read_text(encoding="utf-8"))) + if not sources: + print("ERROR: no ledger source could be read", file=sys.stderr) + return 2 + + items = open_items(sources, status_check) + scanned = ", ".join( + f"{label} ({sum(1 for i in items if i.source == label)} open)" for label, _ in sources + ) + if args.item: + wanted = set(args.item) + items = [i for i in items if i.num in wanted] + + reports: list[ItemReport] = [] + symbols_skipped = 0 + for item in items: + allow = probe_symbols and repo.symbol_probes < args.max_symbol_probes + if probe_symbols and not allow: + symbols_skipped += 1 + reports.append( + screen_item( + item.num, + item.heading, + item.body, + repo, + source=item.source, + probe_symbols=allow, + ) + ) + reports.sort(key=lambda r: r.rank) + + # THE LEDGER CONTROL. A control that stops applying must say so; it must not read like a pass. + control_lines: list[str] = [] + control_failed = False + by_num = {r.num: r for r in reports} + for num in LEDGER_CONTROLS: + hit = by_num.get(num) + if hit is None: + control_lines.append( + f" #{num}: RETIRED as a control -- no longer an OPEN item in the scanned sources " + f"(or excluded by --item). It is not evidence either way." + ) + elif hit.verdict == "candidate": + control_lines.append(f" #{num}: FIRED as expected ({len(hit.signals)} signal(s))") + else: + control_failed = True + control_lines.append( + f" #{num}: DID NOT FIRE -- verdict {hit.verdict}. This is a KNOWN-TRUE case, so " + f"the screen is under-firing and its empty findings mean nothing." + ) + + counts = { + verdict: sum(1 for r in reports if r.verdict == verdict) + for verdict in ("candidate", "weak-candidate", "no-signal") + } + totals = {kind: sum(r.subject_counts[kind] for r in reports) for kind in _SUBJECT_KINDS} + + if args.as_json: + print( + json.dumps( + { + "ref": args.ref, + "ref_sha": ref_sha, + "shallow": repo.is_shallow(), + "scanned": scanned, + "items_examined": len(reports), + "subjects_extracted": totals, + "symbol_probes": repo.symbol_probes - control_probes, + "control_symbol_probes": control_probes, + "items_with_symbols_skipped_by_cap": symbols_skipped, + "verdicts": counts, + "ledger_controls": control_lines, + "items": [ + { + "num": r.num, + "verdict": r.verdict, + "heading": r.heading, + "last_read": r.last_read, + "signals": [ + {"strength": s.strength, "code": s.code, "detail": s.detail} + for s in r.signals + ], + } + for r in reports + if r.verdict != "no-signal" or args.verbose + ], + }, + indent=2, + ) + ) + return 1 if control_failed else 0 + + print(f" scanned: {scanned}") + print(f" items examined (OPEN only): {len(reports)}") + print( + " subjects extracted: " + + ", ".join(f"{k}={totals[k]}" for k in _SUBJECT_KINDS) + + f" (total {sum(totals.values())})" + ) + print( + f" symbol probes run: {repo.symbol_probes - control_probes} " + f"(plus {control_probes} by the control)" + + ( + f"; items whose symbols were skipped by the {args.max_symbol_probes} cap: " + f"{symbols_skipped}" + if symbols_skipped + else "" + ) + + ("" if probe_symbols else " (--no-symbols)") + ) + print( + f" verdicts: {counts['candidate']} candidate, {counts['weak-candidate']} weak-candidate, " + f"{counts['no-signal']} no-signal" + ) + print(" ledger controls (the two measured cases):") + for line in control_lines: + print(line) + + body = _render(reports, include_weak=args.include_weak, verbose=args.verbose) + if body: + print("\n--- candidates, for a person to read. THIS TOOL FLIPS NOTHING. ---") + for line in body: + print(line) + else: + print("\nNo rows to list at this verdict threshold. The controls above are what make that") + print("readable as a clean ledger rather than a screen that stopped matching.") + + if control_failed: + print( + "\nERROR: a known-true ledger control did not fire; treat the list above as unsound.", + file=sys.stderr, + ) + return 1 + return 0 + + +if __name__ == "__main__": # pragma: no cover + raise SystemExit(main()) diff --git a/tests/test_subject_exists_screen.py b/tests/test_subject_exists_screen.py new file mode 100644 index 000000000..3f2bf9e06 --- /dev/null +++ b/tests/test_subject_exists_screen.py @@ -0,0 +1,531 @@ +# SPDX-License-Identifier: AGPL-3.0-or-later +# Copyright (C) 2026 MessageFoundry Organization and contributors +"""The subject-exists screen (BACKLOG #1426). + +EVERY TEST HERE RUNS AGAINST A FAKE REPOSITORY, and that is a decision rather than convenience. The +screen's job is to answer "is this subject on `origin/main`", so a test driving real git would assert +facts about whichever clone happens to run it -- facts that change under every merge and vanish under +a shallow fetch. The fake pins the SCREEN's behaviour; the live probes are pinned separately by the +tool's own structural control, which runs on every invocation and has a negative arm. + +TWO CONTROL LAYERS ARE TESTED, NOT ONE. A screen that finds nothing must be distinguishable from a +screen that is broken, so it is not enough that the controls pass on good code -- they must FAIL on +broken code. `test_the_probe_control_fails_against_a_broken_probe` and +`test_the_extractor_control_fails_when_a_subject_kind_is_lost` are those negative arms. Without them +the controls are one-sided instruments, which is the failure mode this repository has recorded +repeatedly. + +THE TWO KNOWN-TRUE CASES ARE FIXTURES HERE, ABRIDGED FROM THE REAL ROWS. #1229 and #1040 were each +dispatched as a build on 2026-09-03 and each was already complete. Their shapes are pinned so a +change that stops matching one of them reds here rather than showing up as a shorter candidate list +nobody can tell from a cleaner ledger. +""" + +from __future__ import annotations + +import importlib.util +import sys +from pathlib import Path + +import pytest + +_SPEC = importlib.util.spec_from_file_location( + "_subject_exists_screen", + Path(__file__).resolve().parents[1] / "scripts" / "docs" / "subject_exists_screen.py", +) +assert _SPEC is not None and _SPEC.loader is not None +ses = importlib.util.module_from_spec(_SPEC) +# REGISTERED BEFORE EXECUTION, and it is not optional here as it is in the sibling loaders. A +# `@dataclass` in the loaded module sends `dataclasses._is_type` to `sys.modules[cls.__module__]` +# while the class body is being processed; an unregistered module makes that None and collection +# dies with `AttributeError: 'NoneType' object has no attribute '__dict__'`, which names neither +# this file nor the real cause. +sys.modules[_SPEC.name] = ses +_SPEC.loader.exec_module(ses) + + +# --- a fake repository ---------------------------------------------------------------------------- + + +class FakeRepo: + """A `RepoReader` whose every answer is declared, so a test asserts the SCREEN, not a clone.""" + + def __init__( + self, + *, + shallow: bool = False, + commits: set[str] | None = None, + ancestors: set[str] | None = None, + tree: set[str] | None = None, + added: dict[str, str] | None = None, + changed: dict[str, str] | None = None, + prs: dict[str, tuple[str, str]] | None = None, + symbols: set[str] | None = None, + ) -> None: + self._shallow = shallow + self._commits = commits or set() + self._ancestors = ancestors or set() + self._tree = tree or set() + self._added = added or {} + self._changed = changed or {} + self._prs = prs or {} + self._symbols = symbols or set() + + def is_shallow(self) -> bool: + return self._shallow + + def is_commit(self, sha: str) -> bool: + return sha in self._commits or sha in self._ancestors + + def is_ancestor(self, sha: str) -> bool: + return sha in self._ancestors + + def path_on_main(self, path: str) -> str | None: + if path in self._tree: + return path + if "/" in path: + return None + matches = [p for p in self._tree if p.rsplit("/", 1)[-1] == path] + return matches[0] if len(matches) == 1 else None + + def path_added(self, path: str) -> str | None: + return self._added.get(path) + + def path_last_changed(self, path: str) -> str | None: + return self._changed.get(path) + + def pr_merged(self, number: str) -> tuple[str, str] | None: + return self._prs.get(number) + + def symbol_on_main(self, symbol: str) -> bool: + return symbol in self._symbols + + +def _values(body: str, kind: str) -> set[str]: + return {s.value for s in ses.subjects_in(body) if s.kind == kind} + + +def _codes(report: object) -> set[str]: + return {s.code for s in report.signals} # type: ignore[attr-defined] + + +# --- what a subject IS ---------------------------------------------------------------------------- + + +def test_a_sha_is_found() -> None: + assert _values("landed in c7f0e308 today", "sha") == {"c7f0e308"} + + +def test_a_full_length_sha_is_found() -> None: + sha = "889dd9409" + "a" * 31 + assert _values(f"see {sha}", "sha") == {sha} + + +def test_a_path_with_directories_is_found() -> None: + assert "messagefoundry/store/store.py" in _values("see messagefoundry/store/store.py", "path") + + +def test_a_path_inside_a_code_span_is_found_WHOLE() -> None: + """REGRESSION, and it was found by running the tool rather than predicted. + + The first lookbehind refused a backtick, so a path written the way this ledger usually writes one + could not match at its start and the regex matched a SUFFIX instead: the #1229 row's + `scripts/hooks/worktree_gate.ps1` came out `hooks/worktree_gate.ps1`. That resolves against + nothing and reported as ABSENT -- a truncation rendering as a confident negative, which is the + worst of the three outcomes. + """ + got = _values("the gate is `scripts/hooks/worktree_gate.ps1:382-383` today", "path") + assert "scripts/hooks/worktree_gate.ps1" in got + assert "hooks/worktree_gate.ps1" not in got + + +def test_a_hyphenated_filename_is_not_truncated_at_its_hyphen() -> None: + """Same defect one character along: `install-git-hooks.ps1` must not yield `git-hooks.ps1`.""" + got = _values("install-git-hooks.ps1 copies them", "path") + assert "install-git-hooks.ps1" in got + assert "git-hooks.ps1" not in got + + +def test_a_markdown_link_path_loses_its_relative_prefix() -> None: + body = "see [`store.py`](../messagefoundry/store/store.py) for it" + assert "messagefoundry/store/store.py" in _values(body, "path") + + +def test_a_bare_filename_with_a_line_range_is_found() -> None: + """`worktree_gate.ps1:347-388` is how the #1229 row names its own subject.""" + assert "worktree_gate.ps1" in _values("blanks at worktree_gate.ps1:347-388", "path") + + +def test_a_bare_filename_is_not_repeated_when_a_full_path_on_the_line_covers_it() -> None: + got = _values("messagefoundry/store/store.py is the file", "path") + assert got == {"messagefoundry/store/store.py"} + + +def test_a_backticked_snake_case_symbol_is_found() -> None: + assert _values("the reader is `reset_stale_inflight` here", "symbol") == { + "reset_stale_inflight" + } + + +def test_a_powershell_verb_noun_symbol_is_found() -> None: + assert _values("calls `Remove-QuotedSpans` first", "symbol") == {"Remove-QuotedSpans"} + + +def test_a_symbol_keeps_one_identity_with_or_without_call_parens() -> None: + assert _values("`parse_items()` and `parse_items`", "symbol") == {"parse_items"} + + +# --- what a subject IS NOT ------------------------------------------------------------------------ + + +def test_a_bare_hash_number_is_never_a_pull_request() -> None: + """`#N` spells a pull request and a backlog item identically. + + A security record entry reading "the build is #156" resolved to a pull request while backlog #156 + was unrelated work. Guessing is what produced that, so neither form is guessed.""" + assert _values("see #547 and #1229 for context", "pr") == set() + + +def test_a_backlog_cross_reference_is_never_a_pull_request() -> None: + assert _values("closes BACKLOG #547 eventually", "pr") == set() + + +def test_only_the_literal_pr_form_resolves_a_pull_request() -> None: + assert _values("landed in PR #547 and pull request #560", "pr") == {"547", "560"} + + +def test_a_cross_reference_is_reported_as_context_and_not_as_a_subject() -> None: + body = "blocked on BACKLOG #1268 and BACKLOG #1229" + assert ses.crossrefs_in(body) == [1229, 1268] + assert _values(body, "sha") == set() + + +def test_a_decimal_run_is_not_read_as_a_sha() -> None: + assert _values("the count was 12345678 rows", "sha") == set() + + +def test_an_item_heading_does_not_name_itself_as_a_file() -> None: + assert _values("## 1229. the worktree gate blanks spans", "path") == set() + + +def test_a_short_identifier_is_not_a_symbol() -> None: + """Under eight characters an identifier is too common for its presence on main to mean anything.""" + assert _values("`db_read` and `x_y`", "symbol") == set() + + +def test_a_backticked_command_is_not_a_symbol() -> None: + assert _values("run `git merge-base --is-ancestor`", "symbol") == set() + + +def test_prose_naming_nothing_yields_no_subject() -> None: + """The extractor's negative arm. A probe validated on one input is not validated.""" + assert ses.subjects_in("this row is entirely prose and names no code at all") == [] + + +# --- wording, dates ------------------------------------------------------------------------------- + + +def test_landing_wording_is_recognised() -> None: + (subject,) = ses.subjects_in("that LANDED on 2026-08-23 in c7f0e308") + assert subject.wording == "landing" + + +def test_measurement_wording_is_recognised() -> None: + (subject,) = ses.subjects_in("Measured at efe061a3f on this branch") + assert subject.wording == "measurement" + + +def test_landing_wins_a_line_that_carries_both() -> None: + """Ranking "measured after it landed" as a base ref would drop the strongest signal there is.""" + (subject,) = ses.subjects_in("measured after it landed in c7f0e308") + assert subject.wording == "landing" + + +def test_a_later_landing_mention_upgrades_an_earlier_neutral_one() -> None: + body = "the base is c7f0e308\nand c7f0e308 landed last week" + (subject,) = [s for s in ses.subjects_in(body) if s.kind == "sha"] + assert subject.wording == "landing" + + +def test_the_newest_date_in_the_row_is_taken() -> None: + assert ses.newest_date_in("filed 2026-08-12, re-scored 2026-08-20") == "2026-08-20" + + +def test_a_row_with_no_date_reports_none() -> None: + assert ses.newest_date_in("no dates at all here") is None + + +# --- screening ------------------------------------------------------------------------------------ + + +def test_an_ancestor_sha_on_a_landing_line_is_a_strong_signal() -> None: + repo = FakeRepo(ancestors={"c7f0e308"}) + report = ses.screen_item(1, "h", "it landed in c7f0e308 on 2026-08-20", repo) + assert report.verdict == "candidate" + assert "sha-ancestor-landing" in _codes(report) + + +def test_an_ancestor_sha_cited_as_a_base_ref_is_only_medium() -> None: + """ "Measured at `efe061a3f`" is an ordinary citation of an ancestor and must not read as news.""" + repo = FakeRepo(ancestors={"efe061a3f"}) + report = ses.screen_item(1, "h", "Measured at efe061a3f on 2026-08-20", repo) + assert report.verdict == "weak-candidate" + assert "sha-ancestor" in _codes(report) + + +def test_a_merged_pull_request_on_a_landing_line_is_a_strong_signal() -> None: + repo = FakeRepo(prs={"547": ("889dd9409", "2026-08-23T12:56:57Z")}) + report = ses.screen_item(1, "h", "That LANDED in PR #547 on 2026-08-20", repo) + assert "pr-merged-landing" in _codes(report) + assert report.verdict == "candidate" + + +def test_an_unmerged_pull_request_produces_no_signal() -> None: + repo = FakeRepo() + report = ses.screen_item(1, "h", "opened as PR #999 on 2026-08-20", repo) + assert report.verdict == "no-signal" + assert any("PR #999" in u for u in report.unresolved) + + +def test_a_path_added_after_the_rows_newest_date_is_a_strong_signal() -> None: + """The sharpest shape there is: the file did not exist when anyone last read the row.""" + repo = FakeRepo( + tree={"tests/test_conftest_name_collision_guard.py"}, + added={"tests/test_conftest_name_collision_guard.py": "2026-08-26T10:00:00Z"}, + ) + body = "re-scored 2026-08-25; the fix needs tests/test_conftest_name_collision_guard.py" + report = ses.screen_item(1, "h", body, repo) + assert "path-added-after" in _codes(report) + assert report.verdict == "candidate" + + +def test_a_path_added_before_the_rows_newest_date_is_not_that_signal() -> None: + repo = FakeRepo( + tree={"tests/test_thing.py"}, + added={"tests/test_thing.py": "2026-08-01T10:00:00Z"}, + changed={"tests/test_thing.py": "2026-08-01T10:00:00Z"}, + ) + report = ses.screen_item(1, "h", "re-scored 2026-08-25; see tests/test_thing.py", repo) + assert "path-added-after" not in _codes(report) + + +def test_one_file_named_two_ways_yields_one_signal() -> None: + """`scripts/hooks/worktree_gate.ps1` and a bare `worktree_gate.ps1:654` are ONE piece of evidence. + + Deduplicating on the raw text instead of the resolved path reported the file's date twice, which + renders as two independent findings for a single fact.""" + repo = FakeRepo( + tree={"scripts/hooks/worktree_gate.ps1"}, + changed={"scripts/hooks/worktree_gate.ps1": "2026-09-01T00:00:00Z"}, + ) + body = ( + "re-scored 2026-08-20; `scripts/hooks/worktree_gate.ps1` is called at worktree_gate.ps1:654" + ) + report = ses.screen_item(1, "h", body, repo) + assert [s.code for s in report.signals].count("path-changed-after") == 1 + + +def test_a_row_with_no_date_is_surfaced_rather_than_skipped() -> None: + """No date means the date comparisons cannot run, and silence is the under-firing direction.""" + report = ses.screen_item(1, "h", "a row naming no date whatsoever", FakeRepo()) + assert "no-date-anchor" in _codes(report) + assert report.verdict == "weak-candidate" + + +def test_a_symbol_present_on_the_ref_is_only_a_weak_signal() -> None: + repo = FakeRepo(symbols={"reset_stale_inflight"}) + report = ses.screen_item(1, "h", "on 2026-08-20 `reset_stale_inflight` matters", repo) + assert "symbol-on-main" in _codes(report) + assert report.verdict == "no-signal" + + +def test_symbol_probing_can_be_switched_off() -> None: + repo = FakeRepo(symbols={"reset_stale_inflight"}) + report = ses.screen_item( + 1, "h", "on 2026-08-20 `reset_stale_inflight` matters", repo, probe_symbols=False + ) + assert "symbol-on-main" not in _codes(report) + + +# --- the shallow-clone trap ----------------------------------------------------------------------- + + +def test_under_a_shallow_clone_an_unresolvable_sha_is_UNKNOWN_not_absent() -> None: + """Constraint 7. This clone really is shallow -- 16 graft points, measured 2026-09-03 -- so a + walk that stops at a boundary must not render as "not on main".""" + repo = FakeRepo(shallow=True) + report = ses.screen_item(1, "h", "it landed in abc1234 on 2026-08-20", repo) + assert "sha-unverifiable-shallow" in _codes(report) + assert report.unresolved == [] + + +def test_under_a_shallow_clone_a_false_ancestry_answer_is_UNKNOWN() -> None: + """A TRUE ancestry answer is sound under a graft; a FALSE one may only mean the walk stopped.""" + repo = FakeRepo(shallow=True, commits={"abc1234"}) + report = ses.screen_item(1, "h", "it landed in abc1234 on 2026-08-20", repo) + assert "sha-unverifiable-shallow" in _codes(report) + + +def test_in_a_COMPLETE_clone_the_same_sha_is_reported_as_absent() -> None: + """The other arm. Without it, "unknown" would be indistinguishable from a screen that never + reports a negative at all.""" + repo = FakeRepo(shallow=False, commits={"abc1234"}) + report = ses.screen_item(1, "h", "it landed in abc1234 on 2026-08-20", repo) + assert "sha-unverifiable-shallow" not in _codes(report) + assert any("abc1234" in u for u in report.unresolved) + + +# --- the two known-true cases, as fixtures -------------------------------------------------------- + +#: Abridged from the real row. The escape limb shipped 2026-08-22 in `3c5cb9885`, whose subject names +#: BACKLOG #1268 rather than #1229, so a search keyed on the item number never finds it. +CASE_1229 = """## 1229. the worktree gate blanks double-quoted spans FIRST + +> Re-scored 2026-08-20 -> P2. The ORDERING limb is genuinely shipped: Remove-QuotedSpans +> (worktree_gate.ps1:347-388) is CALLED on the scan path at :654. Landed in c7f0e308 naming #1229. +> The TEST limbs are shipped too -- tests/test_worktree_gate_quote_straddle.py asserts it, and it is +> registered in tests/tooling_manifest.txt:109. +> Filed 2026-08-12 -- a LIVE FAIL-OPEN in `scripts/hooks/worktree_gate.ps1:382-383`. +""" + +#: Abridged from the real row. It stayed open behind a note saying the banner was left for an archive +#: pass, while an older re-score beneath it still described the landed work as outstanding. +CASE_1040 = """## 1040. Hook deny text is attacker-influenceable output + +> BOTH REMAINING SURFACES ARE CLOSED 2026-08-27; banner left open for the archive pass. +> THIS ROW'S OWN COORDINATES WERE STALE. It still pointed at the `claim_check.py` site as +> outstanding. That LANDED on 2026-08-23 in `889dd9409` (PR #547). +""" + + +def test_the_1229_case_comes_out_a_candidate() -> None: + repo = FakeRepo( + ancestors={"c7f0e308"}, + tree={ + "scripts/hooks/worktree_gate.ps1", + "tests/test_worktree_gate_quote_straddle.py", + "tests/tooling_manifest.txt", + }, + changed={"scripts/hooks/worktree_gate.ps1": "2026-09-01T00:00:00Z"}, + symbols={"Remove-QuotedSpans"}, + ) + report = ses.screen_item(1229, "h", CASE_1229, repo) + assert report.verdict == "candidate" + assert "sha-ancestor-landing" in _codes(report) + + +def test_the_1040_case_comes_out_a_candidate() -> None: + repo = FakeRepo( + ancestors={"889dd9409"}, + tree={"scripts/hooks/claim_check.py"}, + prs={"547": ("889dd9409", "2026-08-23T12:56:57Z")}, + ) + report = ses.screen_item(1040, "h", CASE_1040, repo) + assert report.verdict == "candidate" + assert {"sha-ancestor-landing", "pr-merged-landing"} <= _codes(report) + + +@pytest.mark.parametrize("case", [CASE_1229, CASE_1040]) +def test_neither_case_fires_against_a_ref_that_carries_none_of_it(case: str) -> None: + """The negative arm on the fixtures themselves. + + Both rows come out candidates above. If they came out candidates here too, the fixtures would be + proving that the screen fires, not that it fires ON THE RIGHT THING.""" + report = ses.screen_item(1, "h", case, FakeRepo()) + assert report.verdict == "no-signal" + + +# --- the controls, in BOTH directions ------------------------------------------------------------- + + +def test_the_extractor_control_passes_on_the_shipped_extractor() -> None: + assert ses.extractor_control() == [] + + +def test_the_extractor_control_fails_when_a_subject_kind_is_lost( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The control's own negative arm. + + A control that has never failed is indistinguishable from one that cannot. Neutering the symbol + reader must red the control rather than quietly shrinking every future report.""" + monkeypatch.setattr(ses, "_symbol_from_span", lambda span: None) + failures = ses.extractor_control() + assert failures, "the extractor control passed with the symbol reader neutered" + assert any("symbol" in f for f in failures) + + +def test_the_probe_control_passes_against_a_sound_fake() -> None: + repo = FakeRepo( + ancestors={"deadbee"}, + tree={"docs/BACKLOG.md"}, + changed={"docs/BACKLOG.md": "2026-09-02T00:00:00Z"}, + symbols={"parse_items"}, + ) + assert ses.probe_control(repo, "deadbee", probe_symbols=True) == [] + + +def test_the_probe_control_fails_against_a_probe_that_cannot_say_no() -> None: + """A probe validated on one input is not validated. This one answers TRUE to everything, which + is exactly the shape that makes a screen report a clean ledger forever.""" + + class AlwaysYes(FakeRepo): + def is_commit(self, sha: str) -> bool: + return True + + def path_on_main(self, path: str) -> str | None: + return path + + def symbol_on_main(self, symbol: str) -> bool: + return True + + repo = AlwaysYes(ancestors={"deadbee"}, changed={"docs/BACKLOG.md": "2026-09-02T00:00:00Z"}) + failures = ses.probe_control(repo, "deadbee", probe_symbols=True) + assert len(failures) >= 3, failures + assert any("cannot say no" in f for f in failures) + + +def test_the_probe_control_fails_against_a_probe_that_cannot_say_yes() -> None: + failures = ses.probe_control(FakeRepo(), "deadbee", probe_symbols=True) + assert failures + assert any("is not an ancestor" in f for f in failures) + + +def test_the_named_ledger_controls_are_the_two_measured_cases() -> None: + """Pinned so that dropping one is a visible edit rather than a quietly weaker screen.""" + assert ses.LEDGER_CONTROLS == (1229, 1040) + + +# --- reading the ledger --------------------------------------------------------------------------- + + +def test_open_items_uses_parse_items_and_carries_each_body() -> None: + """`parse_items` DEFINES item status; a hand-rolled scan is a second, silently different + definition (CLAUDE.md section 11). This asserts the screen goes through it.""" + status_check = ses._load_status_check() + text = ( + "# Backlog\n\n" + "## 10. an open one\n\n" + "> \N{INPUT SYMBOL FOR NUMBERS} **Re-scored 2026-08-20.**\n\n" + "body naming messagefoundry/store/store.py\n\n" + "## 11. a closed one\n\n" + "> \N{WHITE HEAVY CHECK MARK} **SHIPPED.**\n\n" + "other body\n" + ) + items = ses.open_items([("docs/BACKLOG.md", text)], status_check) + assert [i.num for i in items] == [10] + assert "messagefoundry/store/store.py" in items[0].body + assert items[0].source == "docs/BACKLOG.md" + + +def test_the_live_ledger_still_parses_and_still_has_open_items() -> None: + """The corpus control. Every assertion above is satisfied just as well by an empty ledger.""" + root = Path(__file__).resolve().parents[1] + status_check = ses._load_status_check() + text = (root / "docs" / "BACKLOG.md").read_text(encoding="utf-8") + items = ses.open_items([("docs/BACKLOG.md", text)], status_check) + print(f"open items in the live ledger: {len(items)}") + assert len(items) >= 50, ( + "the live ledger yielded almost no open items -- the reader is narrowing" + ) + assert any(ses.subjects_in(i.body) for i in items), "no open row names any code-side subject" diff --git a/tests/tooling_manifest.txt b/tests/tooling_manifest.txt index bb5029a6f..0eb4466d5 100644 --- a/tests/tooling_manifest.txt +++ b/tests/tooling_manifest.txt @@ -127,6 +127,7 @@ tests/test_session_mail.py tests/test_session_registry.py tests/test_setup_leak_gate_reports_source.py tests/test_stalled_prs.py +tests/test_subject_exists_screen.py tests/test_testpaths_webconsole_coverage.py tests/test_verdict_divergence_check.py tests/test_workflow_local_action_check.py