From 77a413424d277a3820288905e0e6ca5a35561e53 Mon Sep 17 00:00:00 2001 From: Dailen Gunter Date: Sat, 5 Sep 2026 14:04:35 -0400 Subject: [PATCH 1/2] fix: ask the harness question before naming a fault, and name the CI artifact Three defects, one shape: a step asked for a conclusion without saying what evidence would support it, so one got invented. Step 0's four mechanical checks all pass when the real cause is a harness that never reads hooks/hooks.json, and it then asked for "the likely cause" with a worked example naming Node. Two different invented causes reached a real project's CONTINUE.md that way; the actual cause was harness/omp/forge-bridge.ts never having been copied to ~/.omp/agent/extensions/, which this plugin's own harness/README.md documents. Step 0 now separates a broken machine from an intact one nobody is driving, resolves the harness from a marker table that harness/README.md owns, and reports a missing adapter as one line naming the file to copy. Not knowing the harness still produces a complete report. CLAUDECODE=1 does not establish Claude Code: omp sets it deliberately for tool compatibility alongside its own OMPCODE=1, so keying on it would return a confident wrong answer in exactly the situation being diagnosed. next_action took the lowest-numbered open task because status was the only field it had. The record set had no way to say why an open task is waiting, so the three backlog dispositions forge-standards defines all collapsed to open and "leave Deferred items alone" could not be evaluated. Task records gain a disposition field, absent reading as ready, with a Step 2a backfill row so the default is offered once rather than carried silently forever. The release CI gate now names its artifact, read for the commit being released and quoted with its run ids. A recorded gate result whose subject is a commit is absent, not stale, once commits land on top of it; one whose subject is a surface is not. Closes #8, closes #9, closes #10 Co-Authored-By: Claude Opus 5 (1M context) --- .claude-plugin/plugin.json | 2 +- CHANGELOG.md | 60 +++++++++++++++++ docs/index.html | 56 ++++++++-------- harness/README.md | 28 ++++++++ skills/forge-code/SKILL.md | 4 +- skills/forge-standards/SKILL.md | 7 +- skills/forge/SKILL.md | 24 +++++-- templates/TODO.md | 4 +- templates/forge-index.js | 12 +++- templates/forge-records-lib.js | 34 ++++++++++ templates/forge-records-migrate.js | 21 ++++++ templates/forge-views.js | 5 +- tests/harness-and-gates.test.js | 100 +++++++++++++++++++++++++++++ tests/records.test.js | 78 +++++++++++++++++++++- 14 files changed, 389 insertions(+), 46 deletions(-) create mode 100644 tests/harness-and-gates.test.js diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index e247b12..0e00ba6 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "forge-workflow", "displayName": "Forge", - "version": "1.5.2", + "version": "1.6.0", "description": "Gated spec-to-ship lifecycle for Claude Code. One command determines where a project stands and does the next thing: requirements discovery to 95 percent confidence, toolchain and repository bootstrap, then test-driven implementation with enforced resumability, documentation drift gates, and tagged releases.", "author": { "name": "Dailen Gunter", diff --git a/CHANGELOG.md b/CHANGELOG.md index 7aa5948..935a8b6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,66 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [1.6.0] - 2026-09-05 + +### Fixed + +- `/forge` Step 0 asks the harness question before it names a fault. Its four + mechanical checks all pass when the real cause is a harness that never reads + `hooks/hooks.json`, and it then asked for "the likely cause" with a worked + example naming Node, so a cause got invented. On one project two different + invented causes reached `CONTINUE.md` and were believed for a day; the actual + cause was `harness/omp/forge-bridge.ts` never having been copied to + `~/.omp/agent/extensions/`, which the plugin's own `harness/README.md` + documents. Step 0 now separates a broken machine from an intact one nobody is + driving, resolves the harness from a marker table in `harness/README.md`, and + reports the missing adapter as one line naming the file to copy. When it + cannot identify the harness it says so and reports the consequence anyway; + naming a cause the diagnostics did not establish is now explicitly refused. + `CLAUDECODE=1` does not establish Claude Code: omp sets it deliberately for + tool compatibility alongside its own `OMPCODE=1`. Issue #8 +- `next_action` in `.forge/index.json` no longer names a deliberately parked + task. It took the lowest-numbered open task, because `status` was the only + field it had. The cause was one level up: the record set had no way to say why + an open task is waiting, so the three backlog dispositions `forge-standards` + defines all collapsed to `open` and the detection ladder's "leave Deferred + items alone" could not be evaluated at all. Issue #9 +- The release CI gate names its artifact. Every other release gate implies an act + of measurement; CI runs elsewhere, so a recorded "CI green" line was the + cheapest place to look, and a release was cut on top of two pushes with a red + job. Issue #10 + +### Added + +- `disposition` on task records: `ready`, `needs-decision`, or `deferred`, with + an absent field reading as ready so a project that predates it still has a next + action. The index skips anything not ready, `docs/views/open-work.md` shows the + column, the linter rejects an unknown value or one on a non-task record, and + the migration keeps the TODO section a bullet sat under so `## Blocked` survives + as `needs-decision`. A `/forge` Step 2a backfill row offers to fill in the open + tasks once, since absent-means-ready is otherwise a silent wrong default forever +- `harness/README.md` gains a harness marker table (harness, environment marker, + adapter, extension path), the single owner of that knowledge, plus a note that + forge's enforcement degrades silently: a project can carry a sentence asserting + a `PreToolUse` hook denies writes under `docs/views/` that is simply false under + a harness which never loads it. Adding a harness is one row and one directory +- A rule for recorded gate results, in `/forge` Step 2 reconcile: a result whose + subject is a commit (CI, suite, coverage) is evidence only for the commit it was + measured against, and is absent rather than stale once commits land on top of + it. A result whose subject is a surface, such as the polish pass, does not + expire this way + +### Changed + +- The `docs/views/**` guarantee is stated as the portable one. The pre-push + `06_views` check is an ordinary program the project runs and holds under any + harness; the `PreToolUse` hook is the fast local deny wherever the harness loads + it. `forge-code` points at the mechanism `forge-standards` owns rather than + restating it +- `.forge/index.json` `open.tasks` carries `{ id, disposition }` rather than bare + IDs, matching how `open.uxd` already carries its severity. The file is generated + and gitignored, so `forge-index.js build` reproduces it + ## [1.5.2] - 2026-09-01 ### Changed diff --git a/docs/index.html b/docs/index.html index 04df4e1..f5162db 100644 --- a/docs/index.html +++ b/docs/index.html @@ -276,7 +276,7 @@

A guide for people who are not developersForge

AudienceTechnical, not a developer
Runs inClaude Code
Reading timeAbout 20 minutes
-
Editionv1.5.2  /  for Claude 5
+
Editionv1.6.0  /  for Claude 5
@@ -311,41 +311,39 @@

A guide for people who are not developersForge

-

New in v1.5.0  /  one note per fact, and the summaries write themselves

+

New in v1.6.0  /  "I cannot tell you why" beats a confident wrong answer

- v1.4.0 put a size limit on the "where we are now" file. This release fixes the reason it - grew. Until now, finishing one piece of work meant writing the same news into four or five - different files by hand: the to-do list, the decisions file, the test map, the design - notes. Nothing said which file owned which fact, so over time they drifted apart and - disagreed, and the one you happened to read first was the one you believed. On one real - project a single fixed bug had been written up seventeen times across three files. + Three fixes, all the same shape: a step that asked for a conclusion without saying what + evidence would support it, so one got invented.

- Forge can now keep each note in its own small file, named after the thing it is about, and - generate the summaries from them: the test map, the list of open work, the archive - of finished work. You never edit a generated summary, and Forge will not let you: it stops - the edit and points you at the note to change instead. Change the note, and every summary - that mentions it updates itself. + Forge runs some of its own machinery through Claude Code hooks, and one of them puts your + project state in front of it the moment a session starts. Some other tools that can run + Forge do not support that kind of hook at all. When the state never arrived, Forge used to + run four checks that all passed and then name a cause anyway. On one real project that + wrong cause was written into the project notes and believed for a day. It now separates + the two cases: something broken, which it names, or an intact setup that this tool is not + reading, in which case it says exactly that, points at the adapter that closes the gap, and + tells you what you lose until it is installed. If it cannot tell, it says it cannot tell.

- That also settles which answer is current. When a decision replaces an earlier one, the - earlier one is marked as replaced and drops out of every summary, while staying on file if - you ever want to look it up. No more reading a stale answer at the top of a long file and - a corrected one further down. + The "what should I do next" line used to pick whichever open job had the lowest number, + which on a long-running project meant it kept naming something deliberately parked. Jobs + now record why they are waiting: ready to start, waiting on a decision, or parked + at your request. Forge skips the last two, and says so plainly when everything left is + parked, rather than pointing you at work nobody meant to do.

- What you should notice: finishing a piece of work takes fewer steps and produces fewer - contradictions, the summaries are always current because nobody maintains them, and asking - "what is left to do" reads a short generated list instead of a very long history. Before - anything is pushed, Forge checks that every note still points at something real. + And before a release, "the automated checks passed" is no longer something Forge can take + from its own notes. It now looks up the result for the exact version of the code being + released and quotes it. A release went out over a failing check because that line had been + written a day earlier and nobody re-read it. More generally, a recorded check result older + than the code it covers now counts as missing rather than as proof.

- Already have a project on an older Forge? Nothing breaks and nothing is - treated as damaged. Your project keeps its current files until you say otherwise. Forge - offers the change once, explains what it costs, and takes "not now" or "no" for an answer. - Moving an existing project over is done as its own separate piece of work, never in the - middle of something else, and it never deletes your old files: it moves them aside. - See walking away and coming back. + Already have a project on an older Forge? Nothing breaks. Existing jobs + with no recorded reason are treated as ready to start, exactly as before, and Forge offers + to fill them in with you once. See walking away and coming back.

@@ -544,7 +542,7 @@

Which edition to install

A current Claude model
Claude 5 era. This is almost certainly you. - forge-workflow
v1.5.2 + forge-workflow
v1.6.0 /forge @@ -1289,7 +1287,7 @@

If you only remember four things

-
FORGE  /  GUIDE FOR NON-DEVELOPERS  /  REVISED FOR v1.5.2
+
FORGE  /  GUIDE FOR NON-DEVELOPERS  /  REVISED FOR v1.6.0
diff --git a/harness/README.md b/harness/README.md index 68fa198..75ca4eb 100644 --- a/harness/README.md +++ b/harness/README.md @@ -27,6 +27,34 @@ forge's enforcement keeps moving into `templates/` programs, and why the `docs/views/` guard being a `PreToolUse` hook rather than a `PostToolUse` one also happens to be the portable choice. +**Degradation is silent, and a project may assert otherwise in its own docs.** forge is +written for Claude Code, so a project can carry a sentence like "nothing under +`docs/views/` may be hand-edited, a PreToolUse hook denies it" that is simply false +under a harness which never loads that hook, with nothing anywhere to say so. The rule +that keeps a guarantee honest is to place it in a program the project runs rather than in +a hook the harness may not provide: `templates/lefthook.yml` runs +`forge-views.js check` as `06_views` under `pre-push`, which holds everywhere, and +the PreToolUse hook is the fast local deny where it is available. Word it that way round. + +## Telling which harness you are in + +A skill cannot ask the harness what it is, but every harness so far marks its own shell +environment. `/forge` Step 0 reads this table when the SessionStart hook did not fire +and the four mechanical checks all pass. + +| Harness | Environment marker | Adapter | Install to | +|---|---|---|---| +| Claude Code | `CLAUDE_CODE_ENTRYPOINT`, `CLAUDE_CODE_SESSION_ID` | none needed | n/a | +| omp (Oh My Pi) | `OMPCODE=1` | `harness/omp/forge-bridge.ts` | `~/.omp/agent/extensions/` | + +**`CLAUDECODE=1` does not identify Claude Code.** omp sets it deliberately, for tool +compatibility, in the same environment where it sets `OMPCODE=1`. Read it as necessary +and not sufficient, and match a harness's own marker first. A harness absent from this +table is the expected case rather than an error, and the honest report is that the +manifest is probably not being read, with no cause invented. + +Adding a harness is one row here and one directory beside this file. + ## omp (Oh My Pi) `omp/forge-bridge.ts`. Copy to `~/.omp/agent/extensions/forge-bridge.ts` and diff --git a/skills/forge-code/SKILL.md b/skills/forge-code/SKILL.md index 8cbcddb..cead310 100644 --- a/skills/forge-code/SKILL.md +++ b/skills/forge-code/SKILL.md @@ -86,7 +86,7 @@ A slice is done when all of these hold: - The failure paths log what the SRS observability section says they log, and you have read that output - The pre-push hook passes without `--no-verify` - `TODO.md` and `CONTINUE.md` reflect reality: this slice's entry has left the backlog for Completed, `CONTINUE.md` states the next action rather than the story of this one, and neither restates what `docs/DECISIONS.md` or `docs/traceability.md` already owns -- Under `records: backfilled` that closure is two edits, not five: set the task record's `status` and its `satisfies`, and add one decision record carrying the reasoning with `decided_in` naming the task. Then `node .forge/forge-records-lint.js`, `node .forge/forge-index.js build`, `node .forge/forge-views.js render`, and commit the records with the regenerated views. The views are generated, so writing the closure into them by hand is denied by a hook rather than merely discouraged +- Under `records: backfilled` that closure is two edits, not five: set the task record's `status` and its `satisfies`, and add one decision record carrying the reasoning with `decided_in` naming the task. Then `node .forge/forge-records-lint.js`, `node .forge/forge-index.js build`, `node .forge/forge-views.js render`, and commit the records with the regenerated views. The views are generated, so writing the closure into them by hand is caught rather than merely discouraged, by the mechanism `forge-standards` names Then merge to `main` with `--no-ff`, delete the branch, and push. @@ -122,7 +122,7 @@ Cadence, gates, and the release checklist are in `forge-standards`. The last str The execution order in this phase: 1. Run the `forge-design` polish pass over every surface this milestone touched. Clear the `finish` and `degrades` items or get a decision to slip each one, and record the run, the findings, and their dispositions in the polish log in `docs/DESIGN.md`. In the same pass, sweep every open UX debt row whose slip target is at or below this version: closed, or re-anchored forward with the user's agreement -2. CI green and the full suite green, coverage reported +2. CI conclusion read for HEAD's SHA and quoted with its run ids, per `forge-standards`, plus the full suite green and coverage reported. The release proposal cannot be made without that reading; a "CI green" line carried in from an earlier session is not it 3. Traceability matrix complete for every requirement claimed in this release, `UX-nnn` included 4. Regenerate API reference and architecture docs from CodeGraph queries 5. Confirm no screenshots are STALE or MISSING diff --git a/skills/forge-standards/SKILL.md b/skills/forge-standards/SKILL.md index 033a21b..6fc03a4 100644 --- a/skills/forge-standards/SKILL.md +++ b/skills/forge-standards/SKILL.md @@ -38,7 +38,7 @@ Under `backfilled`, `docs/records/` is the system of record and the monoliths ar |---|---| | `docs/records/{decisions,tasks,requirements,uxd}/.md` | Authoritative. One record, one file, one ID | | `docs/records/VOCABULARY.md` | Authoritative. Declares every legal ID prefix | -| `docs/views/**` | **Generated. Never hand-edited.** A PreToolUse hook denies the write | +| `docs/views/**` | **Generated. Never hand-edited.** The pre-push `06_views` check is the guarantee; a PreToolUse hook denies the write outright wherever the harness loads it | | `.forge/index.json` | Generated, gitignored, rebuildable. Authoritative about nothing | - **Write the fact once, in the record that owns it, then regenerate.** A slice close edits its task record and adds one decision record. It does not restate the closure in four files, because the views derive it. This is the whole point: on the project that motivated this, one closed defect was narrated seventeen times across three files. @@ -84,6 +84,8 @@ A backlog entry in `TODO.md` states why it is waiting, in one of three explicit - **Needs decision: ** - a named design or scope question blocks starting it. - **Deferred at your request** - the user said later or park it; do not raise it again until they do. +Under `records: backfilled` the disposition is the task record's `disposition` field, one of `ready`, `needs-decision`, `deferred`, and an absent field reads as ready. That field is what makes "leave Deferred items alone" something a tool can evaluate rather than a sentence: `.forge/index.json` skips anything not ready when it derives `next_action`, and the open-work view shows the column. While the backlog is still monolithic, the section and the prose in `TODO.md` carry it. + "Not for autonomous pickup" is not a disposition: it describes who acts, not why, and reads as deprioritized when it usually is not. Opening a Ready item never gets a second confirmation. The go-ahead already happened when it was captured as Ready; asking again at pickup time is the FLOW-mode confirmation prompt FLOW exists to remove, and doing it anyway is the same failure the disposition rule fixes, just moved one step later. @@ -218,9 +220,10 @@ Documentation ships in the same commit as the code it describes. - v1.0.0 only when every functional requirement is closed with a passing mapped test, and every `UX-nnn` requirement is verified by its named method. - Pre-1.0 means the interface may break; the README should say so. - Never tag with CI red. Never move an existing tag. +- **The CI gate names an artifact, and a recorded claim is not one.** Read it at proposal time with `gh run list --commit --json headSha,name,status,conclusion,databaseId`, and quote the run ids in the release proposal. Every run of a workflow the project names as a release gate must be `completed` with conclusion `success`; a run still `in_progress` has no conclusion yet and is not green. One commit routinely has several runs, since a tag push and a branch push each trigger their own, so "the latest run" is the wrong question. Which workflows gate a release is the project's to say, recorded in `docs/ENVIRONMENT.md`, defaulting to every workflow that runs on a push to the default branch. When `gh` is unavailable the gate is unmeasurable rather than met, and the proposal says so. - **A user override of a proposed version sweeps the version being skipped.** When the user picks a different version than the one proposed, every open row anchored to the skipped version is re-anchored or closed at that moment, while someone is looking. A skipped version never arrives, so nothing downstream will ever check it. -At each release: CI green, docs and API reference regenerated, no STALE or MISSING screenshots, traceability matrix complete for the requirements claimed, the `forge-design` polish pass run over every surface the milestone touched with its findings recorded in the polish log, no open `blocks` or `degrades` UX debt on those surfaces except what the user agreed to slip, no open UX debt whose recorded slip target is at or below this version (each such row is closed, or re-anchored forward with the user's agreement, before the tag), git-cliff run, annotated tag, GitHub Release published, the artifact the SRS distribution section names built, smoke-tested on a clean target, and attached to that release (or one line in the notes saying why none applies, plus the unsigned-binary warning whenever an exe or MSI is attached), and a plain-language MILESTONE paragraph at the top of the notes stating what this release proves the project can now do. +At each release: CI conclusion read for the exact commit being released and quoted in the proposal, docs and API reference regenerated, no STALE or MISSING screenshots, traceability matrix complete for the requirements claimed, the `forge-design` polish pass run over every surface the milestone touched with its findings recorded in the polish log, no open `blocks` or `degrades` UX debt on those surfaces except what the user agreed to slip, no open UX debt whose recorded slip target is at or below this version (each such row is closed, or re-anchored forward with the user's agreement, before the tag), git-cliff run, annotated tag, GitHub Release published, the artifact the SRS distribution section names built, smoke-tested on a clean target, and attached to that release (or one line in the notes saying why none applies, plus the unsigned-binary warning whenever an exe or MSI is attached), and a plain-language MILESTONE paragraph at the top of the notes stating what this release proves the project can now do. ## Explaining the process diff --git a/skills/forge/SKILL.md b/skills/forge/SKILL.md index f3eaabc..dfbc4df 100644 --- a/skills/forge/SKILL.md +++ b/skills/forge/SKILL.md @@ -13,23 +13,32 @@ You are the entry point for this project. The user should never need to remember Confirm the plugin's own machinery is working, silently. Report only when something is wrong. -**The check:** if `CONTINUE.md` exists in the project but no `=== FORGE: PROJECT STATE ===` block appeared in this session's context, the SessionStart hook did not fire. - -That single observation covers every cause: Node missing from PATH, `hooks/` changes not reloaded, a non-executable script on Unix-like systems, malformed `hooks.json`, or a wrong directory structure. Node being present does not prove the hook ran, so it is not a substitute check. +**The check:** if `CONTINUE.md` exists in the project but no `=== FORGE: PROJECT STATE ===` block appeared in this session's context, the SessionStart hook did not fire. Node being present does not prove the hook ran, so it is not a substitute check. If `CONTINUE.md` does not exist yet, the hook is correctly silent. Skip to Step 1. -**When the hook did not fire**, diagnose before continuing. Run these directly rather than through a script, since a broken Node runtime would prevent a diagnostic script from running at all: +**When the hook did not fire**, separate a broken machine from an intact one nobody is driving, before saying anything. Run these directly rather than through a script, since a broken Node runtime would prevent a diagnostic script from running at all: 1. `node --version` and confirm it resolves 2. Confirm `hooks/hooks.json` exists at the plugin root and parses as JSON 3. Confirm `scripts/session-start.js` exists at the plugin root 4. On Linux or macOS, confirm both scripts under `scripts/` are executable -Then report concisely, name the likely cause, and state the consequence: +**If one of these fails, that is the cause.** Report it concisely with its consequence, and continue: > The SessionStart hook is not firing, because Node is not on PATH. The workflow still works, but `CONTINUE.md` will no longer be loaded automatically at session start, so cold-start resumption depends on me remembering to read it rather than being guaranteed. Install Node, or say the word and I will convert the hooks to PowerShell. +**If all four pass, nothing is broken and the likely answer is that the hook manifest is never read.** `hooks/hooks.json` is a Claude Code file. Another agent harness can load forge's skills, commands, and MCP servers perfectly while providing no SessionStart or Stop event at all, and it does so silently. Establish the harness rather than naming a fault, using the marker table in `harness/README.md` at the plugin root: + +- **A recognised marker with a matching `harness//` directory.** Check whether that adapter is installed at the extension path the table names. If it is not, that is the finding, and the report is one line naming the file to copy. +- **Anything else**, including a marker you do not recognise and no marker at all. Say so, and report the consequence anyway. It does not depend on knowing which harness this is. + +`CLAUDECODE=1` does not establish Claude Code. At least one harness sets it deliberately for tool compatibility while setting its own marker alongside, so read it as necessary and not sufficient, and look for a harness's own marker first. + +Never name a cause the diagnostics did not establish. An invented cause is worse than no cause, because it gets written into `CONTINUE.md` and the next session starts from it: + +> No project state block was injected this session and all four mechanical checks pass, so forge's machinery is intact and this harness is probably not reading `hooks/hooks.json`. I cannot tell from inside the session which harness this is. If it is not Claude Code, check `harness/` in the plugin for an adapter and whether it is installed. Until then the SessionStart and Stop guarantees are absent here: cold-start resumption depends on me remembering to read `CONTINUE.md` rather than being guaranteed, and the Stop-time resumability check does not run. + Then continue with Step 1. Degraded hooks are a real regression but not a blocker. Report this once per session. ## Step 1: Establish actual state @@ -55,7 +64,7 @@ Observe: - `git tag --sort=-v:refname` - Whether a remote exists (`git remote -v`) - The test suite result, if a test command exists in `CLAUDE.md` -- Latest CI conclusion, if `gh` is available +- The CI conclusion for HEAD, if `gh` is available: `gh run list --commit --json headSha,name,status,conclusion,databaseId`. The newest run for the repository is a different question and often a different commit ## Step 2: Reconcile before deciding @@ -72,6 +81,8 @@ Commit identity comes from git, never from the record. Older projects carry a `L **If the record and reality disagree, stop.** Report the discrepancy and ask how to reconcile. Do not trust either side, and do not build on an unexplained inconsistency. This override applies in FLOW mode too. +A recorded gate result whose subject is a commit is evidence only for the commit it was measured against. CI conclusion, suite result, and coverage are all of this kind: if commits have landed since the line was written, it is absent rather than stale, and re-measuring is the next action rather than a courtesy. A gate whose subject is a surface does not expire this way, because a push that does not touch that surface does not invalidate it; the polish pass is the one that matters here. Quoting a recorded "CI green" over a commit it never saw is the failure this rule exists to catch. + A recorded gate, backlog item, or slip target naming a version already published is a discrepancy of this kind. The version moved and the item did not, so the item is no longer gating anything: it reads as pending work behind a tag that has already shipped, or behind one the user chose to skip. Re-anchor it to the next real target, or close it, before continuing. Test counts and commits are not the only things that go stale; version anchors do, and they do it silently. If `CONTINUE.md` is missing, badly stale, or contradicted by the tree, reconstructing accurate state IS the next action. Rebuild it from git history, the tests, and the code, report what you found, rewrite the file, then continue. @@ -91,6 +102,7 @@ Check for these, cheaply, from what you already read: | Surface verification tooling | `docs/ENVIRONMENT.md` records none, and a `UX-` requirement needs it | `forge-env` Step 11a, for the tiers in play | | External design tools | a GUI tier, and `docs/DECISIONS.md` records no choice about them | Offer the list from `forge-design` "External design tools" once, take "none" as an answer, and record it. Nothing installs without the user asking, and nothing is uploaded to a hosted service without a gate | | Structured records | no `docs/records/` directory | `forge-standards` "Structured records", as its own slice. Migration rewrites every artifact the lifecycle reads, so it never runs mid-slice | +| Task dispositions | `docs/records/` exists and no task record carries `disposition` | One pass over the open tasks with the user: ready, needs-decision, or deferred. Until then every open task reads as ready, so row 7 and `next_action` both pick the lowest-numbered one whatever it is | Raise the whole set once, in one message, ordered by what it would change about the work in front of you. For each: what it is, what backfilling costs now, and what shipping without it means. Then offer three answers, and say which you recommend: diff --git a/templates/TODO.md b/templates/TODO.md index 85fb72b..8ded7e9 100644 --- a/templates/TODO.md +++ b/templates/TODO.md @@ -10,7 +10,9 @@ Task IDs are permanent. Never renumber. Requirement IDs refer to docs/SRS.md. ## Needed - +", or "Deferred at your request".> - **T-001** Description. Closes: FR-001, FR-002. - **T-002** Description. Closes: FR-003. diff --git a/templates/forge-index.js b/templates/forge-index.js index 3632f4b..3ec6c11 100644 --- a/templates/forge-index.js +++ b/templates/forge-index.js @@ -129,7 +129,15 @@ function build(opts) { .filter((r) => r.data.type === "requirement" && lib.isLive(r) && !satisfied.has(r.data.id)) .map((r) => r.data.id); - const nextAction = openTasks.find((r) => r.data.status === "in-progress") || openTasks[0] || null; + // The in-progress task, else the first task that is actually pickable. Taking + // openTasks[0] was wrong in a way that trained people to distrust the file: a + // deliberately parked post-1.0 item with a low number outranked every ready + // slice, forever. A null next_action is a real answer and means every open + // task is parked or waiting on a decision. + const nextAction = + openTasks.find((r) => r.data.status === "in-progress") || + openTasks.find((r) => lib.dispositionOf(r) === "ready") || + null; return { generated: new Date().toISOString().replace(/\.[0-9]{3}Z$/, "Z"), @@ -139,7 +147,7 @@ function build(opts) { mode: ladder.mode, capabilities: ladder.capabilities, open: { - tasks: openTasks.map((r) => r.data.id), + tasks: openTasks.map((r) => ({ id: r.data.id, disposition: lib.dispositionOf(r) })), uxd: openUxd.map((r) => ({ id: r.data.id, severity: r.data.severity || null })), }, counts, diff --git a/templates/forge-records-lib.js b/templates/forge-records-lib.js index 1a509aa..7c08fe1 100644 --- a/templates/forge-records-lib.js +++ b/templates/forge-records-lib.js @@ -36,6 +36,7 @@ const FIELDS = { superseded_by: "idornull", decided_in: "idornull", severity: "enum", + disposition: "enum", }; const REQUIRED = ["id", "type", "status", "date", "title"]; @@ -51,6 +52,14 @@ const STATUS_BY_TYPE = { const SEVERITIES = ["blocks", "degrades", "finish"]; +// Why an open task is waiting, which is a different question from whether it is +// open. The three are forge-standards' backlog dispositions; without the field +// they all collapse to `open`, and "open the next Ready slice, leave Deferred +// items alone" becomes an instruction nothing can evaluate. +// Absent means `ready`. A project that predates the field would otherwise have +// no next action at all, which is a worse answer than a stale one. +const DISPOSITIONS = ["ready", "needs-decision", "deferred"]; + // Which statuses count as still-open work, per type. Used by the index and the // open-work view so the two cannot disagree about what "open" means. const OPEN_STATUS = { @@ -283,6 +292,19 @@ function validateRecord(rec, vocabulary) { problems.push({ file, field: "severity", severity: "error", message: "a uxd record requires a severity" }); } + if (d.disposition !== undefined && d.disposition !== null) { + if (d.type !== "task") { + problems.push({ file, field: "disposition", severity: "error", message: "disposition belongs to task records only" }); + } else if (!DISPOSITIONS.includes(d.disposition)) { + problems.push({ + file, + field: "disposition", + severity: "error", + message: "`" + d.disposition + "` is not one of " + DISPOSITIONS.join(", "), + }); + } + } + for (const key of ["closes", "satisfies"]) { const list = d[key]; if (list === undefined) continue; @@ -396,6 +418,16 @@ function isLive(rec) { return by === undefined || by === null || by === ""; } +/* + * The disposition of an open task, defaulted. Read this rather than the raw + * field so the default lives in one place. + */ +function dispositionOf(rec) { + if (rec.data.type !== "task") return null; + const value = rec.data.disposition; + return DISPOSITIONS.includes(value) ? value : "ready"; +} + function isOpen(rec) { const statuses = OPEN_STATUS[rec.data.type] || []; return isLive(rec) && statuses.includes(rec.data.status); @@ -425,6 +457,7 @@ module.exports = { TYPES, STATUS_BY_TYPE, SEVERITIES, + DISPOSITIONS, OPEN_STATUS, TYPE_DIRS, ID_SHAPE, @@ -436,5 +469,6 @@ module.exports = { readRecords, isLive, isOpen, + dispositionOf, recordsHash, }; diff --git a/templates/forge-records-migrate.js b/templates/forge-records-migrate.js index a6d3d7f..ea9743d 100644 --- a/templates/forge-records-migrate.js +++ b/templates/forge-records-migrate.js @@ -205,9 +205,20 @@ if (todoText === null) { const seenTask = new Set(); const seenUxd = new Set(); + // The section a bullet sits under is the only place TODO.md records WHY a + // task is waiting, and the split below flattens headings and bullets into + // one stream, so carry the heading forward. Losing it is how a bullet under + // "## Blocked" became indistinguishable from the next ready slice. + let section = ""; + for (const block of todoText.split(/\n(?=[-*]\s|#{2,4}\s)/)) { const head = block.split(/\r?\n/)[0] || ""; + const headText = head.trim(); + if (headText.startsWith("#")) { + section = headText.replace(/^#+/, "").trim().toLowerCase(); + } + const uxd = /\b(UXD-[0-9]+[a-z]?)\b/.exec(head); if (uxd && !seenUxd.has(uxd[1])) { seenUxd.add(uxd[1]); @@ -243,10 +254,20 @@ if (todoText === null) { if (/\bclosed\b|\bdone\b|\bshipped\b|\[x\]/.test(lower)) status = "closed"; else if (/\babandoned\b|\bdropped\b|\bwon't do\b|\bwont do\b/.test(lower)) status = "abandoned"; else if (/\bin progress\b|\bin-progress\b|\bstarted\b/.test(lower)) status = "in-progress"; + // Disposition is asserted only where the source actually says something. + // Everything else is left absent, which the index reads as ready, so the + // silent default and an explicit one cannot disagree. + let disposition = null; + if (section.includes("blocked") || lower.includes("needs decision")) { + disposition = "needs-decision"; + } else if (lower.includes("deferred") || lower.includes("parked") || lower.includes("post-1.0")) { + disposition = "deferred"; + } emit("tasks", task[1], { id: task[1], type: "task", status, + ...(disposition ? { disposition } : {}), date: today, title: escapeTitle(head), closes: [], diff --git a/templates/forge-views.js b/templates/forge-views.js index c3373d4..0fd2c1e 100644 --- a/templates/forge-views.js +++ b/templates/forge-views.js @@ -147,12 +147,13 @@ function renderOpenWork(records) { out.push("## Tasks (" + tasks.length + ")"); out.push(""); if (tasks.length) { - out.push("| Task | Status | Title | Satisfies |"); - out.push("|---|---|---|---|"); + out.push("| Task | Status | Disposition | Title | Satisfies |"); + out.push("|---|---|---|---|---|"); for (const t of tasks) { out.push( "| " + cell(t.data.id) + " | " + cell(t.data.status) + + " | " + cell(lib.dispositionOf(t)) + " | " + cell(t.data.title) + " | " + cell((t.data.satisfies || []).join(", ")) + " |" diff --git a/tests/harness-and-gates.test.js b/tests/harness-and-gates.test.js new file mode 100644 index 0000000..5fda090 --- /dev/null +++ b/tests/harness-and-gates.test.js @@ -0,0 +1,100 @@ +"use strict"; + +/* + * Two contracts that failed in the field, both for the same reason: a step that + * asked for a conclusion without naming the evidence, so the model supplied one. + * + * - Step 0's self-check. Its four mechanical checks all pass when the real + * cause is a harness that never reads hooks.json, and it then asked for "the + * likely cause" with a worked example naming Node. Two different invented + * causes reached a real project's CONTINUE.md that way. + * - The release CI gate. Every other gate in the list implies an act of + * measurement; CI runs elsewhere, so the record was the cheapest place to + * look, and a release was cut on a day-old "CI green" line over a red run. + * + * These are prompt files, so what is testable is structural: the honest branch + * exists, the evidence is named, and the knowledge has exactly one owner. + */ + +const test = require("node:test"); +const assert = require("node:assert"); +const fs = require("fs"); +const path = require("path"); + +const { REPO_ROOT } = require("./helpers/sandbox.js"); + +function skill(name) { + return fs.readFileSync(path.join(REPO_ROOT, "skills", name, "SKILL.md"), "utf8"); +} + +const FORGE = skill("forge"); +const CODE = skill("forge-code"); +const STANDARDS = skill("forge-standards"); +const HARNESS = fs.readFileSync(path.join(REPO_ROOT, "harness", "README.md"), "utf8"); +const LEFTHOOK = fs.readFileSync(path.join(REPO_ROOT, "templates", "lefthook.yml"), "utf8"); + +const STEP0 = FORGE.slice(FORGE.indexOf("## Step 0"), FORGE.indexOf("## Step 1")); + +/* ------------------------------- Step 0 -------------------------------- */ + +test("Step 0 asks the harness question before it names a fault", () => { + assert.match(STEP0, /If all four pass/); + assert.match(STEP0, /harness/i); + assert.match(STEP0, /harness\/README\.md/, "the marker table has one owner and Step 0 points at it"); +}); + +test("Step 0 keeps a fail-safe branch that does not depend on recognising the harness", () => { + assert.match(STEP0, /Never name a cause the diagnostics did not establish/); + assert.match(STEP0, /cannot tell/i, "not knowing the harness must still produce a complete report"); +}); + +test("Step 0 does not treat CLAUDECODE as proof of Claude Code", () => { + // omp sets CLAUDECODE=1 alongside OMPCODE=1 for tool compatibility, so keying + // on it would return a confident wrong answer in exactly the situation being + // diagnosed. + assert.match(STEP0, /CLAUDECODE=1.{0,4} does not establish Claude Code/); +}); + +test("the harness README carries the marker table, and only it does", () => { + assert.match(HARNESS, /## Telling which harness you are in/); + assert.match(HARNESS, /OMPCODE/); + assert.match(HARNESS, /~\/\.omp\/agent\/extensions\//); + assert.ok( + !STANDARDS.includes("OMPCODE") && !CODE.includes("OMPCODE"), + "one owner: the marker vocabulary must not be restated in a skill" + ); +}); + +test("the views guarantee names the portable check rather than a hook the harness may not load", () => { + assert.match(STANDARDS, /06_views/, "the guarantee is the pre-push program, not the PreToolUse hook"); + const prePush = LEFTHOOK.slice(LEFTHOOK.indexOf("pre-push:"), LEFTHOOK.indexOf("pre-commit:")); + assert.match(prePush, /06_views/, "and that check really is a pre-push one"); + assert.ok( + !CODE.includes("denied by a hook"), + "forge-code points at the mechanism forge-standards owns rather than restating it" + ); +}); + +/* ------------------------------ the CI gate ----------------------------- */ + +test("every step that reads CI anchors it to the commit rather than to the newest run", () => { + // The command is stated in the two places that actually run it, and + // forge-code points at forge-standards rather than restating it a third time. + assert.match(FORGE, /--commit/, "Step 1 gathers the conclusion for HEAD"); + assert.match(STANDARDS, /--commit/, "the release gate owns the wording"); + assert.match(CODE, /HEAD/, "the execution order names the reading it depends on"); + assert.ok(!/Latest CI conclusion/.test(FORGE), "the newest run for the repo is a different question"); +}); + +test("the release gate names the artifact and refuses a quoted claim", () => { + assert.match(STANDARDS, /gh run list --commit/); + assert.match(STANDARDS, /run ids/); + assert.match(STANDARDS, /in_progress/, "a pending run has no conclusion and is not green"); + assert.match(CODE, /cannot be made without that reading/); +}); + +test("a commit-subject gate result expires against later commits, and a surface-subject one does not", () => { + const reconcile = FORGE.slice(FORGE.indexOf("## Step 2:"), FORGE.indexOf("## Step 2a")); + assert.match(reconcile, /absent rather than stale/); + assert.match(reconcile, /polish pass/, "the boundary is stated, or the rule over-fires on every push"); +}); diff --git a/tests/records.test.js b/tests/records.test.js index cdf7b77..c61555f 100644 --- a/tests/records.test.js +++ b/tests/records.test.js @@ -316,7 +316,7 @@ test("the index carries live state and never enumerates history", () => { assert.equal(index.gate, "PASSED"); assert.equal(index.mode, "FLOW"); assert.equal(index.capabilities.records, "backfilled"); - assert.deepEqual(index.open.tasks, ["T-1"]); + assert.deepEqual(index.open.tasks, [{ id: "T-1", disposition: "ready" }]); assert.equal(index.counts.task, 39); assert.deepEqual(index.requirement_gaps, ["FR-GW-002"]); @@ -327,6 +327,55 @@ test("the index carries live state and never enumerates history", () => { bp.removeDisposable(root); }); +test("next_action skips a parked task and prefers the one in progress", () => { + const root = makeProject(); + // The reported failure: a low-numbered post-1.0 item outranking every ready + // slice, forever, because the only field the index had was `status`. + task(root, "T-059", { disposition: "deferred" }); + task(root, "T-101"); + assert.equal(run(INDEX, ["build"], root).status, 0); + let index = JSON.parse(fs.readFileSync(path.join(root, ".forge", "index.json"), "utf8")); + assert.equal(index.next_action, "T-101"); + + task(root, "T-200", { status: "in-progress" }); + assert.equal(run(INDEX, ["build"], root).status, 0); + index = JSON.parse(fs.readFileSync(path.join(root, ".forge", "index.json"), "utf8")); + assert.equal(index.next_action, "T-200", "an in-progress task outranks a ready one"); + bp.removeDisposable(root); +}); + +test("next_action is null when every open task is parked or waiting on a decision", () => { + const root = makeProject(); + task(root, "T-1", { disposition: "deferred" }); + task(root, "T-2", { disposition: "needs-decision" }); + assert.equal(run(INDEX, ["build"], root).status, 0); + const index = JSON.parse(fs.readFileSync(path.join(root, ".forge", "index.json"), "utf8")); + assert.equal(index.next_action, null, "null is a real answer, not a missing one"); + assert.equal(index.open.tasks.length, 2, "parked work is still open work"); + bp.removeDisposable(root); +}); + +test("an absent disposition reads as ready, so a project predating the field still has a next action", () => { + const root = makeProject(); + task(root, "T-1"); + assert.equal(run(INDEX, ["build"], root).status, 0); + const index = JSON.parse(fs.readFileSync(path.join(root, ".forge", "index.json"), "utf8")); + assert.equal(index.next_action, "T-1"); + assert.equal(index.open.tasks[0].disposition, "ready"); + bp.removeDisposable(root); +}); + +test("the linter rejects an unknown disposition and one on a non-task record", () => { + const root = makeProject(); + task(root, "T-1", { disposition: "someday" }); + requirement(root, "FR-GW-001", { disposition: "ready" }); + const res = run(LINT, [], root); + assert.equal(res.status, 1, res.stdout); + assert.match(res.stdout, /is not one of ready, needs-decision, deferred/); + assert.match(res.stdout, /disposition belongs to task records only/); + bp.removeDisposable(root); +}); + test("index check reports staleness and ignores its own timestamp", () => { const root = makeProject(); task(root, "T-1"); @@ -499,6 +548,33 @@ test("the migration reports orphans and moves originals aside rather than deleti bp.removeDisposable(root); }); +test("the migration keeps the TODO section a task sat under, so Blocked survives as a disposition", () => { + const root = makeProject(); + fs.rmSync(path.join(root, "docs", "records"), { recursive: true, force: true }); + fs.writeFileSync(path.join(root, "TODO.md"), [ + "## Needed", + "", + "- T-1 build the thing", + "- T-2 the post-1.0 idea, deferred at your request", + "", + "## Blocked", + "", + "- T-3 waiting on the vendor", + "", + ].join("\n")); + assert.equal(run(MIGRATE, ["--force-dirty"], root).status, 0); + + const read = (id) => fs.readFileSync(path.join(root, "docs", "records", "tasks", id + ".md"), "utf8"); + assert.ok(!/disposition:/.test(read("T-1")), "an unremarkable backlog item asserts nothing and reads as ready"); + assert.match(read("T-2"), /disposition: deferred/); + assert.match(read("T-3"), /disposition: needs-decision/, "the heading is the only place the reason was written"); + + assert.equal(run(INDEX, ["build"], root).status, 0); + const index = JSON.parse(fs.readFileSync(path.join(root, ".forge", "index.json"), "utf8")); + assert.equal(index.next_action, "T-1"); + bp.removeDisposable(root); +}); + test("--dry-run writes nothing", () => { const root = makeProject(); fs.rmSync(path.join(root, "docs", "records"), { recursive: true, force: true }); From 0d88b6172cbcf5066dd5d073440d8e6a655d66c2 Mon Sep 17 00:00:00 2001 From: Dailen Gunter Date: Sat, 5 Sep 2026 14:07:06 -0400 Subject: [PATCH 2/2] fix: resolve the plugin root from the registry, and widen the one-owner guard Step 0's harness branch reads harness/README.md at the plugin root, but CLAUDE_PLUGIN_ROOT is set for hook processes rather than for the session running the skill, and listing the cache directory picks a version at random: on the machine forge-bridge.ts was written for, the cache held five and the registry resolved a build seven versions old. Step 0 now names ~/.claude/plugins/installed_plugins.json, so the one branch that fires when hooks are not working cannot dead-end on a path it was never taught to find. The marker vocabulary test guarded two skills by name, which looks like a one-owner guard without being one. It now sweeps every skill and template. Step 2a's disposition row says to expect a stale records_hash on the first index check after the field set grows, rather than reading it as a discrepancy. Co-Authored-By: Claude Opus 5 (1M context) --- skills/forge/SKILL.md | 4 +++- tests/harness-and-gates.test.js | 28 ++++++++++++++++++++++------ 2 files changed, 25 insertions(+), 7 deletions(-) diff --git a/skills/forge/SKILL.md b/skills/forge/SKILL.md index dfbc4df..784e49f 100644 --- a/skills/forge/SKILL.md +++ b/skills/forge/SKILL.md @@ -17,6 +17,8 @@ Confirm the plugin's own machinery is working, silently. Report only when someth If `CONTINUE.md` does not exist yet, the hook is correctly silent. Skip to Step 1. +Every check below reads the plugin root, and `${CLAUDE_PLUGIN_ROOT}` is set for hook processes rather than for this session, so resolve it first: `~/.claude/plugins/installed_plugins.json` names the live install path. Read the registry rather than listing `~/.claude/plugins/cache/`, which holds every version ever installed and will hand you the wrong one. + **When the hook did not fire**, separate a broken machine from an intact one nobody is driving, before saying anything. Run these directly rather than through a script, since a broken Node runtime would prevent a diagnostic script from running at all: 1. `node --version` and confirm it resolves @@ -102,7 +104,7 @@ Check for these, cheaply, from what you already read: | Surface verification tooling | `docs/ENVIRONMENT.md` records none, and a `UX-` requirement needs it | `forge-env` Step 11a, for the tiers in play | | External design tools | a GUI tier, and `docs/DECISIONS.md` records no choice about them | Offer the list from `forge-design` "External design tools" once, take "none" as an answer, and record it. Nothing installs without the user asking, and nothing is uploaded to a hosted service without a gate | | Structured records | no `docs/records/` directory | `forge-standards` "Structured records", as its own slice. Migration rewrites every artifact the lifecycle reads, so it never runs mid-slice | -| Task dispositions | `docs/records/` exists and no task record carries `disposition` | One pass over the open tasks with the user: ready, needs-decision, or deferred. Until then every open task reads as ready, so row 7 and `next_action` both pick the lowest-numbered one whatever it is | +| Task dispositions | `docs/records/` exists and no task record carries `disposition` | One pass over the open tasks with the user: ready, needs-decision, or deferred. Until then every open task reads as ready, so row 7 and `next_action` both pick the lowest-numbered one whatever it is. `forge-index.js check` reports `records_hash` stale on the first run after the field set grows; rebuild rather than reading it as a discrepancy | Raise the whole set once, in one message, ordered by what it would change about the work in front of you. For each: what it is, what backfilling costs now, and what shipping without it means. Then offer three answers, and say which you recommend: diff --git a/tests/harness-and-gates.test.js b/tests/harness-and-gates.test.js index 5fda090..7787220 100644 --- a/tests/harness-and-gates.test.js +++ b/tests/harness-and-gates.test.js @@ -43,6 +43,13 @@ test("Step 0 asks the harness question before it names a fault", () => { assert.match(STEP0, /harness\/README\.md/, "the marker table has one owner and Step 0 points at it"); }); +test("Step 0 says how to resolve the plugin root it keeps reading", () => { + // CLAUDE_PLUGIN_ROOT is set for hook processes, not for the session running + // this skill, and the cache directory holds every version ever installed. + assert.ok(STEP0.includes("installed_plugins.json"), "the registry, which says which version is live"); + assert.match(STEP0, /rather than listing/); +}); + test("Step 0 keeps a fail-safe branch that does not depend on recognising the harness", () => { assert.match(STEP0, /Never name a cause the diagnostics did not establish/); assert.match(STEP0, /cannot tell/i, "not knowing the harness must still produce a complete report"); @@ -55,14 +62,23 @@ test("Step 0 does not treat CLAUDECODE as proof of Claude Code", () => { assert.match(STEP0, /CLAUDECODE=1.{0,4} does not establish Claude Code/); }); -test("the harness README carries the marker table, and only it does", () => { +test("the harness README carries the marker table, and is the only file that does", () => { assert.match(HARNESS, /## Telling which harness you are in/); assert.match(HARNESS, /OMPCODE/); - assert.match(HARNESS, /~\/\.omp\/agent\/extensions\//); - assert.ok( - !STANDARDS.includes("OMPCODE") && !CODE.includes("OMPCODE"), - "one owner: the marker vocabulary must not be restated in a skill" - ); + assert.ok(HARNESS.includes("~/.omp/agent/extensions/"), "the table names where the adapter goes"); + + // One owner. A marker restated in a skill or a template is a second copy to + // keep current, and the one that goes stale is the one somebody reads. + const owned = []; + for (const dir of ["skills", "templates"]) { + for (const name of fs.readdirSync(path.join(REPO_ROOT, dir))) { + const file = dir === "skills" ? path.join(dir, name, "SKILL.md") : path.join(dir, name); + const full = path.join(REPO_ROOT, file); + if (!fs.existsSync(full) || fs.statSync(full).isDirectory()) continue; + if (/OMPCODE|CLAUDE_CODE_ENTRYPOINT/.test(fs.readFileSync(full, "utf8"))) owned.push(file); + } + } + assert.deepEqual(owned, [], "the marker vocabulary belongs to harness/README.md alone"); }); test("the views guarantee names the portable check rather than a hook the harness may not load", () => {