diff --git a/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md new file mode 100644 index 0000000..fbee4c7 --- /dev/null +++ b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md @@ -0,0 +1,71 @@ + +# Changes Log: Claracle Relaunch Follow-Up Execution + +## Related Plans + +* .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md +* .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md + +## Implementation Date + +2026-08-02 + +## Summary + +Published the PR review correction, reconciled production GA4/GSC observations, refreshed acceptance evidence, created one owner-action register, and produced implementation-ready rollout and cost plans. External account actions and human approvals remain owner-gated. + +## Added + +* .copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md +* .copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md +* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md +* .copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md +* .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md +* .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md +* .copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md +* .copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md +* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md +* docs/review/data-observatory-relaunch/owner-action-register.md + +## Modified + +* hugo.toml +* docs/growth/ga4-gsc-baseline-2026-07-29.md +* docs/prds/claracle-data-observatory-relaunch.md +* docs/review/data-observatory-relaunch/README.md +* docs/review/data-observatory-relaunch/security-review.md +* docs/review/data-observatory-relaunch/status-of-record.md +* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md +* .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md +* .copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md + +## Completed Work + +* Pushed correction commit `8fddceb` and resolved both PR #647 review threads +* Confirmed production GA configuration on the main site and standalone embed without relying on checked-in identifiers +* Recorded owner-confirmed GA4 stream operation, GSC verification, root sitemap submission, and GA4-to-GSC product link; FR-035 is complete +* Reconciled SEC-01 and SEC-04 with current sanitization and lifecycle tests +* Implemented SEC-02 with a no-referrer official iframe snippet and frame-local explicit-consent tests +* Implemented SEC-03 exact CSV, metadata, nested-object, and source-path allowlists +* Documented the SEC-05 defense-in-depth recommendation and limitations without recording acceptance +* Classified #622 as non-blocking polish and #626 as independent hardening +* Added exact owner actions for analytics, security, accessibility, protected Podcaster, visual, and sponsor evidence +* Planned report-only cost attribution, one dynamic-topic canary, and repository-page activation with rollback + +## Validation + +* Full pytest: 1,389 passed, 19 skipped, 34 subtests passed +* Focused acceptance suite: 45 passed, 4 skipped +* Ruff lint and format: passed +* Data-page, public dataset, and trend-export checks: passed +* PR #647 at `8fddceb`: 13 successful checks, including Production site +* Editor diagnostics and `git diff --check`: passed +* Security closure focused suite: 217 passed +* Rendered embed/export suite with Hugo 0.161.1: 10 passed +* Public dataset freshness, Hugo production build, internal links, Ruff, and diff whitespace: passed +* Local Playwright analytics execution was attempted but the host lacks Chromium runtime libraries; CI browser execution remains required +* Final local suite after Squad and Google evidence updates: 1,392 passed, 19 skipped, 34 subtests passed + +## Known Inherited Discrepancy + +`discover_topic_candidates.py --check` reports the candidate registry stale at the inherited commit. Temporary regeneration preserves 2,173 total candidates and the same five eligible candidates while rotating four sanitized keys. The owning publish workflow should refresh this generated state. diff --git a/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md new file mode 100644 index 0000000..a60cf64 --- /dev/null +++ b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md @@ -0,0 +1,51 @@ + +# Changes Log: Claracle Relaunch Readiness Reconciliation + +## Related Plan + +.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md + +## Implementation Date + +2026-08-02 + +## Summary + +Reconciled the Claracle relaunch plans, PRD, BRD, and issue evidence into one status of record. One review iteration corrected stale issue-state claims for closed issues #599 and #644 while preserving the outstanding GA4/GSC launch gate. + +## Changes by Category + +### Added + +* .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md +* .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md +* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md +* .copilot-tracking/research/2026-08-01/restore-consistency-640-research.md +* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md +* docs/review/data-observatory-relaunch/status-of-record.md + +### Modified + +* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md +* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md +* docs/brds/claracle-data-observatory-relaunch-brd.md +* docs/prds/claracle-data-observatory-relaunch.md + +### Removed + +* None + +## Review Iteration + +PR #647 review found that #599 and #644 were described as open after both had closed as completed on 2026-08-01. The research and status-of-record artifacts now show the final dispositions. FR-035 remains partial because #599 closed with GSC, platform-receipt, and baseline actions still outstanding; later production verification confirmed secret-backed GA configuration is present. + +## Validation + +* Focused documentation tests: 10 passed +* PR #647 status checks: 13 passed, 0 failed +* Editor diagnostics: no errors in the corrected files +* Git whitespace validation: passed + +## Release Summary + +The repository now has one evidence-backed relaunch readiness view. Delivered remediation work is distinguished from pending external acceptance gates, and closed issue state is no longer used as evidence that GA4/GSC acceptance work shipped. \ No newline at end of file diff --git a/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md b/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md new file mode 100644 index 0000000..64c23ea --- /dev/null +++ b/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md @@ -0,0 +1,35 @@ + +# Implementation Details: Claracle Gated Rollouts and Cost Measurement + +## Cost Experiment Contract + +Every variant starts from a clean destination and the same hydrated source state. The machine-readable record must include main SHA, publish SHA, workload variant, source counts by page class, Hugo and Pagefind versions and raw durations, rendered and indexed counts, output bytes, runner identity, exit state, and Actions URL. + +Use cumulative variants so marginal cost can be calculated without changing generator logic: + +1. Observatory generated classes excluded +2. Five checked-in topic hubs included +3. Three generated data pages included +4. 263 checked-in repository pages included +5. Optional approved dynamic canary included + +Do not derive a blocking budget from one run. Retain at least three comparable runs and calculate median plus nearest-rank p95 separately for Hugo and Pagefind. + +## Dynamic Preview Contract + +A preview must evaluate the same eligible-candidate and assignment path as write mode while performing no filesystem mutation. Its structured output must identify candidate slug, title, evidence weeks, supporting sources, proposed hub path, proposed weekly assignments, registry effect, and skip reason. Tests must compare preview output with the corresponding isolated write transaction. + +The first canary uses explicit deferrals in `ignore_topics`; no threshold change is permitted. Threshold-based canaries are unsafe because repository generation can classify existing pages as obsolete. + +## Repository Activation Contract + +The isolated enabled preflight must preserve the existing recurrence threshold and hydrated publish state. A reviewer must disposition every obsolete or expired path. No removal is accepted from mere crawl absence. The second generation must be byte-stable. + +Rollback has two parts: + +1. Disable the production flag to stop future mutation. +2. Revert the generated-state transaction to undo pages, ledgers, registries, assignments, and logs already committed. + +## Approval Contract + +Hermes approves security and lifecycle policy. URL approves workflows, secret scope, and retained artifacts. jmservera separately approves the dynamic-topic canary and repository-page activation. Each approval identifies the exact revision, evidence, conditions, rollback owner, and date. diff --git a/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md b/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md new file mode 100644 index 0000000..5212963 --- /dev/null +++ b/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md @@ -0,0 +1,87 @@ + +# Implementation Details: Claracle Relaunch Follow-Up Execution + +## Phase 1: Publish Review Corrections + +Commit and push the reviewed #599/#644 state corrections, then resolve the two PR #647 threads only after the changed diff is visible remotely. + +Success: commit `8fddceb` is on the PR branch and both threads are resolved. + +## Phase 2: Reconcile GA4/GSC Evidence + +Keep both checked-in Hugo defaults empty. Record only presence-level production observations and secret names. Never record the GA identifier or GSC token. + +Owner handoff completed on 2026-08-02: + +1. The deployed ID maps to the intended Claracle stream. +2. The GSC property is verified without requiring the optional HTML-tag secret path. +3. `https://claracle.com/sitemap.xml` was submitted and the GA4 stream was linked to GSC. +4. GA4 is operational, and a GSC performance export was supplied. + +Remaining evidence work: + +1. Transcribe the supplied GSC performance values once the attachment is available as a readable file. +2. Retain denied and granted production consent observations. +3. Confirm GSC processing and review indexed and excluded URL counts. + +Success: FR-035 connection and submission are complete; baseline transcription and NFR-008 production consent evidence remain open. + +## Phase 3: Refresh Acceptance Evidence + +Update the security record to acknowledge implemented candidate-title sanitization and lifecycle fixtures while retaining Hermes disposition requirements. Add owner-ready evidence records for manual accessibility, visual review, protected Podcaster execution, and sponsor decisions. Do not mark a human gate complete from automated tests. + +Protected Podcaster sequence: + +1. Confirm downstream idempotency or authorize a specific eligible week. +2. Define required reviewers and branch policy for a real-generation environment. +3. Bind the real generation job to that environment through a separately reviewed workflow change. +4. Run once and retain the approver, week, manifest run, article digest, Actions URL, downstream job ID, and final conclusion. + +Success: the acceptance index identifies current automated evidence and exact remaining owner actions. + +Repository-executable security closure added on 2026-08-02: + +1. SEC-02: generated iframe snippets use `referrerpolicy="no-referrer"`; analytics remains disabled + until explicit consent inside the Claracle frame. Tests cover rendered markup, default-off wiring, + and the existing browser consent behavior. Publisher edits and third-party storage remain stated + limitations. +2. SEC-03: production export code defines and validates exact CSV, metadata, nested ranking, weekly + count, and source-path allowlists. Schema expansion now requires an intentional code and test + change. +3. SEC-05: the record recommends defense-in-depth acceptance for human review while retaining + sanitization, fencing, canary, output/frontmatter validation, prompt lint, and red-team controls. + Semantic paraphrases remain outside phrase-matching guarantees. + +These changes provide implementation evidence only. Hermes, URL, and sponsor sign-off remain pending. + +## Phase 4: Plan Gated Rollouts and Cost Measurement + +Cost experiment: + +1. Use one main SHA and one hydrated publish SHA for every workload variant. +2. Measure baseline, topic hubs, data pages, repository pages, and optionally the reviewed dynamic canary. +3. Collect at least three comparable CI runs, preferably five. +4. Retain raw Hugo and Pagefind samples, workload counts, output sizes, medians, nearest-rank p95, absolute deltas, and per-added-page deltas. +5. Keep thresholds report-only until an owner approves the budget and enforcement date. + +Repository-page activation: + +1. Resolve stable GitHub identity risk or record an explicit accepted-risk disposition. +2. Seed lifecycle parity twice while disabled and require byte-identical output. +3. Run enabled checks and two generations in an isolated checkout at the unchanged threshold. +4. Review every created, rewritten, obsolete, and expired path. +5. Obtain Hermes, URL, and sponsor approval for the exact revision. + +Dynamic-topic canary: + +1. Review the five eligible candidates and select one unambiguous canary. +2. Add the other four to `ignore_topics` as explicit deferrals. +3. Preview the exact mutation in an isolated checkout because current `--dry-run` is a no-op. +4. Validate hub output, registry changes, weekly assignments, taxonomy, log event, rendered output, and rollback behavior. +5. Obtain security and sponsor approval for one publish transaction. + +Success: both rollouts have bounded, reversible execution plans and production flags remain disabled. + +## Phase 5: Validate and Review + +Run focused tests for workflow mapping, internal links, sanitization, lifecycle, taxonomy, export policy, and documentation. Use PR CI for Hugo, browser, axe, and Lighthouse validation when local binaries or system libraries are unavailable. diff --git a/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md b/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md new file mode 100644 index 0000000..b661e43 --- /dev/null +++ b/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md @@ -0,0 +1,207 @@ + +# Implementation Details: Claracle Relaunch Readiness Reconciliation + +## Context Reference + +Sources: .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md; conversation gap analysis (2026-08-02); repository inspection. + +## Implementation Phase 1: Triage the Live Deploy Failure (#644) + + + +### Step 1.1: Diagnose deploy run 30718600607 and classify the failure + +Pull the failed step logs and classify: is it (a) the 2026-W31 restore interacting with the just-merged #646 preservation, (b) a dangling embed/source_page (the #627 class), (c) a publish/main hydration divergence, or (d) unrelated content. + +Commands: +* `gh run view 30718600607 --log-failed | grep -iE "error|fail|does not exist|errorf|source_page|not found" | head -40` +* `gh issue view 644 --json body,createdAt` +* Compare timing against the restore run and #646 merge. + +Files: +* .github/workflows/deploy-site.yml - hydration list +* layouts/embeds/single.html, layouts/shortcodes/observatory-chart.html - errorf sources + +Success criteria: +* Root cause identified and classified against the known failure classes. + +Dependencies: +* gh CLI access. + +### Step 1.2: Apply or plan the fix and confirm a green deploy + +If the fix is minor and reversible (e.g., a missing embed source_page, a hydration path), apply it via a PR to main. If it requires a larger change (e.g., restore-fix interaction), scope a dedicated fix plan instead of inline changes. + +Files: +* .github/workflows/deploy-site.yml or content/embeds/* or content/data/* depending on cause + +Success criteria: +* A deploy run completes successfully, or a scoped fix plan is handed off. + +Dependencies: +* Step 1.1 classification. + +### Step 1.3: Close #644 with the root-cause note + +Comment the root cause and resolution on #644 and close it once the deploy is green. + +Success criteria: +* #644 closed with a documented root cause, or explicitly kept open with a linked fix plan. + +## Implementation Phase 2: Reconcile Plan Checklists to Delivered State + + + +### Step 2.1: Update the 2026-07-31 deploy-hydration plan checkboxes to match shipped PRs + +Mark Phase 3 (deploy hydration restored via #637) and Phase 4 Step 1 (embed source_page guard shipped as check_embed_sources.py in #641) complete; annotate Phase 5 with #627 closed. Keep genuinely open items unchecked. + +Files: +* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md + +Success criteria: +* Checkboxes match the merged PRs (#628/#632/#634/#637/#641) with PR references. + +Dependencies: +* None (documentation only). + +### Step 2.2: Update the 2026-07-29 and 2026-07-30 plan checkboxes for completed work + +Reflect Podcaster smoke reusability + release evidence (#636/#639/#643), and restore-consistency (#640/#646) against the relevant phases (2026-07-29 Phase 8; 2026-07-30 Phases 6-8 where satisfied). Do not mark human/external gates that remain pending. + +Files: +* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md +* .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md + +Success criteria: +* Only items with merged evidence are checked; each carries a PR/issue reference. + +Dependencies: +* None. + +### Step 2.3: Create one status-of-record reconciling delivered vs pending + +Author a single status-of-record (under docs/review/data-observatory-relaunch/ or .copilot-tracking) that maps every relaunch requirement/phase to Done/Pending with evidence, superseding the fragmented view across three plans and covering the final dispositions of epic issues #644/#626/#622/#599/#594. + +Files: +* docs/review/data-observatory-relaunch/status-of-record.md (new) or an agreed location + +Success criteria: +* A reader can determine relaunch readiness from one document. + +Dependencies: +* Steps 2.1-2.2. + +## Implementation Phase 3: Reconcile Product Documents + + + +### Step 3.1: Fix BRD version drift + +Reconcile the BRD Document Control version (v1.1) with the Acceptance section and the PRD REF-1 reference (both cite v1.0). Choose the authoritative version, update cross-references, and add a BRD changelog row if missing. + +Files: +* docs/brds/claracle-data-observatory-relaunch-brd.md +* docs/prds/claracle-data-observatory-relaunch.md (REF-1) + +Success criteria: +* BRD version and all cross-references agree. + +Dependencies: +* None. + +### Step 3.2: Add PRD v1.3 changelog + restore-consistency behavior + +Add a v1.3 changelog entry covering #627-#646 (deploy/hydration cascade, Podcaster smoke hardening, restore-consistency). Extend NFR-002 or add a new NFR capturing that restore preserves the published weekly transaction (article, summary, promotion record, rollups) and does not corrupt provenance. + +Files: +* docs/prds/claracle-data-observatory-relaunch.md (sections 7, 15) + +Success criteria: +* PRD reflects the delivered restore/Podcaster behavior with a dated changelog row, and FR-041's partial status (test-level link check only, no CI link tool) is recorded. + +Dependencies: +* None. + +### Step 3.3: Add a sponsor-approval + launch-gate register + +Add (or link) an artifact that records sponsor approval status and the launch-gate register so the rollout flags have a traceable approval path. + +Files: +* docs/review/data-observatory-relaunch/ (approval + gate register) +* docs/prds/claracle-data-observatory-relaunch.md (link from Acceptance Status / section 13) + +Success criteria: +* Approval status and gate ownership are recorded and linked from the PRD. + +Dependencies: +* None. + +## Implementation Phase 4: Sequence Remaining Launch Gates + + + +### Step 4.1: Consolidate pending gates into the register + +For each pending gate, record owner, dependency, and evidence path: +* GA4/GSC connection - FR-035 and the human actions recorded on closed #599 (jmservera) - blocks OBJ-2/4 baselines +* NFR-004 security sign-off - Hermes +* NFR-005 accessibility evidence - Amy/Fry +* Podcaster downstream run - URL (NFR-002/R-04) +* Refreshed visual acceptance +* Q-01/NFR-009 incremental generation cost +* #626 Lighthouse quality-gate follow-ups (disposition: readiness scope or explicit out-of-scope) +* #622 post-review UX polish (disposition: readiness scope or explicit out-of-scope) + +Files: +* the launch-gate register from Step 3.3 + +Success criteria: +* Every pending gate has owner + dependency + evidence path. + +Dependencies: +* Step 3.3. + +### Step 4.2: Record deferred scope requiring its own plan + +Capture as follow-on planning items: GA4/GSC implementation (#599), repo_pages rollout, dynamic topic-creation rollout, incremental-generation-cost design spike (Q-01). Note each needs its own plan and (for rollout) sponsor approval. + +Files: +* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md (Suggested Follow-On Work) + +Success criteria: +* Deferred workstreams are enumerated with dependencies. + +Dependencies: +* None. + +## Implementation Phase 5: Validation and Re-Review + + + +### Step 5.1: Validate all edited docs + +Run markdown lint and verify internal references, version numbers, and changelog consistency across the edited docs and plans. + +Validation commands: +* markdown lint per .mega-linter.yml on changed .md files +* grep for stale version/reference strings (BRD v1.0 vs v1.1; PRD version) + +Success criteria: +* No lint errors; version/reference consistency verified. + +### Step 5.2: Fix minor validation issues + +Apply straightforward lint/reference fixes directly. + +### Step 5.3: Report blocking issues and hand off deferred plans + +Summarize residual blockers (e.g., #644 if not fixed inline) and hand off the deferred planning items. + +## Dependencies + +* gh CLI, repository write access, markdown lint tooling, sponsor input. + +## Success Criteria + +* Docs and plans are consistent, current, and validated; pending gates and deferred workstreams are enumerated with owners and dependencies. diff --git a/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md b/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md index c4f6b58..877c29e 100644 --- a/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md +++ b/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md @@ -217,13 +217,13 @@ tests/ * [ ] Step 7.3: Measure Hugo and Pagefind separately * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 357-372) -### [ ] Implementation Phase 8: Podcaster Release Smoke +### [x] Implementation Phase 8: Podcaster Release Smoke * [x] Step 8.1: Make the smoke workflow reusable * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 377-392) -* [ ] Step 8.2: Invoke and retain release evidence +* [x] Step 8.2: Invoke and retain release evidence — smoke wired as a blocking post-deploy gate in `deploy-site.yml`; hardened via `#636` (API key), `#639`/`#643`/`#645` (tooling + source-manifest hydration); deploy-site smoke green since 2026-08-01. Note: the real protected Podcaster downstream run (NFR-002 / R-04) remains a pending launch gate (see 2026-07-30 Step 6.3). * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 393-407) ### [ ] Implementation Phase 9: Documentation and Acceptance Evidence diff --git a/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md b/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md index 788985b..04af1cd 100644 --- a/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md +++ b/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md @@ -110,37 +110,37 @@ docs/ * File: `tests/test_pipeline.py` * [x] Step 1.3: Validate locally: `pytest tests/test_pipeline.py`, zizmor on the workflow, and a clean `hugo --minify` build with the embed and its data page rendered -### [ ] Implementation Phase 2: Regenerate Observatory Content onto Publish +### [x] Implementation Phase 2: Regenerate Observatory Content onto Publish -* [ ] Step 2.1: Trigger `crawl-and-publish.yml` (or the appropriate generation workflow) via `workflow_dispatch` and confirm the `generate` job commits `content/data/` and the chart-embed dependencies to `publish` -* [ ] Step 2.2: Confirm `git ls-tree -r --name-only origin/publish -- content/data/` lists `fastest-growing-ai-repositories-this-year/` and the other ranking pages -* [ ] Step 2.3: Confirm the embed dependency resolves against the `publish` content set (the observatory-chart shortcode finds the data page) +* [x] Step 2.1: Trigger `crawl-and-publish.yml` (or the appropriate generation workflow) via `workflow_dispatch` and confirm the `generate` job commits `content/data/` and the chart-embed dependencies to `publish` — regenerated via crawl runs; `#634` stages only existing generated paths in the publish commit +* [x] Step 2.2: Confirm `git ls-tree -r --name-only origin/publish -- content/data/` lists `fastest-growing-ai-repositories-this-year/` and the other ranking pages +* [x] Step 2.3: Confirm the embed dependency resolves against the `publish` content set (the observatory-chart shortcode finds the data page) -### [ ] Implementation Phase 3: Restore Consistent Deploy Hydration +### [x] Implementation Phase 3: Restore Consistent Deploy Hydration -* [ ] Step 3.1: Re-add `content/data/` to the deploy hydration list and revert the interim comment once `publish` reliably carries the pages -* [ ] Step 3.2: Restore the provenance invariant test to require `content/data/` in deploy hydration -* [ ] Step 3.3: Confirm a full deploy build succeeds against the hydrated `publish` content set +* [x] Step 3.1: Re-add `content/data/` to the deploy hydration list and revert the interim comment once `publish` reliably carries the pages — restored via `#637` (generalize safe hydration guard and restore content/data deploy) +* [x] Step 3.2: Restore the provenance invariant test to require `content/data/` in deploy hydration — `#637` +* [x] Step 3.3: Confirm a full deploy build succeeds against the hydrated `publish` content set — deploy-site runs green since 2026-08-01 -### [ ] Implementation Phase 4: CI Deploy-Parity Guard +### [x] Implementation Phase 4: CI Deploy-Parity Guard -* [ ] Step 4.1: Add a CI build that reproduces the deploy publish-hydration (or a lightweight check that every `content/embeds/*` `source_page` resolves to an existing data page in the built content set) -* [ ] Step 4.2: Wire the guard into the production-site job so `main`/`publish` divergence fails CI, not the deploy -* [ ] Step 4.3: Add or extend tests covering the guard behavior +* [x] Step 4.1: Add a CI build that reproduces the deploy publish-hydration (or a lightweight check that every `content/embeds/*` `source_page` resolves to an existing data page in the built content set) — shipped as `scripts/check_embed_sources.py` (`#641`) +* [x] Step 4.2: Wire the guard into the production-site job so `main`/`publish` divergence fails CI, not the deploy — wired into `.github/workflows/ci.yml` ("Validate embed source pages" step) via `#641` +* [x] Step 4.3: Add or extend tests covering the guard behavior — `tests/test_embed_sources.py` (`#641`) ### [ ] Implementation Phase 5: Validation and Re-Review -* [ ] Step 5.1: Run full validation: `pytest tests/`, ruff, zizmor on changed workflows, and a clean Hugo build -* [ ] Step 5.2: Confirm the production deploy succeeds end to end and close issue #627 -* [ ] Step 5.3: Reconcile PRD NFR-011/012, R-08, and Q-03 status with the delivered state +* [x] Step 5.1: Run full validation: `pytest tests/`, ruff, zizmor on changed workflows, and a clean Hugo build — validated per PR CI (`#628`/`#634`/`#637`/`#641`) +* [x] Step 5.2: Confirm the production deploy succeeds end to end and close issue #627 — `#627` CLOSED; deploy-site green +* [ ] Step 5.3: Reconcile PRD NFR-011/012, R-08, and Q-03 status with the delivered state — handled by the 2026-08-02 relaunch-readiness reconciliation (PRD Phase 3) ## Parallelization Summary diff --git a/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md new file mode 100644 index 0000000..c7084eb --- /dev/null +++ b/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md @@ -0,0 +1,110 @@ +--- +applyTo: '.copilot-tracking/changes/2026-08-02/claracle-gated-rollout-cost-changes.md' +--- + +# Implementation Plan: Claracle Gated Rollouts and Cost Measurement + +## User Requests + +* Plan the `repo_pages` rollout +* Plan the `dynamic_topic_creation` rollout +* Plan the incremental generation cost spike for Q-01/NFR-009 + +## Preconditions + +* Keep `repo_pages.enabled = false` and `topic_hubs.dynamic_creation.enabled = false` during planning and preflight. +* Use one reviewed main SHA and one hydrated publish SHA for comparable experiments. +* Preserve Lighthouse thresholds and keep new cost thresholds report-only until approved. +* Require separate sponsor decisions for each rollout flag. + +## Implementation Checklist + +### [ ] Phase 1: Build a Report-Only Cost Experiment + + + +* [ ] Add a manually dispatchable experiment that creates clean, isolated workload variants from the same main and publish revisions +* [ ] Measure baseline, topic hubs, data pages, repository pages, and an optional approved dynamic canary +* [ ] Record Hugo and Pagefind versions, durations, page counts, indexed counts, output bytes, variant, runner, and both SHAs +* [ ] Retain at least three comparable runs, preferably five +* [ ] Aggregate median, nearest-rank p95, absolute delta, percent delta, and marginal milliseconds per added source page +* [ ] Publish the report without a blocking threshold + +Success: Q-01 has reproducible page-class attribution and an owner-reviewable report. + +### [ ] Phase 2: Close Repository Identity and Lifecycle Preconditions + + + +* [ ] Obtain stable GitHub IDs for the production corpus or record an explicit accepted-risk disposition for fallback name identity +* [ ] Hydrate the target publish revision and seed lifecycle parity twice while production generation remains disabled +* [ ] Require 263 qualified histories, pages, and derived identities, with byte-identical second output +* [ ] Exercise and review one rename, archive, confirmed deletion, retention, and expiry transition against production-shaped data +* [ ] Record Hermes and sponsor dispositions for identity and deletion policy + +Success: FR-020 through FR-022 have corpus-level identity and lifecycle evidence rather than fixture-only proof. + +### [ ] Phase 3: Add a Safe Dynamic-Topic Preview and Canary + + + +* [ ] Change `--dry-run` from an early exit into a non-mutating proposed-change report, or add an equivalent preview command +* [ ] Test that preview reads candidates but writes no hub, registry, weekly frontmatter, taxonomy, or log changes +* [ ] Review the five currently eligible candidates and choose one unambiguous canary +* [ ] Add all non-canary candidates to `ignore_topics` as explicit temporary deferrals +* [ ] Generate and review the exact canary transaction in an isolated checkout +* [ ] Validate sanitization, structured YAML, evidence-backed assignments, taxonomy, logging, rendering, and disabled rollback +* [ ] Obtain Hermes and sponsor approval for the exact canary revision + +Success: one bounded candidate can be promoted without exposing all eligible candidates to the same transaction. + +### [ ] Phase 4: Preflight Repository Regeneration + + + +* [ ] Enable the existing repository config only in an isolated checkout at the unchanged recurrence threshold +* [ ] Run enabled `--check`, then two full generations +* [ ] Review every created, rewritten, obsolete, and expired path +* [ ] Require byte-stable second generation and no unapproved removals +* [ ] Run Hugo, pinned Pagefind, rendered metadata, internal links, axe, Lighthouse, and the cost experiment +* [ ] Obtain Hermes, URL, and sponsor approval for the exact activation revision + +Success: the first production run is a reviewed 263-page regeneration transaction with known cost and rollback. + +### [ ] Phase 5: Execute and Observe Rollouts + + + +* [ ] Enable only the separately approved flag and run one publish transaction +* [ ] Inspect the committed generated-state diff before deployment +* [ ] Confirm production rendering, lifecycle, telemetry, and downstream smoke +* [ ] For rollback, disable the flag and revert the generated transaction; disabling alone does not undo durable mutations +* [ ] Expand dynamic candidates one reviewed item at a time +* [ ] Add blocking budgets only after the report-only observation window and explicit owner approval + +Success: each rollout is independently approved, observable, and reversible. + +## Validation Commands + +* `python -m pytest tests/test_observatory_repos.py tests/test_topic_hubs.py tests/test_taxonomy_registry.py` +* `python scripts/discover_topic_candidates.py --check` +* `python scripts/generate_data_pages.py --check` +* `python scripts/export_observatory_dataset.py --check` +* `python scripts/export_trend_explorer_data.py --check` +* `hugo --minify` +* `npx "pagefind@1.5.2" --site public/` +* `python scripts/check_internal_links.py public --base-url "https://claracle.com/"` +* `python -m pytest tests/` +* `ruff check .` +* `ruff format --check .` + +Workflow changes also require Zizmor and Checkov. Browser and Lighthouse validation may run in GitHub CI when local system dependencies are unavailable. + +## Dependencies + +* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md +* docs/review/data-observatory-relaunch/owner-action-register.md +* Stable identity decision +* Hermes security disposition +* URL workflow review +* Separate jmservera sponsor decisions diff --git a/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md new file mode 100644 index 0000000..f9de48d --- /dev/null +++ b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md @@ -0,0 +1,76 @@ +--- +applyTo: '.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md' +--- + +# Implementation Plan: Claracle Relaunch Follow-Up Execution + +## User Requests + +* Publish review corrections: commit, push, and resolve PR threads +* Complete GA4/GSC setup +* Close security, accessibility, Podcaster, visual, and sponsor acceptance gates +* Plan repository-page, dynamic-topic, and generation-cost work + +## Context Summary + +Research shows that the repository implementation is ahead of the acceptance record. The remaining work mixes executable repository evidence with actions that require Google credentials, protected environment policy, downstream authorization, assistive technology, security review, visual review, and sponsor authority. + +## Implementation Checklist + +### [x] Phase 1: Publish Review Corrections + + + +* [x] Commit the issue-state correction and RPI logs as `8fddceb` +* [x] Push the active PR branch +* [x] Resolve both addressed PR #647 review threads + +### [ ] Phase 2: Reconcile GA4/GSC Evidence + + + +* [x] Verify production GA configuration presence, GSC metadata absence, sitemap response, and secret names without exposing values +* [x] Correct the baseline and status of record to distinguish deployed wiring from external acceptance +* [x] Clarify that production GA configuration is injected through an Actions secret +* [x] Complete Google property verification, sitemap submission, Realtime confirmation, and product link (owner-confirmed by jmservera on 2026-08-02) +* [ ] Transcribe the supplied GSC performance export and retain production consent observations + +### [x] Phase 3: Refresh Acceptance Evidence + + + +* [x] Reconcile the security review with current SEC-01 and SEC-04 implementation evidence without granting sign-off +* [x] Record current CI, environment, Podcaster, accessibility, and visual evidence boundaries +* [x] Correct #622 to non-blocking polish and keep #626 as independent quality hardening +* [x] Provide owner-ready records for manual accessibility, protected Podcaster, visual, security, and sponsor decisions + +### [x] Phase 4: Plan Gated Rollouts and Cost Measurement + + + +* [x] Plan a report-only workload-variant experiment for Q-01/NFR-009 +* [x] Plan repository-page activation with identity, lifecycle, diff, and rollback gates +* [x] Plan one reviewed dynamic-topic canary with explicit deferrals +* [x] Keep both production rollout flags disabled pending approval + +### [x] Phase 5: Validate and Review + + + +* [x] Run focused and full executable validation and inspect PR checks +* [x] Record completed work, owner-gated blockers, and final review disposition + +## Dependencies + +* Research documents under .copilot-tracking/research/2026-08-02/ and .copilot-tracking/research/subagents/2026-08-02/ +* GitHub repository and Actions metadata access +* Google account access for FR-035 external acceptance +* Podcaster maintainer authorization and protected environment policy +* Hermes, URL, Amy, Fry, and jmservera review authority + +## Success Criteria + +* Review corrections are published and review threads resolved. +* Repository records accurately separate the completed GA4/GSC connection from pending baseline and production consent evidence. +* Every acceptance gate has current evidence, a concrete owner action, and no unsupported completion claim. +* Rollout and cost work is implementation-ready while both flags remain disabled. diff --git a/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md new file mode 100644 index 0000000..f84a4e3 --- /dev/null +++ b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md @@ -0,0 +1,120 @@ +--- +applyTo: '.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md' +--- + +# Implementation Plan: Claracle Relaunch Readiness Reconciliation + +## Overview + +Triage the live deploy failure, reconcile the three relaunch plans and the PRD/BRD with the delivered repository state, and produce a single sequenced launch-gate register so the relaunch decision is grounded in accurate, current evidence. + +## Objectives + +### User Requirements + +* Review the plans, PRD, and BRD and address what is missing - Source: user request, 2026-08-02 +* Produce implementation-ready planning artifacts from the gap analysis - Source: task-plan prompt, 2026-08-02 + +### Derived Objectives + +* Resolve the live deploy failure before reconciling status so the record reflects a green pipeline - Derived from: open issue #644 +* Bring the three overlapping plan checklists to the true delivered state and collapse them into one status-of-record - Derived from: stale checkboxes across the 2026-07-29/30/31 plans +* Make the PRD and BRD internally consistent and current with the #627-#646 workstream - Derived from: BRD version drift and PRD changelog lag +* Consolidate the remaining launch gates into one owner/evidence register rather than leaving them scattered - Derived from: pending NFR-004/005/007, Podcaster run, visuals, and sponsor approval + +## Context Summary + +### Project Files + +* docs/prds/claracle-data-observatory-relaunch.md - PRD v1.2; changelog, NFRs, rollout flags, open questions +* docs/brds/claracle-data-observatory-relaunch-brd.md - BRD v1.1 with v1.0 cross-references (drift) +* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md - Phases 7-10 partial +* .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md - Phases 6-8 open +* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md - Phases 2-5 open; Phase 4 shipped but unmarked +* config/observatory.toml - repo_pages flag disabled (confirmed) +* hugo.toml - fork-safe GA4/GSC defaults are empty; production configuration and platform acceptance require separate evidence +* docs/review/data-observatory-relaunch/ - bounded acceptance evidence and pending gates + +### References + +* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md - gap analysis and verified findings +* Issue #644 - live deploy failure (run 30718600607) +* Issue #599 - Connect GA4 + Google Search Console (FR-035) +* Issue #594 - Epic: Claracle Data Observatory Relaunch + +### Standards References + +* .github/copilot-instructions.md - testing, workflow security, cross-repository conventions +* .github/instructions/hve-core/markdown.instructions.md - Markdown requirements +* .github/instructions/hve-core/writing-style.instructions.md - documentation voice and style + +## Implementation Checklist + +### [x] Implementation Phase 1: Triage the Live Deploy Failure (#644) + + + +* [x] Step 1.1: Diagnose deploy run 30718600607 and classify the failure — dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) broke the Podcaster smoke gate (class b/dangling reference) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 12-30) +* [x] Step 1.2: Apply or plan the fix and confirm a green deploy — resolved by already-merged `#645`/`#646`; deploy-site green since 2026-08-01 (runs 30720064394, 30721575540) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 31-43) +* [x] Step 1.3: Close #644 with the root-cause note — #644 already CLOSED (COMPLETED) + +### [x] Implementation Phase 2: Reconcile Plan Checklists to Delivered State + + + +* [x] Step 2.1: Update the 2026-07-31 deploy-hydration plan checkboxes to match shipped PRs (#628/#632/#634/#637/#641) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 55-67) +* [x] Step 2.2: Update the 2026-07-29 and 2026-07-30 plan checkboxes for completed Podcaster/smoke/restore work (#639/#643/#640/#646) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 68-81) +* [x] Step 2.3: Create one status-of-record reconciling delivered vs pending across all three plans, including the final dispositions of epic issues #644/#626/#622/#599/#594 — docs/review/data-observatory-relaunch/status-of-record.md + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 82-94) + +### [x] Implementation Phase 3: Reconcile Product Documents + + + +* [x] Step 3.1: Fix BRD version drift (Document Control vs Acceptance/PRD v1.0 references) — BRD bumped to v1.2 with a Change History; PRD cross-reference aligned to v1.2 + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 99-112) +* [x] Step 3.2: Add PRD v1.3 changelog + restore-consistency behavior for the #627-#646 workstream, and record the FR-041 link-check partial status (test-level only, no CI link tool) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 113-125) +* [x] Step 3.3: Add a sponsor-approval + launch-gate register with owners and evidence links — register in the status-of-record, linked from PRD Acceptance Status, section 13, and REF-10 + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 126-139) + +### [x] Implementation Phase 4: Sequence Remaining Launch Gates + + + +* [x] Step 4.1: Consolidate pending gates (GA4/GSC baseline and consent, external metadata and feed validation, NFR-004 security, NFR-005 a11y, Podcaster run, visuals, Q-01 cost) plus epic issues #626 (Lighthouse) and #622 (UX polish) into the register with owner, dependency, and evidence path — status-of-record launch-gate register + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 144-164) +* [x] Step 4.2: Record deferred scope requiring its own plan (GA4/GSC baseline and consent evidence, repo_pages rollout, dynamic topic rollout, cost spike) — planning log Suggested Follow-On Work (WI-01/03/04/05) + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 165-181) + +### [x] Implementation Phase 5: Validation and Re-Review + + + +* [x] Step 5.1: Validate all edited docs (markdown lint, internal link/reference integrity, changelog/version consistency) — `test_internal_link_checker`/`test_embed_sources` green; referenced files verified; no stale version strings + * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 182-192) +* [x] Step 5.2: Fix minor validation issues — none required +* [x] Step 5.3: Report blocking issues and hand off deferred plans — no blockers; #644 resolved; deferred plans in the log (WI-01/03/04/05) + +## Planning Log + +See .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md for discrepancy tracking, implementation paths considered, and suggested follow-on work. + +## Dependencies + +* gh CLI access for #644 diagnosis and issue updates +* Repository write access for docs/ and .copilot-tracking/ edits +* Markdown lint tooling per .mega-linter.yml +* Sponsor (jmservera) input for the approval artifact and gate ownership + +## Success Criteria + +* Issue #644 is diagnosed and either fixed with a green deploy or handed off with a scoped fix plan - Traces to: open issue #644 +* All three relaunch plans reflect true delivered state and a single status-of-record exists - Traces to: stale checkbox findings +* PRD and BRD are internally consistent, version-correct, and current through the #627-#646 workstream - Traces to: BRD version drift and PRD changelog lag +* A single launch-gate register lists every pending gate with owner, dependency, and evidence path - Traces to: pending NFR-004/005/007, Podcaster, visuals, sponsor approval +* Deferred implementation workstreams are captured for separate planning - Traces to: #599, repo_pages/dynamic rollout, Q-01 diff --git a/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md new file mode 100644 index 0000000..3063bd7 --- /dev/null +++ b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md @@ -0,0 +1,42 @@ + +# Planning Log: Claracle Relaunch Follow-Up Execution + +## Selected Path + +Execute repository-verifiable work and prepare owner-ready handoffs. Do not hardcode analytics identifiers, configure Google properties without account access, trigger duplicate-prone podcast generation, grant human sign-off, or enable rollout flags. + +## Discrepancies + +* The empty checked-in GA default is fork-safe configuration, not proof that production GA4 is disconnected. +* Issue #599 closed after agent-side wiring, while its human-action checklist remains incomplete. +* Issue #622 calls itself non-blocking polish; it must not be represented as a mandatory launch gate without a sponsor decision. +* A real Podcaster run and an environment-bound dry run exist, but no run combines both properties. +* Hugo/Pagefind timing separation is shipped; Q-01 requires workload attribution and retained statistics rather than another timer split. +* `discover_topic_candidates.py --check` fails against the inherited registry at commit `8fddceb`. A temporary regeneration retains 2,173 total candidates and the same five eligible candidates, while rotating four sanitized keys. This branch does not rewrite publish-derived state; refresh it through the owning generation workflow. + +## Deferred Owner Actions + +* jmservera: GA4/GSC connection actions completed; GSC export transcription, production consent observations, processed sitemap review, and rollout decisions remain +* Hermes: SEC-01 through SEC-06 dispositions and NFR-004 sign-off +* URL: protected environment and secret-scope review +* Podcaster maintainer: idempotency or one-run authorization +* Fry and accessibility reviewer: manual keyboard and screen-reader record +* Amy: final visual matrix and acceptance conclusion + +## Repository-Executable Security Closure + +* SEC-02 implementation evidence is complete: official snippets use no-referrer and iframe analytics + requires explicit frame-local Claracle consent. Hermes privacy disposition remains pending. +* SEC-03 implementation evidence is complete: exact public export and safe source-path allowlists are + enforced by production code and tests. Hermes field-policy approval remains pending. +* SEC-05 has an explicit defense-in-depth recommendation with retained executable controls and stated + semantic limitations. No accepted-risk decision has been recorded. +* Squad agent implemented SEC-02/03/05. Fry rejected the first SEC-02 browser assertion, Hermes tightened the cross-origin default-off proof, and Fry approved the revised executable closure. This is quality approval, not Hermes security sign-off. + +## Safety Decisions + +* Keep `GA_MEASUREMENT_ID` and `GSC_SITE_VERIFICATION` values out of source and evidence. +* Keep `repo_pages.enabled` and `topic_hubs.dynamic_creation.enabled` false. +* Keep quality thresholds unchanged. +* Keep cost thresholds report-only until approved. +* Do not dispatch real downstream generation during planning. diff --git a/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md new file mode 100644 index 0000000..05d2a8d --- /dev/null +++ b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md @@ -0,0 +1,74 @@ + +# Planning Log: Claracle Relaunch Readiness Reconciliation + +## Discrepancy Log + +Gaps and differences identified between research findings and the implementation plan. + +### Unaddressed Research Items + +* DR-01: Full GA4/GSC connection (setting ga_measurement_id, verifying the property, submitting the sitemap, capturing the baseline) + * Source: research 2026-08-02 (Unmet launch gates); issue #599 + * Reason: Issue #599 closed as completed on 2026-08-01 with a human-action checklist still outstanding; this plan only registers and sequences that external acceptance work + * Impact: high (blocks OBJ-2/OBJ-4 baselines) +* DR-02: repo_pages and dynamic_topic_creation rollout enablement + * Source: config/observatory.toml flags disabled; PRD section 13 + * Reason: Requires separate sponsor approval and its own rollout plan + * Impact: medium (Wave 2/dynamic scope not live) +* DR-03: Incremental generation cost/time design spike (Q-01/NFR-009) + * Source: PRD section 14; BRD section 13 + * Reason: Needs a measurement spike, not documentation reconciliation + * Impact: medium (capacity risk unquantified) +* DR-04: Open issues #626 (Lighthouse follow-ups) and #622 (UX polish) reconciliation - RESOLVED 2026-08-02 + * Source: research 2026-08-02 (Open issues); issues #626, #622 + * Resolution: Plan Step 2.3 now lists #644/#626/#622/#599/#594 in the status-of-record; Step 4.1 adds #626 and #622 to the gate register with a disposition (readiness scope or explicit out-of-scope). Details Step 2.3 (status-of-record scope) and Step 4.1 (gate list) both enumerate #626/#622. + * Impact: closed (readiness view and gate register now cover the epic's open work) +* DR-05: FR-041 internal link-checker partial status not reconciled in the PRD/status-of-record - RESOLVED 2026-08-02 + * Source: research 2026-08-02 (Verified findings, FR-041) + * Resolution: Plan Step 3.2 now records the FR-041 link-check partial status (test-level only, no CI link tool); Details Step 3.2 success criteria captures the same partial-satisfaction statement. + * Impact: closed (FR-041 traceability now explicitly marked partial) + +### Plan Deviations from Research + +* DD-01: External/human launch gates (security sign-off, accessibility, Podcaster run, visuals, sponsor approval) are registered and sequenced rather than executed + * Research recommends: close the gates + * Plan implements: consolidate into one owner/evidence register with sequencing + * Rationale: these gates depend on humans and external platforms outside a planning/documentation change; execution belongs to their owners with dated evidence + +## Implementation Paths Considered + +### Selected: Single reconciliation plan (docs + status-of-record) with #644 triage first + +* Approach: triage the live deploy failure, correct the three plan checklists, reconcile PRD/BRD, and consolidate a launch-gate register; defer external gate execution +* Rationale: directly answers "what's missing" with accurate state and a single readiness view; low-risk and mostly documentation +* Evidence: research 2026-08-02 (Planning approach) + +### IP-01: One mega-plan that also implements every launch gate + +* Approach: fold GA4/GSC, security, a11y, Podcaster run, and rollout into one plan +* Trade-offs: comprehensive but mixes documentation with external/human execution; long-lived and hard to validate +* Rejection rationale: each external gate merits its own plan and owner; a mega-plan would stall on human dependencies + +### IP-02: Skip planning and edit docs directly + +* Approach: immediately edit PRD/BRD/plans +* Trade-offs: faster but loses traceability and the #644 dependency ordering +* Rejection rationale: the reconciliation touches multiple documents and a live blocker; a checklist keeps it ordered and reviewable + +## Suggested Follow-On Work + +* WI-01: GA4 + GSC connection implementation plan (high) - set ga_measurement_id, verify property, submit sitemap, capture dated baseline + * Source: research (GA4/GSC); human-action checklist on closed issue #599 + * Dependency: sponsor/platform access +* WI-02: Deploy failure #644 dedicated fix plan (high) - NOT NEEDED. #644 is CLOSED: root cause was a dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) breaking the Podcaster smoke gate; resolved by `#645`/`#646`, deploy-site green since 2026-08-01. No dedicated fix plan required. + * Source: open issue #644 (now closed) + * Dependency: none +* WI-03: repo_pages rollout plan (medium) - enable flag, lifecycle acceptance, sponsor approval + * Source: config/observatory.toml; PRD FR-020-022 + * Dependency: sponsor approval, security/lifecycle evidence +* WI-04: Dynamic topic-creation rollout plan (medium) + * Source: PRD FR-004 + * Dependency: sponsor approval, security evidence +* WI-05: Incremental-generation-cost design spike (medium) - quantify hub/data/repo build cost (Q-01/NFR-009) + * Source: PRD section 14; BRD section 13 + * Dependency: none diff --git a/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/handoff.md b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/handoff.md new file mode 100644 index 0000000..dab746f --- /dev/null +++ b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/handoff.md @@ -0,0 +1,75 @@ + +# PR Review Handoff: reconcile-claracle-relaunch-readiness-2026-08-02 + +## PR Overview + +PR #647 reconciles the Claracle relaunch PRD, BRD, implementation tasks, delivered repository state, and remaining launch gates. The review found and resolved one synchronization issue before merge. + +* Branch: `reconcile/claracle-relaunch-readiness-2026-08-02` +* Base Branch: `main` +* Reviewed Source Commit: `46f5fb3f4f1d4a402953876b4f5dbda7d2a953b1` +* Total Source Files Changed: 35 +* Total Review Findings: 1 +* Open Review Findings: 0 + +## PR Comments Ready for Submission + +No unresolved PR comments remain. RI-001 was corrected directly before merge. + +## Resolved Finding + +### RI-001: Synchronize product summaries and task ownership + +* Category: Functional correctness / Documentation +* Severity: Medium +* Status: ✅ Resolved + +The PRD and BRD still described the GA4/GSC connection as pending after owner confirmation recorded it complete. Four external validation rows also lacked a corresponding owner action, and the follow-up evidence phase was marked complete while one child task remained unchecked. + +Resolution: + +* Record the GA4/GSC connection as complete while preserving numeric baseline and production consent evidence as pending +* Add owners and actions for social preview, Rich Results, Schema.org, and production feed validation +* Mark the partially complete GA4/GSC evidence phase unchecked +* Replace stale implementation and access/verification task wording +* Keep `dynamic_topic_creation` and `repo_pages` disabled with separate sponsor decisions required + +## Validation + +* Focused tests: 9 passed, 1 skipped +* Full tests: 1,392 passed, 19 skipped, 34 subtests passed +* Ruff lint: passed +* Ruff format: passed +* Diff whitespace: passed +* Secret-value scan: clean +* CI Python: passed +* CI Production site, Hugo, Playwright, accessibility, and Lighthouse: passed +* Checkov, CodeQL, Bandit, zizmor, Squad CI, and preview: passed +* Merge state: clean + +## Review Summary by Category + +* Security Issues: 0 open +* Code Quality: 0 open +* Convention Violations: 0 open +* Documentation: 1 resolved + +## Instruction Compliance + +* ✅ Repository instructions: tests and cross-repository boundaries preserved +* ✅ Markdown instructions: links and current-state wording validated +* ✅ Prompt-builder instructions: task parent and child states aligned +* ✅ Merge instructions: remote refs refreshed, no conflicts, clean merge candidate + +## Residual Product Gates + +These are documented follow-up work, not merge blockers for PR #647: + +* GA4/GSC baseline transcription and production consent observations +* Hermes security disposition +* Keyboard and screen-reader accessibility acceptance +* Protected real Podcaster run +* External metadata, social preview, structured-data, and feed validation +* Refreshed visual acceptance +* Report-only generation cost experiment +* Separate sponsor decisions for each disabled rollout flag diff --git a/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/in-progress-review.md b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/in-progress-review.md new file mode 100644 index 0000000..24aa91b --- /dev/null +++ b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/in-progress-review.md @@ -0,0 +1,100 @@ + +# PR Review Status: reconcile-claracle-relaunch-readiness-2026-08-02 + +## Review Status + +* Phase: 4 - Complete +* Last Updated: 2026-08-02T21:12:00Z +* Summary: PRD, BRD, status register, owner actions, and task plans are synchronized; all source checks pass and no findings remain open. + +## Branch and Metadata + +* Normalized Branch: `reconcile-claracle-relaunch-readiness-2026-08-02` +* Source Branch: `reconcile/claracle-relaunch-readiness-2026-08-02` +* Base Branch: `main` +* Pull Request: [#647](https://github.com/jmservera/SquadScope/pull/647) +* Linked Work Items: #594, #599, #622, #626, #644 +* Author Intent: Reconcile the relaunch PRD, BRD, plans, delivered state, and launch-gate ownership before merge. + +## Diff Mapping + +The full 3,700-line structured diff and commit history are retained in [pr-reference.xml](pr-reference.xml). Review focus covered these requirement and acceptance surfaces: + +| File | Type | Review Focus | Status | +|------|------|--------------|--------| +| [PRD](../../../../docs/prds/claracle-data-observatory-relaunch.md) | Modified | FR-035, NFR-004/005/007/008, dependencies, rollout state | ✅ Reviewed | +| [BRD](../../../../docs/brds/claracle-data-observatory-relaunch-brd.md) | Modified | Acceptance status, metrics, open actions, sponsor authority | ✅ Reviewed | +| [Status of record](../../../../docs/review/data-observatory-relaunch/status-of-record.md) | Added | Delivered and pending gates, owners, dependencies | ✅ Reviewed | +| [Acceptance index](../../../../docs/review/data-observatory-relaunch/README.md) | Modified | External evidence matrix | ✅ Reviewed | +| [Owner action register](../../../../docs/review/data-observatory-relaunch/owner-action-register.md) | Added | Human and protected-environment actions | ✅ Reviewed | +| [Readiness plan](../../../../.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md) | Added | Reconciliation completion and deferred scope | ✅ Reviewed | +| [Follow-up plan](../../../../.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md) | Added | Google evidence and acceptance tasks | ✅ Reviewed | +| [Rollout plan](../../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) | Added | Disabled flags, preflight, cost measurement | ✅ Reviewed | + +## Instruction Files Reviewed + +* `.github/copilot-instructions.md`: Repository testing, cross-repository, and DevSecOps requirements +* `AGENTS.md`: Squad is the repository default reviewer +* `hve-core/markdown.instructions.md`: Markdown structure and link rules +* `hve-core/writing-style.instructions.md`: Documentation voice and terminology +* `hve-core/prompt-builder.instructions.md`: Task-plan artifact conventions +* `hve-core/git-merge.instructions.md`: Clean-worktree, fetch, conflict, and completion controls + +## Review Items + +### 🔍 In Review + +None. + +### ✅ Approved for PR Comment + +#### RI-001: Product summaries and task ownership drifted from the status of record + +* Category: Functional correctness / Documentation +* Severity: Medium +* Decision: Approved and resolved in the working tree +* Resolution: Record GA4/GSC connection as complete while retaining baseline and consent evidence as pending; add explicit owners and actions for social preview, Rich Results, Schema.org, and production feed validation; mark the partially complete evidence phase unchecked; update stale implementation and dependency wording. +* Evidence: Independent Squad review plus focused repository validation. + +### ❌ Rejected / No Action + +* Historical research snapshots remain unchanged where they accurately describe evidence available at their original capture time. +* Rollout flags remain disabled; no acceptance or sponsor decision was inferred from repository automation. + +## Action Log + +* Generated `pr-reference.xml` against the merge base with `origin/main`. +* Parsed changed files and commit history with the pr-reference skill. +* Compared PRD, BRD, acceptance index, status register, owner-action register, and all current relaunch task plans. +* Ran independent Squad synchronization review. +* Applied RI-001 corrections across current product, acceptance, and task artifacts. +* Ran focused tests: 9 passed, 1 skipped. +* Ran full tests: 1,392 passed, 19 skipped, 34 subtests passed. +* Ran Ruff lint and format checks: passed. +* Ran `git diff --check`: passed. +* Scanned the diff for GA measurement IDs and verification token values: clean. +* Attempted an isolated Hugo build: local `hugo` executable unavailable; fresh PR CI remains required. +* Refreshed remote refs and checked merge-tree conflict markers: none found. +* Published synchronization correction as `46f5fb3`. +* Confirmed fresh CI Python and Production site checks passed, including Hugo, Playwright, accessibility, and Lighthouse. +* Confirmed every PR status check passed and merge state is clean. +* Finalized [handoff.md](handoff.md). + +## Residual Manual Gates + +* Numeric GA4/GSC baseline transcription and production consent observations +* Hermes security disposition +* Keyboard and screen-reader accessibility acceptance +* Protected real Podcaster run +* External metadata, social preview, structured-data, and feed validation +* Refreshed visual acceptance +* Separate sponsor decisions for `dynamic_topic_creation` and `repo_pages` +* Report-only incremental generation cost experiment + +## Next Steps + +* [x] Run full pytest, Ruff, format, and diff validation +* [x] Confirm Hugo, browser, accessibility, and Lighthouse in fresh PR CI +* [x] Commit and publish synchronization corrections +* [x] Confirm fresh PR checks and mergeability +* [ ] Merge PR #647 diff --git a/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/pr-reference.xml b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/pr-reference.xml new file mode 100644 index 0000000..d4992bd --- /dev/null +++ b/.copilot-tracking/pr/review/reconcile-claracle-relaunch-readiness-2026-08-02/pr-reference.xml @@ -0,0 +1,3706 @@ + + +reconcile/claracle-relaunch-readiness-2026-08-02 + + + + origin/main + + + +<\![CDATA[chore(docs): synchronize relaunch acceptance records]]><\![CDATA[- align GA4/GSC completion with pending baseline evidence +- add owners for metadata and feed validation +- correct task phase and dependency states + +📝 - Generated by Copilot +]]> +<\![CDATA[test(analytics): replay queued GA events]]><\![CDATA[Model gtag.js queue processing so consent-time frame events reach the intercepted collection endpoint without duplicating dataLayer. + +🔒 - Generated by Copilot +]]> +<\![CDATA[test(analytics): drive frame-local consent API]]><\![CDATA[Keep cross-origin privacy assertions independent of CookieConsent modal suppression inside third-party frames. + +🔒 - Generated by Copilot +]]> +<\![CDATA[test(analytics): open frame consent explicitly]]><\![CDATA[Exercise the frame-local consent UI without assuming that CookieConsent auto-displays its modal in a cross-origin iframe. + +🔒 - Generated by Copilot +]]> +<\![CDATA[fix(relaunch): enforce embed privacy and export policy]]><\![CDATA[- record completed GA4 and Search Console setup +- add frame-local consent and no-referrer embed contracts +- enforce exact public dataset allowlists + +🔒 - Generated by Copilot +]]> +<\![CDATA[docs(relaunch): reconcile follow-up acceptance and rollout evidence]]><\![CDATA[📝 - Generated by Copilot +]]> +<\![CDATA[docs(relaunch): clarify closed analytics issue state]]><\![CDATA[📝 - Generated by Copilot +]]> +<\![CDATA[docs(relaunch): reconcile Claracle relaunch readiness across plans, PRD, and BRD]]><\![CDATA[Triage the live deploy failure and reconcile the three relaunch plans, the +PRD, and the BRD with the delivered repository state, producing a single +sequenced launch-gate register. + +- Confirm #644 root cause (dangling source_manifest.path breaking the + Podcaster smoke) resolved by #645/#646; deploy-site green since 2026-08-01 +- Update 2026-07-31 deploy-hydration plan (Phases 2-4 delivered via + #634/#637/#641) and 2026-07-29 Phase 8 (smoke gate hardened) +- Add docs/review/.../status-of-record.md: delivered-vs-pending view plus a + launch-gate register with owner, dependency, and evidence path +- BRD v1.2: add Change History, align the PRD cross-reference, link the + status of record +- PRD v1.3: record the #627-#646 workstream, restore-consistency under + NFR-002, FR-041 partial status, resolve Q-03, link the register (REF-10) + +Validation: pytest tests/ (1389 passed), internal link + embed-source checks green. +]]> + + +diff --git a/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md +new file mode 100644 +index 0000000..fbee4c7 +--- /dev/null ++++ b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md +@@ -0,0 +1,71 @@ ++ ++# Changes Log: Claracle Relaunch Follow-Up Execution ++ ++## Related Plans ++ ++* .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md ++* .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md ++ ++## Implementation Date ++ ++2026-08-02 ++ ++## Summary ++ ++Published the PR review correction, reconciled production GA4/GSC observations, refreshed acceptance evidence, created one owner-action register, and produced implementation-ready rollout and cost plans. External account actions and human approvals remain owner-gated. ++ ++## Added ++ ++* .copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md ++* .copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md ++* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md ++* .copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md ++* .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md ++* .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md ++* .copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md ++* .copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md ++* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md ++* docs/review/data-observatory-relaunch/owner-action-register.md ++ ++## Modified ++ ++* hugo.toml ++* docs/growth/ga4-gsc-baseline-2026-07-29.md ++* docs/prds/claracle-data-observatory-relaunch.md ++* docs/review/data-observatory-relaunch/README.md ++* docs/review/data-observatory-relaunch/security-review.md ++* docs/review/data-observatory-relaunch/status-of-record.md ++* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md ++* .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md ++* .copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md ++ ++## Completed Work ++ ++* Pushed correction commit `8fddceb` and resolved both PR #647 review threads ++* Confirmed production GA configuration on the main site and standalone embed without relying on checked-in identifiers ++* Recorded owner-confirmed GA4 stream operation, GSC verification, root sitemap submission, and GA4-to-GSC product link; FR-035 is complete ++* Reconciled SEC-01 and SEC-04 with current sanitization and lifecycle tests ++* Implemented SEC-02 with a no-referrer official iframe snippet and frame-local explicit-consent tests ++* Implemented SEC-03 exact CSV, metadata, nested-object, and source-path allowlists ++* Documented the SEC-05 defense-in-depth recommendation and limitations without recording acceptance ++* Classified #622 as non-blocking polish and #626 as independent hardening ++* Added exact owner actions for analytics, security, accessibility, protected Podcaster, visual, and sponsor evidence ++* Planned report-only cost attribution, one dynamic-topic canary, and repository-page activation with rollback ++ ++## Validation ++ ++* Full pytest: 1,389 passed, 19 skipped, 34 subtests passed ++* Focused acceptance suite: 45 passed, 4 skipped ++* Ruff lint and format: passed ++* Data-page, public dataset, and trend-export checks: passed ++* PR #647 at `8fddceb`: 13 successful checks, including Production site ++* Editor diagnostics and `git diff --check`: passed ++* Security closure focused suite: 217 passed ++* Rendered embed/export suite with Hugo 0.161.1: 10 passed ++* Public dataset freshness, Hugo production build, internal links, Ruff, and diff whitespace: passed ++* Local Playwright analytics execution was attempted but the host lacks Chromium runtime libraries; CI browser execution remains required ++* Final local suite after Squad and Google evidence updates: 1,392 passed, 19 skipped, 34 subtests passed ++ ++## Known Inherited Discrepancy ++ ++`discover_topic_candidates.py --check` reports the candidate registry stale at the inherited commit. Temporary regeneration preserves 2,173 total candidates and the same five eligible candidates while rotating four sanitized keys. The owning publish workflow should refresh this generated state. +diff --git a/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md +new file mode 100644 +index 0000000..a60cf64 +--- /dev/null ++++ b/.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md +@@ -0,0 +1,51 @@ ++ ++# Changes Log: Claracle Relaunch Readiness Reconciliation ++ ++## Related Plan ++ ++.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md ++ ++## Implementation Date ++ ++2026-08-02 ++ ++## Summary ++ ++Reconciled the Claracle relaunch plans, PRD, BRD, and issue evidence into one status of record. One review iteration corrected stale issue-state claims for closed issues #599 and #644 while preserving the outstanding GA4/GSC launch gate. ++ ++## Changes by Category ++ ++### Added ++ ++* .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md ++* .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md ++* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md ++* .copilot-tracking/research/2026-08-01/restore-consistency-640-research.md ++* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md ++* docs/review/data-observatory-relaunch/status-of-record.md ++ ++### Modified ++ ++* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md ++* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md ++* docs/brds/claracle-data-observatory-relaunch-brd.md ++* docs/prds/claracle-data-observatory-relaunch.md ++ ++### Removed ++ ++* None ++ ++## Review Iteration ++ ++PR #647 review found that #599 and #644 were described as open after both had closed as completed on 2026-08-01. The research and status-of-record artifacts now show the final dispositions. FR-035 remains partial because #599 closed with GSC, platform-receipt, and baseline actions still outstanding; later production verification confirmed secret-backed GA configuration is present. ++ ++## Validation ++ ++* Focused documentation tests: 10 passed ++* PR #647 status checks: 13 passed, 0 failed ++* Editor diagnostics: no errors in the corrected files ++* Git whitespace validation: passed ++ ++## Release Summary ++ ++The repository now has one evidence-backed relaunch readiness view. Delivered remediation work is distinguished from pending external acceptance gates, and closed issue state is no longer used as evidence that GA4/GSC acceptance work shipped. +\ No newline at end of file +diff --git a/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md b/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md +new file mode 100644 +index 0000000..64c23ea +--- /dev/null ++++ b/.copilot-tracking/details/2026-08-02/claracle-gated-rollout-cost-details.md +@@ -0,0 +1,35 @@ ++ ++# Implementation Details: Claracle Gated Rollouts and Cost Measurement ++ ++## Cost Experiment Contract ++ ++Every variant starts from a clean destination and the same hydrated source state. The machine-readable record must include main SHA, publish SHA, workload variant, source counts by page class, Hugo and Pagefind versions and raw durations, rendered and indexed counts, output bytes, runner identity, exit state, and Actions URL. ++ ++Use cumulative variants so marginal cost can be calculated without changing generator logic: ++ ++1. Observatory generated classes excluded ++2. Five checked-in topic hubs included ++3. Three generated data pages included ++4. 263 checked-in repository pages included ++5. Optional approved dynamic canary included ++ ++Do not derive a blocking budget from one run. Retain at least three comparable runs and calculate median plus nearest-rank p95 separately for Hugo and Pagefind. ++ ++## Dynamic Preview Contract ++ ++A preview must evaluate the same eligible-candidate and assignment path as write mode while performing no filesystem mutation. Its structured output must identify candidate slug, title, evidence weeks, supporting sources, proposed hub path, proposed weekly assignments, registry effect, and skip reason. Tests must compare preview output with the corresponding isolated write transaction. ++ ++The first canary uses explicit deferrals in `ignore_topics`; no threshold change is permitted. Threshold-based canaries are unsafe because repository generation can classify existing pages as obsolete. ++ ++## Repository Activation Contract ++ ++The isolated enabled preflight must preserve the existing recurrence threshold and hydrated publish state. A reviewer must disposition every obsolete or expired path. No removal is accepted from mere crawl absence. The second generation must be byte-stable. ++ ++Rollback has two parts: ++ ++1. Disable the production flag to stop future mutation. ++2. Revert the generated-state transaction to undo pages, ledgers, registries, assignments, and logs already committed. ++ ++## Approval Contract ++ ++Hermes approves security and lifecycle policy. URL approves workflows, secret scope, and retained artifacts. jmservera separately approves the dynamic-topic canary and repository-page activation. Each approval identifies the exact revision, evidence, conditions, rollback owner, and date. +diff --git a/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md b/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md +new file mode 100644 +index 0000000..5212963 +--- /dev/null ++++ b/.copilot-tracking/details/2026-08-02/claracle-relaunch-followup-execution-details.md +@@ -0,0 +1,87 @@ ++ ++# Implementation Details: Claracle Relaunch Follow-Up Execution ++ ++## Phase 1: Publish Review Corrections ++ ++Commit and push the reviewed #599/#644 state corrections, then resolve the two PR #647 threads only after the changed diff is visible remotely. ++ ++Success: commit `8fddceb` is on the PR branch and both threads are resolved. ++ ++## Phase 2: Reconcile GA4/GSC Evidence ++ ++Keep both checked-in Hugo defaults empty. Record only presence-level production observations and secret names. Never record the GA identifier or GSC token. ++ ++Owner handoff completed on 2026-08-02: ++ ++1. The deployed ID maps to the intended Claracle stream. ++2. The GSC property is verified without requiring the optional HTML-tag secret path. ++3. `https://claracle.com/sitemap.xml` was submitted and the GA4 stream was linked to GSC. ++4. GA4 is operational, and a GSC performance export was supplied. ++ ++Remaining evidence work: ++ ++1. Transcribe the supplied GSC performance values once the attachment is available as a readable file. ++2. Retain denied and granted production consent observations. ++3. Confirm GSC processing and review indexed and excluded URL counts. ++ ++Success: FR-035 connection and submission are complete; baseline transcription and NFR-008 production consent evidence remain open. ++ ++## Phase 3: Refresh Acceptance Evidence ++ ++Update the security record to acknowledge implemented candidate-title sanitization and lifecycle fixtures while retaining Hermes disposition requirements. Add owner-ready evidence records for manual accessibility, visual review, protected Podcaster execution, and sponsor decisions. Do not mark a human gate complete from automated tests. ++ ++Protected Podcaster sequence: ++ ++1. Confirm downstream idempotency or authorize a specific eligible week. ++2. Define required reviewers and branch policy for a real-generation environment. ++3. Bind the real generation job to that environment through a separately reviewed workflow change. ++4. Run once and retain the approver, week, manifest run, article digest, Actions URL, downstream job ID, and final conclusion. ++ ++Success: the acceptance index identifies current automated evidence and exact remaining owner actions. ++ ++Repository-executable security closure added on 2026-08-02: ++ ++1. SEC-02: generated iframe snippets use `referrerpolicy="no-referrer"`; analytics remains disabled ++ until explicit consent inside the Claracle frame. Tests cover rendered markup, default-off wiring, ++ and the existing browser consent behavior. Publisher edits and third-party storage remain stated ++ limitations. ++2. SEC-03: production export code defines and validates exact CSV, metadata, nested ranking, weekly ++ count, and source-path allowlists. Schema expansion now requires an intentional code and test ++ change. ++3. SEC-05: the record recommends defense-in-depth acceptance for human review while retaining ++ sanitization, fencing, canary, output/frontmatter validation, prompt lint, and red-team controls. ++ Semantic paraphrases remain outside phrase-matching guarantees. ++ ++These changes provide implementation evidence only. Hermes, URL, and sponsor sign-off remain pending. ++ ++## Phase 4: Plan Gated Rollouts and Cost Measurement ++ ++Cost experiment: ++ ++1. Use one main SHA and one hydrated publish SHA for every workload variant. ++2. Measure baseline, topic hubs, data pages, repository pages, and optionally the reviewed dynamic canary. ++3. Collect at least three comparable CI runs, preferably five. ++4. Retain raw Hugo and Pagefind samples, workload counts, output sizes, medians, nearest-rank p95, absolute deltas, and per-added-page deltas. ++5. Keep thresholds report-only until an owner approves the budget and enforcement date. ++ ++Repository-page activation: ++ ++1. Resolve stable GitHub identity risk or record an explicit accepted-risk disposition. ++2. Seed lifecycle parity twice while disabled and require byte-identical output. ++3. Run enabled checks and two generations in an isolated checkout at the unchanged threshold. ++4. Review every created, rewritten, obsolete, and expired path. ++5. Obtain Hermes, URL, and sponsor approval for the exact revision. ++ ++Dynamic-topic canary: ++ ++1. Review the five eligible candidates and select one unambiguous canary. ++2. Add the other four to `ignore_topics` as explicit deferrals. ++3. Preview the exact mutation in an isolated checkout because current `--dry-run` is a no-op. ++4. Validate hub output, registry changes, weekly assignments, taxonomy, log event, rendered output, and rollback behavior. ++5. Obtain security and sponsor approval for one publish transaction. ++ ++Success: both rollouts have bounded, reversible execution plans and production flags remain disabled. ++ ++## Phase 5: Validate and Review ++ ++Run focused tests for workflow mapping, internal links, sanitization, lifecycle, taxonomy, export policy, and documentation. Use PR CI for Hugo, browser, axe, and Lighthouse validation when local binaries or system libraries are unavailable. +diff --git a/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md b/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md +new file mode 100644 +index 0000000..b661e43 +--- /dev/null ++++ b/.copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md +@@ -0,0 +1,207 @@ ++ ++# Implementation Details: Claracle Relaunch Readiness Reconciliation ++ ++## Context Reference ++ ++Sources: .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md; conversation gap analysis (2026-08-02); repository inspection. ++ ++## Implementation Phase 1: Triage the Live Deploy Failure (#644) ++ ++ ++ ++### Step 1.1: Diagnose deploy run 30718600607 and classify the failure ++ ++Pull the failed step logs and classify: is it (a) the 2026-W31 restore interacting with the just-merged #646 preservation, (b) a dangling embed/source_page (the #627 class), (c) a publish/main hydration divergence, or (d) unrelated content. ++ ++Commands: ++* `gh run view 30718600607 --log-failed | grep -iE "error|fail|does not exist|errorf|source_page|not found" | head -40` ++* `gh issue view 644 --json body,createdAt` ++* Compare timing against the restore run and #646 merge. ++ ++Files: ++* .github/workflows/deploy-site.yml - hydration list ++* layouts/embeds/single.html, layouts/shortcodes/observatory-chart.html - errorf sources ++ ++Success criteria: ++* Root cause identified and classified against the known failure classes. ++ ++Dependencies: ++* gh CLI access. ++ ++### Step 1.2: Apply or plan the fix and confirm a green deploy ++ ++If the fix is minor and reversible (e.g., a missing embed source_page, a hydration path), apply it via a PR to main. If it requires a larger change (e.g., restore-fix interaction), scope a dedicated fix plan instead of inline changes. ++ ++Files: ++* .github/workflows/deploy-site.yml or content/embeds/* or content/data/* depending on cause ++ ++Success criteria: ++* A deploy run completes successfully, or a scoped fix plan is handed off. ++ ++Dependencies: ++* Step 1.1 classification. ++ ++### Step 1.3: Close #644 with the root-cause note ++ ++Comment the root cause and resolution on #644 and close it once the deploy is green. ++ ++Success criteria: ++* #644 closed with a documented root cause, or explicitly kept open with a linked fix plan. ++ ++## Implementation Phase 2: Reconcile Plan Checklists to Delivered State ++ ++ ++ ++### Step 2.1: Update the 2026-07-31 deploy-hydration plan checkboxes to match shipped PRs ++ ++Mark Phase 3 (deploy hydration restored via #637) and Phase 4 Step 1 (embed source_page guard shipped as check_embed_sources.py in #641) complete; annotate Phase 5 with #627 closed. Keep genuinely open items unchecked. ++ ++Files: ++* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md ++ ++Success criteria: ++* Checkboxes match the merged PRs (#628/#632/#634/#637/#641) with PR references. ++ ++Dependencies: ++* None (documentation only). ++ ++### Step 2.2: Update the 2026-07-29 and 2026-07-30 plan checkboxes for completed work ++ ++Reflect Podcaster smoke reusability + release evidence (#636/#639/#643), and restore-consistency (#640/#646) against the relevant phases (2026-07-29 Phase 8; 2026-07-30 Phases 6-8 where satisfied). Do not mark human/external gates that remain pending. ++ ++Files: ++* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md ++* .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md ++ ++Success criteria: ++* Only items with merged evidence are checked; each carries a PR/issue reference. ++ ++Dependencies: ++* None. ++ ++### Step 2.3: Create one status-of-record reconciling delivered vs pending ++ ++Author a single status-of-record (under docs/review/data-observatory-relaunch/ or .copilot-tracking) that maps every relaunch requirement/phase to Done/Pending with evidence, superseding the fragmented view across three plans and covering the final dispositions of epic issues #644/#626/#622/#599/#594. ++ ++Files: ++* docs/review/data-observatory-relaunch/status-of-record.md (new) or an agreed location ++ ++Success criteria: ++* A reader can determine relaunch readiness from one document. ++ ++Dependencies: ++* Steps 2.1-2.2. ++ ++## Implementation Phase 3: Reconcile Product Documents ++ ++ ++ ++### Step 3.1: Fix BRD version drift ++ ++Reconcile the BRD Document Control version (v1.1) with the Acceptance section and the PRD REF-1 reference (both cite v1.0). Choose the authoritative version, update cross-references, and add a BRD changelog row if missing. ++ ++Files: ++* docs/brds/claracle-data-observatory-relaunch-brd.md ++* docs/prds/claracle-data-observatory-relaunch.md (REF-1) ++ ++Success criteria: ++* BRD version and all cross-references agree. ++ ++Dependencies: ++* None. ++ ++### Step 3.2: Add PRD v1.3 changelog + restore-consistency behavior ++ ++Add a v1.3 changelog entry covering #627-#646 (deploy/hydration cascade, Podcaster smoke hardening, restore-consistency). Extend NFR-002 or add a new NFR capturing that restore preserves the published weekly transaction (article, summary, promotion record, rollups) and does not corrupt provenance. ++ ++Files: ++* docs/prds/claracle-data-observatory-relaunch.md (sections 7, 15) ++ ++Success criteria: ++* PRD reflects the delivered restore/Podcaster behavior with a dated changelog row, and FR-041's partial status (test-level link check only, no CI link tool) is recorded. ++ ++Dependencies: ++* None. ++ ++### Step 3.3: Add a sponsor-approval + launch-gate register ++ ++Add (or link) an artifact that records sponsor approval status and the launch-gate register so the rollout flags have a traceable approval path. ++ ++Files: ++* docs/review/data-observatory-relaunch/ (approval + gate register) ++* docs/prds/claracle-data-observatory-relaunch.md (link from Acceptance Status / section 13) ++ ++Success criteria: ++* Approval status and gate ownership are recorded and linked from the PRD. ++ ++Dependencies: ++* None. ++ ++## Implementation Phase 4: Sequence Remaining Launch Gates ++ ++ ++ ++### Step 4.1: Consolidate pending gates into the register ++ ++For each pending gate, record owner, dependency, and evidence path: ++* GA4/GSC connection - FR-035 and the human actions recorded on closed #599 (jmservera) - blocks OBJ-2/4 baselines ++* NFR-004 security sign-off - Hermes ++* NFR-005 accessibility evidence - Amy/Fry ++* Podcaster downstream run - URL (NFR-002/R-04) ++* Refreshed visual acceptance ++* Q-01/NFR-009 incremental generation cost ++* #626 Lighthouse quality-gate follow-ups (disposition: readiness scope or explicit out-of-scope) ++* #622 post-review UX polish (disposition: readiness scope or explicit out-of-scope) ++ ++Files: ++* the launch-gate register from Step 3.3 ++ ++Success criteria: ++* Every pending gate has owner + dependency + evidence path. ++ ++Dependencies: ++* Step 3.3. ++ ++### Step 4.2: Record deferred scope requiring its own plan ++ ++Capture as follow-on planning items: GA4/GSC implementation (#599), repo_pages rollout, dynamic topic-creation rollout, incremental-generation-cost design spike (Q-01). Note each needs its own plan and (for rollout) sponsor approval. ++ ++Files: ++* .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md (Suggested Follow-On Work) ++ ++Success criteria: ++* Deferred workstreams are enumerated with dependencies. ++ ++Dependencies: ++* None. ++ ++## Implementation Phase 5: Validation and Re-Review ++ ++ ++ ++### Step 5.1: Validate all edited docs ++ ++Run markdown lint and verify internal references, version numbers, and changelog consistency across the edited docs and plans. ++ ++Validation commands: ++* markdown lint per .mega-linter.yml on changed .md files ++* grep for stale version/reference strings (BRD v1.0 vs v1.1; PRD version) ++ ++Success criteria: ++* No lint errors; version/reference consistency verified. ++ ++### Step 5.2: Fix minor validation issues ++ ++Apply straightforward lint/reference fixes directly. ++ ++### Step 5.3: Report blocking issues and hand off deferred plans ++ ++Summarize residual blockers (e.g., #644 if not fixed inline) and hand off the deferred planning items. ++ ++## Dependencies ++ ++* gh CLI, repository write access, markdown lint tooling, sponsor input. ++ ++## Success Criteria ++ ++* Docs and plans are consistent, current, and validated; pending gates and deferred workstreams are enumerated with owners and dependencies. +diff --git a/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md b/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md +index c4f6b58..877c29e 100644 +--- a/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md ++++ b/.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md +@@ -217,13 +217,13 @@ tests/ + * [ ] Step 7.3: Measure Hugo and Pagefind separately + * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 357-372) + +-### [ ] Implementation Phase 8: Podcaster Release Smoke ++### [x] Implementation Phase 8: Podcaster Release Smoke + + + + * [x] Step 8.1: Make the smoke workflow reusable + * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 377-392) +-* [ ] Step 8.2: Invoke and retain release evidence ++* [x] Step 8.2: Invoke and retain release evidence — smoke wired as a blocking post-deploy gate in `deploy-site.yml`; hardened via `#636` (API key), `#639`/`#643`/`#645` (tooling + source-manifest hydration); deploy-site smoke green since 2026-08-01. Note: the real protected Podcaster downstream run (NFR-002 / R-04) remains a pending launch gate (see 2026-07-30 Step 6.3). + * Details: `.copilot-tracking/details/2026-07-29/claracle-data-observatory-relaunch-remediation-details.md` (Lines 393-407) + + ### [ ] Implementation Phase 9: Documentation and Acceptance Evidence +diff --git a/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md b/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md +index 788985b..04af1cd 100644 +--- a/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md ++++ b/.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md +@@ -110,37 +110,37 @@ docs/ + * File: `tests/test_pipeline.py` + * [x] Step 1.3: Validate locally: `pytest tests/test_pipeline.py`, zizmor on the workflow, and a clean `hugo --minify` build with the embed and its data page rendered + +-### [ ] Implementation Phase 2: Regenerate Observatory Content onto Publish ++### [x] Implementation Phase 2: Regenerate Observatory Content onto Publish + + + +-* [ ] Step 2.1: Trigger `crawl-and-publish.yml` (or the appropriate generation workflow) via `workflow_dispatch` and confirm the `generate` job commits `content/data/` and the chart-embed dependencies to `publish` +-* [ ] Step 2.2: Confirm `git ls-tree -r --name-only origin/publish -- content/data/` lists `fastest-growing-ai-repositories-this-year/` and the other ranking pages +-* [ ] Step 2.3: Confirm the embed dependency resolves against the `publish` content set (the observatory-chart shortcode finds the data page) ++* [x] Step 2.1: Trigger `crawl-and-publish.yml` (or the appropriate generation workflow) via `workflow_dispatch` and confirm the `generate` job commits `content/data/` and the chart-embed dependencies to `publish` — regenerated via crawl runs; `#634` stages only existing generated paths in the publish commit ++* [x] Step 2.2: Confirm `git ls-tree -r --name-only origin/publish -- content/data/` lists `fastest-growing-ai-repositories-this-year/` and the other ranking pages ++* [x] Step 2.3: Confirm the embed dependency resolves against the `publish` content set (the observatory-chart shortcode finds the data page) + +-### [ ] Implementation Phase 3: Restore Consistent Deploy Hydration ++### [x] Implementation Phase 3: Restore Consistent Deploy Hydration + + + +-* [ ] Step 3.1: Re-add `content/data/` to the deploy hydration list and revert the interim comment once `publish` reliably carries the pages +-* [ ] Step 3.2: Restore the provenance invariant test to require `content/data/` in deploy hydration +-* [ ] Step 3.3: Confirm a full deploy build succeeds against the hydrated `publish` content set ++* [x] Step 3.1: Re-add `content/data/` to the deploy hydration list and revert the interim comment once `publish` reliably carries the pages — restored via `#637` (generalize safe hydration guard and restore content/data deploy) ++* [x] Step 3.2: Restore the provenance invariant test to require `content/data/` in deploy hydration — `#637` ++* [x] Step 3.3: Confirm a full deploy build succeeds against the hydrated `publish` content set — deploy-site runs green since 2026-08-01 + +-### [ ] Implementation Phase 4: CI Deploy-Parity Guard ++### [x] Implementation Phase 4: CI Deploy-Parity Guard + + + +-* [ ] Step 4.1: Add a CI build that reproduces the deploy publish-hydration (or a lightweight check that every `content/embeds/*` `source_page` resolves to an existing data page in the built content set) +-* [ ] Step 4.2: Wire the guard into the production-site job so `main`/`publish` divergence fails CI, not the deploy +-* [ ] Step 4.3: Add or extend tests covering the guard behavior ++* [x] Step 4.1: Add a CI build that reproduces the deploy publish-hydration (or a lightweight check that every `content/embeds/*` `source_page` resolves to an existing data page in the built content set) — shipped as `scripts/check_embed_sources.py` (`#641`) ++* [x] Step 4.2: Wire the guard into the production-site job so `main`/`publish` divergence fails CI, not the deploy — wired into `.github/workflows/ci.yml` ("Validate embed source pages" step) via `#641` ++* [x] Step 4.3: Add or extend tests covering the guard behavior — `tests/test_embed_sources.py` (`#641`) + + ### [ ] Implementation Phase 5: Validation and Re-Review + + + +-* [ ] Step 5.1: Run full validation: `pytest tests/`, ruff, zizmor on changed workflows, and a clean Hugo build +-* [ ] Step 5.2: Confirm the production deploy succeeds end to end and close issue #627 +-* [ ] Step 5.3: Reconcile PRD NFR-011/012, R-08, and Q-03 status with the delivered state ++* [x] Step 5.1: Run full validation: `pytest tests/`, ruff, zizmor on changed workflows, and a clean Hugo build — validated per PR CI (`#628`/`#634`/`#637`/`#641`) ++* [x] Step 5.2: Confirm the production deploy succeeds end to end and close issue #627 — `#627` CLOSED; deploy-site green ++* [ ] Step 5.3: Reconcile PRD NFR-011/012, R-08, and Q-03 status with the delivered state — handled by the 2026-08-02 relaunch-readiness reconciliation (PRD Phase 3) + + ## Parallelization Summary + +diff --git a/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md +new file mode 100644 +index 0000000..c7084eb +--- /dev/null ++++ b/.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md +@@ -0,0 +1,110 @@ ++--- ++applyTo: '.copilot-tracking/changes/2026-08-02/claracle-gated-rollout-cost-changes.md' ++--- ++ ++# Implementation Plan: Claracle Gated Rollouts and Cost Measurement ++ ++## User Requests ++ ++* Plan the `repo_pages` rollout ++* Plan the `dynamic_topic_creation` rollout ++* Plan the incremental generation cost spike for Q-01/NFR-009 ++ ++## Preconditions ++ ++* Keep `repo_pages.enabled = false` and `topic_hubs.dynamic_creation.enabled = false` during planning and preflight. ++* Use one reviewed main SHA and one hydrated publish SHA for comparable experiments. ++* Preserve Lighthouse thresholds and keep new cost thresholds report-only until approved. ++* Require separate sponsor decisions for each rollout flag. ++ ++## Implementation Checklist ++ ++### [ ] Phase 1: Build a Report-Only Cost Experiment ++ ++ ++ ++* [ ] Add a manually dispatchable experiment that creates clean, isolated workload variants from the same main and publish revisions ++* [ ] Measure baseline, topic hubs, data pages, repository pages, and an optional approved dynamic canary ++* [ ] Record Hugo and Pagefind versions, durations, page counts, indexed counts, output bytes, variant, runner, and both SHAs ++* [ ] Retain at least three comparable runs, preferably five ++* [ ] Aggregate median, nearest-rank p95, absolute delta, percent delta, and marginal milliseconds per added source page ++* [ ] Publish the report without a blocking threshold ++ ++Success: Q-01 has reproducible page-class attribution and an owner-reviewable report. ++ ++### [ ] Phase 2: Close Repository Identity and Lifecycle Preconditions ++ ++ ++ ++* [ ] Obtain stable GitHub IDs for the production corpus or record an explicit accepted-risk disposition for fallback name identity ++* [ ] Hydrate the target publish revision and seed lifecycle parity twice while production generation remains disabled ++* [ ] Require 263 qualified histories, pages, and derived identities, with byte-identical second output ++* [ ] Exercise and review one rename, archive, confirmed deletion, retention, and expiry transition against production-shaped data ++* [ ] Record Hermes and sponsor dispositions for identity and deletion policy ++ ++Success: FR-020 through FR-022 have corpus-level identity and lifecycle evidence rather than fixture-only proof. ++ ++### [ ] Phase 3: Add a Safe Dynamic-Topic Preview and Canary ++ ++ ++ ++* [ ] Change `--dry-run` from an early exit into a non-mutating proposed-change report, or add an equivalent preview command ++* [ ] Test that preview reads candidates but writes no hub, registry, weekly frontmatter, taxonomy, or log changes ++* [ ] Review the five currently eligible candidates and choose one unambiguous canary ++* [ ] Add all non-canary candidates to `ignore_topics` as explicit temporary deferrals ++* [ ] Generate and review the exact canary transaction in an isolated checkout ++* [ ] Validate sanitization, structured YAML, evidence-backed assignments, taxonomy, logging, rendering, and disabled rollback ++* [ ] Obtain Hermes and sponsor approval for the exact canary revision ++ ++Success: one bounded candidate can be promoted without exposing all eligible candidates to the same transaction. ++ ++### [ ] Phase 4: Preflight Repository Regeneration ++ ++ ++ ++* [ ] Enable the existing repository config only in an isolated checkout at the unchanged recurrence threshold ++* [ ] Run enabled `--check`, then two full generations ++* [ ] Review every created, rewritten, obsolete, and expired path ++* [ ] Require byte-stable second generation and no unapproved removals ++* [ ] Run Hugo, pinned Pagefind, rendered metadata, internal links, axe, Lighthouse, and the cost experiment ++* [ ] Obtain Hermes, URL, and sponsor approval for the exact activation revision ++ ++Success: the first production run is a reviewed 263-page regeneration transaction with known cost and rollback. ++ ++### [ ] Phase 5: Execute and Observe Rollouts ++ ++ ++ ++* [ ] Enable only the separately approved flag and run one publish transaction ++* [ ] Inspect the committed generated-state diff before deployment ++* [ ] Confirm production rendering, lifecycle, telemetry, and downstream smoke ++* [ ] For rollback, disable the flag and revert the generated transaction; disabling alone does not undo durable mutations ++* [ ] Expand dynamic candidates one reviewed item at a time ++* [ ] Add blocking budgets only after the report-only observation window and explicit owner approval ++ ++Success: each rollout is independently approved, observable, and reversible. ++ ++## Validation Commands ++ ++* `python -m pytest tests/test_observatory_repos.py tests/test_topic_hubs.py tests/test_taxonomy_registry.py` ++* `python scripts/discover_topic_candidates.py --check` ++* `python scripts/generate_data_pages.py --check` ++* `python scripts/export_observatory_dataset.py --check` ++* `python scripts/export_trend_explorer_data.py --check` ++* `hugo --minify` ++* `npx "pagefind@1.5.2" --site public/` ++* `python scripts/check_internal_links.py public --base-url "https://claracle.com/"` ++* `python -m pytest tests/` ++* `ruff check .` ++* `ruff format --check .` ++ ++Workflow changes also require Zizmor and Checkov. Browser and Lighthouse validation may run in GitHub CI when local system dependencies are unavailable. ++ ++## Dependencies ++ ++* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md ++* docs/review/data-observatory-relaunch/owner-action-register.md ++* Stable identity decision ++* Hermes security disposition ++* URL workflow review ++* Separate jmservera sponsor decisions +diff --git a/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md +new file mode 100644 +index 0000000..f9de48d +--- /dev/null ++++ b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md +@@ -0,0 +1,76 @@ ++--- ++applyTo: '.copilot-tracking/changes/2026-08-02/claracle-relaunch-followup-execution-changes.md' ++--- ++ ++# Implementation Plan: Claracle Relaunch Follow-Up Execution ++ ++## User Requests ++ ++* Publish review corrections: commit, push, and resolve PR threads ++* Complete GA4/GSC setup ++* Close security, accessibility, Podcaster, visual, and sponsor acceptance gates ++* Plan repository-page, dynamic-topic, and generation-cost work ++ ++## Context Summary ++ ++Research shows that the repository implementation is ahead of the acceptance record. The remaining work mixes executable repository evidence with actions that require Google credentials, protected environment policy, downstream authorization, assistive technology, security review, visual review, and sponsor authority. ++ ++## Implementation Checklist ++ ++### [x] Phase 1: Publish Review Corrections ++ ++ ++ ++* [x] Commit the issue-state correction and RPI logs as `8fddceb` ++* [x] Push the active PR branch ++* [x] Resolve both addressed PR #647 review threads ++ ++### [ ] Phase 2: Reconcile GA4/GSC Evidence ++ ++ ++ ++* [x] Verify production GA configuration presence, GSC metadata absence, sitemap response, and secret names without exposing values ++* [x] Correct the baseline and status of record to distinguish deployed wiring from external acceptance ++* [x] Clarify that production GA configuration is injected through an Actions secret ++* [x] Complete Google property verification, sitemap submission, Realtime confirmation, and product link (owner-confirmed by jmservera on 2026-08-02) ++* [ ] Transcribe the supplied GSC performance export and retain production consent observations ++ ++### [x] Phase 3: Refresh Acceptance Evidence ++ ++ ++ ++* [x] Reconcile the security review with current SEC-01 and SEC-04 implementation evidence without granting sign-off ++* [x] Record current CI, environment, Podcaster, accessibility, and visual evidence boundaries ++* [x] Correct #622 to non-blocking polish and keep #626 as independent quality hardening ++* [x] Provide owner-ready records for manual accessibility, protected Podcaster, visual, security, and sponsor decisions ++ ++### [x] Phase 4: Plan Gated Rollouts and Cost Measurement ++ ++ ++ ++* [x] Plan a report-only workload-variant experiment for Q-01/NFR-009 ++* [x] Plan repository-page activation with identity, lifecycle, diff, and rollback gates ++* [x] Plan one reviewed dynamic-topic canary with explicit deferrals ++* [x] Keep both production rollout flags disabled pending approval ++ ++### [x] Phase 5: Validate and Review ++ ++ ++ ++* [x] Run focused and full executable validation and inspect PR checks ++* [x] Record completed work, owner-gated blockers, and final review disposition ++ ++## Dependencies ++ ++* Research documents under .copilot-tracking/research/2026-08-02/ and .copilot-tracking/research/subagents/2026-08-02/ ++* GitHub repository and Actions metadata access ++* Google account access for FR-035 external acceptance ++* Podcaster maintainer authorization and protected environment policy ++* Hermes, URL, Amy, Fry, and jmservera review authority ++ ++## Success Criteria ++ ++* Review corrections are published and review threads resolved. ++* Repository records accurately separate the completed GA4/GSC connection from pending baseline and production consent evidence. ++* Every acceptance gate has current evidence, a concrete owner action, and no unsupported completion claim. ++* Rollout and cost work is implementation-ready while both flags remain disabled. +diff --git a/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md +new file mode 100644 +index 0000000..f84a4e3 +--- /dev/null ++++ b/.copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md +@@ -0,0 +1,120 @@ ++--- ++applyTo: '.copilot-tracking/changes/2026-08-02/claracle-relaunch-readiness-reconciliation-changes.md' ++--- ++ ++# Implementation Plan: Claracle Relaunch Readiness Reconciliation ++ ++## Overview ++ ++Triage the live deploy failure, reconcile the three relaunch plans and the PRD/BRD with the delivered repository state, and produce a single sequenced launch-gate register so the relaunch decision is grounded in accurate, current evidence. ++ ++## Objectives ++ ++### User Requirements ++ ++* Review the plans, PRD, and BRD and address what is missing - Source: user request, 2026-08-02 ++* Produce implementation-ready planning artifacts from the gap analysis - Source: task-plan prompt, 2026-08-02 ++ ++### Derived Objectives ++ ++* Resolve the live deploy failure before reconciling status so the record reflects a green pipeline - Derived from: open issue #644 ++* Bring the three overlapping plan checklists to the true delivered state and collapse them into one status-of-record - Derived from: stale checkboxes across the 2026-07-29/30/31 plans ++* Make the PRD and BRD internally consistent and current with the #627-#646 workstream - Derived from: BRD version drift and PRD changelog lag ++* Consolidate the remaining launch gates into one owner/evidence register rather than leaving them scattered - Derived from: pending NFR-004/005/007, Podcaster run, visuals, and sponsor approval ++ ++## Context Summary ++ ++### Project Files ++ ++* docs/prds/claracle-data-observatory-relaunch.md - PRD v1.2; changelog, NFRs, rollout flags, open questions ++* docs/brds/claracle-data-observatory-relaunch-brd.md - BRD v1.1 with v1.0 cross-references (drift) ++* .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md - Phases 7-10 partial ++* .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md - Phases 6-8 open ++* .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md - Phases 2-5 open; Phase 4 shipped but unmarked ++* config/observatory.toml - repo_pages flag disabled (confirmed) ++* hugo.toml - fork-safe GA4/GSC defaults are empty; production configuration and platform acceptance require separate evidence ++* docs/review/data-observatory-relaunch/ - bounded acceptance evidence and pending gates ++ ++### References ++ ++* .copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md - gap analysis and verified findings ++* Issue #644 - live deploy failure (run 30718600607) ++* Issue #599 - Connect GA4 + Google Search Console (FR-035) ++* Issue #594 - Epic: Claracle Data Observatory Relaunch ++ ++### Standards References ++ ++* .github/copilot-instructions.md - testing, workflow security, cross-repository conventions ++* .github/instructions/hve-core/markdown.instructions.md - Markdown requirements ++* .github/instructions/hve-core/writing-style.instructions.md - documentation voice and style ++ ++## Implementation Checklist ++ ++### [x] Implementation Phase 1: Triage the Live Deploy Failure (#644) ++ ++ ++ ++* [x] Step 1.1: Diagnose deploy run 30718600607 and classify the failure — dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) broke the Podcaster smoke gate (class b/dangling reference) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 12-30) ++* [x] Step 1.2: Apply or plan the fix and confirm a green deploy — resolved by already-merged `#645`/`#646`; deploy-site green since 2026-08-01 (runs 30720064394, 30721575540) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 31-43) ++* [x] Step 1.3: Close #644 with the root-cause note — #644 already CLOSED (COMPLETED) ++ ++### [x] Implementation Phase 2: Reconcile Plan Checklists to Delivered State ++ ++ ++ ++* [x] Step 2.1: Update the 2026-07-31 deploy-hydration plan checkboxes to match shipped PRs (#628/#632/#634/#637/#641) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 55-67) ++* [x] Step 2.2: Update the 2026-07-29 and 2026-07-30 plan checkboxes for completed Podcaster/smoke/restore work (#639/#643/#640/#646) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 68-81) ++* [x] Step 2.3: Create one status-of-record reconciling delivered vs pending across all three plans, including the final dispositions of epic issues #644/#626/#622/#599/#594 — docs/review/data-observatory-relaunch/status-of-record.md ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 82-94) ++ ++### [x] Implementation Phase 3: Reconcile Product Documents ++ ++ ++ ++* [x] Step 3.1: Fix BRD version drift (Document Control vs Acceptance/PRD v1.0 references) — BRD bumped to v1.2 with a Change History; PRD cross-reference aligned to v1.2 ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 99-112) ++* [x] Step 3.2: Add PRD v1.3 changelog + restore-consistency behavior for the #627-#646 workstream, and record the FR-041 link-check partial status (test-level only, no CI link tool) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 113-125) ++* [x] Step 3.3: Add a sponsor-approval + launch-gate register with owners and evidence links — register in the status-of-record, linked from PRD Acceptance Status, section 13, and REF-10 ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 126-139) ++ ++### [x] Implementation Phase 4: Sequence Remaining Launch Gates ++ ++ ++ ++* [x] Step 4.1: Consolidate pending gates (GA4/GSC baseline and consent, external metadata and feed validation, NFR-004 security, NFR-005 a11y, Podcaster run, visuals, Q-01 cost) plus epic issues #626 (Lighthouse) and #622 (UX polish) into the register with owner, dependency, and evidence path — status-of-record launch-gate register ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 144-164) ++* [x] Step 4.2: Record deferred scope requiring its own plan (GA4/GSC baseline and consent evidence, repo_pages rollout, dynamic topic rollout, cost spike) — planning log Suggested Follow-On Work (WI-01/03/04/05) ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 165-181) ++ ++### [x] Implementation Phase 5: Validation and Re-Review ++ ++ ++ ++* [x] Step 5.1: Validate all edited docs (markdown lint, internal link/reference integrity, changelog/version consistency) — `test_internal_link_checker`/`test_embed_sources` green; referenced files verified; no stale version strings ++ * Details: .copilot-tracking/details/2026-08-02/claracle-relaunch-readiness-reconciliation-details.md (Lines 182-192) ++* [x] Step 5.2: Fix minor validation issues — none required ++* [x] Step 5.3: Report blocking issues and hand off deferred plans — no blockers; #644 resolved; deferred plans in the log (WI-01/03/04/05) ++ ++## Planning Log ++ ++See .copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md for discrepancy tracking, implementation paths considered, and suggested follow-on work. ++ ++## Dependencies ++ ++* gh CLI access for #644 diagnosis and issue updates ++* Repository write access for docs/ and .copilot-tracking/ edits ++* Markdown lint tooling per .mega-linter.yml ++* Sponsor (jmservera) input for the approval artifact and gate ownership ++ ++## Success Criteria ++ ++* Issue #644 is diagnosed and either fixed with a green deploy or handed off with a scoped fix plan - Traces to: open issue #644 ++* All three relaunch plans reflect true delivered state and a single status-of-record exists - Traces to: stale checkbox findings ++* PRD and BRD are internally consistent, version-correct, and current through the #627-#646 workstream - Traces to: BRD version drift and PRD changelog lag ++* A single launch-gate register lists every pending gate with owner, dependency, and evidence path - Traces to: pending NFR-004/005/007, Podcaster, visuals, sponsor approval ++* Deferred implementation workstreams are captured for separate planning - Traces to: #599, repo_pages/dynamic rollout, Q-01 +diff --git a/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md +new file mode 100644 +index 0000000..3063bd7 +--- /dev/null ++++ b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-followup-execution-log.md +@@ -0,0 +1,42 @@ ++ ++# Planning Log: Claracle Relaunch Follow-Up Execution ++ ++## Selected Path ++ ++Execute repository-verifiable work and prepare owner-ready handoffs. Do not hardcode analytics identifiers, configure Google properties without account access, trigger duplicate-prone podcast generation, grant human sign-off, or enable rollout flags. ++ ++## Discrepancies ++ ++* The empty checked-in GA default is fork-safe configuration, not proof that production GA4 is disconnected. ++* Issue #599 closed after agent-side wiring, while its human-action checklist remains incomplete. ++* Issue #622 calls itself non-blocking polish; it must not be represented as a mandatory launch gate without a sponsor decision. ++* A real Podcaster run and an environment-bound dry run exist, but no run combines both properties. ++* Hugo/Pagefind timing separation is shipped; Q-01 requires workload attribution and retained statistics rather than another timer split. ++* `discover_topic_candidates.py --check` fails against the inherited registry at commit `8fddceb`. A temporary regeneration retains 2,173 total candidates and the same five eligible candidates, while rotating four sanitized keys. This branch does not rewrite publish-derived state; refresh it through the owning generation workflow. ++ ++## Deferred Owner Actions ++ ++* jmservera: GA4/GSC connection actions completed; GSC export transcription, production consent observations, processed sitemap review, and rollout decisions remain ++* Hermes: SEC-01 through SEC-06 dispositions and NFR-004 sign-off ++* URL: protected environment and secret-scope review ++* Podcaster maintainer: idempotency or one-run authorization ++* Fry and accessibility reviewer: manual keyboard and screen-reader record ++* Amy: final visual matrix and acceptance conclusion ++ ++## Repository-Executable Security Closure ++ ++* SEC-02 implementation evidence is complete: official snippets use no-referrer and iframe analytics ++ requires explicit frame-local Claracle consent. Hermes privacy disposition remains pending. ++* SEC-03 implementation evidence is complete: exact public export and safe source-path allowlists are ++ enforced by production code and tests. Hermes field-policy approval remains pending. ++* SEC-05 has an explicit defense-in-depth recommendation with retained executable controls and stated ++ semantic limitations. No accepted-risk decision has been recorded. ++* Squad agent implemented SEC-02/03/05. Fry rejected the first SEC-02 browser assertion, Hermes tightened the cross-origin default-off proof, and Fry approved the revised executable closure. This is quality approval, not Hermes security sign-off. ++ ++## Safety Decisions ++ ++* Keep `GA_MEASUREMENT_ID` and `GSC_SITE_VERIFICATION` values out of source and evidence. ++* Keep `repo_pages.enabled` and `topic_hubs.dynamic_creation.enabled` false. ++* Keep quality thresholds unchanged. ++* Keep cost thresholds report-only until approved. ++* Do not dispatch real downstream generation during planning. +diff --git a/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md +new file mode 100644 +index 0000000..05d2a8d +--- /dev/null ++++ b/.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md +@@ -0,0 +1,74 @@ ++ ++# Planning Log: Claracle Relaunch Readiness Reconciliation ++ ++## Discrepancy Log ++ ++Gaps and differences identified between research findings and the implementation plan. ++ ++### Unaddressed Research Items ++ ++* DR-01: Full GA4/GSC connection (setting ga_measurement_id, verifying the property, submitting the sitemap, capturing the baseline) ++ * Source: research 2026-08-02 (Unmet launch gates); issue #599 ++ * Reason: Issue #599 closed as completed on 2026-08-01 with a human-action checklist still outstanding; this plan only registers and sequences that external acceptance work ++ * Impact: high (blocks OBJ-2/OBJ-4 baselines) ++* DR-02: repo_pages and dynamic_topic_creation rollout enablement ++ * Source: config/observatory.toml flags disabled; PRD section 13 ++ * Reason: Requires separate sponsor approval and its own rollout plan ++ * Impact: medium (Wave 2/dynamic scope not live) ++* DR-03: Incremental generation cost/time design spike (Q-01/NFR-009) ++ * Source: PRD section 14; BRD section 13 ++ * Reason: Needs a measurement spike, not documentation reconciliation ++ * Impact: medium (capacity risk unquantified) ++* DR-04: Open issues #626 (Lighthouse follow-ups) and #622 (UX polish) reconciliation - RESOLVED 2026-08-02 ++ * Source: research 2026-08-02 (Open issues); issues #626, #622 ++ * Resolution: Plan Step 2.3 now lists #644/#626/#622/#599/#594 in the status-of-record; Step 4.1 adds #626 and #622 to the gate register with a disposition (readiness scope or explicit out-of-scope). Details Step 2.3 (status-of-record scope) and Step 4.1 (gate list) both enumerate #626/#622. ++ * Impact: closed (readiness view and gate register now cover the epic's open work) ++* DR-05: FR-041 internal link-checker partial status not reconciled in the PRD/status-of-record - RESOLVED 2026-08-02 ++ * Source: research 2026-08-02 (Verified findings, FR-041) ++ * Resolution: Plan Step 3.2 now records the FR-041 link-check partial status (test-level only, no CI link tool); Details Step 3.2 success criteria captures the same partial-satisfaction statement. ++ * Impact: closed (FR-041 traceability now explicitly marked partial) ++ ++### Plan Deviations from Research ++ ++* DD-01: External/human launch gates (security sign-off, accessibility, Podcaster run, visuals, sponsor approval) are registered and sequenced rather than executed ++ * Research recommends: close the gates ++ * Plan implements: consolidate into one owner/evidence register with sequencing ++ * Rationale: these gates depend on humans and external platforms outside a planning/documentation change; execution belongs to their owners with dated evidence ++ ++## Implementation Paths Considered ++ ++### Selected: Single reconciliation plan (docs + status-of-record) with #644 triage first ++ ++* Approach: triage the live deploy failure, correct the three plan checklists, reconcile PRD/BRD, and consolidate a launch-gate register; defer external gate execution ++* Rationale: directly answers "what's missing" with accurate state and a single readiness view; low-risk and mostly documentation ++* Evidence: research 2026-08-02 (Planning approach) ++ ++### IP-01: One mega-plan that also implements every launch gate ++ ++* Approach: fold GA4/GSC, security, a11y, Podcaster run, and rollout into one plan ++* Trade-offs: comprehensive but mixes documentation with external/human execution; long-lived and hard to validate ++* Rejection rationale: each external gate merits its own plan and owner; a mega-plan would stall on human dependencies ++ ++### IP-02: Skip planning and edit docs directly ++ ++* Approach: immediately edit PRD/BRD/plans ++* Trade-offs: faster but loses traceability and the #644 dependency ordering ++* Rejection rationale: the reconciliation touches multiple documents and a live blocker; a checklist keeps it ordered and reviewable ++ ++## Suggested Follow-On Work ++ ++* WI-01: GA4 + GSC connection implementation plan (high) - set ga_measurement_id, verify property, submit sitemap, capture dated baseline ++ * Source: research (GA4/GSC); human-action checklist on closed issue #599 ++ * Dependency: sponsor/platform access ++* WI-02: Deploy failure #644 dedicated fix plan (high) - NOT NEEDED. #644 is CLOSED: root cause was a dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) breaking the Podcaster smoke gate; resolved by `#645`/`#646`, deploy-site green since 2026-08-01. No dedicated fix plan required. ++ * Source: open issue #644 (now closed) ++ * Dependency: none ++* WI-03: repo_pages rollout plan (medium) - enable flag, lifecycle acceptance, sponsor approval ++ * Source: config/observatory.toml; PRD FR-020-022 ++ * Dependency: sponsor approval, security/lifecycle evidence ++* WI-04: Dynamic topic-creation rollout plan (medium) ++ * Source: PRD FR-004 ++ * Dependency: sponsor approval, security evidence ++* WI-05: Incremental-generation-cost design spike (medium) - quantify hub/data/repo build cost (Q-01/NFR-009) ++ * Source: PRD section 14; BRD section 13 ++ * Dependency: none +diff --git a/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md b/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md +new file mode 100644 +index 0000000..ea2a068 +--- /dev/null ++++ b/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md +@@ -0,0 +1,69 @@ ++ ++# Research: Durable restore-consistency fix (#640) ++ ++## Scope ++ ++`run_mode=restore` (intended to refresh observatory surfaces from stored raw ++evidence) also regenerates and republishes the week's weekly article. The ++`sync-publish-to-main.yml` job wipes and re-syncs `content/weekly/` from ++`publish` to `main`, so a restore overwrites the original published article ++with a non-deterministic LLM regeneration. The same restore also rewrote the ++promotion record to reference a candidate manifest that was never persisted, ++breaking the Podcaster handoff smoke (remediated reactively for 2026-W31). ++ ++## Success criteria (from #640) ++ ++- A restore intended to refresh `content/data` does not change any already-published `content/weekly/*` article. ++- Topic frontmatter on prior weeks is preserved through restore+sync. ++- Regression test covers the chosen behavior. ++ ++## Evidence log ++ ++- `.github/workflows/crawl-and-publish.yml` ++ - `generate` job: hydrates prior state from `publish`, downloads analyzed + ++ candidate + raw artifacts, then `Generate weekly content candidate` → ++ `Rehash and promote final weekly content` (rewrites `content/weekly//W.md`), ++ then observatory steps, then `Commit generated content to data branch`. ++ - Commit step resets working tree to `origin/publish` (`git checkout -f -B publish origin/publish`), ++ extracts regenerated files (`tar -xf generated-state.tar`), stages ++ `GENERATED_PATHS`, and force-pushes to `publish`. ++ - The commit step copies the current-branch tooling to `publish-safety-tool.py` ++ BEFORE the reset, so a current-version helper is available after the reset. ++- `.github/workflows/sync-publish-to-main.yml`: `rm -rf ... content/weekly ...` ++ then `git checkout origin/publish -- content/weekly/ ...` → publish is canonical. ++- `scripts/rerun_modes.py`: mode validation; restore action = "restore published ++ artifacts for and regenerate through guarded promotion". ++- Published weekly transaction = 3 files: `content/weekly//W.md`, ++ `data/analyzed/-summary.md`, `data/published//promotion-manifest.json`. ++- `scripts/podcaster_handoff.py` (`promotion_transaction_v1`) verifies the ++ promotion record's `source_manifest.path` bytes → dangling references break it. ++ ++## Alternatives evaluated ++ ++- **A — Data-only restore mode (gate article steps):** skip article/promotion ++ steps entirely in restore. Cleanest semantically but touches many interdependent ++ steps in a high-risk force-pushing workflow; hard to test end-to-end. ++- **B — Preserve published transaction (SELECTED):** after regeneration and the ++ branch reset, in restore mode revert the 3 published-transaction files to their ++ `publish` versions (`git checkout HEAD -- ` where HEAD == origin/publish) ++ before staging. Observatory surfaces stay regenerated. The commit — and thus the ++ sync — leave the published article, summary, and promotion record byte-identical. ++ Minimal, testable, and also prevents the dangling-manifest class. ++- **C — Guard the sync:** sync has no knowledge of run_mode; detecting a ++ restore-driven rewrite there is fragile. Rejected. ++ ++## Selected approach ++ ++Option B. Add a small unit-tested helper `weekly_transaction_paths(week)` + ++`weekly-transaction-paths` CLI to `scripts/publish_safety.py`. In the commit ++step, gated on `run_mode == restore`, iterate those paths and ++`git checkout HEAD -- ` for each that exists on `publish`, before staging. ++Update `rerun_modes.py` restore action wording. Add tests. ++ ++## Next steps ++ ++1. Helper + CLI in `publish_safety.py`. ++2. Gated preservation block in the commit step (pass `RUN_MODE` env). ++3. Update `rerun_modes.py` restore action string. ++4. Tests: `test_publish_safety.py` (paths), `test_pipeline.py` (workflow guard). ++5. Validate: pytest, ruff, zizmor, yaml. +diff --git a/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md b/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md +new file mode 100644 +index 0000000..d6a9f19 +--- /dev/null ++++ b/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md +@@ -0,0 +1,46 @@ ++ ++# Research: Claracle Relaunch Follow-Up Execution ++ ++## Scope ++ ++Execute all four follow-up items from the relaunch reconciliation: publish review corrections, complete repository-side GA4/GSC work, close executable acceptance evidence gaps, and plan gated rollouts plus incremental cost measurement. ++ ++## Source Research ++ ++* .copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md ++* .copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md ++* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md ++ ++## Verified Findings ++ ++* Review correction commit `8fddceb` is pushed to PR #647 and both review threads are resolved. ++* Production renders GA configuration, the `GA_MEASUREMENT_ID` secret name exists, and the sitemap returns HTTP 200 with `application/xml`. ++* Production does not render GSC verification metadata and no `GSC_SITE_VERIFICATION` secret name exists. ++* Repository-side GA4/GSC wiring, consent gating, GSC metadata rendering, workflow injection, and tests already exist. Google account verification and baseline capture require jmservera. ++* Focused acceptance research passed 102 security tests plus 42 UX/lifecycle tests; six Hugo-dependent tests skipped locally. ++* Real Podcaster generation and environment-bound smoke evidence exist separately. No run combines real generation with a protected environment, and `podcaster-release-smoke` has no protection rules. ++* Issue #622 explicitly classifies its work as non-blocking polish. Issue #626 is independent hardening and forbids lowering quality thresholds. ++* Both rollout flags remain disabled. Repository activation lacks stable production GitHub IDs and lifecycle-transition evidence. Dynamic activation has five eligible candidates and no useful preview-only dry run. ++* Hugo and Pagefind timing are already separated in CI, but Q-01 lacks workload variants, retained comparable samples, aggregation, and an approved budget. ++ ++## Selected Approach ++ ++1. Preserve the pushed review correction and verify PR checks. ++2. Correct GA4/GSC evidence to distinguish fork-safe checked-in defaults from observed production configuration. ++3. Refresh the acceptance package with current automated evidence and implementation dispositions without granting human sign-off. ++4. Create implementation-ready owner checklists for external acceptance, protected real Podcaster execution, rollout canaries, and report-only cost measurement. ++5. Keep both rollout flags disabled and do not trigger external side effects. ++ ++## External Boundaries ++ ++* Google property inspection, GSC token retrieval, ownership verification, sitemap submission, Realtime confirmation, and baseline values require jmservera or delegated Google access. ++* Environment protection configuration requires repository administration and named reviewer policy. ++* Real podcast generation requires Podcaster maintainer authorization because duplicate suppression is not evidenced in this repository. ++* Manual keyboard, screen-reader, visual, security, and sponsor conclusions require named human reviewers. ++ ++## Validation ++ ++* Focused Python tests for workflow mapping, links, security, lifecycle, and exports ++* Public presence-only production probes that do not print identifiers or tokens ++* GitHub PR status checks and environment metadata ++* Documentation diagnostics, stale-claim scans, and `git diff --check` +diff --git a/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md b/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md +new file mode 100644 +index 0000000..eb6cad1 +--- /dev/null ++++ b/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md +@@ -0,0 +1,61 @@ ++ ++# Research: Claracle Relaunch Readiness Reconciliation (gap analysis) ++ ++## Scope ++ ++Review of the relaunch plans, PRD, and BRD against actual repository state to ++identify what is missing and plan its closure. Verified via repository inspection ++on 2026-08-02. ++ ++## Source documents ++ ++- BRD: docs/brds/claracle-data-observatory-relaunch-brd.md (BRD-CLARACLE-002, Document Control says v1.1) ++- PRD: docs/prds/claracle-data-observatory-relaunch.md (v1.2, 2026-07-31) ++- Plans: ++ - .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md (Phases 1-6 done; 7-10 partial/open) ++ - .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md (Phases 1-5 done; 6-8 open) ++ - .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md (Phase 1 done; 2-5 open) ++ ++## Verified findings (evidence) ++ ++- Initial blocker: issue #644 "Deploy Hugo site failed (run 30718600607)" was the incident under investigation. Final verification found it closed as completed on 2026-08-01 after #645/#646 restored a green deploy. ++- Initial repository-only observation: hugo.toml uses an empty fork-safe GA4 default. Follow-up production verification found secret-backed GA configuration present, while issue #599's recorded GSC, platform-receipt, and baseline actions remain outstanding (FR-035/DR-002/NFR-007). Growth KPI baselines (OBJ-2/3/4, G-002/003/004) were not captured. ++- Repo pages gated off: config/observatory.toml `[repo_pages] enabled = false` and `[repo_pages.lifecycle] enabled = false` (FR-020-022 not live; matches PRD flag `repo_pages`). ++- Internal link checking exists as tests/test_internal_link_checker.py (FR-041 partially satisfied at test level, not a separate CI link tool). ++- Issue dispositions: #644 and #599 closed as completed on 2026-08-01; #626 (Lighthouse follow-ups), #622 (UX polish), and #594 (Epic) remained open at final verification. Closing #599 did not complete its human-action checklist. ++ ++## Session work NOT reflected in docs ++ ++- Deploy/hydration cascade #627-#637; Podcaster smoke #639/#643; restore-consistency #640/#646 all merged but absent from the PRD changelog and plan checkboxes. ++- Deploy-hydration plan Phase 4 (embed source_page guard) shipped as check_embed_sources.py (#641) but left unmarked. ++ ++## Document/consistency gaps ++ ++- BRD version drift: Document Control = v1.1 but Acceptance section + PRD REF-1 cite BRD v1.0. ++- No single status-of-record reconciling the three overlapping plans; checkboxes are stale. ++- PRD changelog behind reality (no v1.3 for #627-#646 workstream; NFR-002 restore sub-behavior undocumented). ++- No sponsor-approval artifact exists (BRD notes none recorded); both rollout flags cannot flip without it. ++- DR-002 dated baseline snapshot never captured -> success currently unmeasurable. ++ ++## Unmet launch gates (PRD/BRD) ++ ++- NFR-004 Hermes security sign-off (pending) ++- NFR-005 accessibility evidence (pending) ++- NFR-002 / R-04 real Podcaster downstream run evidence (pending) ++- Refreshed visual acceptance (pending) ++- Q-01 / NFR-009 incremental generation cost/time (still TBD) ++- Sponsor approval to enable dynamic_topic_creation + repo_pages ++ ++## Planning approach ++ ++Single reconciliation plan: ++1. Triage/resolve #644 (live blocker) first. ++2. Reconcile the three plan checklists to delivered state + produce one status-of-record. ++3. Reconcile product docs (PRD v1.3 changelog + restore NFR; fix BRD version drift; add sponsor-approval + launch-gate register). ++4. Sequence remaining launch gates into one owner/evidence register (do not fully implement external/human gates here). ++5. Validate docs (markdown lint, link integrity) + re-review. ++ ++Deferred to separate plans (out of scope here): ++- GA4/GSC connection implementation (FR-035; continue the human-action checklist on closed issue #599) ++- repo_pages rollout + dynamic topic rollout (require sponsor approval) ++- Incremental-generation-cost design spike (Q-01) +diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md +new file mode 100644 +index 0000000..6a7c338 +--- /dev/null ++++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md +@@ -0,0 +1,320 @@ ++ ++# Claracle Acceptance Gates Follow-Up Research ++ ++## Research Scope ++ ++Assess current evidence and executable work for: ++ ++* NFR-004 security sign-off ++* NFR-005 accessibility ++* A real protected Podcaster downstream run ++* Refreshed visual acceptance ++* Lighthouse issue #626 ++* UX issue #622 ++* Sponsor approval ++ ++Separate checks that can run locally or through GitHub now from actions that require protected environments or human approval. Inspect the relaunch review package, relevant tests, workflows, scripts, screenshots, PRD, BRD, and current repository patterns. ++ ++## Working Hypothesis ++ ++The status-of-record identifies the correct pending gates, but most acceptance evidence remains either stale, test-level only, or dependent on protected GitHub environments and named human approvers. Nearby scripts and workflows may provide executable checks without satisfying the final approval gates. ++ ++## Evidence Inventory ++ ++| Surface | Current evidence | What it proves | Boundary | ++| --- | --- | --- | --- | ++| Acceptance status | docs/review/data-observatory-relaunch/README.md and status-of-record.md both say release acceptance is pending | The seven requested gates are launch-readiness scope and both rollout flags must remain disabled | These records do not supply the missing approvals | ++| Product requirements | docs/prds/claracle-data-observatory-relaunch.md v1.3 and docs/brds/claracle-data-observatory-relaunch-brd.md v1.2 | NFR-004 is Must, NFR-005 is Should, NFR-001 requires Lighthouse Performance >= 90, and each rollout flag needs separate sponsor approval | PRD and BRD record intent, not execution | ++| Automated browser quality | .github/workflows/ci.yml production-site job | CI installs pinned Playwright 1.54.2, axe-playwright 4.10.2, and Lighthouse 12.8.2; runs responsive, axe, analytics, and Lighthouse checks; uploads reports for 30 days | The server is a local production build, not the public production origin | ++| Current GitHub CI | CI run 30723119836 succeeded for the reconciliation branch; run 30742507113 was in progress when checked on 2026-08-02 | The branch has a recent complete green baseline and a current GitHub check is available | The current in-progress run must finish before its revision is cited as green | ++| Lighthouse implementation | scripts/design/lighthouse-gates.mjs | Nine routes run mobile Lighthouse three times; medians must meet performance 0.90, accessibility 0.95, best practices 0.95, and CLS <= 0.1 | This does not complete the five open improvements in issue #626 | ++| Accessibility implementation | tests/visual/observatory-a11y.spec.mjs and tests/visual/a11y-perf.spec.mjs | Serious and critical WCAG 2.1 A/AA axe findings fail; keyboard focus, modal focus handling, chart alternatives, overflow, and 44 px targets receive automated coverage | No retained production-origin audit, manual keyboard record, or screen-reader findings exist | ++| Protected dry-run smoke | Deploy run 30721575540, job 91426194097 | The post-deploy reusable smoke succeeded on 2026-08-01 using exact retained promotion evidence and `--podcaster-dry-run` | It did not create a real downstream episode | ++| Real downstream run | Trigger Podcast run 30202586031 | Week 2026-W30 used manifest run 29744859230 and Podcaster returned `status=accepted` with a retained job ID on 2026-07-26 | .github/workflows/trigger-podcast.yml declares no GitHub environment | ++| GitHub environments | GitHub API response on 2026-08-02 | `podcaster-release-smoke` exists, but has no protection rules or deployment branch policy | A named environment is not evidence of protected approval controls | ++| Visual evidence | Ten PNGs under docs/review/data-observatory-relaunch/screenshots plus its README | Historical desktop captures show the intended surfaces and the defects cited by issue #622 | They lack revision, viewport, theme, interaction metadata, mobile/dark variants, populated topic membership, and unobscured states | ++| Security review | docs/review/data-observatory-relaunch/security-review.md | SEC-01 through SEC-06 and three signatories define the acceptance boundary | The review is stale against some current controls and all sign-off rows remain Pending | ++ ++## Gate Findings ++ ++### NFR-004 Security Sign-Off ++ ++Status: blocked on review reconciliation and named human dispositions, not on a single test command. ++ ++* Hermes has not dispositioned SEC-01 through SEC-06; URL and jmservera sign-off rows are also Pending. ++* SEC-01 is stale as written. scripts/manage_topic_hubs.py now applies `sanitize_text`, rejects line breaks, boundary markers, HTML, Markdown, control characters, and injection phrases. tests/test_topic_hubs.py includes adversarial cases for those payloads. ++* SEC-02 remains substantively open. layouts/partials/visuals/observatory-chart.html emits an iframe snippet without `referrerpolicy`, and standalone embeds can load the common consent-gated analytics adapter. A privacy decision and test are still required. ++* SEC-03 is partially implemented. scripts/export_observatory_dataset.py centralizes a fixed `CSV_COLUMNS` list and emits a public-exposure statement, while the ADR requires review for new fields. There is no separately approved field policy or automated assertion that the published fields equal that policy. ++* SEC-04 has strong executable evidence in tests/test_observatory_repos.py for rename, archive, confirmed deletion, retention, and expiry. Hermes still must accept the operator-override policy. ++* SEC-05 is an explicit accepted-risk decision for phrase-based semantic false negatives. Tests can verify defense in depth, but only Hermes can record the risk disposition. ++* SEC-06 requires production analytics evidence and protected Podcaster secret-scope evidence. Repository tests cannot close it. ++ ++### NFR-005 Accessibility ++ ++Status: automated implementation exists; acceptance evidence is incomplete. ++ ++* CI already runs axe against topic, data, repository, chart, and tool routes and rejects serious or critical WCAG 2.1 A/AA violations. ++* Existing Playwright tests cover labels, visible focus, keyboard operation, consent-dialog focus trapping and restoration, chart text alternatives, responsive overflow, and touch target size. ++* The required acceptance record does not exist. It must identify revision, public production URLs, browser and assistive technology, automated results, manual keyboard findings, screen-reader findings, reviewer, date, and dispositions. ++ ++### Real Protected Podcaster Downstream Run ++ ++Status: real and environment-bound evidence each exist separately; the required combination does not. ++ ++* Run 30202586031 proves a real accepted downstream request for week 2026-W30. ++* Run 30721575540 proves a successful post-deploy dry-run through the `podcaster-release-smoke` environment. ++* The real workflow and the post-merge path in .github/workflows/sync-publish-to-main.yml do not bind to an environment. The named smoke environment currently has no protection rules. Therefore no current run proves real generation after protected-environment approval. ++ ++### Refreshed Visual Acceptance ++ ++Status: blocked on final UI/data state and human visual review. ++ ++* The historical set contains only ten PNGs and records a generated date of 2026-05-25 in the rendered footer. ++* 02-topics-index.png shows no topic hubs with issue matches; 03-topic-hub-mcp.png shows zero recent weekly issues; 08-star-velocity-tool.png shows ambiguous rounded bars; all three are obscured by the consent banner. ++* tests/visual/visual.spec.mjs covers only home, about, weekly, monthly, yearly, and cover-card baselines. It does not automate the relaunch capture matrix in docs/review/data-observatory-relaunch/screenshots/README.md. ++* Acceptance requires a replacement matrix with revision, date, browser, viewport, theme, consent state, interaction state, source week, populated content, and reviewer conclusion. ++ ++### Lighthouse Issue #626 ++ ++Status: open on GitHub with no assignee and five independently shippable items. ++ ++* Extend page-scoped CSS splitting to search, about, methodology, and privacy. ++* Add Brotli negotiation to scripts/serve_static.py. ++* Document the median-of-three and compressed-server methodology in docs/qa-gates.md. ++* Reserve chart space to reduce the data-page CLS margin from its cited approximately 0.040 value. ++* Parallelize per-page Lighthouse execution to reduce the roughly ten-minute production-site job. ++* Acceptance explicitly forbids lowering the current thresholds. ++ ++### UX Issue #622 ++ ++Status: open on GitHub with no assignee; the issue calls the work non-blocking polish, while the relaunch status-of-record includes it in the readiness register. ++ ++* Determine whether Star Velocity Explorer bars are intentionally normalized per row or incorrectly bound. Current screenshots are not sufficient to decide. ++* Verify topic index aggregation against a production-representative dataset. Historical captures show the mismatch, but current weekly files W21 through W31 contain canonical `topics` frontmatter, including MCP Ecosystem in W21 and W26. This is likely stale visual evidence unless a fresh Hugo build still renders zero matches. ++* Verify mobile consent-banner placement. Existing automated modal tests cover focus behavior but not visual content occlusion. ++ ++### Sponsor Approval ++ ++Status: human-only and absent. ++ ++* Repository search found no dated approval artifact. ++* BRD approval of the business requirements is not rollout approval. ++* jmservera must approve or reject `dynamic_topic_creation` and `repo_pages` separately, identify the reviewed evidence and revision, and date the decision. Both flags must remain disabled until then. ++ ++## Executable Checks Available Now ++ ++### Local Without Protected Access ++ ++* Pure-Python security controls, topic lifecycle, repository lifecycle, embed contracts, and trend-tool behavior run through the existing `.venv` with `uv run --no-sync`. ++* The focused UX/lifecycle run completed with 42 passed and 6 skipped. Every skip required Hugo, which is not installed on this host. ++* The focused sanitization and defense-chain run completed with 102 passed. ++* Node packages match CI: Playwright 1.54.2, axe-playwright 4.10.2, and Lighthouse 12.8.2. ++* Local Playwright execution is blocked before page navigation because the host lacks `libnspr4` and `libnss3`. Installing Playwright system dependencies requires elevated access. Lighthouse is subject to the same Chromium host dependency. ++* The checked-in `public/` tree can be served with scripts/serve_static.py, but it is not sufficient acceptance evidence unless it is rebuilt from and tied to the tested source revision. ++ ++### GitHub Without Protected Secrets Or Human Approval ++ ++* Opening or updating a PR to `main` runs .github/workflows/ci.yml. It builds Hugo, runs rendered contracts, axe, responsive checks, analytics tests, and Lighthouse, and retains production-quality reports for 30 days. ++* CI run 30742507113 had passed Python, rendered contracts, internal links, and axe/responsive gates when inspected. Lighthouse was still running. Completed run 30723119836 is the latest full green run for the same reconciliation branch lineage. ++* Issues #626 and #622 can be researched, assigned, decomposed, and implemented through normal PRs without protected environment access. ++* The dry-run smoke workflow is manually dispatchable, but it requires exact retained publish evidence and repository environment secrets. ++ ++### GitHub Actions That Cause External Effects ++ ++* .github/workflows/trigger-podcast.yml can dispatch a real generation request for an eligible week and publish run. It is not a read-only validation command. ++* No idempotency or duplicate-suppression key is visible in scripts/podcaster_handoff.py. A rerun must be approved by the Podcaster maintainer or target a deliberately authorized episode. ++* .github/workflows/deploy-site.yml can redeploy and then run the dry-run smoke. It changes production deployment state and is not needed merely to prove local code quality. ++ ++## Protected Or Human Actions ++ ++| Action | Required actor or access | Why it cannot be closed locally | ++| --- | --- | --- | ++| Protect the Podcaster environment | Repository administrator | `podcaster-release-smoke` currently has no protection rules or branch policy | ++| Bind real generation to the protected environment | Workflow change reviewed by URL and Hermes | Current real and post-merge jobs declare no environment | ++| Authorize a real downstream target week | Podcaster maintainer | A duplicate episode may be created; this repository has no visible idempotency guard | ++| Execute and verify the real protected run | Environment approver and Podcaster maintainer | Requires secret-bearing environment and downstream access | ++| Manual accessibility review | Fry plus an accessibility reviewer with production browser and assistive technology | Automated axe and keyboard tests do not provide screen-reader findings | ++| Visual acceptance | Amy or named visual reviewer | Screenshot generation does not equal a human acceptance conclusion | ++| NFR-004 disposition | Hermes, URL, and jmservera | SEC-02, SEC-05, SEC-06 and policy acceptance require threat, workflow, and production-owner judgment | ++| Sponsor rollout decision | jmservera | Approval authority is explicitly human and must address each flag separately | ++ ++## Exact Evidence Gaps ++ ++1. NFR-004 has no dated Hermes disposition for SEC-01 through SEC-06 and no completed URL or sponsor sign-off row. The review also does not acknowledge that SEC-01 sanitization and SEC-04 lifecycle fixtures now exist. ++2. SEC-02 lacks a chosen embed privacy model, implementation evidence, and a test for referrer and cross-origin consent behavior. ++3. SEC-03 lacks an approved field-level publication policy and an automated comparison between that policy and exported fields. `CSV_COLUMNS` plus a self-authored PASS string is useful implementation evidence but not independent privacy approval. ++4. SEC-05 lacks an accepted-risk rationale for semantic prompt-injection false negatives. ++5. SEC-06 lacks dated production analytics network/cookie observations and environment-scoped Podcaster secret review. ++6. NFR-005 lacks a retained accessibility review record with public URLs, revision, browser, assistive technology, axe results, full keyboard findings, screen-reader findings, reviewer, date, and dispositions. ++7. No Actions run combines real Podcaster generation with a protected GitHub environment. Existing evidence is split between real unbound run 30202586031 and environment-bound dry-run 30721575540. ++8. The environment named `podcaster-release-smoke` has zero protection rules. Its name alone does not satisfy protected execution. ++9. The visual set lacks current revision metadata, mobile and dark captures, consent-resolved feature states, populated topic membership, complete interaction states, empty/error states, and a dated reviewer conclusion. ++10. Issue #626 remains open with all five checklist items unchecked. No issue comment or linked PR records partial completion. ++11. Issue #622 remains open. Fresh source data suggests the topic aggregation screenshot is stale, but no current rendered capture proves it; bar semantics and mobile consent occlusion remain unclassified. ++12. No dated sponsor artifact approves or rejects `dynamic_topic_creation` and `repo_pages` separately. Both are correctly still `enabled = false` in config/observatory.toml. ++13. The status-of-record calls #622 a readiness gate while the issue body calls it non-blocking polish and says the epic is accepted. A human release owner must resolve which statement controls launch acceptance. ++ ++## Suggested Implementation Sequence ++ ++1. Let CI run 30742507113 finish and retain its production-quality artifact URLs. Do not call the revision fully green until Lighthouse completes. ++2. Reconcile the security review with current code. Mark SEC-01 implementation-ready for Hermes verification, attach the adversarial test result, attach SEC-04 lifecycle test evidence, and replace stale statements without marking NFR-004 accepted. ++3. Resolve SEC-02 and SEC-03 through small independent changes: choose the embed privacy model first, then codify and test the public export field policy. Preserve disabled rollout flags. ++4. Resolve issue #622’s factual questions before visual recapture. Rebuild with current W21-W31 frontmatter, confirm topic counts, document Star Velocity normalization semantics, and test consent placement at mobile viewports. ++5. Implement issue #626 before final screenshots because CSS splitting and CLS reservation can change layout. Land documentation and Brotli support independently; preserve every threshold. Parallelize Lighthouse only after deterministic per-page artifact naming and failure aggregation are designed. ++6. Run the complete GitHub production-site job and review the uploaded axe, responsive, Lighthouse, and Playwright reports. Then perform the manual keyboard and screen-reader review against the final production revision. ++7. Capture the complete visual matrix only after #622 and layout-affecting #626 work is final. Record metadata beside every image and obtain Amy's dated acceptance conclusion. ++8. Create a protected environment for real Podcaster generation, or add real mode to a separately named protected environment. Require approved branches and the intended reviewer. Bind the real job to it and have URL/Hermes review secret scope. ++9. With Podcaster-maintainer authorization, run one real eligible week through the protected environment. Retain the Actions URL, week, manifest run ID, article SHA-256, downstream job ID, final downstream conclusion, approver, and date without recording secret values. ++10. Have Hermes, URL, and jmservera complete the security sign-off table after external evidence is attached. ++11. Resolve whether #622 is blocking, then have jmservera issue a dated sponsor decision for each rollout flag. Enabling either flag is a separate product change after approval, not part of the approval artifact itself. ++ ++## Blockers ++ ++* Hugo is absent locally, so rendered source validation cannot run on this host without installing the pinned binary. ++* Playwright's Node packages are installed, but required Chromium host libraries are absent and need elevated package installation. GitHub CI is the available executable browser path now. ++* The real Podcaster workflow is not environment-bound, and the existing named environment has no protection rules. ++* A protected real rerun can create duplicate downstream work; Podcaster maintainer authorization is required because no local idempotency contract was found. ++* Production accessibility, screen-reader, visual, security, and sponsor conclusions require named humans. ++* Issue #622 has contradictory launch semantics between its GitHub body and the status-of-record. ++ ++## Validation Commands ++ ++### Local Pure-Python Baseline ++ ++```bash ++uv run --no-sync pytest -q \ ++ tests/test_topic_hubs.py \ ++ tests/test_trend_explorer_tool.py \ ++ tests/test_observatory_repos.py \ ++ tests/test_observatory_embeds.py ++``` ++ ++```bash ++uv run --no-sync pytest -q \ ++ tests/test_sanitize_repo_content.py \ ++ tests/test_prompt_injection_redteam.py \ ++ tests/test_defense_chain_e2e.py ++``` ++ ++```bash ++uv run --no-sync pytest -q tests/test_export_observatory_dataset.py ++uv run --no-sync python scripts/export_observatory_dataset.py --check ++uv run --no-sync python scripts/export_trend_explorer_data.py --check ++``` ++ ++### Full Local Source Build When Hugo And Browser Libraries Are Available ++ ++```bash ++hugo --minify --baseURL http://127.0.0.1:1313/ ++npx pagefind@1.5.2 --site public/ ++uv run --no-sync python scripts/serve_static.py --directory public --bind 127.0.0.1 --port 1313 ++``` ++ ++Run the server in one terminal, then run: ++ ++```bash ++BASE_URL=http://127.0.0.1:1313 \ ++ npx --no-install playwright test \ ++ --config tests/visual/playwright.config.mjs \ ++ tests/visual/a11y-perf.spec.mjs \ ++ tests/visual/observatory-a11y.spec.mjs \ ++ tests/visual/observatory-analytics.spec.mjs ++``` ++ ++```bash ++node scripts/design/lighthouse-gates.mjs --base http://127.0.0.1:1313 ++``` ++ ++### GitHub Evidence Checks ++ ++```bash ++gh run view 30742507113 --repo jmservera/SquadScope \ ++ --json status,conclusion,headSha,jobs,url ++``` ++ ++```bash ++gh run view 30721575540 --repo jmservera/SquadScope \ ++ --json conclusion,headSha,jobs,url ++``` ++ ++```bash ++gh run view 30202586031 --repo jmservera/SquadScope --log ++``` ++ ++```bash ++gh api repos/jmservera/SquadScope/environments \ ++ --jq '.environments[] | {name, protection_rules, deployment_branch_policy}' ++``` ++ ++### Side-Effecting Dispatches Requiring Approval ++ ++Do not run either command as a read-only check. Substitute an authorized week and retained publish run only after environment protection, workflow binding, and Podcaster-maintainer approval. ++ ++```bash ++gh workflow run podcaster-handoff-smoke.yml \ ++ --repo jmservera/SquadScope \ ++ --ref main \ ++ -f week=YYYY-WNN \ ++ -f article_url=https://claracle.com/weekly/YYYY/WNN/ \ ++ -f article_path=content/weekly/YYYY/WNN.md \ ++ -f article_sha256=LOWERCASE_SHA256 \ ++ -f promotion_reference=data/published/YYYY-WNN/promotion-manifest.json ++``` ++ ++```bash ++gh workflow run trigger-podcast.yml \ ++ --repo jmservera/SquadScope \ ++ --ref main \ ++ -f week=YYYY-WNN \ ++ -f publish_run_id=RUN_ID ++``` ++ ++## References ++ ++### Repository Evidence ++ ++* docs/review/data-observatory-relaunch/README.md ++* docs/review/data-observatory-relaunch/status-of-record.md ++* docs/review/data-observatory-relaunch/security-review.md ++* docs/review/data-observatory-relaunch/screenshots/README.md ++* docs/review/data-observatory-relaunch/screenshots/01-home.png through 10-internal-linking-block.png ++* docs/prds/claracle-data-observatory-relaunch.md ++* docs/brds/claracle-data-observatory-relaunch-brd.md ++* .github/workflows/ci.yml ++* .github/workflows/deploy-site.yml ++* .github/workflows/podcaster-handoff-smoke.yml ++* .github/workflows/trigger-podcast.yml ++* .github/workflows/sync-publish-to-main.yml ++* scripts/design/lighthouse-gates.mjs ++* scripts/manage_topic_hubs.py ++* scripts/export_observatory_dataset.py ++* scripts/podcaster_handoff.py ++* tests/visual/a11y-perf.spec.mjs ++* tests/visual/observatory-a11y.spec.mjs ++* tests/visual/observatory-analytics.spec.mjs ++* tests/visual/visual.spec.mjs ++ ++### External Evidence ++ ++* [Issue #626](https://github.com/jmservera/SquadScope/issues/626) ++* [Issue #622](https://github.com/jmservera/SquadScope/issues/622) ++* [Latest confirmed green deploy and dry-run smoke](https://github.com/jmservera/SquadScope/actions/runs/30721575540) ++* [Real accepted Podcaster run](https://github.com/jmservera/SquadScope/actions/runs/30202586031) ++* [Current reconciliation CI](https://github.com/jmservera/SquadScope/actions/runs/30742507113) ++ ++## Follow-On Questions ++ ++* [ ] Inspect the completed Lighthouse results and uploaded quality artifacts from CI run 30742507113 after it finishes ++* [ ] Confirm whether the Podcaster service deduplicates week or manifest requests outside this repository ++* [ ] Confirm the intended required reviewer and branch policy for real Podcaster generation ++* [ ] Render current W21-W31 content and verify that topic counts replace the historical empty states ++* [ ] Determine the intended Star Velocity normalization model from product/design ownership ++* [ ] Decide and document the embed referrer and cross-origin consent policy ++* [ ] Resolve the blocking versus non-blocking status of issue #622 ++ ++## Clarifying Questions ++ ++* Who should approve the protected real Podcaster environment: URL, Hermes, jmservera, or a Podcaster maintainer? ++* Is issue #622 a mandatory relaunch gate as recorded in status-of-record.md, or non-blocking polish as stated in the issue body? ++* Does SquadScope-Podcaster guarantee idempotency for a repeated week or manifest, and where is that contract retained? ++* Which screen reader and browser combination is the required NFR-005 manual acceptance target? +diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md +new file mode 100644 +index 0000000..0ccee6e +--- /dev/null ++++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md +@@ -0,0 +1,289 @@ ++ ++# Claracle GA4 and Google Search Console Follow-up Research ++ ++## Status ++ ++Complete as of 2026-08-02 ++ ++## Research Questions ++ ++* What FR-035 repository-side work can be completed without Google credentials? ++* Which GA4 and Google Search Console actions require jmservera account access? ++* Should `ga_measurement_id` be driven by Hugo configuration, environment variables, or both? ++* What are the cheapest executable validations for the selected approach? ++ ++## Scope ++ ++* `hugo.toml` ++* Consent-gating layouts and scripts ++* Growth evidence and FR-035 planning records ++* GitHub issue #599, when available ++* Relevant tests and repository instructions ++ ++## Findings ++ ++### Executive conclusion ++ ++FR-035 is partially implemented, not wholly unimplemented. The repository already has the required ++fork-safe GA4 parameter path, dynamic consent-gated loading, GSC verification metadata path, Hugo ++sitemap generation, workflow secret injection, unit coverage for workflow mapping and GSC rendering, ++and a blocking Playwright consent suite. No new production analytics implementation is required before ++the account-side work. ++ ++The remaining acceptance work is external. A names-only GitHub secret query showed that ++`GA_MEASUREMENT_ID` exists and was last updated on 2026-06-13, while `GSC_SITE_VERIFICATION` does not ++exist. A credential-free production probe on 2026-08-02 found GA configuration in the rendered home ++page, no GSC verification meta tag, and a successful XML sitemap response: ++ ++```text ++ga_config_present=yes ++gsc_meta_present=no ++sitemap_status=200 ++sitemap_content_type=application/xml ++``` ++ ++These observations disprove the status-of-record rationale that an empty checked-in ++`ga_measurement_id` means GA4 is not deployed. They do not prove that the deployed ID belongs to the ++intended property, that GA4 receives events, or that the required baseline has been captured. ++ ++### Current repository implementation ++ ++| Surface | Current behavior | Evidence | ++| --- | --- | --- | ++| Checked-in GA4 configuration | Empty by design for fork safety | `hugo.toml:22-24` | ++| Checked-in GSC configuration | Empty by design | `hugo.toml:25-26` | ++| Production parameter injection | Actions secrets map to Hugo environment overrides | `.github/workflows/deploy-site.yml:36-40` | ++| GA4 render path | Supports flat config and Hugo's nested environment mapping; renders only when configured | `layouts/partials/analytics.html:1-11` | ++| Consent gate | GA is disabled first; `gtag.js` is appended only after analytics consent | `layouts/partials/cookie-consent.html:44-101` | ++| Custom events | Event names and fields are allowlisted; dispatch returns before consent | `assets/js/observatory-analytics.js:4-78` | ++| GSC render path | Supports the new secret-backed parameter plus the legacy theme fallback | `layouts/partials/head.html:20-29` | ++| Workflow contract test | Verifies both Actions secret mappings and CI's GA test ID | `tests/test_pipeline.py:265-284` | ++| GSC render tests | Verify absent-by-default, environment override, precedence, and legacy fallback | `tests/test_rendered_seo_metadata.py:447-539` | ++| Consent browser tests | Verify denied, accepted, reload, withdrawal, cookie clearing, and bounded events | `tests/visual/observatory-analytics.spec.mjs:1-180` | ++ ++### Repository-side work possible without Google credentials ++ ++No product behavior must be added to connect the existing paths. The following repository work can be ++completed without signing in to Google: ++ ++1. Correct stale evidence and status wording after the account observations are supplied. In ++ particular, `docs/review/data-observatory-relaunch/status-of-record.md:70` and ++ `docs/prds/claracle-data-observatory-relaunch.md:138` should not use the empty fork-safe default as ++ evidence that production GA4 is disconnected. ++2. Update `docs/growth/ga4-gsc-baseline-2026-07-29.md` with dated, redacted observations supplied by ++ jmservera. Repository authors can prepare and review the evidence structure without platform access, ++ but must not invent values or copy secret tokens. ++3. Clarify the comment in `hugo.toml:23`. It currently suggests writing the production ID into the ++ file, while the implemented security decision and `docs/setup-secrets.md` require secret-backed ++ environment injection and an empty checked-in default. ++4. Run the workflow mapping test, Hugo rendering tests, local consent browser test, sitemap HTTP probe, ++ and names-only secret inventory. These checks need repository, network, or GitHub access, but no ++ Google credentials. ++5. Optionally add a small Hugo render test for the GA parameter itself. The current blocking Playwright ++ path already exercises it end to end with `G-TEST-OBSERVATORY`, so this is a test-speed improvement, ++ not an FR-035 blocker. ++ ++Setting `GSC_SITE_VERIFICATION` in GitHub is also repository-side, but the token must first be obtained ++from a Search Console property. The person setting it needs GitHub secret-management rights; they do ++not need Google access if jmservera supplies the token through an approved private channel. ++ ++### Issue 599 context ++ ++GitHub issue #599 was opened by jmservera on 2026-07-29 and closed as completed on 2026-08-01. Its ++acceptance criteria require GA4 receipt, a verified GSC property, sitemap submission, and a dated ++baseline. The only owner follow-up comment records five remaining human actions: create or select the ++Claracle GA4 property and web stream, set the production measurement ID, verify `claracle.com` in GSC, ++submit the sitemap, and capture baseline values. The issue contains no later comment proving those ++actions. Closure therefore records completion of agent-side wiring, not FR-035 operational acceptance. ++ ++Issue URL: [#599](https://github.com/jmservera/SquadScope/issues/599) ++ ++## Selected Approach ++ ++Keep the existing config-plus-environment design: ++ ++* Keep `params.ga_measurement_id = ""` and `params.gsc_site_verification = ""` in `hugo.toml` ++* Inject production values through `HUGO_PARAMS_GA_MEASUREMENT_ID` and ++ `HUGO_PARAMS_GSC_SITE_VERIFICATION` from GitHub Actions secrets ++* Continue resolving both flat checked-in keys and Hugo's nested environment-key representation in the ++ templates ++* Never hardcode the production GA measurement ID or GSC verification token in source ++ ++For GSC, use the already-implemented URL-prefix property and HTML meta-tag verification flow for ++`https://claracle.com/`. This is the cheapest path because it needs no DNS change and the template, ++workflow mapping, documentation, and tests already exist. A Domain property would broaden coverage but ++requires DNS-provider access and does not use the repository's current verification path. ++ ++Recommended completion sequence: ++ ++1. jmservera confirms that the existing deployed GA ID belongs to the intended Claracle property and ++ web stream. Do not rotate the GitHub secret unless it is wrong. ++2. jmservera adds or selects the GSC URL-prefix property and privately obtains its HTML-tag token. ++3. A repository administrator sets `GSC_SITE_VERIFICATION`, deploys the reviewed revision, and confirms ++ only the presence of the production meta tag. ++4. jmservera clicks Verify in GSC and submits `https://claracle.com/sitemap.xml`. ++5. jmservera performs a consented production visit, confirms GA4 Realtime receipt, and captures GA4/GSC ++ values for one explicitly dated baseline window. ++6. Hermes or the designated privacy reviewer records denied and granted production browser behavior. ++7. A repository-only follow-up updates the baseline, status of record, and PRD acceptance note with ++ redacted evidence and marks FR-035 complete only when every acceptance item is evidenced. ++ ++## Credential and Ownership Boundary ++ ++### Requires jmservera or delegated Google account access ++ ++* Create, select, or inspect the GA4 account, property, and web data stream. Google requires the Editor ++ role to create properties or streams. ++* Confirm that the deployed measurement ID maps to the intended Claracle stream. ++* Observe Realtime receipt and capture GA4 acquisition/session baseline values. ++* Add or select the Search Console property and obtain its verification token. ++* Complete GSC ownership verification. A verified owner has the highest Search Console permission. ++* Submit the sitemap in the verified property and capture submission, processing, performance, and ++ indexing evidence. ++ ++These actions are assigned to jmservera by `docs/data-observatory-runbook.md:32-45` and issue #599. ++They can be delegated only by granting the appropriate GA4 role and GSC owner/user access. ++ ++### Requires GitHub access but not Google credentials ++ ++* List secret names and timestamps without viewing values ++* Set or rotate `GA_MEASUREMENT_ID` after an authorized person supplies the ID ++* Set `GSC_SITE_VERIFICATION` after an authorized person supplies the token ++* Trigger or inspect the Pages deployment and retain the Actions URL ++ ++GitHub does not expose Actions secret values after creation. Secret-name presence proves protected ++configuration exists, not that it is correct. ++ ++### Requires neither Google nor privileged GitHub access ++ ++* Inspect and test templates, consent code, workflow expressions, and generated output with test IDs ++* Probe the public home page for configuration presence without printing the ID ++* Probe the public sitemap response and content type ++* Prepare documentation and evidence placeholders ++ ++### Blockers ++ ++* `GSC_SITE_VERIFICATION` is absent from the repository secret inventory ++* Production has no GSC verification meta tag as of 2026-08-02 ++* No retained evidence proves GA4 Realtime receipt or the intended property/stream mapping ++* No retained evidence proves GSC ownership verification or sitemap submission ++* All GA4/GSC baseline values remain pending ++* Local Hugo render tests could not execute in this session because the `hugo` binary is absent ++ ++## Validation Commands ++ ++### Cheapest repository contract ++ ++```bash ++python3 -m pytest -q \ ++ tests/test_pipeline.py::WorkflowConfigTests::test_deploy_workflow_maps_analytics_and_gsc_secrets_to_hugo_params ++``` ++ ++Observed result: `1 passed` when run as part of the focused four-test selection. ++ ++### GSC rendering contract ++ ++```bash ++python3 -m pytest -q tests/test_rendered_seo_metadata.py -k gsc_site_verification ++``` ++ ++Requires Hugo. The three selected GSC tests skipped locally because `hugo` is not installed. CI installs ++Hugo 0.161.1 and runs this file in the blocking production-site job. ++ ++### Full static production build ++ ++```bash ++HUGO_PARAMS_GA_MEASUREMENT_ID=G-TEST-OBSERVATORY \ ++HUGO_PARAMS_GSC_SITE_VERIFICATION=testtoken123 \ ++hugo --minify --quiet --destination /tmp/claracle-fr035 ++``` ++ ++Inspect only the test values in `/tmp/claracle-fr035/index.html`. Also build once with both variables ++unset and confirm neither analytics configuration nor the GSC meta tag is emitted. ++ ++### Consent behavior ++ ++Use the same pinned dependencies and local server setup as `.github/workflows/ci.yml:110-177`, then run: ++ ++```bash ++BASE_URL=http://127.0.0.1:1313 \ ++npx --no-install playwright test \ ++ --config tests/visual/playwright.config.mjs \ ++ tests/visual/observatory-analytics.spec.mjs \ ++ --project desktop-light ++``` ++ ++This is narrower than the full axe, responsive, and Lighthouse suite. It intercepts Google endpoints, ++so it does not need a real GA property or send test events to Google. ++ ++### Protected configuration metadata ++ ++```bash ++gh secret list --repo jmservera/SquadScope --json name,updatedAt \ ++ | jq '[.[] | select(.name == "GA_MEASUREMENT_ID" or .name == "GSC_SITE_VERIFICATION")]' ++``` ++ ++This command must never be replaced with a command that prints secret values. ++ ++### Credential-free production smoke ++ ++```bash ++html="$(curl --fail --silent --show-error --location https://claracle.com/)" ++printf 'ga_config_present=%s\n' "$(if grep -q 'gaMeasurementId' <<<"$html"; then echo yes; else echo no; fi)" ++printf 'gsc_meta_present=%s\n' "$(if grep -q 'name=google-site-verification' <<<"$html"; then echo yes; else echo no; fi)" ++curl --fail --silent --show-error --location --output /dev/null \ ++ --write-out 'sitemap_status=%{http_code}\nsitemap_content_type=%{content_type}\n' \ ++ https://claracle.com/sitemap.xml ++``` ++ ++This probe deliberately reports presence only and does not print the measurement ID or verification ++token. ++ ++## References and Evidence ++ ++### Repository references ++ ++* `.github/copilot-instructions.md` ++* `.github/workflows/ci.yml` ++* `.github/workflows/deploy-site.yml` ++* `.squad/decisions-archive.md:1303-1327` ++* `assets/js/observatory-analytics.js` ++* `docs/data-observatory-runbook.md` ++* `docs/growth/ga4-gsc-baseline-2026-07-29.md` ++* `docs/prds/claracle-data-observatory-relaunch.md` ++* `docs/review/data-observatory-relaunch/security-review.md` ++* `docs/review/data-observatory-relaunch/status-of-record.md` ++* `docs/setup-secrets.md` ++* `hugo.toml` ++* `layouts/partials/analytics.html` ++* `layouts/partials/cookie-consent.html` ++* `layouts/partials/head.html` ++* `tests/test_pipeline.py` ++* `tests/test_rendered_seo_metadata.py` ++* `tests/visual/observatory-analytics.spec.mjs` ++ ++### External references ++ ++* [Google Analytics setup](https://support.google.com/analytics/answer/9304153): signed-in Google ++ account; Editor role required to create a property or add a data stream ++* [Search Console ownership verification](https://support.google.com/webmasters/answer/9008080): ++ property addition, verification methods, owner permissions, and verification persistence ++* [Search Console Sitemaps report](https://support.google.com/webmasters/answer/7451001): submission, ++ processing status, and the distinction between submitted and automatically discovered sitemaps ++ ++## Follow-on Questions ++ ++* Does the existing `GA_MEASUREMENT_ID` map to a dedicated Claracle production web stream? ++* Has GSC already collected pre-verification data for a property that only needs ownership completion? ++* Which approved private evidence location should retain redacted GA4 Realtime and GSC screenshots or ++ exports? ++ ++## Clarifying Questions ++ ++These questions require jmservera account observations and cannot be answered from repository or public ++evidence: ++ ++* Is the intended GA4 property already receiving consented production events? ++* Does a Claracle GSC property already exist under the jmservera account? ++* Should the dated baseline window begin at first verified receipt, relaunch approval, or a fixed ++ calendar boundary? +\ No newline at end of file +diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md +new file mode 100644 +index 0000000..cf6614a +--- /dev/null ++++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md +@@ -0,0 +1,422 @@ ++ ++# Claracle Rollout and Cost Follow-Up Research ++ ++## Research Scope ++ ++* Investigate existing `repo_pages` rollout behavior and safety dependencies ++* Investigate existing `dynamic_topic_creation` rollout behavior and safety dependencies ++* Determine how Hugo and Pagefind timings are separated today ++* Determine what measurable instrumentation exists for incremental generation cost Q-01/NFR-009 ++* Identify the smallest implementation-ready phases and precise validations ++ ++## Executive Findings ++ ++* Both production creation controls are off. `repo_pages.enabled = false` and ++ `topic_hubs.dynamic_creation.enabled = false` are defined in ++ `config/observatory.toml:1-33`. ++* The flags freeze mutation, not visibility. The repository contains 263 generated ++ repository pages plus `content/repo/_index.md`, five seed topic hubs, and three ++ generated data-page leaves. Hugo still renders those durable files while the ++ generators are disabled. ++* Repository rollout is an all-current-state activation, not a first publication of ++ 263 pages. The July 29 implementation commits generated pages; commit `879c42c` ++ added `repo_pages.enabled = false`, changed dynamic creation from true to false, ++ and preserved the generated state and lifecycle ledger. ++* Dynamic activation currently has five eligible candidates among 2,173 candidates: ++ AI Memory, fable, inference, Local First, and Self Hosted. Enabling the flag without ++ changing the ignore list can promote all five in one publish transaction. ++* Hugo and Pagefind are timed separately only in the CI `production-site` job. CI ++ writes one report-only `reports/build-timing.json` artifact with separate ++ `duration_ms` values and no threshold. This is total build timing, not incremental ++ cost attribution by hubs, data pages, or repository pages. ++* Q-01/NFR-009 remains open. One local observation exists, but there is no retained ++ representative series, median/p95 report, approved budget, per-page-class workload ++ metadata, or blocking threshold. ++* The smallest safe sequence is: measure immutable workload variants first, close ++ security and identity blockers, canary one reviewed dynamic topic through the ++ existing ignore list, then activate repository regeneration as one reviewed ++ transaction. Repository threshold changes are not a safe canary because generated ++ pages outside the new expected set are treated as obsolete and can be deleted. ++ ++## Existing Flag Behavior ++ ++### Repository pages ++ ++`scripts/observatory_repos.py:205-220` loads the flag, recurrence threshold, three-year ++retention, lifecycle overrides, and lifecycle-ledger path. The threshold is strictly ++`>` and defaults to more than three distinct weekly issues, equivalent to four weeks. ++ ++`scripts/observatory_repos.py:997-1020` returns immediately when disabled. It does not ++read or update the ledger, generate derived data, create pages, delete durable pages, ++or validate staleness. The same early return applies to `--check`, so a green disabled ++freshness check does not prove repository outputs are current. ++ ++`scripts/observatory_repos.py:922-994` computes all expected page and derived outputs ++when enabled. In write mode it creates or rewrites expected outputs and removes ++obsolete generated pages. In check mode it returns stale, obsolete, and expired paths ++without writing. This means recurrence-threshold reduction or a small threshold-based ++canary can classify existing generated pages as obsolete. ++ ++`scripts/observatory_repos.py:813-859` provides a separate lifecycle seed operation. ++It requires production `repo_pages.enabled = false`, validates parity among qualified ++histories, repository pages, and derived repository data, then atomically writes only ++`data/derived/observatory/repository-lifecycle.json`. It is byte-stable when repeated. ++ ++Current durable state, corroborated by ++`tests/test_observatory_repos.py:570-581`, is: ++ ++| Measure | Current value | ++|---------|--------------:| ++| Lifecycle histories | 2,242 | ++| Qualified histories | 263 | ++| Generated repository page leaves | 263 | ++| Histories with stable `github_id` | 0 | ++| Lifecycle statuses | 2,242 active; no checked-in rename/archive/delete transition | ++ ++The code can absorb fallback name history into a stable GitHub ID later, as tested in ++`tests/test_observatory_repos.py:584-627`, but production inputs have not supplied those ++IDs. Stable canonical rename behavior is therefore implemented but not evidenced on ++the production corpus. ++ ++### Dynamic topic creation ++ ++Candidate discovery is independent from promotion. The publish workflow always runs ++`scripts/discover_topic_candidates.py`, which derives a byte-stable registry from ++weekly content, analyzed summaries, and raw observations. It uses the configured ++four-week threshold, 62-day lookback, ignore list, and repository recurrence threshold ++from `scripts/discover_topic_candidates.py:53-67`. ++ ++`scripts/manage_topic_hubs.py:407-419` exits before reading candidates or writing a log ++when the creation flag is disabled. `--dry-run` also exits at this point when enabled; ++it is a no-op safety switch, not a preview of proposed changes. ++ ++When enabled, `scripts/manage_topic_hubs.py:421-486` can perform one transaction that: ++ ++* Creates `content/topics//_index.md` ++* Promotes the term in `data/taxonomy/topics.json` ++* Assigns the topic to historical weekly frontmatter supported by evidence ++* Reassigns already promoted topics from current source evidence ++* Refreshes taxonomy registries ++* Appends decision and summary entries to ++ `data/topic-hubs/dynamic-topic-creation.log` ++ ++Creation is additive. Quiet or subsequently ineligible hubs are not deleted. Turning ++the flag off stops future mutation but does not reverse promoted registry entries, ++topic pages, weekly assignments, or logs. ++ ++The current candidate registry in `data/taxonomy/topic-candidates.json` has 2,173 ++candidates and five eligible candidates. Each eligible candidate has at least four ++weekly issues and at least one supporting signal. The threshold alone is not editorial ++approval; `fable` and `inference`, for example, are broad terms requiring human review. ++ ++The existing `ignore_topics` list can implement a configuration-only canary by adding ++four reviewed deferrals and allowing one candidate. There is no positive allowlist, ++maximum creations per run, or useful non-mutating preview mode. ++ ++## Rollout Safety Dependencies ++ ++### Shared dependencies ++ ++* Generated state must be hydrated from `publish` before any preflight or measurement. ++ The publish generator does this in ++ `.github/workflows/crawl-and-publish.yml:1048-1077`; deployment does it in ++ `.github/workflows/deploy-site.yml:88-122`. ++* The `weekly-crawl` concurrency group uses `cancel-in-progress: false` at ++ `.github/workflows/crawl-and-publish.yml:64-66`, preventing overlapping publish ++ mutations. ++* Generated content, taxonomy, topic logs, and repository derived state are committed ++ together by `.github/workflows/crawl-and-publish.yml:1224-1263` and must be reviewed ++ as one transaction. ++* Hermes security acceptance, URL workflow review, and jmservera sponsor approval are ++ all pending in ++ `docs/review/data-observatory-relaunch/security-review.md:151-169`. ++* The PRD requires separate sponsor-approved rollouts after security and lifecycle ++ evidence in `docs/prds/claracle-data-observatory-relaunch.md:260-267`. ++ ++### Repository-specific dependencies ++ ++* Run lifecycle parity seed twice while disabled against the hydrated publish revision. ++* Acquire stable GitHub identity fields or explicitly disposition fallback name identity ++ risk before claiming FR-020 stable canonical URLs or FR-022 rename safety. ++* Exercise reviewed rename, archive, and deletion evidence. Current production ledger ++ contains only active histories, while fixture coverage exists in ++ `tests/test_observatory_repos.py:664-761`. ++* Run enabled `--check` and a full generation in a disposable worktree before changing ++ production config. Review every created, rewritten, obsolete, and expired path. ++* Preserve the current threshold during activation. A threshold canary is unsafe because ++ the generator removes obsolete generated pages. ++ ++### Dynamic-topic-specific dependencies ++ ++* Hermes must disposition SEC-01. Structured YAML and adversarial title rejection are ++ implemented in `tests/test_topic_hubs.py:527-586`, but the security review still ++ records dynamic title handling as rollout-blocking. ++* Review each eligible candidate's evidence, semantics, aliases, and affected weekly ++ files before promotion. ++* Use a disposable worktree with a temporary enabled config to obtain the proposed diff, ++ because current `--dry-run` does not evaluate candidates. ++* Select a single canary through `ignore_topics`, retain the other eligible candidates as ++ explicit deferrals, and obtain sponsor approval for that exact config and diff. ++ ++## Hugo and Pagefind Timing Separation ++ ++The CI production-site job has separate timed steps: ++ ++* Hugo Extended 0.161.1 at `.github/workflows/ci.yml:127-134` ++* Pagefind 1.5.2 at `.github/workflows/ci.yml:136-143` ++* JSON report writing at `.github/workflows/ci.yml:145-156` ++* Artifact retention under `production-quality-reports` at ++ `.github/workflows/ci.yml:181-191` ++ ++The report schema records commit, report-only mode, tool versions, and separate ++durations. `blocking_threshold_ms` is null. This correctly prevents an unapproved ++budget from becoming a gate. ++ ++Timing comparability gaps remain: ++ ++* CI builds the checked-out branch and does not hydrate generated state from `publish`. ++ Deploy and crawl do hydrate it, so CI timing can measure a different workload. ++* Crawl, deploy, and preview invoke unpinned `npx pagefind`; only CI pins Pagefind 1.5.2. ++ References: `.github/workflows/crawl-and-publish.yml:1446-1449`, ++ `.github/workflows/deploy-site.yml:178-181`, and ++ `.github/workflows/site-preview.yml:116-119`. ++* The report omits source-page counts, rendered-page counts, HTML files scanned, indexed ++ pages, output bytes, runner identity, hydration source SHA, and workload variant. ++* The artifact is ephemeral and no repository process aggregates comparable reports. ++* Hugo and Pagefind are sequentially separated, but neither generator execution time nor ++ incremental page-class contribution is measured. ++ ++## Existing Cost and Measurement Evidence ++ ++`docs/design/data-observatory-model.md:400-449` records one local report-only sample: ++ ++| Stage | Duration | Workload evidence | ++|-------|---------:|-------------------| ++| Hugo 0.161.1 | 6,668 ms | 2,669 rendered pages | ++| Pagefind 1.5.2 | 6,207 ms | 1,477 HTML files scanned; 288 pages indexed | ++ ++The design document correctly labels this observation as insufficient and requires ++three comparable external CI reports, median and p95, a proposed blocking budget, and ++owner approval. Prior validation reaches the same conclusion in ++`.copilot-tracking/reviews/rpi/2026-07-30/claracle-data-observatory-relaunch-remediation-plan-007-validation.md:128-203`. ++ ++Reusable percentile code exists in `scripts/baseline_telemetry.py:69-124`, with tests in ++`tests/test_baseline_telemetry.py`, but it is scoped to crawl/analysis observability ++ledgers and uses a five-run readiness baseline. It does not consume ++`build-timing.json` or calculate incremental generation cost. ++ ++The current repository provides useful workload counters but no integrated cost record: ++ ++* 5 seed topic hubs under `content/topics/` ++* 3 generated data-page leaves under `content/data/` ++* 263 generated repository page leaves under `content/repo/` ++* 5 currently eligible dynamic topic candidates ++* Generator summaries such as `Generated repository pages` and ++ `dynamic-topic-summary created= skipped=` ++ ++Q-01 is therefore measurable with existing tools, but not answered by existing data. ++ ++## Selected Approach ++ ++### Phase 1: Implement a report-only incremental cost experiment ++ ++Use a disposable worktree hydrated from the same `publish` SHA for every variant. Keep ++Hugo, Pagefind, runner image, config, and source revision fixed. Build into clean, ++variant-specific destinations so tracked `public/` content cannot contaminate results. ++ ++Measure these cumulative variants: ++ ++1. Observatory generated page classes excluded ++2. Add the five checked-in topic hubs ++3. Add the three generated data pages ++4. Add the 263 checked-in repository pages ++5. Optionally add the exact reviewed dynamic canary diff ++ ++Run at least three comparable CI repetitions because that is the documented acceptance ++minimum; five runs aligns with the repository's existing baseline-telemetry convention. ++For each repetition and variant, record: ++ ++* Main SHA and hydrated publish SHA ++* Runner image and tool versions ++* Source Markdown counts by page class ++* Hugo duration and rendered page count ++* Pagefind duration, HTML files scanned, and indexed page count ++* Hugo destination bytes and Pagefind index bytes ++* Exit status and run/artifact URL ++ ++Aggregate median, nearest-rank p95, absolute delta, percent delta, and marginal ++milliseconds/added source page separately for Hugo and Pagefind. Do not enforce a budget ++in this phase. Publish the report and obtain owner approval before replacing the null ++threshold. ++ ++### Phase 2: Prepare repository activation without enabling production ++ ++1. Resolve or explicitly accept the missing stable-ID risk and review lifecycle evidence. ++2. Hydrate the target publish revision and run lifecycle seed twice while disabled. ++3. Confirm 263-way parity and byte-identical second seed. ++4. In a disposable worktree, enable the existing config without changing the threshold. ++5. Run `observatory_repos.py --check`, then generation twice, and review all output diffs. ++6. Run repository tests, Hugo, pinned Pagefind, rendered SEO/link checks, internal links, ++ Lighthouse, axe, and the Q-01 workload measurement. ++7. Obtain Hermes, URL, and sponsor sign-offs for the exact revision and diff. ++ ++### Phase 3: Canary one dynamic topic ++ ++1. Review all five eligible candidates and select one unambiguous canary. `local-first` ++ currently has the strongest evidence breadth with six weekly issues and five ++ supporting signals, but editorial/security review must make the final choice. ++2. Add the other four candidates to `ignore_topics` as explicit temporary deferrals. ++3. Produce the enabled result in a disposable worktree and review the hub, canonical ++ registry change, historical weekly assignments, taxonomy changes, and log event. ++4. Run the focused and rendered validation suite and capture incremental timing. ++5. Obtain approval, enable the flag for one publish run, inspect the committed generated ++ transaction, and turn the flag off or retain the restricted ignore list according to ++ the approved rollout decision. ++ ++### Phase 4: Activate repository regeneration ++ ++Enable `repo_pages` at the unchanged threshold only after Phases 1 and 2 pass. Treat the ++first publish as a single 263-page lifecycle activation. Verify no unexpected obsolete or ++expired paths before promotion. Rollback requires both disabling the flag and reverting ++the generated transaction; disabling alone preserves mutations already committed. ++ ++### Phase 5: Expand and enforce ++ ++Remove dynamic candidate deferrals one reviewed candidate at a time. After enough stable ++timing samples and explicit owner approval, add separate Hugo and Pagefind budgets with a ++report-only observation period before making either threshold blocking. ++ ++## Dependencies and Blockers ++ ++### Blocking ++ ++* Hermes NFR-004/security sign-off is pending, including SEC-01 dynamic-title disposition ++ and SEC-04 lifecycle deletion policy. ++* URL workflow/security sign-off and jmservera sponsor rollout approval are pending. ++* All 2,242 production repository histories lack stable GitHub IDs; stable rename identity ++ is not evidenced. ++* Production lifecycle state has no reviewed rename, archive, or deletion transition. ++* Dynamic `--dry-run` cannot preview the mutation set; a disposable worktree is required ++ unless preview semantics are implemented first. ++* Q-01 lacks comparable CI samples, aggregation, approved budgets, and page-class ++ attribution. ++ ++### Non-blocking implementation dependencies ++ ++* Access to the canonical `publish` branch and retained workflow artifacts ++* Hugo Extended 0.161.1, Pagefind 1.5.2, Python 3.12, and Node.js 24 ++* A clean disposable worktree or equivalent isolated checkout per workload variant ++* Existing generated-state transaction paths in crawl, deploy, and freshness workflows ++* Existing tests for disabled-state preservation, deterministic generation, structured ++ YAML, adversarial titles, lifecycle retention, stable-ID migration, and Hugo rendering ++ ++## Precise Validations ++ ++### Flag and generator behavior ++ ++```bash ++python -m pytest tests/test_observatory_repos.py tests/test_topic_hubs.py tests/test_taxonomy_registry.py ++python scripts/discover_topic_candidates.py --check ++python scripts/generate_data_pages.py --check ++python scripts/export_observatory_dataset.py --check ++python scripts/export_trend_explorer_data.py --check ++``` ++ ++Run repository freshness against a temporary config with `repo_pages.enabled = true`; ++the production disabled config makes `observatory_repos.py --check` a no-op. ++ ++For repository activation, require: ++ ++* Lifecycle seed parity is 263 qualified histories, 263 page identities, and 263 derived ++ identities ++* The second lifecycle seed is byte-identical ++* Enabled check reports no unexplained stale, obsolete, or expired paths ++* Two enabled generations produce byte-identical outputs ++* No page removal occurs without reviewed positive lifecycle evidence and elapsed ++ retention ++ ++For dynamic canary activation, require: ++ ++* Exactly one approved hub is created ++* Only evidence-backed weekly files receive the topic assignment ++* Registry YAML/JSON remains parseable and canonical aliases resolve ++* A structured promotion log records evidence weeks, sources, and assigned paths ++* A second run is additive and byte-stable except for explicitly designed append-log ++ behavior ++* Disabled rollback creates or deletes nothing ++ ++### Rendered and pipeline behavior ++ ++```bash ++hugo --minify ++npx "pagefind@1.5.2" --site public/ ++python scripts/check_internal_links.py public --base-url "https://claracle.com/" ++python -m pytest tests/test_rendered_seo_metadata.py tests/test_rendered_weekly_links.py tests/test_internal_link_checker.py ++python -m pytest tests/ ++ruff check . ++ruff format --check . ++``` ++ ++Also run the existing Lighthouse and axe route matrices, which include a topic, data, and ++repository page at `scripts/design/lighthouse-gates.mjs:28-36` and the Observatory visual ++tests. If a workflow changes, run Zizmor and Checkov under the repository guardrails. ++ ++### Cost acceptance ++ ++* Every timing artifact identifies both main and publish SHAs and the workload variant ++* Hugo and Pagefind retain separate raw samples and statistics ++* Each variant starts from a clean destination and uses pinned versions ++* At least three comparable CI repetitions exist; five are preferred ++* Median and p95 are reproducible from retained machine-readable samples ++* Incremental deltas are reported by page class, not inferred from one total build ++* The approved budget names its owner, sample window, headroom rationale, and enforcement ++ date ++* Thresholds remain report-only until approval is recorded ++ ++## Evidence Index ++ ++* `config/observatory.toml:1-33` controls both disabled rollouts ++* `scripts/observatory_repos.py:205-220,813-859,922-1020` defines config, lifecycle seed, ++ writes/checks, and disabled behavior ++* `scripts/discover_topic_candidates.py:53-67` defines candidate policy inputs ++* `scripts/manage_topic_hubs.py:407-486` defines disabled, dry-run, and mutation behavior ++* `tests/test_observatory_repos.py:570-790` proves frozen-corpus parity, stable-ID ++ migration, disabled preservation, lifecycle rendering, and Hugo output ++* `tests/test_topic_hubs.py:204-376,444-588` proves additive creation, persistence, ++ disabled preservation, unsafe-title rejection, and structured YAML ++* `.github/workflows/crawl-and-publish.yml:1048-1077,1139-1210,1224-1263` hydrates, ++ generates, checks, and commits the generated transaction ++* `.github/workflows/ci.yml:127-156,181-191` emits and uploads separate timing data ++* `.github/workflows/deploy-site.yml:88-122,178-181` hydrates production generated state ++ and builds Hugo/Pagefind ++* `.github/workflows/generate-data-pages.yml:1-66` hydrates publish state for monthly ++ freshness checks ++* `docs/data-observatory-runbook.md:18-132` defines operating boundaries, generation ++ order, lifecycle policy, and seed procedure ++* `docs/design/data-observatory-model.md:400-449` records the only local timing sample and ++ pending cost acceptance ++* `docs/prds/claracle-data-observatory-relaunch.md:121-176,260-278` defines FR-004, ++ FR-020-022, NFR-009, rollout flags, and Q-01 ++* `docs/review/data-observatory-relaunch/security-review.md:140-169` records open security ++ findings and pending sign-offs ++* `.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md:10-18,64-74` ++ records these workstreams as separate follow-on plans ++* Git commits `0baae6d`, `f9a17c8`, and `879c42c` establish generated-page, dynamic-topic, ++ and rollout-freeze provenance ++ ++## Recommended Next Research ++ ++* [ ] Inspect authenticated retained `production-quality-reports` artifacts from at least ++ three comparable successful runs; repository source cannot supply their raw values ++* [ ] Confirm whether upstream crawl payloads can begin persisting `id`/`node_id`, archive, ++ disabled, and rename evidence before repository activation ++* [ ] Have editorial/security owners disposition the five currently eligible topic ++ candidates and nominate an exact canary ++* [ ] Obtain the named owner and approved method for the Hugo/Pagefind regression budget ++ ++## Clarifying Questions ++ ++* Will missing stable GitHub IDs block `repo_pages` activation, or will the sponsor accept ++ fallback name identity for the first activation window? ++* Which eligible dynamic topic, if any, is approved as the first canary? ++* Who gives final approval for separate Hugo and Pagefind budgets after the timing series? +diff --git a/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md +new file mode 100644 +index 0000000..7739ee8 +--- /dev/null ++++ b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md +@@ -0,0 +1,54 @@ ++ ++# Review: Claracle Relaunch Follow-Up Execution ++ ++## Metadata ++ ++* Plan: .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md ++* Rollout plan: .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md ++* Pull request: #647 ++* Date: 2026-08-02 ++* Iterations: 3 ++ ++## User Request Fulfillment ++ ++* Complete: review corrections committed, pushed, and both PR threads resolved ++* Complete: GA4 stream operation, GSC verification, root sitemap submission, and GA4-to-GSC link are owner-confirmed; baseline transcription and consent evidence remain separate measurement work ++* Partial, owner-gated: automated acceptance evidence is current and owner actions are implementation-ready; manual and protected-environment approvals remain pending ++* Complete: repository-page, dynamic-topic, and generation-cost work has an implementation-ready gated plan ++ ++## Executive Findings ++ ++1. Production GA configuration is present through the protected secret path. The empty checked-in Hugo value is an intentional fork-safe default, not evidence of disconnection. ++2. GSC is verified through the owner's selected method; the optional HTML-tag secret path is not required. The submitted root sitemap is a complete URL set, so no child sitemap submissions exist. ++3. SEC-01 implementation is complete and tested; SEC-04 has comprehensive lifecycle fixtures. Hermes still owns policy verification and NFR-004 sign-off. ++4. Accessibility automation is strong, but NFR-005 lacks manual keyboard and screen-reader evidence. ++5. Real Podcaster generation and environment-bound smoke evidence exist separately. The environment has no protection rules, and no run combines approval with real generation. ++6. #622 is explicitly non-blocking polish. #626 is independent hardening with unchanged thresholds. Both can affect final visual recapture but are not approval evidence. ++7. Both rollout flags remain disabled. Repository pages lack production stable IDs and lifecycle transitions; dynamic creation lacks a non-mutating preview and has five eligible candidates. ++8. Hugo/Pagefind timing separation is complete. Q-01 still needs comparable workload variants, retained samples, aggregation, and budget approval. ++9. The standalone embed currently renders the same secret-backed GA configuration as the main site. A dedicated contract now prevents its separate base template from dropping analytics or consent wiring. ++10. Squad implemented SEC-02/03/05. The first SEC-02 browser assertion was rejected, revised to prove cross-origin default-off behavior, then approved by Fry. Human Hermes disposition remains pending. ++ ++## Validation ++ ++* `python3 -m pytest -q tests/`: 1,392 passed, 19 skipped, 34 subtests passed ++* Focused analytics, topic, lifecycle, export, and link suite: 45 passed, 4 Hugo-dependent skips ++* `ruff check .` and `ruff format --check .`: passed ++* Data-page, public dataset, and trend-export checks: passed ++* Two rollout flags confirmed disabled ++* Editor diagnostics: no errors ++* `git diff --check`: passed ++* PR #647 checks for `8fddceb`: 13 successful checks ++ ++## Blockers Requiring Named Owners ++ ++* jmservera: GSC export transcription, production consent observations, sitemap processing review, and separate rollout decisions ++* Hermes: SEC-01 through SEC-06 dispositions and NFR-004 sign-off ++* URL and repository administrator: protected real-generation environment and secret scope ++* Podcaster maintainer: idempotency confirmation or one-run authorization ++* Fry and accessibility reviewer: manual NFR-005 evidence ++* Amy: refreshed visual matrix and acceptance conclusion ++ ++## Overall Status ++ ++Complete for repository-executable work and planning. External acceptance remains owner-gated and is not falsely marked complete. The branch is ready to publish and validate through PR CI. +diff --git a/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md +new file mode 100644 +index 0000000..90023eb +--- /dev/null ++++ b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md +@@ -0,0 +1,40 @@ ++ ++# Review: Claracle Relaunch Readiness Reconciliation ++ ++## Review Metadata ++ ++* Plan: .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md ++* Pull request: #647 ++* Reviewer: RPI Agent ++* Date: 2026-08-02 ++* Iterations: 1 ++ ++## User Request Fulfillment ++ ++* Complete: reviewed the three relaunch plans, PRD, and BRD against delivered repository state ++* Complete: reconciled stale plan checklists and created one status of record ++* Complete: aligned PRD v1.3 and BRD v1.2 with the delivered #627-#646 workstream ++* Complete: consolidated pending launch gates with owners, dependencies, and evidence paths ++* Complete: captured deferred implementation work as follow-on items ++ ++## Review Findings ++ ++PR review identified two stale state claims. Issues #599 and #644 closed as completed on 2026-08-01, but the initial research still called both open and the status register presented #599 as a pending issue anchor. The corrected record distinguishes issue disposition from acceptance state: #599 is closed, while its human-action checklist and FR-035 acceptance evidence remain pending. ++ ++No placement or architecture concerns remain. The status of record is the owning readiness view, while research, plans, and logs retain supporting traceability. ++ ++## Validation ++ ++* `pytest -q tests/test_internal_link_checker.py tests/test_embed_sources.py`: 10 passed ++* PR #647 checks: 13 successful checks, no failures, no approval requirement, and no requested changes ++* Local `hugo --minify`: unavailable because Hugo is not installed; PR #647 Production site check passed ++* Editor diagnostics: no errors in the corrected files ++* `git diff --check`: passed ++ ++## Pull Request Threads ++ ++Both review comments are addressed locally. The threads remain unresolved until the correction commit is pushed so reviewers can inspect the updated PR diff. ++ ++## Overall Status ++ ++Complete. The implementation fulfills the recorded user requests and local validation passes. Committing and pushing the review corrections is the only remaining delivery action. +\ No newline at end of file +diff --git a/content/privacy/_index.md b/content/privacy/_index.md +index 58a03e8..ad32206 100644 +--- a/content/privacy/_index.md ++++ b/content/privacy/_index.md +@@ -29,6 +29,12 @@ Claracle uses Google Analytics 4 (GA4) **only if you accept the analytics catego + + After consent, GA4 helps us understand whether the site is useful: page views, referrers, session duration, device/browser information, and approximate location derived from network data. The GA4 measurement ID is configured per deployment through a repository secret, not hard-coded in this page. Google's processing is governed by [Google's Privacy Policy](https://policies.google.com/privacy). You can also use the [Google Analytics opt-out browser add-on](https://tools.google.com/dlpage/gaoptout). + ++### Charts embedded on other sites ++ ++Claracle's official chart iframe snippet uses `referrerpolicy="no-referrer"`. If a publisher uses the snippet unchanged, the embedding page URL is not sent as the iframe request referrer. Publishers control their copied HTML and may alter that attribute. ++ ++Analytics in an embedded Claracle chart starts off. It can be enabled only when you explicitly accept Claracle analytics in the consent controls shown inside the iframe. A choice made on the embedding website is not treated as Claracle consent. Some browsers block third-party storage, so an iframe choice may not persist and the prompt may reappear; storage failure does not turn analytics on. ++ + ### Google Fonts + + Claracle loads Inter and JetBrains Mono from Google Fonts. When your browser requests those font files, Google may receive request metadata such as your IP address and user-agent under [Google's Privacy Policy](https://policies.google.com/privacy). +@@ -103,8 +109,9 @@ GitHub and Google may process data in countries outside your own. GA4 data may b + + ## Changes to this policy + +-Last updated: 2026-06-12. Changes are announced through the git history of this page in the public SquadScope repository, so you can review what changed and when. ++Last updated: 2026-08-02. Changes are announced through the git history of this page in the public SquadScope repository, so you can review what changed and when. + ++**2026-08-02:** Documented the no-referrer iframe snippet and frame-local, explicit analytics consent model. + **2026-06-12:** Added Signal Check podcast section covering TTS provider, staging storage, and platform disclosures. + + ## Contact +diff --git a/docs/brds/claracle-data-observatory-relaunch-brd.md b/docs/brds/claracle-data-observatory-relaunch-brd.md +index 1938c68..5e39c0b 100644 +--- a/docs/brds/claracle-data-observatory-relaunch-brd.md ++++ b/docs/brds/claracle-data-observatory-relaunch-brd.md +@@ -2,7 +2,7 @@ + title: "Claracle Data Observatory Relaunch — Business Requirements Document" + description: "BRD for the next version of the Claracle site, repositioning it from weekly AI-generated summaries into a discoverable, linkable public database of GitHub technology trends to solve the discovery/SEO problem." + author: "BRD Builder (facilitated)" +-ms.date: 2026-07-30 ++ms.date: 2026-08-02 + ms.topic: reference + --- + +@@ -12,17 +12,25 @@ ms.topic: reference + |-------|-------| + | BRD ID | BRD-CLARACLE-002 | + | Status | Acceptance and sponsor approval pending | +-| Version | 1.1 | ++| Version | 1.2 | + | Author | BRD Builder (facilitated) | + | Sponsor | jmservera (also the human approval authority) | +-| Last updated | 2026-07-29 | ++| Last updated | 2026-08-02 | + | Related repositories | SquadScope, SquadScope-Podcaster, SquadScope-Coordinator | + ++### Change History ++ ++| Version | Date | Author | Summary | ++|---------|------|--------|---------| ++| 1.0 | 2026-07-29 | BRD Builder (facilitated) | Initial BRD repositioning Claracle into a data observatory | ++| 1.1 | 2026-07-30 | BRD Builder (facilitated) | Reconciled acceptance status with pending security, analytics, production, Podcaster, accessibility, visual, and rollout gates | ++| 1.2 | 2026-08-02 | SquadScope Squad | Added this change history, aligned the PRD cross-reference, linked the status of record, and reconciled the completed GA4/GSC connection | ++ + --- + + ## Acceptance Status + +-The business requirements remain approved as requirements, but no repository artifact records sponsor approval to enable either rollout flag. Security sign-off, analytics and search evidence, production responses, Podcaster execution, accessibility review, and refreshed visual acceptance remain pending. Dynamic topic creation and repository-page creation must be approved separately. ++The business requirements remain approved as requirements, but no repository artifact records sponsor approval to enable either rollout flag. The GA4/GSC connection is complete by owner confirmation; dated baseline values and production consent observations remain pending. Security sign-off, remaining production responses, Podcaster execution, accessibility review, and refreshed visual acceptance also remain pending. Dynamic topic creation and repository-page creation must be approved separately. Current delivered-versus-pending status is tracked in the [relaunch status of record](../review/data-observatory-relaunch/status-of-record.md). + + --- + +@@ -316,7 +324,7 @@ Aligned to the SEO analysis phasing; final sequencing is a delivery decision. + Resolved in elicitation (2026-07-29): + + - ✅ Sponsor and approval authority: **jmservera** (human approval authority). +-- Analytics measurement stack: **GA4 + GSC selected**; property state and numeric baseline remain pending external evidence. ++- ✅ Analytics measurement stack and connection: **GA4 + GSC selected and connected**; the numeric baseline and production consent observations remain pending. + - ✅ Wave 1 topic hubs: AI Coding Agents, MCP Ecosystem, Open-Source LLMs, Developer Tools, plus one vertical (e.g., AI Agents in Healthcare) — kept **trend-aligned and dynamic** (BR-004). + - ✅ Dataset licensing: **MIT**, cite all references (BR-050). + - ✅ Free tool: **client-side-only**, specific tool chosen via design spike (BR-052). +@@ -330,5 +338,5 @@ Still open: + + 1. Quantify **incremental generation cost/time** for hubs, data, and repository pages. + 2. Record Hermes security sign-off and disposition of open review findings. +-3. Capture GA4, GSC, production, Podcaster, accessibility, and visual acceptance evidence. ++3. Capture the GA4/GSC dated baseline, production consent observations, remaining production responses, Podcaster, accessibility, and visual acceptance evidence. + 4. Obtain separate sponsor approval before enabling dynamic topic creation or repository-page creation. +diff --git a/docs/data-observatory-runbook.md b/docs/data-observatory-runbook.md +index 808edea..db41ba7 100644 +--- a/docs/data-observatory-runbook.md ++++ b/docs/data-observatory-runbook.md +@@ -171,17 +171,22 @@ Review these signals after an approved production deployment: + ## Cross-origin embed privacy + + Claracle's consent controls govern scripts, cookies, and telemetry on the Claracle origin. They +-cannot inspect, suppress, or withdraw storage and network activity initiated inside a +-cross-origin iframe. The repository does not currently contain approved evidence that a +-third-party embed receives or honors Claracle's consent state. +- +-Treat analytics-bearing cross-origin embeds as blocked until Hermes records an approved consent +-and referrer policy with reproducible browser evidence. Before publication, test the fresh, +-denied, granted, reloaded, and withdrawn consent states while capturing iframe requests and +-storage behavior. If an embed sends requests before approval, continues after withdrawal, or +-cannot be inspected reliably, remove or disable the embed, stop publication of the affected +-page, preserve the request evidence and deployed revision, and escalate to Hermes and URL. Do +-not describe the embed as consent-compliant while that review is pending. ++cannot inherit or inspect consent collected by an embedding site. Use the generated official ++snippet unchanged: it includes `referrerpolicy="no-referrer"`. Publishers can alter copied iframe ++attributes, so production review must inspect the markup actually deployed by the publisher. ++ ++The iframe starts Claracle analytics disabled. Only an explicit analytics choice in the Claracle ++consent UI inside that frame can enable telemetry; parent-page consent is not inferred or ++transferred. Third-party storage blocking may prevent the choice from persisting and cause another ++prompt, but it must never enable analytics. ++ ++Before publication, test the fresh, denied, granted, reloaded, and withdrawn frame-local consent ++states while capturing iframe requests and storage behavior. Confirm the deployed iframe retains ++`no-referrer`. If an embed sends requests before Claracle consent, continues after withdrawal, or ++cannot be inspected reliably, remove or disable the embed, stop publication of the affected page, ++preserve the request evidence and deployed revision, and escalate to Hermes and URL. Repository ++tests establish the implementation contract; Hermes approval and production evidence remain ++pending. + + ## Failed-run recovery + +diff --git a/docs/growth/ga4-gsc-baseline-2026-07-29.md b/docs/growth/ga4-gsc-baseline-2026-07-29.md +index 43161fa..366f1d4 100644 +--- a/docs/growth/ga4-gsc-baseline-2026-07-29.md ++++ b/docs/growth/ga4-gsc-baseline-2026-07-29.md +@@ -2,7 +2,7 @@ + title: GA4 and GSC Launch Baseline for 2026-07-29 + description: Dated Claracle analytics and search baseline record that separates repository wiring from pending external platform evidence + author: SquadScope Squad +-ms.date: 2026-07-30 ++ms.date: 2026-08-02 + ms.topic: reference + keywords: + - google analytics 4 +@@ -14,22 +14,34 @@ estimated_reading_time: 6 + + ## Baseline status + +-The 2026-07-29 launch baseline has not been captured from GA4 or Google Search +-Console. No numeric baseline value is asserted in this record. The repository contains +-conditional integration paths, but repository inspection cannot prove property setup, +-secret presence, consent behavior in production, data receipt, verification, sitemap +-submission, or indexing. ++The production GA4 stream and Google Search Console property are connected. On ++2026-08-02, jmservera confirmed the intended GA4 stream, verified the GSC property, ++submitted the root sitemap, and linked the GA4 stream to GSC. A Search Console ++performance export was supplied for the dated baseline, but its numeric values have not ++yet been transcribed into this record. + + The configured production target is `https://claracle.com/`, and the expected standard +-sitemap target is `https://claracle.com/sitemap.xml`. Their production responses remain +-unverified for this acceptance record. ++sitemap target is `https://claracle.com/sitemap.xml`. ++ ++## Credential-free production observations ++ ++| Observation | Result | Date | Acceptance boundary | ++| ----------- | ------ | ---- | ------------------- | ++| GA configuration rendered | Present | 2026-08-02 | Does not reveal or validate the identifier, property, stream, consent behavior, or receipt | ++| GSC verification meta tag | Absent | 2026-08-02 | GSC ownership remains unverified | ++| Sitemap response | HTTP 200, `application/xml` | 2026-08-02 | Does not prove submission or processing in GSC | ++| `GA_MEASUREMENT_ID` secret name | Present | 2026-08-02 | Secret value is not observable and must not be recorded | ++| GSC property verification | Complete by owner attestation | 2026-08-02 | Property verified without requiring a public verification meta tag | ++| Root sitemap submission | Complete by owner attestation | 2026-08-02 | Root `sitemap.xml` is a complete ``, not an index of child sitemaps | ++| GA4 and GSC product link | Complete by owner attestation | 2026-08-02 | Does not replace GA4 or GSC baseline values | ++| Standalone embed GA configuration | Present | 2026-08-02 | The affected embed renders the same secret-backed configuration as the main site | + + ## Repository-verifiable implementation + + | Surface | Repository status | Evidence boundary | + | -------------------------- | ----------------------------------------- | ----------------------------------------------------------------------------------------------- | +-| GA4 build parameter | Implemented conditionally | `deploy-site.yml` reads `GA_MEASUREMENT_ID`; secret existence and value are not observable | +-| GSC verification parameter | Implemented conditionally | `deploy-site.yml` reads `GSC_SITE_VERIFICATION`; verification is not observable | ++| GA4 build parameter | Implemented and present in production | `deploy-site.yml` reads `GA_MEASUREMENT_ID`; public presence does not validate the protected value | ++| GSC verification parameter | Available but not required by the completed verification method | `deploy-site.yml` supports `GSC_SITE_VERIFICATION` when HTML-tag verification is selected | + | Fork-safe defaults | Implemented | Hugo parameters default empty, so an unconfigured build does not inherit production identifiers | + | Analytics consent gate | Implemented in templates and browser code | Production requests and cookies require browser evidence | + | Observatory events | Implemented with a consent check | GA4 receipt and payload inspection require browser and Realtime evidence | +@@ -43,13 +55,14 @@ the secret. + + | Evidence | Status | Owner | Required proof | + | --------------------------- | ------- | -------------------- | -------------------------------------------------------------------------------- | +-| GA4 property and web stream | Pending | jmservera | Dated property or stream evidence with identifiers redacted where appropriate | ++| GA4 property and web stream | Complete | jmservera | Owner confirmed the intended production stream on 2026-08-02 | + | Consent denied behavior | Pending | jmservera and Hermes | Private first-visit network and cookie evidence showing no GA4 request or cookie | + | Consent granted behavior | Pending | jmservera and Hermes | Network evidence showing the expected GA4 request after consent | +-| GA4 Realtime receipt | Pending | jmservera | Dated Realtime evidence correlated to the consented test visit | +-| GSC property verification | Pending | jmservera | Dated verified-property evidence | +-| GSC sitemap submission | Pending | jmservera | Submission URL, date, and platform status | +-| Production sitemap response | Pending | jmservera | Dated response status and content-type evidence | ++| GA4 Realtime receipt | Complete by owner attestation | jmservera | GA4 declared operational on 2026-08-02; retain a redacted platform capture if formal audit evidence is required | ++| GSC property verification | Complete | jmservera | Owner confirmed verified property on 2026-08-02 | ++| GSC sitemap submission | Complete | jmservera | `https://claracle.com/sitemap.xml` submitted on 2026-08-02 | ++| GA4 and GSC product link | Complete | jmservera | Owner confirmed the production stream is linked to the GSC property | ++| Production sitemap response | Complete | jmservera | HTTP 200 with `application/xml` observed on 2026-08-02 | + | Production feed responses | Pending | jmservera | Dated site and topic feed response status and content types | + + ## Baseline values +@@ -76,13 +89,15 @@ values with estimates such as “near zero.” + 3. Record consent-denied network and cookie behavior in a private browser session. + 4. Grant analytics consent and record the expected request. + 5. Correlate that visit with GA4 Realtime and record the observation date. +-6. Verify the GSC property, submit the configured sitemap, and record platform status. +-7. Capture GA4 acquisition values and GSC performance values for the same documented ++6. Confirm the submitted sitemap is processed successfully and review indexed versus ++ excluded URLs in GSC. ++7. Capture GA4 acquisition values and transcribe the supplied GSC performance export for the same documented + baseline window. + 8. Link the evidence from the relaunch review index and retain redacted artifacts in the + approved evidence location. + + ## Acceptance rule + +-NFR-007, FR-035, and the analytics portion of NFR-008 remain pending. They may be marked +-accepted only after the external evidence matrix contains dated proof and actual values. ++FR-035 connection and submission are complete. NFR-007 baseline measurement and the ++production-consent portion of NFR-008 remain pending until the supplied performance ++export is transcribed and dated consent observations are retained. +diff --git a/docs/prds/claracle-data-observatory-relaunch.md b/docs/prds/claracle-data-observatory-relaunch.md +index af076e6..927f2d9 100644 +--- a/docs/prds/claracle-data-observatory-relaunch.md ++++ b/docs/prds/claracle-data-observatory-relaunch.md +@@ -2,12 +2,12 @@ + title: Claracle Data Observatory Relaunch Product Requirements Document + description: Product requirements, delivery state, rollout controls, risks, and acceptance gates for the Claracle Data Observatory relaunch + author: SquadScope Squad +-ms.date: 2026-07-31 ++ms.date: 2026-08-02 + ms.topic: reference + --- + + +-Version 1.2 | Status Acceptance pending | Owner jmservera | Team SquadScope Squad | Target Wave 1 (foundation) | Lifecycle Definition ++Version 1.3 | Status Acceptance pending | Owner jmservera | Team SquadScope Squad | Target Wave 1 (foundation) | Lifecycle Definition + + ## Progress Tracker + | Phase | Done | Gaps | Updated | +@@ -18,14 +18,14 @@ Version 1.2 | Status Acceptance pending | Owner jmservera | Team SquadScope Squa + | Requirements | Yes | Incremental generation cost still to quantify | 2026-07-30 | + | Metrics & Risks | Yes | None | 2026-07-29 | + | Operationalization | Yes | Star Velocity Explorer selected; production evidence pending | 2026-07-30 | +-| Finalization | No | Security, external platform, Podcaster, accessibility, visual, and sponsor gates remain open | 2026-07-30 | +-Unresolved launch gates: 6 | TBDs: 1 (incremental generation cost) ++| Finalization | No | Security, baseline and consent, Podcaster, accessibility, visual, and sponsor gates remain open | 2026-08-02 | ++Unresolved launch gates: See the launch-gate register | TBDs: 1 (incremental generation cost) + + ## Acceptance Status + +-Repository implementation is present, but GA acceptance is pending. Hermes has not signed NFR-004; GA4/GSC, production responses, external schema and social debuggers, a downstream Podcaster run, accessibility review, and refreshed visual evidence are not recorded. `dynamic_topic_creation` and `repo_pages` remain off and require separate sponsor-approved rollout changes. ++Repository implementation is present, and the GA4/GSC connection is complete by owner confirmation. Dated baseline values and production consent observations remain pending. Hermes has not signed NFR-004; remaining production responses, external schema and social debuggers, a downstream Podcaster run, accessibility review, and refreshed visual evidence are not recorded. `dynamic_topic_creation` and `repo_pages` remain off and require separate sponsor-approved rollout changes. Delivered-versus-pending status and the launch-gate register are tracked in the [relaunch status of record](../review/data-observatory-relaunch/status-of-record.md). + +-Derived from: `docs/brds/claracle-data-observatory-relaunch-brd.md` (BRD-CLARACLE-002, v1.0). ++Derived from: `docs/brds/claracle-data-observatory-relaunch-brd.md` (BRD-CLARACLE-002, v1.2). + + ## 1. Executive Summary + ### Context +@@ -135,9 +135,9 @@ Extends the existing PaperMod theme and topic layouts. New page types: topic hub + | FR-032 | OG + Twitter cards | Emit Open Graph and Twitter/X tags including image, `og:image:alt`, width/height, `article:author`, `twitter:creator`, plus a homepage/fallback OG image. | G-005 | Data citer | Must | FB/Twitter debuggers render valid previews with image on homepage and articles; closes gaps 1-7 in distribution-strategy.md. | Uses existing `default_social_image` param. | + | FR-033 | Structured data | Article pages emit Schema.org Article; hub/data/repo pages emit appropriate schema; all hierarchical pages emit Breadcrumb schema. | G-005,G-004 | Search visitor | Must | Google Rich Results Test validates Article + Breadcrumb with no errors. | Extends `schema_json.html`. | + | FR-034 | Sitemap + RSS | Publish Hugo's built-in `sitemap.xml` and RSS feeds (site-wide + per topic). No news sitemap. | G-005,G-002 | Search visitor | Must | Sitemap and feeds reachable/valid; per-topic feeds resolve. | Hugo `outputs` already emit taxonomy RSS. | +-| FR-035 | Search Console | Connect and verify Google Search Console (and confirm GA4) for the production domain; submit sitemap. | G-002,G-004 | Search visitor | Must | GSC property verified, sitemap submitted; GA4 receiving data. | `ga_measurement_id` currently empty. | ++| FR-035 | Search Console | Connect and verify Google Search Console (and confirm GA4) for the production domain; submit sitemap. | G-002,G-004 | Search visitor | Must | GSC property verified, sitemap submitted; GA4 receiving data. | Complete by owner confirmation on 2026-08-02: GA4 stream operational, GSC verified, root sitemap submitted, and products linked. | + | FR-040 | Internal linking | Every weekly article links to previous/next week, its topic hubs, and referenced repository/technology pages. | G-001,G-004 | Signal-seeker | Must | Rendered weekly pages contain prev/next, topic-hub, and repo links where applicable. | PaperMod `ShowPostNavLinks` already on. | +-| FR-041 | Link-check gate | Validate internal links in CI; broken internal links fail the build gate. | G-005 | - | Should | CI link-check runs and fails on broken internal links. | New CI step. | ++| FR-041 | Link-check gate | Validate internal links in CI; broken internal links fail the build gate. | G-005 | - | Should | CI link-check runs and fails on broken internal links. | Partial: satisfied at test level (`tests/test_internal_link_checker.py`); no standalone CI link tool. | + | FR-050 | Downloadable datasets | Offer MIT-licensed downloadable datasets (e.g., CSV of top projects, weekly trend archive) with all sources cited and a stable link. | G-003 | Data citer | Should | >= 1 dataset published under MIT with citation/attribution note and stable download URL. | Maps BR-050. | + | FR-051 | Embeddable charts | Generate charts (growth curves, rankings, momentum) with an embed snippet that links back to Claracle. | G-003 | Data citer | Should | >= 1 chart type embeddable via provided snippet with backlink. | Shortcode/layout, not raw HTML. | + | FR-052 | Client-side tool | Provide >= 1 free, client-side-only interactive tool (e.g., trend explorer, star-velocity tracker), selected via a design spike weighing discoverability value, effort, and static-hosting fit. | G-003,G-002 | Search visitor, Data citer | Could | Design spike recommends one tool with rationale; tool ships and runs fully in-browser with no backend. | Maps BR-052. | +@@ -165,7 +165,7 @@ Discovery engine + | NFR ID | Category | Requirement | Metric/Target | Priority | Validation | Notes | + |--------|----------|------------|--------------|----------|-----------|-------| + | NFR-001 | Performance | New page types keep the site fast on static hosting | Lighthouse Performance >= 90 on hub/data/repo pages | Should | Lighthouse CI or manual audit | Static pre-rendered; charts lazy-loaded | +-| NFR-002 | Reliability | No regression to weekly pipeline or handoff | Weekly pipeline success + handoff smoke unchanged (G-006) | Must | `podcaster-handoff-smoke.yml`, pytest | Crawl untouched | ++| NFR-002 | Reliability | No regression to weekly pipeline or handoff | Weekly pipeline success + handoff smoke unchanged (G-006) | Must | `podcaster-handoff-smoke.yml`, pytest | Crawl untouched; restore mode preserves the published weekly transaction (article, summary, promotion record, rollups) rather than overwriting provenance (`#640`/`#646`) | + | NFR-003 | Maintainability | Thresholds are configuration, not code | Topic (FR-004) and repo (FR-021) thresholds set via config | Must | Change threshold with no code edit in review | | + | NFR-004 | Security | No raw HTML in AI content; no secrets in client tool | `unsafe=false` retained; tool ships no secrets; inputs sanitized | Must | Build config check; Hermes review; `sanitize_repo_content` path | Follows prompt-injection guardrails | + | NFR-005 | Accessibility | New pages meet WCAG 2.1 AA basics | Images have alt; charts have text alternative; contrast passes | Should | axe/Lighthouse a11y audit | OG alt also required (FR-032) | +@@ -208,7 +208,7 @@ Generated Hugo content: topic hubs (taxonomy terms), data pages, repository page + | `generate_content.py` | Internal code | High | Farnsworth | Regression on frontmatter | Add `topics` behind tests (FR-002) | + | Hugo `topic` taxonomy + `layouts/topics` | Platform | High | Amy | Layout gaps | Extend existing templates | + | SEO partials (`opengraph`, `twitter_cards`, `schema_json`) | Platform | Medium | Amy | Incomplete tags | Close catalogued gaps (FR-032/033) | +-| GA4 + GSC | External | High | jmservera | Access/verification | Verify property early (FR-035) | ++| GA4 + GSC | External | High | jmservera | Baseline and consent evidence | Capture dated values and production consent observations (FR-035, NFR-007/008) | + | Podcaster handoff contract | Cross-repo | High | URL/Hermes | Contract break | Keep payload unchanged; smoke test | + | Client-side charting/tool library | External | Medium | Amy | Static-hosting fit | Design spike (FR-051/052) | + +@@ -265,6 +265,8 @@ AI-generated content must not render raw HTML (`unsafe=false`); repo-derived tex + | dynamic_topic_creation | Gate auto-creation of new topic hubs | Off; no rollout approval recorded | Separate sponsor approval after security and acceptance evidence | + | repo_pages | Gate repository-page generation | Off; no rollout approval recorded | Separate sponsor approval after lifecycle and acceptance evidence | + ++Owners, dependencies, and evidence paths for every launch gate (including sponsor rollout approval) are consolidated in the [relaunch status of record launch-gate register](../review/data-observatory-relaunch/status-of-record.md#launch-gate-register). ++ + ### Communication Plan (Optional) + Use the existing per-week distribution playbook (`docs/growth/distribution-strategy.md`) for launch posts; announce Wave 2 dataset and Wave 3 tool on developer communities (HN, Lobsters, dev.to) when genuinely useful. + +@@ -273,11 +275,12 @@ Use the existing per-week distribution playbook (`docs/growth/distribution-strat + |------|----------|-------|---------|--------| + | Q-01 | Quantify incremental generation cost/time for hubs, data, and repo pages | Leela | Design spike | Open | + | Q-02 | Which client-side tool to build first (FR-052) | Amy | 2026-07-30 | Resolved: Star Velocity Explorer; see ADR | +-| Q-03 | When can `content/data/` deploy hydration be restored (once the crawl reliably publishes observatory pages to `publish`)? | Bender | Post-#627 crawl run | Open | ++| Q-03 | When can `content/data/` deploy hydration be restored (once the crawl reliably publishes observatory pages to `publish`)? | Bender | Post-#627 crawl run | Resolved: hydration restored via `#637` after the crawl repopulated `publish`; CI embed-source guard (`#641`) prevents recurrence | + + ## 15. Changelog + | Version | Date | Author | Summary | Type | + |---------|------|-------|---------|------| ++| 1.3 | 2026-08-02 | SquadScope Squad | Reconciled the #627-#646 workstream: deploy/hydration parity restored and CI embed-source guard shipped (`#634`/`#637`/`#641`), Podcaster smoke hardened (`#636`/`#639`/`#643`/`#645`), restore preserves the published weekly transaction (NFR-002; `#640`/`#646`); recorded FR-041 partial status and linked the status of record | Updated | + | 1.2 | 2026-07-31 | SquadScope Squad | Recorded the deploy hydration content-provenance failure (issue #627), the interim `content/data` fix, and the deploy/CI parity requirement (NFR-011/012, R-08) | Updated | + | 1.1 | 2026-07-30 | SquadScope Squad | Reconciled repository delivery with pending external, security, visual, accessibility, Podcaster, and rollout gates | Updated | + | 1.0 | 2026-07-29 | PRD Builder (facilitated) | Initial PRD derived from BRD-CLARACLE-002 v1.0 | Created | +@@ -294,6 +297,7 @@ Use the existing per-week distribution playbook (`docs/growth/distribution-strat + | REF-7 | Analysis | External SEO analysis (user-provided, 2026-07) | Discovery-first strategy | Primary driver | + | REF-8 | ADR | `docs/decisions/adr-star-velocity-explorer.md` | FR-052 selection and static-hosting rationale | Resolves Q-02 | + | REF-9 | Review | `docs/review/data-observatory-relaunch/README.md` | Bounded acceptance evidence and pending gates | Source of release status | ++| REF-10 | Review | `docs/review/data-observatory-relaunch/status-of-record.md` | Reconciled delivered-versus-pending status and launch-gate register | Single readiness view | + + ### Citation Usage + Functional requirements cite BRD requirement IDs (BR-xxx) inline; technical claims cite the repository files above. +diff --git a/docs/review/data-observatory-relaunch/README.md b/docs/review/data-observatory-relaunch/README.md +index f10f883..f03471d 100644 +--- a/docs/review/data-observatory-relaunch/README.md ++++ b/docs/review/data-observatory-relaunch/README.md +@@ -2,7 +2,7 @@ + title: Data Observatory Relaunch Acceptance Evidence + description: Bounded evidence index for repository implementation, external launch gates, security review, and visual acceptance of the Claracle relaunch + author: SquadScope Squad +-ms.date: 2026-07-30 ++ms.date: 2026-08-02 + ms.topic: reference + keywords: + - acceptance evidence +@@ -18,8 +18,12 @@ Repository implementation evidence is available, but relaunch acceptance is inco + Dynamic topic creation and repository-page creation remain disabled in + `config/observatory.toml`. This index does not authorize either rollout. + +-External platform, production, cross-repository run, security sign-off, accessibility +-review, and visual acceptance evidence remain pending as listed below. ++External baseline and consent, remaining production responses, cross-repository run, ++security sign-off, accessibility review, and visual acceptance evidence remain pending ++as listed below. The GA4/GSC connection itself is complete. ++ ++The [owner action register](owner-action-register.md) sequences the remaining human and ++protected-environment work without treating repository automation as approval evidence. + + ## Evidence principles + +@@ -38,30 +42,37 @@ review, and visual acceptance evidence remain pending as listed below. + | FR-052 tool selection and architecture rationale | Complete | [Star Velocity Explorer ADR](../../decisions/adr-star-velocity-explorer.md) | + | Security and privacy surface review | Complete with open findings | [Security review](security-review.md) | + | Hermes security acceptance | Pending | Security review sign-off table | +-| GA4 and GSC repository wiring | Implemented conditionally | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | ++| GA4 and GSC connection | Complete | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | + | GA4 and GSC external baseline values | Pending | Dated baseline external evidence matrix | + | Product delivery and rollout status | Pending acceptance | [PRD](../../prds/claracle-data-observatory-relaunch.md) | + | Sponsor-approved lifecycle state | Pending | [BRD](../../brds/claracle-data-observatory-relaunch-brd.md) | + | Visual capture requirements | Pending | [Screenshot capture checklist](screenshots/README.md) | ++| Owner-gated acceptance actions | Pending | [Owner action register](owner-action-register.md) | + + ## External acceptance matrix + + | Gate | Status | Actor or access needed | Required evidence | + | ------------------------------------- | ------- | --------------------------------------------------- | ---------------------------------------------------------- | +-| GSC property verification | Pending | jmservera with GSC access | Dated verified-property evidence | +-| GSC sitemap submission | Pending | jmservera with GSC access | Submitted sitemap target and platform status | ++| GSC property verification | Complete | jmservera | Owner confirmed verification on 2026-08-02 | ++| GSC sitemap submission | Complete | jmservera | Root `sitemap.xml` submitted on 2026-08-02 | + | GA4 consent-denied behavior | Pending | jmservera and Hermes with production browser access | Network and cookie evidence from a private first visit | + | GA4 consent-granted behavior | Pending | jmservera and Hermes with production browser access | Expected request after consent | +-| GA4 Realtime receipt | Pending | jmservera with GA4 access | Dated Realtime observation correlated to test visit | ++| GA4 property, stream, and receipt | Complete by owner attestation | jmservera | Intended production stream confirmed operational | + | Social preview debuggers | Pending | Reviewer with external debugger access | Homepage and article conclusions with retained links | + | Rich Results Test | Pending | Reviewer with external debugger access | Article and breadcrumb conclusions with retained links | + | Schema.org validator | Pending | Reviewer with external debugger access | Relevant page-type conclusions with retained links | +-| Production sitemap and feed responses | Pending | jmservera with production access | Status, content type, date, and tested target | ++| Production sitemap response | Complete | jmservera | HTTP 200 `application/xml` observed on 2026-08-02 | ++| Production feed responses | Pending | jmservera with production access | Status, content type, date, and tested target | + | Podcaster downstream run | Pending | Podcaster maintainer and protected environment | Successful downstream run conclusion and Actions link | + | Accessibility review | Pending | Fry and accessibility reviewer | Automated results plus keyboard and screen-reader findings | + | Hermes sign-off | Pending | Hermes | Dated disposition of security findings and NFR-004 | + | Sponsor rollout approval | Pending | jmservera | Dated approval identifying each flag separately | + ++Issue #622 is non-blocking UX polish according to its issue contract. Issue #626 is ++independent quality hardening whose existing thresholds remain unchanged. Both should be ++completed before final visual recapture where their changes affect the rendered result, ++but neither is represented as an unevidenced acceptance approval. ++ + ## Visual evidence status + + The existing ten PNG files are retained as historical local captures. Their current index +diff --git a/docs/review/data-observatory-relaunch/owner-action-register.md b/docs/review/data-observatory-relaunch/owner-action-register.md +new file mode 100644 +index 0000000..2801d92 +--- /dev/null ++++ b/docs/review/data-observatory-relaunch/owner-action-register.md +@@ -0,0 +1,133 @@ ++--- ++title: Data Observatory Relaunch Owner Action Register ++description: Sequenced owner actions and evidence requirements for Claracle relaunch gates that cannot be completed by repository automation ++author: SquadScope Squad ++ms.date: 2026-08-02 ++ms.topic: reference ++keywords: ++ - launch gates ++ - acceptance evidence ++ - owner actions ++ - rollout approval ++estimated_reading_time: 7 ++--- ++ ++## Purpose ++ ++Repository automation proves implementation behavior, not external platform state or ++human approval. This register defines the remaining actions, actors, and completion ++evidence without recording secret values or granting approval by implication. ++ ++## Analytics and search acceptance ++ ++Owner: jmservera, with Hermes reviewing production consent behavior. ++ ++Current evidence: ++ ++* Production renders secret-backed GA configuration on the main site and standalone embed ++* The `GA_MEASUREMENT_ID` secret name exists ++* Production serves `https://claracle.com/sitemap.xml` as HTTP 200 and `application/xml` ++* jmservera confirmed the intended GA4 stream, verified GSC property, root sitemap submission, and GA4-to-GSC product link on 2026-08-02 ++* The root sitemap is one complete `` rather than a sitemap index, so there are no child sitemaps to submit ++ ++Required actions: ++ ++1. Record denied and granted production consent behavior without exposing identifiers. ++2. Transcribe the supplied GSC performance export and capture GA4 values for one explicit date range. ++3. Confirm GSC finishes processing the sitemap and review indexed and excluded URL counts. ++4. Update the [dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) with redacted conclusions and actual values. ++ ++Completion evidence still needed: consent observations, processed sitemap conclusion, ++numeric baseline date range, and reviewer/date. ++ ++## Security acceptance ++ ++Owners: Hermes, URL, and jmservera. ++ ++Required actions: ++ ++1. Hermes records a disposition for SEC-01 through SEC-06 in the [security review](security-review.md). ++2. Review the implemented SEC-02 no-referrer and frame-local consent model, including its publisher-markup and third-party-storage limitations. ++3. Approve, reject, or amend the SEC-03 exact public export field and source-path allowlists. ++4. Approve, reject, or require additional controls for the SEC-05 defense-in-depth accepted-risk recommendation; no acceptance is currently recorded. ++5. URL reviews protected workflow and secret scope after the real Podcaster environment change. ++6. jmservera records the production-owner conclusion after external evidence is linked. ++ ++Completion evidence: dated sign-off rows with finding-level dispositions and linked test, ++workflow, or production observations. ++ ++## Accessibility acceptance ++ ++Owners: Fry and a named accessibility reviewer. ++ ++Required actions: ++ ++1. Identify the tested revision, production URLs, browser, operating system, screen reader, and viewport. ++2. Review the retained axe and responsive reports from the final CI revision. ++3. Complete keyboard-only navigation for primary navigation, consent, filters, charts, tools, and related links. ++4. Complete screen-reader review for headings, landmarks, labels, status changes, chart alternatives, and errors. ++5. Record each finding, severity, disposition, reviewer, and date. ++ ++Completion evidence: a retained review record combining automated results with keyboard ++and screen-reader conclusions. Automated axe success alone does not close NFR-005. ++ ++## Protected real Podcaster run ++ ++Owners: URL, Hermes, a repository administrator, the Podcaster maintainer, and the ++environment approver. ++ ++Current evidence is split: run `30202586031` proves real accepted generation, while run ++`30721575540` proves an environment-bound dry run. The `podcaster-release-smoke` ++environment has no protection rules, and the real workflow is not environment-bound. ++ ++Required actions: ++ ++1. Confirm downstream idempotency or authorize one exact eligible week and manifest. ++2. Define required reviewers and branch policy for a real-generation environment. ++3. Bind the real generation job to that environment in a separately reviewed workflow change. ++4. Review secret scope without recording secret values. ++5. Approve and execute one real run. ++6. Retain the approver, week, manifest run ID, article digest, Actions URL, downstream job ID, and final conclusion. ++ ++Completion evidence: one successful real downstream run after environment approval. ++ ++## Visual acceptance ++ ++Owner: Amy or another named visual reviewer. ++ ++Complete issue #622's factual checks and any layout-affecting #626 work before capture. ++Then follow the [screenshot capture checklist](screenshots/README.md) against the final ++revision with populated content, resolved consent state, desktop and mobile viewports, ++light and dark themes, interaction states, and a dated reviewer conclusion. ++ ++Completion evidence: replacement visual matrix with revision metadata and an explicit ++accept or reject conclusion. Screenshots alone are not approval. ++ ++## External metadata and feed validation ++ ++Owners: Amy for rendered metadata and jmservera for production access. ++ ++Required actions: ++ ++1. Validate the homepage and one representative article in supported social preview debuggers. ++2. Validate representative article and breadcrumb markup with Google Rich Results Test. ++3. Validate each relevant page type with Schema.org Validator. ++4. Record HTTP status and content type for the site and topic feeds in production. ++5. Retain the tested URLs, revision, tool conclusions, reviewer, and date without exposing credentials. ++ ++Completion evidence: dated social preview, structured-data, and production feed ++conclusions with retained links or redacted records. ++ ++## Sponsor rollout decision ++ ++Owner: jmservera. ++ ++Record a separate decision for each flag. Do not use one blanket approval. ++ ++| Flag | Decision | Reviewed revision and evidence | Conditions | Date | ++| ---- | -------- | ------------------------------ | ---------- | ---- | ++| `dynamic_topic_creation` | Pending | Pending | Security disposition and approved canary required | Pending | ++| `repo_pages` | Pending | Pending | Stable identity and lifecycle evidence required | Pending | ++ ++Completion evidence: dated approve, reject, or defer decisions identifying the exact ++revision, evidence, conditions, and rollback owner for each flag. +diff --git a/docs/review/data-observatory-relaunch/security-review.md b/docs/review/data-observatory-relaunch/security-review.md +index 60569a5..fba8782 100644 +--- a/docs/review/data-observatory-relaunch/security-review.md ++++ b/docs/review/data-observatory-relaunch/security-review.md +@@ -2,7 +2,7 @@ + title: Data Observatory Relaunch Security Review + description: Repository security and privacy review of Observatory generation, lifecycle, datasets, embeds, browser tools, analytics, and deployment secrets + author: SquadScope Squad +-ms.date: 2026-07-30 ++ms.date: 2026-08-02 + ms.topic: reference + keywords: + - security review +@@ -14,7 +14,7 @@ estimated_reading_time: 10 + + ## Review status + +-Repository review is complete as of 2026-07-30. Hermes review and sign-off are pending. ++Repository review was reconciled with current controls on 2026-08-02. Hermes review and sign-off are pending. + NFR-004 is not accepted, and the relaunch security gate remains open until Hermes records a + disposition for every open finding. + +@@ -54,17 +54,25 @@ introduce arbitrary raw HTML through normal rendering. + Residual risk remains because phrase matching cannot identify every semantic injection. New external + fields must pass through the same sanitization and boundary path before prompt use. + ++**SEC-05 recommendation for human decision:** accept the semantic false-negative risk only as a ++defense-in-depth residual risk while retaining input sanitization, untrusted-content fencing, closing ++prompt constraints, canary leak detection, output and frontmatter validation, prompt lint, and the ++red-team corpus. Phrase matching detects known lexical patterns; it cannot reliably identify novel ++wording, translation, encoding, or semantic paraphrases with equivalent intent. The retained controls ++reduce the chance that one miss reaches publication but do not prove semantic detection. This is an ++implementation-supported recommendation, not an accepted risk; Hermes must approve, reject, or ++require an additional semantic classifier. ++ + ### Candidate-title abuse + + Candidate discovery combines repository-controlled topics, weekly tags, and analyzed headings. +-Canonical and ignored terms reduce noise, and dynamic creation is disabled. However, +-`manage_topic_hubs.py` constructs generated hub frontmatter and prose from the candidate title with +-quote escaping rather than `sanitize_text()` or structured YAML serialization for the complete +-document. ++Canonical and ignored terms reduce noise, and dynamic creation is disabled. `manage_topic_hubs.py` ++now bounds candidate titles through `sanitize_text()`, rejects line breaks, boundary markers, HTML, ++Markdown syntax, control characters, and known injection phrases, and serializes frontmatter through ++structured YAML. `tests/test_topic_hubs.py` verifies that unsafe titles fail before any mutation. + +-This path is contained while `topic_hubs.dynamic_creation.enabled = false`. It must remain disabled +-until the title is sanitized, bounded, tested with control characters and Markdown payloads, and +-reviewed by Hermes. Candidate promotion also requires a human diff review of evidence and output. ++The implementation condition for this finding is complete. Dynamic creation remains disabled until ++Hermes verifies the control and a human reviews the evidence and exact output for the approved canary. + + ### Lifecycle evidence and deletion + +@@ -77,6 +85,11 @@ The remaining risk is operator error in a lifecycle override. Review must pair t + evidence, ledger diff, aliases, generated page, and any expiry removal. Hermes must review deletion + evidence policy before NFR-004 acceptance. + ++`tests/test_observatory_repos.py` now exercises rename aliases, archive evidence, confirmed deletion, ++three-year retention, expiry removal, absence that fails closed, and stable-ID migration. These ++fixtures prove implementation behavior, not that the production corpus contains stable IDs or a ++reviewed lifecycle transition. ++ + ### Public dataset exposure + + Observatory JSON and CSV outputs are intentionally public. They contain public repository metadata, +@@ -84,21 +97,41 @@ weekly observations, topics, derived metrics, and provenance. They must not cont + private repository data, prompt transcripts, credentials, email addresses, analytics identifiers, + or local filesystem paths. + +-Publication is a data-classification boundary. Bender owns a field-level diff for new exports; +-Hermes owns privacy disposition for new fields. Derived output should remain bounded to the minimum +-needed by pages and tools. ++`scripts/export_observatory_dataset.py` now defines exact production allowlists for the CSV, ++top-repository metadata objects, and the metadata document. It also restricts `source_files` to the ++eleven expected checked-in paths under `data/raw/` and ++`data/archive/recovered-W23-W29/`. Runtime validation rejects added or missing keys, keeps ++`metadata.fields` synchronized with the CSV schema, and requires weekly count keys to equal the ++exported week list. ++ ++| Classification | Allowed fields | ++| -------------- | -------------- | ++| Public source identity | `repository`, `url`, `primary_language`, `latest_license`, `top_topics` | ++| Public observations | `latest_stars`, `first_observed_stars`, `max_forks_observed`, `seen_in_trending`, `seen_in_new` | ++| Derived public metrics | `rank_by_latest_stars`, `first_seen_week`, `last_seen_week`, `weeks_observed`, `observed_star_change` | ++| Release metadata | Dataset/version/timestamp/source/selection/license, bounded counts and rankings, exact CSV fields, allowlisted source paths, exposure statement | ++ ++Publication remains a data-classification boundary. Any new CSV, metadata, or nested-object field ++requires an intentional allowlist change, an exact-schema test update, and Hermes privacy review. ++The executable policy is implementation evidence, not approval. + + ### Embed privacy and attribution + +-Embeddable charts are static Claracle iframe endpoints with visible attribution. The provided snippet +-does not include a sandbox or `referrerpolicy` attribute. Loading the iframe can disclose the +-embedding page through normal request referrer behavior, and the embed page includes the common +-analytics partial. Consent state does not automatically cross site origins. ++Embeddable charts are static Claracle iframe endpoints with visible attribution. The official ++snippet now sets `referrerpolicy="no-referrer"`, so a publisher using it unchanged does not send the ++embedding page URL as the iframe request referrer. Publishers control their own markup and can remove ++or replace this attribute; Claracle cannot enforce the policy after a snippet is copied. ++ ++Analytics inside the iframe is frame-local, default-off, and enabled only after the visitor explicitly ++accepts Claracle analytics in the consent UI rendered inside that frame. Consent collected by the ++embedding site is neither inferred nor transferred. Browser third-party-storage restrictions may ++prevent the Claracle consent choice from persisting, which can cause the frame to ask again, but ++storage failure never enables analytics. The adapter records `chart_embed_view` only after the ++frame-local consent callback enables it. + +-The existing analytics adapter records `chart_embed_view` only when analytics consent is active in +-the frame. Hermes must decide whether embedded endpoints should omit analytics entirely or enforce a +-referrer policy and a documented consent model. Until disposition, embed privacy acceptance is +-pending. ++Rendered-snippet assertions, consent-wiring tests, and the Observatory browser analytics test provide ++repository-executable evidence for this model. Production network/storage behavior and publisher ++modifications remain outside repository control, so Hermes privacy disposition is still pending. + + ### Browser tool URL and DOM handling + +@@ -137,11 +170,11 @@ does not prove protected-environment configuration or a downstream Podcaster run + + | ID | Finding | Severity | Owner | Disposition | + | ------ | ------------------------------------------------------------------------------------------ | ------------- | --------------------- | ----------------------------------------------------------------------------------------------------------- | +-| SEC-01 | Dynamic hub candidate titles bypass the standard text sanitizer | High | Farnsworth and Hermes | Open, rollout-blocking; keep dynamic creation off, sanitize and add adversarial tests before review | +-| SEC-02 | Embed snippets omit an explicit referrer policy and cross-origin consent does not transfer | Medium | Amy and Hermes | Open; decide no-analytics embed or explicit privacy policy before acceptance | +-| SEC-03 | Public export fields need a documented allowlist to prevent future accidental expansion | Medium | Bender and Hermes | Open; review current schema and add a field-level publication policy | +-| SEC-04 | Lifecycle deletion depends on manually reviewed overrides | Medium | Bender and Hermes | Controlled by disabled flag, persisted evidence, retention, and diff review; Hermes disposition pending | +-| SEC-05 | Phrase-based injection detection has known semantic false-negative risk | Medium | Hermes and Farnsworth | Accepted only as defense in depth after Hermes review; retain fencing, canary, output validation, and tests | ++| SEC-01 | Dynamic hub candidate titles require bounded sanitization and structured serialization | High | Farnsworth and Hermes | Implemented; adversarial rejection and structured YAML are tested, Hermes verification pending | ++| SEC-02 | Embed snippets require an explicit referrer policy and cross-origin consent does not transfer | Medium | Amy and Hermes | Implemented and tested: official snippet uses no-referrer; frame-local analytics remains default-off until explicit Claracle consent; Hermes disposition pending | ++| SEC-03 | Public export fields need a documented allowlist to prevent future accidental expansion | Medium | Bender and Hermes | Implemented and tested: exact CSV, metadata, nested-object, and source-path allowlists; Hermes policy approval pending | ++| SEC-04 | Lifecycle deletion depends on manually reviewed overrides | Medium | Bender and Hermes | Rename, archive, deletion, retention, expiry, and fail-closed fixtures pass; production-policy disposition pending | ++| SEC-05 | Phrase-based injection detection has known semantic false-negative risk | Medium | Hermes and Farnsworth | Defense-in-depth accepted-risk recommendation is documented and executable controls are retained; no risk acceptance has been granted | + | SEC-06 | GA4, GSC, and Podcaster secret behavior is not proven by repository inspection | Medium | URL and jmservera | External verification pending; never record secret values | + | SEC-07 | Browser tool uses safe DOM and a restricted outbound URL policy | Informational | Amy | Repository control verified; production and accessibility behavior pending | + | SEC-08 | Raw HTML rendering remains disabled | Informational | Amy and Hermes | Repository control verified; Hermes sign-off pending | +@@ -149,7 +182,7 @@ does not prove protected-environment configuration or a downstream Podcaster run + ## Required evidence before acceptance + + - Hermes records approval, rejection, or accepted-risk rationale for SEC-01 through SEC-06 +-- Candidate-title sanitizer and adversarial tests pass before dynamic topic creation is enabled ++- Hermes verifies the implemented candidate-title sanitizer and adversarial rejection before dynamic topic creation is enabled + - Embed privacy behavior has a documented and tested disposition + - Public dataset schema receives a field-level privacy review + - Lifecycle fixtures demonstrate rename, archive, confirmed deletion, retention, and expiry +diff --git a/docs/review/data-observatory-relaunch/status-of-record.md b/docs/review/data-observatory-relaunch/status-of-record.md +new file mode 100644 +index 0000000..c2cb650 +--- /dev/null ++++ b/docs/review/data-observatory-relaunch/status-of-record.md +@@ -0,0 +1,114 @@ ++--- ++title: Data Observatory Relaunch Status of Record ++description: Single reconciled view of delivered versus pending relaunch work across the three remediation plans, the PRD, and the BRD ++author: SquadScope Squad ++ms.date: 2026-08-02 ++ms.topic: reference ++keywords: ++ - status of record ++ - data observatory ++ - relaunch readiness ++ - reconciliation ++estimated_reading_time: 7 ++--- ++ ++## Purpose ++ ++This document is the single reconciled view of the Claracle Data Observatory ++relaunch. It supersedes the fragmented checkbox state across the three remediation ++plans by mapping each workstream to its delivered or pending status with evidence. ++It complements the [acceptance evidence index](README.md), which owns the external ++gate matrix and the acceptance decision. ++ ++- Epic: [#594](https://github.com/jmservera/SquadScope/issues/594) ++- PRD: [claracle-data-observatory-relaunch.md](../../prds/claracle-data-observatory-relaunch.md) ++- BRD: [claracle-data-observatory-relaunch-brd.md](../../brds/claracle-data-observatory-relaunch-brd.md) ++ ++Reconciled on 2026-08-02. Release acceptance remains **pending** per the ++[acceptance decision](README.md#acceptance-decision); both rollout flags stay disabled. ++ ++## Source plans ++ ++| Plan | Scope | Reconciled state | ++| --------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | ----------------------------- | ++| [2026-07-29 remediation](../../../.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md) | Core relaunch build (Phases 1-10) | Phases 1-6, 8 done; 7, 9, 10 open | ++| [2026-07-30 review remediation](../../../.copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md) | Post-review corrections (Phases 1-8) | Phases 1-5 done; 6-8 open | ++| [2026-07-31 deploy hydration](../../../.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md) | Deploy/hydration incident (Phases 1-5) | Phases 1-4 done; 5.3 open | ++ ++## Delivered since the plans were written ++ ++| Workstream | Evidence (merged) | Requirement trace | ++| --------------------------------------- | ---------------------------------------------------------------- | -------------------------- | ++| Deploy stopped hydrating empty `content/data` | `#627` → `#628` (interim Path B) | NFR-011 | ++| Publish commit stages only existing generated paths | `#634` | NFR-011, provenance | ++| Safe hydration guard generalized; `content/data` deploy restored | `#633` → `#637` | NFR-011, NFR-012, R-08 | ++| Embed `source_page` validation before build (CI guard) | `scripts/check_embed_sources.py`, `tests/test_embed_sources.py`, wired in `ci.yml` (`#641`) | NFR-011, NFR-012 | ++| Embeddable-charts demo linked from the data landing page | `#642` | FR-052 | ++| Podcaster smoke: API key passed to reusable workflow | `#636` | NFR-002 | ++| Podcaster smoke: tooling checked out from default branch | `#639` → `#643` | NFR-002, R-04 | ++| Podcaster smoke: hydrate source manifest referenced by promotion record | `#639` → `#645` | NFR-002, R-04 | ++| Restore preserves the published weekly transaction | `#640` → `#646` | NFR-002 (restore integrity) | ++| Live deploy failure (run 30718600607) | `#644` CLOSED — root cause was a dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) breaking the Podcaster smoke gate; resolved by `#645`/`#646`; deploy-site green since 2026-08-01 | — | ++ ++## Requirement status summary ++ ++| Area | Status | Notes | ++| --------------------------------------- | --------- | -------------------------------------------------------------------------------- | ++| Weekly topic through-line | Done | 2026-07-29 Phase 2 | ++| Repository lifecycle (identity, retention) | Done | 2026-07-29 Phase 3 | ++| Atomic publish transaction | Done | 2026-07-29 Phase 4; restore integrity hardened by `#640`/`#646` | ++| SEO / rendered link contracts | Done | 2026-07-29 Phase 5 | ++| Consent-gated analytics API | Done | 2026-07-29 Phase 6 | ++| Deploy / hydration parity + CI guard | Done | 2026-07-31 Phases 1-4 (`#628`/`#634`/`#637`/`#641`) | ++| Podcaster release smoke (dry-run gate) | Done | Blocking post-deploy gate green (`#636`/`#639`/`#643`/`#645`) | ++| FR-041 internal link checking | Partial | Satisfied at test level (`tests/test_internal_link_checker.py`); no standalone CI link tool | ++| Hugo/Pagefind timing separation | Done | CI records separate report-only Hugo and Pagefind durations; Q-01 workload attribution remains pending | ++| Security sign-off (NFR-004) | Pending | SEC-02 and SEC-03 repository controls are implemented and tested; SEC-05 has a defense-in-depth recommendation; Hermes finding dispositions remain required | ++| Accessibility evidence (NFR-005) | Pending | Amy/Fry; 2026-07-30 Step 7.2 | ++| Real Podcaster downstream run (NFR-002 / R-04) | Pending | Protected environment; 2026-07-30 Step 6.3 | ++| Refreshed visual acceptance | Pending | Screenshot capture checklist | ++| GA4 + GSC connection + baseline (FR-035) | Partial | Connection complete: GA4 stream confirmed, GSC verified, root sitemap submitted, and products linked; numeric baseline transcription and consent evidence remain pending | ++| External metadata and feed validation | Pending | Social previews, Rich Results, Schema.org, and production feed responses require retained conclusions | ++| Incremental generation cost (Q-01 / NFR-009) | Pending | Design spike required | ++| `repo_pages` rollout (FR-020-022) | Deferred | Flag disabled; needs sponsor approval + own plan | ++| `dynamic_topic_creation` rollout (FR-004) | Deferred | Flag disabled; needs sponsor approval + own plan | ++| Sponsor rollout approval | Pending | jmservera; see [launch-gate register](#launch-gate-register) | ++ ++## Epic issue dispositions ++ ++| Issue | Title | Disposition | ++| ------------------------------------------------- | ------------------------------------------- | ----------------------------------------------------------------- | ++| [#594](https://github.com/jmservera/SquadScope/issues/594) | Epic: Claracle Data Observatory Relaunch | Open — tracks overall relaunch | ++| [#644](https://github.com/jmservera/SquadScope/issues/644) | Deploy Hugo site failed (run 30718600607) | CLOSED — resolved by `#645`/`#646` | ++| [#626](https://github.com/jmservera/SquadScope/issues/626) | Lighthouse / performance quality-gate follow-ups | Independent hardening; keep thresholds unchanged | ++| [#622](https://github.com/jmservera/SquadScope/issues/622) | Post-review UX polish | Non-blocking polish; resolve factual questions before final visual recapture | ++| [#599](https://github.com/jmservera/SquadScope/issues/599) | Connect GA4 + Google Search Console (FR-035) | Closed; owner confirmed GA4, GSC verification, sitemap submission, and product link on 2026-08-02 | ++ ++## Launch-gate register ++ ++Each gate lists its owner, blocking dependency, and the evidence path that closes it. ++The external acceptance matrix in the [acceptance evidence index](README.md#external-acceptance-matrix) ++holds the platform-level rows; this register adds ownership and sequencing. ++ ++| Gate | Owner | Dependency | Evidence path | ++| ------------------------------------------ | ----------- | -------------------------------------------- | ------------------------------------------------------------------- | ++| GA4 + GSC dated baseline and consent evidence (FR-035/NFR-007/008) | jmservera | Transcribe supplied export and retain production consent observations | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | ++| Security sign-off (NFR-004) | Hermes | Security review disposition | [Security review](security-review.md) | ++| Accessibility evidence (NFR-005) | Amy / Fry | Production browser and assistive technology | [Owner action register](owner-action-register.md#accessibility-acceptance) | ++| Real Podcaster downstream run (NFR-002 / R-04) | URL | Protected environment policy and maintainer authorization | [Owner action register](owner-action-register.md#protected-real-podcaster-run) | ++| Refreshed visual acceptance | Amy | Populated content render | [Screenshot capture checklist](screenshots/README.md) | ++| External metadata and feed validation | Amy / jmservera | External debugger and production access | [Owner action register](owner-action-register.md#external-metadata-and-feed-validation) | ++| Incremental generation cost (Q-01 / NFR-009) | URL | Comparable workload variants and budget owner | [Gated rollout and cost plan](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) | ++| Lighthouse follow-ups (`#626`) | Amy / Fry | None; independent hardening | `#626` | ++| Post-review UX polish (`#622`) | Amy | None; non-blocking | `#622` | ++| Sponsor rollout approval | jmservera | Required gate evidence for each flag | [Owner action register](owner-action-register.md#sponsor-rollout-decision) | ++ ++## Deferred to separate plans ++ ++These are out of scope for the readiness reconciliation and each needs its own plan ++(see the [reconciliation planning log](../../../.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md#suggested-follow-on-work)): ++ ++- GA4/GSC baseline transcription and production consent evidence (connection and sitemap submission are complete) ++- [`repo_pages` rollout](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (requires identity, lifecycle, security, and sponsor approval) ++- [`dynamic_topic_creation` rollout](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (requires preview, canary, security, and sponsor approval) ++- [Incremental-generation-cost experiment](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (Q-01 / NFR-009) +diff --git a/hugo.toml b/hugo.toml +index 29293dc..4ca8445 100644 +--- a/hugo.toml ++++ b/hugo.toml +@@ -20,7 +20,7 @@ rssLimit = 52 + term = ['HTML', 'RSS'] + + [params] +- # jmservera: set to the production GA4 web stream ID (for example, G-XXXXXXXXXX) after creating the Google Analytics property. ++ # Production injects this through the GA_MEASUREMENT_ID Actions secret; keep the checked-in default empty for forks. + ga_measurement_id = "" + # jmservera: set via GSC_SITE_VERIFICATION secret. + gsc_site_verification = "" +diff --git a/layouts/partials/visuals/observatory-chart.html b/layouts/partials/visuals/observatory-chart.html +index 12bf3e0..6047460 100644 +--- a/layouts/partials/visuals/observatory-chart.html ++++ b/layouts/partials/visuals/observatory-chart.html +@@ -28,7 +28,7 @@ + {{- $sourcePermalink := $page.Permalink -}} + {{- $embedSnippet := "" -}} + {{- with $embedURL -}} +- {{- $embedSnippet = printf `` ($title | htmlEscape) (. | absURL) -}} ++ {{- $embedSnippet = printf `` ($title | htmlEscape) (. | absURL) -}} + {{- end -}} +
+
+diff --git a/scripts/export_observatory_dataset.py b/scripts/export_observatory_dataset.py +index 9917b38..d8518b4 100644 +--- a/scripts/export_observatory_dataset.py ++++ b/scripts/export_observatory_dataset.py +@@ -24,7 +24,7 @@ DATASET_VERSION = "2026-W31" + DEFAULT_OUTPUT_DIR = PROJECT_ROOT / "static" / "datasets" / DATASET_SLUG + PREFERRED_RAW_WEEKS = ("2026-W21", "2026-W22", "2026-W29", "2026-W30", "2026-W31") + RECOVERED_WEEKS = ("2026-W23", "2026-W24", "2026-W25", "2026-W26", "2026-W27", "2026-W28") +-CSV_COLUMNS = [ ++PUBLIC_CSV_FIELDS = ( + "rank_by_latest_stars", + "repository", + "url", +@@ -40,7 +40,37 @@ CSV_COLUMNS = [ + "seen_in_trending", + "seen_in_new", + "top_topics", +-] ++) ++PUBLIC_METADATA_FIELDS = ( ++ "dataset", ++ "version", ++ "generated_at", ++ "source", ++ "selection_rule", ++ "license", ++ "row_count", ++ "weeks", ++ "weekly_observation_counts", ++ "exported_repo_observations", ++ "total_source_repo_observations_screened", ++ "recurring_repository_count_min_4_weeks", ++ "repositories_seen_in_trending", ++ "repositories_seen_in_new", ++ "top_languages_by_repository_count", ++ "top_licenses_by_repository_count", ++ "top_topics_by_repository_mentions", ++ "top_repositories_by_latest_stars", ++ "fields", ++ "source_files", ++ "public_exposure_review", ++) ++PUBLIC_TOP_REPOSITORY_FIELDS = ( ++ "repository", ++ "latest_stars", ++ "observed_star_change", ++ "weeks_observed", ++ "url", ++) + AI_KEYWORDS = { + "agent", + "agents", +@@ -136,7 +166,7 @@ class RepoAggregate: + return self.latest_stars - self.first_stars + + def row(self, rank: int) -> dict[str, str | int]: +- return { ++ row = { + "rank_by_latest_stars": rank, + "repository": self.repository, + "url": self.url, +@@ -153,6 +183,57 @@ class RepoAggregate: + "seen_in_new": "true" if "new_repos" in self.buckets else "false", + "top_topics": "|".join(topic for topic, _count in self.topics.most_common(8)), + } ++ validate_exact_keys(row, PUBLIC_CSV_FIELDS, "CSV row") ++ return row ++ ++ ++def validate_exact_keys( ++ payload: dict[str, Any], allowed_fields: tuple[str, ...], label: str ++) -> None: ++ actual = set(payload) ++ expected = set(allowed_fields) ++ if actual != expected: ++ added = sorted(actual - expected) ++ missing = sorted(expected - actual) ++ raise ValueError( ++ f"{label} fields violate the public allowlist: added={added}, missing={missing}" ++ ) ++ ++ ++def public_source_path(path: Path, data_root: Path) -> str: ++ try: ++ relative_path = path.resolve().relative_to(data_root.resolve()) ++ except ValueError as error: ++ raise ValueError(f"Source path is outside the public export policy: {path}") from error ++ allowed_paths = {Path("raw") / f"{week}.json" for week in PREFERRED_RAW_WEEKS} | { ++ Path("archive") / "recovered-W23-W29" / week / f"{week}.json" for week in RECOVERED_WEEKS ++ } ++ if relative_path not in allowed_paths: ++ raise ValueError(f"Source path is outside the public export policy: {relative_path}") ++ return (Path("data") / relative_path).as_posix() ++ ++ ++def validate_public_summary(summary: dict[str, Any]) -> None: ++ validate_exact_keys(summary, PUBLIC_METADATA_FIELDS, "dataset metadata") ++ if summary["fields"] != list(PUBLIC_CSV_FIELDS): ++ raise ValueError("Metadata fields must exactly match the public CSV allowlist") ++ if set(summary["weekly_observation_counts"]) != set(summary["weeks"]): ++ raise ValueError("Weekly observation metadata must exactly match the exported weeks") ++ for metadata_field in ( ++ "top_languages_by_repository_count", ++ "top_licenses_by_repository_count", ++ "top_topics_by_repository_mentions", ++ ): ++ for item in summary[metadata_field]: ++ if ( ++ not isinstance(item, (list, tuple)) ++ or len(item) != 2 ++ or not isinstance(item[0], str) ++ or not isinstance(item[1], int) ++ ): ++ raise ValueError(f"{metadata_field} entries must be public label/count pairs") ++ for repository in summary["top_repositories_by_latest_stars"]: ++ validate_exact_keys(repository, PUBLIC_TOP_REPOSITORY_FIELDS, "top repository metadata") + + + def discover_source_paths(data_root: Path) -> list[Path]: +@@ -225,6 +306,7 @@ def build_summary( + source_paths: list[Path], + observation_counts: dict[str, int], + total_source_observations: int, ++ data_root: Path = PROJECT_ROOT / "data", + ) -> dict[str, Any]: + language_counts: Counter[str] = Counter() + license_counts: Counter[str] = Counter() +@@ -249,7 +331,7 @@ def build_summary( + str(json.loads(path.read_text(encoding="utf-8")).get("crawled_at") or "") + for path in source_paths + ) +- return { ++ summary = { + "dataset": DATASET_SLUG, + "version": DATASET_VERSION, + "generated_at": generated_at, +@@ -281,19 +363,26 @@ def build_summary( + } + for repo in repos[:25] + ], +- "fields": CSV_COLUMNS, +- "source_files": [str(path.relative_to(PROJECT_ROOT)) for path in source_paths], ++ "fields": list(PUBLIC_CSV_FIELDS), ++ "source_files": [public_source_path(path, data_root) for path in source_paths], + "public_exposure_review": ( + "PASS: exported fields are repository names, GitHub URLs, public language/license/topic " + "metadata, public star/fork counts, and derived weekly aggregates from public GitHub crawl records. " + "No tokens, private repo data, cache payloads, user account data, or unpublished crawl calls are included." + ), + } ++ validate_public_summary(summary) ++ return summary + + + def write_csv(path: Path, repos: list[RepoAggregate]) -> None: + with path.open("w", encoding="utf-8", newline="") as output: +- writer = csv.DictWriter(output, fieldnames=CSV_COLUMNS, lineterminator="\n") ++ writer = csv.DictWriter( ++ output, ++ fieldnames=PUBLIC_CSV_FIELDS, ++ extrasaction="raise", ++ lineterminator="\n", ++ ) + writer.writeheader() + for rank, repo in enumerate(repos, start=1): + writer.writerow(repo.row(rank)) +@@ -364,7 +453,13 @@ def export_dataset( + source_paths = discover_source_paths(data_root) + aggregates, observation_counts, total_source_observations = load_aggregates(source_paths) + repos = sorted_aggregates(aggregates) +- summary = build_summary(repos, source_paths, observation_counts, total_source_observations) ++ summary = build_summary( ++ repos, ++ source_paths, ++ observation_counts, ++ total_source_observations, ++ data_root, ++ ) + + output_dir.mkdir(parents=True, exist_ok=True) + write_csv(output_dir / "top-github-projects.csv", repos) +@@ -379,7 +474,9 @@ def export_dataset( + + def check_dataset(output_dir: Path, data_root: Path) -> list[Path]: + """Return generated dataset files whose checked-in bytes are stale.""" +- with tempfile.TemporaryDirectory() as temporary_directory: ++ workspace_root = PROJECT_ROOT / ".test-workspaces" ++ workspace_root.mkdir(exist_ok=True) ++ with tempfile.TemporaryDirectory(dir=workspace_root) as temporary_directory: + expected_dir = Path(temporary_directory) + export_dataset(expected_dir, data_root) + expected_paths = sorted(path for path in expected_dir.rglob("*") if path.is_file()) +diff --git a/tests/test_export_observatory_dataset.py b/tests/test_export_observatory_dataset.py +index 1e285af..3760cb0 100644 +--- a/tests/test_export_observatory_dataset.py ++++ b/tests/test_export_observatory_dataset.py +@@ -13,6 +13,91 @@ WORKSPACE_ROOT = Path(".test-workspaces") + + + class ExportObservatoryDatasetTests(unittest.TestCase): ++ def test_public_export_allowlists_are_exact_and_synchronized(self) -> None: ++ self.assertEqual( ++ export_observatory_dataset.PUBLIC_CSV_FIELDS, ++ ( ++ "rank_by_latest_stars", ++ "repository", ++ "url", ++ "primary_language", ++ "latest_license", ++ "first_seen_week", ++ "last_seen_week", ++ "weeks_observed", ++ "latest_stars", ++ "first_observed_stars", ++ "observed_star_change", ++ "max_forks_observed", ++ "seen_in_trending", ++ "seen_in_new", ++ "top_topics", ++ ), ++ ) ++ self.assertEqual( ++ export_observatory_dataset.PUBLIC_TOP_REPOSITORY_FIELDS, ++ ( ++ "repository", ++ "latest_stars", ++ "observed_star_change", ++ "weeks_observed", ++ "url", ++ ), ++ ) ++ self.assertEqual( ++ set(export_observatory_dataset.PUBLIC_METADATA_FIELDS), ++ { ++ "dataset", ++ "version", ++ "generated_at", ++ "source", ++ "selection_rule", ++ "license", ++ "row_count", ++ "weeks", ++ "weekly_observation_counts", ++ "exported_repo_observations", ++ "total_source_repo_observations_screened", ++ "recurring_repository_count_min_4_weeks", ++ "repositories_seen_in_trending", ++ "repositories_seen_in_new", ++ "top_languages_by_repository_count", ++ "top_licenses_by_repository_count", ++ "top_topics_by_repository_mentions", ++ "top_repositories_by_latest_stars", ++ "fields", ++ "source_files", ++ "public_exposure_review", ++ }, ++ ) ++ ++ def test_public_export_rejects_unlisted_fields_and_source_paths(self) -> None: ++ valid_row = export_observatory_dataset.RepoAggregate(repository="owner/repo").row(1) ++ with self.assertRaisesRegex(ValueError, "public allowlist"): ++ export_observatory_dataset.validate_exact_keys( ++ {**valid_row, "private_note": "not public"}, ++ export_observatory_dataset.PUBLIC_CSV_FIELDS, ++ "CSV row", ++ ) ++ ++ with self.assertRaisesRegex(ValueError, "outside the public export policy"): ++ export_observatory_dataset.public_source_path( ++ export_observatory_dataset.PROJECT_ROOT / "data/private/repos.json", ++ export_observatory_dataset.PROJECT_ROOT / "data", ++ ) ++ ++ summary = export_observatory_dataset.export_dataset( ++ output_dir=WORKSPACE_ROOT / "allowlist-validation" ++ ) ++ self.addCleanup( ++ lambda: shutil.rmtree(WORKSPACE_ROOT / "allowlist-validation", ignore_errors=True) ++ ) ++ summary["top_languages_by_repository_count"] = [ ++ {"language": "Python", "count": 1, "private_note": "not public"} ++ ] ++ with self.assertRaisesRegex(ValueError, "public label/count pairs"): ++ export_observatory_dataset.validate_public_summary(summary) ++ + def test_check_reports_stale_external_output_path(self) -> None: + external_path = Path("/tmp/claracle-dataset/dataset-metadata.json") + stderr = io.StringIO() +@@ -54,13 +139,32 @@ class ExportObservatoryDatasetTests(unittest.TestCase): + with csv_path.open(encoding="utf-8", newline="") as handle: + rows = list(csv.DictReader(handle)) + self.assertEqual(len(rows), summary["row_count"]) ++ self.assertEqual(tuple(rows[0]), export_observatory_dataset.PUBLIC_CSV_FIELDS) + self.assertEqual(rows[0]["repository"], "openclaw/openclaw") + self.assertEqual(rows[0]["seen_in_trending"], "true") + self.assertIn("ai", rows[0]["top_topics"]) + + metadata = json.loads(metadata_path.read_text(encoding="utf-8")) ++ self.assertEqual(set(metadata), set(export_observatory_dataset.PUBLIC_METADATA_FIELDS)) ++ self.assertEqual(metadata["fields"], list(export_observatory_dataset.PUBLIC_CSV_FIELDS)) ++ self.assertTrue( ++ all( ++ set(repository) == set(export_observatory_dataset.PUBLIC_TOP_REPOSITORY_FIELDS) ++ for repository in metadata["top_repositories_by_latest_stars"] ++ ) ++ ) ++ self.assertEqual( ++ set(metadata["weekly_observation_counts"]), ++ set(metadata["weeks"]), ++ ) + self.assertEqual(metadata["source_files"], summary["source_files"]) + self.assertEqual(len(metadata["source_files"]), 11) ++ self.assertTrue( ++ all( ++ source.startswith(("data/raw/", "data/archive/recovered-W23-W29/")) ++ for source in metadata["source_files"] ++ ) ++ ) + self.assertIn("MIT License", license_path.read_text(encoding="utf-8")) + citation = citation_path.read_text(encoding="utf-8") + self.assertIn( +diff --git a/tests/test_observatory_embeds.py b/tests/test_observatory_embeds.py +index 2ab8b2b..ec035ea 100644 +--- a/tests/test_observatory_embeds.py ++++ b/tests/test_observatory_embeds.py +@@ -42,6 +42,21 @@ def test_copy_button_handles_clipboard_rejections() -> None: + assert "Copy failed" in script + + ++def test_embed_layout_keeps_consent_gated_analytics_wiring() -> None: ++ base_layout = (ROOT / "layouts/embeds/baseof.html").read_text(encoding="utf-8") ++ analytics = (ROOT / "assets/js/observatory-analytics.js").read_text(encoding="utf-8") ++ consent = (ROOT / "layouts/partials/cookie-consent.html").read_text(encoding="utf-8") ++ ++ assert 'partial "analytics.html"' in base_layout ++ assert 'resources.Get "js/observatory-analytics.js"' in base_layout ++ assert 'partial "cookie-consent.html"' in base_layout ++ assert "let analyticsConsent = false;" in analytics ++ assert "analyticsConsent = enabled === true;" in analytics ++ assert "if (!analyticsConsent" in analytics ++ assert "CookieConsent.acceptedCategory('analytics')" in consent ++ assert "setObservatoryAnalyticsConsent(false);" in consent ++ ++ + def test_rendered_embed_contains_backlink_and_chart_data(tmp_path: Path) -> None: + if shutil.which("hugo") is None: + pytest.skip("Hugo binary is required to render embed fixtures") +@@ -66,3 +81,4 @@ def test_rendered_embed_contains_backlink_and_chart_data(tmp_path: Path) -> None + assert "observatory-chart__data" in embed_html + assert "https://claracle.com/embeds/fastest-growing-ai-repositories-chart/" in demo_html + assert "<iframe" in demo_html ++ assert "referrerpolicy="no-referrer"" in demo_html +diff --git a/tests/visual/observatory-analytics.spec.mjs b/tests/visual/observatory-analytics.spec.mjs +index b370bf0..03e106c 100644 +--- a/tests/visual/observatory-analytics.spec.mjs ++++ b/tests/visual/observatory-analytics.spec.mjs +@@ -17,20 +17,25 @@ async function interceptGoogleEndpoints(page) { + contentType: 'application/javascript', + body: ` + window.SquadScopeGA4TestStubLoaded = true; +- window.gtag = function () { +- window.dataLayer = window.dataLayer || []; +- window.dataLayer.push(arguments); +- if (arguments[0] === 'event') { +- var params = new URLSearchParams({ en: arguments[1] }); +- Object.keys(arguments[2] || {}).forEach(function (key) { +- params.set('ep.' + key, arguments[2][key]); ++ function sendEvent(args) { ++ if (args[0] === 'event') { ++ var params = new URLSearchParams({ en: args[1] }); ++ Object.keys(args[2] || {}).forEach(function (key) { ++ params.set('ep.' + key, args[2][key]); + }); + fetch('https://www.google-analytics.com/g/collect?' + params, { + mode: 'no-cors', + keepalive: true + }); + } ++ } ++ var queuedEntries = (window.dataLayer || []).slice(); ++ window.gtag = function () { ++ window.dataLayer = window.dataLayer || []; ++ window.dataLayer.push(arguments); ++ sendEvent(arguments); + }; ++ queuedEntries.forEach(sendEvent); + `, + }); + }); +@@ -182,15 +187,67 @@ test('tool interactions use real handlers and bounded fields', async ({ page }, + expect(JSON.stringify(events)).not.toContain('token=secret'); + }); + +-test('standalone chart view fires only after UI acceptance', async ({ page }, testInfo) => { ++test('standalone frame uses only its own explicit analytics consent', async ({ page }, testInfo) => { + desktopOnly(testInfo); +- await interceptGoogleEndpoints(page); +- await page.goto('/embeds/fastest-growing-ai-repositories-chart/'); +- await waitForConsentUi(page); ++ const requests = await interceptGoogleEndpoints(page); ++ const embedUrl = new URL( ++ '/embeds/fastest-growing-ai-repositories-chart/', ++ testInfo.project.use.baseURL, ++ ); ++ const publisherUrl = new URL('/charts/embeddable-rankings/', embedUrl); ++ publisherUrl.hostname = embedUrl.hostname === 'localhost' ? '127.0.0.1' : 'localhost'; ++ expect(publisherUrl.origin).not.toBe(embedUrl.origin); + +- expect(await customEvents(page)).toEqual([]); ++ await page.goto(publisherUrl.href); + await acceptAnalytics(page); +- expect(await customEvents(page)).toEqual([ ++ const parentRequestCount = requests.length; ++ const parentAnalyticsCookies = await analyticsCookies(page); ++ expect(requests.filter(({ kind }) => kind === 'script')).toHaveLength(1); ++ expect(parentAnalyticsCookies).toEqual([]); ++ ++ await page.locator('body').evaluate((body, src) => { ++ const iframe = document.createElement('iframe'); ++ iframe.title = 'Cross-origin Claracle chart'; ++ iframe.src = src; ++ iframe.referrerPolicy = 'no-referrer'; ++ body.appendChild(iframe); ++ }, embedUrl.href); ++ ++ await expect ++ .poll(() => ++ page ++ .frames() ++ .some((candidate) => ++ candidate.url().includes('/embeds/fastest-growing-ai-repositories-chart/'), ++ ), ++ ) ++ .toBe(true); ++ const frame = page ++ .frames() ++ .find((candidate) => ++ candidate.url().includes('/embeds/fastest-growing-ai-repositories-chart/'), ++ ); ++ expect(frame).toBeDefined(); ++ await waitForConsentUi(frame); ++ ++ expect(await customEvents(frame)).toEqual([]); ++ await expect(frame.locator(`script[src*="gtag/js?id=${TEST_MEASUREMENT_ID}"]`)).toHaveCount(0); ++ expect(requests.slice(parentRequestCount)).toEqual([]); ++ expect(await analyticsCookies(page)).toEqual(parentAnalyticsCookies); ++ ++ await frame.evaluate(() => window.CookieConsent.acceptCategory('all')); ++ await expect ++ .poll(() => frame.evaluate(() => window.CookieConsent.acceptedCategory('analytics'))) ++ .toBe(true); ++ await expect ++ .poll(() => frame.locator(`script[src*="gtag/js?id=${TEST_MEASUREMENT_ID}"]`).count()) ++ .toBe(1); ++ await frame.waitForFunction(() => window.SquadScopeGA4TestStubLoaded === true); ++ await expect ++ .poll(() => requests.slice(parentRequestCount).filter(({ kind }) => kind === 'collect').length) ++ .toBe(1); ++ expect(requests.slice(parentRequestCount).filter(({ kind }) => kind === 'script')).toHaveLength(1); ++ expect(await customEvents(frame)).toEqual([ + { + name: 'chart_embed_view', + payload: { + + diff --git a/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md b/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md new file mode 100644 index 0000000..ea2a068 --- /dev/null +++ b/.copilot-tracking/research/2026-08-01/restore-consistency-640-research.md @@ -0,0 +1,69 @@ + +# Research: Durable restore-consistency fix (#640) + +## Scope + +`run_mode=restore` (intended to refresh observatory surfaces from stored raw +evidence) also regenerates and republishes the week's weekly article. The +`sync-publish-to-main.yml` job wipes and re-syncs `content/weekly/` from +`publish` to `main`, so a restore overwrites the original published article +with a non-deterministic LLM regeneration. The same restore also rewrote the +promotion record to reference a candidate manifest that was never persisted, +breaking the Podcaster handoff smoke (remediated reactively for 2026-W31). + +## Success criteria (from #640) + +- A restore intended to refresh `content/data` does not change any already-published `content/weekly/*` article. +- Topic frontmatter on prior weeks is preserved through restore+sync. +- Regression test covers the chosen behavior. + +## Evidence log + +- `.github/workflows/crawl-and-publish.yml` + - `generate` job: hydrates prior state from `publish`, downloads analyzed + + candidate + raw artifacts, then `Generate weekly content candidate` → + `Rehash and promote final weekly content` (rewrites `content/weekly//W.md`), + then observatory steps, then `Commit generated content to data branch`. + - Commit step resets working tree to `origin/publish` (`git checkout -f -B publish origin/publish`), + extracts regenerated files (`tar -xf generated-state.tar`), stages + `GENERATED_PATHS`, and force-pushes to `publish`. + - The commit step copies the current-branch tooling to `publish-safety-tool.py` + BEFORE the reset, so a current-version helper is available after the reset. +- `.github/workflows/sync-publish-to-main.yml`: `rm -rf ... content/weekly ...` + then `git checkout origin/publish -- content/weekly/ ...` → publish is canonical. +- `scripts/rerun_modes.py`: mode validation; restore action = "restore published + artifacts for and regenerate through guarded promotion". +- Published weekly transaction = 3 files: `content/weekly//W.md`, + `data/analyzed/-summary.md`, `data/published//promotion-manifest.json`. +- `scripts/podcaster_handoff.py` (`promotion_transaction_v1`) verifies the + promotion record's `source_manifest.path` bytes → dangling references break it. + +## Alternatives evaluated + +- **A — Data-only restore mode (gate article steps):** skip article/promotion + steps entirely in restore. Cleanest semantically but touches many interdependent + steps in a high-risk force-pushing workflow; hard to test end-to-end. +- **B — Preserve published transaction (SELECTED):** after regeneration and the + branch reset, in restore mode revert the 3 published-transaction files to their + `publish` versions (`git checkout HEAD -- ` where HEAD == origin/publish) + before staging. Observatory surfaces stay regenerated. The commit — and thus the + sync — leave the published article, summary, and promotion record byte-identical. + Minimal, testable, and also prevents the dangling-manifest class. +- **C — Guard the sync:** sync has no knowledge of run_mode; detecting a + restore-driven rewrite there is fragile. Rejected. + +## Selected approach + +Option B. Add a small unit-tested helper `weekly_transaction_paths(week)` + +`weekly-transaction-paths` CLI to `scripts/publish_safety.py`. In the commit +step, gated on `run_mode == restore`, iterate those paths and +`git checkout HEAD -- ` for each that exists on `publish`, before staging. +Update `rerun_modes.py` restore action wording. Add tests. + +## Next steps + +1. Helper + CLI in `publish_safety.py`. +2. Gated preservation block in the commit step (pass `RUN_MODE` env). +3. Update `rerun_modes.py` restore action string. +4. Tests: `test_publish_safety.py` (paths), `test_pipeline.py` (workflow guard). +5. Validate: pytest, ruff, zizmor, yaml. diff --git a/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md b/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md new file mode 100644 index 0000000..d6a9f19 --- /dev/null +++ b/.copilot-tracking/research/2026-08-02/claracle-relaunch-followup-execution-research.md @@ -0,0 +1,46 @@ + +# Research: Claracle Relaunch Follow-Up Execution + +## Scope + +Execute all four follow-up items from the relaunch reconciliation: publish review corrections, complete repository-side GA4/GSC work, close executable acceptance evidence gaps, and plan gated rollouts plus incremental cost measurement. + +## Source Research + +* .copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md +* .copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md +* .copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md + +## Verified Findings + +* Review correction commit `8fddceb` is pushed to PR #647 and both review threads are resolved. +* Production renders GA configuration, the `GA_MEASUREMENT_ID` secret name exists, and the sitemap returns HTTP 200 with `application/xml`. +* Production does not render GSC verification metadata and no `GSC_SITE_VERIFICATION` secret name exists. +* Repository-side GA4/GSC wiring, consent gating, GSC metadata rendering, workflow injection, and tests already exist. Google account verification and baseline capture require jmservera. +* Focused acceptance research passed 102 security tests plus 42 UX/lifecycle tests; six Hugo-dependent tests skipped locally. +* Real Podcaster generation and environment-bound smoke evidence exist separately. No run combines real generation with a protected environment, and `podcaster-release-smoke` has no protection rules. +* Issue #622 explicitly classifies its work as non-blocking polish. Issue #626 is independent hardening and forbids lowering quality thresholds. +* Both rollout flags remain disabled. Repository activation lacks stable production GitHub IDs and lifecycle-transition evidence. Dynamic activation has five eligible candidates and no useful preview-only dry run. +* Hugo and Pagefind timing are already separated in CI, but Q-01 lacks workload variants, retained comparable samples, aggregation, and an approved budget. + +## Selected Approach + +1. Preserve the pushed review correction and verify PR checks. +2. Correct GA4/GSC evidence to distinguish fork-safe checked-in defaults from observed production configuration. +3. Refresh the acceptance package with current automated evidence and implementation dispositions without granting human sign-off. +4. Create implementation-ready owner checklists for external acceptance, protected real Podcaster execution, rollout canaries, and report-only cost measurement. +5. Keep both rollout flags disabled and do not trigger external side effects. + +## External Boundaries + +* Google property inspection, GSC token retrieval, ownership verification, sitemap submission, Realtime confirmation, and baseline values require jmservera or delegated Google access. +* Environment protection configuration requires repository administration and named reviewer policy. +* Real podcast generation requires Podcaster maintainer authorization because duplicate suppression is not evidenced in this repository. +* Manual keyboard, screen-reader, visual, security, and sponsor conclusions require named human reviewers. + +## Validation + +* Focused Python tests for workflow mapping, links, security, lifecycle, and exports +* Public presence-only production probes that do not print identifiers or tokens +* GitHub PR status checks and environment metadata +* Documentation diagnostics, stale-claim scans, and `git diff --check` diff --git a/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md b/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md new file mode 100644 index 0000000..eb6cad1 --- /dev/null +++ b/.copilot-tracking/research/2026-08-02/claracle-relaunch-readiness-reconciliation-research.md @@ -0,0 +1,61 @@ + +# Research: Claracle Relaunch Readiness Reconciliation (gap analysis) + +## Scope + +Review of the relaunch plans, PRD, and BRD against actual repository state to +identify what is missing and plan its closure. Verified via repository inspection +on 2026-08-02. + +## Source documents + +- BRD: docs/brds/claracle-data-observatory-relaunch-brd.md (BRD-CLARACLE-002, Document Control says v1.1) +- PRD: docs/prds/claracle-data-observatory-relaunch.md (v1.2, 2026-07-31) +- Plans: + - .copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md (Phases 1-6 done; 7-10 partial/open) + - .copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md (Phases 1-5 done; 6-8 open) + - .copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md (Phase 1 done; 2-5 open) + +## Verified findings (evidence) + +- Initial blocker: issue #644 "Deploy Hugo site failed (run 30718600607)" was the incident under investigation. Final verification found it closed as completed on 2026-08-01 after #645/#646 restored a green deploy. +- Initial repository-only observation: hugo.toml uses an empty fork-safe GA4 default. Follow-up production verification found secret-backed GA configuration present, while issue #599's recorded GSC, platform-receipt, and baseline actions remain outstanding (FR-035/DR-002/NFR-007). Growth KPI baselines (OBJ-2/3/4, G-002/003/004) were not captured. +- Repo pages gated off: config/observatory.toml `[repo_pages] enabled = false` and `[repo_pages.lifecycle] enabled = false` (FR-020-022 not live; matches PRD flag `repo_pages`). +- Internal link checking exists as tests/test_internal_link_checker.py (FR-041 partially satisfied at test level, not a separate CI link tool). +- Issue dispositions: #644 and #599 closed as completed on 2026-08-01; #626 (Lighthouse follow-ups), #622 (UX polish), and #594 (Epic) remained open at final verification. Closing #599 did not complete its human-action checklist. + +## Session work NOT reflected in docs + +- Deploy/hydration cascade #627-#637; Podcaster smoke #639/#643; restore-consistency #640/#646 all merged but absent from the PRD changelog and plan checkboxes. +- Deploy-hydration plan Phase 4 (embed source_page guard) shipped as check_embed_sources.py (#641) but left unmarked. + +## Document/consistency gaps + +- BRD version drift: Document Control = v1.1 but Acceptance section + PRD REF-1 cite BRD v1.0. +- No single status-of-record reconciling the three overlapping plans; checkboxes are stale. +- PRD changelog behind reality (no v1.3 for #627-#646 workstream; NFR-002 restore sub-behavior undocumented). +- No sponsor-approval artifact exists (BRD notes none recorded); both rollout flags cannot flip without it. +- DR-002 dated baseline snapshot never captured -> success currently unmeasurable. + +## Unmet launch gates (PRD/BRD) + +- NFR-004 Hermes security sign-off (pending) +- NFR-005 accessibility evidence (pending) +- NFR-002 / R-04 real Podcaster downstream run evidence (pending) +- Refreshed visual acceptance (pending) +- Q-01 / NFR-009 incremental generation cost/time (still TBD) +- Sponsor approval to enable dynamic_topic_creation + repo_pages + +## Planning approach + +Single reconciliation plan: +1. Triage/resolve #644 (live blocker) first. +2. Reconcile the three plan checklists to delivered state + produce one status-of-record. +3. Reconcile product docs (PRD v1.3 changelog + restore NFR; fix BRD version drift; add sponsor-approval + launch-gate register). +4. Sequence remaining launch gates into one owner/evidence register (do not fully implement external/human gates here). +5. Validate docs (markdown lint, link integrity) + re-review. + +Deferred to separate plans (out of scope here): +- GA4/GSC connection implementation (FR-035; continue the human-action checklist on closed issue #599) +- repo_pages rollout + dynamic topic rollout (require sponsor approval) +- Incremental-generation-cost design spike (Q-01) diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md new file mode 100644 index 0000000..6a7c338 --- /dev/null +++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-acceptance-gates-followup-research.md @@ -0,0 +1,320 @@ + +# Claracle Acceptance Gates Follow-Up Research + +## Research Scope + +Assess current evidence and executable work for: + +* NFR-004 security sign-off +* NFR-005 accessibility +* A real protected Podcaster downstream run +* Refreshed visual acceptance +* Lighthouse issue #626 +* UX issue #622 +* Sponsor approval + +Separate checks that can run locally or through GitHub now from actions that require protected environments or human approval. Inspect the relaunch review package, relevant tests, workflows, scripts, screenshots, PRD, BRD, and current repository patterns. + +## Working Hypothesis + +The status-of-record identifies the correct pending gates, but most acceptance evidence remains either stale, test-level only, or dependent on protected GitHub environments and named human approvers. Nearby scripts and workflows may provide executable checks without satisfying the final approval gates. + +## Evidence Inventory + +| Surface | Current evidence | What it proves | Boundary | +| --- | --- | --- | --- | +| Acceptance status | docs/review/data-observatory-relaunch/README.md and status-of-record.md both say release acceptance is pending | The seven requested gates are launch-readiness scope and both rollout flags must remain disabled | These records do not supply the missing approvals | +| Product requirements | docs/prds/claracle-data-observatory-relaunch.md v1.3 and docs/brds/claracle-data-observatory-relaunch-brd.md v1.2 | NFR-004 is Must, NFR-005 is Should, NFR-001 requires Lighthouse Performance >= 90, and each rollout flag needs separate sponsor approval | PRD and BRD record intent, not execution | +| Automated browser quality | .github/workflows/ci.yml production-site job | CI installs pinned Playwright 1.54.2, axe-playwright 4.10.2, and Lighthouse 12.8.2; runs responsive, axe, analytics, and Lighthouse checks; uploads reports for 30 days | The server is a local production build, not the public production origin | +| Current GitHub CI | CI run 30723119836 succeeded for the reconciliation branch; run 30742507113 was in progress when checked on 2026-08-02 | The branch has a recent complete green baseline and a current GitHub check is available | The current in-progress run must finish before its revision is cited as green | +| Lighthouse implementation | scripts/design/lighthouse-gates.mjs | Nine routes run mobile Lighthouse three times; medians must meet performance 0.90, accessibility 0.95, best practices 0.95, and CLS <= 0.1 | This does not complete the five open improvements in issue #626 | +| Accessibility implementation | tests/visual/observatory-a11y.spec.mjs and tests/visual/a11y-perf.spec.mjs | Serious and critical WCAG 2.1 A/AA axe findings fail; keyboard focus, modal focus handling, chart alternatives, overflow, and 44 px targets receive automated coverage | No retained production-origin audit, manual keyboard record, or screen-reader findings exist | +| Protected dry-run smoke | Deploy run 30721575540, job 91426194097 | The post-deploy reusable smoke succeeded on 2026-08-01 using exact retained promotion evidence and `--podcaster-dry-run` | It did not create a real downstream episode | +| Real downstream run | Trigger Podcast run 30202586031 | Week 2026-W30 used manifest run 29744859230 and Podcaster returned `status=accepted` with a retained job ID on 2026-07-26 | .github/workflows/trigger-podcast.yml declares no GitHub environment | +| GitHub environments | GitHub API response on 2026-08-02 | `podcaster-release-smoke` exists, but has no protection rules or deployment branch policy | A named environment is not evidence of protected approval controls | +| Visual evidence | Ten PNGs under docs/review/data-observatory-relaunch/screenshots plus its README | Historical desktop captures show the intended surfaces and the defects cited by issue #622 | They lack revision, viewport, theme, interaction metadata, mobile/dark variants, populated topic membership, and unobscured states | +| Security review | docs/review/data-observatory-relaunch/security-review.md | SEC-01 through SEC-06 and three signatories define the acceptance boundary | The review is stale against some current controls and all sign-off rows remain Pending | + +## Gate Findings + +### NFR-004 Security Sign-Off + +Status: blocked on review reconciliation and named human dispositions, not on a single test command. + +* Hermes has not dispositioned SEC-01 through SEC-06; URL and jmservera sign-off rows are also Pending. +* SEC-01 is stale as written. scripts/manage_topic_hubs.py now applies `sanitize_text`, rejects line breaks, boundary markers, HTML, Markdown, control characters, and injection phrases. tests/test_topic_hubs.py includes adversarial cases for those payloads. +* SEC-02 remains substantively open. layouts/partials/visuals/observatory-chart.html emits an iframe snippet without `referrerpolicy`, and standalone embeds can load the common consent-gated analytics adapter. A privacy decision and test are still required. +* SEC-03 is partially implemented. scripts/export_observatory_dataset.py centralizes a fixed `CSV_COLUMNS` list and emits a public-exposure statement, while the ADR requires review for new fields. There is no separately approved field policy or automated assertion that the published fields equal that policy. +* SEC-04 has strong executable evidence in tests/test_observatory_repos.py for rename, archive, confirmed deletion, retention, and expiry. Hermes still must accept the operator-override policy. +* SEC-05 is an explicit accepted-risk decision for phrase-based semantic false negatives. Tests can verify defense in depth, but only Hermes can record the risk disposition. +* SEC-06 requires production analytics evidence and protected Podcaster secret-scope evidence. Repository tests cannot close it. + +### NFR-005 Accessibility + +Status: automated implementation exists; acceptance evidence is incomplete. + +* CI already runs axe against topic, data, repository, chart, and tool routes and rejects serious or critical WCAG 2.1 A/AA violations. +* Existing Playwright tests cover labels, visible focus, keyboard operation, consent-dialog focus trapping and restoration, chart text alternatives, responsive overflow, and touch target size. +* The required acceptance record does not exist. It must identify revision, public production URLs, browser and assistive technology, automated results, manual keyboard findings, screen-reader findings, reviewer, date, and dispositions. + +### Real Protected Podcaster Downstream Run + +Status: real and environment-bound evidence each exist separately; the required combination does not. + +* Run 30202586031 proves a real accepted downstream request for week 2026-W30. +* Run 30721575540 proves a successful post-deploy dry-run through the `podcaster-release-smoke` environment. +* The real workflow and the post-merge path in .github/workflows/sync-publish-to-main.yml do not bind to an environment. The named smoke environment currently has no protection rules. Therefore no current run proves real generation after protected-environment approval. + +### Refreshed Visual Acceptance + +Status: blocked on final UI/data state and human visual review. + +* The historical set contains only ten PNGs and records a generated date of 2026-05-25 in the rendered footer. +* 02-topics-index.png shows no topic hubs with issue matches; 03-topic-hub-mcp.png shows zero recent weekly issues; 08-star-velocity-tool.png shows ambiguous rounded bars; all three are obscured by the consent banner. +* tests/visual/visual.spec.mjs covers only home, about, weekly, monthly, yearly, and cover-card baselines. It does not automate the relaunch capture matrix in docs/review/data-observatory-relaunch/screenshots/README.md. +* Acceptance requires a replacement matrix with revision, date, browser, viewport, theme, consent state, interaction state, source week, populated content, and reviewer conclusion. + +### Lighthouse Issue #626 + +Status: open on GitHub with no assignee and five independently shippable items. + +* Extend page-scoped CSS splitting to search, about, methodology, and privacy. +* Add Brotli negotiation to scripts/serve_static.py. +* Document the median-of-three and compressed-server methodology in docs/qa-gates.md. +* Reserve chart space to reduce the data-page CLS margin from its cited approximately 0.040 value. +* Parallelize per-page Lighthouse execution to reduce the roughly ten-minute production-site job. +* Acceptance explicitly forbids lowering the current thresholds. + +### UX Issue #622 + +Status: open on GitHub with no assignee; the issue calls the work non-blocking polish, while the relaunch status-of-record includes it in the readiness register. + +* Determine whether Star Velocity Explorer bars are intentionally normalized per row or incorrectly bound. Current screenshots are not sufficient to decide. +* Verify topic index aggregation against a production-representative dataset. Historical captures show the mismatch, but current weekly files W21 through W31 contain canonical `topics` frontmatter, including MCP Ecosystem in W21 and W26. This is likely stale visual evidence unless a fresh Hugo build still renders zero matches. +* Verify mobile consent-banner placement. Existing automated modal tests cover focus behavior but not visual content occlusion. + +### Sponsor Approval + +Status: human-only and absent. + +* Repository search found no dated approval artifact. +* BRD approval of the business requirements is not rollout approval. +* jmservera must approve or reject `dynamic_topic_creation` and `repo_pages` separately, identify the reviewed evidence and revision, and date the decision. Both flags must remain disabled until then. + +## Executable Checks Available Now + +### Local Without Protected Access + +* Pure-Python security controls, topic lifecycle, repository lifecycle, embed contracts, and trend-tool behavior run through the existing `.venv` with `uv run --no-sync`. +* The focused UX/lifecycle run completed with 42 passed and 6 skipped. Every skip required Hugo, which is not installed on this host. +* The focused sanitization and defense-chain run completed with 102 passed. +* Node packages match CI: Playwright 1.54.2, axe-playwright 4.10.2, and Lighthouse 12.8.2. +* Local Playwright execution is blocked before page navigation because the host lacks `libnspr4` and `libnss3`. Installing Playwright system dependencies requires elevated access. Lighthouse is subject to the same Chromium host dependency. +* The checked-in `public/` tree can be served with scripts/serve_static.py, but it is not sufficient acceptance evidence unless it is rebuilt from and tied to the tested source revision. + +### GitHub Without Protected Secrets Or Human Approval + +* Opening or updating a PR to `main` runs .github/workflows/ci.yml. It builds Hugo, runs rendered contracts, axe, responsive checks, analytics tests, and Lighthouse, and retains production-quality reports for 30 days. +* CI run 30742507113 had passed Python, rendered contracts, internal links, and axe/responsive gates when inspected. Lighthouse was still running. Completed run 30723119836 is the latest full green run for the same reconciliation branch lineage. +* Issues #626 and #622 can be researched, assigned, decomposed, and implemented through normal PRs without protected environment access. +* The dry-run smoke workflow is manually dispatchable, but it requires exact retained publish evidence and repository environment secrets. + +### GitHub Actions That Cause External Effects + +* .github/workflows/trigger-podcast.yml can dispatch a real generation request for an eligible week and publish run. It is not a read-only validation command. +* No idempotency or duplicate-suppression key is visible in scripts/podcaster_handoff.py. A rerun must be approved by the Podcaster maintainer or target a deliberately authorized episode. +* .github/workflows/deploy-site.yml can redeploy and then run the dry-run smoke. It changes production deployment state and is not needed merely to prove local code quality. + +## Protected Or Human Actions + +| Action | Required actor or access | Why it cannot be closed locally | +| --- | --- | --- | +| Protect the Podcaster environment | Repository administrator | `podcaster-release-smoke` currently has no protection rules or branch policy | +| Bind real generation to the protected environment | Workflow change reviewed by URL and Hermes | Current real and post-merge jobs declare no environment | +| Authorize a real downstream target week | Podcaster maintainer | A duplicate episode may be created; this repository has no visible idempotency guard | +| Execute and verify the real protected run | Environment approver and Podcaster maintainer | Requires secret-bearing environment and downstream access | +| Manual accessibility review | Fry plus an accessibility reviewer with production browser and assistive technology | Automated axe and keyboard tests do not provide screen-reader findings | +| Visual acceptance | Amy or named visual reviewer | Screenshot generation does not equal a human acceptance conclusion | +| NFR-004 disposition | Hermes, URL, and jmservera | SEC-02, SEC-05, SEC-06 and policy acceptance require threat, workflow, and production-owner judgment | +| Sponsor rollout decision | jmservera | Approval authority is explicitly human and must address each flag separately | + +## Exact Evidence Gaps + +1. NFR-004 has no dated Hermes disposition for SEC-01 through SEC-06 and no completed URL or sponsor sign-off row. The review also does not acknowledge that SEC-01 sanitization and SEC-04 lifecycle fixtures now exist. +2. SEC-02 lacks a chosen embed privacy model, implementation evidence, and a test for referrer and cross-origin consent behavior. +3. SEC-03 lacks an approved field-level publication policy and an automated comparison between that policy and exported fields. `CSV_COLUMNS` plus a self-authored PASS string is useful implementation evidence but not independent privacy approval. +4. SEC-05 lacks an accepted-risk rationale for semantic prompt-injection false negatives. +5. SEC-06 lacks dated production analytics network/cookie observations and environment-scoped Podcaster secret review. +6. NFR-005 lacks a retained accessibility review record with public URLs, revision, browser, assistive technology, axe results, full keyboard findings, screen-reader findings, reviewer, date, and dispositions. +7. No Actions run combines real Podcaster generation with a protected GitHub environment. Existing evidence is split between real unbound run 30202586031 and environment-bound dry-run 30721575540. +8. The environment named `podcaster-release-smoke` has zero protection rules. Its name alone does not satisfy protected execution. +9. The visual set lacks current revision metadata, mobile and dark captures, consent-resolved feature states, populated topic membership, complete interaction states, empty/error states, and a dated reviewer conclusion. +10. Issue #626 remains open with all five checklist items unchecked. No issue comment or linked PR records partial completion. +11. Issue #622 remains open. Fresh source data suggests the topic aggregation screenshot is stale, but no current rendered capture proves it; bar semantics and mobile consent occlusion remain unclassified. +12. No dated sponsor artifact approves or rejects `dynamic_topic_creation` and `repo_pages` separately. Both are correctly still `enabled = false` in config/observatory.toml. +13. The status-of-record calls #622 a readiness gate while the issue body calls it non-blocking polish and says the epic is accepted. A human release owner must resolve which statement controls launch acceptance. + +## Suggested Implementation Sequence + +1. Let CI run 30742507113 finish and retain its production-quality artifact URLs. Do not call the revision fully green until Lighthouse completes. +2. Reconcile the security review with current code. Mark SEC-01 implementation-ready for Hermes verification, attach the adversarial test result, attach SEC-04 lifecycle test evidence, and replace stale statements without marking NFR-004 accepted. +3. Resolve SEC-02 and SEC-03 through small independent changes: choose the embed privacy model first, then codify and test the public export field policy. Preserve disabled rollout flags. +4. Resolve issue #622’s factual questions before visual recapture. Rebuild with current W21-W31 frontmatter, confirm topic counts, document Star Velocity normalization semantics, and test consent placement at mobile viewports. +5. Implement issue #626 before final screenshots because CSS splitting and CLS reservation can change layout. Land documentation and Brotli support independently; preserve every threshold. Parallelize Lighthouse only after deterministic per-page artifact naming and failure aggregation are designed. +6. Run the complete GitHub production-site job and review the uploaded axe, responsive, Lighthouse, and Playwright reports. Then perform the manual keyboard and screen-reader review against the final production revision. +7. Capture the complete visual matrix only after #622 and layout-affecting #626 work is final. Record metadata beside every image and obtain Amy's dated acceptance conclusion. +8. Create a protected environment for real Podcaster generation, or add real mode to a separately named protected environment. Require approved branches and the intended reviewer. Bind the real job to it and have URL/Hermes review secret scope. +9. With Podcaster-maintainer authorization, run one real eligible week through the protected environment. Retain the Actions URL, week, manifest run ID, article SHA-256, downstream job ID, final downstream conclusion, approver, and date without recording secret values. +10. Have Hermes, URL, and jmservera complete the security sign-off table after external evidence is attached. +11. Resolve whether #622 is blocking, then have jmservera issue a dated sponsor decision for each rollout flag. Enabling either flag is a separate product change after approval, not part of the approval artifact itself. + +## Blockers + +* Hugo is absent locally, so rendered source validation cannot run on this host without installing the pinned binary. +* Playwright's Node packages are installed, but required Chromium host libraries are absent and need elevated package installation. GitHub CI is the available executable browser path now. +* The real Podcaster workflow is not environment-bound, and the existing named environment has no protection rules. +* A protected real rerun can create duplicate downstream work; Podcaster maintainer authorization is required because no local idempotency contract was found. +* Production accessibility, screen-reader, visual, security, and sponsor conclusions require named humans. +* Issue #622 has contradictory launch semantics between its GitHub body and the status-of-record. + +## Validation Commands + +### Local Pure-Python Baseline + +```bash +uv run --no-sync pytest -q \ + tests/test_topic_hubs.py \ + tests/test_trend_explorer_tool.py \ + tests/test_observatory_repos.py \ + tests/test_observatory_embeds.py +``` + +```bash +uv run --no-sync pytest -q \ + tests/test_sanitize_repo_content.py \ + tests/test_prompt_injection_redteam.py \ + tests/test_defense_chain_e2e.py +``` + +```bash +uv run --no-sync pytest -q tests/test_export_observatory_dataset.py +uv run --no-sync python scripts/export_observatory_dataset.py --check +uv run --no-sync python scripts/export_trend_explorer_data.py --check +``` + +### Full Local Source Build When Hugo And Browser Libraries Are Available + +```bash +hugo --minify --baseURL http://127.0.0.1:1313/ +npx pagefind@1.5.2 --site public/ +uv run --no-sync python scripts/serve_static.py --directory public --bind 127.0.0.1 --port 1313 +``` + +Run the server in one terminal, then run: + +```bash +BASE_URL=http://127.0.0.1:1313 \ + npx --no-install playwright test \ + --config tests/visual/playwright.config.mjs \ + tests/visual/a11y-perf.spec.mjs \ + tests/visual/observatory-a11y.spec.mjs \ + tests/visual/observatory-analytics.spec.mjs +``` + +```bash +node scripts/design/lighthouse-gates.mjs --base http://127.0.0.1:1313 +``` + +### GitHub Evidence Checks + +```bash +gh run view 30742507113 --repo jmservera/SquadScope \ + --json status,conclusion,headSha,jobs,url +``` + +```bash +gh run view 30721575540 --repo jmservera/SquadScope \ + --json conclusion,headSha,jobs,url +``` + +```bash +gh run view 30202586031 --repo jmservera/SquadScope --log +``` + +```bash +gh api repos/jmservera/SquadScope/environments \ + --jq '.environments[] | {name, protection_rules, deployment_branch_policy}' +``` + +### Side-Effecting Dispatches Requiring Approval + +Do not run either command as a read-only check. Substitute an authorized week and retained publish run only after environment protection, workflow binding, and Podcaster-maintainer approval. + +```bash +gh workflow run podcaster-handoff-smoke.yml \ + --repo jmservera/SquadScope \ + --ref main \ + -f week=YYYY-WNN \ + -f article_url=https://claracle.com/weekly/YYYY/WNN/ \ + -f article_path=content/weekly/YYYY/WNN.md \ + -f article_sha256=LOWERCASE_SHA256 \ + -f promotion_reference=data/published/YYYY-WNN/promotion-manifest.json +``` + +```bash +gh workflow run trigger-podcast.yml \ + --repo jmservera/SquadScope \ + --ref main \ + -f week=YYYY-WNN \ + -f publish_run_id=RUN_ID +``` + +## References + +### Repository Evidence + +* docs/review/data-observatory-relaunch/README.md +* docs/review/data-observatory-relaunch/status-of-record.md +* docs/review/data-observatory-relaunch/security-review.md +* docs/review/data-observatory-relaunch/screenshots/README.md +* docs/review/data-observatory-relaunch/screenshots/01-home.png through 10-internal-linking-block.png +* docs/prds/claracle-data-observatory-relaunch.md +* docs/brds/claracle-data-observatory-relaunch-brd.md +* .github/workflows/ci.yml +* .github/workflows/deploy-site.yml +* .github/workflows/podcaster-handoff-smoke.yml +* .github/workflows/trigger-podcast.yml +* .github/workflows/sync-publish-to-main.yml +* scripts/design/lighthouse-gates.mjs +* scripts/manage_topic_hubs.py +* scripts/export_observatory_dataset.py +* scripts/podcaster_handoff.py +* tests/visual/a11y-perf.spec.mjs +* tests/visual/observatory-a11y.spec.mjs +* tests/visual/observatory-analytics.spec.mjs +* tests/visual/visual.spec.mjs + +### External Evidence + +* [Issue #626](https://github.com/jmservera/SquadScope/issues/626) +* [Issue #622](https://github.com/jmservera/SquadScope/issues/622) +* [Latest confirmed green deploy and dry-run smoke](https://github.com/jmservera/SquadScope/actions/runs/30721575540) +* [Real accepted Podcaster run](https://github.com/jmservera/SquadScope/actions/runs/30202586031) +* [Current reconciliation CI](https://github.com/jmservera/SquadScope/actions/runs/30742507113) + +## Follow-On Questions + +* [ ] Inspect the completed Lighthouse results and uploaded quality artifacts from CI run 30742507113 after it finishes +* [ ] Confirm whether the Podcaster service deduplicates week or manifest requests outside this repository +* [ ] Confirm the intended required reviewer and branch policy for real Podcaster generation +* [ ] Render current W21-W31 content and verify that topic counts replace the historical empty states +* [ ] Determine the intended Star Velocity normalization model from product/design ownership +* [ ] Decide and document the embed referrer and cross-origin consent policy +* [ ] Resolve the blocking versus non-blocking status of issue #622 + +## Clarifying Questions + +* Who should approve the protected real Podcaster environment: URL, Hermes, jmservera, or a Podcaster maintainer? +* Is issue #622 a mandatory relaunch gate as recorded in status-of-record.md, or non-blocking polish as stated in the issue body? +* Does SquadScope-Podcaster guarantee idempotency for a repeated week or manifest, and where is that contract retained? +* Which screen reader and browser combination is the required NFR-005 manual acceptance target? diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md new file mode 100644 index 0000000..0ccee6e --- /dev/null +++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-ga4-gsc-followup-research.md @@ -0,0 +1,289 @@ + +# Claracle GA4 and Google Search Console Follow-up Research + +## Status + +Complete as of 2026-08-02 + +## Research Questions + +* What FR-035 repository-side work can be completed without Google credentials? +* Which GA4 and Google Search Console actions require jmservera account access? +* Should `ga_measurement_id` be driven by Hugo configuration, environment variables, or both? +* What are the cheapest executable validations for the selected approach? + +## Scope + +* `hugo.toml` +* Consent-gating layouts and scripts +* Growth evidence and FR-035 planning records +* GitHub issue #599, when available +* Relevant tests and repository instructions + +## Findings + +### Executive conclusion + +FR-035 is partially implemented, not wholly unimplemented. The repository already has the required +fork-safe GA4 parameter path, dynamic consent-gated loading, GSC verification metadata path, Hugo +sitemap generation, workflow secret injection, unit coverage for workflow mapping and GSC rendering, +and a blocking Playwright consent suite. No new production analytics implementation is required before +the account-side work. + +The remaining acceptance work is external. A names-only GitHub secret query showed that +`GA_MEASUREMENT_ID` exists and was last updated on 2026-06-13, while `GSC_SITE_VERIFICATION` does not +exist. A credential-free production probe on 2026-08-02 found GA configuration in the rendered home +page, no GSC verification meta tag, and a successful XML sitemap response: + +```text +ga_config_present=yes +gsc_meta_present=no +sitemap_status=200 +sitemap_content_type=application/xml +``` + +These observations disprove the status-of-record rationale that an empty checked-in +`ga_measurement_id` means GA4 is not deployed. They do not prove that the deployed ID belongs to the +intended property, that GA4 receives events, or that the required baseline has been captured. + +### Current repository implementation + +| Surface | Current behavior | Evidence | +| --- | --- | --- | +| Checked-in GA4 configuration | Empty by design for fork safety | `hugo.toml:22-24` | +| Checked-in GSC configuration | Empty by design | `hugo.toml:25-26` | +| Production parameter injection | Actions secrets map to Hugo environment overrides | `.github/workflows/deploy-site.yml:36-40` | +| GA4 render path | Supports flat config and Hugo's nested environment mapping; renders only when configured | `layouts/partials/analytics.html:1-11` | +| Consent gate | GA is disabled first; `gtag.js` is appended only after analytics consent | `layouts/partials/cookie-consent.html:44-101` | +| Custom events | Event names and fields are allowlisted; dispatch returns before consent | `assets/js/observatory-analytics.js:4-78` | +| GSC render path | Supports the new secret-backed parameter plus the legacy theme fallback | `layouts/partials/head.html:20-29` | +| Workflow contract test | Verifies both Actions secret mappings and CI's GA test ID | `tests/test_pipeline.py:265-284` | +| GSC render tests | Verify absent-by-default, environment override, precedence, and legacy fallback | `tests/test_rendered_seo_metadata.py:447-539` | +| Consent browser tests | Verify denied, accepted, reload, withdrawal, cookie clearing, and bounded events | `tests/visual/observatory-analytics.spec.mjs:1-180` | + +### Repository-side work possible without Google credentials + +No product behavior must be added to connect the existing paths. The following repository work can be +completed without signing in to Google: + +1. Correct stale evidence and status wording after the account observations are supplied. In + particular, `docs/review/data-observatory-relaunch/status-of-record.md:70` and + `docs/prds/claracle-data-observatory-relaunch.md:138` should not use the empty fork-safe default as + evidence that production GA4 is disconnected. +2. Update `docs/growth/ga4-gsc-baseline-2026-07-29.md` with dated, redacted observations supplied by + jmservera. Repository authors can prepare and review the evidence structure without platform access, + but must not invent values or copy secret tokens. +3. Clarify the comment in `hugo.toml:23`. It currently suggests writing the production ID into the + file, while the implemented security decision and `docs/setup-secrets.md` require secret-backed + environment injection and an empty checked-in default. +4. Run the workflow mapping test, Hugo rendering tests, local consent browser test, sitemap HTTP probe, + and names-only secret inventory. These checks need repository, network, or GitHub access, but no + Google credentials. +5. Optionally add a small Hugo render test for the GA parameter itself. The current blocking Playwright + path already exercises it end to end with `G-TEST-OBSERVATORY`, so this is a test-speed improvement, + not an FR-035 blocker. + +Setting `GSC_SITE_VERIFICATION` in GitHub is also repository-side, but the token must first be obtained +from a Search Console property. The person setting it needs GitHub secret-management rights; they do +not need Google access if jmservera supplies the token through an approved private channel. + +### Issue 599 context + +GitHub issue #599 was opened by jmservera on 2026-07-29 and closed as completed on 2026-08-01. Its +acceptance criteria require GA4 receipt, a verified GSC property, sitemap submission, and a dated +baseline. The only owner follow-up comment records five remaining human actions: create or select the +Claracle GA4 property and web stream, set the production measurement ID, verify `claracle.com` in GSC, +submit the sitemap, and capture baseline values. The issue contains no later comment proving those +actions. Closure therefore records completion of agent-side wiring, not FR-035 operational acceptance. + +Issue URL: [#599](https://github.com/jmservera/SquadScope/issues/599) + +## Selected Approach + +Keep the existing config-plus-environment design: + +* Keep `params.ga_measurement_id = ""` and `params.gsc_site_verification = ""` in `hugo.toml` +* Inject production values through `HUGO_PARAMS_GA_MEASUREMENT_ID` and + `HUGO_PARAMS_GSC_SITE_VERIFICATION` from GitHub Actions secrets +* Continue resolving both flat checked-in keys and Hugo's nested environment-key representation in the + templates +* Never hardcode the production GA measurement ID or GSC verification token in source + +For GSC, use the already-implemented URL-prefix property and HTML meta-tag verification flow for +`https://claracle.com/`. This is the cheapest path because it needs no DNS change and the template, +workflow mapping, documentation, and tests already exist. A Domain property would broaden coverage but +requires DNS-provider access and does not use the repository's current verification path. + +Recommended completion sequence: + +1. jmservera confirms that the existing deployed GA ID belongs to the intended Claracle property and + web stream. Do not rotate the GitHub secret unless it is wrong. +2. jmservera adds or selects the GSC URL-prefix property and privately obtains its HTML-tag token. +3. A repository administrator sets `GSC_SITE_VERIFICATION`, deploys the reviewed revision, and confirms + only the presence of the production meta tag. +4. jmservera clicks Verify in GSC and submits `https://claracle.com/sitemap.xml`. +5. jmservera performs a consented production visit, confirms GA4 Realtime receipt, and captures GA4/GSC + values for one explicitly dated baseline window. +6. Hermes or the designated privacy reviewer records denied and granted production browser behavior. +7. A repository-only follow-up updates the baseline, status of record, and PRD acceptance note with + redacted evidence and marks FR-035 complete only when every acceptance item is evidenced. + +## Credential and Ownership Boundary + +### Requires jmservera or delegated Google account access + +* Create, select, or inspect the GA4 account, property, and web data stream. Google requires the Editor + role to create properties or streams. +* Confirm that the deployed measurement ID maps to the intended Claracle stream. +* Observe Realtime receipt and capture GA4 acquisition/session baseline values. +* Add or select the Search Console property and obtain its verification token. +* Complete GSC ownership verification. A verified owner has the highest Search Console permission. +* Submit the sitemap in the verified property and capture submission, processing, performance, and + indexing evidence. + +These actions are assigned to jmservera by `docs/data-observatory-runbook.md:32-45` and issue #599. +They can be delegated only by granting the appropriate GA4 role and GSC owner/user access. + +### Requires GitHub access but not Google credentials + +* List secret names and timestamps without viewing values +* Set or rotate `GA_MEASUREMENT_ID` after an authorized person supplies the ID +* Set `GSC_SITE_VERIFICATION` after an authorized person supplies the token +* Trigger or inspect the Pages deployment and retain the Actions URL + +GitHub does not expose Actions secret values after creation. Secret-name presence proves protected +configuration exists, not that it is correct. + +### Requires neither Google nor privileged GitHub access + +* Inspect and test templates, consent code, workflow expressions, and generated output with test IDs +* Probe the public home page for configuration presence without printing the ID +* Probe the public sitemap response and content type +* Prepare documentation and evidence placeholders + +### Blockers + +* `GSC_SITE_VERIFICATION` is absent from the repository secret inventory +* Production has no GSC verification meta tag as of 2026-08-02 +* No retained evidence proves GA4 Realtime receipt or the intended property/stream mapping +* No retained evidence proves GSC ownership verification or sitemap submission +* All GA4/GSC baseline values remain pending +* Local Hugo render tests could not execute in this session because the `hugo` binary is absent + +## Validation Commands + +### Cheapest repository contract + +```bash +python3 -m pytest -q \ + tests/test_pipeline.py::WorkflowConfigTests::test_deploy_workflow_maps_analytics_and_gsc_secrets_to_hugo_params +``` + +Observed result: `1 passed` when run as part of the focused four-test selection. + +### GSC rendering contract + +```bash +python3 -m pytest -q tests/test_rendered_seo_metadata.py -k gsc_site_verification +``` + +Requires Hugo. The three selected GSC tests skipped locally because `hugo` is not installed. CI installs +Hugo 0.161.1 and runs this file in the blocking production-site job. + +### Full static production build + +```bash +HUGO_PARAMS_GA_MEASUREMENT_ID=G-TEST-OBSERVATORY \ +HUGO_PARAMS_GSC_SITE_VERIFICATION=testtoken123 \ +hugo --minify --quiet --destination /tmp/claracle-fr035 +``` + +Inspect only the test values in `/tmp/claracle-fr035/index.html`. Also build once with both variables +unset and confirm neither analytics configuration nor the GSC meta tag is emitted. + +### Consent behavior + +Use the same pinned dependencies and local server setup as `.github/workflows/ci.yml:110-177`, then run: + +```bash +BASE_URL=http://127.0.0.1:1313 \ +npx --no-install playwright test \ + --config tests/visual/playwright.config.mjs \ + tests/visual/observatory-analytics.spec.mjs \ + --project desktop-light +``` + +This is narrower than the full axe, responsive, and Lighthouse suite. It intercepts Google endpoints, +so it does not need a real GA property or send test events to Google. + +### Protected configuration metadata + +```bash +gh secret list --repo jmservera/SquadScope --json name,updatedAt \ + | jq '[.[] | select(.name == "GA_MEASUREMENT_ID" or .name == "GSC_SITE_VERIFICATION")]' +``` + +This command must never be replaced with a command that prints secret values. + +### Credential-free production smoke + +```bash +html="$(curl --fail --silent --show-error --location https://claracle.com/)" +printf 'ga_config_present=%s\n' "$(if grep -q 'gaMeasurementId' <<<"$html"; then echo yes; else echo no; fi)" +printf 'gsc_meta_present=%s\n' "$(if grep -q 'name=google-site-verification' <<<"$html"; then echo yes; else echo no; fi)" +curl --fail --silent --show-error --location --output /dev/null \ + --write-out 'sitemap_status=%{http_code}\nsitemap_content_type=%{content_type}\n' \ + https://claracle.com/sitemap.xml +``` + +This probe deliberately reports presence only and does not print the measurement ID or verification +token. + +## References and Evidence + +### Repository references + +* `.github/copilot-instructions.md` +* `.github/workflows/ci.yml` +* `.github/workflows/deploy-site.yml` +* `.squad/decisions-archive.md:1303-1327` +* `assets/js/observatory-analytics.js` +* `docs/data-observatory-runbook.md` +* `docs/growth/ga4-gsc-baseline-2026-07-29.md` +* `docs/prds/claracle-data-observatory-relaunch.md` +* `docs/review/data-observatory-relaunch/security-review.md` +* `docs/review/data-observatory-relaunch/status-of-record.md` +* `docs/setup-secrets.md` +* `hugo.toml` +* `layouts/partials/analytics.html` +* `layouts/partials/cookie-consent.html` +* `layouts/partials/head.html` +* `tests/test_pipeline.py` +* `tests/test_rendered_seo_metadata.py` +* `tests/visual/observatory-analytics.spec.mjs` + +### External references + +* [Google Analytics setup](https://support.google.com/analytics/answer/9304153): signed-in Google + account; Editor role required to create a property or add a data stream +* [Search Console ownership verification](https://support.google.com/webmasters/answer/9008080): + property addition, verification methods, owner permissions, and verification persistence +* [Search Console Sitemaps report](https://support.google.com/webmasters/answer/7451001): submission, + processing status, and the distinction between submitted and automatically discovered sitemaps + +## Follow-on Questions + +* Does the existing `GA_MEASUREMENT_ID` map to a dedicated Claracle production web stream? +* Has GSC already collected pre-verification data for a property that only needs ownership completion? +* Which approved private evidence location should retain redacted GA4 Realtime and GSC screenshots or + exports? + +## Clarifying Questions + +These questions require jmservera account observations and cannot be answered from repository or public +evidence: + +* Is the intended GA4 property already receiving consented production events? +* Does a Claracle GSC property already exist under the jmservera account? +* Should the dated baseline window begin at first verified receipt, relaunch approval, or a fixed + calendar boundary? \ No newline at end of file diff --git a/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md b/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md new file mode 100644 index 0000000..cf6614a --- /dev/null +++ b/.copilot-tracking/research/subagents/2026-08-02/claracle-rollout-cost-followup-research.md @@ -0,0 +1,422 @@ + +# Claracle Rollout and Cost Follow-Up Research + +## Research Scope + +* Investigate existing `repo_pages` rollout behavior and safety dependencies +* Investigate existing `dynamic_topic_creation` rollout behavior and safety dependencies +* Determine how Hugo and Pagefind timings are separated today +* Determine what measurable instrumentation exists for incremental generation cost Q-01/NFR-009 +* Identify the smallest implementation-ready phases and precise validations + +## Executive Findings + +* Both production creation controls are off. `repo_pages.enabled = false` and + `topic_hubs.dynamic_creation.enabled = false` are defined in + `config/observatory.toml:1-33`. +* The flags freeze mutation, not visibility. The repository contains 263 generated + repository pages plus `content/repo/_index.md`, five seed topic hubs, and three + generated data-page leaves. Hugo still renders those durable files while the + generators are disabled. +* Repository rollout is an all-current-state activation, not a first publication of + 263 pages. The July 29 implementation commits generated pages; commit `879c42c` + added `repo_pages.enabled = false`, changed dynamic creation from true to false, + and preserved the generated state and lifecycle ledger. +* Dynamic activation currently has five eligible candidates among 2,173 candidates: + AI Memory, fable, inference, Local First, and Self Hosted. Enabling the flag without + changing the ignore list can promote all five in one publish transaction. +* Hugo and Pagefind are timed separately only in the CI `production-site` job. CI + writes one report-only `reports/build-timing.json` artifact with separate + `duration_ms` values and no threshold. This is total build timing, not incremental + cost attribution by hubs, data pages, or repository pages. +* Q-01/NFR-009 remains open. One local observation exists, but there is no retained + representative series, median/p95 report, approved budget, per-page-class workload + metadata, or blocking threshold. +* The smallest safe sequence is: measure immutable workload variants first, close + security and identity blockers, canary one reviewed dynamic topic through the + existing ignore list, then activate repository regeneration as one reviewed + transaction. Repository threshold changes are not a safe canary because generated + pages outside the new expected set are treated as obsolete and can be deleted. + +## Existing Flag Behavior + +### Repository pages + +`scripts/observatory_repos.py:205-220` loads the flag, recurrence threshold, three-year +retention, lifecycle overrides, and lifecycle-ledger path. The threshold is strictly +`>` and defaults to more than three distinct weekly issues, equivalent to four weeks. + +`scripts/observatory_repos.py:997-1020` returns immediately when disabled. It does not +read or update the ledger, generate derived data, create pages, delete durable pages, +or validate staleness. The same early return applies to `--check`, so a green disabled +freshness check does not prove repository outputs are current. + +`scripts/observatory_repos.py:922-994` computes all expected page and derived outputs +when enabled. In write mode it creates or rewrites expected outputs and removes +obsolete generated pages. In check mode it returns stale, obsolete, and expired paths +without writing. This means recurrence-threshold reduction or a small threshold-based +canary can classify existing generated pages as obsolete. + +`scripts/observatory_repos.py:813-859` provides a separate lifecycle seed operation. +It requires production `repo_pages.enabled = false`, validates parity among qualified +histories, repository pages, and derived repository data, then atomically writes only +`data/derived/observatory/repository-lifecycle.json`. It is byte-stable when repeated. + +Current durable state, corroborated by +`tests/test_observatory_repos.py:570-581`, is: + +| Measure | Current value | +|---------|--------------:| +| Lifecycle histories | 2,242 | +| Qualified histories | 263 | +| Generated repository page leaves | 263 | +| Histories with stable `github_id` | 0 | +| Lifecycle statuses | 2,242 active; no checked-in rename/archive/delete transition | + +The code can absorb fallback name history into a stable GitHub ID later, as tested in +`tests/test_observatory_repos.py:584-627`, but production inputs have not supplied those +IDs. Stable canonical rename behavior is therefore implemented but not evidenced on +the production corpus. + +### Dynamic topic creation + +Candidate discovery is independent from promotion. The publish workflow always runs +`scripts/discover_topic_candidates.py`, which derives a byte-stable registry from +weekly content, analyzed summaries, and raw observations. It uses the configured +four-week threshold, 62-day lookback, ignore list, and repository recurrence threshold +from `scripts/discover_topic_candidates.py:53-67`. + +`scripts/manage_topic_hubs.py:407-419` exits before reading candidates or writing a log +when the creation flag is disabled. `--dry-run` also exits at this point when enabled; +it is a no-op safety switch, not a preview of proposed changes. + +When enabled, `scripts/manage_topic_hubs.py:421-486` can perform one transaction that: + +* Creates `content/topics//_index.md` +* Promotes the term in `data/taxonomy/topics.json` +* Assigns the topic to historical weekly frontmatter supported by evidence +* Reassigns already promoted topics from current source evidence +* Refreshes taxonomy registries +* Appends decision and summary entries to + `data/topic-hubs/dynamic-topic-creation.log` + +Creation is additive. Quiet or subsequently ineligible hubs are not deleted. Turning +the flag off stops future mutation but does not reverse promoted registry entries, +topic pages, weekly assignments, or logs. + +The current candidate registry in `data/taxonomy/topic-candidates.json` has 2,173 +candidates and five eligible candidates. Each eligible candidate has at least four +weekly issues and at least one supporting signal. The threshold alone is not editorial +approval; `fable` and `inference`, for example, are broad terms requiring human review. + +The existing `ignore_topics` list can implement a configuration-only canary by adding +four reviewed deferrals and allowing one candidate. There is no positive allowlist, +maximum creations per run, or useful non-mutating preview mode. + +## Rollout Safety Dependencies + +### Shared dependencies + +* Generated state must be hydrated from `publish` before any preflight or measurement. + The publish generator does this in + `.github/workflows/crawl-and-publish.yml:1048-1077`; deployment does it in + `.github/workflows/deploy-site.yml:88-122`. +* The `weekly-crawl` concurrency group uses `cancel-in-progress: false` at + `.github/workflows/crawl-and-publish.yml:64-66`, preventing overlapping publish + mutations. +* Generated content, taxonomy, topic logs, and repository derived state are committed + together by `.github/workflows/crawl-and-publish.yml:1224-1263` and must be reviewed + as one transaction. +* Hermes security acceptance, URL workflow review, and jmservera sponsor approval are + all pending in + `docs/review/data-observatory-relaunch/security-review.md:151-169`. +* The PRD requires separate sponsor-approved rollouts after security and lifecycle + evidence in `docs/prds/claracle-data-observatory-relaunch.md:260-267`. + +### Repository-specific dependencies + +* Run lifecycle parity seed twice while disabled against the hydrated publish revision. +* Acquire stable GitHub identity fields or explicitly disposition fallback name identity + risk before claiming FR-020 stable canonical URLs or FR-022 rename safety. +* Exercise reviewed rename, archive, and deletion evidence. Current production ledger + contains only active histories, while fixture coverage exists in + `tests/test_observatory_repos.py:664-761`. +* Run enabled `--check` and a full generation in a disposable worktree before changing + production config. Review every created, rewritten, obsolete, and expired path. +* Preserve the current threshold during activation. A threshold canary is unsafe because + the generator removes obsolete generated pages. + +### Dynamic-topic-specific dependencies + +* Hermes must disposition SEC-01. Structured YAML and adversarial title rejection are + implemented in `tests/test_topic_hubs.py:527-586`, but the security review still + records dynamic title handling as rollout-blocking. +* Review each eligible candidate's evidence, semantics, aliases, and affected weekly + files before promotion. +* Use a disposable worktree with a temporary enabled config to obtain the proposed diff, + because current `--dry-run` does not evaluate candidates. +* Select a single canary through `ignore_topics`, retain the other eligible candidates as + explicit deferrals, and obtain sponsor approval for that exact config and diff. + +## Hugo and Pagefind Timing Separation + +The CI production-site job has separate timed steps: + +* Hugo Extended 0.161.1 at `.github/workflows/ci.yml:127-134` +* Pagefind 1.5.2 at `.github/workflows/ci.yml:136-143` +* JSON report writing at `.github/workflows/ci.yml:145-156` +* Artifact retention under `production-quality-reports` at + `.github/workflows/ci.yml:181-191` + +The report schema records commit, report-only mode, tool versions, and separate +durations. `blocking_threshold_ms` is null. This correctly prevents an unapproved +budget from becoming a gate. + +Timing comparability gaps remain: + +* CI builds the checked-out branch and does not hydrate generated state from `publish`. + Deploy and crawl do hydrate it, so CI timing can measure a different workload. +* Crawl, deploy, and preview invoke unpinned `npx pagefind`; only CI pins Pagefind 1.5.2. + References: `.github/workflows/crawl-and-publish.yml:1446-1449`, + `.github/workflows/deploy-site.yml:178-181`, and + `.github/workflows/site-preview.yml:116-119`. +* The report omits source-page counts, rendered-page counts, HTML files scanned, indexed + pages, output bytes, runner identity, hydration source SHA, and workload variant. +* The artifact is ephemeral and no repository process aggregates comparable reports. +* Hugo and Pagefind are sequentially separated, but neither generator execution time nor + incremental page-class contribution is measured. + +## Existing Cost and Measurement Evidence + +`docs/design/data-observatory-model.md:400-449` records one local report-only sample: + +| Stage | Duration | Workload evidence | +|-------|---------:|-------------------| +| Hugo 0.161.1 | 6,668 ms | 2,669 rendered pages | +| Pagefind 1.5.2 | 6,207 ms | 1,477 HTML files scanned; 288 pages indexed | + +The design document correctly labels this observation as insufficient and requires +three comparable external CI reports, median and p95, a proposed blocking budget, and +owner approval. Prior validation reaches the same conclusion in +`.copilot-tracking/reviews/rpi/2026-07-30/claracle-data-observatory-relaunch-remediation-plan-007-validation.md:128-203`. + +Reusable percentile code exists in `scripts/baseline_telemetry.py:69-124`, with tests in +`tests/test_baseline_telemetry.py`, but it is scoped to crawl/analysis observability +ledgers and uses a five-run readiness baseline. It does not consume +`build-timing.json` or calculate incremental generation cost. + +The current repository provides useful workload counters but no integrated cost record: + +* 5 seed topic hubs under `content/topics/` +* 3 generated data-page leaves under `content/data/` +* 263 generated repository page leaves under `content/repo/` +* 5 currently eligible dynamic topic candidates +* Generator summaries such as `Generated repository pages` and + `dynamic-topic-summary created= skipped=` + +Q-01 is therefore measurable with existing tools, but not answered by existing data. + +## Selected Approach + +### Phase 1: Implement a report-only incremental cost experiment + +Use a disposable worktree hydrated from the same `publish` SHA for every variant. Keep +Hugo, Pagefind, runner image, config, and source revision fixed. Build into clean, +variant-specific destinations so tracked `public/` content cannot contaminate results. + +Measure these cumulative variants: + +1. Observatory generated page classes excluded +2. Add the five checked-in topic hubs +3. Add the three generated data pages +4. Add the 263 checked-in repository pages +5. Optionally add the exact reviewed dynamic canary diff + +Run at least three comparable CI repetitions because that is the documented acceptance +minimum; five runs aligns with the repository's existing baseline-telemetry convention. +For each repetition and variant, record: + +* Main SHA and hydrated publish SHA +* Runner image and tool versions +* Source Markdown counts by page class +* Hugo duration and rendered page count +* Pagefind duration, HTML files scanned, and indexed page count +* Hugo destination bytes and Pagefind index bytes +* Exit status and run/artifact URL + +Aggregate median, nearest-rank p95, absolute delta, percent delta, and marginal +milliseconds/added source page separately for Hugo and Pagefind. Do not enforce a budget +in this phase. Publish the report and obtain owner approval before replacing the null +threshold. + +### Phase 2: Prepare repository activation without enabling production + +1. Resolve or explicitly accept the missing stable-ID risk and review lifecycle evidence. +2. Hydrate the target publish revision and run lifecycle seed twice while disabled. +3. Confirm 263-way parity and byte-identical second seed. +4. In a disposable worktree, enable the existing config without changing the threshold. +5. Run `observatory_repos.py --check`, then generation twice, and review all output diffs. +6. Run repository tests, Hugo, pinned Pagefind, rendered SEO/link checks, internal links, + Lighthouse, axe, and the Q-01 workload measurement. +7. Obtain Hermes, URL, and sponsor sign-offs for the exact revision and diff. + +### Phase 3: Canary one dynamic topic + +1. Review all five eligible candidates and select one unambiguous canary. `local-first` + currently has the strongest evidence breadth with six weekly issues and five + supporting signals, but editorial/security review must make the final choice. +2. Add the other four candidates to `ignore_topics` as explicit temporary deferrals. +3. Produce the enabled result in a disposable worktree and review the hub, canonical + registry change, historical weekly assignments, taxonomy changes, and log event. +4. Run the focused and rendered validation suite and capture incremental timing. +5. Obtain approval, enable the flag for one publish run, inspect the committed generated + transaction, and turn the flag off or retain the restricted ignore list according to + the approved rollout decision. + +### Phase 4: Activate repository regeneration + +Enable `repo_pages` at the unchanged threshold only after Phases 1 and 2 pass. Treat the +first publish as a single 263-page lifecycle activation. Verify no unexpected obsolete or +expired paths before promotion. Rollback requires both disabling the flag and reverting +the generated transaction; disabling alone preserves mutations already committed. + +### Phase 5: Expand and enforce + +Remove dynamic candidate deferrals one reviewed candidate at a time. After enough stable +timing samples and explicit owner approval, add separate Hugo and Pagefind budgets with a +report-only observation period before making either threshold blocking. + +## Dependencies and Blockers + +### Blocking + +* Hermes NFR-004/security sign-off is pending, including SEC-01 dynamic-title disposition + and SEC-04 lifecycle deletion policy. +* URL workflow/security sign-off and jmservera sponsor rollout approval are pending. +* All 2,242 production repository histories lack stable GitHub IDs; stable rename identity + is not evidenced. +* Production lifecycle state has no reviewed rename, archive, or deletion transition. +* Dynamic `--dry-run` cannot preview the mutation set; a disposable worktree is required + unless preview semantics are implemented first. +* Q-01 lacks comparable CI samples, aggregation, approved budgets, and page-class + attribution. + +### Non-blocking implementation dependencies + +* Access to the canonical `publish` branch and retained workflow artifacts +* Hugo Extended 0.161.1, Pagefind 1.5.2, Python 3.12, and Node.js 24 +* A clean disposable worktree or equivalent isolated checkout per workload variant +* Existing generated-state transaction paths in crawl, deploy, and freshness workflows +* Existing tests for disabled-state preservation, deterministic generation, structured + YAML, adversarial titles, lifecycle retention, stable-ID migration, and Hugo rendering + +## Precise Validations + +### Flag and generator behavior + +```bash +python -m pytest tests/test_observatory_repos.py tests/test_topic_hubs.py tests/test_taxonomy_registry.py +python scripts/discover_topic_candidates.py --check +python scripts/generate_data_pages.py --check +python scripts/export_observatory_dataset.py --check +python scripts/export_trend_explorer_data.py --check +``` + +Run repository freshness against a temporary config with `repo_pages.enabled = true`; +the production disabled config makes `observatory_repos.py --check` a no-op. + +For repository activation, require: + +* Lifecycle seed parity is 263 qualified histories, 263 page identities, and 263 derived + identities +* The second lifecycle seed is byte-identical +* Enabled check reports no unexplained stale, obsolete, or expired paths +* Two enabled generations produce byte-identical outputs +* No page removal occurs without reviewed positive lifecycle evidence and elapsed + retention + +For dynamic canary activation, require: + +* Exactly one approved hub is created +* Only evidence-backed weekly files receive the topic assignment +* Registry YAML/JSON remains parseable and canonical aliases resolve +* A structured promotion log records evidence weeks, sources, and assigned paths +* A second run is additive and byte-stable except for explicitly designed append-log + behavior +* Disabled rollback creates or deletes nothing + +### Rendered and pipeline behavior + +```bash +hugo --minify +npx "pagefind@1.5.2" --site public/ +python scripts/check_internal_links.py public --base-url "https://claracle.com/" +python -m pytest tests/test_rendered_seo_metadata.py tests/test_rendered_weekly_links.py tests/test_internal_link_checker.py +python -m pytest tests/ +ruff check . +ruff format --check . +``` + +Also run the existing Lighthouse and axe route matrices, which include a topic, data, and +repository page at `scripts/design/lighthouse-gates.mjs:28-36` and the Observatory visual +tests. If a workflow changes, run Zizmor and Checkov under the repository guardrails. + +### Cost acceptance + +* Every timing artifact identifies both main and publish SHAs and the workload variant +* Hugo and Pagefind retain separate raw samples and statistics +* Each variant starts from a clean destination and uses pinned versions +* At least three comparable CI repetitions exist; five are preferred +* Median and p95 are reproducible from retained machine-readable samples +* Incremental deltas are reported by page class, not inferred from one total build +* The approved budget names its owner, sample window, headroom rationale, and enforcement + date +* Thresholds remain report-only until approval is recorded + +## Evidence Index + +* `config/observatory.toml:1-33` controls both disabled rollouts +* `scripts/observatory_repos.py:205-220,813-859,922-1020` defines config, lifecycle seed, + writes/checks, and disabled behavior +* `scripts/discover_topic_candidates.py:53-67` defines candidate policy inputs +* `scripts/manage_topic_hubs.py:407-486` defines disabled, dry-run, and mutation behavior +* `tests/test_observatory_repos.py:570-790` proves frozen-corpus parity, stable-ID + migration, disabled preservation, lifecycle rendering, and Hugo output +* `tests/test_topic_hubs.py:204-376,444-588` proves additive creation, persistence, + disabled preservation, unsafe-title rejection, and structured YAML +* `.github/workflows/crawl-and-publish.yml:1048-1077,1139-1210,1224-1263` hydrates, + generates, checks, and commits the generated transaction +* `.github/workflows/ci.yml:127-156,181-191` emits and uploads separate timing data +* `.github/workflows/deploy-site.yml:88-122,178-181` hydrates production generated state + and builds Hugo/Pagefind +* `.github/workflows/generate-data-pages.yml:1-66` hydrates publish state for monthly + freshness checks +* `docs/data-observatory-runbook.md:18-132` defines operating boundaries, generation + order, lifecycle policy, and seed procedure +* `docs/design/data-observatory-model.md:400-449` records the only local timing sample and + pending cost acceptance +* `docs/prds/claracle-data-observatory-relaunch.md:121-176,260-278` defines FR-004, + FR-020-022, NFR-009, rollout flags, and Q-01 +* `docs/review/data-observatory-relaunch/security-review.md:140-169` records open security + findings and pending sign-offs +* `.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md:10-18,64-74` + records these workstreams as separate follow-on plans +* Git commits `0baae6d`, `f9a17c8`, and `879c42c` establish generated-page, dynamic-topic, + and rollout-freeze provenance + +## Recommended Next Research + +* [ ] Inspect authenticated retained `production-quality-reports` artifacts from at least + three comparable successful runs; repository source cannot supply their raw values +* [ ] Confirm whether upstream crawl payloads can begin persisting `id`/`node_id`, archive, + disabled, and rename evidence before repository activation +* [ ] Have editorial/security owners disposition the five currently eligible topic + candidates and nominate an exact canary +* [ ] Obtain the named owner and approved method for the Hugo/Pagefind regression budget + +## Clarifying Questions + +* Will missing stable GitHub IDs block `repo_pages` activation, or will the sponsor accept + fallback name identity for the first activation window? +* Which eligible dynamic topic, if any, is approved as the first canary? +* Who gives final approval for separate Hugo and Pagefind budgets after the timing series? diff --git a/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md new file mode 100644 index 0000000..7739ee8 --- /dev/null +++ b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-followup-execution-plan-review.md @@ -0,0 +1,54 @@ + +# Review: Claracle Relaunch Follow-Up Execution + +## Metadata + +* Plan: .copilot-tracking/plans/2026-08-02/claracle-relaunch-followup-execution-plan.instructions.md +* Rollout plan: .copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md +* Pull request: #647 +* Date: 2026-08-02 +* Iterations: 3 + +## User Request Fulfillment + +* Complete: review corrections committed, pushed, and both PR threads resolved +* Complete: GA4 stream operation, GSC verification, root sitemap submission, and GA4-to-GSC link are owner-confirmed; baseline transcription and consent evidence remain separate measurement work +* Partial, owner-gated: automated acceptance evidence is current and owner actions are implementation-ready; manual and protected-environment approvals remain pending +* Complete: repository-page, dynamic-topic, and generation-cost work has an implementation-ready gated plan + +## Executive Findings + +1. Production GA configuration is present through the protected secret path. The empty checked-in Hugo value is an intentional fork-safe default, not evidence of disconnection. +2. GSC is verified through the owner's selected method; the optional HTML-tag secret path is not required. The submitted root sitemap is a complete URL set, so no child sitemap submissions exist. +3. SEC-01 implementation is complete and tested; SEC-04 has comprehensive lifecycle fixtures. Hermes still owns policy verification and NFR-004 sign-off. +4. Accessibility automation is strong, but NFR-005 lacks manual keyboard and screen-reader evidence. +5. Real Podcaster generation and environment-bound smoke evidence exist separately. The environment has no protection rules, and no run combines approval with real generation. +6. #622 is explicitly non-blocking polish. #626 is independent hardening with unchanged thresholds. Both can affect final visual recapture but are not approval evidence. +7. Both rollout flags remain disabled. Repository pages lack production stable IDs and lifecycle transitions; dynamic creation lacks a non-mutating preview and has five eligible candidates. +8. Hugo/Pagefind timing separation is complete. Q-01 still needs comparable workload variants, retained samples, aggregation, and budget approval. +9. The standalone embed currently renders the same secret-backed GA configuration as the main site. A dedicated contract now prevents its separate base template from dropping analytics or consent wiring. +10. Squad implemented SEC-02/03/05. The first SEC-02 browser assertion was rejected, revised to prove cross-origin default-off behavior, then approved by Fry. Human Hermes disposition remains pending. + +## Validation + +* `python3 -m pytest -q tests/`: 1,392 passed, 19 skipped, 34 subtests passed +* Focused analytics, topic, lifecycle, export, and link suite: 45 passed, 4 Hugo-dependent skips +* `ruff check .` and `ruff format --check .`: passed +* Data-page, public dataset, and trend-export checks: passed +* Two rollout flags confirmed disabled +* Editor diagnostics: no errors +* `git diff --check`: passed +* PR #647 checks for `8fddceb`: 13 successful checks + +## Blockers Requiring Named Owners + +* jmservera: GSC export transcription, production consent observations, sitemap processing review, and separate rollout decisions +* Hermes: SEC-01 through SEC-06 dispositions and NFR-004 sign-off +* URL and repository administrator: protected real-generation environment and secret scope +* Podcaster maintainer: idempotency confirmation or one-run authorization +* Fry and accessibility reviewer: manual NFR-005 evidence +* Amy: refreshed visual matrix and acceptance conclusion + +## Overall Status + +Complete for repository-executable work and planning. External acceptance remains owner-gated and is not falsely marked complete. The branch is ready to publish and validate through PR CI. diff --git a/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md new file mode 100644 index 0000000..90023eb --- /dev/null +++ b/.copilot-tracking/reviews/2026-08-02/claracle-relaunch-readiness-reconciliation-plan-review.md @@ -0,0 +1,40 @@ + +# Review: Claracle Relaunch Readiness Reconciliation + +## Review Metadata + +* Plan: .copilot-tracking/plans/2026-08-02/claracle-relaunch-readiness-reconciliation-plan.instructions.md +* Pull request: #647 +* Reviewer: RPI Agent +* Date: 2026-08-02 +* Iterations: 1 + +## User Request Fulfillment + +* Complete: reviewed the three relaunch plans, PRD, and BRD against delivered repository state +* Complete: reconciled stale plan checklists and created one status of record +* Complete: aligned PRD v1.3 and BRD v1.2 with the delivered #627-#646 workstream +* Complete: consolidated pending launch gates with owners, dependencies, and evidence paths +* Complete: captured deferred implementation work as follow-on items + +## Review Findings + +PR review identified two stale state claims. Issues #599 and #644 closed as completed on 2026-08-01, but the initial research still called both open and the status register presented #599 as a pending issue anchor. The corrected record distinguishes issue disposition from acceptance state: #599 is closed, while its human-action checklist and FR-035 acceptance evidence remain pending. + +No placement or architecture concerns remain. The status of record is the owning readiness view, while research, plans, and logs retain supporting traceability. + +## Validation + +* `pytest -q tests/test_internal_link_checker.py tests/test_embed_sources.py`: 10 passed +* PR #647 checks: 13 successful checks, no failures, no approval requirement, and no requested changes +* Local `hugo --minify`: unavailable because Hugo is not installed; PR #647 Production site check passed +* Editor diagnostics: no errors in the corrected files +* `git diff --check`: passed + +## Pull Request Threads + +Both review comments are addressed locally. The threads remain unresolved until the correction commit is pushed so reviewers can inspect the updated PR diff. + +## Overall Status + +Complete. The implementation fulfills the recorded user requests and local validation passes. Committing and pushing the review corrections is the only remaining delivery action. \ No newline at end of file diff --git a/content/privacy/_index.md b/content/privacy/_index.md index 58a03e8..ad32206 100644 --- a/content/privacy/_index.md +++ b/content/privacy/_index.md @@ -29,6 +29,12 @@ Claracle uses Google Analytics 4 (GA4) **only if you accept the analytics catego After consent, GA4 helps us understand whether the site is useful: page views, referrers, session duration, device/browser information, and approximate location derived from network data. The GA4 measurement ID is configured per deployment through a repository secret, not hard-coded in this page. Google's processing is governed by [Google's Privacy Policy](https://policies.google.com/privacy). You can also use the [Google Analytics opt-out browser add-on](https://tools.google.com/dlpage/gaoptout). +### Charts embedded on other sites + +Claracle's official chart iframe snippet uses `referrerpolicy="no-referrer"`. If a publisher uses the snippet unchanged, the embedding page URL is not sent as the iframe request referrer. Publishers control their copied HTML and may alter that attribute. + +Analytics in an embedded Claracle chart starts off. It can be enabled only when you explicitly accept Claracle analytics in the consent controls shown inside the iframe. A choice made on the embedding website is not treated as Claracle consent. Some browsers block third-party storage, so an iframe choice may not persist and the prompt may reappear; storage failure does not turn analytics on. + ### Google Fonts Claracle loads Inter and JetBrains Mono from Google Fonts. When your browser requests those font files, Google may receive request metadata such as your IP address and user-agent under [Google's Privacy Policy](https://policies.google.com/privacy). @@ -103,8 +109,9 @@ GitHub and Google may process data in countries outside your own. GA4 data may b ## Changes to this policy -Last updated: 2026-06-12. Changes are announced through the git history of this page in the public SquadScope repository, so you can review what changed and when. +Last updated: 2026-08-02. Changes are announced through the git history of this page in the public SquadScope repository, so you can review what changed and when. +**2026-08-02:** Documented the no-referrer iframe snippet and frame-local, explicit analytics consent model. **2026-06-12:** Added Signal Check podcast section covering TTS provider, staging storage, and platform disclosures. ## Contact diff --git a/docs/brds/claracle-data-observatory-relaunch-brd.md b/docs/brds/claracle-data-observatory-relaunch-brd.md index 1938c68..5e39c0b 100644 --- a/docs/brds/claracle-data-observatory-relaunch-brd.md +++ b/docs/brds/claracle-data-observatory-relaunch-brd.md @@ -2,7 +2,7 @@ title: "Claracle Data Observatory Relaunch — Business Requirements Document" description: "BRD for the next version of the Claracle site, repositioning it from weekly AI-generated summaries into a discoverable, linkable public database of GitHub technology trends to solve the discovery/SEO problem." author: "BRD Builder (facilitated)" -ms.date: 2026-07-30 +ms.date: 2026-08-02 ms.topic: reference --- @@ -12,17 +12,25 @@ ms.topic: reference |-------|-------| | BRD ID | BRD-CLARACLE-002 | | Status | Acceptance and sponsor approval pending | -| Version | 1.1 | +| Version | 1.2 | | Author | BRD Builder (facilitated) | | Sponsor | jmservera (also the human approval authority) | -| Last updated | 2026-07-29 | +| Last updated | 2026-08-02 | | Related repositories | SquadScope, SquadScope-Podcaster, SquadScope-Coordinator | +### Change History + +| Version | Date | Author | Summary | +|---------|------|--------|---------| +| 1.0 | 2026-07-29 | BRD Builder (facilitated) | Initial BRD repositioning Claracle into a data observatory | +| 1.1 | 2026-07-30 | BRD Builder (facilitated) | Reconciled acceptance status with pending security, analytics, production, Podcaster, accessibility, visual, and rollout gates | +| 1.2 | 2026-08-02 | SquadScope Squad | Added this change history, aligned the PRD cross-reference, linked the status of record, and reconciled the completed GA4/GSC connection | + --- ## Acceptance Status -The business requirements remain approved as requirements, but no repository artifact records sponsor approval to enable either rollout flag. Security sign-off, analytics and search evidence, production responses, Podcaster execution, accessibility review, and refreshed visual acceptance remain pending. Dynamic topic creation and repository-page creation must be approved separately. +The business requirements remain approved as requirements, but no repository artifact records sponsor approval to enable either rollout flag. The GA4/GSC connection is complete by owner confirmation; dated baseline values and production consent observations remain pending. Security sign-off, remaining production responses, Podcaster execution, accessibility review, and refreshed visual acceptance also remain pending. Dynamic topic creation and repository-page creation must be approved separately. Current delivered-versus-pending status is tracked in the [relaunch status of record](../review/data-observatory-relaunch/status-of-record.md). --- @@ -316,7 +324,7 @@ Aligned to the SEO analysis phasing; final sequencing is a delivery decision. Resolved in elicitation (2026-07-29): - ✅ Sponsor and approval authority: **jmservera** (human approval authority). -- Analytics measurement stack: **GA4 + GSC selected**; property state and numeric baseline remain pending external evidence. +- ✅ Analytics measurement stack and connection: **GA4 + GSC selected and connected**; the numeric baseline and production consent observations remain pending. - ✅ Wave 1 topic hubs: AI Coding Agents, MCP Ecosystem, Open-Source LLMs, Developer Tools, plus one vertical (e.g., AI Agents in Healthcare) — kept **trend-aligned and dynamic** (BR-004). - ✅ Dataset licensing: **MIT**, cite all references (BR-050). - ✅ Free tool: **client-side-only**, specific tool chosen via design spike (BR-052). @@ -330,5 +338,5 @@ Still open: 1. Quantify **incremental generation cost/time** for hubs, data, and repository pages. 2. Record Hermes security sign-off and disposition of open review findings. -3. Capture GA4, GSC, production, Podcaster, accessibility, and visual acceptance evidence. +3. Capture the GA4/GSC dated baseline, production consent observations, remaining production responses, Podcaster, accessibility, and visual acceptance evidence. 4. Obtain separate sponsor approval before enabling dynamic topic creation or repository-page creation. diff --git a/docs/data-observatory-runbook.md b/docs/data-observatory-runbook.md index 808edea..db41ba7 100644 --- a/docs/data-observatory-runbook.md +++ b/docs/data-observatory-runbook.md @@ -171,17 +171,22 @@ Review these signals after an approved production deployment: ## Cross-origin embed privacy Claracle's consent controls govern scripts, cookies, and telemetry on the Claracle origin. They -cannot inspect, suppress, or withdraw storage and network activity initiated inside a -cross-origin iframe. The repository does not currently contain approved evidence that a -third-party embed receives or honors Claracle's consent state. - -Treat analytics-bearing cross-origin embeds as blocked until Hermes records an approved consent -and referrer policy with reproducible browser evidence. Before publication, test the fresh, -denied, granted, reloaded, and withdrawn consent states while capturing iframe requests and -storage behavior. If an embed sends requests before approval, continues after withdrawal, or -cannot be inspected reliably, remove or disable the embed, stop publication of the affected -page, preserve the request evidence and deployed revision, and escalate to Hermes and URL. Do -not describe the embed as consent-compliant while that review is pending. +cannot inherit or inspect consent collected by an embedding site. Use the generated official +snippet unchanged: it includes `referrerpolicy="no-referrer"`. Publishers can alter copied iframe +attributes, so production review must inspect the markup actually deployed by the publisher. + +The iframe starts Claracle analytics disabled. Only an explicit analytics choice in the Claracle +consent UI inside that frame can enable telemetry; parent-page consent is not inferred or +transferred. Third-party storage blocking may prevent the choice from persisting and cause another +prompt, but it must never enable analytics. + +Before publication, test the fresh, denied, granted, reloaded, and withdrawn frame-local consent +states while capturing iframe requests and storage behavior. Confirm the deployed iframe retains +`no-referrer`. If an embed sends requests before Claracle consent, continues after withdrawal, or +cannot be inspected reliably, remove or disable the embed, stop publication of the affected page, +preserve the request evidence and deployed revision, and escalate to Hermes and URL. Repository +tests establish the implementation contract; Hermes approval and production evidence remain +pending. ## Failed-run recovery diff --git a/docs/growth/ga4-gsc-baseline-2026-07-29.md b/docs/growth/ga4-gsc-baseline-2026-07-29.md index 43161fa..366f1d4 100644 --- a/docs/growth/ga4-gsc-baseline-2026-07-29.md +++ b/docs/growth/ga4-gsc-baseline-2026-07-29.md @@ -2,7 +2,7 @@ title: GA4 and GSC Launch Baseline for 2026-07-29 description: Dated Claracle analytics and search baseline record that separates repository wiring from pending external platform evidence author: SquadScope Squad -ms.date: 2026-07-30 +ms.date: 2026-08-02 ms.topic: reference keywords: - google analytics 4 @@ -14,22 +14,34 @@ estimated_reading_time: 6 ## Baseline status -The 2026-07-29 launch baseline has not been captured from GA4 or Google Search -Console. No numeric baseline value is asserted in this record. The repository contains -conditional integration paths, but repository inspection cannot prove property setup, -secret presence, consent behavior in production, data receipt, verification, sitemap -submission, or indexing. +The production GA4 stream and Google Search Console property are connected. On +2026-08-02, jmservera confirmed the intended GA4 stream, verified the GSC property, +submitted the root sitemap, and linked the GA4 stream to GSC. A Search Console +performance export was supplied for the dated baseline, but its numeric values have not +yet been transcribed into this record. The configured production target is `https://claracle.com/`, and the expected standard -sitemap target is `https://claracle.com/sitemap.xml`. Their production responses remain -unverified for this acceptance record. +sitemap target is `https://claracle.com/sitemap.xml`. + +## Credential-free production observations + +| Observation | Result | Date | Acceptance boundary | +| ----------- | ------ | ---- | ------------------- | +| GA configuration rendered | Present | 2026-08-02 | Does not reveal or validate the identifier, property, stream, consent behavior, or receipt | +| GSC verification meta tag | Absent | 2026-08-02 | GSC ownership remains unverified | +| Sitemap response | HTTP 200, `application/xml` | 2026-08-02 | Does not prove submission or processing in GSC | +| `GA_MEASUREMENT_ID` secret name | Present | 2026-08-02 | Secret value is not observable and must not be recorded | +| GSC property verification | Complete by owner attestation | 2026-08-02 | Property verified without requiring a public verification meta tag | +| Root sitemap submission | Complete by owner attestation | 2026-08-02 | Root `sitemap.xml` is a complete ``, not an index of child sitemaps | +| GA4 and GSC product link | Complete by owner attestation | 2026-08-02 | Does not replace GA4 or GSC baseline values | +| Standalone embed GA configuration | Present | 2026-08-02 | The affected embed renders the same secret-backed configuration as the main site | ## Repository-verifiable implementation | Surface | Repository status | Evidence boundary | | -------------------------- | ----------------------------------------- | ----------------------------------------------------------------------------------------------- | -| GA4 build parameter | Implemented conditionally | `deploy-site.yml` reads `GA_MEASUREMENT_ID`; secret existence and value are not observable | -| GSC verification parameter | Implemented conditionally | `deploy-site.yml` reads `GSC_SITE_VERIFICATION`; verification is not observable | +| GA4 build parameter | Implemented and present in production | `deploy-site.yml` reads `GA_MEASUREMENT_ID`; public presence does not validate the protected value | +| GSC verification parameter | Available but not required by the completed verification method | `deploy-site.yml` supports `GSC_SITE_VERIFICATION` when HTML-tag verification is selected | | Fork-safe defaults | Implemented | Hugo parameters default empty, so an unconfigured build does not inherit production identifiers | | Analytics consent gate | Implemented in templates and browser code | Production requests and cookies require browser evidence | | Observatory events | Implemented with a consent check | GA4 receipt and payload inspection require browser and Realtime evidence | @@ -43,13 +55,14 @@ the secret. | Evidence | Status | Owner | Required proof | | --------------------------- | ------- | -------------------- | -------------------------------------------------------------------------------- | -| GA4 property and web stream | Pending | jmservera | Dated property or stream evidence with identifiers redacted where appropriate | +| GA4 property and web stream | Complete | jmservera | Owner confirmed the intended production stream on 2026-08-02 | | Consent denied behavior | Pending | jmservera and Hermes | Private first-visit network and cookie evidence showing no GA4 request or cookie | | Consent granted behavior | Pending | jmservera and Hermes | Network evidence showing the expected GA4 request after consent | -| GA4 Realtime receipt | Pending | jmservera | Dated Realtime evidence correlated to the consented test visit | -| GSC property verification | Pending | jmservera | Dated verified-property evidence | -| GSC sitemap submission | Pending | jmservera | Submission URL, date, and platform status | -| Production sitemap response | Pending | jmservera | Dated response status and content-type evidence | +| GA4 Realtime receipt | Complete by owner attestation | jmservera | GA4 declared operational on 2026-08-02; retain a redacted platform capture if formal audit evidence is required | +| GSC property verification | Complete | jmservera | Owner confirmed verified property on 2026-08-02 | +| GSC sitemap submission | Complete | jmservera | `https://claracle.com/sitemap.xml` submitted on 2026-08-02 | +| GA4 and GSC product link | Complete | jmservera | Owner confirmed the production stream is linked to the GSC property | +| Production sitemap response | Complete | jmservera | HTTP 200 with `application/xml` observed on 2026-08-02 | | Production feed responses | Pending | jmservera | Dated site and topic feed response status and content types | ## Baseline values @@ -76,13 +89,15 @@ values with estimates such as “near zero.” 3. Record consent-denied network and cookie behavior in a private browser session. 4. Grant analytics consent and record the expected request. 5. Correlate that visit with GA4 Realtime and record the observation date. -6. Verify the GSC property, submit the configured sitemap, and record platform status. -7. Capture GA4 acquisition values and GSC performance values for the same documented +6. Confirm the submitted sitemap is processed successfully and review indexed versus + excluded URLs in GSC. +7. Capture GA4 acquisition values and transcribe the supplied GSC performance export for the same documented baseline window. 8. Link the evidence from the relaunch review index and retain redacted artifacts in the approved evidence location. ## Acceptance rule -NFR-007, FR-035, and the analytics portion of NFR-008 remain pending. They may be marked -accepted only after the external evidence matrix contains dated proof and actual values. +FR-035 connection and submission are complete. NFR-007 baseline measurement and the +production-consent portion of NFR-008 remain pending until the supplied performance +export is transcribed and dated consent observations are retained. diff --git a/docs/prds/claracle-data-observatory-relaunch.md b/docs/prds/claracle-data-observatory-relaunch.md index af076e6..927f2d9 100644 --- a/docs/prds/claracle-data-observatory-relaunch.md +++ b/docs/prds/claracle-data-observatory-relaunch.md @@ -2,12 +2,12 @@ title: Claracle Data Observatory Relaunch Product Requirements Document description: Product requirements, delivery state, rollout controls, risks, and acceptance gates for the Claracle Data Observatory relaunch author: SquadScope Squad -ms.date: 2026-07-31 +ms.date: 2026-08-02 ms.topic: reference --- -Version 1.2 | Status Acceptance pending | Owner jmservera | Team SquadScope Squad | Target Wave 1 (foundation) | Lifecycle Definition +Version 1.3 | Status Acceptance pending | Owner jmservera | Team SquadScope Squad | Target Wave 1 (foundation) | Lifecycle Definition ## Progress Tracker | Phase | Done | Gaps | Updated | @@ -18,14 +18,14 @@ Version 1.2 | Status Acceptance pending | Owner jmservera | Team SquadScope Squa | Requirements | Yes | Incremental generation cost still to quantify | 2026-07-30 | | Metrics & Risks | Yes | None | 2026-07-29 | | Operationalization | Yes | Star Velocity Explorer selected; production evidence pending | 2026-07-30 | -| Finalization | No | Security, external platform, Podcaster, accessibility, visual, and sponsor gates remain open | 2026-07-30 | -Unresolved launch gates: 6 | TBDs: 1 (incremental generation cost) +| Finalization | No | Security, baseline and consent, Podcaster, accessibility, visual, and sponsor gates remain open | 2026-08-02 | +Unresolved launch gates: See the launch-gate register | TBDs: 1 (incremental generation cost) ## Acceptance Status -Repository implementation is present, but GA acceptance is pending. Hermes has not signed NFR-004; GA4/GSC, production responses, external schema and social debuggers, a downstream Podcaster run, accessibility review, and refreshed visual evidence are not recorded. `dynamic_topic_creation` and `repo_pages` remain off and require separate sponsor-approved rollout changes. +Repository implementation is present, and the GA4/GSC connection is complete by owner confirmation. Dated baseline values and production consent observations remain pending. Hermes has not signed NFR-004; remaining production responses, external schema and social debuggers, a downstream Podcaster run, accessibility review, and refreshed visual evidence are not recorded. `dynamic_topic_creation` and `repo_pages` remain off and require separate sponsor-approved rollout changes. Delivered-versus-pending status and the launch-gate register are tracked in the [relaunch status of record](../review/data-observatory-relaunch/status-of-record.md). -Derived from: `docs/brds/claracle-data-observatory-relaunch-brd.md` (BRD-CLARACLE-002, v1.0). +Derived from: `docs/brds/claracle-data-observatory-relaunch-brd.md` (BRD-CLARACLE-002, v1.2). ## 1. Executive Summary ### Context @@ -135,9 +135,9 @@ Extends the existing PaperMod theme and topic layouts. New page types: topic hub | FR-032 | OG + Twitter cards | Emit Open Graph and Twitter/X tags including image, `og:image:alt`, width/height, `article:author`, `twitter:creator`, plus a homepage/fallback OG image. | G-005 | Data citer | Must | FB/Twitter debuggers render valid previews with image on homepage and articles; closes gaps 1-7 in distribution-strategy.md. | Uses existing `default_social_image` param. | | FR-033 | Structured data | Article pages emit Schema.org Article; hub/data/repo pages emit appropriate schema; all hierarchical pages emit Breadcrumb schema. | G-005,G-004 | Search visitor | Must | Google Rich Results Test validates Article + Breadcrumb with no errors. | Extends `schema_json.html`. | | FR-034 | Sitemap + RSS | Publish Hugo's built-in `sitemap.xml` and RSS feeds (site-wide + per topic). No news sitemap. | G-005,G-002 | Search visitor | Must | Sitemap and feeds reachable/valid; per-topic feeds resolve. | Hugo `outputs` already emit taxonomy RSS. | -| FR-035 | Search Console | Connect and verify Google Search Console (and confirm GA4) for the production domain; submit sitemap. | G-002,G-004 | Search visitor | Must | GSC property verified, sitemap submitted; GA4 receiving data. | `ga_measurement_id` currently empty. | +| FR-035 | Search Console | Connect and verify Google Search Console (and confirm GA4) for the production domain; submit sitemap. | G-002,G-004 | Search visitor | Must | GSC property verified, sitemap submitted; GA4 receiving data. | Complete by owner confirmation on 2026-08-02: GA4 stream operational, GSC verified, root sitemap submitted, and products linked. | | FR-040 | Internal linking | Every weekly article links to previous/next week, its topic hubs, and referenced repository/technology pages. | G-001,G-004 | Signal-seeker | Must | Rendered weekly pages contain prev/next, topic-hub, and repo links where applicable. | PaperMod `ShowPostNavLinks` already on. | -| FR-041 | Link-check gate | Validate internal links in CI; broken internal links fail the build gate. | G-005 | - | Should | CI link-check runs and fails on broken internal links. | New CI step. | +| FR-041 | Link-check gate | Validate internal links in CI; broken internal links fail the build gate. | G-005 | - | Should | CI link-check runs and fails on broken internal links. | Partial: satisfied at test level (`tests/test_internal_link_checker.py`); no standalone CI link tool. | | FR-050 | Downloadable datasets | Offer MIT-licensed downloadable datasets (e.g., CSV of top projects, weekly trend archive) with all sources cited and a stable link. | G-003 | Data citer | Should | >= 1 dataset published under MIT with citation/attribution note and stable download URL. | Maps BR-050. | | FR-051 | Embeddable charts | Generate charts (growth curves, rankings, momentum) with an embed snippet that links back to Claracle. | G-003 | Data citer | Should | >= 1 chart type embeddable via provided snippet with backlink. | Shortcode/layout, not raw HTML. | | FR-052 | Client-side tool | Provide >= 1 free, client-side-only interactive tool (e.g., trend explorer, star-velocity tracker), selected via a design spike weighing discoverability value, effort, and static-hosting fit. | G-003,G-002 | Search visitor, Data citer | Could | Design spike recommends one tool with rationale; tool ships and runs fully in-browser with no backend. | Maps BR-052. | @@ -165,7 +165,7 @@ Discovery engine | NFR ID | Category | Requirement | Metric/Target | Priority | Validation | Notes | |--------|----------|------------|--------------|----------|-----------|-------| | NFR-001 | Performance | New page types keep the site fast on static hosting | Lighthouse Performance >= 90 on hub/data/repo pages | Should | Lighthouse CI or manual audit | Static pre-rendered; charts lazy-loaded | -| NFR-002 | Reliability | No regression to weekly pipeline or handoff | Weekly pipeline success + handoff smoke unchanged (G-006) | Must | `podcaster-handoff-smoke.yml`, pytest | Crawl untouched | +| NFR-002 | Reliability | No regression to weekly pipeline or handoff | Weekly pipeline success + handoff smoke unchanged (G-006) | Must | `podcaster-handoff-smoke.yml`, pytest | Crawl untouched; restore mode preserves the published weekly transaction (article, summary, promotion record, rollups) rather than overwriting provenance (`#640`/`#646`) | | NFR-003 | Maintainability | Thresholds are configuration, not code | Topic (FR-004) and repo (FR-021) thresholds set via config | Must | Change threshold with no code edit in review | | | NFR-004 | Security | No raw HTML in AI content; no secrets in client tool | `unsafe=false` retained; tool ships no secrets; inputs sanitized | Must | Build config check; Hermes review; `sanitize_repo_content` path | Follows prompt-injection guardrails | | NFR-005 | Accessibility | New pages meet WCAG 2.1 AA basics | Images have alt; charts have text alternative; contrast passes | Should | axe/Lighthouse a11y audit | OG alt also required (FR-032) | @@ -208,7 +208,7 @@ Generated Hugo content: topic hubs (taxonomy terms), data pages, repository page | `generate_content.py` | Internal code | High | Farnsworth | Regression on frontmatter | Add `topics` behind tests (FR-002) | | Hugo `topic` taxonomy + `layouts/topics` | Platform | High | Amy | Layout gaps | Extend existing templates | | SEO partials (`opengraph`, `twitter_cards`, `schema_json`) | Platform | Medium | Amy | Incomplete tags | Close catalogued gaps (FR-032/033) | -| GA4 + GSC | External | High | jmservera | Access/verification | Verify property early (FR-035) | +| GA4 + GSC | External | High | jmservera | Baseline and consent evidence | Capture dated values and production consent observations (FR-035, NFR-007/008) | | Podcaster handoff contract | Cross-repo | High | URL/Hermes | Contract break | Keep payload unchanged; smoke test | | Client-side charting/tool library | External | Medium | Amy | Static-hosting fit | Design spike (FR-051/052) | @@ -265,6 +265,8 @@ AI-generated content must not render raw HTML (`unsafe=false`); repo-derived tex | dynamic_topic_creation | Gate auto-creation of new topic hubs | Off; no rollout approval recorded | Separate sponsor approval after security and acceptance evidence | | repo_pages | Gate repository-page generation | Off; no rollout approval recorded | Separate sponsor approval after lifecycle and acceptance evidence | +Owners, dependencies, and evidence paths for every launch gate (including sponsor rollout approval) are consolidated in the [relaunch status of record launch-gate register](../review/data-observatory-relaunch/status-of-record.md#launch-gate-register). + ### Communication Plan (Optional) Use the existing per-week distribution playbook (`docs/growth/distribution-strategy.md`) for launch posts; announce Wave 2 dataset and Wave 3 tool on developer communities (HN, Lobsters, dev.to) when genuinely useful. @@ -273,11 +275,12 @@ Use the existing per-week distribution playbook (`docs/growth/distribution-strat |------|----------|-------|---------|--------| | Q-01 | Quantify incremental generation cost/time for hubs, data, and repo pages | Leela | Design spike | Open | | Q-02 | Which client-side tool to build first (FR-052) | Amy | 2026-07-30 | Resolved: Star Velocity Explorer; see ADR | -| Q-03 | When can `content/data/` deploy hydration be restored (once the crawl reliably publishes observatory pages to `publish`)? | Bender | Post-#627 crawl run | Open | +| Q-03 | When can `content/data/` deploy hydration be restored (once the crawl reliably publishes observatory pages to `publish`)? | Bender | Post-#627 crawl run | Resolved: hydration restored via `#637` after the crawl repopulated `publish`; CI embed-source guard (`#641`) prevents recurrence | ## 15. Changelog | Version | Date | Author | Summary | Type | |---------|------|-------|---------|------| +| 1.3 | 2026-08-02 | SquadScope Squad | Reconciled the #627-#646 workstream: deploy/hydration parity restored and CI embed-source guard shipped (`#634`/`#637`/`#641`), Podcaster smoke hardened (`#636`/`#639`/`#643`/`#645`), restore preserves the published weekly transaction (NFR-002; `#640`/`#646`); recorded FR-041 partial status and linked the status of record | Updated | | 1.2 | 2026-07-31 | SquadScope Squad | Recorded the deploy hydration content-provenance failure (issue #627), the interim `content/data` fix, and the deploy/CI parity requirement (NFR-011/012, R-08) | Updated | | 1.1 | 2026-07-30 | SquadScope Squad | Reconciled repository delivery with pending external, security, visual, accessibility, Podcaster, and rollout gates | Updated | | 1.0 | 2026-07-29 | PRD Builder (facilitated) | Initial PRD derived from BRD-CLARACLE-002 v1.0 | Created | @@ -294,6 +297,7 @@ Use the existing per-week distribution playbook (`docs/growth/distribution-strat | REF-7 | Analysis | External SEO analysis (user-provided, 2026-07) | Discovery-first strategy | Primary driver | | REF-8 | ADR | `docs/decisions/adr-star-velocity-explorer.md` | FR-052 selection and static-hosting rationale | Resolves Q-02 | | REF-9 | Review | `docs/review/data-observatory-relaunch/README.md` | Bounded acceptance evidence and pending gates | Source of release status | +| REF-10 | Review | `docs/review/data-observatory-relaunch/status-of-record.md` | Reconciled delivered-versus-pending status and launch-gate register | Single readiness view | ### Citation Usage Functional requirements cite BRD requirement IDs (BR-xxx) inline; technical claims cite the repository files above. diff --git a/docs/review/data-observatory-relaunch/README.md b/docs/review/data-observatory-relaunch/README.md index f10f883..f03471d 100644 --- a/docs/review/data-observatory-relaunch/README.md +++ b/docs/review/data-observatory-relaunch/README.md @@ -2,7 +2,7 @@ title: Data Observatory Relaunch Acceptance Evidence description: Bounded evidence index for repository implementation, external launch gates, security review, and visual acceptance of the Claracle relaunch author: SquadScope Squad -ms.date: 2026-07-30 +ms.date: 2026-08-02 ms.topic: reference keywords: - acceptance evidence @@ -18,8 +18,12 @@ Repository implementation evidence is available, but relaunch acceptance is inco Dynamic topic creation and repository-page creation remain disabled in `config/observatory.toml`. This index does not authorize either rollout. -External platform, production, cross-repository run, security sign-off, accessibility -review, and visual acceptance evidence remain pending as listed below. +External baseline and consent, remaining production responses, cross-repository run, +security sign-off, accessibility review, and visual acceptance evidence remain pending +as listed below. The GA4/GSC connection itself is complete. + +The [owner action register](owner-action-register.md) sequences the remaining human and +protected-environment work without treating repository automation as approval evidence. ## Evidence principles @@ -38,30 +42,37 @@ review, and visual acceptance evidence remain pending as listed below. | FR-052 tool selection and architecture rationale | Complete | [Star Velocity Explorer ADR](../../decisions/adr-star-velocity-explorer.md) | | Security and privacy surface review | Complete with open findings | [Security review](security-review.md) | | Hermes security acceptance | Pending | Security review sign-off table | -| GA4 and GSC repository wiring | Implemented conditionally | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | +| GA4 and GSC connection | Complete | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | | GA4 and GSC external baseline values | Pending | Dated baseline external evidence matrix | | Product delivery and rollout status | Pending acceptance | [PRD](../../prds/claracle-data-observatory-relaunch.md) | | Sponsor-approved lifecycle state | Pending | [BRD](../../brds/claracle-data-observatory-relaunch-brd.md) | | Visual capture requirements | Pending | [Screenshot capture checklist](screenshots/README.md) | +| Owner-gated acceptance actions | Pending | [Owner action register](owner-action-register.md) | ## External acceptance matrix | Gate | Status | Actor or access needed | Required evidence | | ------------------------------------- | ------- | --------------------------------------------------- | ---------------------------------------------------------- | -| GSC property verification | Pending | jmservera with GSC access | Dated verified-property evidence | -| GSC sitemap submission | Pending | jmservera with GSC access | Submitted sitemap target and platform status | +| GSC property verification | Complete | jmservera | Owner confirmed verification on 2026-08-02 | +| GSC sitemap submission | Complete | jmservera | Root `sitemap.xml` submitted on 2026-08-02 | | GA4 consent-denied behavior | Pending | jmservera and Hermes with production browser access | Network and cookie evidence from a private first visit | | GA4 consent-granted behavior | Pending | jmservera and Hermes with production browser access | Expected request after consent | -| GA4 Realtime receipt | Pending | jmservera with GA4 access | Dated Realtime observation correlated to test visit | +| GA4 property, stream, and receipt | Complete by owner attestation | jmservera | Intended production stream confirmed operational | | Social preview debuggers | Pending | Reviewer with external debugger access | Homepage and article conclusions with retained links | | Rich Results Test | Pending | Reviewer with external debugger access | Article and breadcrumb conclusions with retained links | | Schema.org validator | Pending | Reviewer with external debugger access | Relevant page-type conclusions with retained links | -| Production sitemap and feed responses | Pending | jmservera with production access | Status, content type, date, and tested target | +| Production sitemap response | Complete | jmservera | HTTP 200 `application/xml` observed on 2026-08-02 | +| Production feed responses | Pending | jmservera with production access | Status, content type, date, and tested target | | Podcaster downstream run | Pending | Podcaster maintainer and protected environment | Successful downstream run conclusion and Actions link | | Accessibility review | Pending | Fry and accessibility reviewer | Automated results plus keyboard and screen-reader findings | | Hermes sign-off | Pending | Hermes | Dated disposition of security findings and NFR-004 | | Sponsor rollout approval | Pending | jmservera | Dated approval identifying each flag separately | +Issue #622 is non-blocking UX polish according to its issue contract. Issue #626 is +independent quality hardening whose existing thresholds remain unchanged. Both should be +completed before final visual recapture where their changes affect the rendered result, +but neither is represented as an unevidenced acceptance approval. + ## Visual evidence status The existing ten PNG files are retained as historical local captures. Their current index diff --git a/docs/review/data-observatory-relaunch/owner-action-register.md b/docs/review/data-observatory-relaunch/owner-action-register.md new file mode 100644 index 0000000..2801d92 --- /dev/null +++ b/docs/review/data-observatory-relaunch/owner-action-register.md @@ -0,0 +1,133 @@ +--- +title: Data Observatory Relaunch Owner Action Register +description: Sequenced owner actions and evidence requirements for Claracle relaunch gates that cannot be completed by repository automation +author: SquadScope Squad +ms.date: 2026-08-02 +ms.topic: reference +keywords: + - launch gates + - acceptance evidence + - owner actions + - rollout approval +estimated_reading_time: 7 +--- + +## Purpose + +Repository automation proves implementation behavior, not external platform state or +human approval. This register defines the remaining actions, actors, and completion +evidence without recording secret values or granting approval by implication. + +## Analytics and search acceptance + +Owner: jmservera, with Hermes reviewing production consent behavior. + +Current evidence: + +* Production renders secret-backed GA configuration on the main site and standalone embed +* The `GA_MEASUREMENT_ID` secret name exists +* Production serves `https://claracle.com/sitemap.xml` as HTTP 200 and `application/xml` +* jmservera confirmed the intended GA4 stream, verified GSC property, root sitemap submission, and GA4-to-GSC product link on 2026-08-02 +* The root sitemap is one complete `` rather than a sitemap index, so there are no child sitemaps to submit + +Required actions: + +1. Record denied and granted production consent behavior without exposing identifiers. +2. Transcribe the supplied GSC performance export and capture GA4 values for one explicit date range. +3. Confirm GSC finishes processing the sitemap and review indexed and excluded URL counts. +4. Update the [dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) with redacted conclusions and actual values. + +Completion evidence still needed: consent observations, processed sitemap conclusion, +numeric baseline date range, and reviewer/date. + +## Security acceptance + +Owners: Hermes, URL, and jmservera. + +Required actions: + +1. Hermes records a disposition for SEC-01 through SEC-06 in the [security review](security-review.md). +2. Review the implemented SEC-02 no-referrer and frame-local consent model, including its publisher-markup and third-party-storage limitations. +3. Approve, reject, or amend the SEC-03 exact public export field and source-path allowlists. +4. Approve, reject, or require additional controls for the SEC-05 defense-in-depth accepted-risk recommendation; no acceptance is currently recorded. +5. URL reviews protected workflow and secret scope after the real Podcaster environment change. +6. jmservera records the production-owner conclusion after external evidence is linked. + +Completion evidence: dated sign-off rows with finding-level dispositions and linked test, +workflow, or production observations. + +## Accessibility acceptance + +Owners: Fry and a named accessibility reviewer. + +Required actions: + +1. Identify the tested revision, production URLs, browser, operating system, screen reader, and viewport. +2. Review the retained axe and responsive reports from the final CI revision. +3. Complete keyboard-only navigation for primary navigation, consent, filters, charts, tools, and related links. +4. Complete screen-reader review for headings, landmarks, labels, status changes, chart alternatives, and errors. +5. Record each finding, severity, disposition, reviewer, and date. + +Completion evidence: a retained review record combining automated results with keyboard +and screen-reader conclusions. Automated axe success alone does not close NFR-005. + +## Protected real Podcaster run + +Owners: URL, Hermes, a repository administrator, the Podcaster maintainer, and the +environment approver. + +Current evidence is split: run `30202586031` proves real accepted generation, while run +`30721575540` proves an environment-bound dry run. The `podcaster-release-smoke` +environment has no protection rules, and the real workflow is not environment-bound. + +Required actions: + +1. Confirm downstream idempotency or authorize one exact eligible week and manifest. +2. Define required reviewers and branch policy for a real-generation environment. +3. Bind the real generation job to that environment in a separately reviewed workflow change. +4. Review secret scope without recording secret values. +5. Approve and execute one real run. +6. Retain the approver, week, manifest run ID, article digest, Actions URL, downstream job ID, and final conclusion. + +Completion evidence: one successful real downstream run after environment approval. + +## Visual acceptance + +Owner: Amy or another named visual reviewer. + +Complete issue #622's factual checks and any layout-affecting #626 work before capture. +Then follow the [screenshot capture checklist](screenshots/README.md) against the final +revision with populated content, resolved consent state, desktop and mobile viewports, +light and dark themes, interaction states, and a dated reviewer conclusion. + +Completion evidence: replacement visual matrix with revision metadata and an explicit +accept or reject conclusion. Screenshots alone are not approval. + +## External metadata and feed validation + +Owners: Amy for rendered metadata and jmservera for production access. + +Required actions: + +1. Validate the homepage and one representative article in supported social preview debuggers. +2. Validate representative article and breadcrumb markup with Google Rich Results Test. +3. Validate each relevant page type with Schema.org Validator. +4. Record HTTP status and content type for the site and topic feeds in production. +5. Retain the tested URLs, revision, tool conclusions, reviewer, and date without exposing credentials. + +Completion evidence: dated social preview, structured-data, and production feed +conclusions with retained links or redacted records. + +## Sponsor rollout decision + +Owner: jmservera. + +Record a separate decision for each flag. Do not use one blanket approval. + +| Flag | Decision | Reviewed revision and evidence | Conditions | Date | +| ---- | -------- | ------------------------------ | ---------- | ---- | +| `dynamic_topic_creation` | Pending | Pending | Security disposition and approved canary required | Pending | +| `repo_pages` | Pending | Pending | Stable identity and lifecycle evidence required | Pending | + +Completion evidence: dated approve, reject, or defer decisions identifying the exact +revision, evidence, conditions, and rollback owner for each flag. diff --git a/docs/review/data-observatory-relaunch/security-review.md b/docs/review/data-observatory-relaunch/security-review.md index 60569a5..fba8782 100644 --- a/docs/review/data-observatory-relaunch/security-review.md +++ b/docs/review/data-observatory-relaunch/security-review.md @@ -2,7 +2,7 @@ title: Data Observatory Relaunch Security Review description: Repository security and privacy review of Observatory generation, lifecycle, datasets, embeds, browser tools, analytics, and deployment secrets author: SquadScope Squad -ms.date: 2026-07-30 +ms.date: 2026-08-02 ms.topic: reference keywords: - security review @@ -14,7 +14,7 @@ estimated_reading_time: 10 ## Review status -Repository review is complete as of 2026-07-30. Hermes review and sign-off are pending. +Repository review was reconciled with current controls on 2026-08-02. Hermes review and sign-off are pending. NFR-004 is not accepted, and the relaunch security gate remains open until Hermes records a disposition for every open finding. @@ -54,17 +54,25 @@ introduce arbitrary raw HTML through normal rendering. Residual risk remains because phrase matching cannot identify every semantic injection. New external fields must pass through the same sanitization and boundary path before prompt use. +**SEC-05 recommendation for human decision:** accept the semantic false-negative risk only as a +defense-in-depth residual risk while retaining input sanitization, untrusted-content fencing, closing +prompt constraints, canary leak detection, output and frontmatter validation, prompt lint, and the +red-team corpus. Phrase matching detects known lexical patterns; it cannot reliably identify novel +wording, translation, encoding, or semantic paraphrases with equivalent intent. The retained controls +reduce the chance that one miss reaches publication but do not prove semantic detection. This is an +implementation-supported recommendation, not an accepted risk; Hermes must approve, reject, or +require an additional semantic classifier. + ### Candidate-title abuse Candidate discovery combines repository-controlled topics, weekly tags, and analyzed headings. -Canonical and ignored terms reduce noise, and dynamic creation is disabled. However, -`manage_topic_hubs.py` constructs generated hub frontmatter and prose from the candidate title with -quote escaping rather than `sanitize_text()` or structured YAML serialization for the complete -document. +Canonical and ignored terms reduce noise, and dynamic creation is disabled. `manage_topic_hubs.py` +now bounds candidate titles through `sanitize_text()`, rejects line breaks, boundary markers, HTML, +Markdown syntax, control characters, and known injection phrases, and serializes frontmatter through +structured YAML. `tests/test_topic_hubs.py` verifies that unsafe titles fail before any mutation. -This path is contained while `topic_hubs.dynamic_creation.enabled = false`. It must remain disabled -until the title is sanitized, bounded, tested with control characters and Markdown payloads, and -reviewed by Hermes. Candidate promotion also requires a human diff review of evidence and output. +The implementation condition for this finding is complete. Dynamic creation remains disabled until +Hermes verifies the control and a human reviews the evidence and exact output for the approved canary. ### Lifecycle evidence and deletion @@ -77,6 +85,11 @@ The remaining risk is operator error in a lifecycle override. Review must pair t evidence, ledger diff, aliases, generated page, and any expiry removal. Hermes must review deletion evidence policy before NFR-004 acceptance. +`tests/test_observatory_repos.py` now exercises rename aliases, archive evidence, confirmed deletion, +three-year retention, expiry removal, absence that fails closed, and stable-ID migration. These +fixtures prove implementation behavior, not that the production corpus contains stable IDs or a +reviewed lifecycle transition. + ### Public dataset exposure Observatory JSON and CSV outputs are intentionally public. They contain public repository metadata, @@ -84,21 +97,41 @@ weekly observations, topics, derived metrics, and provenance. They must not cont private repository data, prompt transcripts, credentials, email addresses, analytics identifiers, or local filesystem paths. -Publication is a data-classification boundary. Bender owns a field-level diff for new exports; -Hermes owns privacy disposition for new fields. Derived output should remain bounded to the minimum -needed by pages and tools. +`scripts/export_observatory_dataset.py` now defines exact production allowlists for the CSV, +top-repository metadata objects, and the metadata document. It also restricts `source_files` to the +eleven expected checked-in paths under `data/raw/` and +`data/archive/recovered-W23-W29/`. Runtime validation rejects added or missing keys, keeps +`metadata.fields` synchronized with the CSV schema, and requires weekly count keys to equal the +exported week list. + +| Classification | Allowed fields | +| -------------- | -------------- | +| Public source identity | `repository`, `url`, `primary_language`, `latest_license`, `top_topics` | +| Public observations | `latest_stars`, `first_observed_stars`, `max_forks_observed`, `seen_in_trending`, `seen_in_new` | +| Derived public metrics | `rank_by_latest_stars`, `first_seen_week`, `last_seen_week`, `weeks_observed`, `observed_star_change` | +| Release metadata | Dataset/version/timestamp/source/selection/license, bounded counts and rankings, exact CSV fields, allowlisted source paths, exposure statement | + +Publication remains a data-classification boundary. Any new CSV, metadata, or nested-object field +requires an intentional allowlist change, an exact-schema test update, and Hermes privacy review. +The executable policy is implementation evidence, not approval. ### Embed privacy and attribution -Embeddable charts are static Claracle iframe endpoints with visible attribution. The provided snippet -does not include a sandbox or `referrerpolicy` attribute. Loading the iframe can disclose the -embedding page through normal request referrer behavior, and the embed page includes the common -analytics partial. Consent state does not automatically cross site origins. +Embeddable charts are static Claracle iframe endpoints with visible attribution. The official +snippet now sets `referrerpolicy="no-referrer"`, so a publisher using it unchanged does not send the +embedding page URL as the iframe request referrer. Publishers control their own markup and can remove +or replace this attribute; Claracle cannot enforce the policy after a snippet is copied. + +Analytics inside the iframe is frame-local, default-off, and enabled only after the visitor explicitly +accepts Claracle analytics in the consent UI rendered inside that frame. Consent collected by the +embedding site is neither inferred nor transferred. Browser third-party-storage restrictions may +prevent the Claracle consent choice from persisting, which can cause the frame to ask again, but +storage failure never enables analytics. The adapter records `chart_embed_view` only after the +frame-local consent callback enables it. -The existing analytics adapter records `chart_embed_view` only when analytics consent is active in -the frame. Hermes must decide whether embedded endpoints should omit analytics entirely or enforce a -referrer policy and a documented consent model. Until disposition, embed privacy acceptance is -pending. +Rendered-snippet assertions, consent-wiring tests, and the Observatory browser analytics test provide +repository-executable evidence for this model. Production network/storage behavior and publisher +modifications remain outside repository control, so Hermes privacy disposition is still pending. ### Browser tool URL and DOM handling @@ -137,11 +170,11 @@ does not prove protected-environment configuration or a downstream Podcaster run | ID | Finding | Severity | Owner | Disposition | | ------ | ------------------------------------------------------------------------------------------ | ------------- | --------------------- | ----------------------------------------------------------------------------------------------------------- | -| SEC-01 | Dynamic hub candidate titles bypass the standard text sanitizer | High | Farnsworth and Hermes | Open, rollout-blocking; keep dynamic creation off, sanitize and add adversarial tests before review | -| SEC-02 | Embed snippets omit an explicit referrer policy and cross-origin consent does not transfer | Medium | Amy and Hermes | Open; decide no-analytics embed or explicit privacy policy before acceptance | -| SEC-03 | Public export fields need a documented allowlist to prevent future accidental expansion | Medium | Bender and Hermes | Open; review current schema and add a field-level publication policy | -| SEC-04 | Lifecycle deletion depends on manually reviewed overrides | Medium | Bender and Hermes | Controlled by disabled flag, persisted evidence, retention, and diff review; Hermes disposition pending | -| SEC-05 | Phrase-based injection detection has known semantic false-negative risk | Medium | Hermes and Farnsworth | Accepted only as defense in depth after Hermes review; retain fencing, canary, output validation, and tests | +| SEC-01 | Dynamic hub candidate titles require bounded sanitization and structured serialization | High | Farnsworth and Hermes | Implemented; adversarial rejection and structured YAML are tested, Hermes verification pending | +| SEC-02 | Embed snippets require an explicit referrer policy and cross-origin consent does not transfer | Medium | Amy and Hermes | Implemented and tested: official snippet uses no-referrer; frame-local analytics remains default-off until explicit Claracle consent; Hermes disposition pending | +| SEC-03 | Public export fields need a documented allowlist to prevent future accidental expansion | Medium | Bender and Hermes | Implemented and tested: exact CSV, metadata, nested-object, and source-path allowlists; Hermes policy approval pending | +| SEC-04 | Lifecycle deletion depends on manually reviewed overrides | Medium | Bender and Hermes | Rename, archive, deletion, retention, expiry, and fail-closed fixtures pass; production-policy disposition pending | +| SEC-05 | Phrase-based injection detection has known semantic false-negative risk | Medium | Hermes and Farnsworth | Defense-in-depth accepted-risk recommendation is documented and executable controls are retained; no risk acceptance has been granted | | SEC-06 | GA4, GSC, and Podcaster secret behavior is not proven by repository inspection | Medium | URL and jmservera | External verification pending; never record secret values | | SEC-07 | Browser tool uses safe DOM and a restricted outbound URL policy | Informational | Amy | Repository control verified; production and accessibility behavior pending | | SEC-08 | Raw HTML rendering remains disabled | Informational | Amy and Hermes | Repository control verified; Hermes sign-off pending | @@ -149,7 +182,7 @@ does not prove protected-environment configuration or a downstream Podcaster run ## Required evidence before acceptance - Hermes records approval, rejection, or accepted-risk rationale for SEC-01 through SEC-06 -- Candidate-title sanitizer and adversarial tests pass before dynamic topic creation is enabled +- Hermes verifies the implemented candidate-title sanitizer and adversarial rejection before dynamic topic creation is enabled - Embed privacy behavior has a documented and tested disposition - Public dataset schema receives a field-level privacy review - Lifecycle fixtures demonstrate rename, archive, confirmed deletion, retention, and expiry diff --git a/docs/review/data-observatory-relaunch/status-of-record.md b/docs/review/data-observatory-relaunch/status-of-record.md new file mode 100644 index 0000000..c2cb650 --- /dev/null +++ b/docs/review/data-observatory-relaunch/status-of-record.md @@ -0,0 +1,114 @@ +--- +title: Data Observatory Relaunch Status of Record +description: Single reconciled view of delivered versus pending relaunch work across the three remediation plans, the PRD, and the BRD +author: SquadScope Squad +ms.date: 2026-08-02 +ms.topic: reference +keywords: + - status of record + - data observatory + - relaunch readiness + - reconciliation +estimated_reading_time: 7 +--- + +## Purpose + +This document is the single reconciled view of the Claracle Data Observatory +relaunch. It supersedes the fragmented checkbox state across the three remediation +plans by mapping each workstream to its delivered or pending status with evidence. +It complements the [acceptance evidence index](README.md), which owns the external +gate matrix and the acceptance decision. + +- Epic: [#594](https://github.com/jmservera/SquadScope/issues/594) +- PRD: [claracle-data-observatory-relaunch.md](../../prds/claracle-data-observatory-relaunch.md) +- BRD: [claracle-data-observatory-relaunch-brd.md](../../brds/claracle-data-observatory-relaunch-brd.md) + +Reconciled on 2026-08-02. Release acceptance remains **pending** per the +[acceptance decision](README.md#acceptance-decision); both rollout flags stay disabled. + +## Source plans + +| Plan | Scope | Reconciled state | +| --------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------- | ----------------------------- | +| [2026-07-29 remediation](../../../.copilot-tracking/plans/2026-07-29/claracle-data-observatory-relaunch-remediation-plan.instructions.md) | Core relaunch build (Phases 1-10) | Phases 1-6, 8 done; 7, 9, 10 open | +| [2026-07-30 review remediation](../../../.copilot-tracking/plans/2026-07-30/claracle-data-observatory-relaunch-review-remediation-plan.instructions.md) | Post-review corrections (Phases 1-8) | Phases 1-5 done; 6-8 open | +| [2026-07-31 deploy hydration](../../../.copilot-tracking/plans/2026-07-31/claracle-deploy-hydration-remediation-plan.instructions.md) | Deploy/hydration incident (Phases 1-5) | Phases 1-4 done; 5.3 open | + +## Delivered since the plans were written + +| Workstream | Evidence (merged) | Requirement trace | +| --------------------------------------- | ---------------------------------------------------------------- | -------------------------- | +| Deploy stopped hydrating empty `content/data` | `#627` → `#628` (interim Path B) | NFR-011 | +| Publish commit stages only existing generated paths | `#634` | NFR-011, provenance | +| Safe hydration guard generalized; `content/data` deploy restored | `#633` → `#637` | NFR-011, NFR-012, R-08 | +| Embed `source_page` validation before build (CI guard) | `scripts/check_embed_sources.py`, `tests/test_embed_sources.py`, wired in `ci.yml` (`#641`) | NFR-011, NFR-012 | +| Embeddable-charts demo linked from the data landing page | `#642` | FR-052 | +| Podcaster smoke: API key passed to reusable workflow | `#636` | NFR-002 | +| Podcaster smoke: tooling checked out from default branch | `#639` → `#643` | NFR-002, R-04 | +| Podcaster smoke: hydrate source manifest referenced by promotion record | `#639` → `#645` | NFR-002, R-04 | +| Restore preserves the published weekly transaction | `#640` → `#646` | NFR-002 (restore integrity) | +| Live deploy failure (run 30718600607) | `#644` CLOSED — root cause was a dangling `source_manifest.path` (`data/candidates/2026-W31/30669054860/publish-manifest.json`) breaking the Podcaster smoke gate; resolved by `#645`/`#646`; deploy-site green since 2026-08-01 | — | + +## Requirement status summary + +| Area | Status | Notes | +| --------------------------------------- | --------- | -------------------------------------------------------------------------------- | +| Weekly topic through-line | Done | 2026-07-29 Phase 2 | +| Repository lifecycle (identity, retention) | Done | 2026-07-29 Phase 3 | +| Atomic publish transaction | Done | 2026-07-29 Phase 4; restore integrity hardened by `#640`/`#646` | +| SEO / rendered link contracts | Done | 2026-07-29 Phase 5 | +| Consent-gated analytics API | Done | 2026-07-29 Phase 6 | +| Deploy / hydration parity + CI guard | Done | 2026-07-31 Phases 1-4 (`#628`/`#634`/`#637`/`#641`) | +| Podcaster release smoke (dry-run gate) | Done | Blocking post-deploy gate green (`#636`/`#639`/`#643`/`#645`) | +| FR-041 internal link checking | Partial | Satisfied at test level (`tests/test_internal_link_checker.py`); no standalone CI link tool | +| Hugo/Pagefind timing separation | Done | CI records separate report-only Hugo and Pagefind durations; Q-01 workload attribution remains pending | +| Security sign-off (NFR-004) | Pending | SEC-02 and SEC-03 repository controls are implemented and tested; SEC-05 has a defense-in-depth recommendation; Hermes finding dispositions remain required | +| Accessibility evidence (NFR-005) | Pending | Amy/Fry; 2026-07-30 Step 7.2 | +| Real Podcaster downstream run (NFR-002 / R-04) | Pending | Protected environment; 2026-07-30 Step 6.3 | +| Refreshed visual acceptance | Pending | Screenshot capture checklist | +| GA4 + GSC connection + baseline (FR-035) | Partial | Connection complete: GA4 stream confirmed, GSC verified, root sitemap submitted, and products linked; numeric baseline transcription and consent evidence remain pending | +| External metadata and feed validation | Pending | Social previews, Rich Results, Schema.org, and production feed responses require retained conclusions | +| Incremental generation cost (Q-01 / NFR-009) | Pending | Design spike required | +| `repo_pages` rollout (FR-020-022) | Deferred | Flag disabled; needs sponsor approval + own plan | +| `dynamic_topic_creation` rollout (FR-004) | Deferred | Flag disabled; needs sponsor approval + own plan | +| Sponsor rollout approval | Pending | jmservera; see [launch-gate register](#launch-gate-register) | + +## Epic issue dispositions + +| Issue | Title | Disposition | +| ------------------------------------------------- | ------------------------------------------- | ----------------------------------------------------------------- | +| [#594](https://github.com/jmservera/SquadScope/issues/594) | Epic: Claracle Data Observatory Relaunch | Open — tracks overall relaunch | +| [#644](https://github.com/jmservera/SquadScope/issues/644) | Deploy Hugo site failed (run 30718600607) | CLOSED — resolved by `#645`/`#646` | +| [#626](https://github.com/jmservera/SquadScope/issues/626) | Lighthouse / performance quality-gate follow-ups | Independent hardening; keep thresholds unchanged | +| [#622](https://github.com/jmservera/SquadScope/issues/622) | Post-review UX polish | Non-blocking polish; resolve factual questions before final visual recapture | +| [#599](https://github.com/jmservera/SquadScope/issues/599) | Connect GA4 + Google Search Console (FR-035) | Closed; owner confirmed GA4, GSC verification, sitemap submission, and product link on 2026-08-02 | + +## Launch-gate register + +Each gate lists its owner, blocking dependency, and the evidence path that closes it. +The external acceptance matrix in the [acceptance evidence index](README.md#external-acceptance-matrix) +holds the platform-level rows; this register adds ownership and sequencing. + +| Gate | Owner | Dependency | Evidence path | +| ------------------------------------------ | ----------- | -------------------------------------------- | ------------------------------------------------------------------- | +| GA4 + GSC dated baseline and consent evidence (FR-035/NFR-007/008) | jmservera | Transcribe supplied export and retain production consent observations | [Dated baseline](../../growth/ga4-gsc-baseline-2026-07-29.md) | +| Security sign-off (NFR-004) | Hermes | Security review disposition | [Security review](security-review.md) | +| Accessibility evidence (NFR-005) | Amy / Fry | Production browser and assistive technology | [Owner action register](owner-action-register.md#accessibility-acceptance) | +| Real Podcaster downstream run (NFR-002 / R-04) | URL | Protected environment policy and maintainer authorization | [Owner action register](owner-action-register.md#protected-real-podcaster-run) | +| Refreshed visual acceptance | Amy | Populated content render | [Screenshot capture checklist](screenshots/README.md) | +| External metadata and feed validation | Amy / jmservera | External debugger and production access | [Owner action register](owner-action-register.md#external-metadata-and-feed-validation) | +| Incremental generation cost (Q-01 / NFR-009) | URL | Comparable workload variants and budget owner | [Gated rollout and cost plan](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) | +| Lighthouse follow-ups (`#626`) | Amy / Fry | None; independent hardening | `#626` | +| Post-review UX polish (`#622`) | Amy | None; non-blocking | `#622` | +| Sponsor rollout approval | jmservera | Required gate evidence for each flag | [Owner action register](owner-action-register.md#sponsor-rollout-decision) | + +## Deferred to separate plans + +These are out of scope for the readiness reconciliation and each needs its own plan +(see the [reconciliation planning log](../../../.copilot-tracking/plans/logs/2026-08-02/claracle-relaunch-readiness-reconciliation-log.md#suggested-follow-on-work)): + +- GA4/GSC baseline transcription and production consent evidence (connection and sitemap submission are complete) +- [`repo_pages` rollout](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (requires identity, lifecycle, security, and sponsor approval) +- [`dynamic_topic_creation` rollout](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (requires preview, canary, security, and sponsor approval) +- [Incremental-generation-cost experiment](../../../.copilot-tracking/plans/2026-08-02/claracle-gated-rollout-cost-plan.instructions.md) (Q-01 / NFR-009) diff --git a/hugo.toml b/hugo.toml index 29293dc..4ca8445 100644 --- a/hugo.toml +++ b/hugo.toml @@ -20,7 +20,7 @@ rssLimit = 52 term = ['HTML', 'RSS'] [params] - # jmservera: set to the production GA4 web stream ID (for example, G-XXXXXXXXXX) after creating the Google Analytics property. + # Production injects this through the GA_MEASUREMENT_ID Actions secret; keep the checked-in default empty for forks. ga_measurement_id = "" # jmservera: set via GSC_SITE_VERIFICATION secret. gsc_site_verification = "" diff --git a/layouts/partials/visuals/observatory-chart.html b/layouts/partials/visuals/observatory-chart.html index 12bf3e0..6047460 100644 --- a/layouts/partials/visuals/observatory-chart.html +++ b/layouts/partials/visuals/observatory-chart.html @@ -28,7 +28,7 @@ {{- $sourcePermalink := $page.Permalink -}} {{- $embedSnippet := "" -}} {{- with $embedURL -}} - {{- $embedSnippet = printf `` ($title | htmlEscape) (. | absURL) -}} + {{- $embedSnippet = printf `` ($title | htmlEscape) (. | absURL) -}} {{- end -}}
diff --git a/scripts/export_observatory_dataset.py b/scripts/export_observatory_dataset.py index 9917b38..d8518b4 100644 --- a/scripts/export_observatory_dataset.py +++ b/scripts/export_observatory_dataset.py @@ -24,7 +24,7 @@ DEFAULT_OUTPUT_DIR = PROJECT_ROOT / "static" / "datasets" / DATASET_SLUG PREFERRED_RAW_WEEKS = ("2026-W21", "2026-W22", "2026-W29", "2026-W30", "2026-W31") RECOVERED_WEEKS = ("2026-W23", "2026-W24", "2026-W25", "2026-W26", "2026-W27", "2026-W28") -CSV_COLUMNS = [ +PUBLIC_CSV_FIELDS = ( "rank_by_latest_stars", "repository", "url", @@ -40,7 +40,37 @@ "seen_in_trending", "seen_in_new", "top_topics", -] +) +PUBLIC_METADATA_FIELDS = ( + "dataset", + "version", + "generated_at", + "source", + "selection_rule", + "license", + "row_count", + "weeks", + "weekly_observation_counts", + "exported_repo_observations", + "total_source_repo_observations_screened", + "recurring_repository_count_min_4_weeks", + "repositories_seen_in_trending", + "repositories_seen_in_new", + "top_languages_by_repository_count", + "top_licenses_by_repository_count", + "top_topics_by_repository_mentions", + "top_repositories_by_latest_stars", + "fields", + "source_files", + "public_exposure_review", +) +PUBLIC_TOP_REPOSITORY_FIELDS = ( + "repository", + "latest_stars", + "observed_star_change", + "weeks_observed", + "url", +) AI_KEYWORDS = { "agent", "agents", @@ -136,7 +166,7 @@ def observed_star_change(self) -> int: return self.latest_stars - self.first_stars def row(self, rank: int) -> dict[str, str | int]: - return { + row = { "rank_by_latest_stars": rank, "repository": self.repository, "url": self.url, @@ -153,6 +183,57 @@ def row(self, rank: int) -> dict[str, str | int]: "seen_in_new": "true" if "new_repos" in self.buckets else "false", "top_topics": "|".join(topic for topic, _count in self.topics.most_common(8)), } + validate_exact_keys(row, PUBLIC_CSV_FIELDS, "CSV row") + return row + + +def validate_exact_keys( + payload: dict[str, Any], allowed_fields: tuple[str, ...], label: str +) -> None: + actual = set(payload) + expected = set(allowed_fields) + if actual != expected: + added = sorted(actual - expected) + missing = sorted(expected - actual) + raise ValueError( + f"{label} fields violate the public allowlist: added={added}, missing={missing}" + ) + + +def public_source_path(path: Path, data_root: Path) -> str: + try: + relative_path = path.resolve().relative_to(data_root.resolve()) + except ValueError as error: + raise ValueError(f"Source path is outside the public export policy: {path}") from error + allowed_paths = {Path("raw") / f"{week}.json" for week in PREFERRED_RAW_WEEKS} | { + Path("archive") / "recovered-W23-W29" / week / f"{week}.json" for week in RECOVERED_WEEKS + } + if relative_path not in allowed_paths: + raise ValueError(f"Source path is outside the public export policy: {relative_path}") + return (Path("data") / relative_path).as_posix() + + +def validate_public_summary(summary: dict[str, Any]) -> None: + validate_exact_keys(summary, PUBLIC_METADATA_FIELDS, "dataset metadata") + if summary["fields"] != list(PUBLIC_CSV_FIELDS): + raise ValueError("Metadata fields must exactly match the public CSV allowlist") + if set(summary["weekly_observation_counts"]) != set(summary["weeks"]): + raise ValueError("Weekly observation metadata must exactly match the exported weeks") + for metadata_field in ( + "top_languages_by_repository_count", + "top_licenses_by_repository_count", + "top_topics_by_repository_mentions", + ): + for item in summary[metadata_field]: + if ( + not isinstance(item, (list, tuple)) + or len(item) != 2 + or not isinstance(item[0], str) + or not isinstance(item[1], int) + ): + raise ValueError(f"{metadata_field} entries must be public label/count pairs") + for repository in summary["top_repositories_by_latest_stars"]: + validate_exact_keys(repository, PUBLIC_TOP_REPOSITORY_FIELDS, "top repository metadata") def discover_source_paths(data_root: Path) -> list[Path]: @@ -225,6 +306,7 @@ def build_summary( source_paths: list[Path], observation_counts: dict[str, int], total_source_observations: int, + data_root: Path = PROJECT_ROOT / "data", ) -> dict[str, Any]: language_counts: Counter[str] = Counter() license_counts: Counter[str] = Counter() @@ -249,7 +331,7 @@ def build_summary( str(json.loads(path.read_text(encoding="utf-8")).get("crawled_at") or "") for path in source_paths ) - return { + summary = { "dataset": DATASET_SLUG, "version": DATASET_VERSION, "generated_at": generated_at, @@ -281,19 +363,26 @@ def build_summary( } for repo in repos[:25] ], - "fields": CSV_COLUMNS, - "source_files": [str(path.relative_to(PROJECT_ROOT)) for path in source_paths], + "fields": list(PUBLIC_CSV_FIELDS), + "source_files": [public_source_path(path, data_root) for path in source_paths], "public_exposure_review": ( "PASS: exported fields are repository names, GitHub URLs, public language/license/topic " "metadata, public star/fork counts, and derived weekly aggregates from public GitHub crawl records. " "No tokens, private repo data, cache payloads, user account data, or unpublished crawl calls are included." ), } + validate_public_summary(summary) + return summary def write_csv(path: Path, repos: list[RepoAggregate]) -> None: with path.open("w", encoding="utf-8", newline="") as output: - writer = csv.DictWriter(output, fieldnames=CSV_COLUMNS, lineterminator="\n") + writer = csv.DictWriter( + output, + fieldnames=PUBLIC_CSV_FIELDS, + extrasaction="raise", + lineterminator="\n", + ) writer.writeheader() for rank, repo in enumerate(repos, start=1): writer.writerow(repo.row(rank)) @@ -364,7 +453,13 @@ def export_dataset( source_paths = discover_source_paths(data_root) aggregates, observation_counts, total_source_observations = load_aggregates(source_paths) repos = sorted_aggregates(aggregates) - summary = build_summary(repos, source_paths, observation_counts, total_source_observations) + summary = build_summary( + repos, + source_paths, + observation_counts, + total_source_observations, + data_root, + ) output_dir.mkdir(parents=True, exist_ok=True) write_csv(output_dir / "top-github-projects.csv", repos) @@ -379,7 +474,9 @@ def export_dataset( def check_dataset(output_dir: Path, data_root: Path) -> list[Path]: """Return generated dataset files whose checked-in bytes are stale.""" - with tempfile.TemporaryDirectory() as temporary_directory: + workspace_root = PROJECT_ROOT / ".test-workspaces" + workspace_root.mkdir(exist_ok=True) + with tempfile.TemporaryDirectory(dir=workspace_root) as temporary_directory: expected_dir = Path(temporary_directory) export_dataset(expected_dir, data_root) expected_paths = sorted(path for path in expected_dir.rglob("*") if path.is_file()) diff --git a/tests/test_export_observatory_dataset.py b/tests/test_export_observatory_dataset.py index 1e285af..3760cb0 100644 --- a/tests/test_export_observatory_dataset.py +++ b/tests/test_export_observatory_dataset.py @@ -13,6 +13,91 @@ class ExportObservatoryDatasetTests(unittest.TestCase): + def test_public_export_allowlists_are_exact_and_synchronized(self) -> None: + self.assertEqual( + export_observatory_dataset.PUBLIC_CSV_FIELDS, + ( + "rank_by_latest_stars", + "repository", + "url", + "primary_language", + "latest_license", + "first_seen_week", + "last_seen_week", + "weeks_observed", + "latest_stars", + "first_observed_stars", + "observed_star_change", + "max_forks_observed", + "seen_in_trending", + "seen_in_new", + "top_topics", + ), + ) + self.assertEqual( + export_observatory_dataset.PUBLIC_TOP_REPOSITORY_FIELDS, + ( + "repository", + "latest_stars", + "observed_star_change", + "weeks_observed", + "url", + ), + ) + self.assertEqual( + set(export_observatory_dataset.PUBLIC_METADATA_FIELDS), + { + "dataset", + "version", + "generated_at", + "source", + "selection_rule", + "license", + "row_count", + "weeks", + "weekly_observation_counts", + "exported_repo_observations", + "total_source_repo_observations_screened", + "recurring_repository_count_min_4_weeks", + "repositories_seen_in_trending", + "repositories_seen_in_new", + "top_languages_by_repository_count", + "top_licenses_by_repository_count", + "top_topics_by_repository_mentions", + "top_repositories_by_latest_stars", + "fields", + "source_files", + "public_exposure_review", + }, + ) + + def test_public_export_rejects_unlisted_fields_and_source_paths(self) -> None: + valid_row = export_observatory_dataset.RepoAggregate(repository="owner/repo").row(1) + with self.assertRaisesRegex(ValueError, "public allowlist"): + export_observatory_dataset.validate_exact_keys( + {**valid_row, "private_note": "not public"}, + export_observatory_dataset.PUBLIC_CSV_FIELDS, + "CSV row", + ) + + with self.assertRaisesRegex(ValueError, "outside the public export policy"): + export_observatory_dataset.public_source_path( + export_observatory_dataset.PROJECT_ROOT / "data/private/repos.json", + export_observatory_dataset.PROJECT_ROOT / "data", + ) + + summary = export_observatory_dataset.export_dataset( + output_dir=WORKSPACE_ROOT / "allowlist-validation" + ) + self.addCleanup( + lambda: shutil.rmtree(WORKSPACE_ROOT / "allowlist-validation", ignore_errors=True) + ) + summary["top_languages_by_repository_count"] = [ + {"language": "Python", "count": 1, "private_note": "not public"} + ] + with self.assertRaisesRegex(ValueError, "public label/count pairs"): + export_observatory_dataset.validate_public_summary(summary) + def test_check_reports_stale_external_output_path(self) -> None: external_path = Path("/tmp/claracle-dataset/dataset-metadata.json") stderr = io.StringIO() @@ -54,13 +139,32 @@ def test_export_dataset_writes_public_mit_dataset(self) -> None: with csv_path.open(encoding="utf-8", newline="") as handle: rows = list(csv.DictReader(handle)) self.assertEqual(len(rows), summary["row_count"]) + self.assertEqual(tuple(rows[0]), export_observatory_dataset.PUBLIC_CSV_FIELDS) self.assertEqual(rows[0]["repository"], "openclaw/openclaw") self.assertEqual(rows[0]["seen_in_trending"], "true") self.assertIn("ai", rows[0]["top_topics"]) metadata = json.loads(metadata_path.read_text(encoding="utf-8")) + self.assertEqual(set(metadata), set(export_observatory_dataset.PUBLIC_METADATA_FIELDS)) + self.assertEqual(metadata["fields"], list(export_observatory_dataset.PUBLIC_CSV_FIELDS)) + self.assertTrue( + all( + set(repository) == set(export_observatory_dataset.PUBLIC_TOP_REPOSITORY_FIELDS) + for repository in metadata["top_repositories_by_latest_stars"] + ) + ) + self.assertEqual( + set(metadata["weekly_observation_counts"]), + set(metadata["weeks"]), + ) self.assertEqual(metadata["source_files"], summary["source_files"]) self.assertEqual(len(metadata["source_files"]), 11) + self.assertTrue( + all( + source.startswith(("data/raw/", "data/archive/recovered-W23-W29/")) + for source in metadata["source_files"] + ) + ) self.assertIn("MIT License", license_path.read_text(encoding="utf-8")) citation = citation_path.read_text(encoding="utf-8") self.assertIn( diff --git a/tests/test_observatory_embeds.py b/tests/test_observatory_embeds.py index 2ab8b2b..ec035ea 100644 --- a/tests/test_observatory_embeds.py +++ b/tests/test_observatory_embeds.py @@ -42,6 +42,21 @@ def test_copy_button_handles_clipboard_rejections() -> None: assert "Copy failed" in script +def test_embed_layout_keeps_consent_gated_analytics_wiring() -> None: + base_layout = (ROOT / "layouts/embeds/baseof.html").read_text(encoding="utf-8") + analytics = (ROOT / "assets/js/observatory-analytics.js").read_text(encoding="utf-8") + consent = (ROOT / "layouts/partials/cookie-consent.html").read_text(encoding="utf-8") + + assert 'partial "analytics.html"' in base_layout + assert 'resources.Get "js/observatory-analytics.js"' in base_layout + assert 'partial "cookie-consent.html"' in base_layout + assert "let analyticsConsent = false;" in analytics + assert "analyticsConsent = enabled === true;" in analytics + assert "if (!analyticsConsent" in analytics + assert "CookieConsent.acceptedCategory('analytics')" in consent + assert "setObservatoryAnalyticsConsent(false);" in consent + + def test_rendered_embed_contains_backlink_and_chart_data(tmp_path: Path) -> None: if shutil.which("hugo") is None: pytest.skip("Hugo binary is required to render embed fixtures") @@ -66,3 +81,4 @@ def test_rendered_embed_contains_backlink_and_chart_data(tmp_path: Path) -> None assert "observatory-chart__data" in embed_html assert "https://claracle.com/embeds/fastest-growing-ai-repositories-chart/" in demo_html assert "<iframe" in demo_html + assert "referrerpolicy="no-referrer"" in demo_html diff --git a/tests/visual/observatory-analytics.spec.mjs b/tests/visual/observatory-analytics.spec.mjs index b370bf0..03e106c 100644 --- a/tests/visual/observatory-analytics.spec.mjs +++ b/tests/visual/observatory-analytics.spec.mjs @@ -17,20 +17,25 @@ async function interceptGoogleEndpoints(page) { contentType: 'application/javascript', body: ` window.SquadScopeGA4TestStubLoaded = true; - window.gtag = function () { - window.dataLayer = window.dataLayer || []; - window.dataLayer.push(arguments); - if (arguments[0] === 'event') { - var params = new URLSearchParams({ en: arguments[1] }); - Object.keys(arguments[2] || {}).forEach(function (key) { - params.set('ep.' + key, arguments[2][key]); + function sendEvent(args) { + if (args[0] === 'event') { + var params = new URLSearchParams({ en: args[1] }); + Object.keys(args[2] || {}).forEach(function (key) { + params.set('ep.' + key, args[2][key]); }); fetch('https://www.google-analytics.com/g/collect?' + params, { mode: 'no-cors', keepalive: true }); } + } + var queuedEntries = (window.dataLayer || []).slice(); + window.gtag = function () { + window.dataLayer = window.dataLayer || []; + window.dataLayer.push(arguments); + sendEvent(arguments); }; + queuedEntries.forEach(sendEvent); `, }); }); @@ -182,15 +187,67 @@ test('tool interactions use real handlers and bounded fields', async ({ page }, expect(JSON.stringify(events)).not.toContain('token=secret'); }); -test('standalone chart view fires only after UI acceptance', async ({ page }, testInfo) => { +test('standalone frame uses only its own explicit analytics consent', async ({ page }, testInfo) => { desktopOnly(testInfo); - await interceptGoogleEndpoints(page); - await page.goto('/embeds/fastest-growing-ai-repositories-chart/'); - await waitForConsentUi(page); + const requests = await interceptGoogleEndpoints(page); + const embedUrl = new URL( + '/embeds/fastest-growing-ai-repositories-chart/', + testInfo.project.use.baseURL, + ); + const publisherUrl = new URL('/charts/embeddable-rankings/', embedUrl); + publisherUrl.hostname = embedUrl.hostname === 'localhost' ? '127.0.0.1' : 'localhost'; + expect(publisherUrl.origin).not.toBe(embedUrl.origin); - expect(await customEvents(page)).toEqual([]); + await page.goto(publisherUrl.href); await acceptAnalytics(page); - expect(await customEvents(page)).toEqual([ + const parentRequestCount = requests.length; + const parentAnalyticsCookies = await analyticsCookies(page); + expect(requests.filter(({ kind }) => kind === 'script')).toHaveLength(1); + expect(parentAnalyticsCookies).toEqual([]); + + await page.locator('body').evaluate((body, src) => { + const iframe = document.createElement('iframe'); + iframe.title = 'Cross-origin Claracle chart'; + iframe.src = src; + iframe.referrerPolicy = 'no-referrer'; + body.appendChild(iframe); + }, embedUrl.href); + + await expect + .poll(() => + page + .frames() + .some((candidate) => + candidate.url().includes('/embeds/fastest-growing-ai-repositories-chart/'), + ), + ) + .toBe(true); + const frame = page + .frames() + .find((candidate) => + candidate.url().includes('/embeds/fastest-growing-ai-repositories-chart/'), + ); + expect(frame).toBeDefined(); + await waitForConsentUi(frame); + + expect(await customEvents(frame)).toEqual([]); + await expect(frame.locator(`script[src*="gtag/js?id=${TEST_MEASUREMENT_ID}"]`)).toHaveCount(0); + expect(requests.slice(parentRequestCount)).toEqual([]); + expect(await analyticsCookies(page)).toEqual(parentAnalyticsCookies); + + await frame.evaluate(() => window.CookieConsent.acceptCategory('all')); + await expect + .poll(() => frame.evaluate(() => window.CookieConsent.acceptedCategory('analytics'))) + .toBe(true); + await expect + .poll(() => frame.locator(`script[src*="gtag/js?id=${TEST_MEASUREMENT_ID}"]`).count()) + .toBe(1); + await frame.waitForFunction(() => window.SquadScopeGA4TestStubLoaded === true); + await expect + .poll(() => requests.slice(parentRequestCount).filter(({ kind }) => kind === 'collect').length) + .toBe(1); + expect(requests.slice(parentRequestCount).filter(({ kind }) => kind === 'script')).toHaveLength(1); + expect(await customEvents(frame)).toEqual([ { name: 'chart_embed_view', payload: {