From eab2a10cd51047f6355a871fdb7cf4cef2cc85c3 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:57:38 +0200 Subject: [PATCH 01/65] fix(hooks): drop \bux\b from design-toolchain pattern (FR prose FPs) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 3rd tightening pass (series LRN-1005/1007): bare "ux" matched inside French prose (2 logged FPs, both FR — latest "changement ux vu"). \bui\b kept: zero logged FP, one logged true positive, now locked by a must-fire test row. Flip-tested: quiet row fired pre-change. --- hooks/design-toolchain-reminder.sh | 6 +++++- lib/tests/design-toolchain-reminder.test.sh | 2 ++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/hooks/design-toolchain-reminder.sh b/hooks/design-toolchain-reminder.sh index e39b76a..68547eb 100755 --- a/hooks/design-toolchain-reminder.sh +++ b/hooks/design-toolchain-reminder.sh @@ -44,7 +44,11 @@ lc="$(printf '%s' "$prompt" | tr '[:upper:]' '[:lower:]')" # "design system", "redesign", "front-?end design". dashboard -> \bdashboard\b # so a filename like ecc_dashboard.py no longer matches while "admin dashboard" # still does. animation kept (rarely non-UI). -pattern='redesign|refonte|refont|ui/ux|ux/ui|\bui\b|\bux\b|ui kit|design system|design-system|front-?end design|\bnavbar\b|\bsidebar\b|\bmodal\b|\bbouton\b|\bbutton\b|formulaire|\bhero\b|\bheader\b|\bfooter\b|dropdown|tooltip|\bbadge\b|\bchart\b|graphique|accordion|carousel|\bslider\b|landing|\bdashboard\b|homepage|home page|\baccueil\b|\bécran\b|\becran\b|portfolio|maquette|mockup|wireframe|prototype|\bjoli\b|\bjolie\b|\bbeau\b|\bbelle\b|esth[eé]tique|aesthetic|\bvisuel\b|\bvisual\b|embellir|fignol|peaufin|polish|styliser|styling|stylesheet|\bskin\b|charte graphique|\bbrand\b|branding|\blogo\b|favicon|ic[oô]ne|\bicon\b|\bcss\b|tailwind|shadcn|couleur|gradient|d[eé]grad[eé]|\bombre\b|spacing|espacement|\bmarge\b|\bpadding\b|\bmargin\b|\bradius\b|arrondi|\bhover\b|dark mode|light mode|typograph|\bfont\b|\bfonts\b|font pairing|\bpolice\b|animation|\bmotion\b|micro-interaction|keyframe|glassmorph|neumorph|claymorph|skeuomorph|brutalis|bento|minimalis|responsive|figma' +# Tightened 2026-07-30 (3rd pass): dropped \bux\b — bare "ux" matched inside +# French prose ("changement ux vu…"; 2 logged FPs, both FR). \bui\b KEPT +# (zero logged FP, one logged true positive). NB: the log records only the +# FIRST match per fire (head -1), so per-token FP rates aren't derivable. +pattern='redesign|refonte|refont|ui/ux|ux/ui|\bui\b|ui kit|design system|design-system|front-?end design|\bnavbar\b|\bsidebar\b|\bmodal\b|\bbouton\b|\bbutton\b|formulaire|\bhero\b|\bheader\b|\bfooter\b|dropdown|tooltip|\bbadge\b|\bchart\b|graphique|accordion|carousel|\bslider\b|landing|\bdashboard\b|homepage|home page|\baccueil\b|\bécran\b|\becran\b|portfolio|maquette|mockup|wireframe|prototype|\bjoli\b|\bjolie\b|\bbeau\b|\bbelle\b|esth[eé]tique|aesthetic|\bvisuel\b|\bvisual\b|embellir|fignol|peaufin|polish|styliser|styling|stylesheet|\bskin\b|charte graphique|\bbrand\b|branding|\blogo\b|favicon|ic[oô]ne|\bicon\b|\bcss\b|tailwind|shadcn|couleur|gradient|d[eé]grad[eé]|\bombre\b|spacing|espacement|\bmarge\b|\bpadding\b|\bmargin\b|\bradius\b|arrondi|\bhover\b|dark mode|light mode|typograph|\bfont\b|\bfonts\b|font pairing|\bpolice\b|animation|\bmotion\b|micro-interaction|keyframe|glassmorph|neumorph|claymorph|skeuomorph|brutalis|bento|minimalis|responsive|figma' if printf '%s' "$lc" | grep -Eq "$pattern"; then # Counter: log the fire (time, matched token, excerpt) — best-effort, never blocks. diff --git a/lib/tests/design-toolchain-reminder.test.sh b/lib/tests/design-toolchain-reminder.test.sh index 959882f..53977cb 100644 --- a/lib/tests/design-toolchain-reminder.test.sh +++ b/lib/tests/design-toolchain-reminder.test.sh @@ -22,6 +22,7 @@ check D8-dash-file "$(fire 'ecc_dashboard.py')" quiet # --- Harness-generated inputs must be QUIET even with UI tokens --- check D9-tasknotif "$(fire ' x add css header fonts')" quiet check D10-notif-file "$(fire ' design-motion-principles keyframe done')" quiet +check D11-bare-ux "$(fire 'changement ux vu de tes trouvailles')" quiet # --- Real UI signals must still FIRE --- check F1-button "$(fire 'add a button')" fire @@ -33,6 +34,7 @@ check F6-frontdesign "$(fire 'frontend design work')" fire check F7-admin-dash "$(fire 'admin dashboard screen')" fire check F8-animation "$(fire 'add an animation')" fire check F9-designsys "$(fire 'our design system')" fire +check F10-bare-ui "$(fire 'revois l'\''ui du panneau admin')" fire # --- Fire is logged (time + token + excerpt) --- tmp="$(mktemp -d)" From 0f7b565bb042b942759156dd9bdf63a6b0cb1e9e Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:58:06 +0200 Subject: [PATCH 02/65] feat(global): recalibrate instruction layer for Claude 5 family MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Delegation block: model-neutral when-guidance replaces the Opus 4.8 under-delegation counter (LRN-030 trait inverted on Opus 5; Claude Code injects its own anti-delegation prompt there, #80988). Gates carve-out keeps verifier/security/challenge dispatch mandatory. - Drop the 'staff engineer' self-check bar (Opus 5 over-verification trigger per official migration guide); honest-reporting steps stay. - Deviations bullet: finish-whole-task clause (Opus 5 scope-expansion counter), scoped so 'gone wrong → STOP' still wins. - Written-deliverable length rule (Opus 5 writes ~30-40% longer). --- CLAUDE.global.md | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/CLAUDE.global.md b/CLAUDE.global.md index 3aa4c2d..3ab8178 100644 --- a/CLAUDE.global.md +++ b/CLAUDE.global.md @@ -22,6 +22,8 @@ Apply unless repo-specific instructions override. - Document intent, not mechanics. Use project doc style (docstring, JSDoc…). - Explicit, consistent, meaningful names. Straight control flow, no hidden side effects. +- Written deliverables (docs, reports, .md): length matched to what + the task needs — no filler sections, no boilerplate summaries. ## Refactoring - Priority: safety → readability → consistency. @@ -40,11 +42,12 @@ Apply unless repo-specific instructions override. - Confirm before implementing only when real trade-offs exist (multiple valid approaches, breaking change, destructive action) — else proceed. - Minimal changes unless broader refactor requested. State trade-offs. -- Sub-agents keep main context clean — one task per sub-agent. - More compute on hard problems. Task fans out across independent - items (many files, parallel searches, multi-point checks) → delegate - to sub-agents, don't iterate serially. Default to delegation for - multi-file exploration. Counters model tendency to under-delegate. +- Sub-agents: one task per sub-agent, main context stays clean. + Delegate genuinely independent, sizeable tracks (wide multi-file + exploration, parallel audits) — not work doable in a few tool + calls. Skill-mandated gates (fresh verifier/security/challenge) + always dispatch as written. Don't redo delegated work by hand — + failed gates re-dispatch fresh executors instead. - One question upfront if needed — don't interrupt mid-task. *Exception: skill-mandated gates and checkpoints (orchestrator validation gates, approval gates, darwin checkpoints) always fire.* @@ -53,6 +56,8 @@ Apply unless repo-specific instructions override. - Something goes wrong → STOP, re-plan. Never push through. - Deviations: minor or clearly justified → do, explain after. Significant or shaky justification → ask before deviating. + Finish the whole task: blocked on an independent sub-part → do + the rest, state what's missing. Gone WRONG → still STOP, re-plan. - Root causes only. No temp fixes. Never assume — verify paths, APIs, variables before use. @@ -77,7 +82,6 @@ Apply unless repo-specific instructions override. 2. Report what verified, what not. 3. List remaining risks, surviving deviations. 4. Don't mark complete without proof it works. - Bar: "would staff engineer approve?" 5. Correction or notable event → capitalize to right registry (see "Memory registries"). From c3d3f4d4658a388d0827dd8e3452aa57c7ae374d Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:58:29 +0200 Subject: [PATCH 03/65] =?UTF-8?q?feat(agents):=20plan-challenger=20?= =?UTF-8?q?=E2=80=94=20route=20grounded=20doubts=20to=20[MINOR]?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Opus 5 follows conservative-reporting clauses literally; 'a manufactured concern is a failure' risked suppressing real low-confidence findings. In-place reword: ungrounded stays noise, grounded-but-uncertain files as [MINOR] with the uncertainty in WHY:. OUTPUT grammar byte-identical; census row added. --- agents/plan-challenger.md | 5 +++-- lib/tests/plan-challenger.test.sh | 1 + 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/agents/plan-challenger.md b/agents/plan-challenger.md index e5c0fda..d4ef552 100644 --- a/agents/plan-challenger.md +++ b/agents/plan-challenger.md @@ -79,8 +79,9 @@ PROOF: read files, inspected , checked plan §<…> - Report-only. Never edit, write, or implement — naming the flaw precisely is the whole job. -- No invention. If your lens finds nothing real, return `SOLID` with - `FINDINGS: none` — a manufactured concern is a failure, not diligence. +- No invention — ungrounded is noise. Silently dropping a grounded doubt is + equally a failure: file it as `[MINOR]` with the uncertainty stated in + `WHY:`. Nothing real at all → `SOLID` with `FINDINGS: none`. - `PROOF` is MANDATORY. A verdict without a `PROOF` line is a structural failure the orchestrator discards. - Stay in your lens. A finding outside it belongs to another challenger. diff --git a/lib/tests/plan-challenger.test.sh b/lib/tests/plan-challenger.test.sh index fa70adc..9733f4f 100644 --- a/lib/tests/plan-challenger.test.sh +++ b/lib/tests/plan-challenger.test.sh @@ -24,6 +24,7 @@ has "$A" "correctness" has "$A" "robustness" has "$A" "simplicity" has "$A" "Report-only" +has "$A" "grounded doubt" # uncertain findings → [MINOR], not self-censored (Opus 5 literalism) # 2) reusable phase — the mechanism lives here (one canonical include) has "$L" 'subagent_type="plan-challenger"' From 550b39043e18fb71e2ad4fd81d2ef7c89a1cdac6 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 13:02:55 +0200 Subject: [PATCH 04/65] chore(memory): BDR-081 + LRN-139 + journal + CHANGELOG + plan (opus5 tuning) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Capitalizes the Claude-5-family config recalibration: decision record, trait-inversion learning (LRN-030 superseded premise, #80988 injections, no-effort-hold trap), journal line, CHANGELOG Unreleased entries, and the challenged plan (3 blind Opus 5 lenses, synthesis in §5bis). --- .claude/memory/decisions.md | 3 + .claude/memory/journal.md | 3 + .claude/memory/learnings.md | 6 + .claude/tasks/TODO.md | 25 ++ .../2026-07-30-opus5-config-tuning-1238.md | 255 ++++++++++++++++++ CHANGELOG.md | 14 + 6 files changed, 306 insertions(+) create mode 100644 .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md diff --git a/.claude/memory/decisions.md b/.claude/memory/decisions.md index 15c51ba..c4a99fa 100644 --- a/.claude/memory/decisions.md +++ b/.claude/memory/decisions.md @@ -1073,3 +1073,6 @@ Audit (user ask "profile toggles externals both ways?"): ASYMMETRIC. Enable side ### BDR-080 — bug routing inverted: /bugfix primary, /investigate explicit-only [accepted] (2026-07-21) Old routing "Bug → investigate (bugfix if gstack off)" + gstack ON by default → every bug took path bypassing own quality pipeline (gitflow aiguillage, contract, fresh verifier + security gates, doc-sync, `.claude/memory` registries) — /bugfix relegated to near-never fallback. Skill comparison: same core doctrine (root-cause iron law, hypothesis loop, regression test, 3-strike stop, >5-files alert) but incompatible wrappers — investigate monolithic (same context investigates+fixes+verifies, ~1075-line SKILL.md w/ gstack preamble/telemetry/onboarding, capitalizes to `~/.gstack` learnings.jsonl framework never reads at session start); bugfix orchestrator (reflection inline, sonnet bugfixer executor, fresh gates — BDR-066, LRN-083). Composition rejected: skills superpose in context, don't compose — invoking investigate inside bugfix = two full workflows, two completion protocols, two memory systems loaded at once. Decision: CLAUDE.global.md routing line inverted — bugfix primary; investigate ONLY on explicit ask for gstack ecosystem (cross-project learnings, /freeze scope lock, long no-commit investigation). Alternatives rejected: keep investigate primary (bypasses framework), embed investigate inside bugfix (context conflict, dual memory). Known drift noted at write time: Index table rows BDR-074..079 missing (pre-existing, /prune-memory scope). + +### BDR-081 — Config recalibrated for Claude 5 family (Opus 5 dispatch tier) [accepted] (2026-07-30) +Opus 5 (released 2026-07-24) now backs every `model: opus` pin (BDR-076/077) + any `/model opus` session. Research (official migration guide + web + registries): Opus 5 OVER-delegates (inverts LRN-030 Opus 4.8 trait that CLAUDE.global.md:43-47 compensated), self-verifies (explicit verify instructions → over-verification, "removing them reduces wasted tokens with no loss in quality"), literal following (conservative-reporting clauses depress recall; MUST/CRITICAL over-triggers), scope expansion = named regression, written deliverables +30-40%. Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988, server-gated, no opt-out) — prose caps would triple-stack. Shipped: delegation block → model-neutral WHEN-guidance + explicit gates carve-out (verifier/security/challenge still dispatch as written); "staff engineer" self-check bar dropped; finish-whole-task clause folded into Deviations (gone-WRONG→STOP still wins); deliverable-length rule; design hook `\bux\b` dropped (`\bui\b` KEPT — 0 FP, 1 logged TP, lock-tested); plan-challenger grounded-doubt→[MINOR] in-place reword (grammar byte-identical). Plan challenged by 3 blind Opus 5 plan-challengers: correctness CONCERNS(4) / robustness FATAL(5, BLOCKER: all surfaces symlink-deployed LIVE — gates fire post-deployment) / simplicity CONCERNS(4); every fix adopted as prescribed (scratch-validation before live hook write, minimal diffs, ux-only, MINOR-routing). Alternatives rejected: leave as-is (nudge actively counter-productive); hard spawn caps in prose (harness injects one); confidence axis on challenger grammar (consumer unwired); dropping \bui\b (no evidence). NOT touched: verify-secure-loop + fresh gates (harness architecture BDR-049/050, ≠ model self-check prose); Security/Architecture sections (BDR-021); settings effortLevel xhigh (user pref — Opus 5 carry-over trap → LRN-139); superpowers plugin wording (external upstream). Plan+synthesis: .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md. Branch feature/opus5-config-tuning, unmerged (human gate). diff --git a/.claude/memory/journal.md b/.claude/memory/journal.md index 78a43c9..d489fdb 100644 --- a/.claude/memory/journal.md +++ b/.claude/memory/journal.md @@ -427,3 +427,6 @@ rules: ## 2026-07-22 - User: auto-gitignore+delete transient pipeline artifacts in all projects. Investigation reframed the ask — gitignore = WRONG tool (files read from disk during run; would break superpowers SDD `git add` of spec). BDR-065 already rejected gitignore + its DELETE side was doctrine-only (no code, manual chore slipped once — 655e364). User picks (2 recommended): keep committed-during-run + AUTOMATE delete; keep `.claude/tasks/{contracts,plans}` versioned. - Built `lib/gitflow.sh` `_gitflow_purge_transient` at finish (feature/bugfix, pre-merge, best-effort never-abort, opt-out `GITFLOW_PURGE_TRANSIENT=0`) + `purge-transient` CLI verb. Universal via `~/.claude/lib`→repo symlink. gitflow-test T17 a-d (10 checks, `--full-history` recovery), shellcheck clean, make test exit 0. BDR-065 amendment + [[LRN-138]]. feature/gitflow-auto-purge-transient. + +## 2026-07-30 +- User: Opus 5 "needs more freedom" → analyse config + adapt. Research 3-agent (registries / config audit / web) + official migration guide: over-delegation (inverts LRN-030), over-verification, literal following, scope expansion, #80988 injections. Plan challenged 3 blind Opus 5 plan-challengers — robustness FATAL (BLOCKER: symlink-live deployment), all fixes adopted. Shipped: CLAUDE.global.md recalibrated (delegation when-guidance, staff-bar dropped, finish-whole-task, deliverable-length; 308/320), design hook \bux\b dropped flip-tested (22/0), plan-challenger grounded-doubt→[MINOR] (44/0). BDR-081 + LRN-139. feature/opus5-config-tuning, UNMERGED. diff --git a/.claude/memory/learnings.md b/.claude/memory/learnings.md index 2a4d04a..1c27fa5 100644 --- a/.claude/memory/learnings.md +++ b/.claude/memory/learnings.md @@ -1355,3 +1355,9 @@ rules: - **context**: user asked to gitignore transient planning artifacts (`docs/superpowers/{specs,plans}`, `.claude/tasks/{contracts,plans}`) to stop them merging. BDR-065 had already REJECTED gitignore for docs/superpowers on the git-travel ground; the real gap was the DELETE side never being coded (doctrine-only manual chore, slipped once — 655e364). Built `_gitflow_purge_transient`. - **future application**: "don't merge transient X" → ask: does the run read X from disk? does X travel via git (worktree, foreign checkout)? Yes → auto-purge at finish, not gitignore. Scoped commit `-- ` avoids sweeping a dirty index; `git diff --quiet HEAD -- paths` precheck makes `git rm` all-or-nothing safe; keep the purge best-effort so cleanup NEVER blocks a merge. Prove archive-reachability with `git log --full-history` / `git show :path` — plain `git log -- path` prunes the purged add-commit via history simplification (bit me writing T17). - **link**: [[BDR-065]]. + +## LRN-139 — model-trait compensations invert across generations; state WHEN-guidance, not direction (2026-07-30) +- **pattern**: config rules that COMPENSATE a model trait become counter-productive when the next generation inverts the trait. LRN-030 (Opus 4.8 under-delegates → "Default to delegation… counters under-delegation") inverted by Opus 5 (delegates MORE readily, official guide) — the rule pushed the failure the model now has. Same class: explicit verify instructions → over-verification; conservative-reporting clauses → literal recall suppression; MUST/CRITICAL → over-triggering. +- **Opus 5 traps found**: (a) Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988; server-gated, no opt-out, absent from transcripts) — own prose stacks on top blindly; (b) NO model-default effort hold on Opus 5 — persisted effortLevel (xhigh, settings.json) silently carries over, against "start high, sweep low/medium"; run /effort sweep per model; (c) effort does NOT shorten visible output/deliverables — only prose length rules do (+30-40% docs). +- **future application**: at every model-generation bump, grep config for trait-compensating language ("counters model tendency…", "default to X") and re-verify the premise; prefer WHEN-guidance (conditions where X pays) over directional nudges — survives inversions unchanged. +- **link**: [[LRN-030]] [[BDR-081]]. diff --git a/.claude/tasks/TODO.md b/.claude/tasks/TODO.md index cc2c8cb..9e6c2a4 100644 --- a/.claude/tasks/TODO.md +++ b/.claude/tasks/TODO.md @@ -1,5 +1,30 @@ # TODO +## 2026-07-30 — adapt config for Claude 5 family / Opus 5 (feature/opus5-config-tuning) +User: Opus 5 "needs more freedom" → research (official migration guide + +web + registres) confirms: over-delegates (inverts LRN-030 Opus 4.8 trait), +over-verifies if told to verify, literal instruction following, scope +expansion named regression, harness already injects anti-delegation on +Opus 5 (#80988). Plan: .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md +— to be challenged by 3 blind plan-challengers (opus pins → Opus 5), then +executed on feature branch. NO merge (human gate). +Challenged 2026-07-30: correctness CONCERNS(4) · robustness FATAL(5, 1 +BLOCKER: symlink-live deployment) · simplicity CONCERNS(4) — all fixes +adopted as prescribed (plan §5bis, v2 items below). +- [x] W0 branch first (eab2a10 parent); hook regex validated on scratch copy + (bash -n + shellcheck + 5 replays, HOME sandboxed) before live write +- [x] W1 delegation block v2 (when-guidance + gates carve-out + scoped don't-redo) — 0f7b565 +- [x] W2 "staff engineer" bar line deleted — 0f7b565 +- [x] W3 finish-whole-task folded into Deviations (+ gone-WRONG→STOP) — 0f7b565 +- [x] W4 deliverable-length rule — 0f7b565 +- [x] W5 line budget: 308/320 +- [x] W6 hook \bux\b dropped, \bui\b kept + F10 must-fire lock, D11 quiet row + flip-tested (fire before/quiet after) — eab2a10, suite 22/0 +- [x] W7 plan-challenger :82-83 reworded → [MINOR] routing, census row — c3d3f4d, 44/0 +- [x] W8 BDR-081 + LRN-139 + journal + CHANGELOG +- [ ] W9 final gate: make test full suite +- [ ] W10 no gitflow finish (human gate) — merge only on explicit user signal + ## 2026-07-22 — auto-purge transient superpowers artifacts at finish (feature/gitflow-auto-purge-transient) User: transient planning artifacts (`docs/superpowers/{specs,plans}`) leak into develop; BDR-065 "post-merge cleanup" is DOCTRINE ONLY (no code) — manual chore, diff --git a/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md b/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md new file mode 100644 index 0000000..e9c2c7b --- /dev/null +++ b/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md @@ -0,0 +1,255 @@ +# PLAN — Adapt claude-config for the Claude 5 family (Opus 5 focus) + +Date: 2026-07-30 · Branch (planned): feature/opus5-config-tuning (off develop) +KIND: build-plan · Author: main-loop session (Fable 5) + +## 1. Context & evidence + +Opus 5 (`claude-opus-5`, released 2026-07-24) now backs every `model: opus` +agent pin in this repo (analyzer, plan-challenger, seo/geo-analyzer, +plugin-advisor — BDR-076/077) and any session the user switches to via +`/model opus`. Its documented behavioral profile differs from Opus 4.8 in +ways that make parts of this config counterproductive: + +- E1 **Over-delegation**: Opus 5 "delegates to subagents more readily than + prior models" (official prompting guide). Opus 4.8 had the OPPOSITE trait + (LRN-030), and `CLAUDE.global.md:43-47` was written to counter it + ("Counters model tendency to under-delegate"). The premise is inverted. +- E2 **Anti-delegation already injected by the harness**: Claude Code + v2.1.219 server-gates an Opus-5-only prompt section (`heron_brook` + + `subagent_steer_delegation`, GitHub issue #80988) that says "Do not call + the AgentTool unless the user requested it" and "Subagents multiply cost + and time…". Stacking our own hard cap on top would triple-constrain; + keeping a pro-delegation nudge would fight the injection. Model-neutral + when-guidance is the stable middle. +- E3 **Over-verification**: official guidance — "If your prompt contains + explicit verification instructions … remove them: instructions like these + cause over-verification on Claude Opus 5, and removing them reduces wasted + tokens with no loss in quality." Also true of per-prompt "double-check" + phrasing. Targets PROSE told to the model, not harness-level gates. +- E4 **Scope expansion**: named Opus 5 regression ("can expand the scope of + a task, adding steps that weren't requested"). Anthropic ships a literal + counter-block; tested to reduce scope changes "to nearly zero". +- E5 **Literal instruction following** (since 4.7, stronger now): aggressive + MUST/CRITICAL language over-triggers; conservative-reporting instructions + ("only report high-severity") measurably depress recall in review/challenge + harnesses. +- E6 **Longer written deliverables**: files written to disk run ~30-40% + longer; `effort` does NOT control visible/deliverable length — only prose + instructions do. +- E7 **Overconstraint costs reasoning**: Anthropic removed >80% of Claude + Code's system prompt for Claude-5-generation models "with no measurable + loss"; named mechanism = tokens burned resolving conflicting rules. +- E8 **Hook false positive (today)**: `\bux\b` in + `hooks/design-toolchain-reminder.sh:47` fired on French prose ("changement + ux vu" — matches after apostrophe/slash/space); 2nd `ux` FP in the log, + both French. Continues the LRN-1005/1007 false-positive series. No test + row covers `\bui\b`/`\bux\b`. +- E9 **Effort carry-over trap**: Opus 5 has no model-default effort hold in + Claude Code — a persisted `xhigh` (our `settings.json:333`) silently + carries onto Opus 5 sessions, against Anthropic's "start at high, sweep + low/medium" guidance for that model. + +## 2. Design decisions + +- D1 The global instruction layer must be MODEL-NEUTRAL across the Claude 5 + family (sessions run Fable 5 by default; dispatched judgment agents run + Opus 5; executors Sonnet). Fixes therefore express WHEN-guidance and + outcome bars, not directional compensation for one model's trait. +- D2 Harness-level quality gates (fresh blind verifier + security-auditor, + BDR-049/050; plan-challenge, BDR-075) are architecture, not model + self-check prompting. They stay. E3 applies only to prose that tells the + MODEL to verify its own work. +- D3 Per BDR-021, the Security and Architecture-decisions sections of + CLAUDE.global.md stay verbatim (deliberate policy). No softening there. +- D4 Registries are append-only: LRN-030 is not edited; a new LRN records + the trait inversion and points back to it. +- D5 Deterministic backstops (gitflow pre-commit, Gitea protection, + permissions.deny, rtk pinning) are explicitly out of "more freedom" scope + — community reports show Opus 5 working AROUND soft controls, which argues + for keeping hard ones. + +## 3. Work items + +### W1 — CLAUDE.global.md: rewrite the delegation block (:43-47) +Replace the 5-line block (incl. "Default to delegation for multi-file +exploration. Counters model tendency to under-delegate.") with model-neutral +when-guidance, same footprint (≤5 lines): + +``` +- Sub-agents: one task per sub-agent, main context stays clean. + Delegate genuinely independent, sizeable tracks (wide multi-file + exploration, parallel audits) — not work doable in a few tool + calls, and not self-verification (harness gates own that). Brief + precisely, then commit to the delegation — don't redo its work. +``` +Rationale: E1+E2. No hard spawn cap in prose (harness already injects one on +Opus 5; Fable benefits from delegation). + +### W2 — CLAUDE.global.md: reframe "After code changes" (:75-83) +Keep the concrete quality bar; drop the proof-mandate/self-check phrasing +(E3). Replace steps 2-4 with faithful-outcome reporting: + +``` +## After code changes +1. Run tests, lint, build, type-check if available. +2. Report outcomes faithfully: what passed, what wasn't run, + remaining risks, surviving deviations. Completion claims only + for verified work. +3. Correction or notable event → capitalize to right registry. +``` +Net: -2 lines. "Would staff engineer approve?" bar and "Don't mark complete +without proof" are removed as self-check choreography; honest-reporting +line preserves the intent (grounded completion claims) without mandating an +extra verification pass. + +### W3 — CLAUDE.global.md: add scope fence (Workflow section) +Append (adapted from Anthropic's tested block, caveman-compressed, ~5 lines): + +``` +- Scope: deliver what was asked, at the scope intended. Routine + judgment calls → decide alone; materially different readings → + ask. Better approach spotted → say so in one line, still do the + task as asked. Finish the whole task; genuinely blocked → do the + rest, state plainly what's missing. +``` +Rationale: E4. Complements existing "Scope changes to task — no unrelated +edits" (line ~15) without contradicting it. + +### W4 — CLAUDE.global.md: add deliverable-length rule (Code style / Comments area) +~2 lines: + +``` +- Written deliverables (docs, reports, .md): length matched to what + the task needs — no filler sections, no boilerplate summaries. +``` +Rationale: E6. Registries already covered by caveman rule. + +### W5 — Line budget +After W1-W4: expected ~309 lines. Hard check: `wc -l CLAUDE.global.md` ≤ 320 +(session-start.sh warning threshold at :202-213). + +### W6 — hooks/design-toolchain-reminder.sh: drop `\bui\b` and `\bux\b` +- Remove the two 2-char alternatives from the pattern at :47. Keep + `ui/ux|ux/ui|ui kit` and all other tokens. +- Add a dated header comment (3rd tightening pass, 2026-07-30, cites the + two French-prose `ux` FPs; series LRN-1005/1007). +- Trade-off accepted: a bare "améliore l'ux" prompt with no other design + token goes quiet — the CLAUDE.global.md "Design work" section still + routes it (the hook is a belt, self-described soft nudge). +- Update `lib/tests/design-toolchain-reminder.test.sh`: add 2 quiet rows + (the real FP prompt excerpt; a bare "l'ui" French sentence) — flip-tested + per LRN-096. Existing 9 must-fire rows unaffected (none uses ui/ux). + +### W7 — agents/plan-challenger.md: coverage-first reporting line +Add one clause to the findings rules (add-only, no removal): uncertain or +low-severity findings are REPORTED with an explicit confidence + severity +tag rather than self-censored — severity filtering happens in the +orchestrator's synthesis, not in the challenger. Rationale: E5 (literal +Opus 5 + "manufactured concern is a failure" wording risks suppressing real +low-confidence findings). Must not touch: verdict grammar, MANDATORY PROOF +clause, blind-dispatch rules (test-locked in plan-challenger.test.sh). + +### W8 — Memory + docs capitalization (same branch, follows the work) +- decisions.md: new BDR (config adapted for Claude 5 family — scope, + rationale, alternatives incl. "leave config as-is" and "hard spawn caps" + rejected). +- learnings.md: new LRN — Opus 5 behavioral profile (over-delegation + inverts LRN-030's Opus 4.8 trait; over-verification; literal following; + no effort hold on Opus 5 in Claude Code; heron_brook/#80988 injection). +- journal.md: one line. +- CHANGELOG.md: entry under Unreleased. + +### W9 — Gates (before commit) +- `shellcheck hooks/design-toolchain-reminder.sh` clean. +- Manual flip-test of the hook: FP prompt → quiet; "redesign the navbar" → + fires. +- `make test` full suite green (design-toolchain-reminder.test.sh, + plan-challenger.test.sh, model-routing.test.sh untouched-but-must-pass, + curated-config-guard, loops-light…). +- `wc -l CLAUDE.global.md` ≤ 320. + +### W10 — Gitflow +`bash ~/.claude/lib/gitflow.sh start feature opus5-config-tuning` off +develop; atomic commits (hook+test / CLAUDE.global.md / agent / memory+docs); +NO `gitflow finish` — merge only on explicit human signal. + +## 4. Explicitly NOT doing (considered, rejected) + +- N1 Touching lib/verify-secure-loop.md or the fresh-verifier/security + gates: harness architecture (BDR-049/050, D2), verifies SONNET executor + output — not Opus 5 self-check prose. +- N2 Softening the Security / Architecture sections (BDR-021, D3). +- N3 Editing the superpowers plugin's "1% chance → MUST invoke" language: + external upstream code; flagged as residual over-triggering risk in the + new LRN, revisit as its own decision if observed. +- N4 Changing `settings.json` `effortLevel: "xhigh"`: user preference, + optimal for the Fable 5 session default; the Opus 5 carry-over trap (E9) + is documented in the LRN + surfaced to the user for a manual decision. +- N5 De-prescribing seo-analyzer.md / geo-analyzer.md (1528/1106 lines, + heavy MUST density): separate project, backlog note in TODO.md. +- N6 Removing or session-gating the design/ctx7 reminder hooks: soft + nudges, cheap, deliberately built; tightened only (W6). +- N7 Any model pin change: `model: opus` pins now resolve to Opus 5 — + desired outcome, census (model-routing.test.sh) untouched. +- N8 Committing settings.json for any reason (LRN-098/1049 /model-churn + trap): file is currently clean; keep it out of every commit. + +## 5bis. CHALLENGE SYNTHESIS (2026-07-30) — FINAL amendments (v2) + +Verdicts: correctness CONCERNS(4) · robustness FATAL(5, 1 BLOCKER) · +simplicity CONCERNS(4). Every fix below is the challenger's own named FIX, +adopted as written. No re-challenge pass: scope narrowed, no new dependency; +W0 is an execution-time safety procedure, not a new config mechanism. + +- **W0 (NEW — robustness BLOCKER)**: all edited surfaces are symlink-deployed + LIVE (~/.claude/CLAUDE.md, hooks/, agents/ → this repo); edits take effect + machine-wide at save time, before any W9 gate. Mitigations: + (a) `gitflow start` BEFORE any live-file edit; never checkout develop + mid-work; (b) hook regex change validated on a SCRATCH copy first + (bash -n + shellcheck + pattern replay), then written to the live file in + ONE atomic Edit; (c) named reverts: `git show develop: > `; + escape hatch = remove the hook registration block from settings.json. +- **W1 v2** (robustness#3, correctness#2): replacement text carves out the + mandated gates explicitly and scopes "don't redo": + "Skill-mandated gates (fresh verifier/security/challenge) always dispatch + as written. Don't redo delegated work by hand — failed gates re-dispatch + fresh executors instead." +- **W2 v2** (simplicity#2): minimal diff — delete ONLY the line + `Bar: "would staff engineer approve?"`. Steps 1-4 + capitalize step stay. +- **W3 v2** (simplicity#1, robustness#4): no new bullet. Fold the only new + clause into the existing Deviations bullet: "Finish the whole task: + blocked on an independent sub-part → do the rest, state what's missing. + Gone WRONG → still STOP, re-plan." (net +2 lines, no conflict with :53). +- **W4**: unchanged (+2 lines). Budget v2: 304 +1 −1 +2 +2 = 308 ≤ 320. +- **W6 v2** (all lenses): drop `\bux\b` ONLY — keep `\bui\b` (zero evidenced + FP; one logged true positive). Accepted trade-off: the 2026-07-21 "ameliore + le tutoriel…gamifier" ux row (plausible TP) goes quiet; CLAUDE.global.md + design-routing section remains the router. Header comment notes the log + records `head -1` only → per-token FP rate not fully derivable. Tests: + quiet row = synthetic "changement ux vu…" (verified matches pre-change → + flips); must-fire row = "revois l'ui du panneau admin" (locks `\bui\b`; + apostrophe escaped correctly, doubles as JSON-path control per + robustness#7). No log-excerpt rows (vacuous — 100-char truncation). +- **W7 v2** (all lenses): in-place reword of the `:82-83` sentence (NOT + test-locked; plan v1 misstated that) instead of an add-only clause: + "No invention — ungrounded is noise. Silently dropping a grounded doubt is + equally a failure: file it as `[MINOR]` with the uncertainty stated in + `WHY:`. Nothing real at all → `SOLID` with `FINDINGS: none`." + OUTPUT grammar byte-identical; no confidence axis; no consumer change. + Census: add `has "$A" "grounded doubt"` row to plan-challenger.test.sh in + the same commit. +- **W9 v2**: adds the W0 scratch-validation step; rest unchanged. +- **W10 v2**: branch creation moves FIRST in execution order. + +## 5. Constraints for challengers + +- Registries append-only; curation only via /prune-memory. +- Census tests lock behavior: any hook/agent edit must land with its test + update in the same commit; `make test` must stay green. +- CLAUDE.global.md ≤ 320 lines (runtime warning threshold). +- BDR-021: Security + Architecture sections verbatim. +- Gitflow: feature branch off develop, no merge without human signal. +- The global file serves ALL models (Fable sessions, Opus 5 dispatches, + Sonnet executors read skill/agent prompts instead) — no Opus-5-only + wording in CLAUDE.global.md. diff --git a/CHANGELOG.md b/CHANGELOG.md index e7df2b7..e8769b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,20 @@ Format follows [Keep a Changelog](https://keepachangelog.com/). ## [Unreleased] +### Changed +- **Global instruction layer recalibrated for the Claude 5 family (BDR-081)** — + delegation block is now model-neutral when-guidance (the Opus 4.8 + under-delegation counter inverted on Opus 5, which over-delegates and gets + an injected harness cap); "staff engineer" self-check bar dropped (Opus 5 + over-verification trigger); finish-whole-task clause added to Deviations; + written-deliverable length rule added. 308/320 lines. +- **design-toolchain hook** — dropped `\bux\b` (2 French-prose false + positives; 3rd tightening pass, series LRN-1005/1007); `\bui\b` kept and + locked by a must-fire test row. +- **plan-challenger** — grounded-but-uncertain findings now file as `[MINOR]` + with the uncertainty stated, instead of being self-censored (Opus 5 follows + conservative-reporting clauses literally). + ## [1.4.0] — 2026-07-22 ### Added From 7047adfe77592d1d4c2bf114a379b486f03489d1 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 13:41:37 +0200 Subject: [PATCH 05/65] =?UTF-8?q?chore(memory):=20reconcile=20TODO=20?= =?UTF-8?q?=E2=80=94=20opus5=20branch=20merged=20(709cf9b),=20add=20Claude?= =?UTF-8?q?=205=20follow-on=20chantiers=20C1-C4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .claude/tasks/TODO.md | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/.claude/tasks/TODO.md b/.claude/tasks/TODO.md index 9e6c2a4..4de4c90 100644 --- a/.claude/tasks/TODO.md +++ b/.claude/tasks/TODO.md @@ -22,8 +22,26 @@ adopted as prescribed (plan §5bis, v2 items below). flip-tested (fire before/quiet after) — eab2a10, suite 22/0 - [x] W7 plan-challenger :82-83 reworded → [MINOR] routing, census row — c3d3f4d, 44/0 - [x] W8 BDR-081 + LRN-139 + journal + CHANGELOG -- [ ] W9 final gate: make test full suite -- [ ] W10 no gitflow finish (human gate) — merge only on explicit user signal +- [x] W9 final gate: make test full suite — green except known T6c + (darwin-skill residual → chantier 4 below), 2026-07-30 +- [x] W10 merged on explicit user signal — 709cf9b (2026-07-30 13:28), + branch deleted; confirmed post-merge this session + +## 2026-07-30 — Claude 5 follow-on chantiers (user directive, checkpoint between each) +Order fixed, one branch per chantier, no merge without per-chantier signal. +- [ ] C1 dé-prescription seo-analyzer.md + geo-analyzer.md (opus pins → Opus 5): + separate machine-parsed contracts (fix-bundles, ownership matrices, + output formats — verbatim) from process choreography (MUST/MANDATORY on + "how" → when-guidance). Dedicated plan + 3-lens challenge, census tests + same commits, real /seo dogfood before/after (zero format regression). +- [ ] C2 self-contradiction audit CLAUDE.global.md + own skills: list rule + pairs in tension, propose resolution per pair, apply after user OK. + /doctor as assistant, not authority. +- [ ] C3 superpowers: MEASURE first (skill-invocation log over sessions) + whether "1% chance → MUST invoke" over-triggers; if yes, options + + trade-offs (disable plugin / softer house rule / live with) — user decides. +- [ ] C4 hygiene: reinstall darwin-skill (T6c residual from run-reconcile), + make test 100% green. ## 2026-07-22 — auto-purge transient superpowers artifacts at finish (feature/gitflow-auto-purge-transient) User: transient planning artifacts (`docs/superpowers/{specs,plans}`) leak into From 9681b468e14146a8b4cd6555e47df9bf692a37a0 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Sun, 2 Aug 2026 01:10:47 +0200 Subject: [PATCH 06/65] =?UTF-8?q?test(census):=20seo/geo=20agent=E2=87=84d?= =?UTF-8?q?ispatcher=20contract=20locks=20(C1=20P1,=20pre-reword)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 71 locks, flip-proven (7 scratch mutations → 7 FAILs): judge verdict grammar, FIX BUNDLE + READY-TO-APPLY sentinel, signals handoff, ALL STEP headers (interiors included, conf#5), collect report, bundle item fields parsed by L1 appliers, score labels (BDR-010/LRN-011), scoring blocks, trajectory, envelope keys. Locks existing state — reword commits must keep this green. + plan v3 (challenged 3 lenses FATAL/FATAL/CONCERNS + 1 confirmation pass FATAL(9), every BLOCKER closed by a named change, §5bis record) + directive-language inventory annex (analyzer report). --- ...0-seo-geo-deprescription-1402-inventory.md | 174 +++++++++ .../2026-07-30-seo-geo-deprescription-1402.md | 342 ++++++++++++++++++ lib/tests/seo-geo-contract.test.sh | 82 +++++ 3 files changed, 598 insertions(+) create mode 100644 .claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402-inventory.md create mode 100644 .claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402.md create mode 100644 lib/tests/seo-geo-contract.test.sh diff --git a/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402-inventory.md b/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402-inventory.md new file mode 100644 index 0000000..d572ffc --- /dev/null +++ b/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402-inventory.md @@ -0,0 +1,174 @@ +# ANNEX — directive-language inventory (analyzer report, 2026-07-30) + +Produced by a read-only analyzer dispatch over agents/seo-analyzer.md +(1528 l) + agents/geo-analyzer.md (1106 l), cross-referenced against +every consumer. Referenced by the C1 plan (same folder, -1402.md). + +## 0. Token census (raw) + +| Token family | seo-analyzer.md | geo-analyzer.md | +|---|---|---| +| MUST/must | 12 | 6 | +| MANDATORY/mandatory | 8 | 4 | +| NEVER/never | 42 | 33 | +| ALWAYS/always | 6 | 1 | +| CRITICAL/critical | 3 | 1 | +| Do NOT / do not | 24 | 8 | +| verbatim | 6 | 3 | +| STOP | 3 | 4 | +| refuse/REFUSE | 6 | 4 | +| ⚠️ blocks | 0 | 0 | + +## 1. Test locks on these files (complete list — 6 per file) + +model-routing.test.sh:67-68 `model: opus` (both) · :150-157 `MODE: +collect|judge|template` + `COLLECTION COMPLETE` (both) · +seo-data.test.sh:538-540 `fetch.sh crux` / `fetch.sh queries` / +`Performance GSC` (seo) · :542-543 `fetch.sh schema_gen` / +`fetch.sh content_quality` (geo). +NOT locked by any test: READY-TO-APPLY sentinel, envelope headings, +score-block shapes, JUDGE-ERROR strings, batch labels — contracts by +consumer only; a rewrite can break them silently and make test stays +green. Sibling dispatcher locks: model-routing.test.sh:159-166. +Stale line-number comments (no enforcement): lib/url-guard.sh:9, +url-guard.test.sh:20, source-scope.sh:25, seo-data/README.md:196/309, +drift.py:4, linkgraph.py:4 — all already drifted. + +## 2. Format contract (artifact → consumer) — FREEZE SET + +seo-analyzer: signals `.audit/seo-signals-.md` (+clean/load sites +in /seo) · `COLLECTION COMPLETE — RUNID: ` terminal · +`COLLECT REPORT` w/ `STATUS: DONE|BLOCKED` · `SEO JUDGE — VERDICT: +ERROR()` · judge report forwarded verbatim to template · +`SEO SCORING ()` block w/ `COVERAGE SOURCE:`/`COVERAGE LIVE :` ++ 7 axes + `SEO GLOBAL (weighted): XX.X/20` (score.py:26-37 mirrors +weights) · `TRAJECTORY TO 17/20 (code-only)` · `fetch.sh score` JSON +(`axes.{technical,on-page,seo-local,off-page,social,competitive,legal}`, +severities `critique|haute|moyenne|basse`, `status:"na"`) · `FIX PLAN (N +findings total)` + BATCH A…F (tier-mapping tolerant) · `## FIX BUNDLE +(for dispatcher)` + `### AUTO/### GATED/### USER ACTIONS` + item fields +`id: applier: files: concern: current: expected:` · sentinel `READY TO +APPLY — awaiting dispatcher confirmation` (also reused by /harden:366) · +envelope `SEO AGENT RESULT` + `## SECTION FOR SEO.md §2…§6` + `## ENTRIES +FOR SEO.md §0/§8/§9/§10/§11/§15` · `Automatisation possible avec:` per +§11 entry · standalone `.claude/audits/SEO.md` w/ `**Score SEO** : XX.X +/ 20` (client-handover-writer.md:344 labeled grep) + §0-§15 + Historique. +geo-analyzer: same families with GEO names; envelope `GEO AGENT RESULT` ++ `## SECTION FOR SEO.md §7` (7.1-7.6); `**Score GEO** : XX.X / 20` +(handover parses it only inside SEO.md, allow_fallback=no); G1-G7 +batches (G1-G4/G6 AUTO · G5 GATED · G7 USER). Both: STEP NUMBERS are +addressed by dispatchers (seo: 2-5/6-11/12-14; geo: 0-5/6-12/13-15; +also depth-matrix.md:17-19,37) — renumbering re-points dispatch prompts. +Engine interfaces: fetch.sh verbs {crux,queries,inspect,cannibal, +sitemap,rendercheck,linkgraph,score,schema_gen,content_quality} · +url-guard.sh host|url · source-scope.sh findargs|list · resources/*.md. + +## 3-4. Site classification counts + +| | seo | geo | total | +|---|---|---|---| +| A machine-parsed contract | ~52 | ~41 | ~93 (12 test-locked) | +| B safety/policy invariant | ~30 | ~31 | ~61 | +| C process choreography | ~21 | ~12 | ~33 | +| D other/domain-fact | ~20 | ~13 | ~33 | + +### Class C sites — seo-analyzer.md (rewrite targets) +:61 "First action." · :143-148 CMS-detect-before-edit ordering · +:208-210 "keep the two consistent" (runtime cross-file reconcile) · +:508 "run this BEFORE anything else in STEP 5" (ordering; the refusal +rule itself is B) · :550-553 "Record the denominator BEFORE sampling" +(ordering; honesty rule is B) · :602-604 "Sanity-check the grouping +before trusting it" (self-verify) · :606-618 sampling-method essay · +:661-680 C1a 20-line rationale (rule itself is B at :1493-1501) · +:875 per-item method · :970-971 "Run it twice on the same file before +publishing" (exact BDR-081 over-verification class) · :1147 "AUTO items +are a commitment, not a suggestion." · :1149-1157 P0 CMS-plugin-first +mandate · :1159-1162 P0 Bing mandate (dup of geo :777-786) · :1217 "Do +not proceed to STEP 12 until this plan is printed." · :1260-1261 + +:1342-1350 + :1502-1503 landing-page rule ×3 · :1309-1320 bundle +completeness checklist (10 checkboxes self-audit) · :1504 "Preserve +existing valid SEO." · :1522-1523 WebSearch-on-FULL extra-verify · +:1525-1526 "Transparency. Every automated change logged" (VESTIGIAL — +agent applies nothing, pre-BDR-061). + +### Class C sites — geo-analyzer.md +:48 "copy these patterns" · :124 "First action." + :127-139 ask-block +(unreachable when dispatched) · :230 conditional skip · :262-269 + +:873 + :1063-1065 PERMISSIVE default ×3 · :360 ordering · :394 "20-50 +real customer questions (P0)" · :777-786 MANDATORY AI-index submission +(dup of seo :1159-1162) · :811 "Consolidate EVERY finding" · :823 +"Print the plan before STEP 13" · :1102-1103 WebSearch extra-verify · +:1106 "Every automated change logged in §14" (VESTIGIAL; §15 log is +dispatcher's per :959). + +### Class B anchors (keep obligation, dedup emphasis) +CWD/TARGET MISMATCH twins (seo :117-126 ≈ geo :173-181) · url-guard +mandatory (seo :287-291 ≈ geo :273-277) · R2 refuse-to-score (seo +:519-548, geo :548-557; BDR-072) · NAP direction rule (seo :801-812, +geo :1073-1087; LRN-032-zenquality) · COVERAGE mandatory (seo +:1110-1130, geo :725-729; LRN-133) · never-apply/L1 (BDR-061; LRN-105 +named-ban) · C1a build-output ban · no-invented-content/DGCCRF · +"Compute the scores, do not feel them (I7)" (BDR-073) · §14 mandatory +disclosure lines (backlinks BDR-071, security headers I4) · honest +llms.txt framing · cite-sources (LRN-131). + +## 6. Duplication map (sweep ALL twins — LRN-113) + +seo internal: never-apply ×4 (:1227-1234, :1352-1357, :1468-1472, +:1527-1528) · landing-page ×3 (:1260, :1342, :1502) · shared-file +discipline ×2 (:1254, :1486) · bundle self-containment ×2 (:1249, +:1473) · COVERAGE ×4 (:438, :1000, :1096, :1110) · security-headers- +not-scored ×3 (:281, :977, :994) · 30/70 ×3 (:397, :614, :1165) · +sentinel-verbatim ×3 (:1301, :1304, :1397). +geo internal: PERMISSIVE ×3 · never-apply ×4 (:826-832, :842-848, +:1031-1034, :1104-1105) · tier-mapping ×2 (:824, :850) · +content_quality-advisory ×2 (:584, :622) · shared-file ×2 (:858, +:1047) · llms-honest ×2 (:337, :1066) · cite-sources ×2 (:17, :1089). +Cross-agent twins (stay twins — both files dispatch standalone): +CWD block · url-guard block · MODE DETECTION · MODE BOUNDARY · R2 · +COVERAGE · NAP rule · RULES section skeleton · C1a · automation rule · +Bing/AI-index action · CDN/WAF check. +Agent↔dispatcher duplication (stays — dispatch prompt is per-run +context, agent spec serves standalone/no-MODE paths): NAP ×4 total · +shared-file ×7 · security-headers ×5 · domain split · weights 80/20- +75/25 · Historique · never-re-derive (test-locked dispatcher side). + +## 7. Contradictions / ambiguities found + +1. seo :1525-1526 + geo :1106 vestigial "automated change logged" + (agent applies nothing; geo :959 says dispatcher fills §15). +2. Ask-the-user blocks unreachable in dispatched path (seo :64-75, + :88-112; geo :127-139, :153-168); /geo:41 states it outright. +3. Collect boundary wording: agents "STEP 0-5 ONLY" vs /seo "STEP 2-5 + only (context replaces STEP 0-1)" — works by prompt override. +4. geo judge does live work (sameAs curls :477-492, web_search) unlike + pure-judgment seo judge — asymmetric split, by design. +5. /harden imposes its own output contract (HARDEN.md, /100) the agent + spec never acknowledges; keys on "NARROW-SCOPE" in dispatch prompt. +6. "LRN-032" cite is ambiguous in THIS repo (local LRN-032 = different + lesson; the NAP lesson is zenquality's registry) — keep the + "zenquality" qualifier wherever cited. +7. geo :376-377 uncited FAQ-citation-rate claim vs geo :1089-1097 + cite-sources rule (LRN-131 failure class). +8. Score-label parse fragility: client-handover extract_score fallback + greps FIRST X/20 in file — losing the `Score SEO` label would + silently read `TRAJECTORY TO 17/20` as 17.0. (Latent, downstream.) +9. GEO scoring has no deterministic engine (score.py covers SEO axes + only) — BDR-073 binds only half the pair. + +## 8. Binding memory (from the analyzer's read-before) + +IN FORCE: BDR-081 (premise) · LRN-139 (when-guidance shape) · BDR-061 +(bundle+sentinel decision) · BDR-077 (mode split, fail-closed, locks +survive) · BDR-073 (deterministic scoring) · BDR-072 (R2 refuse) · +BDR-071 (off-page ceiling + §14 line) · BDR-010/LRN-011 (labeled +scores gate) · LRN-133 (omission legible) · LRN-131/132/EVAL-025 +(WebSearch ≠ verification) · LRN-105 (named ban stays explicit) · +LRN-080/088 (measure before delete → dogfood) · LRN-113 (sweep whole +surface) · LRN-093 (no vacuous locks; single-line anchors) · +LRN-126/137 (mode split carries data paths) · BLK-017 (Bing deferred). + +## 9. Open questions → dispatcher decisions (see plan §4b) + +Q1 freeze scope · Q2 census extension · Q3 dedup strategy · +Q4 vestigial lines · Q5 /harden //onboard reconciliation. diff --git a/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402.md b/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402.md new file mode 100644 index 0000000..2664fc2 --- /dev/null +++ b/.claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402.md @@ -0,0 +1,342 @@ +# PLAN v2 — De-prescribe seo-analyzer.md + geo-analyzer.md for Opus 5 + +Date: 2026-07-30 · Branch: feature/seo-geo-deprescription (off develop, started) +KIND: build-plan · Author: main-loop session (Fable 5) +Parent decision: BDR-081 N5 (deferred as separate project) · Method: LRN-139 +v2: revised after the 3-lens challenge (§5bis) — every BLOCKER closed by a +named change; one confirmation challenger pass follows before execution. + +## 1. Context & evidence (v2 — sizing corrected per simplicity#1) + +Both agents are opus-pinned (BDR-076) → every judge phase runs Opus 5. +BDR-081 profile applies: literal following, over-verification when told +to verify, conflicting/duplicated rules burn reasoning tokens. These are +the LONGEST agent files in the repo (1528 + 1106 l) with real downstream +parsers — NOT the densest (measured: ~4.5 directive hits/100 l, ranks +20th/22nd; security-auditor is 17/100). What this pass buys, honestly: +(a) removal of self-output-verification demands (the one pattern the +baseline dogfood caught live: the judge reported "run twice, identical +output" — seo:970 firing), (b) removal of vestigial pre-BDR-061 lines +and 2 real contradictions, (c) small same-audience/same-range dedup, +(d) caps→when-guidance on choreography. The verification apparatus +(census + 3-lens challenge + before/after dogfood) is USER-DIRECTED for +this chantier, not derived from the density premise. + +## 2. Contract surface (v2 — split per correctness#5) + +### 2a. Machine-parsed (named non-LLM consumer: test, script, or literal +grep in a dispatcher step) — byte-frozen +- `model: opus`, `MODE: collect|judge|template`, `COLLECTION COMPLETE` + (model-routing.test.sh:67-68,150-157). +- `fetch.sh crux|queries` + `Performance GSC` (seo), `fetch.sh + schema_gen|content_quality` (geo) (seo-data.test.sh:538-543). +- `SEO|GEO JUDGE — VERDICT: ERROR(` — dispatcher ERROR CONTRACT + fail-closes on it (skills/seo:316-318, skills/geo:65-68). +- `## FIX BUNDLE` + sentinel `READY TO APPLY — awaiting dispatcher + confirmation` — apply step keys on it (skills/seo:524, skills/geo:101; + reused by /harden:366). +- `.audit/-signals-.md` names + fail-closed load. +- STEP numbering: dispatchers address ranges literally (seo 2-5/6-11/ + 12-14; geo 0-5/6-12/13-15; depth-matrix:17-19,37). +- `**Score SEO** : XX.X / 20` / `**Score GEO** : XX.X / 20` labels — + client-handover-writer.md:344-345 labeled grep (BDR-010/LRN-011); + losing the SEO label silently falls back to first-X/20-in-file. +- Bundle item fields `id: applier: files: current: expected:` — pasted + verbatim into hotfixer/feater at L1; `applier: bash` run in-loop. +- url-guard call sites: seo-analyzer.md:287-295, geo-analyzer.md:273-280 + (NOT ":257" as v1 said — robustness#5) + sitemap-URL guard seo:573-582. +- `NARROW-SCOPE` keying of the I4 carve-out (seo:981-983) — /harden's + dispatch prompt relies on it. + +### 2b. LLM-convention contracts (no code consumer; the dispatcher LLM +merges by these shapes) — locked in the census, still frozen +`SEO|GEO AGENT RESULT` envelopes · `## SECTION FOR SEO.md §N` · +`## ENTRIES FOR SEO.md` · `SEO|GEO SCORING (` blocks + `COVERAGE +SOURCE`/`COVERAGE LIVE` lines + `GLOBAL (weighted)` · `TRAJECTORY TO +17/20 (code-only)` · `FIX PLAN (` (seo) · batch labels A-F / G1-G7 +(tier recognition tolerant, labels nominal) · `COLLECT REPORT` + +`STATUS: DONE|BLOCKED` · `Automatisation possible avec:` · §0-§15 +report skeleton + Historique. CROSS-AGENT NOTES emit-instruction lives +in /seo's dispatch prompts (dispatcher-side lock only). + +## 3. Class B invariants — obligation kept, single strongest statement; +security ORDERINGS byte-frozen (robustness#5/#7) + +- Guard-first orderings, frozen verbatim: seo:287-291 / geo:273-277 + ("Guard the domain before it reaches a shell… Run the guard FIRST… + never 'clean up' the value and retry") + seo:573-582 URL loop. +- seo:550 "Record the denominator BEFORE sampling" — the ordering IS + the honesty mechanism (a post-hoc denominator is self-serving); + frozen; only surrounding prose may compress. +- NAP direction rule (LRN-032-zenquality — keep the qualifier, the bare + ID is ambiguous in this repo), R2 refuse-to-score (BDR-072), COVERAGE + obligations (LRN-133 — note :436-439 is a DISTINCT index-reach + obligation, not a repeat), §14 mandatory disclosure lines (BDR-071 + backlinks verbatim line, I4 security-headers), never-apply/L1 + (BDR-061; LRN-105 named ban), C1a build-output ban, no-invented- + content/DGCCRF, deterministic scoring (BDR-073), fail-closed judge, + shared-file Edit-not-Write discipline, honest llms.txt framing, + cite-sources (LRN-131). +- External-freshness checks are NOT self-verification (robustness#6): + seo:1522-1523 + geo:1102-1103 verify a DRIFTING WORLD feeding an + AUTO-tier robots.txt edit — kept, reworded as when-guidance ("crawler + lists shift; cross-check before emitting G1 from the dated resource"). + +## 4. Work items v2 + +- P0 SEQUENCING + LIVE-TREE EXPOSURE (robustness#4, conf#2/#3/#4/#9): + agents/ resolves through ~/.claude symlinks to the WORKING TREE — + edits are live between Edit calls, before any commit. Rules: + (1) the FULL baseline completes before the first agent edit — + signals + judge reports + TEMPLATE envelopes + merged SEO.md + + HUMAN-ACTIONS.md (conf#2: without frozen template artifacts the + template-range edits would have no differential and P0 makes one + unobtainable later); + (2) all baseline artifacts copied to the DURABLE, gitignored + `.audit/dogfood-baseline/` in this repo before the first edit + (conf#9: the session scratchpad dies with the session/reboot; + LRN-124: .audit/** is never committed); + (3) freeze window: no /seo //geo //harden //onboard AND no + /client-handover (spawns /seo — conf#3) nor any skill transitively + dispatching either analyzer, in ANY project, until the after-dogfood + verdict; + (4) aborts (conf#4): mid-reword interrupt or after-dogfood failure → + `git checkout HEAD -- agents/seo-analyzer.md agents/geo-analyzer.md` + (in-flight revert, index-safe); `git checkout develop -- agents/…` + is reserved for a WHOLE-BRANCH abandon; after an abort the named + exit is either (a) fix + re-run the after-dogfood, or (b) present + the static evidence (census + git diff review) to the human who may + accept or abandon at the gate — no open-ended reverted state. +- P1 CENSUS (commit 1, test-only, green pre-reword — compatible with + §7's same-commit rule: it locks EXISTING state and changes no agent + file; reword commits carry any census DELTA): DONE in working tree — + lib/tests/seo-geo-contract.test.sh 54/0, shellcheck clean, real + flip-test run: 7 scratch mutations → 7 FAILs (not "by construction" — + robustness#10). File-qualified locks (correctness#4): `FIX PLAN (` + + `applier: bash` + `Score SEO` seo-only; `Score GEO` geo-only. + Incidental locks dropped (CROSS-AGENT NOTE agent-side, bare + `applier:`). Item fields locked both files. v3 (conf#5): EVERY + `## STEP n —` header locked, interiors included (seo 0-14, geo 0-15) + — census now 71/0. Freeze mechanism for the + ~40 A-sites the census does not cover: reviewed `git diff -U0 + agents/*.md` on each reword commit (simplicity#4). +- P2 REWORD seo-analyzer.md (commit 2): + (a) Self-OUTPUT verification, v3 (conf#1/#8 — neither is deleted + outright): :970-971 "run it twice" → when-guidance integrity + guard ("if the findings JSON changed after scoring, re-run and + explain the move" — score.py is deterministic, so a moving + output means mutated findings: anti-score-shopping, BDR-073; + the unconditional double-run the baseline judge burned goes + away, the guard stays); :1217 "Do not proceed until printed" → + when-guidance scoped to the single-shot path ("single-shot runs + print the FIX PLAN before STEP 12 serializes it" — MODE: judge + stops at 11, but /harden //onboard execute the whole file, + conf#1). The completeness checklist :1309-1320 is NOT deleted: + its routing rows (stock-photo→GATED(E), compression→AUTO(bash) + or §11, aggregateRating→AUTO(hotfixer), structural→GATED(D)…) + are unique routing content (robustness#3) — reshape into a plain + mapping table, drop only the checkbox self-audit framing. + (b) DELETE vestigial :1525-1526 (contradicts BDR-061; Q4). + (c) DEDUP under the invariant (correctness#1 + robustness#1): only + VERBATIM same-AUDIENCE (spec rule / bundle-item payload / + phase-local caveat) same-MODE-RANGE (collect 0-5 / judge 6-11 / + template 12-14 / RULES=global) repeats merge. Expected survivors + per family listed at execution in the commit message; honest + net: never-apply 4→3 (RULES pair merges; template-range + statements stay), sentinel-verbatim reminders 3→2, landing-page + 3→2 (payload instance :1260 + one spec statement; :1342 vs + :1502 merge), bundle-self-containment 2→1 (same range). + NOT deduped (v1 was wrong — distinct rules or cross-range): + COVERAGE ×4, 30/70 ×3, security-headers ×3, shared-file + discipline (payload vs spec audiences). + (d) SOFTEN caps/orderings to when-guidance, keeping semantics: + :61, :508 (gate stays before on-page scoring; emphasis drops), + :875, :1147, :1149-1157 CMS-plugin-first folded together with + :143-148 into ONE statement (correctness#3 — two strengths of + one rule otherwise), :1159-1162 Bing (content rule kept, caps + drop; FULL-only → statically verified), essays :606-618 + + :661-680 compressed keeping the rule + LRN citations; :602-604 + kept as a when-guidance failure detector ("families ≈ URLs → + the heuristic broke — say so"), not deleted (robustness#8). + (e) Dispositions completing the C-list (correctness#3): :208-210 → + static pointer ("the CDN/WAF twin check lives in geo STEP 4"); + :1504 KEEP as-is (one-line scope guard). +- P3 REWORD geo-analyzer.md (commit 3), same invariant: + PERMISSIVE ×3: ALL survive (collect/template/RULES ranges; + :873 is the item-level default guarding an unconfirmed AUTO + robots.txt edit — named survivor, robustness#9). never-apply 4→3 + (RULES pair merges). tier-mapping :824/:850 BOTH stay (judge vs + template ranges). content_quality-advisory 2→1 (same range). + shared-file 2× stays (payload vs spec). llms-honest 2× stays + (collect vs RULES). cite-sources 2× stays (:17 guards the header + stats specifically). :1106 vestigial → reworded to the truth + (dispatcher fills the log — matches :959; Q4). :777-786 caps → + plain content rule (FULL-only). :1102-1103 → freshness + when-guidance (kept — §3). :394 quantity softened ("substantial, + real customer questions"). :48 softened. :124-139 ask-block KEPT + (standalone path). Orderings :360/:811/:823 softened. :376-377 + uncited claim → honest framing (no invented source). +- P4 DOGFOOD AFTER (v3 — ordered by decisiveness, conf#7): fresh copy + of zenquality-frozen; phases in this order so a mid-run death still + leaves the decisive evidence (billing class already realised once): + (ii-first) judges fed the FROZEN baseline signals + (.audit/dogfood-baseline/) → judge reports vs frozen baseline judge + reports, ZERO collect variance — the decisive Opus-judge-prose + differential; (iii) templates on those judge reports → envelopes, + compared against the frozen BASELINE envelopes — the template + verdict anchors on ENVELOPES only (SEO.md/HUMAN-ACTIONS.md are + dispatcher-merged by this authoring session, non-attributable — + conf#10); (i-last) fresh collects, same pre-answered context → + (a) shape check of signals/COLLECT REPORT vs baseline, (b) + FIELD-LEVEL diff of the fresh signals vs baseline signals (record + blocks, COVERAGE counts, denominators — a shape-valid file with a + dropped field must be caught, conf#6), and (c) ONE end-to-end seo + judge on the FRESH signals (the domain with the most collect-range + edits) so the reworded collect→judge handoff runs at least once. + Comparison mechanical-first: presence-assertion script (named home: + `.audit/dogfood-baseline/assert-after.sh`, session-reproducible, + never committed — conf#11) + a FRESH reader agent diffing + before/after WITHOUT this plan in context (correctness#7); the + authoring session only arbitrates its report. If the after-run dies: + P0(4) abort + named exit applies; no merge request meanwhile. +- P5 GATES: make test full suite (census + model-routing + seo-data + + no-vacuous-locks) · shellcheck on touched .sh · per-RANGE grep sweep + for every deduped family (asserts the named survivor lines exist in + their ranges — mode-blind ≥1× sweep is insufficient, robustness#1) · + MEASURED deltas recorded (simplicity#7): wc -l + directive-token + census (annex §0 grep set) per file, before/after, into the BDR. + (v1's manual MODE/STEP sweep dropped — the census asserts it, + simplicity#5.) +- P6 CAPITALIZE: BDR (decision, invariant, deltas, alternatives), LRN + (audience×range dedup invariant — reusable), journal, CHANGELOG. + TODO C1 checked. NO merge (human gate). Checkpoint report includes + the DYNAMICALLY-UNVERIFIED list (§6bis). + +## 4b. Dispatcher decisions (v2) + +- Q1 freeze scope: all §2a byte-frozen + §2b frozen via census; the + remaining unlocked A-prose freeze = per-commit git diff review. +- Q2 census: done (P1), flip-proven. +- Q3 dedup: WITHIN-file, same-AUDIENCE, same-MODE-RANGE, verbatim + repeats only. Cross-agent + agent↔dispatcher twins stay. (Mechanism + note correcting robustness#1's premise: every dispatch loads the FULL + agent file; the risk is ATTENTIONAL — a literal-following model told + "run STEP 13-15" deprioritizes guidance scoped to another step's + body — not access. Same fix either way.) +- Q4 vestigial: seo :1525-1526 DELETE; geo :1106 REWORD to + dispatcher-owns-log (correctness#6 resolved). +- Q5 /harden //onboard: out of scope (N6); their dispatch-prompt + contracts are untouched by agent-file rewording; `NARROW-SCOPE` + keying frozen (§2a). + +## 5. Dogfood protocol (v2) + +Baseline (DONE for collect+judge SEO; geo judge in flight at v2 time): +frozen zenquality copy (no .env), inline pipeline (canonical /seo shape +— the nested-CLI attempt died on the CLI monthly spend limit, recorded), +absolute PROJECT ROOT in every dispatch, `/seo local conservative`, +STEP 0 pre-answered, NAP = NAP-KIT.md (user-confirmed 2026-07-10). +Baseline artifacts frozen under the DURABLE `.audit/dogfood-baseline/` +(gitignored, never committed — conf#9): signals ×2, judge reports ×2, +template ENVELOPES ×2, merged SEO.md, HUMAN-ACTIONS.md (conf#2 — the +template phase runs to completion BEFORE the first agent edit). +After-run per P4. LIMITS stated honestly +(robustness#2): conservative never enters STEP 1b/1.5 (no applier parses +an item this run — the item-field contract is census-locked statically); +LOCAL never executes STEP 3-4/6-7 FULL branches (Bing/AI-index emission +text, live checks — the FULL-only conditionals were exercised and +correctly declined in the baseline judge). These stay on the +§6bis unverified list for the human gate; a FULL/aggressive dry-run is +an OPTION the user may order at checkpoint, not part of this plan. + +## 5bis. CHALLENGE SYNTHESIS (2026-07-30) + +Verdicts: correctness FATAL(3) [1 BLOCKER, 2 MAJOR, 4 MINOR] · +robustness FATAL(9) [3 BLOCKER, 6 MAJOR, 2 MINOR] · simplicity +CONCERNS(3) [3 MAJOR, 4 MINOR]. All three lenses returned. Every +BLOCKER closed by a named v2 change: +- correctness#1 (audience-blind dedup) + robustness#1 (mode-blind + dedup) → §4b Q3 invariant + P2(c)/P3 rewritten + P5 per-range sweep. +- robustness#2 (dogfood can't reach riskiest edits) → §5 honest limits + + §6bis unverified list + P2(d)/P3 minimal-diff on FULL-only sites + + static census cover; FULL/aggressive run offered to the human, not + silently added (billing exposure robustness#11). +- robustness#3 (routing table misfiled as self-check) → P2(a) keeps + routing rows verbatim. +Majors adopted: R4 live-tree abort path (P0) · R5 url-guard anchors +corrected + security orderings frozen (§2a/§3) · R6 external-freshness +kept (§3) · R7 :550 frozen (§3) · R8 :602 kept as detector (P2(d)) · +R9 :873 named survivor (P3) · C2 folded into R2's resolution · C3 full +dispositions (P2(d)/(e), P3) · S1 §1 rewritten · S2 controlled +judge-replay (P4) · S3 mechanical presence script (P4). Minors adopted: +C4 file-qualified locks · C5 §2 split · C6 three inconsistencies +resolved (P1 note, Q4, N1 marker) · C7 fresh-reader diff · S4 diff- +review freeze · S5 sweep dropped · S6+R10 census corrected+flip-proven · +S7 measured deltas. Rejected/scoped: S1's apparatus-shrinking (the +apparatus is user-directed); R1's access premise corrected to +attentional (fix adopted unchanged). + +CONFIRMATION PASS (robustness lens, v2 → v3): FATAL(9) — 2 BLOCKER + +7 MAJOR/MINOR, all targeting the v2 amendments as asked. Closed by +name: conf#1 no-MODE single-shot → §6bis + P2(a) :1217 scoped-softened +· conf#2 missing baseline template artifacts → P0(1) full-baseline +precondition · conf#3 /client-handover freeze → P0(3) · conf#4 abort +HEAD-vs-develop + named exit → P0(4) · conf#5 interior STEP locks → +census extended to all headers (71/0) · conf#6 collect→judge seam → +P4(i) field-diff + one end-to-end seo judge on fresh signals · conf#7 +decisiveness order → P4 reordered (ii)→(iii)→(i) · conf#8 :970 +anti-score-shopping → when-guidance reword, not deletion · conf#9 +volatile baseline → durable .audit/dogfood-baseline/ · conf#10 +dispatcher-owned artifacts → envelope-anchored template verdict · +conf#11 script home named. Challenge budget exhausted (1 re-pass max): +residual risk goes to the human gate with this record. + +## 6. Explicitly NOT doing + +- N1 No dispatcher (SKILL.md) edits. +- N2 No scoring-weight, axis, or depth-matrix changes. +- N3 No model-pin changes (BDR-076). +- N4 No weakening of class-B invariants (§3 hardened in v2: security + orderings byte-frozen). +- N5 No new modes, no pipeline reshaping (BDR-077). +- N6 No /harden //onboard contract reconciliation (annex §7.5). +- N7 No collect-boundary wording fix (works by prompt override). +- N8 No cross-agent shared-resource consolidation. +- N9 No deterministic GEO score engine (annex §7.9). +- N10 No FULL/aggressive dogfood in this plan (user option at gate). + +## 6bis. Dynamically-unverified edit surface (for the human gate) + +Sites edited by P2/P3 that no dogfood run executes: FULL-branch content +(seo :1159-1162 Bing emission, geo :777-786 AI-index emission, both +freshness when-guidances), apply-path parsing (STEP 1b/1.5 — item +pasted into appliers; covered statically by census item-field locks + +frozen bundle templates), STEP 6-7 external-presence prose, and the +no-MODE single-shot path (conf#1: /harden and /onboard dispatch the +agents without a MODE line — "all steps in sequence" — so the whole +reworded body drives those runs; every never-apply and ordering +statement that path relies on keeps a surviving instance, and :1217 +is softened-scoped to it, never deleted). Mitigation: minimal diffs +there (caps→plain only), census locks, git-diff review. + +## 4c. Backlog surfaced (not this branch) + +- Score-label fallback fragility in client-handover-writer.md (can read + `TRAJECTORY TO 17/20` as 17.0 if the label vanishes) — annex §7.8. +- Stale lib/ line-number comments pointing at agent lines (annex §1). +- Baseline judge's gate observation: /client-handover 17/20 gate passes + with an open `critique` finding — "open critique = independent + blocker" is worth its own decision. + +## 7. Constraints for challengers + +- Registries append-only; census green throughout; reword commits keep + 54/0 + model-routing + seo-data locks green. +- Agent files symlink-live INCLUDING between Edit calls (P0 abort path). +- §2a byte-identical; §2b frozen; STEP numbering preserved; §3 security + orderings verbatim. +- Dedup only same-audience + same-mode-range verbatim repeats; named + survivors per family in commit messages; P5 per-range sweep. +- The judge phase is Opus 5; collect/template Sonnet — literal + following applies to all (E5 "since 4.7"). +- Baseline artifacts frozen before first edit; after-run design per P4. diff --git a/lib/tests/seo-geo-contract.test.sh b/lib/tests/seo-geo-contract.test.sh new file mode 100644 index 0000000..85f4f63 --- /dev/null +++ b/lib/tests/seo-geo-contract.test.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +# lib/tests/seo-geo-contract.test.sh — census: seo/geo agent ⇄ dispatcher +# machine contract (C1 de-prescription, 2026-07-30). Locks every string a +# consumer parses BEFORE the choreography reword, so the reword commits +# prove contract preservation by keeping this green. Complements +# model-routing.test.sh (which already locks model pins + MODE:* + +# COLLECTION COMPLETE). +set -u +R="$(cd "$(dirname "$0")/../.." && pwd)" +pass=0; fail=0 +ok() { pass=$((pass+1)); } +ko() { fail=$((fail+1)); printf 'FAIL %s\n' "$1"; } +has() { if grep -qF "$2" "$R/$1"; then ok; else ko "$1 missing: $2"; fi; } + +SEO=agents/seo-analyzer.md +GEO=agents/geo-analyzer.md + +# 1) judge verdict grammar — DISPATCHER ERROR CONTRACT (skills/seo STEP 1, +# skills/geo STEP 1B) fail-closes on this exact shape +has "$SEO" 'SEO JUDGE — VERDICT: ERROR(' +has "$GEO" 'GEO JUDGE — VERDICT: ERROR(' +has "skills/seo/SKILL.md" 'SEO JUDGE — VERDICT: ERROR(' +has "skills/geo/SKILL.md" 'GEO JUDGE — VERDICT: ERROR(' + +# 2) fix-bundle section + apply sentinel — parsed by /seo STEP 1b/1.5 and +# /geo STEP 1b/2 before any L1 apply +for f in "$SEO" "$GEO" skills/seo/SKILL.md skills/geo/SKILL.md; do + has "$f" '## FIX BUNDLE' + has "$f" 'READY TO APPLY — awaiting dispatcher confirmation' +done + +# 3) signals handoff — judge loads the collect artifact fail-closed +has "$SEO" '.audit/seo-signals-' +has "$GEO" '.audit/geo-signals-' +has "skills/seo/SKILL.md" '.audit/seo-signals-.md' +has "skills/geo/SKILL.md" '.audit/geo-signals-.md' + +# 4) STEP numbering — dispatchers reference agent step ranges literally: +# /seo: "STEP 2-5" collect · "STEP 6-11" seo judge · "STEP 6-12" geo +# judge · "STEP 12-14" seo template · "STEP 13-15" geo template; +# /geo: "STEP 0-5". Lock EVERY step header on the agent side (interiors +# too — merging/renumbering one silently re-points the dispatch ranges) +# and the ranges on the dispatcher side. +for n in 0 1 2 3 4 5 6 7 8 9 10 11 12 13 14; do has "$SEO" "## STEP $n —"; done +for n in 0 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15; do has "$GEO" "## STEP $n —"; done +has "skills/seo/SKILL.md" 'STEP 2-5' +has "skills/seo/SKILL.md" 'STEP 6-11' +has "skills/seo/SKILL.md" 'STEP 6-12' +has "skills/seo/SKILL.md" 'STEP 12-14' +has "skills/seo/SKILL.md" 'STEP 13-15' +has "skills/geo/SKILL.md" 'STEP 0-5' +has "skills/geo/SKILL.md" '6-12, report scoring' # "STEP\n6-12" line-wraps +has "skills/geo/SKILL.md" 'STEP 13-15' + +# 5) collect-mode report emission (dispatcher waits on it between phases) +has "$SEO" 'COLLECT REPORT' +has "$GEO" 'COLLECT REPORT' + +# 6) bundle-item routing + the item fields the L1 appliers parse +# (the item is pasted verbatim into hotfixer/feater — /seo STEP 1.5) +for f in "$SEO" "$GEO"; do + has "$f" 'applier: hotfixer' + has "$f" 'applier: feater' + has "$f" ' files:' + has "$f" ' current:' + has "$f" ' expected:' +done +has "$SEO" 'applier: bash' + +# 7) cross-agent escalation block — merged into SEO.md §11 by /seo STEP 2. +# The emit instruction lives in /seo's DISPATCH PROMPTS, not in the agent +# specs (geo-analyzer.md never mentions it; seo-analyzer.md only once, +# incidentally — NOT locked, it is prose). Lock the dispatcher side only. +has "skills/seo/SKILL.md" 'CROSS-AGENT NOTES TO' + +# 8) trajectory block — mandatory in envelopes (/geo audit-end deliverables, +# /seo §1 merge) +has "$SEO" 'TRAJECTORY TO 17/20' +has "$GEO" 'TRAJECTORY TO 17/20' + +printf 'seo-geo contract locks: %d pass, %d fail\n' "$pass" "$fail" +[ "$fail" -eq 0 ] From adafa350da0fc5721c5d986c17ef482f304d063b Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Sun, 2 Aug 2026 01:30:02 +0200 Subject: [PATCH 07/65] feat(agents): de-prescribe seo-analyzer for Opus 5 (C1 P2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Choreography → when-guidance under the audience×mode-range invariant (plan §4b Q3). Census 71/0 + model-routing 133/0 + seo-data 221/0 green throughout; contract surface (§2a/2b) byte-identical; 1528→1503 lines. - self-output verification: ':970 run it twice' → deterministic-engine integrity guard (conf#8); ':1217 do not proceed' → single-shot-scoped (conf#1); grouping sanity-check → when-guidance detector (rob#8) - vestigial pre-BDR-061 'Transparency' line deleted; §15 ownership folded into 'Dispatcher verifies' - dedup (inspection-corrected: most annex 'twins' are distinct obligations — kept): only true same-range dups removed ('Handoff to dispatcher' ≈ sentinel note; 'Landing page rule' block ≈ payload instance + RULES line) - completeness checklist reshaped to routing map, rows verbatim (rob#3) - caps softened: 'P0 rule' MUST/ALWAYS → plain content rules (CMS plugin-first folded with STEP 2 twin, corr#3); First-action/ordering emphasis dropped; C1a + sampling essays compressed (rules + LRN citations kept); WebSearch → drifting-externals when-guidance (rob#6) - FROZEN untouched: guard-first orderings :287, denominator-before- sampling :550, R2 refuse, COVERAGE obligations, all STEP headers - deltas: 'P0 rule' 2→0, ALWAYS 1→0, MUST 5→4; NEVER 9→9 (class-B named bans, kept by design LRN-105) --- agents/seo-analyzer.md | 133 +++++++++++++++++------------------------ 1 file changed, 54 insertions(+), 79 deletions(-) diff --git a/agents/seo-analyzer.md b/agents/seo-analyzer.md index 9c09f16..7233632 100644 --- a/agents/seo-analyzer.md +++ b/agents/seo-analyzer.md @@ -58,8 +58,8 @@ STEP 1-2 business/tech context is consumed by all later steps). ## STEP 0 — AUDIT DEPTH -**First action.** If a parent skill (`/seo` dispatcher) passed depth -in $ARGUMENTS, use it. Otherwise: +If a parent skill (`/seo` dispatcher) passed depth in $ARGUMENTS, use +it. Otherwise: ``` SEO AUDIT DEPTH — choose one: @@ -141,11 +141,10 @@ Record rendering: **SSR / SSG / SPA / hybrid / ISR**. ### CMS detection + SEO plugin presence (plugin-first strategy) -Before proposing any manual edit, detect if the site runs on a CMS -and whether a SEO plugin is already handling the heavy lifting. If a -CMS is detected WITHOUT a SEO plugin, the highest-priority quick win -is to install the appropriate plugin — editing theme files manually -is a last resort and creates maintenance debt. +Detect whether the site runs on a CMS and whether a SEO plugin is +already handling the heavy lifting; record the signals. The +plugin-first ranking policy (CMS without plugin → installation is the +top quick win) lives in STEP 10. ```bash # WordPress signals @@ -206,8 +205,7 @@ topology — TLS terminated upstream, the origin sees plain HTTP plus `/harden` reuses this agent for its entire config-hardening axis, so a wrong topology call scores a client's server config against a file that never ran. -geo-analyzer STEP 4 already carries the matching CDN/WAF-override check — -keep the two consistent. +(The same CDN/WAF-override check lives in geo-analyzer STEP 4.) ```bash # Server / hosting @@ -505,7 +503,7 @@ Fetch rendered HTML. Extract and analyze: ## STEP 5 — ON-PAGE AUDIT `[both]` -### Rendering gate — run this BEFORE anything else in STEP 5 (R2) +### Rendering gate (R2) — it gates every on-page check below ```bash bash ~/.claude/lib/seo-data/fetch.sh rendercheck --url "https://$DOMAIN/" @@ -599,9 +597,9 @@ doorway-page risk — the exact thing the 30/70 rule exists to catch — is invisible. Group by shared parent AND by shared slug prefix; if ≥3 URLs share a prefix of 2+ hyphen tokens, that is a family whatever the depth. -Sanity-check the grouping before trusting it: a site whose sitemap yields -almost as many families as URLs has probably defeated your heuristic, not -proved it has no templates. +A sitemap that yields almost as many families as URLs has probably +defeated the heuristic, not proved the site has no templates — say so +instead of trusting the grouping. **Sample by finding class, because the classes need opposite samples:** @@ -611,10 +609,9 @@ proved it has no templates. | **Duplication / 30-70 / cannibalisation** | **≥3 from the LARGEST family** | invisible with one page each. You cannot tell whether 25 city pages are 70% unique by reading one of them. | | Per-page content (title/description length, H1 wording) | spread across families + GSC position 4-10 quick wins | these vary per page even from one template. | -"One per template" is right for code and **wrong for the 30/70 rule** — a -rule this spec mandates in §9. Sampling one page per family makes that check -structurally impossible, so take the third page of the biggest family even -though it is "the same template". +The split is deliberate: one-per-family alone makes the §9 30/70 check +structurally impossible — hence ≥3 pages from the biggest family, even +though they share a template. An un-sampled family is an un-audited family. Name the ones you skipped. @@ -658,16 +655,11 @@ mapfile -t FEXCL < <(bash ~/.claude/lib/source-scope.sh findargs) find . "${FEXCL[@]}" -type f \( -iname "*.jpg" -o -iname "*.jpeg" -o -iname "*.png" -o -iname "*.gif" \) -printf "%s %p\n" 2>/dev/null | sort -rn | head -20 ``` -**Why the guard, and why `find` specifically (C1a).** `grep` and `find` -disagree about this repo and you use both. Claude Code routes `grep` through -ugrep with `--ignore-files`, so it honours `.gitignore` and never descends -into a gitignored `dist/`. `find` honours nothing. Measured on a real Astro -repo: this command returned **92 images, 45 of them under `dist/`** — every -asset twice, source and generated copy, byte-identical. So "top 20 by size" -was ~10 real images dressed as 20, and a batch-C item -(`cwebp -q 80 -o .webp`) could target `dist/og-image.png`, whose -`.webp` the dispatcher's own `npm run build` then erases. The fix lands, -verification passes, nothing survives. +**Why the guard, and why `find` specifically (C1a).** Claude Code routes +`grep` through ugrep with `--ignore-files` (honours `.gitignore`); `find` +honours nothing. Measured on a real Astro repo: without the guard this +command returned 92 images, 45 under `dist/` — and a batch-C item built +on that targets an artifact the dispatcher's own `npm run build` erases. `FEXCL` MUST be consumed as a quoted array. `find . $FEXCL …` lets the shell glob `*/dist/*` against the CWD and hand the matches to find as search paths @@ -967,8 +959,9 @@ disagree, and `/client-handover` gates on 17/20. **N/A is not a zero** and the engine will not let it behave like one. - `status: "error"` → malformed findings. Fix them; never fall back to eyeballing a number. -- Run it twice on the same file before publishing. If the output moved, your - findings moved, and that is the thing to explain. +- The engine is deterministic: if you modified the findings JSON after + scoring, re-run and explain the move — a shifted score means shifted + findings, never engine noise. **Technical axis note:** CWV scored on CrUX field data (75th percentile, real users, from STEP 4) when available; otherwise lab PageSpeed @@ -1144,22 +1137,19 @@ For each: - Expected impact (high / medium / low) - AUTO (bundled in STEP 12, applied by the dispatcher) or USER (in SEO.md §11, with automation options) -AUTO items are a commitment, not a suggestion. +**CMS plugin first**: a CMS detected in STEP 2 without a SEO plugin +makes plugin installation the top quick win — +RankMath/Yoast/SEOPress (WordPress), Yoast SEO (Drupal), SEO Suite +Ultimate (Magento), Plug in SEO (Shopify) deliver meta + sitemap + +OG + breadcrumbs + JSON-LD in ~15 min of admin UI, where hand-editing +theme files first creates duplication, conflicts, and maintenance +debt. See `~/.claude/agents/resources/automation-catalog.md` CMS +plugins section for the exact install path per CMS. -**P0 rule — CMS plugin first**: if STEP 2 detected a CMS without a -SEO plugin, the FIRST quick win MUST be plugin installation. Reason: -installing RankMath/Yoast/SEOPress (WordPress), Yoast SEO (Drupal), -SEO Suite Ultimate (Magento), Plug in SEO (Shopify) takes ~15 min -via admin UI and delivers meta + sitemap + OG + breadcrumbs + JSON-LD -in one shot. Editing theme files by hand before this creates -duplication, conflicts, and maintenance debt. See -`~/.claude/agents/resources/automation-catalog.md` CMS plugins -section for the exact install path per CMS. - -**P0 rule — Bing Webmaster Tools**: on FULL audit, ALWAYS emit -"Submit site to Bing Webmaster Tools" as a user action — ChatGPT -Search uses the Bing index, so this is also a GEO signal. See -automation-catalog.md for IndexNow + Bing. +**Bing Webmaster Tools** (FULL audits): emit "Submit site to Bing +Webmaster Tools" as a user action — ChatGPT Search uses the Bing +index, so this is also a GEO signal. See automation-catalog.md for +IndexNow + Bing. ### Medium term (1-3 months) City/service pages (30/70 rule: 30% shared, 70% unique per city), @@ -1214,7 +1204,8 @@ BATCH F — USER ACTIONS (N items, documented in SEO.md §11 with automation cat ... ``` -Do not proceed to STEP 12 until this plan is printed. +Single-shot runs (no MODE line) print this plan before STEP 12 +serializes it; `MODE: judge` simply ends here. --- @@ -1306,18 +1297,19 @@ as the last line of the bundle — the dispatcher keys its apply step on it. Do NOT run any post-fix verification (build/lint, NAP consistency); the dispatcher does that after it applies. Your job ends at the sentinel. -### Bundle completeness checklist (did every finding reach the bundle?) +### Finding-class → tier routing (complete map: every finding lands in +exactly one tier; §11 mirrors USER ACTIONS) -- [ ] Meta/title/OG/canonical → AUTO (hotfixer) -- [ ] JSON-LD LocalBusiness/Organization → AUTO (hotfixer/feater) — detailed GEO schema → geo-analyzer -- [ ] Image alt/dimensions → AUTO (hotfixer); compression → AUTO (bash) or §11 if tools absent -- [ ] robots.txt / sitemap.xml → AUTO (hotfixer) — AI-bot directives → geo-analyzer -- [ ] .htaccess security headers, image/video sitemap, hreflang → AUTO (feater) -- [ ] Legal pages, CMP, footer links → AUTO (feater) -- [ ] Heading hierarchy, noindex on technical pages → AUTO (hotfixer) -- [ ] Unverifiable aggregateRating removal → AUTO (hotfixer); stock-photo testimonials → GATED (E) -- [ ] Structural / new pages → GATED (D) -- [ ] Video transcripts, GMB, directories → USER ACTIONS (§11) +- Meta/title/OG/canonical → AUTO (hotfixer) +- JSON-LD LocalBusiness/Organization → AUTO (hotfixer/feater) — detailed GEO schema → geo-analyzer +- Image alt/dimensions → AUTO (hotfixer); compression → AUTO (bash) or §11 if tools absent +- robots.txt / sitemap.xml → AUTO (hotfixer) — AI-bot directives → geo-analyzer +- .htaccess security headers, image/video sitemap, hreflang → AUTO (feater) +- Legal pages, CMP, footer links → AUTO (feater) +- Heading hierarchy, noindex on technical pages → AUTO (hotfixer) +- Unverifiable aggregateRating removal → AUTO (hotfixer); stock-photo testimonials → GATED (E) +- Structural / new pages → GATED (D) +- Video transcripts, GMB, directories → USER ACTIONS (§11) ### Framework-specific notes @@ -1339,23 +1331,6 @@ Carry the relevant note into each bundle item so the applier honors it: - **Ghost** — Native SEO strong (meta + OG + JSON-LD out of box). Usually no plugin needed; handle gaps via `default.hbs` edits. - **Wix / Squarespace / Webflow (hosted CMS)** — No theme file access. ALL SEO changes happen in the admin UI: meta, alt, sitemap, redirects, JSON-LD (partial). Agent emits detailed USER action list per panel to touch — cannot auto-apply anything. -### Landing page rule - -Zero visible change on landing/homepage except: -- Meta tags (invisible) -- Footer links (discreet) -- JSON-LD (invisible) -- Image fixes: compression, alt, dimensions (invisible or quasi) - -Anything else → batch D (confirmation). - -### Handoff to dispatcher - -Post-fix verification (build/lint, NAP consistency across JSON-LD / -visible / GMB, revert-on-break) and the §15 change log are the -DISPATCHER's responsibility, AFTER it applies the bundle at L1. You -emitted the bundle terminated by the sentinel — stop here. - --- ## STEP 13 — OUTPUT `[both]` @@ -1519,10 +1494,10 @@ PROCHAINE ETAPE : ### Process - **Every user action lists automation.** Mandatory from `~/.claude/agents/resources/automation-catalog.md`. -- **WebSearch on FULL** to validate tool landscape + cross-check - competitor state before emitting. +- **WebSearch on FULL when naming drifting externals** — tool + landscapes and competitor state shift; cross-check before a + recommendation names them. - **Iterative SEO.md.** Preserve Historique section. -- **Transparency.** Every automated change logged with file, change, - reason. -- **Dispatcher verifies.** Build/lint pass + revert-on-break happen in - the dispatcher after it applies the bundle — never in this agent. +- **Dispatcher verifies.** Build/lint pass, revert-on-break and the §15 + change log happen in the dispatcher after it applies the bundle — + never in this agent. From c7646a9c8acfe3db86d3e7b4c4abd009ec7fa5cd Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Sun, 2 Aug 2026 01:30:12 +0200 Subject: [PATCH 08/65] feat(agents): de-prescribe geo-analyzer for Opus 5 (C1 P3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same invariant as adafa35. Census 71/0 green; contract surface byte-identical; 1106→1107 lines (single-shot scoping line). - MANDATORY/MUST caps on AI-index submission → plain content rule - 'Print the plan before STEP 13' → single-shot-scoped (conf#1); tier-mapping kept in BOTH judge and template ranges (rob#1/#9, PERMISSIVE :873 named survivor kept) - vestigial ':1106 Transparency §14' reworded to truth: change log is dispatcher's SEO.md §15 (folded into 'Dispatcher verifies') - 'copy these patterns' → reporting-shape-to-match; FAQ '20-50' quantity → 'typically dozens'; 'EVERY finding' → outcome bar - dedup after inspection: ZERO merges (PERMISSIVE ×3 cross-range; never-apply ×4 distinct obligations; content_quality pair = spec rule vs emitted-artifact caveat — annex counts corrected) - FROZEN untouched: guard-first :273, NAP direction rule, cite-sources, WebSearch-freshness (already when-shaped, rob#6), all STEP headers - deltas: MANDATORY 1→0, MUST 4→3, NEVER 9→8 --- agents/geo-analyzer.md | 21 +++++++++++---------- 1 file changed, 11 insertions(+), 10 deletions(-) diff --git a/agents/geo-analyzer.md b/agents/geo-analyzer.md index 8c63520..f23e0e7 100644 --- a/agents/geo-analyzer.md +++ b/agents/geo-analyzer.md @@ -45,7 +45,7 @@ This anchors the agent's output so the user can compare audits over time. effort : weight: <1-5> ``` -Worked examples (1 per axis, copy these patterns when reporting): +Worked examples (1 per axis — the reporting shape to match): ``` [HIGH] [ai-crawlers] GPTBot blocked in robots.txt @@ -391,7 +391,7 @@ Emit finding: FAQ PAGE : present at | absent FAQ SCHEMA : FAQPage (collection) | QAPage (single Q) | none Q&A COUNT : | not applicable -RECOMMENDATION : CREATE /faq with 20-50 real customer questions (P0 for GEO) | ADD schema to existing page | OK +RECOMMENDATION : CREATE /faq with real customer questions (typically dozens — high GEO priority) | ADD schema to existing page | OK ``` If absent and site is informational/service/B2B → emit as MEDIUM-term @@ -774,9 +774,8 @@ High-impact, low-effort. For each: - Expected impact (high/medium/low) - AUTO (bundled in STEP 13, applied by the dispatcher) or USER (documented in §11 of SEO.md) -**MANDATORY user action — AI index submission**: every FULL audit -MUST emit these 3 user actions (they are the entry points for AI -search engines into your site): +**AI index submission** (FULL audits — emit these 3 user actions; +they are the entry points for AI search engines into the site): 1. **Bing Webmaster Tools** — submit + verify sitemap. Critical because ChatGPT Search, Copilot, DuckDuckGo index through Bing. @@ -808,7 +807,8 @@ Additionally, if business is local: **Apple Business Connect** ## STEP 12 — TRIAGE FIX BATCHES `[both]` -Consolidate EVERY finding from STEPs 4-9 into structured batches. +Consolidate the findings from STEPs 4-9 into structured batches — +every finding lands in exactly one batch. | Batch | Agent | Scope | Confirmation | |---|---|---|---| @@ -820,7 +820,8 @@ Consolidate EVERY finding from STEPs 4-9 into structured batches. | **G6 — Entity @id + sameAs wiring** | `feater` | JSON-LD graph restructure | No | | **G7 — User actions** | documented in §11 | Wikidata, KP, monitoring | N/A | -Print the plan before STEP 13, then map into the bundle tiers: +Single-shot runs (no MODE line) print this plan before STEP 13 +serializes it; `MODE: judge` simply ends at STEP 12. Tier mapping: G1–G4/G6 → AUTO, G5 → GATED, G7 → USER ACTIONS. **Apply-vs-report is the DISPATCHER's call, not yours.** You ALWAYS emit @@ -1101,6 +1102,6 @@ PROCHAINE ETAPE : `automation-catalog.md`. No exceptions. - **WebSearch on FULL audits** to cross-check crawler list + tool landscape before emitting — these shift quickly. -- **Dispatcher verifies.** Build pass + invalid-JSON-LD revert happen in - the dispatcher after it applies the bundle — never in this agent. -- **Transparency.** Every automated change logged in §14. +- **Dispatcher verifies.** Build pass, invalid-JSON-LD revert and the + applied-change log (SEO.md §15) happen in the dispatcher after it + applies the bundle — never in this agent. From 325962e0806be1c09bfcfa14584f8133fb1f54a0 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Sun, 2 Aug 2026 17:28:07 +0200 Subject: [PATCH 09/65] chore(memory): BDR-082 + LRN-140 + journal + CHANGELOG + TODO C1 done (seo/geo de-prescription) --- .claude/memory/decisions.md | 3 +++ .claude/memory/journal.md | 3 +++ .claude/memory/learnings.md | 7 +++++++ .claude/tasks/TODO.md | 17 ++++++++++++----- CHANGELOG.md | 17 +++++++++++++++++ 5 files changed, 42 insertions(+), 5 deletions(-) diff --git a/.claude/memory/decisions.md b/.claude/memory/decisions.md index c4a99fa..ad77cd2 100644 --- a/.claude/memory/decisions.md +++ b/.claude/memory/decisions.md @@ -1076,3 +1076,6 @@ Old routing "Bug → investigate (bugfix if gstack off)" + gstack ON by default ### BDR-081 — Config recalibrated for Claude 5 family (Opus 5 dispatch tier) [accepted] (2026-07-30) Opus 5 (released 2026-07-24) now backs every `model: opus` pin (BDR-076/077) + any `/model opus` session. Research (official migration guide + web + registries): Opus 5 OVER-delegates (inverts LRN-030 Opus 4.8 trait that CLAUDE.global.md:43-47 compensated), self-verifies (explicit verify instructions → over-verification, "removing them reduces wasted tokens with no loss in quality"), literal following (conservative-reporting clauses depress recall; MUST/CRITICAL over-triggers), scope expansion = named regression, written deliverables +30-40%. Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988, server-gated, no opt-out) — prose caps would triple-stack. Shipped: delegation block → model-neutral WHEN-guidance + explicit gates carve-out (verifier/security/challenge still dispatch as written); "staff engineer" self-check bar dropped; finish-whole-task clause folded into Deviations (gone-WRONG→STOP still wins); deliverable-length rule; design hook `\bux\b` dropped (`\bui\b` KEPT — 0 FP, 1 logged TP, lock-tested); plan-challenger grounded-doubt→[MINOR] in-place reword (grammar byte-identical). Plan challenged by 3 blind Opus 5 plan-challengers: correctness CONCERNS(4) / robustness FATAL(5, BLOCKER: all surfaces symlink-deployed LIVE — gates fire post-deployment) / simplicity CONCERNS(4); every fix adopted as prescribed (scratch-validation before live hook write, minimal diffs, ux-only, MINOR-routing). Alternatives rejected: leave as-is (nudge actively counter-productive); hard spawn caps in prose (harness injects one); confidence axis on challenger grammar (consumer unwired); dropping \bui\b (no evidence). NOT touched: verify-secure-loop + fresh gates (harness architecture BDR-049/050, ≠ model self-check prose); Security/Architecture sections (BDR-021); settings effortLevel xhigh (user pref — Opus 5 carry-over trap → LRN-139); superpowers plugin wording (external upstream). Plan+synthesis: .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md. Branch feature/opus5-config-tuning, unmerged (human gate). + +### BDR-082 — seo/geo analyzers de-prescribed for Opus 5 (C1) [accepted] (2026-08-02) +BDR-081 N5 follow-on, user-directed apparatus (plan+3-lens challenge+census+dogfood). Method: audience×mode-range invariant — dedup ONLY verbatim same-audience (spec rule / bundle-item payload / phase-local caveat) same-mode-range repeats; cross-mode + agent↔dispatcher twins stay (standalone paths need them). Census-FIRST: lib/tests/seo-geo-contract.test.sh 71 locks (verdict grammar, sentinels, ALL STEP headers incl. interiors, item fields, score labels, envelope keys), flip-proven 7 mutations→7 FAILs, committed BEFORE reword. Shipped: self-output verification removed (":970 run twice"→conditional integrity guard; ":1217"→single-shot-scoped), 2 pre-BDR-061 vestigials fixed, caps softened (P0-rule/MANDATORY/ALWAYS→plain content rules), 2 essays compressed, checklist :1309→routing map rows verbatim (challenger caught it = routing table, NOT self-check), true same-range dups only (seo Handoff+landing-page blocks; geo ZERO — all claimed pairs distinct on inspection). FROZEN: guard-first url-guard orderings, :550 denominator-before-sampling (ordering IS the honesty mechanism), R2/NAP/COVERAGE/citation invariants, external-freshness checks (world drift ≠ self-verification). Deltas: seo 1528→1503 l ("P0 rule" 2→0, ALWAYS 1→0, MUST 5→4, NEVER 9→9 = class-B bans kept); geo 1106→1107 (MANDATORY 1→0, MUST 4→3). Plan challenged correctness FATAL / robustness FATAL(3 BLOCKER) / simplicity CONCERNS + confirmation FATAL(9) — every BLOCKER closed by named change (§5bis record). Dogfood before/after on frozen zenquality copy: judge-replay on frozen signals (zero collect variance) + templates + fresh collects + e2e judge + 42/42 assert battery BOTH sets + blind reader "interchangeable; all deltas = presentation variance both directions OR after MORE spec-conformant". Alternatives rejected: keyword dedup (challengers proved audience/range-blind — most annex "twins" were distinct obligations), FULL/aggressive dogfood (billing gate killed nested CLI; left as user option), banner/shape locks (LLM-convention layers wobble — lock strings only). Evidence: .audit/dogfood-baseline/ (18 artifacts + DOGFOOD-VERDICT.md), plan .claude/tasks/plans/2026-07-30-seo-geo-deprescription-1402.md. Branch feature/seo-geo-deprescription, UNMERGED (human gate). diff --git a/.claude/memory/journal.md b/.claude/memory/journal.md index d489fdb..e690acd 100644 --- a/.claude/memory/journal.md +++ b/.claude/memory/journal.md @@ -430,3 +430,6 @@ rules: ## 2026-07-30 - User: Opus 5 "needs more freedom" → analyse config + adapt. Research 3-agent (registries / config audit / web) + official migration guide: over-delegation (inverts LRN-030), over-verification, literal following, scope expansion, #80988 injections. Plan challenged 3 blind Opus 5 plan-challengers — robustness FATAL (BLOCKER: symlink-live deployment), all fixes adopted. Shipped: CLAUDE.global.md recalibrated (delegation when-guidance, staff-bar dropped, finish-whole-task, deliverable-length; 308/320), design hook \bux\b dropped flip-tested (22/0), plan-challenger grounded-doubt→[MINOR] (44/0). BDR-081 + LRN-139. feature/opus5-config-tuning, UNMERGED. + +## 2026-08-02 +- C1 seo/geo de-prescription EXECUTED end-to-end: census-first 71 locks flip-proven → reword under audience×range invariant (adafa35/c7646a9) → controlled dogfood (judge-replay frozen signals + templates + fresh collects + e2e + blind reader) → 42/42 both sets, zero contract regression, recall improved. Plan survived 4 challenge passes (2 FATAL + confirmation FATAL(9), all closed by name). BDR-082 + LRN-140. Nested-CLI dogfood died on monthly spend limit → inline pipeline (canonical /seo shape). feature/seo-geo-deprescription UNMERGED (human gate). Chantiers C2-C4 pending. diff --git a/.claude/memory/learnings.md b/.claude/memory/learnings.md index 1c27fa5..6a91f43 100644 --- a/.claude/memory/learnings.md +++ b/.claude/memory/learnings.md @@ -1361,3 +1361,10 @@ rules: - **Opus 5 traps found**: (a) Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988; server-gated, no opt-out, absent from transcripts) — own prose stacks on top blindly; (b) NO model-default effort hold on Opus 5 — persisted effortLevel (xhigh, settings.json) silently carries over, against "start high, sweep low/medium"; run /effort sweep per model; (c) effort does NOT shorten visible output/deliverables — only prose length rules do (+30-40% docs). - **future application**: at every model-generation bump, grep config for trait-compensating language ("counters model tendency…", "default to X") and re-verify the premise; prefer WHEN-guidance (conditions where X pays) over directional nudges — survives inversions unchanged. - **link**: [[LRN-030]] [[BDR-081]]. + +## LRN-140 — de-prescription findings: dedup evaporates, self-verify is default, recall survives (2026-08-02) +- **pattern 1 — inventory dedup counts lie**: line-level inspection killed most "duplicate" pairs (seo 9 families→2 real merges; geo 7→0). Twins differ by AUDIENCE (bundle-item payload read by fresh applier vs spec rule) or MODE-RANGE (collect/judge/template/RULES) or are distinct obligations sharing a keyword (30/70 ×3 = three different rules). Dedup rule that survives: verbatim + same-audience + same-range ONLY. +- **pattern 2 — Opus 5 self-verifies unprompted**: "run it twice" instruction REMOVED → after-judge still ran score engine twice, identical output. Removing verify-prose does not remove the behavior; its value = no compounding, no contradiction burn. Confirms BDR-081 E3 mechanism, refines the payoff claim. +- **pattern 3 — de-prescription does NOT depress recall**: reworded collect caught  -encoded phone AT COLLECT (baseline collect missed it); reworded judge found new RGPD finding + self-caught false positive + corrected collect coverage claim 21/21→20/21. Integrity/honesty invariants (kept class B) carry the discipline, not the caps. +- **pattern 4 — lock strings, never shapes**: LLM-convention output layers (banners, fences, table columns, section order) wobble run-to-run in BOTH directions — baseline itself deviated from spec where after conformed (§0 ENTRIES, BUNDLE-before-SCORING). Stable contract = census-locked literal strings; anything unlocked drifts and MUST be tolerated by consumers (tier recognition "by intent" is the right pattern). +- **link**: [[BDR-082]] [[BDR-081]] [[LRN-139]] [[LRN-113]]. diff --git a/.claude/tasks/TODO.md b/.claude/tasks/TODO.md index 4de4c90..e60b70f 100644 --- a/.claude/tasks/TODO.md +++ b/.claude/tasks/TODO.md @@ -29,11 +29,18 @@ adopted as prescribed (plan §5bis, v2 items below). ## 2026-07-30 — Claude 5 follow-on chantiers (user directive, checkpoint between each) Order fixed, one branch per chantier, no merge without per-chantier signal. -- [ ] C1 dé-prescription seo-analyzer.md + geo-analyzer.md (opus pins → Opus 5): - separate machine-parsed contracts (fix-bundles, ownership matrices, - output formats — verbatim) from process choreography (MUST/MANDATORY on - "how" → when-guidance). Dedicated plan + 3-lens challenge, census tests - same commits, real /seo dogfood before/after (zero format regression). +- [x] C1 dé-prescription seo-analyzer.md + geo-analyzer.md — DONE 2026-08-02. + Census-first 71 locks flip-proven (9681b46) → rewords under + audience×range invariant (adafa35 seo, c7646a9 geo) → controlled + before/after dogfood: judge-replay on frozen signals + templates + + fresh collects + e2e judge + blind reader = 42/42 both sets, zero + contract regression, recall improved. Plan challenged 4 passes + (FATAL/FATAL/CONCERNS + confirmation FATAL(9), all closed by name). + BDR-082 + LRN-140. Evidence .audit/dogfood-baseline/ (19 artifacts). + Branch feature/seo-geo-deprescription UNMERGED — human gate. + Residual for gate: §6bis dynamically-unverified list (FULL branches, + apply path — census-locked statically); FULL/aggressive dry-run = user + option; nested-CLI dogfood blocked by monthly spend limit (inline used). - [ ] C2 self-contradiction audit CLAUDE.global.md + own skills: list rule pairs in tension, propose resolution per pair, apply after user OK. /doctor as assistant, not authority. diff --git a/CHANGELOG.md b/CHANGELOG.md index e8769b4..ad138db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,23 @@ Format follows [Keep a Changelog](https://keepachangelog.com/). ## [Unreleased] ### Changed +- **seo-analyzer + geo-analyzer de-prescribed for Opus 5 (BDR-082)** — + process choreography converted to when-guidance under an + audience×mode-range invariant; self-output verification demands removed + (the score-engine "run it twice" became a conditional integrity guard); + two pre-BDR-061 vestigial rules fixed; P0/MANDATORY/ALWAYS caps softened + to plain content rules. Machine contract byte-frozen and locked by the + new `lib/tests/seo-geo-contract.test.sh` census (71 locks, flip-proven); + proven by a controlled before/after `/seo` dogfood — judge replay on + frozen signals, 42/42 presence assertions on both runs, blind structural + reader: interchangeable, recall improved. + +### Added +- **`lib/tests/seo-geo-contract.test.sh`** — census locking the seo/geo + agent ⇄ dispatcher machine contract: judge verdict grammar, FIX BUNDLE + + READY-TO-APPLY sentinel, signals handoff, every STEP header (interiors + included), bundle item fields, score labels, scoring blocks, envelope + keys (46→71 assertions across the C1 chantier). - **Global instruction layer recalibrated for the Claude 5 family (BDR-081)** — delegation block is now model-neutral when-guidance (the Opus 4.8 under-delegation counter inverted on Opus 5, which over-delegates and gets From ae1339d656f83a7c87d2e89fd3039e9316a830d9 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Mon, 24 Aug 2026 12:17:17 +0200 Subject: [PATCH 10/65] chore(config): untrack emil-design-eng skill (machine-owned curl copy) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit skills-external/emil-design-eng/SKILL.md is curl'd from emilkowalski/skill by install-plugins.sh when absent and re-fetched unconditionally by every update-all.sh run, so tracking it produced a repo diff on each upstream edit (latest: Radix vars dropped for Base UI). Same category as frontend-design/ and impeccable/, already ignored on that rationale — a fresh clone re-fetches it, so no offline copy is needed and nothing was pinned here anyway. design-motion-principles/ has the same overwrite-on-update behaviour but NOT the same bootstrap: install-plugins.sh only warns instead of cloning it, so it stays tracked until that gap is closed. --- .gitignore | 6 + plugins.lock.json | 2 +- skills-external/emil-design-eng/SKILL.md | 679 ----------------------- 3 files changed, 7 insertions(+), 680 deletions(-) delete mode 100644 skills-external/emil-design-eng/SKILL.md diff --git a/.gitignore b/.gitignore index e834484..d884dcf 100644 --- a/.gitignore +++ b/.gitignore @@ -142,6 +142,12 @@ desktop.ini # an update. The source is always re-synced, so no offline copy is needed. skills-external/frontend-design/ +# Emil Design Eng — machine-owned copy curl'd from emilkowalski/skill by +# install-plugins.sh (Step 8, when absent) and re-fetched on every update-all.sh +# run. Not vendored: tracking it produced a repo diff each time upstream shipped +# an edit. The source is always re-fetched, so no offline copy is needed. +skills-external/emil-design-eng/ + # Impeccable — machine-owned dist produced by `npx impeccable skills install` # (install-plugins.sh Step 8d, update-all.sh), pinned in plugins.lock.json. # Not vendored: the installer owns the layout and rewrites it on update diff --git a/plugins.lock.json b/plugins.lock.json index 4df45b1..833223c 100644 --- a/plugins.lock.json +++ b/plugins.lock.json @@ -36,7 +36,7 @@ "source": "https://github.com/emilkowalski/skill", "path": "skills/emil-design-eng/SKILL.md", "managed_by": "curl", - "note": "Emil Kowalski's design engineering skill — UI polish, animations, component craft. Downloaded to skills-external/emil-design-eng/, symlinked by link.sh." + "note": "Emil Kowalski's design engineering skill — UI polish, animations, component craft. Machine-owned: curl'd to skills-external/emil-design-eng/ (gitignored, re-fetched by update-all.sh), symlinked by link.sh." }, "impeccable": { "source": "npm:impeccable", diff --git a/skills-external/emil-design-eng/SKILL.md b/skills-external/emil-design-eng/SKILL.md deleted file mode 100644 index 4911235..0000000 --- a/skills-external/emil-design-eng/SKILL.md +++ /dev/null @@ -1,679 +0,0 @@ ---- -name: emil-design-eng -description: This skill encodes Emil Kowalski's philosophy on UI polish, component design, animation decisions, and the invisible details that make software feel great. ---- - -# Design Engineering - -## Initial Response - -When this skill is first invoked without a specific question, respond only with: - -> I'm ready to help you build interfaces that feel right, my knowledge comes from Emil Kowalski's design engineering philosophy. If you want to dive even deeper, check out Emil’s course: [animations.dev](https://animations.dev/). - -Do not provide any other information until the user asks a question. - -You are a design engineer with the craft sensibility. You build interfaces where every detail compounds into something that feels right. You understand that in a world where everyone's software is good enough, taste is the differentiator. - -## Core Philosophy - -### Taste is trained, not innate - -Good taste is not personal preference. It is a trained instinct: the ability to see beyond the obvious and recognize what elevates. You develop it by surrounding yourself with great work, thinking deeply about why something feels good, and practicing relentlessly. - -When building UI, don't just make it work. Study why the best interfaces feel the way they do. Reverse engineer animations. Inspect interactions. Be curious. - -### Unseen details compound - -Most details users never consciously notice. That is the point. When a feature functions exactly as someone assumes it should, they proceed without giving it a second thought. That is the goal. - -> "All those unseen details combine to produce something that's just stunning, like a thousand barely audible voices all singing in tune." - Paul Graham - -Every decision below exists because the aggregate of invisible correctness creates interfaces people love without knowing why. - -### Beauty is leverage - -People select tools based on the overall experience, not just functionality. Good defaults and good animations are real differentiators. Beauty is underutilized in software. Use it as leverage to stand out. - -## Review Format (Required) - -When reviewing UI code, you MUST use a markdown table with Before/After columns. Do NOT use a list with "Before:" and "After:" on separate lines. Always output an actual markdown table like this: - -| Before | After | Why | -| --- | --- | --- | -| `transition: all 300ms` | `transition: transform 200ms ease-out` | Specify exact properties; avoid `all` | -| `transform: scale(0)` | `transform: scale(0.95); opacity: 0` | Nothing in the real world appears from nothing | -| `ease-in` on dropdown | `ease-out` with custom curve | `ease-in` feels sluggish; `ease-out` gives instant feedback | -| No `:active` state on button | `transform: scale(0.97)` on `:active` | Buttons must feel responsive to press | -| `transform-origin: center` on popover | `transform-origin: var(--radix-popover-content-transform-origin)` | Popovers should scale from their trigger (not modals — modals stay centered) | - -Wrong format (never do this): - -``` -Before: transition: all 300ms -After: transition: transform 200ms ease-out -──────────────────────────── -Before: scale(0) -After: scale(0.95) -``` - -Correct format: A single markdown table with | Before | After | Why | columns, one row per issue found. The "Why" column briefly explains the reasoning. - -## The Animation Decision Framework - -Before writing any animation code, answer these questions in order: - -### 1. Should this animate at all? - -**Ask:** How often will users see this animation? - -| Frequency | Decision | -| ----------------------------------------------------------- | ---------------------------- | -| 100+ times/day (keyboard shortcuts, command palette toggle) | No animation. Ever. | -| Tens of times/day (hover effects, list navigation) | Remove or drastically reduce | -| Occasional (modals, drawers, toasts) | Standard animation | -| Rare/first-time (onboarding, feedback forms, celebrations) | Can add delight | - -**Never animate keyboard-initiated actions.** These actions are repeated hundreds of times daily. Animation makes them feel slow, delayed, and disconnected from the user's actions. - -Raycast has no open/close animation. That is the optimal experience for something used hundreds of times a day. - -### 2. What is the purpose? - -Every animation must have a clear answer to "why does this animate?" - -Valid purposes: - -- **Spatial consistency**: toast enters and exits from the same direction, making swipe-to-dismiss feel intuitive -- **State indication**: a morphing feedback button shows the state change -- **Explanation**: a marketing animation that shows how a feature works -- **Feedback**: a button scales down on press, confirming the interface heard the user -- **Preventing jarring changes**: elements appearing or disappearing without transition feel broken - -If the purpose is just "it looks cool" and the user will see it often, don't animate. - -### 3. What easing should it use? - -Is the element entering or exiting? - Yes → ease-out (starts fast, feels responsive) - No → - Is it moving/morphing on screen? - Yes → ease-in-out (natural acceleration/deceleration) - Is it a hover/color change? - Yes → ease - Is it constant motion (marquee, progress bar)? - Yes → linear - Default → ease-out - -**Critical: use custom easing curves.** The built-in CSS easings are too weak. They lack the punch that makes animations feel intentional. - -```css -/* Strong ease-out for UI interactions */ ---ease-out: cubic-bezier(0.23, 1, 0.32, 1); - -/* Strong ease-in-out for on-screen movement */ ---ease-in-out: cubic-bezier(0.77, 0, 0.175, 1); - -/* iOS-like drawer curve (from Ionic Framework) */ ---ease-drawer: cubic-bezier(0.32, 0.72, 0, 1); -``` - -**Never use ease-in for UI animations.** It starts slow, which makes the interface feel sluggish and unresponsive. A dropdown with `ease-in` at 300ms _feels_ slower than `ease-out` at the same 300ms, because ease-in delays the initial movement — the exact moment the user is watching most closely. - -**Easing curve resources:** Don't create curves from scratch. Use [easing.dev](https://easing.dev/) or [easings.co](https://easings.co/) to find stronger custom variants of standard easings. - -### 4. How fast should it be? - -| Element | Duration | -| ------------------------ | ------------- | -| Button press feedback | 100-160ms | -| Tooltips, small popovers | 125-200ms | -| Dropdowns, selects | 150-250ms | -| Modals, drawers | 200-500ms | -| Marketing/explanatory | Can be longer | - -**Rule: UI animations should stay under 300ms.** A 180ms dropdown feels more responsive than a 400ms one. A faster-spinning spinner makes the app feel like it loads faster, even when the load time is identical. - -### Perceived performance - -Speed in animation is not just about feeling snappy — it directly affects how users perceive your app's performance: - -- A **fast-spinning spinner** makes loading feel faster (same load time, different perception) -- A **180ms select** animation feels more responsive than a **400ms** one -- **Instant tooltips** after the first one is open (skip delay + skip animation) make the whole toolbar feel faster - -The perception of speed matters as much as actual speed. Easing amplifies this: `ease-out` at 200ms _feels_ faster than `ease-in` at 200ms because the user sees immediate movement. - -## Spring Animations - -Springs feel more natural than duration-based animations because they simulate real physics. They don't have fixed durations — they settle based on physical parameters. - -### When to use springs - -- Drag interactions with momentum -- Elements that should feel "alive" (like Apple's Dynamic Island) -- Gestures that can be interrupted mid-animation -- Decorative mouse-tracking interactions - -### Spring-based mouse interactions - -Tying visual changes directly to mouse position feels artificial because it lacks motion. Use `useSpring` from Motion (formerly Framer Motion) to interpolate value changes with spring-like behavior instead of updating immediately. - -```jsx -import { useSpring } from 'framer-motion'; - -// Without spring: feels artificial, instant -const rotation = mouseX * 0.1; - -// With spring: feels natural, has momentum -const springRotation = useSpring(mouseX * 0.1, { - stiffness: 100, - damping: 10, -}); -``` - -This works because the animation is **decorative** — it doesn't serve a function. If this were a functional graph in a banking app, no animation would be better. Know when decoration helps and when it hinders. - -### Spring configuration - -**Apple's approach (recommended — easier to reason about):** - -```js -{ type: "spring", duration: 0.5, bounce: 0.2 } -``` - -**Traditional physics (more control):** - -```js -{ type: "spring", mass: 1, stiffness: 100, damping: 10 } -``` - -Keep bounce subtle (0.1-0.3) when used. Avoid bounce in most UI contexts. Use it for drag-to-dismiss and playful interactions. - -### Interruptibility advantage - -Springs maintain velocity when interrupted — CSS animations and keyframes restart from zero. This makes springs ideal for gestures users might change mid-motion. When you click an expanded item and quickly press Escape, a spring-based animation smoothly reverses from its current position. - -## Component Building Principles - -### Buttons must feel responsive - -Add `transform: scale(0.97)` on `:active`. This gives instant feedback, making the UI feel like it is truly listening to the user. - -```css -.button { - transition: transform 160ms ease-out; -} - -.button:active { - transform: scale(0.97); -} -``` - -This applies to any pressable element. The scale should be subtle (0.95-0.98). - -### Never animate from scale(0) - -Nothing in the real world disappears and reappears completely. Elements animating from `scale(0)` look like they come out of nowhere. - -Start from `scale(0.9)` or higher, combined with opacity. Even a barely-visible initial scale makes the entrance feel more natural, like a balloon that has a visible shape even when deflated. - -```css -/* Bad */ -.entering { - transform: scale(0); -} - -/* Good */ -.entering { - transform: scale(0.95); - opacity: 0; -} -``` - -### Make popovers origin-aware - -Popovers should scale in from their trigger, not from center. The default `transform-origin: center` is wrong for almost every popover. **Exception: modals.** Modals should keep `transform-origin: center` because they are not anchored to a specific trigger — they appear centered in the viewport. - -```css -/* Radix UI */ -.popover { - transform-origin: var(--radix-popover-content-transform-origin); -} - -/* Base UI */ -.popover { - transform-origin: var(--transform-origin); -} -``` - -Whether the user notices the difference individually does not matter. In the aggregate, unseen details become visible. They compound. - -### Tooltips: skip delay on subsequent hovers - -Tooltips should delay before appearing to prevent accidental activation. But once one tooltip is open, hovering over adjacent tooltips should open them instantly with no animation. This feels faster without defeating the purpose of the initial delay. - -```css -.tooltip { - transition: transform 125ms ease-out, opacity 125ms ease-out; - transform-origin: var(--transform-origin); -} - -.tooltip[data-starting-style], -.tooltip[data-ending-style] { - opacity: 0; - transform: scale(0.97); -} - -/* Skip animation on subsequent tooltips */ -.tooltip[data-instant] { - transition-duration: 0ms; -} -``` - -### Use CSS transitions over keyframes for interruptible UI - -CSS transitions can be interrupted and retargeted mid-animation. Keyframes restart from zero. For any interaction that can be triggered rapidly (adding toasts, toggling states), transitions produce smoother results. - -```css -/* Interruptible - good for UI */ -.toast { - transition: transform 400ms ease; -} - -/* Not interruptible - avoid for dynamic UI */ -@keyframes slideIn { - from { - transform: translateY(100%); - } - to { - transform: translateY(0); - } -} -``` - -### Use blur to mask imperfect transitions - -When a crossfade between two states feels off despite trying different easings and durations, add subtle `filter: blur(2px)` during the transition. - -**Why blur works:** Without blur, you see two distinct objects during a crossfade — the old state and the new state overlapping. This looks unnatural. Blur bridges the visual gap by blending the two states together, tricking the eye into perceiving a single smooth transformation instead of two objects swapping. - -Combine blur with scale-on-press (`scale(0.97)`) for a polished button state transition: - -```css -.button { - transition: transform 160ms ease-out; -} - -.button:active { - transform: scale(0.97); -} - -.button-content { - transition: filter 200ms ease, opacity 200ms ease; -} - -.button-content.transitioning { - filter: blur(2px); - opacity: 0.7; -} -``` - -Keep blur under 20px. Heavy blur is expensive, especially in Safari. - -### Animate enter states with @starting-style - -The modern CSS way to animate element entry without JavaScript: - -```css -.toast { - opacity: 1; - transform: translateY(0); - transition: opacity 400ms ease, transform 400ms ease; - - @starting-style { - opacity: 0; - transform: translateY(100%); - } -} -``` - -This replaces the common React pattern of using `useEffect` to set `mounted: true` after initial render. Use `@starting-style` when browser support allows; fall back to the `data-mounted` attribute pattern otherwise. - -```jsx -// Legacy pattern (still works everywhere) -useEffect(() => { - setMounted(true); -}, []); -//
-``` - -## CSS Transform Mastery - -### translateY with percentages - -Percentage values in `translate()` are relative to the element's own size. Use `translateY(100%)` to move an element by its own height, regardless of actual dimensions. This is how Sonner positions toasts and how Vaul hides the drawer before animating in. - -```css -/* Works regardless of drawer height */ -.drawer-hidden { - transform: translateY(100%); -} - -/* Works regardless of toast height */ -.toast-enter { - transform: translateY(-100%); -} -``` - -Prefer percentages over hardcoded pixel values. They are less error-prone and adapt to content. - -### scale() scales children too - -Unlike `width`/`height`, `scale()` also scales an element's children. When scaling a button on press, the font size, icons, and content scale proportionally. This is a feature, not a bug. - -### 3D transforms for depth - -`rotateX()`, `rotateY()` with `transform-style: preserve-3d` create real 3D effects in CSS. Orbiting animations, coin flips, and depth effects are all possible without JavaScript. - -```css -.wrapper { - transform-style: preserve-3d; -} - -@keyframes orbit { - from { - transform: translate(-50%, -50%) rotateY(0deg) translateZ(72px) rotateY(360deg); - } - to { - transform: translate(-50%, -50%) rotateY(360deg) translateZ(72px) rotateY(0deg); - } -} -``` - -### transform-origin - -Every element has an anchor point from which transforms execute. The default is center. Set it to match where the trigger lives for origin-aware interactions. - -## clip-path for Animation - -`clip-path` is not just for shapes. It is one of the most powerful animation tools in CSS. - -### The inset shape - -`clip-path: inset(top right bottom left)` defines a rectangular clipping region. Each value "eats" into the element from that side. - -```css -/* Fully hidden from right */ -.hidden { - clip-path: inset(0 100% 0 0); -} - -/* Fully visible */ -.visible { - clip-path: inset(0 0 0 0); -} - -/* Reveal from left to right */ -.overlay { - clip-path: inset(0 100% 0 0); - transition: clip-path 200ms ease-out; -} -.button:active .overlay { - clip-path: inset(0 0 0 0); - transition: clip-path 2s linear; -} -``` - -### Tabs with perfect color transitions - -Duplicate the tab list. Style the copy as "active" (different background, different text color). Clip the copy so only the active tab is visible. Animate the clip on tab change. This creates a seamless color transition that timing individual color transitions can never achieve. - -### Hold-to-delete pattern - -Use `clip-path: inset(0 100% 0 0)` on a colored overlay. On `:active`, transition to `inset(0 0 0 0)` over 2s with linear timing. On release, snap back with 200ms ease-out. Add `scale(0.97)` on the button for press feedback. - -### Image reveals on scroll - -Start with `clip-path: inset(0 0 100% 0)` (hidden from bottom). Animate to `inset(0 0 0 0)` when the element enters the viewport. Use `IntersectionObserver` or Framer Motion's `useInView` with `{ once: true, margin: "-100px" }`. - -### Comparison sliders - -Overlay two images. Clip the top one with `clip-path: inset(0 50% 0 0)`. Adjust the right inset value based on drag position. No extra DOM elements needed, fully hardware-accelerated. - -## Gesture and Drag Interactions - -### Momentum-based dismissal - -Don't require dragging past a threshold. Calculate velocity: `Math.abs(dragDistance) / elapsedTime`. If velocity exceeds ~0.11, dismiss regardless of distance. A quick flick should be enough. - -```js -const timeTaken = new Date().getTime() - dragStartTime.current.getTime(); -const velocity = Math.abs(swipeAmount) / timeTaken; - -if (Math.abs(swipeAmount) >= SWIPE_THRESHOLD || velocity > 0.11) { - dismiss(); -} -``` - -### Damping at boundaries - -When a user drags past the natural boundary (e.g., dragging a drawer up when already at top), apply damping. The more they drag, the less the element moves. Things in real life don't suddenly stop; they slow down first. - -### Pointer capture for drag - -Once dragging starts, set the element to capture all pointer events. This ensures dragging continues even if the pointer leaves the element bounds. - -### Multi-touch protection - -Ignore additional touch points after the initial drag begins. Without this, switching fingers mid-drag causes the element to jump to the new position. - -```js -function onPress() { - if (isDragging) return; - // Start drag... -} -``` - -### Friction instead of hard stops - -Instead of preventing upward drag entirely, allow it with increasing friction. It feels more natural than hitting an invisible wall. - -## Performance Rules - -### Only animate transform and opacity - -These properties skip layout and paint, running on the GPU. Animating `padding`, `margin`, `height`, or `width` triggers all three rendering steps. - -### CSS variables are inheritable - -Changing a CSS variable on a parent recalculates styles for all children. In a drawer with many items, updating `--swipe-amount` on the container causes expensive style recalculation. Update `transform` directly on the element instead. - -```js -// Bad: triggers recalc on all children -element.style.setProperty('--swipe-amount', `${distance}px`); - -// Good: only affects this element -element.style.transform = `translateY(${distance}px)`; -``` - -### Framer Motion hardware acceleration caveat - -Framer Motion's shorthand properties (`x`, `y`, `scale`) are NOT hardware-accelerated. They use `requestAnimationFrame` on the main thread. For hardware acceleration, use the full `transform` string: - -```jsx -// NOT hardware accelerated (convenient but drops frames under load) - - -// Hardware accelerated (stays smooth even when main thread is busy) - -``` - -This matters when the browser is simultaneously loading content, running scripts, or painting. At Vercel, the dashboard tab animation used Shared Layout Animations and dropped frames during page loads. Switching to CSS animations (off main thread) fixed it. - -### CSS animations beat JS under load - -CSS animations run off the main thread. When the browser is busy loading a new page, Framer Motion animations (using `requestAnimationFrame`) drop frames. CSS animations remain smooth. Use CSS for predetermined animations; JS for dynamic, interruptible ones. - -### Use WAAPI for programmatic CSS animations - -The Web Animations API gives you JavaScript control with CSS performance. Hardware-accelerated, interruptible, and no library needed. - -```js -element.animate([{ clipPath: 'inset(0 0 100% 0)' }, { clipPath: 'inset(0 0 0 0)' }], { - duration: 1000, - fill: 'forwards', - easing: 'cubic-bezier(0.77, 0, 0.175, 1)', -}); -``` - -## Accessibility - -### prefers-reduced-motion - -Animations can cause motion sickness. Reduced motion means fewer and gentler animations, not zero. Keep opacity and color transitions that aid comprehension. Remove movement and position animations. - -```css -@media (prefers-reduced-motion: reduce) { - .element { - animation: fade 0.2s ease; - /* No transform-based motion */ - } -} -``` - -```jsx -const shouldReduceMotion = useReducedMotion(); -const closedX = shouldReduceMotion ? 0 : '-100%'; -``` - -### Touch device hover states - -```css -@media (hover: hover) and (pointer: fine) { - .element:hover { - transform: scale(1.05); - } -} -``` - -Touch devices trigger hover on tap, causing false positives. Gate hover animations behind this media query. - -## The Sonner Principles (Building Loved Components) - -These principles come from building Sonner (13M+ weekly npm downloads) and apply to any component: - -1. **Developer experience is key.** No hooks, no context, no complex setup. Insert `` once, call `toast()` from anywhere. The less friction to adopt, the more people will use it. - -2. **Good defaults matter more than options.** Ship beautiful out of the box. Most users never customize. The default easing, timing, and visual design should be excellent. - -3. **Naming creates identity.** "Sonner" (French for "to ring") feels more elegant than "react-toast". Sacrifice discoverability for memorability when appropriate. - -4. **Handle edge cases invisibly.** Pause toast timers when the tab is hidden. Fill gaps between stacked toasts with pseudo-elements to maintain hover state. Capture pointer events during drag. Users never notice these, and that is exactly right. - -5. **Use transitions, not keyframes, for dynamic UI.** Toasts are added rapidly. Keyframes restart from zero on interruption. Transitions retarget smoothly. - -6. **Build a great documentation site.** Let people touch the product, play with it, and understand it before they use it. Interactive examples with ready-to-use code snippets lower the barrier to adoption. - -### Cohesion matters - -Sonner's animation feels satisfying partly because the whole experience is cohesive. The easing and duration fit the vibe of the library. It is slightly slower than typical UI animations and uses `ease` rather than `ease-out` to feel more elegant. The animation style matches the toast design, the page design, the name — everything is in harmony. - -When choosing animation values, consider the personality of the component. A playful component can be bouncier. A professional dashboard should be crisp and fast. Match the motion to the mood. - -### The opacity + height combination - -When items enter and exit a list (like Family's drawer), the opacity change must work well with the height animation. This is often trial and error. There is no formula — you adjust until it feels right. - -### Review your work the next day - -Review animations with fresh eyes. You notice imperfections the next day that you missed during development. Play animations in slow motion or frame by frame to spot timing issues that are invisible at full speed. - -### Asymmetric enter/exit timing - -Pressing should be slow when it needs to be deliberate (hold-to-delete: 2s linear), but release should always be snappy (200ms ease-out). This pattern applies broadly: slow where the user is deciding, fast where the system is responding. - -```css -/* Release: fast */ -.overlay { - transition: clip-path 200ms ease-out; -} - -/* Press: slow and deliberate */ -.button:active .overlay { - transition: clip-path 2s linear; -} -``` - -## Stagger Animations - -When multiple elements enter together, stagger their appearance. Each element animates in with a small delay after the previous one. This creates a cascading effect that feels more natural than everything appearing at once. - -```css -.item { - opacity: 0; - transform: translateY(8px); - animation: fadeIn 300ms ease-out forwards; -} - -.item:nth-child(1) { - animation-delay: 0ms; -} -.item:nth-child(2) { - animation-delay: 50ms; -} -.item:nth-child(3) { - animation-delay: 100ms; -} -.item:nth-child(4) { - animation-delay: 150ms; -} - -@keyframes fadeIn { - to { - opacity: 1; - transform: translateY(0); - } -} -``` - -Keep stagger delays short (30-80ms between items). Long delays make the interface feel slow. Stagger is decorative — never block interaction while stagger animations are playing. - -## Debugging Animations - -### Slow motion testing - -Play animations at reduced speed to spot issues invisible at full speed. Temporarily increase duration to 2-5x normal, or use browser DevTools animation inspector to slow playback. - -Things to look for in slow motion: - -- Do colors transition smoothly, or do you see two distinct states overlapping? -- Does the easing feel right, or does it start/stop abruptly? -- Is the transform-origin correct, or does the element scale from the wrong point? -- Are multiple animated properties (opacity, transform, color) in sync? - -### Frame-by-frame inspection - -Step through animations frame by frame in Chrome DevTools (Animations panel). This reveals timing issues between coordinated properties that you cannot see at full speed. - -### Test on real devices - -For touch interactions (drawers, swipe gestures), test on physical devices. Connect your phone via USB, visit your local dev server by IP address, and use Safari's remote devtools. The Xcode Simulator is an alternative but real hardware is better for gesture testing. - -## Review Checklist - -When reviewing UI code, check for: - -| Issue | Fix | -| ------------------------------------------ | ---------------------------------------------------------------- | -| `transition: all` | Specify exact properties: `transition: transform 200ms ease-out` | -| `scale(0)` entry animation | Start from `scale(0.95)` with `opacity: 0` | -| `ease-in` on UI element | Switch to `ease-out` or custom curve | -| `transform-origin: center` on popover | Set to trigger location or use Radix/Base UI CSS variable (modals are exempt — keep centered) | -| Animation on keyboard action | Remove animation entirely | -| Duration > 300ms on UI element | Reduce to 150-250ms | -| Hover animation without media query | Add `@media (hover: hover) and (pointer: fine)` | -| Keyframes on rapidly-triggered element | Use CSS transitions for interruptibility | -| Framer Motion `x`/`y` props under load | Use `transform: "translateX()"` for hardware acceleration | -| Same enter/exit transition speed | Make exit faster than enter (e.g., enter 2s, exit 200ms) | -| Elements all appear at once | Add stagger delay (30-80ms between items) | From 63310467cab43efa1610f692ca34a1d70834ae24 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Mon, 24 Aug 2026 13:12:38 +0200 Subject: [PATCH 11/65] feat(gates): deterministic floor (GATE 0) under the fresh verifier MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GATE 1 is an LLM dispatch and the verifier's mandatory PROOF: line is a line the verifier writes — nothing structurally stops it being produced without anything being executed. Nothing deterministic sat between the executor and that dispatch. An acceptance criterion can now carry an oracle: indented CHECK: (command), EXPECT: (success-only marker), EVIDENCE: (slot). lib/gates.sh runs them fail-closed — MET requires exit 0 AND the marker, so a nonzero process never passes on its error text carrying the token — and writes the outcome back into the contract, so the fresh verifier reads evidence as fact rather than trusting the executor's report. GATE 0 runs that floor before any verifier is dispatched; a red build sends the executor back for free, on its own iteration budget. ABANDON: turns an impossible criterion into a visible handoff that blocks CONFORME and routes to the human gate, via the new ABANDONED(n) verdict — a distinct token because it routes distinctly, never a dev loop. feater and bugfixer gain a four-pass completion discipline, scoped so a pass can never widen the contract. The runner's parse fails closed on partial oracles, duplicate ids, unindented attributes and runnable criteria with no EVIDENCE: line, and executes nothing at all when the ledger is malformed. status never executes and never writes; run always re-executes, since trusting written evidence is the failure being closed. Adapted from the unlazy skill (Leonxlnx/unlazy, MIT). Its Stop hook, approval store, .unlazy/ tree, depth-tree arithmetic and Node checker were deliberately refused — BDR-083 records each reason. 64 assertions in lib/tests/gates.test.sh, non-execution proved by sentinel with its own positive control asserted first. --- agents/bugfixer.md | 18 ++ agents/feater.md | 19 ++ agents/verifier.md | 45 +++- lib/contract-interview.md | 56 ++++- lib/gates.sh | 323 ++++++++++++++++++++++++++++ lib/tests/contract-verifier.test.sh | 2 +- lib/tests/gates.test.sh | 317 +++++++++++++++++++++++++++ lib/verify-secure-loop.md | 66 ++++-- 8 files changed, 826 insertions(+), 20 deletions(-) create mode 100644 lib/gates.sh create mode 100644 lib/tests/gates.test.sh diff --git a/agents/bugfixer.md b/agents/bugfixer.md index c1771ab..f24cd33 100644 --- a/agents/bugfixer.md +++ b/agents/bugfixer.md @@ -46,6 +46,24 @@ Every choice was made in the plan or is a NEED-DECISION to report. security/verifier dispatch, editing `.claude/**` or memory registries, user questions (you cannot ask — report instead), attribution trailers of any kind. +## FOUR PASSES — over the fix and its test, nothing else + +Loop these until a full pass finds nothing. They apply to the fix and the +regression test ONLY — "keep the fix minimal" above still governs. They make +the minimal fix COMPLETE; they never widen it. + +1. **Complete.** The ROOT CAUSE named in DIAGNOSIS is closed, not just the + reported symptom. No placeholder, no deferred remainder. +2. **Expert reread.** Does the fix hold for the neighbouring inputs and error + paths that reach the same root cause, or only for the one case reported? +3. **Negative control.** Confirm the regression test actually FAILS without + the fix — stash it, run the test, restore. A test that passes both ways + proves nothing, and a green suite then certifies nothing. +4. **Polish.** Naming and comments on what you touched. Nothing else. + +A pass that wants a file outside the contract FILE SCOPE is a +`NEED-DECISION`, not a pass. + ## OUTPUT — end with exactly this report (your final message) ``` diff --git a/agents/feater.md b/agents/feater.md index 6d43318..1346f59 100644 --- a/agents/feater.md +++ b/agents/feater.md @@ -57,6 +57,25 @@ report below is optional on this path (the dispatcher needs the edit applied editing `.claude/**` or memory registries, user questions (you cannot ask — report instead), attribution trailers of any kind. +## FOUR PASSES — before you report DONE + +Do not stop at the first version that runs. Loop these until a full pass +finds nothing: + +1. **Complete.** The whole deliverable the plan names is implemented. No + placeholder, no TODO, no deferred remainder you plan to mention in NOTES. +2. **Expert reread.** Read it as someone who owns this codebase. Where you + took the cheap version of a part, replace it with the one the plan asked + for. +3. **Defect hunt.** Correctness, error paths, integration with the callers + you did NOT touch, portability. Fix what you find. +4. **Polish.** Low-cost only: naming, comment density, dead code you + introduced. + +Every pass stays inside the plan and the contract FILE SCOPE. A pass that +wants to leave either is a `NEED-DECISION`, not a pass — these passes make +the requested work COMPLETE, they never widen it. + ## OUTPUT — end with exactly this report (your final message) ``` diff --git a/agents/verifier.md b/agents/verifier.md index f6fd9bf..ba6a7ae 100644 --- a/agents/verifier.md +++ b/agents/verifier.md @@ -48,6 +48,25 @@ Rules: read the diff AND enough surrounding code to judge behavior; run criterion. Never mark `MET` from naming, comments, or plausibility — only from behavior you observed or code you read. +### Criteria carrying an oracle (`CHECK:` / `EXPECT:` / `EVIDENCE:`) + +`lib/gates.sh run` already executed these and wrote the outcome over the +`EVIDENCE:` line. Read it from the contract and treat it as fact: + +- `EVIDENCE: NOT-MET …` or `EVIDENCE: pending` → the criterion is `NOT-MET`. + Reading the code NEVER overrides a red or unrun oracle. Cite the evidence + line as your evidence. +- `EVIDENCE: MET …` → the declared command passed. That is the strongest + evidence available for that criterion — but it proves the ORACLE, not the + English sentence. Read the `CHECK:` and confirm it observes the artifact + the criterion names. A vacuous oracle (`1. invoices reconcile` + + `CHECK: echo ok`) is `NOT-MET` — reason `vacuous oracle`, quoting the + command. That judgement is yours alone; no command can make it. + +You may re-run a `CHECK:` yourself to settle a doubt (Bash is read-only, and +these commands are observation). You may NOT edit the contract — an evidence +line you disagree with is reported, never rewritten. + ## STEP 3 — SCOPE CHECK List the files actually touched (`git diff --name-only` over `DIFF`). @@ -58,19 +77,30 @@ only enters the contract through a human micro-gate. ## STEP 4 — VERDICT -`CONFORME` ⇔ ALL criteria `MET` AND zero out-of-scope files. -Anything else is `ECARTS(n)` where n = count(NOT-MET) + count(UNVERIFIABLE) -+ count(out-of-scope files). +Read the contract's `ABANDON:` lines. An abandoned criterion is `ABANDONED` +— never `MET`, never counted as a gap the dev can close. + +Precedence, first match wins — fix what is fixable before escalating what +is not: + +1. `ERROR()` — the contract is missing or unreadable. +2. `ECARTS(n)` — n = count(NOT-MET) + count(UNVERIFIABLE) + count(out-of-scope + files). Surface any abandonment in the same report. +3. `ABANDONED(n)` — zero gaps remain, but n abandonments stand. This is NOT + a pass and NOT a dev loop: it routes straight to the human gate. +4. `CONFORME` — ALL criteria `MET`, zero out-of-scope files, zero + abandonments. ## OUTPUT (exact format — machine-parsed by the orchestrator) ``` -VERIFY — VERDICT: CONFORME | ECARTS(n) | ERROR() +VERIFY — VERDICT: CONFORME | ECARTS(n) | ABANDONED(n) | ERROR() CONTRACT: CRITERIA: - 1. — MET — + 1. — MET — 2. — NOT-MET — expected <…> / actual <…> — 3. — UNVERIFIABLE — + 4. — ABANDONED — SCOPE: in-scope files; out-of-scope: PROOF: read files, ran , checked / criteria ``` @@ -82,6 +112,8 @@ PROOF: read files, ran , checked / criteria - `UNVERIFIABLE` ≠ `MET`. A criterion you did not check is `UNVERIFIABLE`, never silently dropped: the checked count in `PROOF` must equal the contract's criteria count. +- `ABANDONED` ≠ `MET`. An abandonment is a visible handoff, never a pass — + report it verbatim even when everything else is green. - `PROOF` is MANDATORY. A `CONFORME` without a `PROOF` line is invalid — the orchestrator discards it as a structural failure (LRN-048: a pass must prove it looked). @@ -103,6 +135,9 @@ loop, never here): with the CRITERIA table (the contract-vs-realized diff). - Remaining `UNVERIFIABLE` while everything else is MET → direct human gate (a dev cannot fix unverifiability). + - `ABANDONED(n)` → direct human gate, never a dev loop. The human either + lifts the abandonment (the criterion was fixable after all) or accepts + the partial delivery; the run is never reported as fully complete. - Structural failure (`ERROR(…)`, missing/duplicated VERDICT line, unparsable output, agent crash, `CONFORME` without `PROOF`) → retry ONCE with a fresh verifier; a 2nd structural failure → human diff --git a/lib/contract-interview.md b/lib/contract-interview.md index 9dcc1c4..684bf85 100644 --- a/lib/contract-interview.md +++ b/lib/contract-interview.md @@ -35,6 +35,38 @@ ask what the repo can answer — verify paths/APIs/behavior yourself first. this conversation. - FILE SCOPE: paths/zones expected to change, or `repo-wide — `. +### ORACLES — a criterion a command can decide carries one + +Give such a criterion an indented `CHECK:` (the command), `EXPECT:` (a +success-only marker), and `EVIDENCE: pending`. +`bash ~/.claude/lib/gates.sh run ` executes it fail-closed — MET +requires exit 0 **AND** the marker — and writes the result back over the +`EVIDENCE:` line. That persisted evidence is what the fresh verifier reads +as fact instead of trusting the executor's report (GATE 0 in +`lib/verify-secure-loop.md`). + +Both attributes or neither. `CHECK:` without `EXPECT:` is a parse error, not +a manual criterion — the runner refuses the whole ledger. Leave a criterion +oracle-free when no command can decide it; the verifier judges those. + +Four authoring rules — a gate that cannot fail proves nothing: + +1. **Observe the named artifact.** The check reads the file, service, or + measurement the criterion's own words name — never a proxy for it. + `1. invoices reconcile` + `CHECK: echo ok` is valid and worthless. +2. **Success-only marker.** The script runs every assertion, exits nonzero + on any failure, and prints the `EXPECT:` string only after all pass. +3. **Positive control before any absence check.** Run the same logic against + a fixture known to trip it and confirm it fails. A missing file, a wrong + path, and a broken pattern all look exactly like valid absence. +4. **Recompute supplied numbers.** Never copy a figure from the request into + `EXPECT:` — the script derives it from source and prints its own marker. + A number that is its own proof proves nothing. + +`CHECK:` is shell code run with our privileges. It is safe only because we +author it in our own repo — never build one out of externally-supplied text +(a scraped URL, a client string); route those through `lib/url-guard.sh`. + ## STEP 4 — WRITE TO DISK (immediately, before any next step) Path: `.claude/tasks/contracts/--.md` @@ -57,8 +89,13 @@ Q: / A: (or: none — request complete) ## ACCEPTANCE CRITERIA -1. -2. +1. + CHECK: + EXPECT: + EVIDENCE: pending +2. + +(ABANDON: — only for a criterion proven impossible) ## FILE SCOPE @@ -78,6 +115,13 @@ Print one line to the user, then continue the flow: this micro-gate: human approves → FILE SCOPE gains the entry `[gated]`; human declines → the dev removes the edit. Without this gate the dev justifies everything and scope constrains nothing. +- **ABANDONMENT**: a criterion proven impossible within the authorized task + is NEVER deleted and never quietly downgraded. Keep it, append + `ABANDON: ` under the criteria, and name it + in the final report. An abandonment is a visible handoff, not a pass: the + verifier cannot return `CONFORME` while one stands, and the run cannot be + described as fully complete. This is the structural half of the house rule + "blocked on an independent sub-part → do the rest, state what's missing". - **Deep re-scope** (the request itself changes): NEW contract file with `supersedes: ` in its header — never a rewrite of the old one. - **Aborted run**: delete the contract file, or commit it with @@ -96,6 +140,14 @@ Print one line to the user, then continue the flow: | init-project | Full. The interviewer's PROJECT BRIEF pours into the contract (V1 features → criteria). | | onboard | Audit-scope contract (interview answers → what to audit, which axes). | +Oracles follow the same proportion. hotfix: the build/tests criterion carries +its `CHECK:`, nothing else. feat / bugfix: the suite criterion at minimum, and +for bugfix the regression test the DIAGNOSIS names — its `CHECK:` runs that +test alone, so a green result means the reproduction actually flipped. +ship-feature / init-project: build, suite, and every criterion a command can +settle. onboard: audit criteria are mostly judgement — leave them oracle-free +rather than invent a check that cannot fail. + ## Hand-off rule Downstream consumers (plan step, dev subagents, verifier) receive the diff --git a/lib/gates.sh b/lib/gates.sh new file mode 100644 index 0000000..36b5404 --- /dev/null +++ b/lib/gates.sh @@ -0,0 +1,323 @@ +#!/usr/bin/env bash +# Deterministic floor under GATE 1: execute the acceptance criteria that the +# contract itself declares as oracles, fail-closed, and persist the evidence +# INTO the contract file. +# +# bash ~/.claude/lib/gates.sh status # parse only, never runs +# bash ~/.claude/lib/gates.sh run # execute + write evidence +# +# rc 0 = MET every runnable criterion passed, no abandonment standing +# 2 = UNMET a runnable criterion failed, or the ledger is malformed +# 3 = ABANDONED runnable criteria all passed, an abandonment still stands +# +# WHY: GATE 1 (lib/verify-secure-loop.md) is an LLM dispatch, and the +# verifier's mandatory `PROOF:` line is a line the verifier WRITES — nothing +# structurally stops it from being produced without anything being executed. +# This runs what the contract declares BEFORE a verifier is ever spawned: a +# red floor sends the executor back for free. Adapted from the `unlazy` skill +# (Leonxlnx/unlazy) — its gate ledger, minus the machinery we do not need. +# +# `run` always re-executes every runnable criterion, including ones already +# recorded MET. Trusting written evidence is exactly the failure this closes, +# so there is no incremental mode to get it wrong with. +# +# TRUST BOUNDARY: `CHECK:` is shell code, run with this process's privileges +# and environment. That is safe here only because the contract is authored by +# our own orchestrator in our own repo — which is why there is no approval +# store (we never execute ledgers inherited from a foreign repo). NEVER build +# a `CHECK:` out of externally-supplied text; route such values through +# lib/url-guard.sh first. +set -uo pipefail + +TIMEOUT="${GATES_TIMEOUT:-120}" +EVIDENCE_CAP=140 + +# Module-level parse tables, index-aligned. Bash has no record type; threading +# eight parallel arrays through every call would cost more readability than +# the explicit data flow buys. +_ID=(); _TEXT=(); _CHECK=(); _EXPECT=(); _EVLINE=(); _EVTEXT=() +_STATUS=(); _EVID=() +_ABANDON_ID=(); _ABANDON_WHY=() +_ERRORS=() +_CUR=-1 + +_die() { printf 'GATES — VERDICT: ERROR(%s)\n' "$1"; exit 2; } +_err() { _ERRORS+=("$1"); } + +_trim() { + local s="$1" + s="${s#"${s%%[![:space:]]*}"}" + printf '%s' "${s%"${s##*[![:space:]]}"}" +} + +# ── parse ─────────────────────────────────────────────────────────────────── + +_new_crit() { # _new_crit + local i + for ((i = 0; i < ${#_ID[@]}; i++)); do + if [ "${_ID[i]}" = "$1" ]; then + _err "duplicate criterion id: $1" + # Orphan what follows instead of aliasing it onto the previous + # criterion, which would hand one gate another gate's oracle. + _CUR=-1 + return 0 + fi + done + _ID+=("$1"); _TEXT+=("$2") + _CHECK+=(""); _EXPECT+=(""); _EVLINE+=("0"); _EVTEXT+=("") + _CUR=$((${#_ID[@]} - 1)) +} + +_set_attr() { # _set_attr + if [ "$_CUR" -lt 0 ]; then + _err "$1 at line $3 belongs to no criterion" + return 0 + fi + case "$1" in + CHECK) _CHECK[_CUR]="$2" ;; + EXPECT) _EXPECT[_CUR]="$2" ;; + EVIDENCE) _EVLINE[_CUR]="$3"; _EVTEXT[_CUR]="$2" ;; + esac +} + +# An UNINDENTED attribute is diagnosed, never absorbed: silently ignoring it +# would demote a runnable criterion to a manual one, which is the one parse +# bug that turns this checker into a rubber stamp. +_absorb() { # _absorb + local body + if [[ "$1" =~ ^([0-9]+)\.[[:space:]]+(.*)$ ]]; then + _new_crit "${BASH_REMATCH[1]}" "${BASH_REMATCH[2]}" + elif [[ "$1" =~ ^ABANDON:[[:space:]]*([0-9]+)?[[:space:]]*(.*)$ ]]; then + _ABANDON_ID+=("${BASH_REMATCH[1]}"); _ABANDON_WHY+=("${BASH_REMATCH[2]}") + elif [[ "$1" =~ ^(CHECK|EXPECT|EVIDENCE): ]]; then + _err "unindented ${BASH_REMATCH[1]}: at line $2" + elif [[ "$1" =~ ^[[:space:]]+(CHECK|EXPECT|EVIDENCE):(.*)$ ]]; then + body="$(_trim "${BASH_REMATCH[2]}")" + _set_attr "${BASH_REMATCH[1]}" "$body" "$2" + fi +} + +_parse() { # _parse + local line n=0 fence=0 inblock=0 + while IFS= read -r line || [ -n "$line" ]; do + n=$((n + 1)) + case "$line" in '```'*) fence=$((1 - fence)); continue ;; esac + [ "$fence" -eq 1 ] && continue + case "$line" in + '## ACCEPTANCE CRITERIA'*) inblock=1; continue ;; + '## '*) inblock=0; continue ;; + esac + [ "$inblock" -eq 1 ] && _absorb "$line" "$n" + done < "$1" +} + +# ── validation ────────────────────────────────────────────────────────────── + +_validate_oracles() { + local i + for ((i = 0; i < ${#_ID[@]}; i++)); do + if [ -n "${_CHECK[i]}" ] && [ -z "${_EXPECT[i]}" ]; then + _err "criterion ${_ID[i]}: CHECK without EXPECT (partial oracle)" + elif [ -z "${_CHECK[i]}" ] && [ -n "${_EXPECT[i]}" ]; then + _err "criterion ${_ID[i]}: EXPECT without CHECK (partial oracle)" + elif [ -n "${_CHECK[i]}" ] && [ "${_EVLINE[i]}" = "0" ]; then + _err "criterion ${_ID[i]}: runnable but has no EVIDENCE: line" + fi + done +} + +_validate_abandons() { + local i j found + for ((i = 0; i < ${#_ABANDON_ID[@]}; i++)); do + found=0 + for ((j = 0; j < ${#_ID[@]}; j++)); do + [ "${_ID[j]}" = "${_ABANDON_ID[i]}" ] && found=1 + done + [ "$found" -eq 1 ] || + _err "ABANDON names unknown criterion: '${_ABANDON_ID[i]}'" + [ -n "$(_trim "${_ABANDON_WHY[i]}")" ] || + _err "ABANDON ${_ABANDON_ID[i]}: blank reason (a handoff needs one)" + done +} + +_is_abandoned() { # _is_abandoned + local i + for ((i = 0; i < ${#_ABANDON_ID[@]}; i++)); do + [ "${_ABANDON_ID[i]}" = "$1" ] && return 0 + done + return 1 +} + +# ── execution ─────────────────────────────────────────────────────────────── + +# One line, capped, newlines flattened: the smallest output that proves the +# outcome. Full logs stay in the terminal, never in the contract. +_decisive() { # _decisive + local flat + flat="$(printf '%s' "$1" | tr '\n\r\t' ' ' | tr -s ' ')" + flat="$(_trim "$flat")" + if [ "${#flat}" -gt "$EVIDENCE_CAP" ]; then + printf '%s…' "${flat:0:$EVIDENCE_CAP}" + else + printf '%s' "$flat" + fi +} + +# Fail-closed: exit 0 AND the marker. A nonzero process never passes because +# its error text happens to contain the expected token. +_run_one() { # _run_one + local i="$1" out rc + out="$(timeout "$TIMEOUT" bash -c "${_CHECK[i]}" 2>&1)" + rc=$? + _STATUS[i]="NOT-MET" + if [ "$rc" -eq 124 ]; then + _EVID[i]="NOT-MET timeout=${TIMEOUT}s" + elif [ "$rc" -ne 0 ]; then + _EVID[i]="NOT-MET exit=$rc (nonzero) :: $(_decisive "$out")" + elif [[ "$out" != *"${_EXPECT[i]}"* ]]; then + _EVID[i]="NOT-MET exit=0 marker-absent :: $(_decisive "$out")" + else + _STATUS[i]="MET" + _EVID[i]="MET exit=0 marker-found :: $(_decisive "$out")" + fi +} + +_run_all() { + local i + for ((i = 0; i < ${#_ID[@]}; i++)); do + _STATUS[i]=""; _EVID[i]="" + [ -n "${_CHECK[i]}" ] && _run_one "$i" + done +} + +_evline_owner() { # _evline_owner — echoes idx, or nothing + local i + for ((i = 0; i < ${#_ID[@]}; i++)); do + if [ "${_EVLINE[i]}" = "$1" ] && [ -n "${_EVID[i]}" ]; then + printf '%s' "$i" + return 0 + fi + done +} + +# Rewrites only the EVIDENCE lines of criteria that actually ran; every other +# byte of the contract is copied through, indentation included. +_write_back() { # _write_back + local tmp line n=0 idx + tmp="$(mktemp)" || _die "mktemp failed" + while IFS= read -r line || [ -n "$line" ]; do + n=$((n + 1)) + idx="$(_evline_owner "$n")" + if [ -n "$idx" ]; then + printf '%s%s\n' "${line%%[![:space:]]*}" "EVIDENCE: ${_EVID[idx]}" + else + printf '%s\n' "$line" + fi + done < "$1" > "$tmp" + cat "$tmp" > "$1" && rm -f "$tmp" +} + +# ── report ────────────────────────────────────────────────────────────────── + +# A recorded `pending`, or a criterion that never ran, is PENDING — never MET. +# `status` reports what the file says; it does not revalidate old evidence. +_row_state() { # _row_state + local i="$1" + _is_abandoned "${_ID[i]}" && { printf 'ABANDONED'; return 0; } + [ -z "${_CHECK[i]}" ] && { printf 'MANUAL'; return 0; } + [ -n "${_STATUS[i]:-}" ] && { printf '%s' "${_STATUS[i]}"; return 0; } + case "${_EVTEXT[i]}" in + MET' '*) printf 'MET-RECORDED' ;; + *) printf 'PENDING' ;; + esac +} + +_report_rows() { + local i state + for ((i = 0; i < ${#_ID[@]}; i++)); do + state="$(_row_state "$i")" + printf ' %-3s %-13s %s\n' "${_ID[i]}" "$state" "${_TEXT[i]}" + done +} + +_report_abandons() { + local i + for ((i = 0; i < ${#_ABANDON_ID[@]}; i++)); do + printf ' ABANDONED %s — %s\n' "${_ABANDON_ID[i]}" "${_ABANDON_WHY[i]}" + done +} + +_count_state() { # _count_state + local i n=0 + for ((i = 0; i < ${#_ID[@]}; i++)); do + [ "$(_row_state "$i")" = "$1" ] && n=$((n + 1)) + done + printf '%s' "$n" +} + +_verdict() { # _verdict — prints the line, returns the rc + local unmet pending abandoned + if [ "${#_ERRORS[@]}" -gt 0 ]; then + printf 'GATES — VERDICT: ERROR(%s)\n' "${#_ERRORS[@]}" + return 2 + fi + unmet="$(_count_state NOT-MET)" + pending="$(_count_state PENDING)" + abandoned="$(_count_state ABANDONED)" + [ "$unmet" -gt 0 ] && + { printf 'GATES — VERDICT: UNMET(%s)\n' "$unmet"; return 2; } + if [ "$1" = "status" ] && [ "$pending" -gt 0 ]; then + printf 'GATES — VERDICT: PENDING(%s)\n' "$pending" + return 2 + fi + [ "$abandoned" -gt 0 ] && + { printf 'GATES — VERDICT: ABANDONED(%s)\n' "$abandoned"; return 3; } + printf 'GATES — VERDICT: MET\n' + return 0 +} + +_report() { # _report + local rc + printf 'GATES — %s (%s)\n' "$2" "$1" + _report_rows + _report_abandons + [ "${#_ERRORS[@]}" -gt 0 ] && printf ' ERROR %s\n' "${_ERRORS[@]}" + printf 'RUNNABLE: %s of %s criteria; timeout %ss\n' \ + "$(_runnable_count)" "${#_ID[@]}" "$TIMEOUT" + _verdict "$1" + rc=$? + return "$rc" +} + +_runnable_count() { + local i n=0 + for ((i = 0; i < ${#_ID[@]}; i++)); do + [ -n "${_CHECK[i]}" ] && n=$((n + 1)) + done + printf '%s' "$n" +} + +# ── entry point ───────────────────────────────────────────────────────────── + +main() { # main + local mode="$1" file="$2" + [ -r "$file" ] || _die "contract unreadable: $file" + _parse "$file" + [ "${#_ID[@]}" -gt 0 ] || + _die "no numbered criteria under ## ACCEPTANCE CRITERIA" + _validate_oracles + _validate_abandons + if [ "$mode" = "run" ] && [ "${#_ERRORS[@]}" -eq 0 ]; then + _run_all + _write_back "$file" + fi + _report "$mode" "$file" +} + +case "${1:-}" in + status|run) + [ $# -eq 2 ] || _die "usage: gates.sh {status|run} " + main "$1" "$2" + ;; + *) _die "usage: gates.sh {status|run} " ;; +esac diff --git a/lib/tests/contract-verifier.test.sh b/lib/tests/contract-verifier.test.sh index af7b99f..ce77347 100644 --- a/lib/tests/contract-verifier.test.sh +++ b/lib/tests/contract-verifier.test.sh @@ -67,7 +67,7 @@ fi tr_ "frontmatter name" "$AGT" "^name: verifier$" tr_ "tools read-only set" "$AGT" "^tools: Read, Grep, Glob, Bash$" tn "no write-capable tools" "$AGT" "^tools:.*(Edit|Write|NotebookEdit)" -tf "verdict grammar" "$AGT" "VERIFY — VERDICT: CONFORME | ECARTS(n) | ERROR()" +tf "verdict grammar" "$AGT" "VERIFY — VERDICT: CONFORME | ECARTS(n) | ABANDONED(n) | ERROR()" tf "blind — no iteration history" "$AGT" "NEVER receive iteration history" tf "blind — complete every time" "$AGT" "every verification is complete and blind" tf "unverifiable is not met" "$AGT" "\`UNVERIFIABLE\` ≠ \`MET\`" diff --git a/lib/tests/gates.test.sh b/lib/tests/gates.test.sh new file mode 100644 index 0000000..76d5fa7 --- /dev/null +++ b/lib/tests/gates.test.sh @@ -0,0 +1,317 @@ +#!/usr/bin/env bash +# ============================================================ +# lib/gates.sh — behavioural tests + structure locks for the +# deterministic floor (GATE 0, lib/verify-secure-loop.md). +# +# Fail-closed is the entire point of this runner, so every +# "looks green but must not pass" case is asserted explicitly: +# nonzero exit carrying the marker, marker absent, timeout, +# unindented attribute silently demoting a gate to manual. +# Non-execution is proved with a sentinel file, and the +# sentinel's own positive control is asserted first — an +# absence check that was never able to fire proves nothing. +# ============================================================ +set -uo pipefail + +REPO="$(cd "$(dirname "$0")/../.." && pwd)" +GATES="$REPO/lib/gates.sh" +WORK="$(mktemp -d)" +trap 'rm -rf "$WORK"' EXIT +PASS=0; FAIL=0; N=0 +LAST="" + +ok() { echo " PASS $1"; PASS=$((PASS + 1)); } +bad() { echo " FAIL $1 — $2"; FAIL=$((FAIL + 1)); } + +# gate