From eab2a10cd51047f6355a871fdb7cf4cef2cc85c3 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:57:38 +0200 Subject: [PATCH 1/4] fix(hooks): drop \bux\b from design-toolchain pattern (FR prose FPs) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 3rd tightening pass (series LRN-1005/1007): bare "ux" matched inside French prose (2 logged FPs, both FR — latest "changement ux vu"). \bui\b kept: zero logged FP, one logged true positive, now locked by a must-fire test row. Flip-tested: quiet row fired pre-change. --- hooks/design-toolchain-reminder.sh | 6 +++++- lib/tests/design-toolchain-reminder.test.sh | 2 ++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/hooks/design-toolchain-reminder.sh b/hooks/design-toolchain-reminder.sh index e39b76a..68547eb 100755 --- a/hooks/design-toolchain-reminder.sh +++ b/hooks/design-toolchain-reminder.sh @@ -44,7 +44,11 @@ lc="$(printf '%s' "$prompt" | tr '[:upper:]' '[:lower:]')" # "design system", "redesign", "front-?end design". dashboard -> \bdashboard\b # so a filename like ecc_dashboard.py no longer matches while "admin dashboard" # still does. animation kept (rarely non-UI). -pattern='redesign|refonte|refont|ui/ux|ux/ui|\bui\b|\bux\b|ui kit|design system|design-system|front-?end design|\bnavbar\b|\bsidebar\b|\bmodal\b|\bbouton\b|\bbutton\b|formulaire|\bhero\b|\bheader\b|\bfooter\b|dropdown|tooltip|\bbadge\b|\bchart\b|graphique|accordion|carousel|\bslider\b|landing|\bdashboard\b|homepage|home page|\baccueil\b|\bécran\b|\becran\b|portfolio|maquette|mockup|wireframe|prototype|\bjoli\b|\bjolie\b|\bbeau\b|\bbelle\b|esth[eé]tique|aesthetic|\bvisuel\b|\bvisual\b|embellir|fignol|peaufin|polish|styliser|styling|stylesheet|\bskin\b|charte graphique|\bbrand\b|branding|\blogo\b|favicon|ic[oô]ne|\bicon\b|\bcss\b|tailwind|shadcn|couleur|gradient|d[eé]grad[eé]|\bombre\b|spacing|espacement|\bmarge\b|\bpadding\b|\bmargin\b|\bradius\b|arrondi|\bhover\b|dark mode|light mode|typograph|\bfont\b|\bfonts\b|font pairing|\bpolice\b|animation|\bmotion\b|micro-interaction|keyframe|glassmorph|neumorph|claymorph|skeuomorph|brutalis|bento|minimalis|responsive|figma' +# Tightened 2026-07-30 (3rd pass): dropped \bux\b — bare "ux" matched inside +# French prose ("changement ux vu…"; 2 logged FPs, both FR). \bui\b KEPT +# (zero logged FP, one logged true positive). NB: the log records only the +# FIRST match per fire (head -1), so per-token FP rates aren't derivable. +pattern='redesign|refonte|refont|ui/ux|ux/ui|\bui\b|ui kit|design system|design-system|front-?end design|\bnavbar\b|\bsidebar\b|\bmodal\b|\bbouton\b|\bbutton\b|formulaire|\bhero\b|\bheader\b|\bfooter\b|dropdown|tooltip|\bbadge\b|\bchart\b|graphique|accordion|carousel|\bslider\b|landing|\bdashboard\b|homepage|home page|\baccueil\b|\bécran\b|\becran\b|portfolio|maquette|mockup|wireframe|prototype|\bjoli\b|\bjolie\b|\bbeau\b|\bbelle\b|esth[eé]tique|aesthetic|\bvisuel\b|\bvisual\b|embellir|fignol|peaufin|polish|styliser|styling|stylesheet|\bskin\b|charte graphique|\bbrand\b|branding|\blogo\b|favicon|ic[oô]ne|\bicon\b|\bcss\b|tailwind|shadcn|couleur|gradient|d[eé]grad[eé]|\bombre\b|spacing|espacement|\bmarge\b|\bpadding\b|\bmargin\b|\bradius\b|arrondi|\bhover\b|dark mode|light mode|typograph|\bfont\b|\bfonts\b|font pairing|\bpolice\b|animation|\bmotion\b|micro-interaction|keyframe|glassmorph|neumorph|claymorph|skeuomorph|brutalis|bento|minimalis|responsive|figma' if printf '%s' "$lc" | grep -Eq "$pattern"; then # Counter: log the fire (time, matched token, excerpt) — best-effort, never blocks. diff --git a/lib/tests/design-toolchain-reminder.test.sh b/lib/tests/design-toolchain-reminder.test.sh index 959882f..53977cb 100644 --- a/lib/tests/design-toolchain-reminder.test.sh +++ b/lib/tests/design-toolchain-reminder.test.sh @@ -22,6 +22,7 @@ check D8-dash-file "$(fire 'ecc_dashboard.py')" quiet # --- Harness-generated inputs must be QUIET even with UI tokens --- check D9-tasknotif "$(fire ' x add css header fonts')" quiet check D10-notif-file "$(fire ' design-motion-principles keyframe done')" quiet +check D11-bare-ux "$(fire 'changement ux vu de tes trouvailles')" quiet # --- Real UI signals must still FIRE --- check F1-button "$(fire 'add a button')" fire @@ -33,6 +34,7 @@ check F6-frontdesign "$(fire 'frontend design work')" fire check F7-admin-dash "$(fire 'admin dashboard screen')" fire check F8-animation "$(fire 'add an animation')" fire check F9-designsys "$(fire 'our design system')" fire +check F10-bare-ui "$(fire 'revois l'\''ui du panneau admin')" fire # --- Fire is logged (time + token + excerpt) --- tmp="$(mktemp -d)" From 0f7b565bb042b942759156dd9bdf63a6b0cb1e9e Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:58:06 +0200 Subject: [PATCH 2/4] feat(global): recalibrate instruction layer for Claude 5 family MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Delegation block: model-neutral when-guidance replaces the Opus 4.8 under-delegation counter (LRN-030 trait inverted on Opus 5; Claude Code injects its own anti-delegation prompt there, #80988). Gates carve-out keeps verifier/security/challenge dispatch mandatory. - Drop the 'staff engineer' self-check bar (Opus 5 over-verification trigger per official migration guide); honest-reporting steps stay. - Deviations bullet: finish-whole-task clause (Opus 5 scope-expansion counter), scoped so 'gone wrong → STOP' still wins. - Written-deliverable length rule (Opus 5 writes ~30-40% longer). --- CLAUDE.global.md | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/CLAUDE.global.md b/CLAUDE.global.md index 3aa4c2d..3ab8178 100644 --- a/CLAUDE.global.md +++ b/CLAUDE.global.md @@ -22,6 +22,8 @@ Apply unless repo-specific instructions override. - Document intent, not mechanics. Use project doc style (docstring, JSDoc…). - Explicit, consistent, meaningful names. Straight control flow, no hidden side effects. +- Written deliverables (docs, reports, .md): length matched to what + the task needs — no filler sections, no boilerplate summaries. ## Refactoring - Priority: safety → readability → consistency. @@ -40,11 +42,12 @@ Apply unless repo-specific instructions override. - Confirm before implementing only when real trade-offs exist (multiple valid approaches, breaking change, destructive action) — else proceed. - Minimal changes unless broader refactor requested. State trade-offs. -- Sub-agents keep main context clean — one task per sub-agent. - More compute on hard problems. Task fans out across independent - items (many files, parallel searches, multi-point checks) → delegate - to sub-agents, don't iterate serially. Default to delegation for - multi-file exploration. Counters model tendency to under-delegate. +- Sub-agents: one task per sub-agent, main context stays clean. + Delegate genuinely independent, sizeable tracks (wide multi-file + exploration, parallel audits) — not work doable in a few tool + calls. Skill-mandated gates (fresh verifier/security/challenge) + always dispatch as written. Don't redo delegated work by hand — + failed gates re-dispatch fresh executors instead. - One question upfront if needed — don't interrupt mid-task. *Exception: skill-mandated gates and checkpoints (orchestrator validation gates, approval gates, darwin checkpoints) always fire.* @@ -53,6 +56,8 @@ Apply unless repo-specific instructions override. - Something goes wrong → STOP, re-plan. Never push through. - Deviations: minor or clearly justified → do, explain after. Significant or shaky justification → ask before deviating. + Finish the whole task: blocked on an independent sub-part → do + the rest, state what's missing. Gone WRONG → still STOP, re-plan. - Root causes only. No temp fixes. Never assume — verify paths, APIs, variables before use. @@ -77,7 +82,6 @@ Apply unless repo-specific instructions override. 2. Report what verified, what not. 3. List remaining risks, surviving deviations. 4. Don't mark complete without proof it works. - Bar: "would staff engineer approve?" 5. Correction or notable event → capitalize to right registry (see "Memory registries"). From c3d3f4d4658a388d0827dd8e3452aa57c7ae374d Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 12:58:29 +0200 Subject: [PATCH 3/4] =?UTF-8?q?feat(agents):=20plan-challenger=20=E2=80=94?= =?UTF-8?q?=20route=20grounded=20doubts=20to=20[MINOR]?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Opus 5 follows conservative-reporting clauses literally; 'a manufactured concern is a failure' risked suppressing real low-confidence findings. In-place reword: ungrounded stays noise, grounded-but-uncertain files as [MINOR] with the uncertainty in WHY:. OUTPUT grammar byte-identical; census row added. --- agents/plan-challenger.md | 5 +++-- lib/tests/plan-challenger.test.sh | 1 + 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/agents/plan-challenger.md b/agents/plan-challenger.md index e5c0fda..d4ef552 100644 --- a/agents/plan-challenger.md +++ b/agents/plan-challenger.md @@ -79,8 +79,9 @@ PROOF: read files, inspected , checked plan §<…> - Report-only. Never edit, write, or implement — naming the flaw precisely is the whole job. -- No invention. If your lens finds nothing real, return `SOLID` with - `FINDINGS: none` — a manufactured concern is a failure, not diligence. +- No invention — ungrounded is noise. Silently dropping a grounded doubt is + equally a failure: file it as `[MINOR]` with the uncertainty stated in + `WHY:`. Nothing real at all → `SOLID` with `FINDINGS: none`. - `PROOF` is MANDATORY. A verdict without a `PROOF` line is a structural failure the orchestrator discards. - Stay in your lens. A finding outside it belongs to another challenger. diff --git a/lib/tests/plan-challenger.test.sh b/lib/tests/plan-challenger.test.sh index fa70adc..9733f4f 100644 --- a/lib/tests/plan-challenger.test.sh +++ b/lib/tests/plan-challenger.test.sh @@ -24,6 +24,7 @@ has "$A" "correctness" has "$A" "robustness" has "$A" "simplicity" has "$A" "Report-only" +has "$A" "grounded doubt" # uncertain findings → [MINOR], not self-censored (Opus 5 literalism) # 2) reusable phase — the mechanism lives here (one canonical include) has "$L" 'subagent_type="plan-challenger"' From 550b39043e18fb71e2ad4fd81d2ef7c89a1cdac6 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Thu, 30 Jul 2026 13:02:55 +0200 Subject: [PATCH 4/4] chore(memory): BDR-081 + LRN-139 + journal + CHANGELOG + plan (opus5 tuning) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Capitalizes the Claude-5-family config recalibration: decision record, trait-inversion learning (LRN-030 superseded premise, #80988 injections, no-effort-hold trap), journal line, CHANGELOG Unreleased entries, and the challenged plan (3 blind Opus 5 lenses, synthesis in §5bis). --- .claude/memory/decisions.md | 3 + .claude/memory/journal.md | 3 + .claude/memory/learnings.md | 6 + .claude/tasks/TODO.md | 25 ++ .../2026-07-30-opus5-config-tuning-1238.md | 255 ++++++++++++++++++ CHANGELOG.md | 14 + 6 files changed, 306 insertions(+) create mode 100644 .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md diff --git a/.claude/memory/decisions.md b/.claude/memory/decisions.md index 15c51ba..c4a99fa 100644 --- a/.claude/memory/decisions.md +++ b/.claude/memory/decisions.md @@ -1073,3 +1073,6 @@ Audit (user ask "profile toggles externals both ways?"): ASYMMETRIC. Enable side ### BDR-080 — bug routing inverted: /bugfix primary, /investigate explicit-only [accepted] (2026-07-21) Old routing "Bug → investigate (bugfix if gstack off)" + gstack ON by default → every bug took path bypassing own quality pipeline (gitflow aiguillage, contract, fresh verifier + security gates, doc-sync, `.claude/memory` registries) — /bugfix relegated to near-never fallback. Skill comparison: same core doctrine (root-cause iron law, hypothesis loop, regression test, 3-strike stop, >5-files alert) but incompatible wrappers — investigate monolithic (same context investigates+fixes+verifies, ~1075-line SKILL.md w/ gstack preamble/telemetry/onboarding, capitalizes to `~/.gstack` learnings.jsonl framework never reads at session start); bugfix orchestrator (reflection inline, sonnet bugfixer executor, fresh gates — BDR-066, LRN-083). Composition rejected: skills superpose in context, don't compose — invoking investigate inside bugfix = two full workflows, two completion protocols, two memory systems loaded at once. Decision: CLAUDE.global.md routing line inverted — bugfix primary; investigate ONLY on explicit ask for gstack ecosystem (cross-project learnings, /freeze scope lock, long no-commit investigation). Alternatives rejected: keep investigate primary (bypasses framework), embed investigate inside bugfix (context conflict, dual memory). Known drift noted at write time: Index table rows BDR-074..079 missing (pre-existing, /prune-memory scope). + +### BDR-081 — Config recalibrated for Claude 5 family (Opus 5 dispatch tier) [accepted] (2026-07-30) +Opus 5 (released 2026-07-24) now backs every `model: opus` pin (BDR-076/077) + any `/model opus` session. Research (official migration guide + web + registries): Opus 5 OVER-delegates (inverts LRN-030 Opus 4.8 trait that CLAUDE.global.md:43-47 compensated), self-verifies (explicit verify instructions → over-verification, "removing them reduces wasted tokens with no loss in quality"), literal following (conservative-reporting clauses depress recall; MUST/CRITICAL over-triggers), scope expansion = named regression, written deliverables +30-40%. Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988, server-gated, no opt-out) — prose caps would triple-stack. Shipped: delegation block → model-neutral WHEN-guidance + explicit gates carve-out (verifier/security/challenge still dispatch as written); "staff engineer" self-check bar dropped; finish-whole-task clause folded into Deviations (gone-WRONG→STOP still wins); deliverable-length rule; design hook `\bux\b` dropped (`\bui\b` KEPT — 0 FP, 1 logged TP, lock-tested); plan-challenger grounded-doubt→[MINOR] in-place reword (grammar byte-identical). Plan challenged by 3 blind Opus 5 plan-challengers: correctness CONCERNS(4) / robustness FATAL(5, BLOCKER: all surfaces symlink-deployed LIVE — gates fire post-deployment) / simplicity CONCERNS(4); every fix adopted as prescribed (scratch-validation before live hook write, minimal diffs, ux-only, MINOR-routing). Alternatives rejected: leave as-is (nudge actively counter-productive); hard spawn caps in prose (harness injects one); confidence axis on challenger grammar (consumer unwired); dropping \bui\b (no evidence). NOT touched: verify-secure-loop + fresh gates (harness architecture BDR-049/050, ≠ model self-check prose); Security/Architecture sections (BDR-021); settings effortLevel xhigh (user pref — Opus 5 carry-over trap → LRN-139); superpowers plugin wording (external upstream). Plan+synthesis: .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md. Branch feature/opus5-config-tuning, unmerged (human gate). diff --git a/.claude/memory/journal.md b/.claude/memory/journal.md index 78a43c9..d489fdb 100644 --- a/.claude/memory/journal.md +++ b/.claude/memory/journal.md @@ -427,3 +427,6 @@ rules: ## 2026-07-22 - User: auto-gitignore+delete transient pipeline artifacts in all projects. Investigation reframed the ask — gitignore = WRONG tool (files read from disk during run; would break superpowers SDD `git add` of spec). BDR-065 already rejected gitignore + its DELETE side was doctrine-only (no code, manual chore slipped once — 655e364). User picks (2 recommended): keep committed-during-run + AUTOMATE delete; keep `.claude/tasks/{contracts,plans}` versioned. - Built `lib/gitflow.sh` `_gitflow_purge_transient` at finish (feature/bugfix, pre-merge, best-effort never-abort, opt-out `GITFLOW_PURGE_TRANSIENT=0`) + `purge-transient` CLI verb. Universal via `~/.claude/lib`→repo symlink. gitflow-test T17 a-d (10 checks, `--full-history` recovery), shellcheck clean, make test exit 0. BDR-065 amendment + [[LRN-138]]. feature/gitflow-auto-purge-transient. + +## 2026-07-30 +- User: Opus 5 "needs more freedom" → analyse config + adapt. Research 3-agent (registries / config audit / web) + official migration guide: over-delegation (inverts LRN-030), over-verification, literal following, scope expansion, #80988 injections. Plan challenged 3 blind Opus 5 plan-challengers — robustness FATAL (BLOCKER: symlink-live deployment), all fixes adopted. Shipped: CLAUDE.global.md recalibrated (delegation when-guidance, staff-bar dropped, finish-whole-task, deliverable-length; 308/320), design hook \bux\b dropped flip-tested (22/0), plan-challenger grounded-doubt→[MINOR] (44/0). BDR-081 + LRN-139. feature/opus5-config-tuning, UNMERGED. diff --git a/.claude/memory/learnings.md b/.claude/memory/learnings.md index 2a4d04a..1c27fa5 100644 --- a/.claude/memory/learnings.md +++ b/.claude/memory/learnings.md @@ -1355,3 +1355,9 @@ rules: - **context**: user asked to gitignore transient planning artifacts (`docs/superpowers/{specs,plans}`, `.claude/tasks/{contracts,plans}`) to stop them merging. BDR-065 had already REJECTED gitignore for docs/superpowers on the git-travel ground; the real gap was the DELETE side never being coded (doctrine-only manual chore, slipped once — 655e364). Built `_gitflow_purge_transient`. - **future application**: "don't merge transient X" → ask: does the run read X from disk? does X travel via git (worktree, foreign checkout)? Yes → auto-purge at finish, not gitignore. Scoped commit `-- ` avoids sweeping a dirty index; `git diff --quiet HEAD -- paths` precheck makes `git rm` all-or-nothing safe; keep the purge best-effort so cleanup NEVER blocks a merge. Prove archive-reachability with `git log --full-history` / `git show :path` — plain `git log -- path` prunes the purged add-commit via history simplification (bit me writing T17). - **link**: [[BDR-065]]. + +## LRN-139 — model-trait compensations invert across generations; state WHEN-guidance, not direction (2026-07-30) +- **pattern**: config rules that COMPENSATE a model trait become counter-productive when the next generation inverts the trait. LRN-030 (Opus 4.8 under-delegates → "Default to delegation… counters under-delegation") inverted by Opus 5 (delegates MORE readily, official guide) — the rule pushed the failure the model now has. Same class: explicit verify instructions → over-verification; conservative-reporting clauses → literal recall suppression; MUST/CRITICAL → over-triggering. +- **Opus 5 traps found**: (a) Claude Code injects Opus-5-only anti-delegation prompt sections (heron_brook + subagent_steer_delegation, issue #80988; server-gated, no opt-out, absent from transcripts) — own prose stacks on top blindly; (b) NO model-default effort hold on Opus 5 — persisted effortLevel (xhigh, settings.json) silently carries over, against "start high, sweep low/medium"; run /effort sweep per model; (c) effort does NOT shorten visible output/deliverables — only prose length rules do (+30-40% docs). +- **future application**: at every model-generation bump, grep config for trait-compensating language ("counters model tendency…", "default to X") and re-verify the premise; prefer WHEN-guidance (conditions where X pays) over directional nudges — survives inversions unchanged. +- **link**: [[LRN-030]] [[BDR-081]]. diff --git a/.claude/tasks/TODO.md b/.claude/tasks/TODO.md index cc2c8cb..9e6c2a4 100644 --- a/.claude/tasks/TODO.md +++ b/.claude/tasks/TODO.md @@ -1,5 +1,30 @@ # TODO +## 2026-07-30 — adapt config for Claude 5 family / Opus 5 (feature/opus5-config-tuning) +User: Opus 5 "needs more freedom" → research (official migration guide + +web + registres) confirms: over-delegates (inverts LRN-030 Opus 4.8 trait), +over-verifies if told to verify, literal instruction following, scope +expansion named regression, harness already injects anti-delegation on +Opus 5 (#80988). Plan: .claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md +— to be challenged by 3 blind plan-challengers (opus pins → Opus 5), then +executed on feature branch. NO merge (human gate). +Challenged 2026-07-30: correctness CONCERNS(4) · robustness FATAL(5, 1 +BLOCKER: symlink-live deployment) · simplicity CONCERNS(4) — all fixes +adopted as prescribed (plan §5bis, v2 items below). +- [x] W0 branch first (eab2a10 parent); hook regex validated on scratch copy + (bash -n + shellcheck + 5 replays, HOME sandboxed) before live write +- [x] W1 delegation block v2 (when-guidance + gates carve-out + scoped don't-redo) — 0f7b565 +- [x] W2 "staff engineer" bar line deleted — 0f7b565 +- [x] W3 finish-whole-task folded into Deviations (+ gone-WRONG→STOP) — 0f7b565 +- [x] W4 deliverable-length rule — 0f7b565 +- [x] W5 line budget: 308/320 +- [x] W6 hook \bux\b dropped, \bui\b kept + F10 must-fire lock, D11 quiet row + flip-tested (fire before/quiet after) — eab2a10, suite 22/0 +- [x] W7 plan-challenger :82-83 reworded → [MINOR] routing, census row — c3d3f4d, 44/0 +- [x] W8 BDR-081 + LRN-139 + journal + CHANGELOG +- [ ] W9 final gate: make test full suite +- [ ] W10 no gitflow finish (human gate) — merge only on explicit user signal + ## 2026-07-22 — auto-purge transient superpowers artifacts at finish (feature/gitflow-auto-purge-transient) User: transient planning artifacts (`docs/superpowers/{specs,plans}`) leak into develop; BDR-065 "post-merge cleanup" is DOCTRINE ONLY (no code) — manual chore, diff --git a/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md b/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md new file mode 100644 index 0000000..e9c2c7b --- /dev/null +++ b/.claude/tasks/plans/2026-07-30-opus5-config-tuning-1238.md @@ -0,0 +1,255 @@ +# PLAN — Adapt claude-config for the Claude 5 family (Opus 5 focus) + +Date: 2026-07-30 · Branch (planned): feature/opus5-config-tuning (off develop) +KIND: build-plan · Author: main-loop session (Fable 5) + +## 1. Context & evidence + +Opus 5 (`claude-opus-5`, released 2026-07-24) now backs every `model: opus` +agent pin in this repo (analyzer, plan-challenger, seo/geo-analyzer, +plugin-advisor — BDR-076/077) and any session the user switches to via +`/model opus`. Its documented behavioral profile differs from Opus 4.8 in +ways that make parts of this config counterproductive: + +- E1 **Over-delegation**: Opus 5 "delegates to subagents more readily than + prior models" (official prompting guide). Opus 4.8 had the OPPOSITE trait + (LRN-030), and `CLAUDE.global.md:43-47` was written to counter it + ("Counters model tendency to under-delegate"). The premise is inverted. +- E2 **Anti-delegation already injected by the harness**: Claude Code + v2.1.219 server-gates an Opus-5-only prompt section (`heron_brook` + + `subagent_steer_delegation`, GitHub issue #80988) that says "Do not call + the AgentTool unless the user requested it" and "Subagents multiply cost + and time…". Stacking our own hard cap on top would triple-constrain; + keeping a pro-delegation nudge would fight the injection. Model-neutral + when-guidance is the stable middle. +- E3 **Over-verification**: official guidance — "If your prompt contains + explicit verification instructions … remove them: instructions like these + cause over-verification on Claude Opus 5, and removing them reduces wasted + tokens with no loss in quality." Also true of per-prompt "double-check" + phrasing. Targets PROSE told to the model, not harness-level gates. +- E4 **Scope expansion**: named Opus 5 regression ("can expand the scope of + a task, adding steps that weren't requested"). Anthropic ships a literal + counter-block; tested to reduce scope changes "to nearly zero". +- E5 **Literal instruction following** (since 4.7, stronger now): aggressive + MUST/CRITICAL language over-triggers; conservative-reporting instructions + ("only report high-severity") measurably depress recall in review/challenge + harnesses. +- E6 **Longer written deliverables**: files written to disk run ~30-40% + longer; `effort` does NOT control visible/deliverable length — only prose + instructions do. +- E7 **Overconstraint costs reasoning**: Anthropic removed >80% of Claude + Code's system prompt for Claude-5-generation models "with no measurable + loss"; named mechanism = tokens burned resolving conflicting rules. +- E8 **Hook false positive (today)**: `\bux\b` in + `hooks/design-toolchain-reminder.sh:47` fired on French prose ("changement + ux vu" — matches after apostrophe/slash/space); 2nd `ux` FP in the log, + both French. Continues the LRN-1005/1007 false-positive series. No test + row covers `\bui\b`/`\bux\b`. +- E9 **Effort carry-over trap**: Opus 5 has no model-default effort hold in + Claude Code — a persisted `xhigh` (our `settings.json:333`) silently + carries onto Opus 5 sessions, against Anthropic's "start at high, sweep + low/medium" guidance for that model. + +## 2. Design decisions + +- D1 The global instruction layer must be MODEL-NEUTRAL across the Claude 5 + family (sessions run Fable 5 by default; dispatched judgment agents run + Opus 5; executors Sonnet). Fixes therefore express WHEN-guidance and + outcome bars, not directional compensation for one model's trait. +- D2 Harness-level quality gates (fresh blind verifier + security-auditor, + BDR-049/050; plan-challenge, BDR-075) are architecture, not model + self-check prompting. They stay. E3 applies only to prose that tells the + MODEL to verify its own work. +- D3 Per BDR-021, the Security and Architecture-decisions sections of + CLAUDE.global.md stay verbatim (deliberate policy). No softening there. +- D4 Registries are append-only: LRN-030 is not edited; a new LRN records + the trait inversion and points back to it. +- D5 Deterministic backstops (gitflow pre-commit, Gitea protection, + permissions.deny, rtk pinning) are explicitly out of "more freedom" scope + — community reports show Opus 5 working AROUND soft controls, which argues + for keeping hard ones. + +## 3. Work items + +### W1 — CLAUDE.global.md: rewrite the delegation block (:43-47) +Replace the 5-line block (incl. "Default to delegation for multi-file +exploration. Counters model tendency to under-delegate.") with model-neutral +when-guidance, same footprint (≤5 lines): + +``` +- Sub-agents: one task per sub-agent, main context stays clean. + Delegate genuinely independent, sizeable tracks (wide multi-file + exploration, parallel audits) — not work doable in a few tool + calls, and not self-verification (harness gates own that). Brief + precisely, then commit to the delegation — don't redo its work. +``` +Rationale: E1+E2. No hard spawn cap in prose (harness already injects one on +Opus 5; Fable benefits from delegation). + +### W2 — CLAUDE.global.md: reframe "After code changes" (:75-83) +Keep the concrete quality bar; drop the proof-mandate/self-check phrasing +(E3). Replace steps 2-4 with faithful-outcome reporting: + +``` +## After code changes +1. Run tests, lint, build, type-check if available. +2. Report outcomes faithfully: what passed, what wasn't run, + remaining risks, surviving deviations. Completion claims only + for verified work. +3. Correction or notable event → capitalize to right registry. +``` +Net: -2 lines. "Would staff engineer approve?" bar and "Don't mark complete +without proof" are removed as self-check choreography; honest-reporting +line preserves the intent (grounded completion claims) without mandating an +extra verification pass. + +### W3 — CLAUDE.global.md: add scope fence (Workflow section) +Append (adapted from Anthropic's tested block, caveman-compressed, ~5 lines): + +``` +- Scope: deliver what was asked, at the scope intended. Routine + judgment calls → decide alone; materially different readings → + ask. Better approach spotted → say so in one line, still do the + task as asked. Finish the whole task; genuinely blocked → do the + rest, state plainly what's missing. +``` +Rationale: E4. Complements existing "Scope changes to task — no unrelated +edits" (line ~15) without contradicting it. + +### W4 — CLAUDE.global.md: add deliverable-length rule (Code style / Comments area) +~2 lines: + +``` +- Written deliverables (docs, reports, .md): length matched to what + the task needs — no filler sections, no boilerplate summaries. +``` +Rationale: E6. Registries already covered by caveman rule. + +### W5 — Line budget +After W1-W4: expected ~309 lines. Hard check: `wc -l CLAUDE.global.md` ≤ 320 +(session-start.sh warning threshold at :202-213). + +### W6 — hooks/design-toolchain-reminder.sh: drop `\bui\b` and `\bux\b` +- Remove the two 2-char alternatives from the pattern at :47. Keep + `ui/ux|ux/ui|ui kit` and all other tokens. +- Add a dated header comment (3rd tightening pass, 2026-07-30, cites the + two French-prose `ux` FPs; series LRN-1005/1007). +- Trade-off accepted: a bare "améliore l'ux" prompt with no other design + token goes quiet — the CLAUDE.global.md "Design work" section still + routes it (the hook is a belt, self-described soft nudge). +- Update `lib/tests/design-toolchain-reminder.test.sh`: add 2 quiet rows + (the real FP prompt excerpt; a bare "l'ui" French sentence) — flip-tested + per LRN-096. Existing 9 must-fire rows unaffected (none uses ui/ux). + +### W7 — agents/plan-challenger.md: coverage-first reporting line +Add one clause to the findings rules (add-only, no removal): uncertain or +low-severity findings are REPORTED with an explicit confidence + severity +tag rather than self-censored — severity filtering happens in the +orchestrator's synthesis, not in the challenger. Rationale: E5 (literal +Opus 5 + "manufactured concern is a failure" wording risks suppressing real +low-confidence findings). Must not touch: verdict grammar, MANDATORY PROOF +clause, blind-dispatch rules (test-locked in plan-challenger.test.sh). + +### W8 — Memory + docs capitalization (same branch, follows the work) +- decisions.md: new BDR (config adapted for Claude 5 family — scope, + rationale, alternatives incl. "leave config as-is" and "hard spawn caps" + rejected). +- learnings.md: new LRN — Opus 5 behavioral profile (over-delegation + inverts LRN-030's Opus 4.8 trait; over-verification; literal following; + no effort hold on Opus 5 in Claude Code; heron_brook/#80988 injection). +- journal.md: one line. +- CHANGELOG.md: entry under Unreleased. + +### W9 — Gates (before commit) +- `shellcheck hooks/design-toolchain-reminder.sh` clean. +- Manual flip-test of the hook: FP prompt → quiet; "redesign the navbar" → + fires. +- `make test` full suite green (design-toolchain-reminder.test.sh, + plan-challenger.test.sh, model-routing.test.sh untouched-but-must-pass, + curated-config-guard, loops-light…). +- `wc -l CLAUDE.global.md` ≤ 320. + +### W10 — Gitflow +`bash ~/.claude/lib/gitflow.sh start feature opus5-config-tuning` off +develop; atomic commits (hook+test / CLAUDE.global.md / agent / memory+docs); +NO `gitflow finish` — merge only on explicit human signal. + +## 4. Explicitly NOT doing (considered, rejected) + +- N1 Touching lib/verify-secure-loop.md or the fresh-verifier/security + gates: harness architecture (BDR-049/050, D2), verifies SONNET executor + output — not Opus 5 self-check prose. +- N2 Softening the Security / Architecture sections (BDR-021, D3). +- N3 Editing the superpowers plugin's "1% chance → MUST invoke" language: + external upstream code; flagged as residual over-triggering risk in the + new LRN, revisit as its own decision if observed. +- N4 Changing `settings.json` `effortLevel: "xhigh"`: user preference, + optimal for the Fable 5 session default; the Opus 5 carry-over trap (E9) + is documented in the LRN + surfaced to the user for a manual decision. +- N5 De-prescribing seo-analyzer.md / geo-analyzer.md (1528/1106 lines, + heavy MUST density): separate project, backlog note in TODO.md. +- N6 Removing or session-gating the design/ctx7 reminder hooks: soft + nudges, cheap, deliberately built; tightened only (W6). +- N7 Any model pin change: `model: opus` pins now resolve to Opus 5 — + desired outcome, census (model-routing.test.sh) untouched. +- N8 Committing settings.json for any reason (LRN-098/1049 /model-churn + trap): file is currently clean; keep it out of every commit. + +## 5bis. CHALLENGE SYNTHESIS (2026-07-30) — FINAL amendments (v2) + +Verdicts: correctness CONCERNS(4) · robustness FATAL(5, 1 BLOCKER) · +simplicity CONCERNS(4). Every fix below is the challenger's own named FIX, +adopted as written. No re-challenge pass: scope narrowed, no new dependency; +W0 is an execution-time safety procedure, not a new config mechanism. + +- **W0 (NEW — robustness BLOCKER)**: all edited surfaces are symlink-deployed + LIVE (~/.claude/CLAUDE.md, hooks/, agents/ → this repo); edits take effect + machine-wide at save time, before any W9 gate. Mitigations: + (a) `gitflow start` BEFORE any live-file edit; never checkout develop + mid-work; (b) hook regex change validated on a SCRATCH copy first + (bash -n + shellcheck + pattern replay), then written to the live file in + ONE atomic Edit; (c) named reverts: `git show develop: > `; + escape hatch = remove the hook registration block from settings.json. +- **W1 v2** (robustness#3, correctness#2): replacement text carves out the + mandated gates explicitly and scopes "don't redo": + "Skill-mandated gates (fresh verifier/security/challenge) always dispatch + as written. Don't redo delegated work by hand — failed gates re-dispatch + fresh executors instead." +- **W2 v2** (simplicity#2): minimal diff — delete ONLY the line + `Bar: "would staff engineer approve?"`. Steps 1-4 + capitalize step stay. +- **W3 v2** (simplicity#1, robustness#4): no new bullet. Fold the only new + clause into the existing Deviations bullet: "Finish the whole task: + blocked on an independent sub-part → do the rest, state what's missing. + Gone WRONG → still STOP, re-plan." (net +2 lines, no conflict with :53). +- **W4**: unchanged (+2 lines). Budget v2: 304 +1 −1 +2 +2 = 308 ≤ 320. +- **W6 v2** (all lenses): drop `\bux\b` ONLY — keep `\bui\b` (zero evidenced + FP; one logged true positive). Accepted trade-off: the 2026-07-21 "ameliore + le tutoriel…gamifier" ux row (plausible TP) goes quiet; CLAUDE.global.md + design-routing section remains the router. Header comment notes the log + records `head -1` only → per-token FP rate not fully derivable. Tests: + quiet row = synthetic "changement ux vu…" (verified matches pre-change → + flips); must-fire row = "revois l'ui du panneau admin" (locks `\bui\b`; + apostrophe escaped correctly, doubles as JSON-path control per + robustness#7). No log-excerpt rows (vacuous — 100-char truncation). +- **W7 v2** (all lenses): in-place reword of the `:82-83` sentence (NOT + test-locked; plan v1 misstated that) instead of an add-only clause: + "No invention — ungrounded is noise. Silently dropping a grounded doubt is + equally a failure: file it as `[MINOR]` with the uncertainty stated in + `WHY:`. Nothing real at all → `SOLID` with `FINDINGS: none`." + OUTPUT grammar byte-identical; no confidence axis; no consumer change. + Census: add `has "$A" "grounded doubt"` row to plan-challenger.test.sh in + the same commit. +- **W9 v2**: adds the W0 scratch-validation step; rest unchanged. +- **W10 v2**: branch creation moves FIRST in execution order. + +## 5. Constraints for challengers + +- Registries append-only; curation only via /prune-memory. +- Census tests lock behavior: any hook/agent edit must land with its test + update in the same commit; `make test` must stay green. +- CLAUDE.global.md ≤ 320 lines (runtime warning threshold). +- BDR-021: Security + Architecture sections verbatim. +- Gitflow: feature branch off develop, no merge without human signal. +- The global file serves ALL models (Fable sessions, Opus 5 dispatches, + Sonnet executors read skill/agent prompts instead) — no Opus-5-only + wording in CLAUDE.global.md. diff --git a/CHANGELOG.md b/CHANGELOG.md index e7df2b7..e8769b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,20 @@ Format follows [Keep a Changelog](https://keepachangelog.com/). ## [Unreleased] +### Changed +- **Global instruction layer recalibrated for the Claude 5 family (BDR-081)** — + delegation block is now model-neutral when-guidance (the Opus 4.8 + under-delegation counter inverted on Opus 5, which over-delegates and gets + an injected harness cap); "staff engineer" self-check bar dropped (Opus 5 + over-verification trigger); finish-whole-task clause added to Deviations; + written-deliverable length rule added. 308/320 lines. +- **design-toolchain hook** — dropped `\bux\b` (2 French-prose false + positives; 3rd tightening pass, series LRN-1005/1007); `\bui\b` kept and + locked by a must-fire test row. +- **plan-challenger** — grounded-but-uncertain findings now file as `[MINOR]` + with the uncertainty stated, instead of being self-censored (Opus 5 follows + conservative-reporting clauses literally). + ## [1.4.0] — 2026-07-22 ### Added