From 557e4cc3172f03beca6ec9e08b0e9cd39a7ada89 Mon Sep 17 00:00:00 2001 From: bastien Date: Mon, 28 Sep 2026 20:02:56 +0200 Subject: [PATCH] feat(effort): max at the verify-secure caps and ship-feature 4b; STOP texts suggest /effort-max --- lib/challenge-plan.md | 2 ++ lib/tests/effort-routing.test.sh | 6 ++++++ lib/verify-secure-loop.md | 9 +++++---- skills/ship-feature/SKILL.md | 7 +++++-- 4 files changed, 18 insertions(+), 6 deletions(-) diff --git a/lib/challenge-plan.md b/lib/challenge-plan.md index 91e72ef..66d4cb1 100644 --- a/lib/challenge-plan.md +++ b/lib/challenge-plan.md @@ -59,6 +59,8 @@ silently downgrade the judgment. (The executor gates stay sonnet.) A challenger that returns a malformed/empty verdict, a missing `PROOF`, or dies → retry ONCE with a fresh challenger; a 2nd failure on that lens → STOP and escalate +(the STOP text names the level reached, `$CLAUDE_EFFORT`, and suggests `/effort-max` +for the relaunch; no shift here: a mute challenger is an infrastructure failure) to the human, NAMING the lens. Never carry "plan challenged" into the gate on a silently dropped lens (`verify-secure-loop.md`: "a mute verifier is NEVER a PASS"). diff --git a/lib/tests/effort-routing.test.sh b/lib/tests/effort-routing.test.sh index 6dc7391..536145e 100755 --- a/lib/tests/effort-routing.test.sh +++ b/lib/tests/effort-routing.test.sh @@ -78,6 +78,12 @@ has "lib/effort-shift.md" 'lone Skill call is a no-op' has "lib/effort-shift.md" 're-applies its' [ "$(grep -c 'a lone Skill call is a no-op' "$R/skills/feat/SKILL.md")" -ge 1 ] && ok || ko "feat INC line must carry the pairing rule" +# ── 7) escalation at max (spec D4) +[ "$(grep -c 'Skill(effort-max)' "$R/lib/verify-secure-loop.md")" -eq 3 ] && ok || ko "verify-secure-loop.md must shift to max at its 3 caps" +has "skills/ship-feature/SKILL.md" 'Skill(effort-max)' +has "lib/challenge-plan.md" '/effort-max' +has "lib/verify-secure-loop.md" '/effort-max' + # ── summary (later tasks insert their locks ABOVE this line) printf 'effort-routing census: %d pass, %d fail\n' "$pass" "$fail" [ "$fail" -eq 0 ] diff --git a/lib/verify-secure-loop.md b/lib/verify-secure-loop.md index 5bb4fd1..2540670 100644 --- a/lib/verify-secure-loop.md +++ b/lib/verify-secure-loop.md @@ -35,7 +35,7 @@ single `GATES — VERDICT:` line: - `UNMET(n)` → hand the dev the CONTRACT path + the `NOT-MET` rows verbatim, nothing else; re-run GATE 0. **No verifier is dispatched** — a red build or a red suite is not a judgement call, and paying an LLM to discover it is - waste. **Max 3 floor iterations** → STOP + human escalation with the rows. + waste. **Max 3 floor iterations** → `Skill(effort-max)` (effort-shift: cap reached, diagnose at max before escalating; send it in the same message as the first tool call that gathers the escalation evidence), then STOP + human escalation with the rows. - `ABANDONED(n)` → floor green but a handoff stands. Continue to GATE 1; the verifier surfaces it and its `ABANDONED(n)` verdict routes to the human gate. @@ -74,7 +74,7 @@ Parse its single `VERIFY — VERDICT:` line: lines (NOT-MET / out-of-scope), nothing else: re-dispatch a FRESH executor with those inputs only, never redo the fix by hand. Then re-run GATE 0 and re-dispatch a FRESH verifier. Repeat. - **Max 3 conformity iterations** → STOP + human escalation with the + **Max 3 conformity iterations** → `Skill(effort-max)` (effort-shift: cap reached, diagnose at max before escalating; send it in the same message as the first tool call that gathers the escalation evidence), then STOP + human escalation with the CRITERIA table (the contract-vs-realized diff). - `ABANDONED(n)` → direct human gate, never a dev loop (a dev cannot close what was proven impossible). The human lifts the abandonment or accepts @@ -104,8 +104,9 @@ Parse its single `SECURITY — VERDICT:` line: (re-dispatch a FRESH executor, never fix by hand). Then re-run GATE 0, then **re-verify the REQUEST first** (GATE 1, fresh verifier) — a security fix can drift the behavior — **then re-run GATE 2** (fresh auditor), in that - order. **Max 3 security iterations** → STOP + human escalation with the - BLOCKING table. + order. **Max 3 security iterations** → `Skill(effort-max)` (effort-shift: cap reached, diagnose at max before escalating; send it in the same message as the first tool call that gathers the escalation evidence), then STOP + human escalation with the + BLOCKING table. Every STOP text names the level reached (`$CLAUDE_EFFORT`) + and suggests `/effort-max` for the relaunch. - `DEGRADED` (semgrep absent) → does NOT block on the tool's absence; surface the checklist result + recommend `make plugin`. A DEGRADED run that still BLOCKs (grep-caught secret/injection) blocks like any other. diff --git a/skills/ship-feature/SKILL.md b/skills/ship-feature/SKILL.md index fb8e3a6..693d724 100644 --- a/skills/ship-feature/SKILL.md +++ b/skills/ship-feature/SKILL.md @@ -191,7 +191,8 @@ this loop. ## STEP 4b — ERROR RECOVERY (if STEP 4 fails) If a subagent returns a build error, failing test, or type error: -1. Load `$HOME/.claude/agents/analyzer.md` in DEBUG MODE on the exact error output. +1. `Skill(effort-max)` (effort-shift: error recovery; send it in the same message as the Read of the analyzer file below), then load + `$HOME/.claude/agents/analyzer.md` in DEBUG MODE on the exact error output. Produce: root cause hypotheses (ordered), affected files, what NOT to touch. 2. Present gate: ``` @@ -207,8 +208,10 @@ OPTIONS : C) Abort feature — preserve work done so far ``` 3. Wait for user choice. Do NOT auto-fix. Do NOT proceed without explicit approval. -4. If A → apply minimal fix, re-run STEP 4 for the failed task only. Max 2 retry attempts. +4. On resume the turn is at the session level (effort-shift: turn reset). + If A → `Skill(effort-medium)` sent with the re-dispatch, apply minimal fix, re-run STEP 4 for the failed task only. Max 2 retry attempts. If still failing after 2 → fall back to options B or C. + If B or C → `Skill(effort-xhigh)` first, sent with the next tool call. If B → before skipping: scan remaining task list for tasks that depend on the failed task (look for references to the same file or function in subsequent tasks). If dependents found → present: "Tasks [N, M] depend on the skipped task.