diff --git a/.claude/memory/blockers.md b/.claude/memory/blockers.md index b0d518c..9f0bc07 100644 --- a/.claude/memory/blockers.md +++ b/.claude/memory/blockers.md @@ -36,6 +36,7 @@ rules: | BLK-014 | 2026-07-01 | `make install` aborts npm EEXIST on `~/.local/bin/claude` when claude already installed via native installer — no presence guard | resolved | | BLK-015 | 2026-07-03 | `gitflow_finish` ignored its ` ` args → merged the CHECKED-OUT branch not the one named → wrong-branch merge (audit LOT3) | resolved | | BLK-016 | 2026-07-04 | rtk compression PATH-dead 30 days — 6/5070 Bash commands compressed (~460K tokens missed); installer sources cargo env so its own check passes, Claude tool shell never gets ~/.cargo/bin | resolved | +| BLK-017 | 2026-07-17 | Bing Webmaster API unusable for a multi-client agency: OAuth swamp (localhost redirect refused, rotated single-use refresh tokens race our parallel dispatch), API key = wrong model (client-owned sites) | open/deferred | --- @@ -201,3 +202,9 @@ rules: - **Status**: resolved. - **Reference**: lesson: a PATH-dependent hook must be verified in the TARGET shell, not the installer's (installer sourcing envs lies to its own checks); usage is MEASURED (`rtk discover`), never assumed. Corroborates [[LRN-047]] (silent degradation → measure) + [[LRN-036]] (hand-managed profile drift); guard interplay [[LRN-089]]-adjacent (ambient-state assumptions). - **backmerge**: entry from release/1.0.0 (2b4e7401); the fix `e58037c` was ALSO missing from develop (rtk was live-broken on develop) — ported to develop 2026-07-08 (review remediation A3, commit follows) so this "resolved" is now true on develop too. + +## BLK-017 — Bing Webmaster API unusable for a multi-client agency (W2 deferred) — 2026-07-17 +- **Friction**: W2 (`bing` verb — free Bing query stats + index status + first-party backlinks) abandoned after 4 challenge rounds. User's model = client sites live on CLIENT Bing accounts. +- **Real cause**: two viable-looking paths, both dead. (API KEY) is per-user not per-site (docs), but IS the account identity → one key per client account, exactly what the user feared; non-scoped, no expiry, passed in query string. (OAuth) is the right delegation model (like GSC) but a swamp: Redirect URI rejects ALL local forms (http/https/127.0.0.1 — user-tested); refresh tokens are ROTATED + single-use, self-described non-compliant with OAuth 2.0 → store rewrite every call, AND our parallel seo‖geo dispatch would race the rotation → `invalid_grant` + dead token; undocumented "Could not extract expected anti-forgery token" on refresh, unanswered on MS Q&A; docs contradict themselves on grant_type + token endpoint; no library. MS's own advisor recommends falling back to the API key. +- **Verified live**: the Webmaster API itself is ALIVE (`GetUserSites?apikey=INVALID` → HTTP 400 `{"ErrorCode":3,"Message":"InvalidApiKey"}`, 0.4s) — distinct from Bing SEARCH API (retired 2025-08-11). So the block is auth/model, not availability. +- **Status**: open/deferred. REVIVAL: a client already on Bing adds the user as Read-Only → test in ~10 min whether one API key sees DELEGATED sites (undocumented, nobody knows). If yes → W2 is cheap+clean (one key, client-owned verification, revocable, read-only, zero OAuth). Value RAISED by [[BDR-071]]: GetUrlLinks is now the only free viable backlink source (first-party only). diff --git a/.claude/memory/decisions.md b/.claude/memory/decisions.md index a00adaf..97e8f1f 100644 --- a/.claude/memory/decisions.md +++ b/.claude/memory/decisions.md @@ -87,6 +87,10 @@ rules: | BDR-064 | 2026-07-14 | global memory split: repo file → CLAUDE.global.md (deployed name unchanged), CLAUDE.md freed for project scope; consumer/maintainer wording rule | accepted | | BDR-065 | 2026-07-14 | transient planning artifacts (superpowers spec/plan): committed during run, deleted post-merge; git history = archive; codified in project CLAUDE.md | accepted | | BDR-066 | 2026-07-15 | Model routing: reflection inline (session big model) + sonnet-pinned executors + blocking gate | accepted | +| BDR-070 | 2026-07-17 | claude-seo: cherry-pick scripts into our tree, never install; /seo stays sole entry | accepted | +| BDR-071 | 2026-07-17 | No viable free backlink source → Off-page axis stays brand-mentions-only (FINAL, not placeholder) | accepted | +| BDR-072 | 2026-07-17 | SPA: honest refuse (On-page N/A, not zero), no headless browser (R2 over R1) | accepted | +| BDR-073 | 2026-07-17 | Scoring: LLM judges findings+severity, engine does the arithmetic (deterministic /20) | accepted | --- @@ -1016,3 +1020,29 @@ rules: - **Alternatives rejected**: (B) narrow glob → weakens `.env.production`; blocked by auto-mode classifier as unauthorized self-modification ([[EVAL-024]]). (C) rename → `env.example` sidesteps glob at zero security cost, but ~30 refs (scaffolder, doc-syncer, init-project, deploy, 3 archetypes, link.sh, install-plugins.sh, toggle-external.sh) + repo's own root `.env.example` + seo-data.test.sh + gitignore `!.env.example` (BDR-030) → refactor, user declined. - **Files**: settings.json, templates/settings/SETTINGS.md (taught the broken `Write()` pattern → fixed at source so /onboard stops propagating it). - **Status**: implemented on chore/fix-inert-write-deny-rules (07ca738), UNMERGED (human gate). + +## BDR-070 — claude-seo (github.com/AgriciDaniel): cherry-pick, never install — 2026-07-17 +- **Decision**: adapt useful scripts into our tree, /seo stays sole entry. Do NOT run install.sh / plugin install. +- **Why**: their CODE is real (326 tests, render_page.py 428l Playwright, url_safety.py 622l SSRF) — their INSTALLERS destroy our work. install.sh:49 `cp -r skills/seo/*` overwrites our SKILL.md. uninstall.sh:45 globs `~/.claude/agents/seo-*.md` → deletes our seo-analyzer.md (42K) it never installed (verified dry-run). extensions/*/install.sh:42 replaces settings.json with `{"env":{...}}` on parse error. skills/seo/SKILL.md:119 injects Skool upsell footer into deliverables (leaks to /client-handover client PDFs). hooks.json registers global PostToolUse exit-2 → blocks our dispatcher mid-bundle. +- **Alternatives rejected**: (plugin install) → both `/seo` coexist namespaced → non-deterministic dispatch, silently loses our FR-legal axis on an unpredictable fraction of runs. (install nothing) → forgoes render_page/url_safety/unlighthouse we lack. +- **Verdict on parity**: their README lies (dual JSON-LD validator = 2 hyperlinks, zero `.py` calls; "zero-network"/"fully offline" false). Our system is more honest; we keep FR-legal (their whole repo: 2 hits), fix-bundle+ownership, trajectory-17/20, NAP anti-seed. +- **Files**: none installed. Findings drove the whole seo-geo-integrity branch (21 commits). + +## BDR-071 — no viable free backlink source: Off-page axis stays brand-mentions-only — 2026-07-17 +- **Decision**: I1's narrowed Off-page axis (brand mentions from STEP 6 only, backlinks+authority declared §14-unauditable) is the FINAL state, not a placeholder awaiting data. +- **Why**: measured, not assumed. GSC has no links endpoint (API = Search Analytics/Sitemaps/Sites/URL-Inspection only; links report UI-only). Common Crawl hyperlinkgraph domain-edges = **17.3 GB gzipped** (+879MB vertices, +2.3GB ranks), HEAD-measured live. Scanning it per-audit is non-viable + abusive to a nonprofit. The reference impl (claude-seo commoncrawl_graph.py:169) caps download at 500 MiB = **2.9% of edges**, sorted by source ID → arbitrary slice reported as a backlink profile, "70/100 health". A random sample dressed as a measurement — the exact failure class the branch removes. +- **Consequence**: B1/B2/B3 all killed. Weight (10-15%) unchanged — re-deriving for an axis that won't widen churns historical scores for nothing. +- **Only free viable source**: Bing GetUrlLinks — first-party only (never a competitor), blocked on client's Bing account → raises W2's value ([[BLK-017]]), does not unblock it. + +## BDR-072 — SPA: honest refuse, no headless browser (R2 chosen over R1) — 2026-07-17 +- **Decision**: rendercheck verdict `client-rendered` → On-page axis N/A, excluded from weighted global, NEVER scored zero. No Playwright, no Chromium. User-arbitrated. +- **Why**: a zero says "your on-page is bad"; N/A says "we couldn't see it" — only one is true, and /client-handover gates on 17/20. curl on a shell returns "missing" for every meta/H1/JSON-LD → a page of FALSE findings + a bundle that "fixes" tags that already exist. STEP 2 recorded `RENDERING: SPA` since forever and NOTHING acted on it. Verdict from what the server SENT (package.json can't tell React-SPA from Next-SSR). +- **GEO angle (sharper)**: AI crawlers (GPTBot/PerplexityBot/ClaudeBot) are WORSE at JS than Googlebot — fetch HTML, largely don't execute. A client-rendered site is near-invisible to the engines the audit serves → §0 alert + SSR/SSG top user action, aligns CLAUDE.global "public sites never SPA". +- **Alternatives rejected**: R1 Playwright (~300MB Chromium, breaks bash+curl purity) — user chose refusal. Refusing IS the finding. +- **Files**: lib/seo-data/render_check.py, seo/geo STEP-5 gates (20d3082). + +## BDR-073 — deterministic scoring: split LLM judgement from arithmetic — 2026-07-17 +- **Decision**: LLM emits WHICH findings + severity (irreducible judgement); engine computes the /20. Reuses /harden's scale (-15/-8/-3/-1, clamp, /5 into /20) → one vocabulary across the family. +- **Why**: /harden had a real scale (SKILL.md:435), /seo had NONE → every axis felt → two runs over identical code diverged, while /client-handover gates on 17/20. H2 sharpened it: once drift reports real change, a self-moving score is visibly noise. Same principle as engine-side cannibalisation grouping — never hand a model 1000 rows to add. +- **Makes computable (was prose)**: "N/A is not a zero" (R2 on-page, I1 off-page) → axis excluded + weights renormalised, verified all-20 with 2 N/A → global 20.0. Prevalence: affected/sampled shift severity ONE step (≥50% escalate, single de-escalate). +- **Files**: lib/seo-data/score.py (4818c61). diff --git a/.claude/memory/evals.md b/.claude/memory/evals.md index daa349e..73a677b 100644 --- a/.claude/memory/evals.md +++ b/.claude/memory/evals.md @@ -36,6 +36,7 @@ rules: | EVAL-013 | 2026-06-30 | /reconcile real-usage on live repo: known gap + 2 unanticipated (header-marker drift class) + false-positive rejected off-fixture, 0 false assertion | keep | | EVAL-018 | 2026-07-06 | job3 docs-drift audit + execution: 46/46 findings verified, 20/23 fixes shipped (B1 blocked, D2-D5+B6 skipped by decision), zero residual on re-sweep | keep | | EVAL-019 | 2026-07-06 | job4 test-gap audit + execution: 11 specs + 5 fixes/seams, every mutation red-green verified, zero residual | keep | +| EVAL-025 | 2026-07-17 | opening seo/geo inventory (subagents): 7/7 verifiable claims false or overstated; real contact corrected all, 6 plan corrections + 4 features killed at measurement | keep | --- @@ -241,3 +242,9 @@ rules: - **A2 (tooling, FALSE POSITIVE)**: security-guidance automated review flagged the same file, HIGH "Agent/Subprocess Permission Bypass", fix = restore the inert `Write()` rules. Wrong — would re-introduce the bug + the 15 startup warnings. Pattern-matched "deny line removed = bypass" with zero knowledge of rule-matching semantics. Rejected with doc citations. - **A3 (subagent, caught)**: claude-code-guide asserted `**/*.lock` matches `package-lock.json`. False (ends `.json`). Caught on read → `package-lock.json`/`pnpm-lock.yaml`/`go.sum` got explicit rules. Don't trust delegated glob reasoning. - **action**: keep — fix landed (07ca738), weakening reverted. Lesson: vague delegation ("je te laisse en juger") authorizes ADDING protection, never REMOVING it; a boundary-loosening edit needs its own explicit ask, doubly so when the boundary is mine. Guardrail signal: the deterministic classifier beat both the LLM reviewer (A2 false pos) and me (A1) — keep it loud. Linked to [[BDR-069]], [[LRN-130]]. + +## EVAL-025 — opening seo/geo inventory (subagent-produced) that founded the 20-point plan — 2026-07-17 +- **output**: the inventory + claude-seo comparison report from 3 Explore subagents, on which the entire seo-geo-integrity plan was built. +- **method**: each verifiable claim confronted DURING execution with a primary source or a live test — CrUX API metric list, web.dev, Search Console API reference, HEAD on data.commoncrawl.org, real curl on 2 live sites (zenquality Astro, lavageangels356 native PHP), 2 real repos. +- **anomalies**: 7/7 of the verifiable claims were false or overstated (VSI exists / Off-page zero-data / stats drive weights / GSC Links API / SPA §0 flag / Twitter 403 / Common Crawl viable). 6 plan corrections mid-execution: I1 over-correction, I6 wrong framing, W1 wrong shape (verb vs extend), C1a false premise (grep already skips gitignore), C1b needless guard, B1 non-viable at 17.3 GB. The REAL corrected every time; re-reading the spec never did. +- **action**: keep — see [[LRN-132]]. 4 features killed at measurement (B1/B2/B3 + W2 deferred) beat 4 false-signal features. The most trustworthy output of the session was the code NOT written. Method that worked: show/measure the real artifact before deciding, mirroring [[LRN-074]]'s watch-the-RED discipline applied to a plan. diff --git a/.claude/memory/journal.md b/.claude/memory/journal.md index 690b9ee..7461677 100644 --- a/.claude/memory/journal.md +++ b/.claude/memory/journal.md @@ -394,3 +394,8 @@ rules: - FIRST PUBLIC RELEASE **v1.0.0** (BDR-067). Versioning RESET: internal v1-4 → pre-release history, public launch = 1.0.0 (override "never restart at v1.0.0" — deliberate public reset = sanctioned exception; NEXT release continues from 1.0.0, not 4.x). Deleted v4.0.0 tag + a STALE abandoned release/1.0.0 branch (July-4 attempt, 227 behind; `git cherry` confirmed nothing orphaned — all real work already in develop). Cut fresh from develop. PUSHED: origin main=dc4f78b, develop=6c23d6f, sole tag v1.0.0. User flips Gitea repo visibility to public separately. Prep done manually (backward version + CHANGELOG restructure beyond the forward-only sonnet release-executor). - /close ritual: LRN-128 (version reset = editorial, not the forward-only executor) + LRN-129 (git cherry proves nothing orphaned before a branch delete) + EVAL-023 (post-merge ronde on the model-routing refactor — clean, 5 edges fixed) capitalized; checked 1 TODO done (Gitea public, user-confirmed). BDR-066/067 + LRN-125/126/127 already logged inline this session (dropped as dup). Index drift (learnings 118-129, evals 020-023) flagged for /prune-memory. - BDR-068 (close-auto-persist) MERGED to develop + pushed. Then cut + pushed **v1.1.0** (minor, that feature). Standard forward bump → sonnet release-executor ran BOTH spans (prep + finish+tag); lineage continued 1.0.0→1.1.0 not 5.x (validates [[BDR-067]]). origin: main=2f8dc6b, develop=21b1e21, tags v1.0.0 + v1.1.0. WATCH-ITEM: a stale local tag `v4.0.0` reappeared during the release — NOT from origin (origin never regained it; `push.followTags` off; its commit unreachable from develop/main). Inert (push targeted main/develop/v1.1.0 explicitly + deleted the local copy; origin verified clean). Mechanism unexplained — if `v4.0.0` resurfaces locally after a `gitflow` op, trace the release lib (gitflow.sh / release-executor) for stray tag re-creation. + +## 2026-07-17 +- seo/geo parity vs github.com/AgriciDaniel/claude-seo (11.5k★, MIT): full 20-point plan built from a 3-subagent inventory, then executed. Verdict cherry-pick-never-install ([[BDR-070]]). 21 commits: Phase 1 (I1-I8 integrity, markdown specs) MERGED to develop (02c7a6f, 8 commits); Phases 2-7 on bugfix/seo-geo-integrity UNMERGED (13 commits, human gate). `fetch.sh` 5→11 verbs (richresults via inspect, sitemap, rendercheck, linkgraph, cannibal, drift, score); seo-data test suite 85→167 pass, 0 fail. Dogfooded on 2 live sites (zenquality Astro + lavageangels356 native PHP) — the second caught 2 bugs Astro hid (image:loc counted as page, flat-URL family heuristic). +- 4 features KILLED at measurement, not built: B1/B2 (Common Crawl edges = 17.3 GB, ref impl reads 2.9% and calls it a profile — [[BDR-071]]), B3 (GSC Links API doesn't exist), W2 (Bing OAuth swamp — [[BLK-017]]). 30/70 similarity refused (needs content extraction), Playwright refused (R2 [[BDR-072]]), defusedxml refused (DTD-reject keeps stdlib-only). The most trustworthy output was the code NOT written ([[EVAL-025]]). +- BDR-070/071/072/073 + LRN-131/132/133 + BLK-017 + EVAL-025 capitalized; checked 14 TODO done (I1-I5,W1,W3,C1-C3,B3,R2,H1,H2), W2+R1 left unchecked (deferred/rejected). 2 learnings dropped as dup of [[LRN-074]] (grep/find gitignore + detector-proof). Red thread [[LRN-133]]: an omission must stay legible. Verification discipline [[LRN-131]]/[[LRN-132]]: WebSearch ≠ verification, subagent summary = claim not fact (7 disproven, 3 self-reproduced). diff --git a/.claude/memory/learnings.md b/.claude/memory/learnings.md index 406f893..628c5b8 100644 --- a/.claude/memory/learnings.md +++ b/.claude/memory/learnings.md @@ -133,6 +133,9 @@ rules: | LRN-115 | 2026-07-08 | analyzer Edit/Write grants (seo/geo/validator) are NOT dead: needed to write the REPORT (VALIDATE/SEO/GEO.md); the "never edit" rule targets CODE, instruction-level (same as the patron) — verified false-positive | do NOT re-flag as a tool-grant defect; a report-only agent keeps Write for its own report | | LRN-116 | 2026-07-08 | memory backfill release→develop: a BLK marked "resolved" can have its RESOLUTION (code) missing from develop — BLK-016 resolved on release but rtk fix e58037c never back-merged → bug LIVE on develop | before backfilling a resolved blocker: verify the fix CODE is on the target branch, not just the registry entry | | LRN-117 | 2026-07-08 | a release/develop fork silently orphans FUNCTIONAL code on develop, not just memory — RC soak fixes (find-skills, make-update TTY, rtk version-guard) lived only on release for the fork's duration; the review's memory back-merge caught only ~half | at release-finish/reconcile: list develop..release commits touching non-registry code (excl. merges/version) for back-merge review — a registry-gap check alone misses code | +| LRN-131 | 2026-07-17 | WebSearch is NOT verification for a number — SEO blogs cross-cite into fake consensus; require primary source + `measured:` field | any stat headed for a client report; verifying a metric/claim exists | +| LRN-132 | 2026-07-17 | a subagent summary is a CLAIM, not a fact — 7 disproven in one session (incl. 3 I reproduced writing the fixes) | before planning on any relayed finding; verify vs primary source / live test first | +| LRN-133 | 2026-07-17 | an omission must stay LEGIBLE, never silent — tool that can't measure says so in its output | designing any audit/measure output; deciding what a cap/refusal/N-A emits | --- @@ -1283,3 +1286,18 @@ rules: - **Also**: `Write(path)` never matches file perms; `Edit(path)` covers ALL file-editing tools (`:242`; `:244` prescribes it). Startup warns on `Write(glob)` — but does NOT warn on a dead `allow` under a `deny`. - **Also**: `Read` deny hits Grep + Glob too (`:242`). Bash NOT covered — `Bash(cat .env)` bypasses `Read(**/.env)` unless separately denied. - **Applied**: [[BDR-069]]. + +## LRN-131 — WebSearch is not verification for a number; require a primary source — 2026-07-17 +- **pattern**: a statistic reaches a client only with ` — — measured: — `. The `measured:` field is what catches the error. +- **context**: "VSI (Visual Stability Index) — new 2026 Core Web Vital" lived in seo-analyzer as a threshold, stated as fact. It does NOT exist — absent from the CrUX API metric list AND web.dev; 10 SEO blogs cross-cited it into apparent consensus, several falsely claiming CrUX already collected it. And EVERY stat in agents/resources/ was real but grafted onto the wrong subject: Aggarwal 40% = ALL methods (pinned on "add stats"); AccuraCast 58.9% = Person-schema PREVALENCE (pinned on QAPage lift, meaning inverted — FAQPage was 1.8%); LLMrefs 3x = brand-mentions-vs-backlinks (pinned on freshness decay). +- **future**: the failure mode is plausible RECOMBINATION — what a model half-remembering a search produces. The old rule "cross-check via WebSearch" LAUNDERS the blog consensus instead of catching it. An API's metric list (e.g. developer.chrome.com/docs/crux) is decisive: a metric the API can't return is one you can't score. See [[LRN-132]] (same family, subagent summaries). + +## LRN-132 — a subagent summary is a claim, not a fact — verify before planning on it — 2026-07-17 +- **pattern**: relaying a subagent's characterisation without checking it propagates plausible-but-false. Treat every relayed finding as a claim to verify against a primary source or a live test. +- **context**: 7 disproven in one seo/geo session — "Off-page has ZERO data" (brand mentions ARE gathered, STEP 6); "the stats drive axis weights" (weight tables carry no citations); "GSC Links API is available" (endpoint doesn't exist); "a SPA-severely-limited §0 flag compensates" (never existed); "X/Twitter returns 403" (returns 200, live-tested); Common Crawl "nearest free source" (17.3 GB dead end); the whole opening inventory that founded the 20-point plan. +- **future**: I reproduced the SAME error 3× while WRITING the fixes (X/Twitter 403 in W3, the two above in I1/I6). Contact with the REAL corrected it every time — the sitemap, the repo, the curl, the primary doc — never re-reading the spec. Measure-first before building. Corroborates [[LRN-074]] (watch the RED go red). + +## LRN-133 — an omission must stay legible, never silent — 2026-07-17 +- **pattern**: when a tool cannot measure something, it says so IN its output — a caller must never read absence as "fine". +- **context**: red thread of 21 commits — NAP with no canonical → finding WITHOUT direction (never pick from source majority); unmeasured backlinks → mandatory §14 line; sample → mandatory COVERAGE ratio; dropped security headers → §14 + "run /harden" pointer; capped crawl → `orphans_withheld` (the cap doesn't degrade the result, it INVALIDATES it — a partial-crawl orphan is a false orphan); SPA → refuse, don't score; N/A ≠ zero in the scorer. +- **future**: the system already HAD the invariant (code-ceiling, §14 Annexe) but applied it in spots. Generalised it. A false signal is worse than a declared gap — the 4 features KILLED at measurement (B1/B2/B3/W2) beat 4 false-signal features. See [[LRN-131]]/[[LRN-132]] (same session, the verification discipline that feeds it). diff --git a/.claude/tasks/TODO.md b/.claude/tasks/TODO.md index 2f78ca7..7645b4a 100644 --- a/.claude/tasks/TODO.md +++ b/.claude/tasks/TODO.md @@ -1,6 +1,76 @@ # TODO -## 2026-07-16 — PLAN seo/geo parity vs claude-seo (not started, awaiting arbitrage) +## 2026-07-17 — STATUS seo/geo parity (branch bugfix/seo-geo-integrity, 10 commits, UNMERGED) +PHASE 1 — integrity: **DONE 7/7**. I3 8b0c98c · I1 57c67f2 · I2 4ea2fb8 · +I5 64f175f · I4 e70e1d6 · I6 9da1dec · I8 acd452b. Plus 9cd7b51 (A1+A2, two +process anomalies surfaced by dogfooding /harden at zenquality.fr from the +wrong CWD). +PHASE 2 — free wins: W3 fe93b79 · W1 a6d423b · **W2 DEFERRED** (see below). +NEXT: H1 (SSRF/injection guard) → C1 (sitemap crawl). Human merge gate: all +10 commits await review; nothing merged to develop. + +### Plan corrections made while executing (the plan was wrong 4×) +- **B3 KILLED** — GSC Links API does not exist. Verified against the API + reference: Search Console v1 exposes exactly Search Analytics, Sitemaps, + Sites, URL Inspection. A subagent hallucinated it; I doubted it in the + plan and the doubt was right. (Its follow-on — "so Common Crawl is the + only free source, and the 70/100 cap is mandatory" — was ALSO wrong: see + B1/B2 KILLED below. Common Crawl is a 17 GB dead end, and Bing's + GetUrlLinks is the only viable free source, first-party only.) +- **I1 was an over-correction** — "Off-page has ZERO data" was overstated + (relayed from a subagent, unverified). Brand mentions ARE gathered + (STEP 6). Narrowed the axis definition instead of N/A-ing it; weights + untouched to avoid churning historical scores twice. +- **I6 framing was wrong** — I claimed 3× that the stats "drive axis + weights". They do not; weight tables carry no citations. They drive Tier + recommendations and, worse, land in CLIENT reports via the "Cite sources" + rule. Reality was worse than my false version. +- **W1 was the wrong shape** — plan said "richresults verb"; a new verb + means a 2nd POST to the same endpoint for a payload already received. + Extended inspect() instead. +- **H1 moved up** (was AXE 5) — it is a PREREQUISITE of C1, not a + follow-up. Today only $DOMAIN (user-typed) is interpolated. After C1, N + URLs from a REMOTE sitemap flow into shell commands and fetch targets. + +### B1/B2 (Common Crawl backlinks) — KILLED 2026-07-17, measured not assumed +The plan said Common Crawl was the free backlink source and the 70/100 cap +was therefore mandatory. Both premises are dead: +- domain-edges.txt.gz = **17.3 GB gzipped** (+879 MB vertices, +2.3 GB + ranks), measured live via HEAD. Finding one domain's inbound links means + scanning all of it, per audit. Non-viable, and abusive toward a nonprofit. +- The implementation everyone cites (claude-seo commoncrawl_graph.py:169) + caps at `500 MiB` = **2.9% of the edges file**, and reports what that + arbitrary slice held as a backlink profile. A random sample presented as a + measurement — the exact failure class this branch exists to remove. We + nearly copied it. +- B2 dies with B1: nothing to cap. +CONSEQUENCE: I1's narrowed Off-page axis (brand mentions only, backlinks + +authority declared unauditable in §14) is the FINAL state, not a placeholder. +Its §14 line was corrected — it used to point at Common Crawl as "nearest +free source", which is a 17 GB dead end. +RAISES W2's VALUE: Bing's GetUrlLinks is now the ONLY free viable backlink +source. First-party only (never a competitor), still blocked on the client's +Bing account. + +### W2 (Bing) — DEFERRED, blocked on a real-world test +Killed after 4 challenge rounds. User's model: client sites live on CLIENT +Bing accounts, so a per-user API key means one key per client account. +OAuth is the right model but is a swamp: +- Redirect URI rejects ALL local forms (http/https/127.0.0.1 — user tested) +- Refresh tokens are **rotated + single-use**, self-described non-compliant + with OAuth 2.0 → store rewrite on every call, AND our parallel + seo/geo dispatch would race the rotation → invalid_grant, dead token +- Undocumented "anti-forgery token" failure on refresh, unanswered on Q&A +- MS's own advisor recommends falling back to the API key +- Doc contradicts itself on grant_type and the token endpoint; no library +REVIVAL CONDITION: a client already on Bing adds the user as a Read-Only +user → test in ~10 min whether the single API key sees DELEGATED sites +(undocumented, nobody knows). If yes → W2 is cheap and clean (one key, +client-owned verification, revocable, read-only, zero OAuth). If no → dead. +Value forgone meanwhile: Bing/DDG/Ecosia query stats + index status + +first-party backlinks. Real but modest; C1 dwarfs it. + +## 2026-07-16 — PLAN seo/geo parity vs claude-seo (superseded by the STATUS above) Source: audit of github.com/AgriciDaniel/claude-seo (11.5k★, MIT, v2.2.0, 5 mo old, 185/197 commits single author). Verdict: cherry-pick, never install (install.sh:49 overwrites our skills/seo/; uninstall.sh:45 glob `seo-*.md` @@ -18,26 +88,26 @@ Seam: `lib/seo-data/fetch.sh` verbs (accounts|crux|queries|inspect|forget) as NEW VERBS. No new architecture. ### AXE 0 — Integrity (no new deps, hours) — the score currently lies -- [ ] I1 Off-page axis scores 10-15% of FULL with ZERO data source (no API, +- [x] I1 Off-page axis scores 10-15% of FULL with ZERO data source (no API, no index) → today fabricated, and it feeds /client-handover. Immediate fix: extend existing LOCAL `N/A — requires FULL audit` pattern to FULL, redistribute weights. Data upgrade later (AXE 3). Honesty now, data after. -- [ ] I2 VSI (Visual Stability Index) listed in CWV thresholds but NO path +- [x] I2 VSI (Visual Stability Index) listed in CWV thresholds but NO path retrieves it — neither CrUX nor PSI expose it. Phantom signal → remove or source. -- [ ] I3 **SAFETY** /geo standalone: geo/SKILL.md (125 l) has no STEP 0, no +- [x] I3 **SAFETY** /geo standalone: geo/SKILL.md (125 l) has no STEP 0, no confirmed-NAP collection — but geo-analyzer OWNS JSON-LD NAP. Standalone /geo on a local business can write unverified NAP with zero LRN-032 protection. Real bug, not cosmetic. -- [ ] I4 Security headers counted 3× (seo-analyzer STEP 4 scores them in +- [x] I4 Security headers counted 3× (seo-analyzer STEP 4 scores them in Technical axis; depth-matrix.md says drop unless indexability; /harden re-audits /100 with 3 validators). Contradiction between dedup rule and agent spec → pick one owner. -- [ ] I5 Report says "audit", measured 5-15 sampled pages. State coverage % +- [x] I5 Report says "audit", measured 5-15 sampled pages. State coverage % explicitly in §0 until AXE 2 lands. ### AXE 1 — Free wins on auth we ALREADY have (fetch.sh verbs) -- [ ] W1 `richresults` verb — GSC URL Inspection already returns +- [x] W1 `richresults` verb — GSC URL Inspection already returns `richResultsResult`; our OAuth already carries the scope. Programmatic rich-results validation on real Google data. **BEATS claude-seo**: their README:314 "dual validator (Rich Results Test + Markup Validator)" is @@ -46,28 +116,29 @@ as NEW VERBS. No new architecture. - [ ] W2 `bing` verb — Bing Webmaster API, free. Closes the Google/Bing asymmetry (Google = full OAuth layer, Bing = manual checklist) while /geo targets ChatGPT Search, which indexes via Bing. Strategic, not cosmetic. -- [ ] W3 `sameas` resolution check — trivial curl loop. entity-seo.md lists +- [x] W3 `sameas` resolution check — trivial curl loop. entity-seo.md lists "sameAs pointing to dead profiles" as a known error class and never checks it. ~10 lines. ### AXE 2 — Coverage (biggest lever: ~97% of a 500-page site unseen today) -- [ ] C1 `crawl` verb — sitemap-driven URL discovery (we ALREADY fetch +- [x] C1 `crawl` verb — sitemap-driven URL discovery (we ALREADY fetch sitemap.xml) + deterministic sampling + coverage % reported. No Chromium, no paid API. Turns "5-15 LLM-chosen pages" into measured coverage. Tradeoff vs claude-seo's link-following 500-page crawl: cheaper, but misses unlinked/unsitemapped pages — accept + disclose. -- [ ] C2 Dupe/cannibalization detection — becomes possible once N pages in +- [x] C2 Dupe/cannibalization detection — becomes possible once N pages in hand: compare titles/H1/canonicals across the set. Free, unblocked by C1. -- [ ] C3 Internal-link graph — orphan pages + 3-click depth are TODAY stated +- [x] C3 Internal-link graph — orphan pages + 3-click depth are TODAY stated as checks with no command to compute them. C1 unblocks real computation. -### AXE 3 — Off-page real (upgrades I1) -- [ ] B1 `backlinks` verb — Common Crawl hyperlinkgraph - (data.commoncrawl.org/projects/hyperlinkgraph), free, no key. -- [ ] B2 Honest cap — steal their idea (free-backlink-sources.md:33: cap - health at 70/100 when only Common Crawl). Fits our code-ceiling doctrine - exactly. -- [ ] B3 VERIFY FIRST: GSC Links API. Subagent claimed "available, OAuth +### AXE 3 — Off-page real (upgrades I1) — SUPERSEDED, see B1/B2 KILLED above +- [x] ~~B1 `backlinks` verb — Common Crawl hyperlinkgraph~~ KILLED: edges file + measured at 17.3 GB gzipped. Non-viable per audit; the reference impl + caps at 500 MiB = 2.9% of the graph and calls the remainder a backlink + profile. +- [x] ~~B2 Honest cap at 70/100~~ KILLED with B1: nothing left to cap. + I1's narrowed axis is the final state. +- [x] B3 VERIFY FIRST: GSC Links API. Subagent claimed "available, OAuth already there" — I doubt it: Search Console API v3 has no links endpoint (links report is UI-only AFAIK). Verify before planning on it. Do not assert. @@ -76,16 +147,16 @@ as NEW VERBS. No new architecture. - [ ] R1 `render` verb — Playwright, GATED on SPA detection (STEP 2 already detects framework + rendering mode). Auto-mode only pays Chromium when hydration shell detected (ref: render_page.py:226 logic, adapt not copy). -- [ ] R2 ARBITRAGE: heavy dep (Chromium ~300MB) vs our bash+curl purity. +- [x] R2 ARBITRAGE: heavy dep (Chromium ~300MB) vs our bash+curl purity. Cheaper honest alternative: on SPA, REFUSE to score on-page rather than score it wrong (today: curl reads source, not hydrated DOM → every meta/JSON-LD/heading/img grep is blind, compensated only by a §0 flag). ### AXE 5 — Hardening + regression (lower priority) -- [ ] H1 SSRF guard on curl paths — both agents curl user-supplied domains. +- [x] H1 SSRF guard on curl paths — both agents curl user-supplied domains. Our own CLAUDE.md doctrine says "never trust user input". url_safety.py (622 l, obfuscated-IPv4 decode, DNS pinning) is a solid reference. -- [ ] H2 `drift` baseline (SQLite) — SEO.md Historique keeps only date+score+ +- [x] H2 `drift` baseline (SQLite) — SEO.md Historique keeps only date+score+ key changes. Their seo-drift is on-page regression detection, NOT rank tracking (common misread). Optional. diff --git a/agents/geo-analyzer.md b/agents/geo-analyzer.md index b3034db..66f845c 100644 --- a/agents/geo-analyzer.md +++ b/agents/geo-analyzer.md @@ -244,8 +244,14 @@ the PERMISSIVE template from `ai-crawlers-2026.md`. ### Live verification `[FULL only]` +**Guard the domain before it reaches a shell — mandatory, not optional.** +`$DOMAIN` is interpolated inside double quotes below, where `$` and backtick +still execute. Run the guard FIRST and use only its output; non-zero exit → +STOP this step and report the refusal, never sanitise-and-retry. + ```bash -DOMAIN="" +DOMAIN="$(bash ~/.claude/lib/url-guard.sh host "")" || { + echo "STEP 4 aborted: domain refused by url-guard"; exit 2; } # Verify robots.txt served curl -s "https://$DOMAIN/robots.txt" | head -50 @@ -431,6 +437,57 @@ Record what exists. For each: - Does `sameAs` on the site point to it? - If yes, does the target resolve and match? +### sameAs resolution `[FULL only]` + +`entity-seo.md:148` says "validate each URL resolves" and nothing did. +A `sameAs` pointing at a dead profile is worse than a missing one: it +asserts an identity link that fails on follow, in the exact graph AI +engines walk to confirm who you are. + +```bash +grep -rhoE '"sameAs"[^]]*\]' \ + --include="*.html" --include="*.astro" --include="*.tsx" --include="*.jsx" \ + --include="*.vue" --include="*.svelte" --include="*.php" --include="*.json" \ + . 2>/dev/null \ + | grep -oE 'https?://[^"]+' | sort -u | while read -r RAW; do + # These URLs come from the audited repo's JSON-LD, not from the operator: + # guard each one before it reaches curl. A refused entry is REPORTED, not + # skipped silently — an unguardable sameAs is itself a finding. + U="$(bash ~/.claude/lib/url-guard.sh url "$RAW" 2>/dev/null)" || { + printf 'REFUSED %s\n' "$RAW"; continue; } + printf '%s %s\n' \ + "$(curl -sIL -o /dev/null -w '%{http_code}' --max-time 10 "$U" 2>/dev/null || echo 000)" \ + "$U" + done +``` + +`REFUSED` rows are not dead links and not live ones — the URL never left the +machine. Report them in §14 with the raw value: a `sameAs` carrying shell +metacharacters or pointing at `localhost` is either broken markup or someone +probing, and both are worth the client knowing. + +**Read the codes honestly — a block is not a death.** Some platforms refuse +non-browser clients: LinkedIn answers `999` (verified 2026-07-16 against a +live company page). A naive check calls that dead and the bundle deletes a +live link — the most valuable node in the graph, since LinkedIn is the +identity anchor for most B2B entities. + +Do NOT assume which platforms block: the same 2026-07-16 check found +`x.com` returning `200`, contradicting the "Twitter always 403" folklore. +Test the code you actually got; classify by code, never by platform +reputation. + +| Code | Verdict | Action | +|---|---|---| +| 2xx / 3xx | alive | none | +| **404 / 410** | **genuinely dead** | finding WITH direction — fix or remove | +| 401 / 403 / 429 / 999 | bot-blocked | **inconclusive — no finding.** Report as unverified, never as dead | +| 000 (DNS/timeout) / 5xx | inconclusive | retry once, then unverified | + +No G2/G6 item may remove a `sameAs` on anything but 404/410. Same rule as +the NAP direction rule: an unreliable signal read confidently is worse than +no signal. Unverified entries → §14, naming the platform and the code. + ### Google Knowledge Panel `[FULL only]` ``` @@ -458,6 +515,17 @@ PRIORITY ACTIONS : ## STEP 8 — CONTENT SHAPE FOR AI `[both]` +**Rendering gate first (R2).** `bash ~/.claude/lib/seo-data/fetch.sh +rendercheck --url "https://$DOMAIN/"`. Verdict `client-rendered` → Content +Shape is `N/A — content not in served HTML`, excluded from the weighted +global, never scored zero. And say the thing that actually matters here: AI +crawlers are **worse** at JS than Googlebot is. GPTBot, PerplexityBot and +ClaudeBot fetch HTML and largely do not execute it, so a client-rendered site +is not just unauditable by us — it is close to invisible to the engines this +whole audit targets. That is a §0 alert and the top user action (SSR/SSG), +not a schema tweak. +Site-wide axes (crawler policy, llms.txt) are unaffected: those are files. + Load: `~/.claude/agents/resources/content-shape-for-ai.md` Sample 5-10 key pages (homepage + top service/blog pages). For each: @@ -488,7 +556,8 @@ sample of a 300-page site says nothing about the other 294. ```bash # Extract H1/H2/H3 from main pages to assess heading style -for f in index.html $(find . -maxdepth 3 -name "*.astro" -o -name "*.tsx" -o -name "*.md" -o -name "*.html" | head -10); do +mapfile -t FEXCL < <(bash ~/.claude/lib/source-scope.sh findargs) # C1a: skip build output +for f in index.html $(find . "${FEXCL[@]}" -maxdepth 3 \( -name "*.astro" -o -name "*.tsx" -o -name "*.md" -o -name "*.html" \) | head -10); do echo "=== $f ===" grep -oE '<(h1|h2|h3)[^>]*>[^<]+|^#{1,3} .+' "$f" 2>/dev/null | head -20 done @@ -595,7 +664,9 @@ Score each axis. Use concrete findings from STEP 2-9. ``` GEO SCORING () -COVERAGE : of sitemap URLs (

%) | pages, total UNKNOWN +COVERAGE SOURCE : of page templates (

%) — bounds Schema.org +COVERAGE LIVE : of sitemap URLs (

%) — bounds Content Shape + | UNKNOWN (no sitemap / fetch degraded) AI Crawlers Policy : XX/20 llms.txt : XX/20 Schema.org for AI : XX/20 @@ -612,6 +683,18 @@ Schema.org. Site-wide axes (AI Crawlers Policy, llms.txt) are unaffected: robots.txt and llms.txt are single files, fully read. Say which is which rather than letting one ratio discredit the whole report. +**Same source/live split as seo-analyzer STEP 9 (C1c), and it cuts your axes +differently.** A JSON-LD block lives in a shared layout, so one sampled page +per URL family proves the SCHEMA for the whole family — SOURCE coverage is +what bounds it. Content Shape does NOT work that way: Definition Lead, TL;DR +and heading wording are written per page, so a template says nothing about +its 25 instances. Bound Schema.org by SOURCE, Content Shape by LIVE, and +never quote the flattering one alone. Get the URL families from +`fetch.sh sitemap`, grouped as seo-analyzer STEP 5 describes — shared parent +path OR shared slug prefix, because both layouts are real: first-segment +alone reads 8 flat `/lavage-auto-` pages as 8 singletons. If `/seo` +already ran it, reuse the count rather than re-fetching. + Per user instruction: **GEO weight in combined SEO+GEO report = 20% for local, 25% for national/SaaS/content.** @@ -913,6 +996,14 @@ PROCHAINE ETAPE : NEVER `Write` on shared templates. `Write` is reserved for files you solely own: robots.txt, llms.txt, llms-full.txt. Full-template refactor → escalate as user action in §11. +- **NEVER emit a bundle item targeting build output (C1a).** No path under + `dist/ build/ .next/ .nuxt/ .output/ _site/ .astro/ .svelte-kit/ out/` — + run `bash ~/.claude/lib/source-scope.sh list` for the authoritative set. + Those files are regenerated: the `npm run build` the dispatcher runs to + VERIFY your fix is what erases it. The fix lands, verification passes, + nothing survives, and the report claims it was applied. Fix the SOURCE + template that generates the file. If you cannot find the source, that is + a finding — say so, do not patch the artifact. - **Respect PERMISSIVE/RESTRICTIVE choice.** geo-analyzer defaults to PERMISSIVE (GEO's goal is AI visibility). Only switch if the client explicitly flags premium/regulated content. diff --git a/agents/seo-analyzer.md b/agents/seo-analyzer.md index 190a865..82cb309 100644 --- a/agents/seo-analyzer.md +++ b/agents/seo-analyzer.md @@ -181,8 +181,9 @@ keep the two consistent. ls .htaccess nginx.conf netlify.toml vercel.json wrangler.toml 2>/dev/null # SEO files ls robots.txt sitemap.xml sitemap-index.xml sitemap-images.xml sitemap-videos.xml 2>/dev/null -# Legal pages -find . -maxdepth 3 \( -iname "*mention*" -o -iname "*legal*" -o -iname "*confidentialite*" -o -iname "*privacy*" -o -iname "*cgv*" -o -iname "*cgu*" \) 2>/dev/null | head -10 +# Legal pages — source only (C1a: find ignores .gitignore, grep does not) +mapfile -t FEXCL < <(bash ~/.claude/lib/source-scope.sh findargs) +find . "${FEXCL[@]}" -maxdepth 3 \( -iname "*mention*" -o -iname "*legal*" -o -iname "*confidentialite*" -o -iname "*privacy*" -o -iname "*cgv*" -o -iname "*cgu*" \) 2>/dev/null | head -10 # Analytics / trackers grep -rl "gtag\|GTM-\|analytics\|matomo\|_paq\|plausible\|umami" --include="*.html" --include="*.js" --include="*.tsx" --include="*.astro" --include="*.php" . 2>/dev/null | head -10 # Cookie consent / CMP @@ -250,8 +251,15 @@ the §14 observed-list. But under `/seo` the security headers themselves are out of scope for scoring: see the Technical axis note in STEP 9. Under `/harden` they are the entire job. Reading is not scoring. +**Guard the domain before it reaches a shell — mandatory, not optional.** +Every curl below interpolates `$DOMAIN` inside double quotes, where `$` and +backtick still execute. Run the guard FIRST and use only its output; if it +exits non-zero, STOP this step and report the refusal — never "clean up" the +value and retry. + ```bash -DOMAIN="" +DOMAIN="$(bash ~/.claude/lib/url-guard.sh host "")" || { + echo "STEP 4 aborted: domain refused by url-guard"; exit 2; } # Headers curl -sI "https://$DOMAIN/" | head -30 @@ -332,13 +340,71 @@ When STEP 0/STEP 1 recorded a GSC account+property (not "none"): ```bash bash ~/.claude/lib/seo-data/fetch.sh queries --account "$GSC_ACCOUNT" --property "$GSC_PROPERTY" --days 90 --dim query bash ~/.claude/lib/seo-data/fetch.sh inspect --account "$GSC_ACCOUNT" --property "$GSC_PROPERTY" --url "https://$DOMAIN/" +bash ~/.claude/lib/seo-data/fetch.sh cannibal --account "$GSC_ACCOUNT" --property "$GSC_PROPERTY" --days 90 ``` +**`cannibal` — keyword cannibalisation, from Google's own data (C2).** Groups +90 days of `query`+`page` rows and returns every query where 2+ of OUR pages +compete, ranked by total impressions. The API always allowed multiple +dimensions; this system only ever asked for one, so the conflict was invisible. + +Read it: +- `conflicts[]` → for each, the strongest page (most impressions) is listed + first. That is usually the one to KEEP; the others either consolidate into + it (301 + merge content) or get differentiated. Never "fix" this by deleting + a page that has clicks — say what competes and let the user choose. +- A conflict with a large impression total and every page beyond position 10 + is the real prize: Google can't decide which page to rank, so none rank. +- `capped: true` → the row window was full; there are conflicts past the cut. + Say so in §14 rather than presenting the list as exhaustive. +- `status: degraded` → no GSC account. Cannibalisation is then **not + auditable** — no substitute exists on-site. §14 line, do not guess it from + title similarity. + +**This is NOT the 30/70 rule, and do not merge the two.** Cannibalisation is +a SERP fact Google measured. The 30/70 duplication rule is a content-similarity +question with **no data source here**: measuring it properly needs main-content +extraction (strip nav/header/footer), and without that a naive comparison of +two same-template pages returns ~95% similar for every site, which is a +confident false positive. So 30/70 stays an explicit LLM judgement over the +≥3 same-family pages STEP 5 now samples for it — label it as judgement in the +report, never as a measurement, and never quote a similarity percentage you +did not compute. + Report: top queries; flag **QUICK WINS** = rows with position between 4 and 10 AND high impressions (candidates to push onto page 1 with a title/meta/content tweak). Report index coverage from `inspect`. All emitted into SEO.md §2 (technical) and §8 (quick wins). +**`inspect` also returns `rich_results` — Google's own structured-data +verdict on the live indexed URL.** It rides the same response (no extra +call, no extra quota). This is the only programmatic JSON-LD validation in +the system; everything else about schema is read by eye. + +``` +rich_results.verdict : PASS | FAIL | NEUTRAL | VERDICT_UNSPECIFIED | ABSENT +rich_results.types[] : {type, items, errors, warnings, issues[]} +``` + +- `FAIL` + a type carrying `errors > 0` → that type **cannot show as a rich + result**. Bundle item, cite the `issues[]` message verbatim — it is + Google's wording, not ours, and geo-analyzer owns the JSON-LD fix + (CROSS-AGENT NOTE). +- `warnings` → recommended fields missing. Report, do not gate on them. +- **`ABSENT` means Google detected no rich results on this URL** — the key + is omitted upstream when nothing is found. It is NOT an error and NOT + proof the markup is broken: a page with no structured data reads the same + as one whose markup Google never parsed. Say "none detected", never + "invalid". +- `ABSENT` while the repo clearly ships JSON-LD → real finding: the markup + is not reaching Google (SPA-rendered, blocked, or malformed). Cross-check + before claiming it. + +**Bound this honestly.** `index:inspect` is per-URL, quota'd, and works only +on a GSC-verified property. It validates the URLs you sampled — not the +site. Its reach is the STEP 9 COVERAGE ratio, and §14 must say so rather +than let one PASS imply site-wide valid markup. + If `status=degraded` → note it in §2 and emit the §11 user action "Connecter GSC: `make seo-connect`". @@ -406,21 +472,118 @@ Fetch rendered HTML. Extract and analyze: ## STEP 5 — ON-PAGE AUDIT `[both]` +### Rendering gate — run this BEFORE anything else in STEP 5 (R2) + +```bash +bash ~/.claude/lib/seo-data/fetch.sh rendercheck --url "https://$DOMAIN/" +``` + +STEP 2 has always recorded `RENDERING: SSR/SSG/SPA/hybrid` and nothing ever +acted on it. This is the rule that does. The verdict comes from what the +server actually sent, not from reading package.json — a React SPA and a +Next.js SSR app are indistinguishable there. + +**`verdict: client-rendered` → REFUSE to score the On-page axis.** Do not +score it low. Do not score it at all: +- On-page → `N/A — content not in served HTML (client-rendered)`. Redistribute + nothing; a missing axis is not a zero. +- Every curl-based meta/H1/JSON-LD check would report "missing" against a site + that may be perfectly correct once hydrated. Those are FALSE findings, and + a bundle built on them would "fix" meta tags that already exist. +- **No bundle item may come from a live on-page check on this site.** Source + greps still apply — the JSX carries the tags — but you cannot tell which + route renders what, so treat them as inventory, not as per-page findings. +- `linkgraph` will refuse too (`no_links_in_html`) — the same blindness. Do + not work around either refusal. + +Still fully auditable, and worth saying so rather than returning an empty +report: robots.txt, sitemap.xml, HTTP headers, redirects, `.htaccess` / +framework config, CWV via CrUX (field data is real-user, hydration included), +GSC queries + index coverage, legal pages, image weights. + +**`verdict: partial`** → shell plus an SSR'd head, or a genuinely thin page. +Score what is present, name what is not, and say which of the two you think +it is. + +**§0 line, mandatory when not server-rendered:** +`Rendering: client-rendered — On-page NOT scored (content absent from served +HTML). Global score excludes it. Fix: SSR/SSG (CLAUDE.md: public sites are +never SPAs).` + +This is the honest half of the R1/R2 call: we do not render JS (no Playwright, +no Chromium), so we do not pretend to see what JS paints. Refusing is the +finding. + **Record the denominator BEFORE sampling.** This step samples; the report -says "audit". Count the URLs in `sitemap.xml` (fetch it in full — the -`head -50` in STEP 4 is a preview, not a count). That count is the coverage -denominator, and it feeds the mandatory COVERAGE line in STEP 9. No sitemap -→ denominator unknown: say so, never let silence imply full coverage. On a -500-page site a 12-page sample is 2.4% — the On-page score is an -extrapolation from it, and the reader cannot know that unless you print it. +says "audit". On a 500-page site a 12-page sample is 2.4% — the On-page score +is an extrapolation from it, and the reader cannot know unless you print it. + +```bash +bash ~/.claude/lib/seo-data/fetch.sh sitemap --url "https://$DOMAIN/sitemap.xml" +``` + +Returns `{count, urls[], index, dropped, ...}` — the coverage denominator and +your sampling frame. It follows a `` one level, dedupes, strips +whitespace, and handles `.xml.gz`. No auth, no venv, no Google. + +Read it honestly: +- `count` → the denominator for the STEP 9 COVERAGE line. +- `dropped > 0` → entries that were not usable URLs. Worth a §14 line: a + sitemap emitting junk is a tooling finding. +- `children_failed > 0` or `children_skipped` → the frame is incomplete. Say + so; do NOT present a partial denominator as the total. +- `status: degraded` → denominator UNKNOWN. Print that, never let silence + imply full coverage. `reason: unsafe_xml_dtd` is not a glitch — a sitemap + carrying a DTD is broken tooling or a billion-laughs aimed at the auditor. + Report it as a finding. + +**Guard every URL before it reaches curl.** These come from the target's own +server, not from the operator — the one place in this audit where a remote +file's bytes flow into a shell: + +```bash +U="$(bash ~/.claude/lib/url-guard.sh url "$RAW_FROM_SITEMAP")" || continue +``` + +The verb applies a garbage filter, not that guard; the guard belongs at the +point of use (same contract as the sameAs check in geo-analyzer). ### Meta tags per page (sample 5-15 key pages) -Sample by risk, not convenience: homepage + top templates (one per page -type: service, city, blog, product, legal) + any page GSC flags as a -position 4-10 quick win. Same template audited twice buys nothing; an -un-sampled template is an un-audited template — name the templates you -skipped. +**Group the sitemap URLs into families first** — a family is "pages one +template renders". You do not need framework routing knowledge to see them, +but you DO need to look at the actual URL shape, because it varies: + +| Layout | Example | Family signal | +|---|---|---| +| Nested | `/creation-site-internet/essonne-91/`, `/creation-site-internet/seine-et-marne-77/` | **shared parent path** → 25 pages, 1 family | +| **Flat** | `/lavage-auto-pomponne`, `/lavage-auto-torcy`, `/lavage-auto-chelles` | **shared slug prefix** → 8 pages, 1 family | + +Both are real, measured on two live sites. First-path-segment alone handles +the nested case and **fails the flat one**: those 8 city pages read as 8 +unrelated singletons, so the largest "family" becomes `/services` (5) and the +doorway-page risk — the exact thing the 30/70 rule exists to catch — is +invisible. Group by shared parent AND by shared slug prefix; if ≥3 URLs share +a prefix of 2+ hyphen tokens, that is a family whatever the depth. + +Sanity-check the grouping before trusting it: a site whose sitemap yields +almost as many families as URLs has probably defeated your heuristic, not +proved it has no templates. + +**Sample by finding class, because the classes need opposite samples:** + +| Looking for | Sample | Why | +|---|---|---| +| Code defects (canonical, OG, `` dims, hreflang) | **1 per family** | one template renders the whole family — a missing canonical in `[dept]/index.astro` breaks all 25 identically. 1 per family ≈ 100% SOURCE coverage for ~8 fetches. | +| **Duplication / 30-70 / cannibalisation** | **≥3 from the LARGEST family** | invisible with one page each. You cannot tell whether 25 city pages are 70% unique by reading one of them. | +| Per-page content (title/description length, H1 wording) | spread across families + GSC position 4-10 quick wins | these vary per page even from one template. | + +"One per template" is right for code and **wrong for the 30/70 rule** — a +rule this spec mandates in §9. Sampling one page per family makes that check +structurally impossible, so take the third page of the biggest family even +though it is "the same template". + +An un-sampled family is an un-audited family. Name the ones you skipped. For each sampled page: ``` @@ -457,10 +620,32 @@ grep -rE ']*>' --include="*.html" --include="*.astro" --include="*.tsx" - # Images missing dimensions (CLS risk) grep -rE ']*>' --include="*.html" --include="*.astro" --include="*.tsx" --include="*.jsx" --include="*.php" . 2>/dev/null | grep -vE 'width=|height=' | head -30 -# Check image asset sizes -find . -type f \( -iname "*.jpg" -o -iname "*.jpeg" -o -iname "*.png" -o -iname "*.gif" \) ! -path "./node_modules/*" ! -path "./.git/*" -printf "%s %p\n" 2>/dev/null | sort -rn | head -20 +# Check image asset sizes — source only, never build output (C1a) +mapfile -t FEXCL < <(bash ~/.claude/lib/source-scope.sh findargs) +find . "${FEXCL[@]}" -type f \( -iname "*.jpg" -o -iname "*.jpeg" -o -iname "*.png" -o -iname "*.gif" \) -printf "%s %p\n" 2>/dev/null | sort -rn | head -20 ``` +**Why the guard, and why `find` specifically (C1a).** `grep` and `find` +disagree about this repo and you use both. Claude Code routes `grep` through +ugrep with `--ignore-files`, so it honours `.gitignore` and never descends +into a gitignored `dist/`. `find` honours nothing. Measured on a real Astro +repo: this command returned **92 images, 45 of them under `dist/`** — every +asset twice, source and generated copy, byte-identical. So "top 20 by size" +was ~10 real images dressed as 20, and a batch-C item +(`cwebp -q 80 -o .webp`) could target `dist/og-image.png`, whose +`.webp` the dispatcher's own `npm run build` then erases. The fix lands, +verification passes, nothing survives. + +`FEXCL` MUST be consumed as a quoted array. `find . $FEXCL …` lets the shell +glob `*/dist/*` against the CWD and hand the matches to find as search paths +— that made the same run return 135 hits and kept every `dist/` file. + +Do NOT add these exclusions to the `grep` lines: the shim already covers +them, `public/` is deliberately kept (it is Astro/Vite/Next SOURCE and holds +`favicon.ico`, `apple-touch-icon.png`, `robots.txt` — the very files STEP 4 +curls), and it is build output only for Hugo/Gatsby, which the script +detects. + Flag images over 100 KB as compression candidates. WebP/AVIF preferred over JPEG/PNG. @@ -481,6 +666,34 @@ Each embedded or self-hosted video should have: ### Internal linking + topic clusters (silos sémantiques) +```bash +bash ~/.claude/lib/seo-data/fetch.sh linkgraph --url "https://$DOMAIN/sitemap.xml" +``` + +**This answers the two questions below, which this spec has always asked and +never had a command for (C3).** Crawls every sitemap URL once, extracts +internal ``, and returns `orphans`, `beyond_3_clicks`, `unreachable`, +`max_depth`. Measured cost: 24 pages in 2.7 s, 86 in 3.8 s — cheap enough to +always run on FULL. + +Read it honestly: +- `orphans` present → real finding, act on it. +- **`orphans_withheld: true` → there is NO orphan list, and you must not + invent one.** It appears when the crawl was capped or any page failed. An + orphan cannot be sampled: proving a page has no inbound link means having + read every other page, so a partial crawl invents orphans. "Page X has no + inbound links" when it does sends the client fixing what is not broken. + §14 line, not a finding. +- `reason: no_links_in_html` → **not a site with zero links; a site whose + links are rendered by JS.** Every page would look orphaned — the worst false + positive this tool could emit — so the verb refuses instead. Flag the SPA in + §0 and stop; do not hand-roll a link audit around it. +- `unreachable` ⊃ `orphans`: a page can have inbound links yet sit outside the + homepage's reach (linked only from another unreachable page). Both matter, + they are not the same finding. +- `max_depth` > 3 → `beyond_3_clicks` names the pages. That is the ":613" + check, now measured rather than asserted. + Sample critical pages. Check: - Every important page reachable within 3 clicks from homepage? - Navigation consistent? @@ -685,6 +898,41 @@ FIX: AUTO () | USER () | Competitive position | 5% | 10% | | | Legal compliance | 10% | 5% | | +**Compute the scores, do not feel them (I7).** Emit your findings, then let +the engine do the arithmetic: + +```bash +bash ~/.claude/lib/seo-data/fetch.sh score --findings /tmp/seo-findings.json +``` + +```json +{"depth":"FULL","profile":"local", + "axes":{"technical":{"findings":[{"severity":"haute","affected":9,"sampled":12}]}, + "on-page":{"status":"na","reason":"client-rendered (R2)"}, + "off-page":{"status":"na","reason":"backlinks unauditable (I1)"}}} +``` + +`profile`: `local` (B2C) | `national` (SaaS/national/content). Severities are +`critique|haute|moyenne|basse` — `/harden`'s scale (-15/-8/-3/-1, clamp, +then /5 into /20), so the whole skill family speaks one vocabulary. + +**The split matters.** WHICH findings exist and how severe each is stays your +judgement — irreducible. The addition is not: same findings in, same score +out. Until now every axis was felt, so two runs over identical code could +disagree, and `/client-handover` gates on 17/20. + +- `affected`/`sampled` (optional) shift severity ONE step: ≥50% of the sample + escalates, a single page de-escalates. A defect on 1 of 12 pages is not the + defect on 12 of 12; pretending so is what made the old numbers wobble. +- `status: "na"` → the axis is EXCLUDED and the remaining weights are + renormalised for you. This is the R2 rule (client-rendered on-page) and the + I1 rule (unauditable off-page), finally computed instead of done by hand. + **N/A is not a zero** and the engine will not let it behave like one. +- `status: "error"` → malformed findings. Fix them; never fall back to + eyeballing a number. +- Run it twice on the same file before publishing. If the output moved, your + findings moved, and that is the thing to explain. + **Technical axis note:** CWV scored on CrUX field data (75th percentile, real users, from STEP 4) when available; otherwise lab PageSpeed Lighthouse run. @@ -714,6 +962,14 @@ owns them (0-100 + Observatory/SecurityHeaders/SSL Labs). Run /harden Name what you saw. An omission has to stay legible — the same reason COVERAGE is mandatory in STEP 9. +**On-page axis note (R2).** `rendercheck` verdict `client-rendered` → this +axis is `N/A — content not in served HTML`, excluded from the weighted global, +NOT scored zero. A zero says "your on-page is bad"; N/A says "we could not +see it", and only one of those is true. Renormalise the remaining weights over +the axes actually scored and say so on the SEO GLOBAL line. The code ceiling +must state that no code fix raises an axis we did not measure — the unlock is +SSR/SSG, and that is a user action, not a bundle item. + **Off-page axis note (I1).** Score ONLY the unlinked brand mentions gathered in STEP 6 (`web_search "" -site:`). Backlink profile and domain authority have NO data source here — no index, @@ -723,13 +979,30 @@ that reaches a client via `/client-handover`. A low mention count is a low mention count — it is NOT evidence of a weak backlink profile. Mandatory §14 line whenever depth=FULL, verbatim: -`Backlinks / domain authority — NOT audited: no backlink index wired. -Nearest free source: Common Crawl hyperlinkgraph. Commercial: Ahrefs / -Semrush / Majestic. The Off-page score above prices in brand mentions only.` +`Backlinks / domain authority — NOT audited: no free backlink index is +practical, and none is wired. Commercial: Ahrefs / Semrush / Majestic. The +Off-page score above prices in brand mentions only.` -Weight deliberately unchanged despite the narrower scope: re-deriving it -now, then again when a backlink source lands, would churn historical -scores twice. Revisit the 10/15% only when the axis widens back. +**This is the final state, not a placeholder (B1 killed, 2026-07-17.)** The +free options were measured, not assumed: +- **GSC has no links endpoint.** The Search Console API exposes exactly + Search Analytics, Sitemaps, Sites, URL Inspection. The Links report is + UI-only. +- **Common Crawl's hyperlinkgraph is 17.3 GB gzipped** for the domain-edges + file alone (+879 MB vertices, +2.3 GB ranks), measured live. Finding one + domain's inbound links means scanning all of it, per audit. Not slow — + non-viable, and abusive toward a nonprofit serving it free. The reference + implementation everyone cites caps its download at 500 MiB, i.e. **2.9% of + the edges file**, and reports whatever that arbitrary slice contained as a + backlink profile. That is a random sample wearing a measurement's clothes, + which is precisely what this axis note exists to prevent. +- **Bing Webmaster's `GetUrlLinks` is the only free, viable source** — but it + is first-party only (your verified properties), so it can never cover a + competitor, and it needs the client's Bing account. See W2, deferred. + +So: no number here beats a fabricated one. Weight deliberately unchanged — +re-deriving it for an axis that is not going to widen would churn historical +scores for nothing. ### LOCAL depth — 4 axes @@ -783,8 +1056,9 @@ misroutes the client-handover gate and the user's effort. ``` SEO SCORING () -COVERAGE : of sitemap URLs (

%) — templates skipped: - | pages, total UNKNOWN (no sitemap) +COVERAGE SOURCE: of page templates (

%) — skipped: +COVERAGE LIVE : of sitemap URLs (

%) — families: + | UNKNOWN (no sitemap / fetch degraded) Technical : XX/20 On-page : XX/20 SEO Local : XX/20 | N/A @@ -796,11 +1070,27 @@ Legal : XX/20 SEO GLOBAL (weighted): XX.X/20 () ``` -**COVERAGE is mandatory, never omitted, never rounded up.** It is the -honesty bound on every page-level axis: On-page and the on-page share of -Technical are extrapolations from the sample. If coverage < 25%, repeat it -in §0 as a major alert — a 17/20 drawn from 3% of a site is not a 17/20, and -`/client-handover` gates on these numbers. +**Both COVERAGE lines are mandatory, never omitted, never rounded up.** They +are the honesty bound on every page-level axis: On-page and the on-page share +of Technical are extrapolations from the sample, and `/client-handover` gates +on these numbers. + +**Report both, because they bound different findings — do not average them +into one comforting number.** +- **SOURCE** bounds CODE findings. One template renders its whole family, so + 1 page per family can legitimately reach 100% here. High SOURCE coverage is + a real claim: the code paths were seen. +- **LIVE** bounds CONTENT findings — title/description wording, thin pages, + 30/70 duplication. It stays low by design and that is fine, as long as it + is printed. Measured on a real site: 12 of 86 URLs is 14% LIVE while the + same 12 pages are 100% SOURCE. Reporting only the 14% understates the audit; + reporting only the 100% oversells it. Both, or neither means anything. +- LIVE < 25% → repeat in §0. A 17/20 for content drawn from 3% of a site is + not a 17/20. +- SOURCE < 100% → name the skipped templates in §0. That is not a sampling + choice, it is code nobody read. +- Denominator UNKNOWN (no sitemap, or `sitemap` degraded) → print UNKNOWN. + Never let silence imply full coverage. Per user instruction: this score represents **80% of the combined final score for local B2C (20% for GEO), or 75% for SaaS/national @@ -1159,6 +1449,15 @@ PROCHAINE ETAPE : `Write` on shared templates. `Write` is reserved for files you solely own: sitemap.xml, .htaccess, legal pages, new city/service pages. Full-template refactor → escalate as user action in §11. +- **NEVER emit a bundle item targeting build output (C1a).** No path under + `dist/ build/ .next/ .nuxt/ .output/ _site/ .astro/ .svelte-kit/ out/` — + `bash ~/.claude/lib/source-scope.sh list` is the authoritative set. Those + files are regenerated: the `npm run build` the dispatcher runs to VERIFY + your fix is what erases it. The fix lands, verification passes, nothing + survives, and the report claims it was applied. This bites batch C hardest + (`cwebp -q 80 -o .webp` on a `dist/` asset writes a `.webp` the + next build deletes). Fix the SOURCE that generates the artifact; if you + cannot find it, that is a finding — say so, do not patch the artifact. - **Landing page protection.** Zero visible change except meta tags, footer links, JSON-LD, image optimization. - **Preserve existing valid SEO.** Don't rewrite correct tags. diff --git a/lib/seo-data/README.md b/lib/seo-data/README.md index b730beb..747a244 100644 --- a/lib/seo-data/README.md +++ b/lib/seo-data/README.md @@ -80,9 +80,162 @@ fetch.sh queries --account client-a --property sc-domain:ex.com [--days 90] [--d → {"status":"degraded","reason":"no_credentials"|"token_revoked"|"network_error"|"rate_limited"} fetch.sh inspect --account client-a --property … --url https://ex.com/page - → {"status":"ok","source":"gsc","indexed":true,"coverage":"…","last_crawl":"…"} + → {"status":"ok","source":"gsc","indexed":true,"coverage":"…","last_crawl":"…", + "rich_results":{"verdict":"PASS|FAIL|NEUTRAL|VERDICT_UNSPECIFIED|ABSENT", + "types":[{"type":"FAQ","items":2,"errors":2,"warnings":1, + "issues":["Missing field 'acceptedAnswer'"]}]}} → {"status":"degraded","reason":"…"} + rich_results rides the SAME URL-Inspection response — Google already sends + it, `inspect` used to discard it. No extra call, quota or OAuth scope. + It is the only programmatic structured-data validation in the system. + • verdict PARTIAL is never emitted — the API reserves it as unused. + • verdict ABSENT is SYNTHETIC (not a Google enum): the API omits + richResultsResult entirely when it detects no rich results. Surfaced + as a value rather than a missing key, because a caller cannot tell an + absent key apart from a check that never ran. ABSENT = "none + detected", never "invalid". + • errors/warnings count issue INSTANCES; issues[] is deduped — the same + issueMessage repeats across every affected item. + +fetch.sh cannibal --account client-a --property … [--days 90] [--rows 1000] + → {"status":"ok","source":"gsc","days":90,"rows_scanned":1000,"capped":true, + "conflict_count":12, + "conflicts":[{"query":"plombier paris","pages":3,"total_impressions":2400, + "urls":[{"url":…,"clicks":…,"impressions":…,"position":…}]}]} + → {"status":"degraded","reason":"…"} # no account → NOT auditable + + Keyword cannibalisation from Google's own data: queries where 2+ of OUR + pages compete. Groups query+page rows; conflicts ranked by total + impressions, and within each the strongest page first. `capped:true` means + the row window was full — more conflicts exist past the cut, say so. + Same auth, same quota family, no new scope: the API always accepted several + dimensions at once, this engine only ever asked for one. + • NOT the 30/70 duplication rule. This is a SERP fact Google measured. + 30/70 is content similarity, which has no data source here — doing it + naively (compare two same-template pages without stripping nav/footer) + returns ~95% similar for every site, a confident false positive. It stays + an LLM judgement, labelled as one. + • `queries` now takes `--dim query,page` (comma-separated) and `--rows`. + Rows gained a `keys` list; `key` stays as keys[0], so the single-dim + consumer is untouched. + +fetch.sh sitemap --url https://ex.com/sitemap.xml + → {"status":"ok","source":"sitemap","index":false,"count":86,"dropped":0, + "urls":["https://ex.com/", …]} + → {"status":"ok","index":true,"children_total":4,"children_read":4, + "children_failed":0,"count":312,…} # , one level deep + → {"status":"degraded","reason":"fetch_failed"|"parse_failed"|"no_urls" + |"unsafe_xml_dtd"} + + No auth, no Google, no venv: stdlib only (urllib + xml.etree + gzip). + Gives STEP 9's COVERAGE line the denominator it was told to print and never + had, and STEP 5 a real sampling frame. Dedupes, strips whitespace, handles + .xml.gz. Caps: 50 children of an index, 50k URLs, 20 MB read — each cut is + REPORTED (children_skipped / truncated), never silent. + + • NOT a security boundary. urllib fetches these, so nothing here reaches a + shell. The CONSUMER interpolates them into curl, so seo-analyzer runs + lib/url-guard.sh at the point of use — same contract as the sameAs check. + A second copy of the guard here would only drift. + • `unsafe_xml_dtd`: a sitemap NEVER has a DTD (sitemaps.org is then + ). Any doctype/entity is refused BEFORE parsing. xml.etree + does not expand external entities, but it IS billion-laughs-vulnerable — + 1 KB expands to gigabytes, and the 20 MB read ceiling bounds the input, + not the expansion. Refusing the construct beats depending on parser + internals AND keeps this stdlib-only; defusedxml would drag in a venv for + a document type that has no legitimate DTD. + +fetch.sh rendercheck --url https://ex.com/ + → {"status":"ok","verdict":"server-rendered"|"client-rendered"|"partial", + "body_text_chars":7650,"h1_in_html":1,"jsonld_in_html":9, + "meta_description_in_html":true,"html_bytes":132447, + "warning":"…"} # warning only when not server-rendered + + R2, the honest half of the SPA call. seo-analyzer has always recorded + `RENDERING: SSR/SSG/SPA` and never acted on it; this is the signal it acts + on. Verdict comes from what the server SENT — package.json cannot tell a + React SPA from a Next.js SSR app. + • client-rendered → the agent REFUSES to score On-page (N/A, not zero: a + zero says "your on-page is bad", N/A says "we could not see it"). Every + curl-based meta/H1/JSON-LD check would report "missing" against a site + that is fine once hydrated — false findings, and a bundle that "fixes" + tags which already exist. + • Does NOT render JS. No Playwright, no Chromium, no venv. Refusing IS the + finding. + • Script/style text is not page text: measured 7 chars on a React shell + whose inline window.__INITIAL_STATE__ is large. Without that, a 200 KB + bundle reads as a rich page. + • Measured 2026-07-17: zenquality 7650 chars/1 h1/9 jsonld and + lavageangels356 13973/1/1 → server-rendered; a Vite shell → 7/0/0. + +fetch.sh linkgraph --url https://ex.com/sitemap.xml [--max 500] + → {"status":"ok","source":"linkgraph","pages_crawled":86,"pages_failed":0, + "total_internal_links":2015,"capped":false,"max_depth":2, + "orphans":[…],"beyond_3_clicks":[…],"unreachable":[…]} + → {"status":"ok",…,"orphans_withheld":true,"reason_withheld":"crawl incomplete…"} + → {"status":"degraded","reason":"no_links_in_html"|"no_pages_fetched"|…} + + Answers seo-analyzer.md:613 ("reachable within 3 clicks?") and :616 ("orphan + pages?") — asked since forever, never computed. Stdlib only (urllib + + html.parser + urljoin), no auth. Measured: 24 pages in 2.7s, 86 in 3.8s. + • EXHAUSTIVE OR NOTHING. Orphans cannot be sampled: proving no inbound + link means having read every other page. If the crawl is capped or any + page failed, orphans are WITHHELD, never truncated — a false orphan + sends a client fixing what is not broken. + • no_links_in_html = a JS-rendered site, not a link-less one. Every page + would read as orphaned, so it REFUSES rather than report that. Does not + render JS by design (see the R1/R2 arbitration). + • Filters what a link graph must never hold: assets (seen live: + /css/main.css?v=1778157313), #anchors, mailto:/tel:/javascript:, other + hosts. Normalises the trailing slash so /blog and /blog/ are one node + rather than a phantom orphan pair. + • Mock is pages.json ({url: html}), not a single page.html: one fixture + cannot express a graph — every node would carry identical links. + +fetch.sh score --findings + → {"status":"ok","axes":{"technical":{"score_20":17.8,"weight":0.2, + "weight_renormalised":0.2857,"findings":2}}, + "na":["off-page","on-page"],"weights_renormalised":true,"global_20":17.6} + → {"status":"error","reason":"unknown severity: 'bogus'"|"bad_findings_json"} + + I7. /harden has a real scale (SKILL.md:435: -15/-8/-3/-1, clamp [0,100]); + /seo had none, so every axis was FELT and two runs over identical code could + disagree — while /client-handover gates on 17/20. Same scale here, /5 into + /20, one vocabulary across the family. + • The split: WHICH findings exist and how severe each is stays the LLM's + judgement. The addition is not. Same findings in, same score out. + • affected/sampled shift severity ONE step: >=50% of the sample escalates, + a single page de-escalates. A defect on 1 of 12 pages is not the defect + on 12 of 12. + • status:"na" → axis EXCLUDED, remaining weights renormalised. This is + R2's rule (client-rendered on-page) and I1's (unauditable off-page), + computed rather than done by hand. N/A is not a zero, and the engine + will not let it act like one. + • Malformed input is an error, never a silently wrong number — unlike the + fetch verbs, a degrade here would mean bad input, not a network fact. + +fetch.sh drift --url https://ex.com/sitemap.xml [--max 500] + → {"status":"ok","baseline":true,"captured":"…","pages":24,"store":"…"} + → {"status":"ok","baseline":false,"since":"…","gone":[…],"new":[…], + "regressions":[{"url":…,"field":"canonical","was":"…","now":null}], + "changes":[{"url":…,"field":"title","was":"…","now":"…"}]} + + On-page drift between audits. seo-analyzer.md:1365 keeps only "date + score + + key changes" as PROSE the LLM writes about its own previous prose: lossy, + unreproducible, machine-uncomparable. So "the redesign silently dropped 40 + canonicals" stays invisible. This snapshots title/description/canonical/ + robots/h1_count/jsonld_types per URL and diffs them. + • NOT rank tracking (the common misread of this feature elsewhere). + Positions come from GSC `queries`. This is regression detection. + • Runs over the WHOLE sitemap, never a sample: a drift over a sample that + changes between runs compares nothing. + • LOSING a signal = regression. CHANGING one = change, possibly intended — + the agent judges that, the engine only says which kind it is. + • Store: ~/.claude/seo-data/drift/.json, 0700, written via + os.replace — never a half-written baseline. Corrupt store → treated as + a first run rather than crashing the audit. + fetch.sh forget --label client-a → {"status":"ok","removed":true|false} # false = label wasn't in the store diff --git a/lib/seo-data/drift.py b/lib/seo-data/drift.py new file mode 100644 index 0000000..d5d4844 --- /dev/null +++ b/lib/seo-data/drift.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +"""On-page drift between audits. Stdlib only. + +seo-analyzer.md:1365 says "on re-run, move current content to Historique +(summary: date + score + key changes)". That is prose the LLM writes about its +own previous prose: lossy, unreproducible, and machine-uncomparable. So "the +redesign silently dropped 40 canonicals" is invisible unless someone happens +to notice. + +This snapshots the machine-readable signals per URL and diffs them. + +NOT rank tracking — a common misread of the same feature elsewhere. Positions +come from GSC (`queries`). This is on-page regression detection: what the site +said last time vs now. + +Runs over the WHOLE sitemap, never a sample: a drift over a sample that +changes between runs compares nothing. +""" +import argparse, json, os, re, time +from html.parser import HTMLParser + +import sitemap as sm + +STORE_DIR = os.path.expanduser("~/.claude/seo-data/drift") +MAX_PAGES = 500 +# Losing a signal is a regression. Changing one may be intentional — the agent +# judges that, we only report which kind it is. +TRACKED = ("title", "description", "canonical", "robots", "h1_count", "jsonld_types") + +class _Signals(HTMLParser): + def __init__(self): + super().__init__(convert_charrefs=True) + self.title, self.description, self.canonical, self.robots = None, None, None, None + self.h1_count, self.jsonld_types = 0, [] + self._in_title, self._in_ld = False, False + + def handle_starttag(self, tag, attrs): + a = dict(attrs) + if tag == "title": + self._in_title = True + elif tag == "h1": + self.h1_count += 1 + elif tag == "meta": + n = (a.get("name") or "").lower() + if n == "description": + self.description = (a.get("content") or "").strip() or None + elif n == "robots": + self.robots = (a.get("content") or "").strip() or None + elif tag == "link" and "canonical" in (a.get("rel") or "").lower(): + self.canonical = (a.get("href") or "").strip() or None + elif tag == "script" and a.get("type") == "application/ld+json": + self._in_ld = True + + def handle_endtag(self, tag): + if tag == "title": + self._in_title = False + elif tag == "script": + self._in_ld = False + + def handle_data(self, data): + if self._in_title and data.strip(): + self.title = re.sub(r"\s+", " ", data.strip()) + elif self._in_ld: + self.jsonld_types.extend(re.findall(r'"@type"\s*:\s*"([^"]+)"', data)) + +def _signals(html): + p = _Signals() + try: + p.feed(html) + except Exception: + pass + return {"title": p.title, "description": p.description, + "canonical": p.canonical, "robots": p.robots, + "h1_count": p.h1_count, "jsonld_types": sorted(set(p.jsonld_types))} + +def _mock_pages(): + """{url: html}, same convention as linkgraph: a single page.html fixture + cannot express a multi-page snapshot — every URL would look identical.""" + raw = sm._mock("pages.json") + return json.loads(raw.decode("utf-8")) if raw else None + +def _capture(urls): + pages = _mock_pages() + snap, failed = {}, 0 + for u in urls: + if pages is not None: + html = pages.get(u) + if html is None: + failed += 1 + continue + else: + try: + html = sm._fetch(u).decode("utf-8", "replace") + except Exception: + failed += 1 + continue + snap[u] = _signals(html) + return snap, failed + +def _store_path(sitemap_url): + from urllib.parse import urlparse + host = urlparse(sitemap_url).netloc.lower() + safe = re.sub(r"[^a-z0-9.-]", "_", host) or "unknown" + return os.path.join(STORE_DIR, safe + ".json") + +def _load(path): + if not os.path.exists(path): + return None + try: + with open(path, encoding="utf-8") as f: + return json.load(f) + except Exception: + return None # corrupt store -> treat as first run + +def _save(path, snap, stamp): + os.makedirs(os.path.dirname(path), mode=0o700, exist_ok=True) + tmp = path + ".tmp" + with open(tmp, "w", encoding="utf-8") as f: + json.dump({"captured": stamp, "pages": snap}, f) + os.replace(tmp, path) # atomic: never a half-written baseline + +def _classify(old, new): + """LOST a signal = regression. Changed it = change. Only the first is + unambiguous; the agent judges the rest.""" + regressions, changes = [], [] + for f in TRACKED: + o, n = old.get(f), new.get(f) + if o == n: + continue + row = {"field": f, "was": o, "now": n} + # Covers every tracked field uniformly: "Titre" -> None, 1 -> 0, + # ["Article"] -> []. Had the value, lost the value. + (regressions if (o and not n) else changes).append(row) + return regressions, changes + +def drift(sitemap_url, max_pages=MAX_PAGES): + sm_res = sm.sitemap(sitemap_url) + if sm_res.get("status") != "ok": + return sm_res + urls = sm_res["urls"][:max_pages] + snap, failed = _capture(urls) + if not snap: + return {"status": "degraded", "reason": "no_pages_fetched"} + stamp = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()) + path = _store_path(sitemap_url) + prev = _load(path) + _save(path, snap, stamp) + if prev is None: + return {"status": "ok", "baseline": True, "captured": stamp, + "pages": len(snap), "pages_failed": failed, "store": path} + old = prev.get("pages", {}) + regressions, changes = [], [] + for u, new in snap.items(): + if u not in old: + continue + r, c = _classify(old[u], new) + for row in r: + regressions.append(dict(row, url=u)) + for row in c: + changes.append(dict(row, url=u)) + return {"status": "ok", "baseline": False, + "since": prev.get("captured"), "captured": stamp, + "pages": len(snap), "pages_failed": failed, + "gone": sorted(set(old) - set(snap)), + "new": sorted(set(snap) - set(old)), + "regressions": regressions, "changes": changes, "store": path} + +def _cli(): + try: + p = argparse.ArgumentParser() + p.add_argument("--url", required=True, help="sitemap URL") + p.add_argument("--max", type=int, default=MAX_PAGES) + p.add_argument("--store", default=None) # accepted+ignored + args = p.parse_args() + print(json.dumps(drift(args.url, args.max), indent=2)) + except SystemExit as e: + if e.code not in (0, None): + print(json.dumps({"status": "error", "reason": "bad_usage"})) + raise + except Exception: + print(json.dumps({"status": "degraded", "reason": "unexpected_error"})) + +if __name__ == "__main__": + _cli() diff --git a/lib/seo-data/fetch.sh b/lib/seo-data/fetch.sh index ede0ee8..e2291be 100644 --- a/lib/seo-data/fetch.sh +++ b/lib/seo-data/fetch.sh @@ -27,8 +27,19 @@ _label_safe() ( LC_ALL=C; case "$1" in ''|[!A-Za-z0-9]*|*[!A-Za-z0-9._-]*) exit cmd="${1:-}"; shift || true case "$cmd" in accounts) exec "$PY" "$HERE/tokenstore.py" list --file "$STORE" ;; - crux|queries|inspect) + crux|queries|inspect|cannibal) exec "$PY" "$HERE/google_seo.py" "$cmd" --store "$STORE" "$@" ;; + # No auth, no Google: stdlib-only, runs even without the venv. + sitemap) + exec "$PY" "$HERE/sitemap.py" --store "$STORE" "$@" ;; + score) + exec "$PY" "$HERE/score.py" --store "$STORE" "$@" ;; + drift) + exec "$PY" "$HERE/drift.py" --store "$STORE" "$@" ;; + rendercheck) + exec "$PY" "$HERE/render_check.py" --store "$STORE" "$@" ;; + linkgraph) + exec "$PY" "$HERE/linkgraph.py" --store "$STORE" "$@" ;; forget) # forget --label

Accueil

", + "https://ex.com/a": "Page A

A

", + "https://ex.com/gone": "Bientot supprimee

G

" +} diff --git a/lib/seo-data/fixtures-drift-v1/sitemap.xml b/lib/seo-data/fixtures-drift-v1/sitemap.xml new file mode 100644 index 0000000..4aafee4 --- /dev/null +++ b/lib/seo-data/fixtures-drift-v1/sitemap.xml @@ -0,0 +1,6 @@ + + + https://ex.com/ + https://ex.com/a + https://ex.com/gone + diff --git a/lib/seo-data/fixtures-drift-v2/pages.json b/lib/seo-data/fixtures-drift-v2/pages.json new file mode 100644 index 0000000..ce06447 --- /dev/null +++ b/lib/seo-data/fixtures-drift-v2/pages.json @@ -0,0 +1,5 @@ +{ + "https://ex.com/": "Accueil refondue

plus de h1, plus de jsonld

", + "https://ex.com/a": "Page A

A

", + "https://ex.com/neuve": "Neuve

N

" +} diff --git a/lib/seo-data/fixtures-drift-v2/sitemap.xml b/lib/seo-data/fixtures-drift-v2/sitemap.xml new file mode 100644 index 0000000..2c99367 --- /dev/null +++ b/lib/seo-data/fixtures-drift-v2/sitemap.xml @@ -0,0 +1,6 @@ + + + https://ex.com/ + https://ex.com/a + https://ex.com/neuve + diff --git a/lib/seo-data/fixtures-linkgraph/pages.json b/lib/seo-data/fixtures-linkgraph/pages.json new file mode 100644 index 0000000..b2d42f5 --- /dev/null +++ b/lib/seo-data/fixtures-linkgraph/pages.json @@ -0,0 +1,9 @@ +{ + "https://ex.com/": "
a b trailing slash anchor asset mail tel external img", + "https://ex.com/a": "home deep", + "https://ex.com/b": "home", + "https://ex.com/deep": "deeper absolute", + "https://ex.com/deeper": "relative", + "https://ex.com/deepest": "home", + "https://ex.com/orphan": "home — links out, nobody links in" +} diff --git a/lib/seo-data/fixtures-linkgraph/sitemap.xml b/lib/seo-data/fixtures-linkgraph/sitemap.xml new file mode 100644 index 0000000..221bec3 --- /dev/null +++ b/lib/seo-data/fixtures-linkgraph/sitemap.xml @@ -0,0 +1,10 @@ + + + https://ex.com/ + https://ex.com/a + https://ex.com/b + https://ex.com/deep + https://ex.com/deeper + https://ex.com/deepest + https://ex.com/orphan + diff --git a/lib/seo-data/fixtures-norich/gsc_inspect.json b/lib/seo-data/fixtures-norich/gsc_inspect.json new file mode 100644 index 0000000..325cacd --- /dev/null +++ b/lib/seo-data/fixtures-norich/gsc_inspect.json @@ -0,0 +1,2 @@ +{"inspectionResult":{"indexStatusResult":{ + "verdict":"PASS","coverageState":"Submitted and indexed","lastCrawlTime":"2026-07-01T10:00:00Z"}}} diff --git a/lib/seo-data/fixtures-sitemap-dtd/sitemap.xml b/lib/seo-data/fixtures-sitemap-dtd/sitemap.xml new file mode 100644 index 0000000..5a58d14 --- /dev/null +++ b/lib/seo-data/fixtures-sitemap-dtd/sitemap.xml @@ -0,0 +1,10 @@ + + + + + +]> + + https://ex.com/&lol4; + diff --git a/lib/seo-data/fixtures-sitemap-index/sitemap.xml b/lib/seo-data/fixtures-sitemap-index/sitemap.xml new file mode 100644 index 0000000..534aa4b --- /dev/null +++ b/lib/seo-data/fixtures-sitemap-index/sitemap.xml @@ -0,0 +1,5 @@ + + + https://ex.com/sitemap-pages.xml + https://ex.com/sitemap-blog.xml + diff --git a/lib/seo-data/fixtures-sitemap-index/sitemap_child.xml b/lib/seo-data/fixtures-sitemap-index/sitemap_child.xml new file mode 100644 index 0000000..aadda6d --- /dev/null +++ b/lib/seo-data/fixtures-sitemap-index/sitemap_child.xml @@ -0,0 +1,5 @@ + + + https://ex.com/child-a + https://ex.com/child-b + diff --git a/lib/seo-data/fixtures-spa/page.html b/lib/seo-data/fixtures-spa/page.html new file mode 100644 index 0000000..b6cc5bb --- /dev/null +++ b/lib/seo-data/fixtures-spa/page.html @@ -0,0 +1,8 @@ + +Mon App + + + +
+ + diff --git a/lib/seo-data/fixtures-ssr/page.html b/lib/seo-data/fixtures-ssr/page.html new file mode 100644 index 0000000..6f96916 --- /dev/null +++ b/lib/seo-data/fixtures-ssr/page.html @@ -0,0 +1,8 @@ + +Lavage auto + + + +

Lavage auto à la main

+

Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique. Lavage automobile à la main à Lagny-sur-Marne, detailing et protection céramique.

+ \ No newline at end of file diff --git a/lib/seo-data/fixtures/gsc_inspect.json b/lib/seo-data/fixtures/gsc_inspect.json index 325cacd..bb5de1f 100644 --- a/lib/seo-data/fixtures/gsc_inspect.json +++ b/lib/seo-data/fixtures/gsc_inspect.json @@ -1,2 +1,11 @@ -{"inspectionResult":{"indexStatusResult":{ - "verdict":"PASS","coverageState":"Submitted and indexed","lastCrawlTime":"2026-07-01T10:00:00Z"}}} +{"inspectionResult":{ + "indexStatusResult":{ + "verdict":"PASS","coverageState":"Submitted and indexed","lastCrawlTime":"2026-07-01T10:00:00Z"}, + "richResultsResult":{"verdict":"FAIL","detectedItems":[ + {"richResultType":"Breadcrumbs","items":[{"name":"Unnamed item","issues":[]}]}, + {"richResultType":"FAQ","items":[ + {"name":"Q1","issues":[ + {"issueMessage":"Missing field 'acceptedAnswer'","severity":"ERROR"}]}, + {"name":"Q2","issues":[ + {"issueMessage":"Missing field 'acceptedAnswer'","severity":"ERROR"}, + {"issueMessage":"Unspecified image","severity":"WARNING"}]}]}]}}} diff --git a/lib/seo-data/fixtures/sitemap.xml b/lib/seo-data/fixtures/sitemap.xml new file mode 100644 index 0000000..68de6af --- /dev/null +++ b/lib/seo-data/fixtures/sitemap.xml @@ -0,0 +1,25 @@ + + + + + https://ex.com/ + weekly + + + https://ex.com/img/logo.png + Logo + + + https://ex.com/img/hero.jpeg + + + https://ex.com/services + https://ex.com/blog + https://ex.com/blog + https://ex.com/spaced + ftp://ex.com/nope + https://ex.com/bad"quote + + diff --git a/lib/seo-data/google_seo.py b/lib/seo-data/google_seo.py index d73277d..8914c67 100644 --- a/lib/seo-data/google_seo.py +++ b/lib/seo-data/google_seo.py @@ -89,13 +89,16 @@ def _gsc_session(store_path, account): return AuthorizedSession(creds) def _norm_queries(raw, dim): + # `keys` is the list the API actually returns (one entry per requested + # dimension); `key` stays as keys[0] so the single-dim consumer that reads + # it keeps working. Additive — nothing to migrate. return {"status": "ok", "source": "gsc", "dimension": dim, "rows": [ - {"key": r["keys"][0], "clicks": r.get("clicks", 0), + {"key": r["keys"][0], "keys": r["keys"], "clicks": r.get("clicks", 0), "impressions": r.get("impressions", 0), "ctr": r.get("ctr", 0), "position": r.get("position")} for r in raw.get("rows", [])]} -def queries(store_path, account, property, days=90, dim="query"): +def queries(store_path, account, property, days=90, dim="query", rows=100): raw = _mock("gsc_queries.json") if raw is None: sess = _gsc_session(store_path, account) @@ -106,14 +109,89 @@ def queries(store_path, account, property, days=90, dim="query"): import urllib.parse url = ("https://searchconsole.googleapis.com/webmasters/v3/sites/" + urllib.parse.quote(property, safe="") + "/searchAnalytics/query") + # dim accepts a comma-separated list: the API groups by several + # dimensions at once ("no limit... but you cannot group by the same + # dimension twice"), and query+page is what exposes cannibalisation. + dims = [d.strip() for d in dim.split(",") if d.strip()] r = sess.post(url, json={"startDate": start.isoformat(), "endDate": end.isoformat(), - "dimensions": [dim], "rowLimit": 100}, timeout=30) + "dimensions": dims, "rowLimit": rows}, timeout=30) if r.status_code == 429: return {"status": "degraded", "reason": "rate_limited"} r.raise_for_status() raw = r.json() return _norm_queries(raw, dim) +def _rollup_issues(items): + """Count issue instances by severity; dedupe messages (they repeat per item).""" + errors = warnings = 0 + msgs = [] + for item in items: + for iss in item.get("issues", []): + sev = iss.get("severity") + if sev == "ERROR": + errors += 1 + elif sev == "WARNING": + warnings += 1 + msg = iss.get("issueMessage") + if msg and msg not in msgs: + msgs.append(msg) + return errors, warnings, msgs + +def _norm_rich(ir): + """richResultsResult → verdict + per-type rollup. Google OMITS the key when + it detects no rich results, so absence is data, not an error: surfaced as the + synthetic verdict ABSENT (not a Google enum) rather than a missing key, which + a caller cannot tell apart from a check that never ran. PARTIAL is never + emitted — the API reserves it as unused.""" + rr = ir.get("richResultsResult") + if rr is None: + return {"verdict": "ABSENT", "types": []} + types = [] + for det in rr.get("detectedItems", []): + errors, warnings, msgs = _rollup_issues(det.get("items", [])) + types.append({"type": det.get("richResultType"), + "items": len(det.get("items", [])), + "errors": errors, "warnings": warnings, "issues": msgs}) + return {"verdict": rr.get("verdict"), "types": types} + +def _group_by_query(rows): + """query+page rows -> {query: [row, …]}. Deterministic aggregation, not + judgement: the agent must not be asked to group 1000 rows by eye.""" + by_q = {} + for r in rows: + keys = r.get("keys") or [] + if len(keys) < 2: + continue + by_q.setdefault(keys[0], []).append( + {"url": keys[1], "clicks": r["clicks"], + "impressions": r["impressions"], "position": r["position"]}) + return by_q + +def cannibal(store_path, account, property, days=90, rows=1000): + """Queries where 2+ of our own pages compete for the same term. + + Google's own data says it; nothing in this system asked. Cannibalisation + is a SERP fact, not a content-similarity guess — do not confuse it with + the 30/70 duplication rule, which has no data source here.""" + res = queries(store_path, account, property, days, "query,page", rows) + if res.get("status") != "ok": + return res + conflicts = [] + for q, pages in _group_by_query(res["rows"]).items(): + if len(pages) < 2: + continue + pages.sort(key=lambda p: p["impressions"], reverse=True) + conflicts.append({"query": q, "pages": len(pages), + "total_impressions": sum(p["impressions"] for p in pages), + "urls": pages}) + conflicts.sort(key=lambda c: c["total_impressions"], reverse=True) + return {"status": "ok", "source": "gsc", "days": days, + "rows_scanned": len(res["rows"]), + # rows_scanned == rows means the window was FULL: there may be more + # conflicts past the cut. Reported, never silently truncated. + "capped": len(res["rows"]) >= rows, + "conflict_count": len(conflicts), "conflicts": conflicts} + def inspect(store_path, account, property, url): raw = _mock("gsc_inspect.json") if raw is None: @@ -126,11 +204,15 @@ def inspect(store_path, account, property, url): return {"status": "degraded", "reason": "rate_limited"} r.raise_for_status() raw = r.json() - isr = raw["inspectionResult"]["indexStatusResult"] + ir = raw["inspectionResult"] + isr = ir["indexStatusResult"] + # rich_results rides the SAME response — Google already sent it and this + # function used to discard it. No extra call, no extra quota, no new scope. return {"status": "ok", "source": "gsc", "indexed": isr.get("verdict") == "PASS", "coverage": isr.get("coverageState"), - "last_crawl": isr.get("lastCrawlTime")} + "last_crawl": isr.get("lastCrawlTime"), + "rich_results": _norm_rich(ir)} def _cli(): try: @@ -145,7 +227,15 @@ def _cli(): pq.add_argument("--account", required=True) pq.add_argument("--property", required=True) pq.add_argument("--days", type=int, default=90) - pq.add_argument("--dim", default="query") + pq.add_argument("--dim", default="query", + help="one dimension, or a comma-separated list (query,page)") + pq.add_argument("--rows", type=int, default=100) + pn = sub.add_parser("cannibal") + pn.add_argument("--store", required=True) + pn.add_argument("--account", required=True) + pn.add_argument("--property", required=True) + pn.add_argument("--days", type=int, default=90) + pn.add_argument("--rows", type=int, default=1000) pi = sub.add_parser("inspect") pi.add_argument("--store", required=True) pi.add_argument("--account", required=True) @@ -156,7 +246,10 @@ def _cli(): print(json.dumps(crux(args.url, args.strategy), indent=2)) elif args.cmd == "queries": print(json.dumps(queries(args.store, args.account, args.property, - args.days, args.dim), indent=2)) + args.days, args.dim, args.rows), indent=2)) + elif args.cmd == "cannibal": + print(json.dumps(cannibal(args.store, args.account, args.property, + args.days, args.rows), indent=2)) elif args.cmd == "inspect": print(json.dumps(inspect(args.store, args.account, args.property, args.url), indent=2)) diff --git a/lib/seo-data/linkgraph.py b/lib/seo-data/linkgraph.py new file mode 100644 index 0000000..96b5d03 --- /dev/null +++ b/lib/seo-data/linkgraph.py @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +"""Internal link graph -> orphans + click depth. Stdlib only. + +seo-analyzer.md asks "Every important page reachable within 3 clicks?" (:613) +and "Orphan pages (no inbound internal links)?" (:616) and has never had a +command that answers either. This is that command. + +EXHAUSTIVE OR NOTHING. You cannot sample orphans: proving a page has no +inbound link means having read every other page. A partial crawl invents +orphans, and "page X has no inbound links" when it does is the worst finding +this tool could emit — it sends a client fixing what is not broken. So when +the cap bites, orphans are WITHHELD, not truncated. + +Does NOT render JS. On a client-side-rendered SPA the links are not in the +HTML, every page looks orphaned, and that is a catastrophic false positive — +so an empty link graph is REFUSED (no_links_in_html), never reported. +""" +import argparse, json +from html.parser import HTMLParser +from urllib.parse import urljoin, urlparse, urldefrag + +import sitemap as sm # sibling module: fetch + parse + +MAX_PAGES = 500 +# Extensions that are assets, not pages. Seen live: /css/main.css?v=1778157313 +ASSET_EXT = (".css", ".js", ".mjs", ".png", ".jpg", ".jpeg", ".gif", ".webp", + ".avif", ".svg", ".ico", ".woff", ".woff2", ".ttf", ".eot", + ".pdf", ".zip", ".mp4", ".webm", ".xml", ".json", ".txt", ".rss") + +class _Links(HTMLParser): + def __init__(self): + super().__init__(convert_charrefs=True) + self.hrefs = [] + def handle_starttag(self, tag, attrs): + if tag != "a": + return + for k, v in attrs: + if k == "href" and v: + self.hrefs.append(v) + +def _norm(u): + """Canonical form for graph identity. Drops the fragment, keeps the query + (?p=2 IS a different page), and unifies the trailing slash so /blog and + /blog/ are one node rather than a phantom orphan pair.""" + u = urldefrag(u)[0] + p = urlparse(u) + path = p.path or "/" + if len(path) > 1 and path.endswith("/"): + path = path[:-1] + out = "%s://%s%s" % (p.scheme, p.netloc.lower(), path) + return out + ("?" + p.query if p.query else "") + +def _page_links(base, html, host): + """Internal page links from one document. Filters what a link graph must + never contain: assets, #anchors, mailto:/tel:, and other hosts.""" + p = _Links() + try: + p.feed(html) + except Exception: + pass # tolerate malformed markup + out = set() + for h in p.hrefs: + h = h.strip() + if not h or h.startswith(("#", "mailto:", "tel:", "javascript:", "data:")): + continue + absu = urljoin(base, h) + pr = urlparse(absu) + if pr.scheme not in ("http", "https") or pr.netloc.lower() != host: + continue + if pr.path.lower().endswith(ASSET_EXT): + continue + out.add(_norm(absu)) + return out + +def _mock_pages(): + """{url: html} for tests. A single page.html fixture cannot express a + GRAPH — every node would carry identical links — so the mock is a map.""" + raw = sm._mock("pages.json") + return json.loads(raw.decode("utf-8")) if raw else None + +def _crawl(urls, host): + """Fetch each page once; return {page: {links}} plus a failure count.""" + pages = _mock_pages() + graph, failed = {}, 0 + for u in urls: + if pages is not None: + html = pages.get(u) + if html is None: + failed += 1 + continue + else: + try: + html = sm._fetch(u).decode("utf-8", "replace") + except Exception: + failed += 1 + continue + graph[_norm(u)] = _page_links(u, html, host) + return graph, failed + +def _depths(graph, root): + """BFS click-depth from the homepage. Absent = unreachable by links.""" + seen, frontier, d = {root: 0}, [root], 0 + while frontier: + d += 1 + nxt = [] + for node in frontier: + for tgt in graph.get(node, ()): + if tgt not in seen: + seen[tgt] = d + nxt.append(tgt) + frontier = nxt + return seen + +def linkgraph(sitemap_url, max_pages=MAX_PAGES): + sm_res = sm.sitemap(sitemap_url) + if sm_res.get("status") != "ok": + return sm_res # propagate the sitemap's own degrade + urls = sm_res["urls"] + capped = len(urls) > max_pages + host = urlparse(urls[0]).netloc.lower() + graph, failed = _crawl(urls[:max_pages], host) + if not graph: + return {"status": "degraded", "reason": "no_pages_fetched"} + total_links = sum(len(v) for v in graph.values()) + if total_links == 0: + # Every page orphaned is never the truth — it is a JS-rendered site. + return {"status": "degraded", "reason": "no_links_in_html", + "pages_crawled": len(graph), + "hint": "links absent from served HTML (SPA?) — see R1/R2"} + inbound = {n: 0 for n in graph} + for src, tgts in graph.items(): + for t in tgts: + if t in inbound and t != src: + inbound[t] += 1 + root = _norm("%s://%s/" % (urlparse(urls[0]).scheme, host)) + depth = _depths(graph, root) + out = {"status": "ok", "source": "linkgraph", + "pages_crawled": len(graph), "pages_failed": failed, + "total_internal_links": total_links, "capped": capped, + "max_depth": max(depth.values()) if depth else 0, + "beyond_3_clicks": sorted(n for n, d in depth.items() if d > 3), + "unreachable": sorted(n for n in graph if n not in depth)} + if capped or failed: + # A page can only be called orphaned if EVERY other page was read. + out["orphans_withheld"] = True + out["reason_withheld"] = ("crawl incomplete (capped=%s, failed=%d) — " + "an orphan from a partial crawl is a false " + "orphan" % (capped, failed)) + else: + out["orphans"] = sorted(n for n, c in inbound.items() + if c == 0 and n != root) + return out + +def _cli(): + try: + p = argparse.ArgumentParser() + p.add_argument("--url", required=True, help="sitemap URL") + p.add_argument("--max", type=int, default=MAX_PAGES) + p.add_argument("--store", default=None) # accepted+ignored + args = p.parse_args() + print(json.dumps(linkgraph(args.url, args.max), indent=2)) + except SystemExit as e: + if e.code not in (0, None): + print(json.dumps({"status": "error", "reason": "bad_usage"})) + raise + except Exception: + print(json.dumps({"status": "degraded", "reason": "unexpected_error"})) + +if __name__ == "__main__": + _cli() diff --git a/lib/seo-data/render_check.py b/lib/seo-data/render_check.py new file mode 100644 index 0000000..c98bac3 --- /dev/null +++ b/lib/seo-data/render_check.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python3 +"""Is the content in the served HTML, or painted by JS? Stdlib only. + +seo-analyzer records `RENDERING: SSR/SSG/SPA/hybrid` and then does nothing +with it. That is the gap this closes. On a client-rendered site `curl` returns +an empty shell, so every meta/H1/JSON-LD check reports "missing" and the audit +emits a page of false findings against a site that may be perfectly fine. + +The verdict is taken from what the server actually sent — not from guessing at +package.json, where a React SPA and a Next.js SSR app look identical. + +R2, not R1: this REPORTS blindness so the agent can refuse to score. It does +not render JS. No Playwright, no Chromium, no venv. +""" +import argparse, json, re +from html.parser import HTMLParser + +import sitemap as sm # sibling: _fetch / _mock + +# A shell can still carry a title + a couple of nav words. These thresholds +# separate "shell" from "page" on the two real sites measured 2026-07-17 +# (server-rendered: 1 h1, thousands of body chars) and on a hydration stub. +MIN_TEXT = 400 +MIN_H1 = 1 + +class _Doc(HTMLParser): + """Collect body text and the tags an SEO audit reads. Script/style content + is NOT text: a 200 KB React bundle would otherwise look like a rich page.""" + SKIP = ("script", "style", "noscript", "template", "svg") + + def __init__(self): + super().__init__(convert_charrefs=True) + self.text, self.h1, self.jsonld, self.meta_desc = [], 0, 0, False + self._skip = 0 + self._ld = False + + def handle_starttag(self, tag, attrs): + a = dict(attrs) + if tag in self.SKIP: + self._skip += 1 + self._ld = tag == "script" and a.get("type") == "application/ld+json" + elif tag == "h1": + self.h1 += 1 + elif tag == "meta" and a.get("name", "").lower() == "description": + self.meta_desc = bool((a.get("content") or "").strip()) + + def handle_endtag(self, tag): + if tag in self.SKIP and self._skip: + self._skip -= 1 + self._ld = False + + def handle_data(self, data): + if self._ld: + self.jsonld += 1 + elif not self._skip: + s = data.strip() + if s: + self.text.append(s) + +def _verdict(text_chars, h1, jsonld): + if text_chars >= MIN_TEXT and h1 >= MIN_H1: + return "server-rendered" + if text_chars < MIN_TEXT and h1 == 0 and jsonld == 0: + return "client-rendered" + return "partial" # shell + some SSR'd head, or thin page + +def render_check(url): + raw = sm._mock("page.html") + if raw is None: + try: + raw = sm._fetch(url) + except Exception: + return {"status": "degraded", "reason": "fetch_failed"} + html = raw.decode("utf-8", "replace") + d = _Doc() + try: + d.feed(html) + except Exception: + pass # tolerate malformed markup + text = re.sub(r"\s+", " ", " ".join(d.text)).strip() + verdict = _verdict(len(text), d.h1, d.jsonld) + out = {"status": "ok", "source": "render_check", "verdict": verdict, + "body_text_chars": len(text), "h1_in_html": d.h1, + "jsonld_in_html": d.jsonld, "meta_description_in_html": d.meta_desc, + "html_bytes": len(raw)} + if verdict != "server-rendered": + out["warning"] = ("content is not in the served HTML — curl-based " + "on-page checks will report false 'missing' findings") + return out + +def _cli(): + try: + p = argparse.ArgumentParser() + p.add_argument("--url", required=True) + p.add_argument("--store", default=None) # accepted+ignored + args = p.parse_args() + print(json.dumps(render_check(args.url), indent=2)) + except SystemExit as e: + if e.code not in (0, None): + print(json.dumps({"status": "error", "reason": "bad_usage"})) + raise + except Exception: + print(json.dumps({"status": "degraded", "reason": "unexpected_error"})) + +if __name__ == "__main__": + _cli() diff --git a/lib/seo-data/score.py b/lib/seo-data/score.py new file mode 100644 index 0000000..775c2e8 --- /dev/null +++ b/lib/seo-data/score.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""Deterministic /20 scoring from a findings list. Stdlib only. + +/harden has a real scale (SKILL.md:435 — Critique -15, Haute -8, Moyenne -3, +Basse -1, clamp [0,100]). /seo has none: every axis is felt, not computed, so +two runs over identical code can produce different scores. That is a +credibility problem on its own, and /client-handover gates on 17/20 — a +wobbling number makes the gate arbitrary. H2 sharpens it further: now that +drift reports what actually changed, a score moving on its own is visibly +noise. + +The split is the point. The LLM keeps the irreducible judgement — WHICH +findings exist and how severe each is. The arithmetic stops being judgement: +same findings in, same score out. Same principle as grouping cannibalisation +rows in the engine rather than asking a model to add up 1000 of them. + +Scale is /harden's, /5 into /20, so the whole skill family speaks one +vocabulary. +""" +import argparse, json, sys + +PENALTY = {"critique": 15, "haute": 8, "moyenne": 3, "basse": 1} + +# STEP 9 weights. FULL = 7 axes, LOCAL = 4 (off-page/social/competitive are +# not audited at that depth). +WEIGHTS = { + ("FULL", "local"): {"technical": .20, "on-page": .20, "seo-local": .25, + "off-page": .10, "social": .10, "competitive": .05, + "legal": .10}, + ("FULL", "national"): {"technical": .30, "on-page": .30, "seo-local": .05, + "off-page": .15, "social": .05, "competitive": .10, + "legal": .05}, + ("LOCAL", "local"): {"technical": .25, "on-page": .35, "seo-local": .20, + "legal": .20}, + ("LOCAL", "national"):{"technical": .35, "on-page": .45, "seo-local": .05, + "legal": .15}, +} + +def _axis_score(findings): + """100 - Σ penalties, clamped, then /5 → /20. Prevalence shifts severity + ONE step, never invents one: a finding on 1 of 12 sampled pages is not the + same defect as one on 12 of 12, and pretending otherwise is what made the + old scores unreproducible.""" + total = 0 + for f in findings: + sev = str(f.get("severity", "")).lower() + if sev not in PENALTY: + raise ValueError("unknown severity: %r" % f.get("severity")) + order = ["basse", "moyenne", "haute", "critique"] + i = order.index(sev) + aff, samp = f.get("affected"), f.get("sampled") + if isinstance(aff, int) and isinstance(samp, int) and samp > 0: + ratio = aff / samp + if ratio >= 0.5: + i = min(i + 1, len(order) - 1) # widespread → escalate + elif aff <= 1: + i = max(i - 1, 0) # isolated → de-escalate + total += PENALTY[order[i]] + return round(max(0, 100 - total) / 5.0, 1) + +def score(payload): + depth = str(payload.get("depth", "FULL")).upper() + profile = str(payload.get("profile", "local")).lower() + key = (depth, profile) + if key not in WEIGHTS: + return {"status": "error", "reason": "unknown depth/profile: %s/%s" + % (depth, profile)} + weights, axes_in = WEIGHTS[key], payload.get("axes", {}) + scored, na = {}, [] + for axis, w in weights.items(): + a = axes_in.get(axis) + if a is None or str(a.get("status", "")).lower() == "na": + na.append(axis) # N/A is not a zero + continue + try: + s = _axis_score(a.get("findings", [])) + except ValueError as e: + return {"status": "error", "reason": str(e)} + scored[axis] = {"score_20": s, "weight": w, + "findings": len(a.get("findings", []))} + if not scored: + return {"status": "degraded", "reason": "no_axis_scored"} + # Renormalise over what was actually measured. R2 mandates this for a + # client-rendered on-page axis and left it to the model to do by hand. + live = sum(v["weight"] for v in scored.values()) + for v in scored.values(): + v["weight_renormalised"] = round(v["weight"] / live, 4) + glob = sum(v["score_20"] * v["weight"] / live for v in scored.values()) + return {"status": "ok", "source": "score", "depth": depth, + "profile": profile, "axes": scored, "na": sorted(na), + "weights_renormalised": round(live, 4) != 1.0, + "global_20": round(glob, 1)} + +def _cli(): + try: + p = argparse.ArgumentParser() + p.add_argument("--findings", default="-", help="JSON path, or - for stdin") + p.add_argument("--store", default=None) # accepted+ignored + args = p.parse_args() + raw = sys.stdin.read() if args.findings == "-" else \ + open(args.findings, encoding="utf-8").read() + print(json.dumps(score(json.loads(raw)), indent=2)) + except SystemExit as e: + if e.code not in (0, None): + print(json.dumps({"status": "error", "reason": "bad_usage"})) + raise + except Exception: + # Unlike the fetch verbs this is pure arithmetic: a degrade here means + # malformed input, never a network fact. + print(json.dumps({"status": "error", "reason": "bad_findings_json"})) + +if __name__ == "__main__": + _cli() diff --git a/lib/seo-data/seo-data.test.sh b/lib/seo-data/seo-data.test.sh index 16e8a81..3c54ba6 100644 --- a/lib/seo-data/seo-data.test.sh +++ b/lib/seo-data/seo-data.test.sh @@ -57,12 +57,182 @@ has "queries position field" "$Q" '"position": 6.3' I="$(SEO_DATA_MOCK_DIR="$MOCK" python3 "$SD/google_seo.py" inspect \ --store "$S2" --account client-a --property sc-domain:ex.com --url https://ex.com/x)" has "inspect indexed true" "$I" '"indexed": true' +# rich_results rides the same URL-Inspection response (no extra call/quota) +has "rich verdict surfaced" "$I" '"verdict": "FAIL"' +has "rich type breadcrumbs" "$I" '"type": "Breadcrumbs"' +has "rich type faq" "$I" '"type": "FAQ"' +has "rich counts error severity" "$I" '"errors": 2' +has "rich counts warn severity" "$I" '"warnings": 1' +has "rich keeps issue message" "$I" "Missing field 'acceptedAnswer'" +# same issueMessage repeats across items — the rollup must collapse it to one +NMSG="$(printf '%s' "$I" | grep -cF "Missing field 'acceptedAnswer'")" +[ "$NMSG" = "1" ] && ok "rich dedupes issue messages" \ + || no "rich dedupes issue messages" "got $NMSG occurrences" +# Google OMITS richResultsResult when it detects none — absence is data, and +# must not KeyError nor vanish into a missing key +NR="$(SEO_DATA_MOCK_DIR="$SD/fixtures-norich" python3 "$SD/google_seo.py" inspect \ + --store "$S2" --account client-a --property sc-domain:ex.com --url https://ex.com/x)" +has "no-rich → synthetic ABSENT" "$NR" '"verdict": "ABSENT"' +has "no-rich keeps index status" "$NR" '"indexed": true' +hasnt "no-rich emits no PARTIAL" "$NR" 'PARTIAL' DEG="$(env -u SEO_DATA_MOCK_DIR python3 "$SD/google_seo.py" queries \ --store "$TMP2/none.json" --account nobody --property sc-domain:ex.com)" has "gsc degrades w/o creds" "$DEG" '"status": "degraded"' has "gsc degrade reason" "$DEG" 'no_credentials' rm -rf "$TMP2" +echo "── cannibalisation ──" +# `keys` is additive: the single-dim consumer that reads `key` must not break +has "queries keeps key (compat)" "$Q" '"key": "plombier paris"' +has "queries adds keys list" "$Q" '"keys"' +CAN="$(SEO_DATA_MOCK_DIR="$SD/fixtures-cannibal" python3 "$SD/google_seo.py" cannibal \ + --store "$S2" --account client-a --property sc-domain:ex.com)" +has "cannibal ok" "$CAN" '"status": "ok"' +# fixture: 3 pages on "urgence fuite", 2 on "plombier paris", 1 on "devis" +has "cannibal finds 2 conflicts" "$CAN" '"conflict_count": 2' +has "cannibal counts pages" "$CAN" '"pages": 3' +has "cannibal sums impressions" "$CAN" '"total_impressions": 2400' +hasnt "single-page query is not a conflict" "$CAN" 'devis plomberie' +# biggest conflict first, and inside it the strongest page first +CAN_FIRST="$(printf '%s' "$CAN" | python3 -c 'import sys,json; d=json.load(sys.stdin); print(d["conflicts"][0]["query"], d["conflicts"][0]["urls"][0]["url"])')" +check_first() { [ "$1" = "$2" ] && ok "$3" || no "$3" "got[$1]"; } +check_first "$CAN_FIRST" "urgence fuite https://ex.com/urgence" "cannibal ranks by impact" +has "cannibal reports the cap" "$CAN" '"capped": false' + +echo "── sitemap ──" +SM="$(SEO_DATA_MOCK_DIR="$MOCK" python3 "$SD/sitemap.py" --url https://ex.com/sitemap.xml)" +has "sitemap ok" "$SM" '"status": "ok"' +has "sitemap not an index" "$SM" '"index": false' +# fixture holds 8 : 1 empty, blog twice, ftp:// and a quoted one to drop +has "sitemap dedupes" "$SM" '"count": 4' +has "sitemap counts drops" "$SM" '"dropped": 2' +has "sitemap strips whitespace" "$SM" '"https://ex.com/spaced"' +hasnt "sitemap drops non-http" "$SM" 'ftp://' +hasnt "sitemap drops shell-meta" "$SM" 'bad"quote' +# namespace-agnostic: real sitemaps carry sitemaps.org xmlns (+ xhtml here) +has "sitemap reads namespaced" "$SM" '"https://ex.com/services"' +# REGRESSION: also ends with '}loc'. An endswith test counted image +# sitemap entries as pages — a real native site returned 27 for 24 , and +# img/logo.png was about to be sampled and audited as a page. +hasnt "image:loc is not a page" "$SM" '/img/logo.png' +hasnt "image:loc jpeg not a page" "$SM" '/img/hero.jpeg' +has "image ns does not inflate count" "$SM" '"count": 4' + +IDX="$(SEO_DATA_MOCK_DIR="$SD/fixtures-sitemap-index" python3 "$SD/sitemap.py" \ + --url https://ex.com/sitemap.xml)" +has "sitemapindex detected" "$IDX" '"index": true' +has "sitemapindex fans out" "$IDX" '"children_read": 2' +has "sitemapindex no child fail" "$IDX" '"children_failed": 0' +has "sitemapindex yields urls" "$IDX" '"https://ex.com/child-a"' + +# A sitemap NEVER has a DTD. Refused at the door: xml.etree does not expand +# external entities but IS billion-laughs-vulnerable, and the 20MB read ceiling +# bounds the input, not the expansion. Refusing beats depending on the parser, +# and keeps this module stdlib-only (no defusedxml, no venv). +DTD="$(SEO_DATA_MOCK_DIR="$SD/fixtures-sitemap-dtd" python3 "$SD/sitemap.py" \ + --url https://ex.com/sitemap.xml)" +has "billion-laughs refused" "$DTD" '"status": "degraded"' +has "dtd reason is distinct" "$DTD" 'unsafe_xml_dtd' +hasnt "dtd never parsed" "$DTD" '"count"' + +echo "── render_check (R2) ──" +SPA="$(SEO_DATA_MOCK_DIR="$SD/fixtures-spa" python3 "$SD/render_check.py" \ + --url https://spa.example/)" +has "spa → client-rendered" "$SPA" '"verdict": "client-rendered"' +has "spa has no h1 in html" "$SPA" '"h1_in_html": 0' +has "spa warns about false negs" "$SPA" 'false' +# the shell carries a fat window.__INITIAL_STATE__ script: script text is NOT +# page text, or a 200KB React bundle would read as a rich page +has "script text is not content" "$SPA" '"body_text_chars": 7' +SSR="$(SEO_DATA_MOCK_DIR="$SD/fixtures-ssr" python3 "$SD/render_check.py" \ + --url https://ssr.example/)" +has "ssr → server-rendered" "$SSR" '"verdict": "server-rendered"' +has "ssr counts jsonld" "$SSR" '"jsonld_in_html": 1' +has "ssr sees meta description" "$SSR" '"meta_description_in_html": true' +hasnt "ssr emits no warning" "$SSR" 'warning' + +echo "── linkgraph ──" +LG="$(SEO_DATA_MOCK_DIR="$SD/fixtures-linkgraph" python3 "$SD/linkgraph.py" \ + --url https://ex.com/sitemap.xml)" +has "linkgraph ok" "$LG" '"status": "ok"' +has "linkgraph crawls all" "$LG" '"pages_crawled": 7' +# THE test: a planted page nobody links to must be found. Two live sites both +# returned zero orphans; without this, "always returns []" looks identical. +has "finds the planted orphan" "$LG" '"https://ex.com/orphan"' +has "orphan is also unreachable" "$LG" '"unreachable"' +has "depth chain measured" "$LG" '"max_depth": 4' +has "flags >3 clicks" "$LG" '"https://ex.com/deepest"' +# home links: /a and /b only. anchor, .css?v=, mailto:, tel:, external, .png +# are not page links — 9 total across the 7 pages. +has "filters non-page links" "$LG" '"total_internal_links": 9' +hasnt "no external host" "$LG" 'other.com' +hasnt "no asset link" "$LG" 'main.css' +hasnt "no image link" "$LG" 'logo.png' +# /b/ in the markup vs /b in the sitemap must be ONE node, not a phantom orphan +hasnt "trailing slash unified" "$LG" '"https://ex.com/b/"' + +# An orphan from a partial crawl is a false orphan: withhold, do not truncate. +CAP="$(SEO_DATA_MOCK_DIR="$SD/fixtures-linkgraph" python3 "$SD/linkgraph.py" \ + --url https://ex.com/sitemap.xml --max 3)" +has "cap is reported" "$CAP" '"capped": true' +has "capped withholds orphans" "$CAP" '"orphans_withheld": true' +hasnt "capped emits no orphans" "$CAP" '"orphans":' + +echo "── score (I7) ──" +sc() { printf '%s' "$1" | python3 "$SD/score.py" --findings -; } +# technical: haute(-8) + moyenne(-3) = 100-11 = 89 → 17.8 +B='{"depth":"FULL","profile":"local","axes":{"technical":{"findings":[{"severity":"haute"},{"severity":"moyenne"}]},"seo-local":{"findings":[]},"off-page":{"findings":[]},"social":{"findings":[]},"competitive":{"findings":[]},"legal":{"findings":[]},"on-page":{"findings":[]}}}' +R="$(sc "$B")" +has "harden scale, /5 into /20" "$R" '"score_20": 17.8' +has "no findings = 20" "$R" '"score_20": 20.0' +has "nothing renormalised" "$R" '"weights_renormalised": false' +# THE point of I7: same findings in, same score out +A1="$(sc "$B" | python3 -c 'import sys,json;print(json.load(sys.stdin)["global_20"])')" +A2="$(sc "$B" | python3 -c 'import sys,json;print(json.load(sys.stdin)["global_20"])')" +[ "$A1" = "$A2" ] && ok "score is reproducible" || no "score is reproducible" "$A1 vs $A2" +# N/A is not a zero, and R2 mandated renormalising by hand — now computed +NA='{"depth":"FULL","profile":"local","axes":{"technical":{"findings":[]},"on-page":{"status":"na"},"seo-local":{"findings":[]},"off-page":{"status":"na"},"social":{"findings":[]},"competitive":{"findings":[]},"legal":{"findings":[]}}}' +RN="$(sc "$NA")" +has "na axes listed" "$RN" '"on-page"' +has "renormalisation flagged" "$RN" '"weights_renormalised": true' +# all axes 20 → global must stay 20: N/A must not drag the mean down +has "na is not a zero" "$RN" '"global_20": 20.0' +# prevalence shifts severity ONE step, both ways +WIDE='{"depth":"LOCAL","profile":"local","axes":{"technical":{"findings":[{"severity":"moyenne","affected":10,"sampled":12}]},"on-page":{"findings":[]},"seo-local":{"findings":[]},"legal":{"findings":[]}}}' +ONE='{"depth":"LOCAL","profile":"local","axes":{"technical":{"findings":[{"severity":"moyenne","affected":1,"sampled":12}]},"on-page":{"findings":[]},"seo-local":{"findings":[]},"legal":{"findings":[]}}}' +has "widespread escalates (-8)" "$(sc "$WIDE")" '"score_20": 18.4' +has "isolated de-escalates (-1)" "$(sc "$ONE")" '"score_20": 19.8' +# malformed input is an error, never a silently wrong number +has "unknown severity rejected" "$(sc '{"depth":"FULL","profile":"local","axes":{"technical":{"findings":[{"severity":"bogus"}]}}}')" '"status": "error"' +has "unknown profile rejected" "$(sc '{"depth":"FULL","profile":"martian","axes":{}}')" '"status": "error"' +has "garbage json is an error" "$(sc 'not json')" '"status": "error"' + +echo "── drift (H2) ──" +DH="$(mktemp -d)" +D1="$(HOME="$DH" SEO_DATA_MOCK_DIR="$SD/fixtures-drift-v1" python3 "$SD/drift.py" \ + --url https://ex.com/sitemap.xml)" +has "first run is a baseline" "$D1" '"baseline": true' +has "baseline captures pages" "$D1" '"pages": 3' +hasnt "baseline diffs nothing" "$D1" '"regressions"' +# v2: canonical lost on /a, h1+jsonld lost on /, title reworded, /gone removed, +# /neuve added. Losses are regressions; a reworded title is not. +D2="$(HOME="$DH" SEO_DATA_MOCK_DIR="$SD/fixtures-drift-v2" python3 "$SD/drift.py" \ + --url https://ex.com/sitemap.xml)" +has "second run diffs" "$D2" '"baseline": false' +has "detects removed url" "$D2" '"https://ex.com/gone"' +has "detects added url" "$D2" '"https://ex.com/neuve"' +has "lost canonical = regression" "$D2" '"canonical"' +has "lost h1 = regression" "$D2" '"h1_count"' +has "lost jsonld = regression" "$D2" '"jsonld_types"' +# the classification IS the feature: losing a signal != changing one +NREG="$(printf '%s' "$D2" | python3 -c 'import sys,json; print(len(json.load(sys.stdin)["regressions"]))')" +NCHG="$(printf '%s' "$D2" | python3 -c 'import sys,json; print(len(json.load(sys.stdin)["changes"]))')" +[ "$NREG" = "3" ] && ok "3 losses classed as regressions" \ + || no "3 losses classed as regressions" "got $NREG" +[ "$NCHG" = "1" ] && ok "reworded title is a change, not a regression" \ + || no "reworded title is a change, not a regression" "got $NCHG" +rm -rf "$DH" + echo "── fetch.sh ──" FETCH="$SD/fetch.sh" # SEO_DATA_ENV_FILE=/dev/null: tests must NEVER source the real ~/.claude/.env — diff --git a/lib/seo-data/sitemap.py b/lib/seo-data/sitemap.py new file mode 100644 index 0000000..99454ab --- /dev/null +++ b/lib/seo-data/sitemap.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +"""Sitemap discovery -> normalized JSON. Stdlib only: no venv, no requests, no +auth. Gives STEP 9 COVERAGE the denominator it was told to report and never +had, and STEP 5 a real sampling frame instead of "5-15 key pages" chosen by +eye. + +Deliberately NOT a security boundary. urllib fetches these URLs, so nothing +here reaches a shell and there is no injection surface to guard. The consumer +is different: seo-analyzer interpolates URLs into curl, so IT must run +lib/url-guard.sh at the point of use (same pattern as the sameAs check). +Duplicating the guard here would just add a second copy to drift. `_sane` +below is a cheap garbage filter, not that guard. +""" +import argparse, gzip, json, os +from urllib.parse import urlparse + +MAX_URLS = 50000 # sitemaps.org caps one file at 50k +MAX_CHILDREN = 50 # sitemapindex fan-out cap: bound the work, report the cut +TIMEOUT = 20 + +def _mock(name): + d = os.environ.get("SEO_DATA_MOCK_DIR") + if not d: + return None + path = os.path.join(d, name) + if not os.path.exists(path): + return None + with open(path, "rb") as f: + return f.read() + +def _fetch(url): + from urllib.request import urlopen, Request # stdlib, lazy + req = Request(url, headers={"User-Agent": "claude-seo-data/1.0"}) + with urlopen(req, timeout=TIMEOUT) as r: # nosec: audited target + raw = r.read(20 * 1024 * 1024) # 20 MB ceiling + if raw[:2] == b"\x1f\x8b": # sitemap.xml.gz is common + raw = gzip.decompress(raw) + return raw + +class UnsafeXML(Exception): + """A DTD reached the parser. Refused before parsing, not mitigated after.""" + +def _refuse_dtd(raw): + """A sitemap NEVER has a DTD: sitemaps.org is then . + So refuse any doctype/entity outright, at the door. + + This is the reason we do not pull in defusedxml. The stdlib parser is not + the problem for XXE — xml.etree.ElementTree does not expand external + entities, it raises on them — but it IS vulnerable to billion-laughs, where + a 1 KB document expands to gigabytes in RAM. The 20 MB read ceiling bounds + the input, not the expansion. Rejecting the construct beats depending on + the parser's internals, and keeps this module stdlib-only: no venv, same as + google_seo.py's mock/degrade paths. A sitemap with a DTD is not a sitemap + we want anyway. + """ + head = raw[:4096].lstrip()[:2048].upper() + if b": sitemaps.org namespace, or namespace-less. + + NOT or . Those live in Google's extension + namespaces and name an ASSET inside a , not a page of its own. An + endswith('}loc') test matches them too — that shipped, and a real site + caught it: 24 + 3 came back as a count of 27, so the + COVERAGE denominator was 12.5% too high and img/logo.png was about to be + sampled and audited as a page. + """ + return tag == SITEMAP_NS + "loc" or tag == "loc" + +def _locs(raw): + """(page texts, is_sitemapindex). + + Walks the DIRECT children of each / rather than root.iter(): + that alone excludes , and the namespace test above + is the second lock. XML comments iterate as elements with no children, so + they fall through harmlessly. + """ + import xml.etree.ElementTree as ET # stdlib, lazy + _refuse_dtd(raw) + root = ET.fromstring(raw) + is_index = root.tag.endswith("sitemapindex") + out = [] + for entry in root: # | + for child in entry: # direct children only + if _is_page_loc(child.tag): + text = (child.text or "").strip() + if text: + out.append(text) + break # one per entry + return out, is_index + +def _sane(u): + """Cheap garbage filter — NOT lib/url-guard.sh. Drops what could never be a + real page URL; the consumer still guards before curling.""" + if not u or len(u) > 2048: + return False + if any(c in u for c in '\n\r\t "\'\\`$<>{}|^'): + return False + return urlparse(u).scheme in ("http", "https") + +def _expand(children): + """Fetch each child sitemap of an index. A child that fails is skipped and + counted, never fatal: one dead child must not lose the other 49.""" + urls, ok, failed = [], 0, 0 + for c in children: + raw = _mock("sitemap_child.xml") + if raw is None: + try: + raw = _fetch(c) + except Exception: + failed += 1 + continue + try: + sub, _ = _locs(raw) + except Exception: + failed += 1 + continue + urls.extend(sub) + ok += 1 + return urls, ok, failed + +def sitemap(url): + raw = _mock("sitemap.xml") + if raw is None: + try: + raw = _fetch(url) + except Exception: + return {"status": "degraded", "reason": "fetch_failed"} + try: + locs, is_index = _locs(raw) + except UnsafeXML: + # Distinct from parse_failed on purpose: this one is a finding, not a + # glitch. A sitemap carrying a DTD is either broken tooling or someone + # aiming a billion-laughs at the auditor. + return {"status": "degraded", "reason": "unsafe_xml_dtd"} + except Exception: + return {"status": "degraded", "reason": "parse_failed"} + out = {"status": "ok", "source": "sitemap", "index": is_index} + if is_index: + out["children_total"] = len(locs) + kids, ok, failed = _expand(locs[:MAX_CHILDREN]) + out["children_read"], out["children_failed"] = ok, failed + if len(locs) > MAX_CHILDREN: # say what was cut + out["children_skipped"] = len(locs) - MAX_CHILDREN + locs = kids + seen, urls, dropped = set(), [], 0 + for u in locs: + if not _sane(u): + dropped += 1 + continue + if u in seen: + continue + seen.add(u) + urls.append(u) + if len(urls) > MAX_URLS: + out["truncated"] = len(urls) - MAX_URLS + urls = urls[:MAX_URLS] + out["count"], out["dropped"], out["urls"] = len(urls), dropped, urls + if not urls: + return {"status": "degraded", "reason": "no_urls"} + return out + +def _cli(): + try: + p = argparse.ArgumentParser() + p.add_argument("--url", required=True) + p.add_argument("--store", default=None) # accepted+ignored: uniform dispatch + args = p.parse_args() + print(json.dumps(sitemap(args.url), indent=2)) + except SystemExit as e: + if e.code not in (0, None): + print(json.dumps({"status": "error", "reason": "bad_usage"})) + raise + except Exception: + # Same fail-open contract as google_seo.py: never a traceback, never + # empty stdout, exit 0 so the audit degrades instead of dying. + print(json.dumps({"status": "degraded", "reason": "unexpected_error"})) + +if __name__ == "__main__": + _cli() diff --git a/lib/source-scope.sh b/lib/source-scope.sh new file mode 100644 index 0000000..ad63839 --- /dev/null +++ b/lib/source-scope.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# Emit the directory exclusions that separate SOURCE from BUILD OUTPUT. +# +# EXCL="$(bash ~/.claude/lib/source-scope.sh grep)" +# grep -rl "gtag" $EXCL --include="*.html" . # note: $EXCL unquoted +# +# mapfile -t FEXCL < <(bash ~/.claude/lib/source-scope.sh findargs) +# find . "${FEXCL[@]}" -iname '*.jpg' -printf '%s %p\n' # quoted array! +# +# findargs emits ONE TOKEN PER LINE and MUST be consumed through a quoted +# array. A flat string does not work: `find . $FEXCL ...` lets the shell glob +# `*/dist/*` against the CWD before find ever sees it, and the matches are then +# passed as search PATHS. Measured on zenquality: that turned 90 hits into 135 +# and kept every dist/ file. The array form passes each token literally. +# +# WHY: grep and find disagree about what is in the repo, and seo-analyzer uses +# both. +# +# grep → Claude Code installs a shell function routing grep to ugrep with +# `--ignore-files`, i.e. .gitignore-aware. A gitignored dist/ is +# invisible to it when recursing from `.`. Verified 2026-07-17. +# find → knows nothing about .gitignore. It sees everything. +# +# So on zenquality (Astro, dist/ gitignored, built locally) the spec's image +# audit at seo-analyzer.md:497 returns 92 images of which 45 live in dist/ — +# every asset listed twice, source and generated copy, identical bytes. Two +# real consequences: +# 1. "top 20 by size" is half generated duplicates: ~10 real images audited +# while 20 are claimed. +# 2. Batch C (`cwebp -q 80 -o .webp`) can target dist/og-image.png. +# The .webp lands in dist/ and the `npm run build` that /seo runs to VERIFY +# the fix erases it. The fix lands, verification passes, nothing survives. +# +# The grep side is already safe by accident — do NOT "fix" it to match find. +# `grep` mode below is defence in depth for the cases the shim misses: a repo +# that COMMITS its build output (no .gitignore entry to honour), or a directory +# that is not a git repo at all. +# +# `public/` is deliberately NOT in the always-list: it is SOURCE for +# Astro/Vite/Next and holds the very files this audit checks — favicon.ico, +# apple-touch-icon.png, robots.txt, OG images. It is build OUTPUT only for +# Hugo and Gatsby, detected below. Blanket-excluding it would blind the audit +# to its own resource checks. +# +# Exclusions are by NAME, not path, so a monorepo's frontend/dist is caught +# exactly like a root ./dist. +set -uo pipefail + +_die() { echo "source-scope: $1" >&2; exit 2; } + +# Build output + tool caches. Never source. +ALWAYS=(node_modules .git dist build .next .nuxt .output _site .astro + .svelte-kit .cache out coverage .vercel .netlify .turbo) + +# public/ is output for exactly these two generators. +_public_is_output() { + find . -maxdepth 3 \( -name "gatsby-config.js" -o -name "gatsby-config.ts" \ + -o -name "gatsby-config.mjs" -o -name "hugo.toml" -o -name "hugo.yaml" \ + -o -name "hugo.json" \) 2>/dev/null | read -r _ && return 0 + # Hugo's legacy config.toml is ambiguous on its own — pair it with archetypes/ + [ -d ./archetypes ] && [ -f ./config.toml ] && return 0 + return 1 +} + +_list() { + printf '%s\n' "${ALWAYS[@]}" + _public_is_output && printf 'public\n' + return 0 +} + +case "${1:-}" in + list) _list ;; + # Safe unquoted: --exclude-dir=NAME carries no glob character. + grep) _list | while read -r d; do printf -- '--exclude-dir=%s ' "$d"; done; echo ;; + # One token per line — consume with mapfile + a QUOTED array, never a flat + # string (see header: the shell would glob */dist/* against the CWD). + findargs) _list | while read -r d; do printf '!\n-path\n*/%s/*\n' "$d"; done ;; + *) _die "usage: source-scope.sh {list|grep|findargs}" ;; +esac diff --git a/lib/tests/source-scope.test.sh b/lib/tests/source-scope.test.sh new file mode 100644 index 0000000..b674b6c --- /dev/null +++ b/lib/tests/source-scope.test.sh @@ -0,0 +1,76 @@ +#!/usr/bin/env bash +# lib/tests/source-scope.test.sh +set -u +S="$(cd "$(dirname "$0")/../.." && pwd)/lib/source-scope.sh" +pass=0; fail=0 +check() { if [ "$2" = "$3" ]; then pass=$((pass+1)); else fail=$((fail+1)); + printf 'FAIL %s: got[%s] want[%s]\n' "$1" "$2" "$3"; fi; } +# does `list` (run inside dir $1) contain the name $2? +listed() { ( cd "$1" && bash "$S" list 2>/dev/null | grep -qxF "$2" ) \ + && echo yes || echo no; } + +TMP="$(mktemp -d)" + +# --- always-excluded build output + caches --- +mkdir -p "$TMP/plain" +for d in node_modules .git dist build .next .nuxt .output _site .astro \ + .svelte-kit .cache out coverage .vercel .netlify .turbo; do + check "A-$d-listed" "$(listed "$TMP/plain" "$d")" yes +done + +# --- public/ is SOURCE by default: Astro/Vite/Next keep favicon.ico, +# apple-touch-icon.png and robots.txt there, and the audit checks them --- +check B1-public-kept-by-default "$(listed "$TMP/plain" public)" no + +# --- public/ is OUTPUT for Gatsby and Hugo only --- +mkdir -p "$TMP/gatsby"; : > "$TMP/gatsby/gatsby-config.js" +check B2-gatsby-js "$(listed "$TMP/gatsby" public)" yes +mkdir -p "$TMP/gatsby2"; : > "$TMP/gatsby2/gatsby-config.ts" +check B3-gatsby-ts "$(listed "$TMP/gatsby2" public)" yes +mkdir -p "$TMP/hugo"; : > "$TMP/hugo/hugo.toml" +check B4-hugo-toml "$(listed "$TMP/hugo" public)" yes +mkdir -p "$TMP/hugo2"; : > "$TMP/hugo2/hugo.yaml" +check B5-hugo-yaml "$(listed "$TMP/hugo2" public)" yes +# legacy config.toml alone is ambiguous (many tools use it) — needs archetypes/ +mkdir -p "$TMP/amb"; : > "$TMP/amb/config.toml" +check B6-config-toml-alone-is-ambiguous "$(listed "$TMP/amb" public)" no +mkdir -p "$TMP/hugo3/archetypes"; : > "$TMP/hugo3/config.toml" +check B7-config-toml-plus-archetypes "$(listed "$TMP/hugo3" public)" yes + +# --- grep mode: flags, and no glob character (safe unquoted) --- +G="$(cd "$TMP/plain" && bash "$S" grep)" +case "$G" in *--exclude-dir=dist*) check C1-grep-has-dist ok ok ;; + *) check C1-grep-has-dist "missing" ok ;; esac +case "$G" in *"*"*) check C2-grep-has-no-glob "has-glob" ok ;; + *) check C2-grep-has-no-glob ok ok ;; esac + +# --- findargs: one token per line, 3 tokens per dir --- +N="$(cd "$TMP/plain" && bash "$S" findargs | wc -l)" +D="$(cd "$TMP/plain" && bash "$S" list | wc -l)" +check D1-findargs-3-tokens-per-dir "$N" "$((D * 3))" +check D2-findargs-first-token "$(cd "$TMP/plain" && bash "$S" findargs | head -1)" '!' + +# --- FUNCTIONAL: the array form actually excludes build output --- +# A flat unquoted string does NOT work here: the shell globs */dist/* against +# the CWD and passes the matches to find as search paths. Measured on a real +# repo, that turned 90 hits into 135 and kept every dist/ file. +W="$TMP/work"; mkdir -p "$W/src" "$W/dist" "$W/public" "$W/node_modules" +: > "$W/src/a.png"; : > "$W/dist/a.png"; : > "$W/public/favicon.ico" +: > "$W/node_modules/dep.png" +cd "$W" || exit 1 +mapfile -t FEXCL < <(bash "$S" findargs) +check E1-excludes-dist "$(find . "${FEXCL[@]}" -name 'a.png' | grep -c '/dist/')" 0 +check E2-keeps-src "$(find . "${FEXCL[@]}" -name 'a.png' | grep -c '/src/')" 1 +check E3-excludes-nodem "$(find . "${FEXCL[@]}" -name '*.png' | grep -c 'node_modules')" 0 +# public/ survives: the audit's own resource checks live there +check E4-keeps-public "$(find . "${FEXCL[@]}" -name 'favicon.ico' | wc -l)" 1 +cd / || exit 1 + +# --- usage --- +bash "$S" >/dev/null 2>&1; check X1-no-args "$?" 2 +bash "$S" bogus >/dev/null 2>&1; check X2-bad-verb "$?" 2 +# `find` was renamed to `findargs` when the flat-string form proved unsafe +bash "$S" find >/dev/null 2>&1; check X3-old-find-verb-gone "$?" 2 + +rm -rf "$TMP" +printf 'PASS=%s FAIL=%s\n' "$pass" "$fail"; [ "$fail" -eq 0 ] diff --git a/lib/tests/url-guard.test.sh b/lib/tests/url-guard.test.sh new file mode 100644 index 0000000..679c832 --- /dev/null +++ b/lib/tests/url-guard.test.sh @@ -0,0 +1,74 @@ +#!/usr/bin/env bash +# lib/tests/url-guard.test.sh +set -u +G="$(cd "$(dirname "$0")/../.." && pwd)/lib/url-guard.sh" +pass=0; fail=0 +check() { if [ "$2" = "$3" ]; then pass=$((pass+1)); else fail=$((fail+1)); + printf 'FAIL %s: got[%s] want[%s]\n' "$1" "$2" "$3"; fi; } +# rc of a guard call, output discarded +rc() { bash "$G" "$1" "$2" >/dev/null 2>&1; return $?; } +# stdout of a guard call (empty on refusal) +out() { bash "$G" "$1" "$2" 2>/dev/null; } + +# --- hosts that must pass, echoing back unchanged --- +rc host "example.com"; check H1-plain "$?" 0 +rc host "www.sub.example.co.uk"; check H2-subdomains "$?" 0 +rc host "my-site.fr"; check H3-hyphen "$?" 0 +check H4-echoes-input "$(out host example.com)" "example.com" + +# --- shell metacharacters: the reason this guard exists --- +# Inside the double quotes seo-analyzer.md:257 uses, $ ` \ " break out. +rc host 'x$(id)'; check H5-cmdsubst "$?" 2 +rc host 'x`id`'; check H6-backtick "$?" 2 +rc host 'x;id'; check H7-semicolon "$?" 2 +rc host 'x|id'; check H8-pipe "$?" 2 +rc host 'x&id'; check H9-ampersand "$?" 2 +rc host 'x"'; check H10-dquote "$?" 2 +rc host "x'"; check H11-squote "$?" 2 +rc host 'x\y'; check H12-backslash "$?" 2 +rc host 'x y'; check H13-space "$?" 2 +rc host 'a +b'; check H14-newline "$?" 2 +# the real payload: read the OAuth vault into a request +rc host 'x$(cat ${HOME}/.claude/.env)'; check H15-env-exfil "$?" 2 +check H16-refusal-is-silent "$(out host 'x$(id)')" "" + +# --- literal local / private / metadata targets --- +rc host "localhost"; check L1-localhost "$?" 2 +rc host "LOCALHOST"; check L2-case-folded "$?" 2 +rc host "127.0.0.1"; check L3-loopback "$?" 2 +rc host "10.1.2.3"; check L4-private-10 "$?" 2 +rc host "192.168.1.1"; check L5-private-192 "$?" 2 +rc host "172.16.0.1"; check L6-private-172-lo "$?" 2 +rc host "172.31.255.254"; check L7-private-172-hi "$?" 2 +rc host "172.32.0.1"; check L8-172-32-is-public "$?" 0 +rc host "169.254.169.254"; check L9-link-local "$?" 2 +rc host "metadata.google.internal"; check L10-gcp-metadata "$?" 2 +rc host "0.0.0.0"; check L11-any-addr "$?" 2 +rc host "printer.local"; check L12-mdns "$?" 2 + +# --- urls --- +rc url "https://example.com/"; check U1-https "$?" 0 +rc url "http://example.com/a/b?x=1&y=2"; check U2-query "$?" 0 +rc url "https://example.com:8443/p"; check U3-port "$?" 0 +rc url "https://example.com/a%20b#frag"; check U4-pct-and-frag "$?" 0 +check U5-echoes-input "$(out url https://example.com/x)" "https://example.com/x" +rc url "ftp://example.com/"; check U6-ftp "$?" 2 +rc url "file:///etc/passwd"; check U7-file "$?" 2 +rc url "gopher://example.com/"; check U8-gopher "$?" 2 +rc url "example.com"; check U9-no-scheme "$?" 2 +rc url 'https://example.com/$(id)'; check U10-cmdsubst "$?" 2 +rc url 'https://example.com/`id`'; check U11-backtick "$?" 2 +rc url "https://localhost/x"; check U12-local "$?" 2 +rc url "https://127.0.0.1:8080/admin"; check U13-loopback "$?" 2 +# authority confusion: the real host is after the @, not before it +rc url "https://trusted.com@127.0.0.1/"; check U14-userinfo-local "$?" 2 +rc url "https://trusted.com@evil.com/"; check U15-userinfo-any "$?" 2 + +# --- usage --- +rc host ""; check X1-host-empty "$?" 2 +bash "$G" >/dev/null 2>&1; check X2-no-args "$?" 2 +bash "$G" bogus x >/dev/null 2>&1; check X3-bad-verb "$?" 2 +bash "$G" host a b >/dev/null 2>&1; check X4-extra-args "$?" 2 + +printf 'PASS=%s FAIL=%s\n' "$pass" "$fail"; [ "$fail" -eq 0 ] diff --git a/lib/url-guard.sh b/lib/url-guard.sh new file mode 100644 index 0000000..61a65cf --- /dev/null +++ b/lib/url-guard.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# Validate a host or URL BEFORE it reaches a shell command or curl. +# Echoes the value on stdout when safe; exits 2 with a reason on stderr. +# +# HOST="$(bash ~/.claude/lib/url-guard.sh host "$RAW")" || exit 2 +# URL="$(bash ~/.claude/lib/url-guard.sh url "$RAW")" || exit 2 +# +# WHY: /seo and /geo interpolate externally-supplied strings into ~10 curl +# commands (seo-analyzer.md:254+, geo-analyzer.md:248+). Today $DOMAIN is typed +# by the operator, so the risk is self-inflicted. The sitemap crawl (C1) changes +# that: URLs then come from the TARGET'S OWN SERVER — a remote file whose bytes +# reach a shell. Inside the double quotes those curls use, the characters that +# break out are $ ` \ " — so a of +# https://x/$(cat ${HOME}/.claude/.env) +# would read GOOGLE_OAUTH_CLIENT_SECRET and CRUX_API_KEY straight out of the +# vault and into a request. Allowlist, per CLAUDE.md: explicit allowlist beats +# implicit denylist. +# +# NOT COVERED, deliberately: DNS-level SSRF. A public hostname that RESOLVES to +# a private address passes this guard. Closing that needs resolve-then-pin at +# the HTTP layer; curl in a shell cannot do it without a TOCTOU window between +# the check and the connection. Literal local targets ARE rejected below. The +# omission is stated rather than silent — see lib/seo-data/README.md. +set -uo pipefail + +_die() { echo "url-guard: $1" >&2; exit 2; } + +# Whole-string charset guards: C locale + POSIX `case`, the same shape as +# fetch.sh:25 _label_safe. Newline-proof and locale-independent, unlike a +# per-line grep. No `$` or backtick inside the patterns, so nothing expands. +_host_charset_ok() ( LC_ALL=C; case "$1" in + ''|[!A-Za-z0-9]*|*[!A-Za-z0-9.-]*) exit 1 ;; esac ) + +# Authority + path + query. Excludes $ ` \ " ' ; | ( ) * ! space and newline — +# none of which a real sitemap URL needs, all of which a shell reads. +_rest_charset_ok() ( LC_ALL=C; case "$1" in + ''|*[!A-Za-z0-9._~:/?#@=\&%+,-]*) exit 1 ;; esac ) + +# Literal local/private/metadata targets. This is a LITERAL check, not a DNS +# one: it stops the obvious, not a hostname that resolves inward. +_host_is_local() ( LC_ALL=C + # ${1,,} not tr: no fork, and no SC2018/SC2019 noise. Safe because the + # charset guard has already run — the string is [A-Za-z0-9.-] by here. + case "${1,,}" in + localhost|*.localhost|*.local|0.0.0.0|broadcasthost) exit 0 ;; + 127.*|10.*|169.254.*|192.168.*) exit 0 ;; + 172.1[6-9].*|172.2[0-9].*|172.3[01].*) exit 0 ;; + metadata.google.internal|metadata) exit 0 ;; + *) exit 1 ;; + esac ) + +_reject_local() { _host_is_local "$1" && _die "local/private target refused: '$1'"; return 0; } + +check_host() { + _host_charset_ok "$1" || _die "host charset (allowed A-Za-z0-9.-): '$1'" + _reject_local "$1" + printf '%s\n' "$1" +} + +check_url() { + local rest host + case "$1" in + https://*) rest="${1#https://}" ;; + http://*) rest="${1#http://}" ;; + *) _die "scheme must be http or https: '$1'" ;; + esac + _rest_charset_ok "$rest" || _die "url charset: '$1'" + host="${rest%%/*}"; host="${host%%\?*}"; host="${host%%#*}" + # user@host hides the real target: https://trusted.com@127.0.0.1/ hits .0.0.1 + case "$host" in *@*) _die "userinfo in authority (confusion vector): '$1'" ;; esac + host="${host%%:*}" # drop :port before validating the host + _host_charset_ok "$host" || _die "host charset: '$host'" + _reject_local "$host" + printf '%s\n' "$1" +} + +case "${1:-}" in + host) [ $# -eq 2 ] || _die "usage: url-guard.sh host "; check_host "$2" ;; + url) [ $# -eq 2 ] || _die "usage: url-guard.sh url "; check_url "$2" ;; + *) _die "usage: url-guard.sh {host|url} " ;; +esac