From dca977bb2732e45ba2a56ff667c175f1dc1d2f75 Mon Sep 17 00:00:00 2001 From: Bastien Chanot Date: Fri, 17 Jul 2026 12:28:36 +0200 Subject: [PATCH] =?UTF-8?q?fix(seo-data,seo):=20backtest=20on=20a=20second?= =?UTF-8?q?,=20native=20site=20=E2=80=94=20two=20real=20bugs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Everything on this branch was grounded on ONE Astro repo. A native PHP site (lavageangels356.fr) broke two things that looked fine there. BUG 1 — sitemap counted images as pages. _locs matched `el.tag.endswith("}loc")`, and from Google's image-sitemap namespace ALSO ends with '}loc'. Astro's sitemap has no image extension, so this was invisible. The native site's does: 24 + 3 came back as count=27. The COVERAGE denominator was 12.5% too high and img/logo.png was about to be sampled and audited as a page. Fixed with two locks: walk the DIRECT children of each / instead of root.iter() (which alone excludes ), and test the sitemaps.org namespace explicitly. Regression fixture carries the image extension; the old endswith code returns 9 URLs against it, the new one 7 with zero images. Verified both sites: native 27 -> 24, zero images; Astro unchanged at 86. BUG 2 — the C1c family heuristic was tuned to one URL layout. "First path segment" works for NESTED city pages (/creation-site-internet/essonne-91/ → 25 pages, 1 family) and FAILS for FLAT ones (/lavage-auto-pomponne, /lavage-auto-torcy → 8 pages, 8 singletons). Consequence: C1c's rule "sample >=3 from the largest family" would have targeted /services (5) and missed the 8 city pages entirely — the exact doorway-page risk the 30/70 rule exists to catch. Family is now "shared parent path OR shared slug prefix (>=3 URLs sharing 2+ hyphen tokens)", with both real layouts as the worked examples, plus a sanity-check: a sitemap yielding almost as many families as URLs has defeated the heuristic, not proved the site has no templates. Fixed in seo-analyzer and in the geo pointer that referenced it. Backtest results on the native site for everything else: url-guard accepts the domain; source-scope excludes only .git (no dist/build/out exists — the exclusions are correctly no-ops, and cache/ holds only .htaccess+.gitignore so it is rightly untouched); the sameAs check runs and finds zero (a real GEO gap for that site, not a tool bug); links are present in the served HTML (PHP is SSR), so C3 is feasible there. Verified: seo-data 119 -> 122 pass, 0 fail; full suite green; py_compile clean. --- agents/geo-analyzer.md | 6 ++++-- agents/seo-analyzer.md | 23 ++++++++++++++++---- lib/seo-data/fixtures/sitemap.xml | 12 ++++++++++- lib/seo-data/seo-data.test.sh | 6 ++++++ lib/seo-data/sitemap.py | 35 ++++++++++++++++++++++++------- 5 files changed, 68 insertions(+), 14 deletions(-) diff --git a/agents/geo-analyzer.md b/agents/geo-analyzer.md index 8d141d7..b9d870f 100644 --- a/agents/geo-analyzer.md +++ b/agents/geo-analyzer.md @@ -679,8 +679,10 @@ what bounds it. Content Shape does NOT work that way: Definition Lead, TL;DR and heading wording are written per page, so a template says nothing about its 25 instances. Bound Schema.org by SOURCE, Content Shape by LIVE, and never quote the flattering one alone. Get the URL families from -`fetch.sh sitemap` (first path segment); if `/seo` already ran it, reuse the -count rather than re-fetching. +`fetch.sh sitemap`, grouped as seo-analyzer STEP 5 describes — shared parent +path OR shared slug prefix, because both layouts are real: first-segment +alone reads 8 flat `/lavage-auto-` pages as 8 singletons. If `/seo` +already ran it, reuse the count rather than re-fetching. Per user instruction: **GEO weight in combined SEO+GEO report = 20% for local, 25% for national/SaaS/content.** diff --git a/agents/seo-analyzer.md b/agents/seo-analyzer.md index ae4843f..5f10cbd 100644 --- a/agents/seo-analyzer.md +++ b/agents/seo-analyzer.md @@ -508,10 +508,25 @@ point of use (same contract as the sameAs check in geo-analyzer). ### Meta tags per page (sample 5-15 key pages) -**Group the sitemap URLs into families first** — first path segment is a -good enough proxy for "same template", and it needs no framework routing -knowledge. Measured on a real Astro site: 86 URLs collapse into 8 families, -and 75 of them (87%) come from just 3 dynamic `[dept]` templates. +**Group the sitemap URLs into families first** — a family is "pages one +template renders". You do not need framework routing knowledge to see them, +but you DO need to look at the actual URL shape, because it varies: + +| Layout | Example | Family signal | +|---|---|---| +| Nested | `/creation-site-internet/essonne-91/`, `/creation-site-internet/seine-et-marne-77/` | **shared parent path** → 25 pages, 1 family | +| **Flat** | `/lavage-auto-pomponne`, `/lavage-auto-torcy`, `/lavage-auto-chelles` | **shared slug prefix** → 8 pages, 1 family | + +Both are real, measured on two live sites. First-path-segment alone handles +the nested case and **fails the flat one**: those 8 city pages read as 8 +unrelated singletons, so the largest "family" becomes `/services` (5) and the +doorway-page risk — the exact thing the 30/70 rule exists to catch — is +invisible. Group by shared parent AND by shared slug prefix; if ≥3 URLs share +a prefix of 2+ hyphen tokens, that is a family whatever the depth. + +Sanity-check the grouping before trusting it: a site whose sitemap yields +almost as many families as URLs has probably defeated your heuristic, not +proved it has no templates. **Sample by finding class, because the classes need opposite samples:** diff --git a/lib/seo-data/fixtures/sitemap.xml b/lib/seo-data/fixtures/sitemap.xml index 5813175..68de6af 100644 --- a/lib/seo-data/fixtures/sitemap.xml +++ b/lib/seo-data/fixtures/sitemap.xml @@ -1,9 +1,19 @@ - + + https://ex.com/ weekly + + https://ex.com/img/logo.png + Logo + + + https://ex.com/img/hero.jpeg + https://ex.com/services https://ex.com/blog diff --git a/lib/seo-data/seo-data.test.sh b/lib/seo-data/seo-data.test.sh index 79cc0f9..097038a 100644 --- a/lib/seo-data/seo-data.test.sh +++ b/lib/seo-data/seo-data.test.sh @@ -111,6 +111,12 @@ hasnt "sitemap drops non-http" "$SM" 'ftp://' hasnt "sitemap drops shell-meta" "$SM" 'bad"quote' # namespace-agnostic: real sitemaps carry sitemaps.org xmlns (+ xhtml here) has "sitemap reads namespaced" "$SM" '"https://ex.com/services"' +# REGRESSION: also ends with '}loc'. An endswith test counted image +# sitemap entries as pages — a real native site returned 27 for 24 , and +# img/logo.png was about to be sampled and audited as a page. +hasnt "image:loc is not a page" "$SM" '/img/logo.png' +hasnt "image:loc jpeg not a page" "$SM" '/img/hero.jpeg' +has "image ns does not inflate count" "$SM" '"count": 4' IDX="$(SEO_DATA_MOCK_DIR="$SD/fixtures-sitemap-index" python3 "$SD/sitemap.py" \ --url https://ex.com/sitemap.xml)" diff --git a/lib/seo-data/sitemap.py b/lib/seo-data/sitemap.py index 3079c10..99454ab 100644 --- a/lib/seo-data/sitemap.py +++ b/lib/seo-data/sitemap.py @@ -57,19 +57,40 @@ def _refuse_dtd(raw): if b": sitemaps.org namespace, or namespace-less. + + NOT or . Those live in Google's extension + namespaces and name an ASSET inside a , not a page of its own. An + endswith('}loc') test matches them too — that shipped, and a real site + caught it: 24 + 3 came back as a count of 27, so the + COVERAGE denominator was 12.5% too high and img/logo.png was about to be + sampled and audited as a page. + """ + return tag == SITEMAP_NS + "loc" or tag == "loc" + def _locs(raw): - """( texts, is_sitemapindex). Namespace-agnostic: real sitemaps carry - the sitemaps.org xmlns and often xhtml too.""" + """(page texts, is_sitemapindex). + + Walks the DIRECT children of each / rather than root.iter(): + that alone excludes , and the namespace test above + is the second lock. XML comments iterate as elements with no children, so + they fall through harmlessly. + """ import xml.etree.ElementTree as ET # stdlib, lazy _refuse_dtd(raw) root = ET.fromstring(raw) is_index = root.tag.endswith("sitemapindex") out = [] - for el in root.iter(): - if el.tag.endswith("}loc") or el.tag == "loc": - text = (el.text or "").strip() - if text: - out.append(text) + for entry in root: # | + for child in entry: # direct children only + if _is_page_loc(child.tag): + text = (child.text or "").strip() + if text: + out.append(text) + break # one per entry return out, is_index def _sane(u):