fix(seo-data): extract real CrUX origin on 404 retry + drop dead import

This commit is contained in:
Bastien Chanot
2026-07-10 01:23:13 +02:00
parent e214da036d
commit 493ecd8806
2 changed files with 10 additions and 2 deletions
+7 -2
View File
@@ -1,7 +1,7 @@
#!/usr/bin/env python3
"""CrUX + GSC fetch → normalized JSON. Third-party imports are LAZY so mock and
degraded paths run stdlib-only (no venv, no network)."""
import argparse, json, os, sys
import argparse, json, os
def _mock(name):
d = os.environ.get("SEO_DATA_MOCK_DIR")
@@ -38,6 +38,11 @@ def _crux_query(key, body):
"https://chromeuxreport.googleapis.com/v1/records:queryRecord?key=" + key,
json=body, timeout=20)
def _origin(url):
from urllib.parse import urlparse # stdlib
p = urlparse(url)
return "%s://%s" % (p.scheme, p.netloc) # strip path — CrUX origin = scheme+host only
def crux(url, strategy="mobile"):
raw = _mock("crux_%s.json" % strategy)
if raw is None:
@@ -47,7 +52,7 @@ def crux(url, strategy="mobile"):
ff = "PHONE" if strategy == "mobile" else "DESKTOP"
r = _crux_query(key, {"url": url, "formFactor": ff})
if r.status_code == 404: # no page-level data → try origin-level
r = _crux_query(key, {"origin": url.rstrip("/"), "formFactor": ff})
r = _crux_query(key, {"origin": _origin(url), "formFactor": ff})
if r.status_code == 404:
return {"status": "degraded", "reason": "no_field_data"}
if r.status_code == 429:
+3
View File
@@ -40,6 +40,9 @@ CRUX_DEG="$(env -u CRUX_API_KEY -u SEO_DATA_MOCK_DIR \
python3 "$SD/google_seo.py" crux --url https://ex.com)"
has "crux degrades w/o key" "$CRUX_DEG" '"status": "degraded"'
has "crux degrade reason" "$CRUX_DEG" 'no_crux_key'
ORIG="$(python3 -c "import sys; sys.path.insert(0,'$SD'); import google_seo; print(google_seo._origin('https://example.com/blog/post'))")"
has "origin strips to host" "$ORIG" 'https://example.com'
hasnt "origin drops the path" "$ORIG" 'blog'
echo ""
echo "seo-data engine: $PASS pass, $FAIL fail"