fix(seo-data): extract real CrUX origin on 404 retry + drop dead import
This commit is contained in:
@@ -1,7 +1,7 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""CrUX + GSC fetch → normalized JSON. Third-party imports are LAZY so mock and
|
"""CrUX + GSC fetch → normalized JSON. Third-party imports are LAZY so mock and
|
||||||
degraded paths run stdlib-only (no venv, no network)."""
|
degraded paths run stdlib-only (no venv, no network)."""
|
||||||
import argparse, json, os, sys
|
import argparse, json, os
|
||||||
|
|
||||||
def _mock(name):
|
def _mock(name):
|
||||||
d = os.environ.get("SEO_DATA_MOCK_DIR")
|
d = os.environ.get("SEO_DATA_MOCK_DIR")
|
||||||
@@ -38,6 +38,11 @@ def _crux_query(key, body):
|
|||||||
"https://chromeuxreport.googleapis.com/v1/records:queryRecord?key=" + key,
|
"https://chromeuxreport.googleapis.com/v1/records:queryRecord?key=" + key,
|
||||||
json=body, timeout=20)
|
json=body, timeout=20)
|
||||||
|
|
||||||
|
def _origin(url):
|
||||||
|
from urllib.parse import urlparse # stdlib
|
||||||
|
p = urlparse(url)
|
||||||
|
return "%s://%s" % (p.scheme, p.netloc) # strip path — CrUX origin = scheme+host only
|
||||||
|
|
||||||
def crux(url, strategy="mobile"):
|
def crux(url, strategy="mobile"):
|
||||||
raw = _mock("crux_%s.json" % strategy)
|
raw = _mock("crux_%s.json" % strategy)
|
||||||
if raw is None:
|
if raw is None:
|
||||||
@@ -47,7 +52,7 @@ def crux(url, strategy="mobile"):
|
|||||||
ff = "PHONE" if strategy == "mobile" else "DESKTOP"
|
ff = "PHONE" if strategy == "mobile" else "DESKTOP"
|
||||||
r = _crux_query(key, {"url": url, "formFactor": ff})
|
r = _crux_query(key, {"url": url, "formFactor": ff})
|
||||||
if r.status_code == 404: # no page-level data → try origin-level
|
if r.status_code == 404: # no page-level data → try origin-level
|
||||||
r = _crux_query(key, {"origin": url.rstrip("/"), "formFactor": ff})
|
r = _crux_query(key, {"origin": _origin(url), "formFactor": ff})
|
||||||
if r.status_code == 404:
|
if r.status_code == 404:
|
||||||
return {"status": "degraded", "reason": "no_field_data"}
|
return {"status": "degraded", "reason": "no_field_data"}
|
||||||
if r.status_code == 429:
|
if r.status_code == 429:
|
||||||
|
|||||||
@@ -40,6 +40,9 @@ CRUX_DEG="$(env -u CRUX_API_KEY -u SEO_DATA_MOCK_DIR \
|
|||||||
python3 "$SD/google_seo.py" crux --url https://ex.com)"
|
python3 "$SD/google_seo.py" crux --url https://ex.com)"
|
||||||
has "crux degrades w/o key" "$CRUX_DEG" '"status": "degraded"'
|
has "crux degrades w/o key" "$CRUX_DEG" '"status": "degraded"'
|
||||||
has "crux degrade reason" "$CRUX_DEG" 'no_crux_key'
|
has "crux degrade reason" "$CRUX_DEG" 'no_crux_key'
|
||||||
|
ORIG="$(python3 -c "import sys; sys.path.insert(0,'$SD'); import google_seo; print(google_seo._origin('https://example.com/blog/post'))")"
|
||||||
|
has "origin strips to host" "$ORIG" 'https://example.com'
|
||||||
|
hasnt "origin drops the path" "$ORIG" 'blog'
|
||||||
|
|
||||||
echo ""
|
echo ""
|
||||||
echo "seo-data engine: $PASS pass, $FAIL fail"
|
echo "seo-data engine: $PASS pass, $FAIL fail"
|
||||||
|
|||||||
Reference in New Issue
Block a user