From 409db51af9cf524b97761b0a58e3f13ab3c61598 Mon Sep 17 00:00:00 2001 From: bastien Date: Sun, 27 Sep 2026 20:17:37 +0200 Subject: [PATCH] test(lib): skill-routing census, TF-IDF collisions across the live catalog Top 10 description pairs, WARN >= 0.50, FAIL >= 0.75, fixture flip-test with a positive control and a sensitivity re-run (2-doc corpora are degenerate, LRN-172). Baseline 2026-09-27: 120 skills, max 0.52 (careful ~ guard). Adapted from agent-skills evals Tier 2. --- lib/skill-routing-census.py | 213 +++++++++++++++++++++++++ lib/tests/skill-routing-census.test.sh | 113 +++++++++++++ 2 files changed, 326 insertions(+) create mode 100644 lib/skill-routing-census.py create mode 100755 lib/tests/skill-routing-census.test.sh diff --git a/lib/skill-routing-census.py b/lib/skill-routing-census.py new file mode 100644 index 0000000..89cb70f --- /dev/null +++ b/lib/skill-routing-census.py @@ -0,0 +1,213 @@ +#!/usr/bin/env python3 +"""Deterministic census of skill-description collisions (TF-IDF cosine). + +Adapted from addyosmani/agent-skills evals Tier 2. Reads every SKILL.md +under the live catalog — `~/.claude/skills/*/` plus plugin skills under +`~/.claude/plugins/cache/*/*/*/skills/*/` and the nested +`.../.claude/skills/*/` layout (override with SKILL_ROUTING_ROOTS, +colon-separated dirs, for fixtures) — extracts the `description` +frontmatter field (scalar or `|`/`>` block), tokenizes it, and scores +every pair by TF-IDF cosine. Two skills with near-identical descriptions +route the same prompt to both: a routing collision, not a naming clash. + +Report: `skills with description: N`, the top 10 pairs, then a WARN line +per pair >= SKILL_ROUTING_WARN (default 0.50) and a FAIL line per pair +>= SKILL_ROUTING_FAIL (default 0.75). Exit 1 iff any pair reached FAIL. +""" +import collections +import glob +import itertools +import math +import os +import re +import sys + +DEFAULT_ROOTS = ["~/.claude/skills", "~/.claude/plugins/cache"] +ROOT_SUFFIXES = [ + "*/SKILL.md", + "*/*/*/skills/*/SKILL.md", + "*/*/*/.claude/skills/*/SKILL.md", +] +STOPWORDS = set( + "a an the and or of to in on for with use when you your this that is are " + "be it as by from at into not no if then use using used uses skill " + "skills user users project projects file files code work".split() +) +BLOCK_MARKERS = ("|", ">", "|-", ">-", "|+", ">+") +WARN_DEFAULT = 0.50 +FAIL_DEFAULT = 0.75 + + +def resolve_roots(): + """Root dirs: SKILL_ROUTING_ROOTS override, else the live catalog.""" + override = os.environ.get("SKILL_ROUTING_ROOTS") + return override.split(":") if override else DEFAULT_ROOTS + + +def discover_paths(roots): + """SKILL.md paths across all roots/suffixes, sorted for determinism.""" + paths = [] + for root in roots: + expanded = os.path.expanduser(root) + for suffix in ROOT_SUFFIXES: + paths.extend(sorted(glob.glob(os.path.join(expanded, suffix)))) + return paths + + +def dedup_by_skill_name(paths): + """name (skill dir) -> path; first path wins (symlinks pre-resolved).""" + by_name = {} + for path in paths: + name = os.path.basename(os.path.dirname(path)) + by_name.setdefault(name, path) + return by_name + + +def _frontmatter(text): + """Raw text between the two leading `---` delimiters, or None.""" + match = re.search(r"^---\r?\n(.*?)\r?\n---", text, re.S) + return match.group(1) if match else None + + +def _strip_quotes(value): + if len(value) >= 2 and value[0] == value[-1] and value[0] in "'\"": + return value[1:-1] + return value + + +def _block_text(lines, start): + """Join a YAML block-scalar's indented continuation lines. + + A blank line inside a block scalar does NOT end it (YAML compares + indentation only against non-blank lines) — only a line back at the + frontmatter's column-0 (the next key) does. + """ + collected = [] + for line in lines[start:]: + if line.strip() == "" or line.startswith((" ", "\t")): + collected.append(line.strip()) + else: + break + return " ".join(part for part in collected if part) + + +def extract_description(path): + """The `description:` frontmatter value (scalar or block), or None.""" + try: + with open(path, encoding="utf-8", errors="ignore") as handle: + text = handle.read() + except OSError: + return None + fm = _frontmatter(text) + if fm is None: + return None + lines = fm.split("\n") + for i, line in enumerate(lines): + match = re.match(r"^description:\s*(.*)$", line) + if not match: + continue + value = match.group(1).strip() + if value in BLOCK_MARKERS: + return _block_text(lines, i + 1) or None + return _strip_quotes(value) or None + return None + + +def stem(word): + """Strip a trailing s/es/ed/ing suffix from a long-enough word.""" + for suffix in ("ing", "ed", "es", "s"): + if len(word) > 4 and word.endswith(suffix): + return word[: -len(suffix)] + return word + + +def tokenize(text): + """Lowercase, keep [a-z][a-z0-9-]+ words, drop stopwords/short, stem.""" + words = re.findall(r"[a-z][a-z0-9-]+", text.lower()) + return [stem(w) for w in words if w not in STOPWORDS and len(w) > 2] + + +def build_docs(paths_by_name): + """name -> Counter(tokens), for every skill with a non-empty description.""" + docs = {} + for name, path in paths_by_name.items(): + desc = extract_description(path) + if not desc: + continue + tokens = tokenize(desc) + if tokens: + docs[name] = collections.Counter(tokens) + return docs + + +def _document_frequencies(docs): + df = collections.Counter() + for counter in docs.values(): + for token in counter: + df[token] += 1 + return df + + +def _tfidf_vector(counter, doc_count, df): + """(1 + log tf) * log(N/df), L2-normalized.""" + weights = { + token: (1 + math.log(n)) * math.log(doc_count / df[token]) + for token, n in counter.items() + } + norm = math.sqrt(sum(w * w for w in weights.values())) or 1 + return {token: w / norm for token, w in weights.items()} + + +def tfidf_vectors(docs): + """name -> {token: TF-IDF weight}, L2-normalized.""" + doc_count = len(docs) + df = _document_frequencies(docs) + return {name: _tfidf_vector(c, doc_count, df) for name, c in docs.items()} + + +def cosine(vec_a, vec_b): + return sum(weight * vec_b.get(token, 0) for token, weight in vec_a.items()) + + +def score_pairs(vectors): + """[(score, a, b), ...] over every pair, sorted by score descending.""" + pairs = [ + (cosine(vectors[a], vectors[b]), a, b) + for a, b in itertools.combinations(sorted(vectors), 2) + ] + pairs.sort(reverse=True) + return pairs + + +def _thresholds(): + warn = float(os.environ.get("SKILL_ROUTING_WARN", WARN_DEFAULT)) + fail = float(os.environ.get("SKILL_ROUTING_FAIL", FAIL_DEFAULT)) + return warn, fail + + +def report(doc_count, pairs): + """Print the census report; return True iff a pair reached FAIL.""" + warn, fail = _thresholds() + print(f"skills with description: {doc_count}") + for score, a, b in pairs[:10]: + print(f"{score:.2f} {a} ~ {b}") + failed = False + for score, a, b in pairs: + if score >= fail: + print(f"FAIL {score:.2f} {a} ~ {b}") + failed = True + elif score >= warn: + print(f"WARN {score:.2f} {a} ~ {b}") + return failed + + +def main(): + by_name = dedup_by_skill_name(discover_paths(resolve_roots())) + docs = build_docs(by_name) + vectors = tfidf_vectors(docs) + failed = report(len(docs), score_pairs(vectors)) + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/lib/tests/skill-routing-census.test.sh b/lib/tests/skill-routing-census.test.sh new file mode 100755 index 0000000..5fe1c41 --- /dev/null +++ b/lib/tests/skill-routing-census.test.sh @@ -0,0 +1,113 @@ +#!/usr/bin/env bash +# lib/tests/skill-routing-census.test.sh — TF-IDF cosine census of +# skill-description collisions across the live catalog (routing ambiguity), +# adapted from addyosmani/agent-skills evals Tier 2. Two skills whose +# descriptions read alike route the same prompt to both — a routing +# collision, not a cosmetic naming clash. +# +# Fixture flip-test math note: with only 2 documents, corpus-wide IDF zeroes +# EVERY term's contribution — a term unique to one doc (df=1) is absent from +# the other vector and can't enter the dot product, a term shared by both +# (df=2) gets idf=log(N/df)=log(1)=0. Cosine is trivially 0 for any 2-doc +# corpus, near-duplicate or not — a bare 2-doc "distinct pair" fixture would +# pass for that reason alone, not because the pair is actually distinct. Both +# fixtures below ride in a 4-doc corpus so IDF carries real signal. The +# distinct-pair fixture also carries its own same-corpus positive control (a +# near-dup pair that must FAIL) and a sensitivity re-run where the "distinct" +# partner is swapped for a near-copy of its counterpart, to prove the WARN/ +# FAIL marker actually fires when the pair collides — not just that it stays +# silent for reasons unrelated to distinctness. +set -u +ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +CENSUS="$ROOT/lib/skill-routing-census.py" +pass=0; fail=0 +check() { if [ "$2" = "$3" ]; then pass=$((pass+1)); else fail=$((fail+1)); + printf 'FAIL %s: got[%s] want[%s]\n' "$1" "$2" "$3"; fi; } + +WORK="$(mktemp -d)"; trap 'rm -rf "$WORK"' EXIT +TRANSLATE_DESC='description: "Translate a PDF, keeping layout and images."' +SECAUDIT_DESC="description: 'Run a security audit for secrets, CVEs, OWASP.'" + +# _mkskill ... — a fixture skill/SKILL.md +_mkskill() { + local dir="$1" name="$2"; shift 2 + mkdir -p "$dir/$name" + { echo "---"; echo "name: $name"; printf '%s\n' "$@"; echo "---"; } \ + > "$dir/$name/SKILL.md" +} + +# _mkcontrol — same-corpus positive control: skill-e/skill-f, a +# near-duplicate deploy-runbook pair that must FAIL wherever it rides. +_mkcontrol() { + local dir="$1" + _mkskill "$dir" skill-e "description: |" \ + " Use when deploying a project via its per-project runbook to ship the" \ + " application to production servers safely with rollback support and" \ + " health checks after each release." + _mkskill "$dir" skill-f "description: >-" \ + " Use when deploying a project via its per-project deployment runbook to" \ + " ship the application to production servers safely with rollback" \ + " support and health checks after each release." +} + +# ── live run: this machine's real catalog (a FAIL line here → suite RED) ── +live_out=$(python3 "$CENSUS" 2>&1); live_rc=$? +printf '%s\n' "$live_out" +check T1-live-run-no-collision "$live_rc" 0 +has_count=$(printf '%s\n' "$live_out" \ + | grep -qE 'skills with description: [0-9]{2,}' && echo yes) +check T2-live-catalog-non-trivial "$has_count" yes + +# ── fixture AB: near-duplicate pair + 2 unrelated fillers (N=4, real signal) ── +AB="$WORK/ab" +_mkskill "$AB" skill-a "description: |" \ + " Use when deploying a project via its per-project runbook to ship the" \ + " application to production servers safely with rollback support and" \ + " health checks after each release." +_mkskill "$AB" skill-b "description: >-" \ + " Use when deploying a project via its per-project deployment runbook to" \ + " ship the application to production servers safely with rollback" \ + " support and health checks after each release." +_mkskill "$AB" filler-translate "$TRANSLATE_DESC" +_mkskill "$AB" filler-secaudit "$SECAUDIT_DESC" +out_ab=$(SKILL_ROUTING_ROOTS="$AB" python3 "$CENSUS" 2>&1) +fail_line=$(printf '%s\n' "$out_ab" \ + | grep -qE '^FAIL 0\.[0-9]{2} skill-a ~ skill-b$' && echo yes) +check T3-collision-fail-line "$fail_line" yes +[ "$fail_line" = yes ] && echo FIXTURE_COLLISION_DETECTED + +# ── fixture CD: distinct pair + same-corpus positive control (N=4) ──────── +# skill-e/skill-f (positive control) must FAIL; skill-c/skill-d (the pair +# under test) must stay silent (no WARN/FAIL line naming them). +CD="$WORK/cd" +_mkcontrol "$CD" +_mkskill "$CD" skill-c "$TRANSLATE_DESC" +_mkskill "$CD" skill-d "$SECAUDIT_DESC" +out_cd=$(SKILL_ROUTING_ROOTS="$CD" python3 "$CENSUS" 2>&1) +control_fail=$(printf '%s\n' "$out_cd" \ + | grep -qE '^FAIL 0\.[0-9]{2} skill-e ~ skill-f$' && echo yes) +check T4-positive-control-fail "$control_fail" yes +distinct_marker=$(printf '%s\n' "$out_cd" \ + | grep -E '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$') +distinct_silent=$([ -z "$distinct_marker" ] && echo yes || echo no) +check T5-distinct-pair-silent "$distinct_silent" yes + +# ── sensitivity re-run: skill-d -> near-copy of skill-c ──────────────────── +# Same corpus, but skill-d's description now near-duplicates skill-c's: the +# marker that stayed silent above must appear here, proving T5 wasn't silent +# by construction (e.g. a broken pattern or a threshold nothing can cross). +CD2="$WORK/cd-nearcopy" +_mkcontrol "$CD2" +_mkskill "$CD2" skill-c "$TRANSLATE_DESC" +_mkskill "$CD2" skill-d \ + 'description: "Translate a PDF document, keeping the layout and images intact."' +out_cd2=$(SKILL_ROUTING_ROOTS="$CD2" python3 "$CENSUS" 2>&1) +sensitivity_marker=$(printf '%s\n' "$out_cd2" \ + | grep -qE '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$' && echo yes) +check T6-sensitivity-marker-appears "$sensitivity_marker" yes + +[ "$control_fail" = yes ] && [ "$distinct_silent" = yes ] \ + && [ "$sensitivity_marker" = yes ] && echo FIXTURE_DISTINCT_OK + +echo "PASS=$pass FAIL=$fail" +[ "$fail" -eq 0 ]