test(lib): skill-routing census, TF-IDF collisions across the live catalog

Top 10 description pairs, WARN >= 0.50, FAIL >= 0.75, fixture flip-test
with a positive control and a sensitivity re-run (2-doc corpora are
degenerate, LRN-172). Baseline 2026-09-27: 120 skills, max 0.52
(careful ~ guard). Adapted from agent-skills evals Tier 2.
This commit is contained in:
bastien
2026-09-27 20:17:37 +02:00
parent 2b25cb4704
commit 409db51af9
2 changed files with 326 additions and 0 deletions
+213
View File
@@ -0,0 +1,213 @@
#!/usr/bin/env python3
"""Deterministic census of skill-description collisions (TF-IDF cosine).
Adapted from addyosmani/agent-skills evals Tier 2. Reads every SKILL.md
under the live catalog — `~/.claude/skills/*/` plus plugin skills under
`~/.claude/plugins/cache/*/*/*/skills/*/` and the nested
`.../.claude/skills/*/` layout (override with SKILL_ROUTING_ROOTS,
colon-separated dirs, for fixtures) — extracts the `description`
frontmatter field (scalar or `|`/`>` block), tokenizes it, and scores
every pair by TF-IDF cosine. Two skills with near-identical descriptions
route the same prompt to both: a routing collision, not a naming clash.
Report: `skills with description: N`, the top 10 pairs, then a WARN line
per pair >= SKILL_ROUTING_WARN (default 0.50) and a FAIL line per pair
>= SKILL_ROUTING_FAIL (default 0.75). Exit 1 iff any pair reached FAIL.
"""
import collections
import glob
import itertools
import math
import os
import re
import sys
DEFAULT_ROOTS = ["~/.claude/skills", "~/.claude/plugins/cache"]
ROOT_SUFFIXES = [
"*/SKILL.md",
"*/*/*/skills/*/SKILL.md",
"*/*/*/.claude/skills/*/SKILL.md",
]
STOPWORDS = set(
"a an the and or of to in on for with use when you your this that is are "
"be it as by from at into not no if then use using used uses skill "
"skills user users project projects file files code work".split()
)
BLOCK_MARKERS = ("|", ">", "|-", ">-", "|+", ">+")
WARN_DEFAULT = 0.50
FAIL_DEFAULT = 0.75
def resolve_roots():
"""Root dirs: SKILL_ROUTING_ROOTS override, else the live catalog."""
override = os.environ.get("SKILL_ROUTING_ROOTS")
return override.split(":") if override else DEFAULT_ROOTS
def discover_paths(roots):
"""SKILL.md paths across all roots/suffixes, sorted for determinism."""
paths = []
for root in roots:
expanded = os.path.expanduser(root)
for suffix in ROOT_SUFFIXES:
paths.extend(sorted(glob.glob(os.path.join(expanded, suffix))))
return paths
def dedup_by_skill_name(paths):
"""name (skill dir) -> path; first path wins (symlinks pre-resolved)."""
by_name = {}
for path in paths:
name = os.path.basename(os.path.dirname(path))
by_name.setdefault(name, path)
return by_name
def _frontmatter(text):
"""Raw text between the two leading `---` delimiters, or None."""
match = re.search(r"^---\r?\n(.*?)\r?\n---", text, re.S)
return match.group(1) if match else None
def _strip_quotes(value):
if len(value) >= 2 and value[0] == value[-1] and value[0] in "'\"":
return value[1:-1]
return value
def _block_text(lines, start):
"""Join a YAML block-scalar's indented continuation lines.
A blank line inside a block scalar does NOT end it (YAML compares
indentation only against non-blank lines) — only a line back at the
frontmatter's column-0 (the next key) does.
"""
collected = []
for line in lines[start:]:
if line.strip() == "" or line.startswith((" ", "\t")):
collected.append(line.strip())
else:
break
return " ".join(part for part in collected if part)
def extract_description(path):
"""The `description:` frontmatter value (scalar or block), or None."""
try:
with open(path, encoding="utf-8", errors="ignore") as handle:
text = handle.read()
except OSError:
return None
fm = _frontmatter(text)
if fm is None:
return None
lines = fm.split("\n")
for i, line in enumerate(lines):
match = re.match(r"^description:\s*(.*)$", line)
if not match:
continue
value = match.group(1).strip()
if value in BLOCK_MARKERS:
return _block_text(lines, i + 1) or None
return _strip_quotes(value) or None
return None
def stem(word):
"""Strip a trailing s/es/ed/ing suffix from a long-enough word."""
for suffix in ("ing", "ed", "es", "s"):
if len(word) > 4 and word.endswith(suffix):
return word[: -len(suffix)]
return word
def tokenize(text):
"""Lowercase, keep [a-z][a-z0-9-]+ words, drop stopwords/short, stem."""
words = re.findall(r"[a-z][a-z0-9-]+", text.lower())
return [stem(w) for w in words if w not in STOPWORDS and len(w) > 2]
def build_docs(paths_by_name):
"""name -> Counter(tokens), for every skill with a non-empty description."""
docs = {}
for name, path in paths_by_name.items():
desc = extract_description(path)
if not desc:
continue
tokens = tokenize(desc)
if tokens:
docs[name] = collections.Counter(tokens)
return docs
def _document_frequencies(docs):
df = collections.Counter()
for counter in docs.values():
for token in counter:
df[token] += 1
return df
def _tfidf_vector(counter, doc_count, df):
"""(1 + log tf) * log(N/df), L2-normalized."""
weights = {
token: (1 + math.log(n)) * math.log(doc_count / df[token])
for token, n in counter.items()
}
norm = math.sqrt(sum(w * w for w in weights.values())) or 1
return {token: w / norm for token, w in weights.items()}
def tfidf_vectors(docs):
"""name -> {token: TF-IDF weight}, L2-normalized."""
doc_count = len(docs)
df = _document_frequencies(docs)
return {name: _tfidf_vector(c, doc_count, df) for name, c in docs.items()}
def cosine(vec_a, vec_b):
return sum(weight * vec_b.get(token, 0) for token, weight in vec_a.items())
def score_pairs(vectors):
"""[(score, a, b), ...] over every pair, sorted by score descending."""
pairs = [
(cosine(vectors[a], vectors[b]), a, b)
for a, b in itertools.combinations(sorted(vectors), 2)
]
pairs.sort(reverse=True)
return pairs
def _thresholds():
warn = float(os.environ.get("SKILL_ROUTING_WARN", WARN_DEFAULT))
fail = float(os.environ.get("SKILL_ROUTING_FAIL", FAIL_DEFAULT))
return warn, fail
def report(doc_count, pairs):
"""Print the census report; return True iff a pair reached FAIL."""
warn, fail = _thresholds()
print(f"skills with description: {doc_count}")
for score, a, b in pairs[:10]:
print(f"{score:.2f} {a} ~ {b}")
failed = False
for score, a, b in pairs:
if score >= fail:
print(f"FAIL {score:.2f} {a} ~ {b}")
failed = True
elif score >= warn:
print(f"WARN {score:.2f} {a} ~ {b}")
return failed
def main():
by_name = dedup_by_skill_name(discover_paths(resolve_roots()))
docs = build_docs(by_name)
vectors = tfidf_vectors(docs)
failed = report(len(docs), score_pairs(vectors))
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(main())
+113
View File
@@ -0,0 +1,113 @@
#!/usr/bin/env bash
# lib/tests/skill-routing-census.test.sh — TF-IDF cosine census of
# skill-description collisions across the live catalog (routing ambiguity),
# adapted from addyosmani/agent-skills evals Tier 2. Two skills whose
# descriptions read alike route the same prompt to both — a routing
# collision, not a cosmetic naming clash.
#
# Fixture flip-test math note: with only 2 documents, corpus-wide IDF zeroes
# EVERY term's contribution — a term unique to one doc (df=1) is absent from
# the other vector and can't enter the dot product, a term shared by both
# (df=2) gets idf=log(N/df)=log(1)=0. Cosine is trivially 0 for any 2-doc
# corpus, near-duplicate or not — a bare 2-doc "distinct pair" fixture would
# pass for that reason alone, not because the pair is actually distinct. Both
# fixtures below ride in a 4-doc corpus so IDF carries real signal. The
# distinct-pair fixture also carries its own same-corpus positive control (a
# near-dup pair that must FAIL) and a sensitivity re-run where the "distinct"
# partner is swapped for a near-copy of its counterpart, to prove the WARN/
# FAIL marker actually fires when the pair collides — not just that it stays
# silent for reasons unrelated to distinctness.
set -u
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
CENSUS="$ROOT/lib/skill-routing-census.py"
pass=0; fail=0
check() { if [ "$2" = "$3" ]; then pass=$((pass+1)); else fail=$((fail+1));
printf 'FAIL %s: got[%s] want[%s]\n' "$1" "$2" "$3"; fi; }
WORK="$(mktemp -d)"; trap 'rm -rf "$WORK"' EXIT
TRANSLATE_DESC='description: "Translate a PDF, keeping layout and images."'
SECAUDIT_DESC="description: 'Run a security audit for secrets, CVEs, OWASP.'"
# _mkskill <root> <name> <frontmatter-line>... — a fixture skill/SKILL.md
_mkskill() {
local dir="$1" name="$2"; shift 2
mkdir -p "$dir/$name"
{ echo "---"; echo "name: $name"; printf '%s\n' "$@"; echo "---"; } \
> "$dir/$name/SKILL.md"
}
# _mkcontrol <root> — same-corpus positive control: skill-e/skill-f, a
# near-duplicate deploy-runbook pair that must FAIL wherever it rides.
_mkcontrol() {
local dir="$1"
_mkskill "$dir" skill-e "description: |" \
" Use when deploying a project via its per-project runbook to ship the" \
" application to production servers safely with rollback support and" \
" health checks after each release."
_mkskill "$dir" skill-f "description: >-" \
" Use when deploying a project via its per-project deployment runbook to" \
" ship the application to production servers safely with rollback" \
" support and health checks after each release."
}
# ── live run: this machine's real catalog (a FAIL line here → suite RED) ──
live_out=$(python3 "$CENSUS" 2>&1); live_rc=$?
printf '%s\n' "$live_out"
check T1-live-run-no-collision "$live_rc" 0
has_count=$(printf '%s\n' "$live_out" \
| grep -qE 'skills with description: [0-9]{2,}' && echo yes)
check T2-live-catalog-non-trivial "$has_count" yes
# ── fixture AB: near-duplicate pair + 2 unrelated fillers (N=4, real signal) ──
AB="$WORK/ab"
_mkskill "$AB" skill-a "description: |" \
" Use when deploying a project via its per-project runbook to ship the" \
" application to production servers safely with rollback support and" \
" health checks after each release."
_mkskill "$AB" skill-b "description: >-" \
" Use when deploying a project via its per-project deployment runbook to" \
" ship the application to production servers safely with rollback" \
" support and health checks after each release."
_mkskill "$AB" filler-translate "$TRANSLATE_DESC"
_mkskill "$AB" filler-secaudit "$SECAUDIT_DESC"
out_ab=$(SKILL_ROUTING_ROOTS="$AB" python3 "$CENSUS" 2>&1)
fail_line=$(printf '%s\n' "$out_ab" \
| grep -qE '^FAIL 0\.[0-9]{2} skill-a ~ skill-b$' && echo yes)
check T3-collision-fail-line "$fail_line" yes
[ "$fail_line" = yes ] && echo FIXTURE_COLLISION_DETECTED
# ── fixture CD: distinct pair + same-corpus positive control (N=4) ────────
# skill-e/skill-f (positive control) must FAIL; skill-c/skill-d (the pair
# under test) must stay silent (no WARN/FAIL line naming them).
CD="$WORK/cd"
_mkcontrol "$CD"
_mkskill "$CD" skill-c "$TRANSLATE_DESC"
_mkskill "$CD" skill-d "$SECAUDIT_DESC"
out_cd=$(SKILL_ROUTING_ROOTS="$CD" python3 "$CENSUS" 2>&1)
control_fail=$(printf '%s\n' "$out_cd" \
| grep -qE '^FAIL 0\.[0-9]{2} skill-e ~ skill-f$' && echo yes)
check T4-positive-control-fail "$control_fail" yes
distinct_marker=$(printf '%s\n' "$out_cd" \
| grep -E '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$')
distinct_silent=$([ -z "$distinct_marker" ] && echo yes || echo no)
check T5-distinct-pair-silent "$distinct_silent" yes
# ── sensitivity re-run: skill-d -> near-copy of skill-c ────────────────────
# Same corpus, but skill-d's description now near-duplicates skill-c's: the
# marker that stayed silent above must appear here, proving T5 wasn't silent
# by construction (e.g. a broken pattern or a threshold nothing can cross).
CD2="$WORK/cd-nearcopy"
_mkcontrol "$CD2"
_mkskill "$CD2" skill-c "$TRANSLATE_DESC"
_mkskill "$CD2" skill-d \
'description: "Translate a PDF document, keeping the layout and images intact."'
out_cd2=$(SKILL_ROUTING_ROOTS="$CD2" python3 "$CENSUS" 2>&1)
sensitivity_marker=$(printf '%s\n' "$out_cd2" \
| grep -qE '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$' && echo yes)
check T6-sensitivity-marker-appears "$sensitivity_marker" yes
[ "$control_fail" = yes ] && [ "$distinct_silent" = yes ] \
&& [ "$sensitivity_marker" = yes ] && echo FIXTURE_DISTINCT_OK
echo "PASS=$pass FAIL=$fail"
[ "$fail" -eq 0 ]