test(lib): skill-routing census, TF-IDF collisions across the live catalog
Top 10 description pairs, WARN >= 0.50, FAIL >= 0.75, fixture flip-test with a positive control and a sensitivity re-run (2-doc corpora are degenerate, LRN-172). Baseline 2026-09-27: 120 skills, max 0.52 (careful ~ guard). Adapted from agent-skills evals Tier 2.
This commit is contained in:
@@ -0,0 +1,213 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Deterministic census of skill-description collisions (TF-IDF cosine).
|
||||||
|
|
||||||
|
Adapted from addyosmani/agent-skills evals Tier 2. Reads every SKILL.md
|
||||||
|
under the live catalog — `~/.claude/skills/*/` plus plugin skills under
|
||||||
|
`~/.claude/plugins/cache/*/*/*/skills/*/` and the nested
|
||||||
|
`.../.claude/skills/*/` layout (override with SKILL_ROUTING_ROOTS,
|
||||||
|
colon-separated dirs, for fixtures) — extracts the `description`
|
||||||
|
frontmatter field (scalar or `|`/`>` block), tokenizes it, and scores
|
||||||
|
every pair by TF-IDF cosine. Two skills with near-identical descriptions
|
||||||
|
route the same prompt to both: a routing collision, not a naming clash.
|
||||||
|
|
||||||
|
Report: `skills with description: N`, the top 10 pairs, then a WARN line
|
||||||
|
per pair >= SKILL_ROUTING_WARN (default 0.50) and a FAIL line per pair
|
||||||
|
>= SKILL_ROUTING_FAIL (default 0.75). Exit 1 iff any pair reached FAIL.
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import glob
|
||||||
|
import itertools
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
DEFAULT_ROOTS = ["~/.claude/skills", "~/.claude/plugins/cache"]
|
||||||
|
ROOT_SUFFIXES = [
|
||||||
|
"*/SKILL.md",
|
||||||
|
"*/*/*/skills/*/SKILL.md",
|
||||||
|
"*/*/*/.claude/skills/*/SKILL.md",
|
||||||
|
]
|
||||||
|
STOPWORDS = set(
|
||||||
|
"a an the and or of to in on for with use when you your this that is are "
|
||||||
|
"be it as by from at into not no if then use using used uses skill "
|
||||||
|
"skills user users project projects file files code work".split()
|
||||||
|
)
|
||||||
|
BLOCK_MARKERS = ("|", ">", "|-", ">-", "|+", ">+")
|
||||||
|
WARN_DEFAULT = 0.50
|
||||||
|
FAIL_DEFAULT = 0.75
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_roots():
|
||||||
|
"""Root dirs: SKILL_ROUTING_ROOTS override, else the live catalog."""
|
||||||
|
override = os.environ.get("SKILL_ROUTING_ROOTS")
|
||||||
|
return override.split(":") if override else DEFAULT_ROOTS
|
||||||
|
|
||||||
|
|
||||||
|
def discover_paths(roots):
|
||||||
|
"""SKILL.md paths across all roots/suffixes, sorted for determinism."""
|
||||||
|
paths = []
|
||||||
|
for root in roots:
|
||||||
|
expanded = os.path.expanduser(root)
|
||||||
|
for suffix in ROOT_SUFFIXES:
|
||||||
|
paths.extend(sorted(glob.glob(os.path.join(expanded, suffix))))
|
||||||
|
return paths
|
||||||
|
|
||||||
|
|
||||||
|
def dedup_by_skill_name(paths):
|
||||||
|
"""name (skill dir) -> path; first path wins (symlinks pre-resolved)."""
|
||||||
|
by_name = {}
|
||||||
|
for path in paths:
|
||||||
|
name = os.path.basename(os.path.dirname(path))
|
||||||
|
by_name.setdefault(name, path)
|
||||||
|
return by_name
|
||||||
|
|
||||||
|
|
||||||
|
def _frontmatter(text):
|
||||||
|
"""Raw text between the two leading `---` delimiters, or None."""
|
||||||
|
match = re.search(r"^---\r?\n(.*?)\r?\n---", text, re.S)
|
||||||
|
return match.group(1) if match else None
|
||||||
|
|
||||||
|
|
||||||
|
def _strip_quotes(value):
|
||||||
|
if len(value) >= 2 and value[0] == value[-1] and value[0] in "'\"":
|
||||||
|
return value[1:-1]
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _block_text(lines, start):
|
||||||
|
"""Join a YAML block-scalar's indented continuation lines.
|
||||||
|
|
||||||
|
A blank line inside a block scalar does NOT end it (YAML compares
|
||||||
|
indentation only against non-blank lines) — only a line back at the
|
||||||
|
frontmatter's column-0 (the next key) does.
|
||||||
|
"""
|
||||||
|
collected = []
|
||||||
|
for line in lines[start:]:
|
||||||
|
if line.strip() == "" or line.startswith((" ", "\t")):
|
||||||
|
collected.append(line.strip())
|
||||||
|
else:
|
||||||
|
break
|
||||||
|
return " ".join(part for part in collected if part)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_description(path):
|
||||||
|
"""The `description:` frontmatter value (scalar or block), or None."""
|
||||||
|
try:
|
||||||
|
with open(path, encoding="utf-8", errors="ignore") as handle:
|
||||||
|
text = handle.read()
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
fm = _frontmatter(text)
|
||||||
|
if fm is None:
|
||||||
|
return None
|
||||||
|
lines = fm.split("\n")
|
||||||
|
for i, line in enumerate(lines):
|
||||||
|
match = re.match(r"^description:\s*(.*)$", line)
|
||||||
|
if not match:
|
||||||
|
continue
|
||||||
|
value = match.group(1).strip()
|
||||||
|
if value in BLOCK_MARKERS:
|
||||||
|
return _block_text(lines, i + 1) or None
|
||||||
|
return _strip_quotes(value) or None
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def stem(word):
|
||||||
|
"""Strip a trailing s/es/ed/ing suffix from a long-enough word."""
|
||||||
|
for suffix in ("ing", "ed", "es", "s"):
|
||||||
|
if len(word) > 4 and word.endswith(suffix):
|
||||||
|
return word[: -len(suffix)]
|
||||||
|
return word
|
||||||
|
|
||||||
|
|
||||||
|
def tokenize(text):
|
||||||
|
"""Lowercase, keep [a-z][a-z0-9-]+ words, drop stopwords/short, stem."""
|
||||||
|
words = re.findall(r"[a-z][a-z0-9-]+", text.lower())
|
||||||
|
return [stem(w) for w in words if w not in STOPWORDS and len(w) > 2]
|
||||||
|
|
||||||
|
|
||||||
|
def build_docs(paths_by_name):
|
||||||
|
"""name -> Counter(tokens), for every skill with a non-empty description."""
|
||||||
|
docs = {}
|
||||||
|
for name, path in paths_by_name.items():
|
||||||
|
desc = extract_description(path)
|
||||||
|
if not desc:
|
||||||
|
continue
|
||||||
|
tokens = tokenize(desc)
|
||||||
|
if tokens:
|
||||||
|
docs[name] = collections.Counter(tokens)
|
||||||
|
return docs
|
||||||
|
|
||||||
|
|
||||||
|
def _document_frequencies(docs):
|
||||||
|
df = collections.Counter()
|
||||||
|
for counter in docs.values():
|
||||||
|
for token in counter:
|
||||||
|
df[token] += 1
|
||||||
|
return df
|
||||||
|
|
||||||
|
|
||||||
|
def _tfidf_vector(counter, doc_count, df):
|
||||||
|
"""(1 + log tf) * log(N/df), L2-normalized."""
|
||||||
|
weights = {
|
||||||
|
token: (1 + math.log(n)) * math.log(doc_count / df[token])
|
||||||
|
for token, n in counter.items()
|
||||||
|
}
|
||||||
|
norm = math.sqrt(sum(w * w for w in weights.values())) or 1
|
||||||
|
return {token: w / norm for token, w in weights.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def tfidf_vectors(docs):
|
||||||
|
"""name -> {token: TF-IDF weight}, L2-normalized."""
|
||||||
|
doc_count = len(docs)
|
||||||
|
df = _document_frequencies(docs)
|
||||||
|
return {name: _tfidf_vector(c, doc_count, df) for name, c in docs.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def cosine(vec_a, vec_b):
|
||||||
|
return sum(weight * vec_b.get(token, 0) for token, weight in vec_a.items())
|
||||||
|
|
||||||
|
|
||||||
|
def score_pairs(vectors):
|
||||||
|
"""[(score, a, b), ...] over every pair, sorted by score descending."""
|
||||||
|
pairs = [
|
||||||
|
(cosine(vectors[a], vectors[b]), a, b)
|
||||||
|
for a, b in itertools.combinations(sorted(vectors), 2)
|
||||||
|
]
|
||||||
|
pairs.sort(reverse=True)
|
||||||
|
return pairs
|
||||||
|
|
||||||
|
|
||||||
|
def _thresholds():
|
||||||
|
warn = float(os.environ.get("SKILL_ROUTING_WARN", WARN_DEFAULT))
|
||||||
|
fail = float(os.environ.get("SKILL_ROUTING_FAIL", FAIL_DEFAULT))
|
||||||
|
return warn, fail
|
||||||
|
|
||||||
|
|
||||||
|
def report(doc_count, pairs):
|
||||||
|
"""Print the census report; return True iff a pair reached FAIL."""
|
||||||
|
warn, fail = _thresholds()
|
||||||
|
print(f"skills with description: {doc_count}")
|
||||||
|
for score, a, b in pairs[:10]:
|
||||||
|
print(f"{score:.2f} {a} ~ {b}")
|
||||||
|
failed = False
|
||||||
|
for score, a, b in pairs:
|
||||||
|
if score >= fail:
|
||||||
|
print(f"FAIL {score:.2f} {a} ~ {b}")
|
||||||
|
failed = True
|
||||||
|
elif score >= warn:
|
||||||
|
print(f"WARN {score:.2f} {a} ~ {b}")
|
||||||
|
return failed
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
by_name = dedup_by_skill_name(discover_paths(resolve_roots()))
|
||||||
|
docs = build_docs(by_name)
|
||||||
|
vectors = tfidf_vectors(docs)
|
||||||
|
failed = report(len(docs), score_pairs(vectors))
|
||||||
|
return 1 if failed else 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Executable
+113
@@ -0,0 +1,113 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# lib/tests/skill-routing-census.test.sh — TF-IDF cosine census of
|
||||||
|
# skill-description collisions across the live catalog (routing ambiguity),
|
||||||
|
# adapted from addyosmani/agent-skills evals Tier 2. Two skills whose
|
||||||
|
# descriptions read alike route the same prompt to both — a routing
|
||||||
|
# collision, not a cosmetic naming clash.
|
||||||
|
#
|
||||||
|
# Fixture flip-test math note: with only 2 documents, corpus-wide IDF zeroes
|
||||||
|
# EVERY term's contribution — a term unique to one doc (df=1) is absent from
|
||||||
|
# the other vector and can't enter the dot product, a term shared by both
|
||||||
|
# (df=2) gets idf=log(N/df)=log(1)=0. Cosine is trivially 0 for any 2-doc
|
||||||
|
# corpus, near-duplicate or not — a bare 2-doc "distinct pair" fixture would
|
||||||
|
# pass for that reason alone, not because the pair is actually distinct. Both
|
||||||
|
# fixtures below ride in a 4-doc corpus so IDF carries real signal. The
|
||||||
|
# distinct-pair fixture also carries its own same-corpus positive control (a
|
||||||
|
# near-dup pair that must FAIL) and a sensitivity re-run where the "distinct"
|
||||||
|
# partner is swapped for a near-copy of its counterpart, to prove the WARN/
|
||||||
|
# FAIL marker actually fires when the pair collides — not just that it stays
|
||||||
|
# silent for reasons unrelated to distinctness.
|
||||||
|
set -u
|
||||||
|
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||||
|
CENSUS="$ROOT/lib/skill-routing-census.py"
|
||||||
|
pass=0; fail=0
|
||||||
|
check() { if [ "$2" = "$3" ]; then pass=$((pass+1)); else fail=$((fail+1));
|
||||||
|
printf 'FAIL %s: got[%s] want[%s]\n' "$1" "$2" "$3"; fi; }
|
||||||
|
|
||||||
|
WORK="$(mktemp -d)"; trap 'rm -rf "$WORK"' EXIT
|
||||||
|
TRANSLATE_DESC='description: "Translate a PDF, keeping layout and images."'
|
||||||
|
SECAUDIT_DESC="description: 'Run a security audit for secrets, CVEs, OWASP.'"
|
||||||
|
|
||||||
|
# _mkskill <root> <name> <frontmatter-line>... — a fixture skill/SKILL.md
|
||||||
|
_mkskill() {
|
||||||
|
local dir="$1" name="$2"; shift 2
|
||||||
|
mkdir -p "$dir/$name"
|
||||||
|
{ echo "---"; echo "name: $name"; printf '%s\n' "$@"; echo "---"; } \
|
||||||
|
> "$dir/$name/SKILL.md"
|
||||||
|
}
|
||||||
|
|
||||||
|
# _mkcontrol <root> — same-corpus positive control: skill-e/skill-f, a
|
||||||
|
# near-duplicate deploy-runbook pair that must FAIL wherever it rides.
|
||||||
|
_mkcontrol() {
|
||||||
|
local dir="$1"
|
||||||
|
_mkskill "$dir" skill-e "description: |" \
|
||||||
|
" Use when deploying a project via its per-project runbook to ship the" \
|
||||||
|
" application to production servers safely with rollback support and" \
|
||||||
|
" health checks after each release."
|
||||||
|
_mkskill "$dir" skill-f "description: >-" \
|
||||||
|
" Use when deploying a project via its per-project deployment runbook to" \
|
||||||
|
" ship the application to production servers safely with rollback" \
|
||||||
|
" support and health checks after each release."
|
||||||
|
}
|
||||||
|
|
||||||
|
# ── live run: this machine's real catalog (a FAIL line here → suite RED) ──
|
||||||
|
live_out=$(python3 "$CENSUS" 2>&1); live_rc=$?
|
||||||
|
printf '%s\n' "$live_out"
|
||||||
|
check T1-live-run-no-collision "$live_rc" 0
|
||||||
|
has_count=$(printf '%s\n' "$live_out" \
|
||||||
|
| grep -qE 'skills with description: [0-9]{2,}' && echo yes)
|
||||||
|
check T2-live-catalog-non-trivial "$has_count" yes
|
||||||
|
|
||||||
|
# ── fixture AB: near-duplicate pair + 2 unrelated fillers (N=4, real signal) ──
|
||||||
|
AB="$WORK/ab"
|
||||||
|
_mkskill "$AB" skill-a "description: |" \
|
||||||
|
" Use when deploying a project via its per-project runbook to ship the" \
|
||||||
|
" application to production servers safely with rollback support and" \
|
||||||
|
" health checks after each release."
|
||||||
|
_mkskill "$AB" skill-b "description: >-" \
|
||||||
|
" Use when deploying a project via its per-project deployment runbook to" \
|
||||||
|
" ship the application to production servers safely with rollback" \
|
||||||
|
" support and health checks after each release."
|
||||||
|
_mkskill "$AB" filler-translate "$TRANSLATE_DESC"
|
||||||
|
_mkskill "$AB" filler-secaudit "$SECAUDIT_DESC"
|
||||||
|
out_ab=$(SKILL_ROUTING_ROOTS="$AB" python3 "$CENSUS" 2>&1)
|
||||||
|
fail_line=$(printf '%s\n' "$out_ab" \
|
||||||
|
| grep -qE '^FAIL 0\.[0-9]{2} skill-a ~ skill-b$' && echo yes)
|
||||||
|
check T3-collision-fail-line "$fail_line" yes
|
||||||
|
[ "$fail_line" = yes ] && echo FIXTURE_COLLISION_DETECTED
|
||||||
|
|
||||||
|
# ── fixture CD: distinct pair + same-corpus positive control (N=4) ────────
|
||||||
|
# skill-e/skill-f (positive control) must FAIL; skill-c/skill-d (the pair
|
||||||
|
# under test) must stay silent (no WARN/FAIL line naming them).
|
||||||
|
CD="$WORK/cd"
|
||||||
|
_mkcontrol "$CD"
|
||||||
|
_mkskill "$CD" skill-c "$TRANSLATE_DESC"
|
||||||
|
_mkskill "$CD" skill-d "$SECAUDIT_DESC"
|
||||||
|
out_cd=$(SKILL_ROUTING_ROOTS="$CD" python3 "$CENSUS" 2>&1)
|
||||||
|
control_fail=$(printf '%s\n' "$out_cd" \
|
||||||
|
| grep -qE '^FAIL 0\.[0-9]{2} skill-e ~ skill-f$' && echo yes)
|
||||||
|
check T4-positive-control-fail "$control_fail" yes
|
||||||
|
distinct_marker=$(printf '%s\n' "$out_cd" \
|
||||||
|
| grep -E '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$')
|
||||||
|
distinct_silent=$([ -z "$distinct_marker" ] && echo yes || echo no)
|
||||||
|
check T5-distinct-pair-silent "$distinct_silent" yes
|
||||||
|
|
||||||
|
# ── sensitivity re-run: skill-d -> near-copy of skill-c ────────────────────
|
||||||
|
# Same corpus, but skill-d's description now near-duplicates skill-c's: the
|
||||||
|
# marker that stayed silent above must appear here, proving T5 wasn't silent
|
||||||
|
# by construction (e.g. a broken pattern or a threshold nothing can cross).
|
||||||
|
CD2="$WORK/cd-nearcopy"
|
||||||
|
_mkcontrol "$CD2"
|
||||||
|
_mkskill "$CD2" skill-c "$TRANSLATE_DESC"
|
||||||
|
_mkskill "$CD2" skill-d \
|
||||||
|
'description: "Translate a PDF document, keeping the layout and images intact."'
|
||||||
|
out_cd2=$(SKILL_ROUTING_ROOTS="$CD2" python3 "$CENSUS" 2>&1)
|
||||||
|
sensitivity_marker=$(printf '%s\n' "$out_cd2" \
|
||||||
|
| grep -qE '^(WARN|FAIL) 0\.[0-9]{2} skill-c ~ skill-d$' && echo yes)
|
||||||
|
check T6-sensitivity-marker-appears "$sensitivity_marker" yes
|
||||||
|
|
||||||
|
[ "$control_fail" = yes ] && [ "$distinct_silent" = yes ] \
|
||||||
|
&& [ "$sensitivity_marker" = yes ] && echo FIXTURE_DISTINCT_OK
|
||||||
|
|
||||||
|
echo "PASS=$pass FAIL=$fail"
|
||||||
|
[ "$fail" -eq 0 ]
|
||||||
Reference in New Issue
Block a user