#!/usr/bin/env python3 """ Skill Curator — 系统化技能管理工具 =================================== 用法: python3 ~/.hermes/scripts/skill-curator.py # 完整报告 python3 ~/.hermes/scripts/skill-curator.py --report # 只出报告,不动 python3 ~/.hermes/scripts/skill-curator.py --clean # 清理 python3 ~/.hermes/scripts/skill-curator.py --check "技能名" # 创建前检查重叠 python3 ~/.hermes/scripts/skill-curator.py --check "技能名" --verbose # 显示匹配详情 技能层级: Tier 0 (core) — 高频使用、有版本号、维护良好 Tier 1 (active) — 偶尔使用、功能完整 Tier 2 (legacy) — 可用但过时,需审查后升级或归档 Tier 3 (archive)— 不再使用,无引用,可安全删除 Tier Q (quarantine) — 已吸收/损坏/被替代,标记待删 生命周期: Create → Draft → Review → Promote (T3→T2→T1→T0) → Deprecate (T1→T2→T3→删除) """ import os, re, glob, json, sys, textwrap from collections import defaultdict from datetime import datetime SKILLS_DIR = os.path.expanduser("~/.hermes/skills") ARCHIVE_DIR = os.path.join(SKILLS_DIR, "_archive") QUARANTINE_DIR = os.path.join(SKILLS_DIR, "_quarantine") REPORT_PATH = os.path.expanduser("~/.hermes/skill-curator-report.json") # ── 读取所有技能 ────────────────────────────────── def load_all_skills(): """遍历 skills/ 下所有子目录,读取 SKILL.md frontmatter""" skills = [] for root, dirs, files in os.walk(SKILLS_DIR): if "_archive" in root or "_quarantine" in root: continue if "SKILL.md" in files: path = os.path.join(root, "SKILL.md") fm = parse_frontmatter(path) fm["dir"] = root fm["rel_dir"] = os.path.relpath(root, SKILLS_DIR) fm["files"] = [f for f in files if f != "SKILL.md"] fm["file_count"] = len(fm["files"]) fm["size_kb"] = sum(os.path.getsize(os.path.join(root, f)) for f in files) // 1024 fm["has_version"] = bool(fm.get("version")) skills.append(fm) return skills def parse_frontmatter(path): """简单 YAML frontmatter 解析""" fm = {"name": "", "description": "", "version": "", "category": "", "tags": []} with open(path) as f: content = f.read() fm["_content_length"] = len(content) # Frontmatter between --- parts = content.split("---") if len(parts) >= 3: yaml_text = parts[1] for line in yaml_text.strip().split("\n"): line = line.strip() if ":" in line: key, val = line.split(":", 1) key = key.strip() val = val.strip().strip('"').strip("'") if key == "tags": # Extract tags from YAML array or inline tag_match = re.findall(r'\[([^\]]+)\]', val) if tag_match: fm["tags"] = [t.strip().strip('"').strip("'") for t in tag_match[0].split(",")] elif val: fm["tags"] = [t.strip() for t in val.replace("[","").replace("]","").split(",") if t.strip()] else: fm["tags"] = re.findall(r'[-\w]+', content.split("tags:")[1].split("\n")[0]) else: fm[key] = val # Extract description from first non-frontmatter paragraph body = parts[2] if len(parts) >= 3 else content body = body.strip() # Take first meaningful line for line in body.split("\n"): line = line.strip().strip("#").strip(">").strip() if line and len(line) > 10 and not line.startswith("```"): fm["description"] = line[:150] break return fm # ── 质量评分 ────────────────────────────────────── def score_skill(s): """对技能质量打分 0-100""" score = 0 # 有版本号 +10 if s.get("version"): score += 10 # description 质量 +0~20 desc = s.get("description", "") if len(desc) > 20: score += 10 if len(desc) > 60: score += 5 if len(desc) > 100: score += 5 # 有 tags +10 if s.get("tags"): score += 10 # 内容长度 +0~20 length = s.get("_content_length", 0) if length > 2000: score += 10 if length > 5000: score += 10 # 有引用文件(脚本/模板等)+10 if s.get("file_count", 0) > 0: score += 10 if s.get("file_count", 0) > 5: score += 5 # 有 category -5 ~ +10 cat = s.get("category", "") if not cat: score -= 5 elif cat not in ("uncategorized", ""): score += 10 # description 具体性 +0~10 generic_words = ["tool", "skill", "use", "when", "building", "for"] if any(w in desc.lower() for w in generic_words): pass # neutral specific_patterns = ["API", "CLI", r"\d+\.\d+", "http", "config", "install"] if any(re.search(p, desc) for p in specific_patterns): score += 10 return min(100, max(0, score)) def tier_from_score(score, has_version): if has_version and score >= 60: return 0, "core" if score >= 40: return 1, "active" if score >= 20: return 2, "legacy" return 3, "archive" # ── 重叠检测 ────────────────────────────────────── def detect_overlaps(skills): """通过描述关键词检测重叠(简化版 TF-IDF)""" overlaps = [] names = [s["name"] for s in skills] descs = [s.get("description", "").lower() for s in skills] for i in range(len(skills)): for j in range(i+1, len(skills)): # Tokenize ti = set(re.findall(r'[a-z0-9\-]+', descs[i])) tj = set(re.findall(r'[a-z0-9\-]+', descs[j])) # Filter common words stopwords = {"use", "when", "for", "the", "and", "with", "that", "this", "from", "via", "to", "in", "of", "a", "an", "is", "are", "on", "at", "by", "or", "as", "be", "it", "its"} ti = ti - stopwords tj = tj - stopwords if not ti or not tj: continue intersection = ti & tj union = ti | tj jaccard = len(intersection) / len(union) if union else 0 if jaccard > 0.25 and names[i] != names[j]: overlaps.append({ "a": names[i], "b": names[j], "score": round(jaccard * 100), "common_terms": list(intersection)[:8], "a_cat": skills[i].get("category", ""), "b_cat": skills[j].get("category", ""), }) overlaps.sort(key=lambda x: -x["score"]) return overlaps # ── 报告生成 ────────────────────────────────────── def generate_report(skills): """生成完整审计报告""" overlaps = detect_overlaps(skills) # 按 tier 分组 scored = [] for s in skills: s["score"] = score_skill(s) s["tier"], s["tier_name"] = tier_from_score(s["score"], s.get("has_version", False)) scored.append(s) tiers = defaultdict(list) for s in scored: tiers[s["tier"]].append(s) # 分类统计 cats = defaultdict(list) for s in scored: cat = s.get("category", "uncategorized") or "uncategorized" cats[cat].append(s["name"]) report = { "generated_at": datetime.now().isoformat(), "summary": { "total": len(skills), "by_tier": {str(k): len(v) for k, v in sorted(tiers.items())}, "by_category": {k: len(v) for k, v in sorted(cats.items())}, "total_files": sum(s.get("file_count", 0) for s in skills), "total_size_kb": sum(s.get("size_kb", 0) for s in skills), }, "overlaps": overlaps[:20], "needs_attention": [ s["name"] for s in scored if s["tier"] >= 2 and s.get("has_version") ][:20], "stale_no_version": [ {"name": s["name"], "category": s.get("category",""), "score": s["score"]} for s in scored if not s.get("has_version") and s["tier"] < 3 ], "tier0_core": [s["name"] for s in scored if s["tier"] == 0], "tier1_active": [s["name"] for s in scored if s["tier"] == 1], "tier2_legacy": [s["name"] for s in scored if s["tier"] == 2], "tier3_archive_candidate": [s["name"] for s in scored if s["tier"] == 3 and s["score"] < 20], "all_skills": sorted([{ "name": s["name"], "version": s.get("version", "N/A"), "category": s.get("category", ""), "tier": s["tier_name"], "tier_num": s["tier"], "score": s["score"], "files": s.get("file_count", 0), "size_kb": s.get("size_kb", 0), } for s in scored], key=lambda x: (-x["tier_num"], -x["score"])), } os.makedirs(os.path.dirname(REPORT_PATH), exist_ok=True) with open(REPORT_PATH, "w") as f: json.dump(report, f, ensure_ascii=False, indent=2) return report # ── 清理执行 ────────────────────────────────────── def execute_cleanup(skills, dry_run=True): """执行清理操作: 1. 删除 'absorbed_into' 标记的技能(SKILL.md 中声明 absorbed_into 且目标存在) 2. 移动低质量无版本技能到 _archive """ actions = [] # Find skills with absorbed_into in their frontmatter for s in skills: absorbed = s.get("absorbed_into", "") if absorbed: target_dir = os.path.join(SKILLS_DIR, s.get("category", ""), absorbed) if os.path.exists(os.path.join(target_dir, "SKILL.md")): action = f"DELETE {s['rel_dir']}: absorbed by {absorbed}" actions.append(action) if not dry_run: # Move to quarantine instead of delete qdir = os.path.join(QUARANTINE_DIR, os.path.basename(s["dir"])) os.makedirs(os.path.dirname(qdir), exist_ok=True) os.rename(s["dir"], qdir) # Archive Tier 3 skills (no version, score < 20, small size) for s in skills: if not s.get("has_version") and s["score"] < 20 and s.get("size_kb", 999) < 100: if s["tier"] >= 2: action = f"ARCHIVE {s['rel_dir']}: stale (no version, score={s['score']})" actions.append(action) if not dry_run: adir = os.path.join(ARCHIVE_DIR, os.path.basename(s["dir"])) os.makedirs(os.path.dirname(adir), exist_ok=True) os.rename(s["dir"], adir) return actions # ── 主入口 ──────────────────────────────────────── def check_overlap(skills, query, verbose=False): """检查新技能名是否与已有技能重叠,返回匹配列表 对英文用 token 匹配,对中文用子串匹配(中文无空格分词)""" query_lower = query.lower() # Extract English tokens and Chinese substrings en_tokens = set(re.findall(r'[a-z0-9]+', query_lower)) # Chinese: use 2-char sliding window (bigrams) cn_chars = re.findall(r'[\u4e00-\u9fff\uff00-\uffef]', query_lower) cn_bigrams = set(cn_chars[i]+cn_chars[i+1] for i in range(len(cn_chars)-1)) if len(cn_chars) > 1 else set(cn_chars) # Also keep individual Chinese chars for short queries cn_unigrams = set(cn_chars) matches = [] for s in skills: name = s.get("name", "").lower() desc = s.get("description", "").lower() text = f"{name} {desc}" score = 0 # English token overlap text_tokens = set(re.findall(r'[a-z0-9]+', text)) if en_tokens and text_tokens: token_overlap = len(en_tokens & text_tokens) token_union = len(en_tokens | text_tokens) score = max(score, token_overlap / max(token_union, 1) * 100) # Chinese bigram overlap text_cn = re.findall(r'[\u4e00-\u9fff\uff00-\uffef]', text) text_bigrams = set(text_cn[i]+text_cn[i+1] for i in range(len(text_cn)-1)) if len(text_cn) > 1 else set(text_cn) if cn_bigrams and text_bigrams: bigram_overlap = len(cn_bigrams & text_bigrams) bigram_union = len(cn_bigrams | text_bigrams) cn_score = bigram_overlap / max(bigram_union, 1) * 100 score = max(score, cn_score) # Chinese substring match (bonus for direct match) if cn_unigrams: # Check if each Chinese char from query appears in text char_ratio = sum(1 for c in cn_unigrams if c in text) / max(len(cn_unigrams), 1) if char_ratio > 0.5: score = max(score, char_ratio * 80) if score > 10: matches.append({ "name": s["name"], "score": round(score), "category": s.get("category", ""), "version": s.get("version", "N/A"), "description": s.get("description", "")[:100] if verbose else "", }) matches.sort(key=lambda x: -x["score"]) return matches def main(): # ── --check mode ── # if "--check" in sys.argv: idx = sys.argv.index("--check") if idx + 1 < len(sys.argv): query = sys.argv[idx + 1] verbose = "--verbose" in sys.argv skills = load_all_skills() matches = check_overlap(skills, query, verbose) print(json.dumps({"query": query, "matches": matches, "total_skills": len(skills)}, ensure_ascii=False, indent=2)) return {"query": query, "matches": matches} else: print('❌ Usage: --check "skill name to check"') return {"error": "no query"} dry_run = "--clean" not in sys.argv show_report = "--report" in sys.argv or not dry_run print("=" * 60) print(f" Hermes Skill Curator") print(f" Mode: {'🟢 DRY RUN (read-only)' if dry_run else '🔴 LIVE (applying changes)'}") print(f" Skills dir: {SKILLS_DIR}") print("=" * 60) skills = load_all_skills() print(f"\n📊 Loaded {len(skills)} active skills\n") report = generate_report(skills) if show_report: print(f"\n{'─'*60}") print(f"📋 REPORT") print(f"{'─'*60}") print(f" Total: {report['summary']['total']}") print(f" Files: {report['summary']['total_files']}") print(f" Size: {report['summary']['total_size_kb']} KB") print(f"\n By Tier:") for t, count in sorted(report['summary']['by_tier'].items()): names = {"0": "Core", "1": "Active", "2": "Legacy", "3": "Archive-Candidate"} print(f" Tier {t} ({names.get(t, '?')}): {count}") print(f"\n 🏆 Core Skills:") for n in report['tier0_core'][:10]: print(f" ✅ {n}") if len(report['tier0_core']) > 10: print(f" ... and {len(report['tier0_core'])-10} more") print(f"\n ⚠️ Stale (no version): {len(report['stale_no_version'])}") for s in report['stale_no_version'][:10]: print(f" {s['name']} (score={s['score']})") print(f"\n 🔗 Overlap Pairs (score > 25%): {len(report['overlaps'])}") for o in report['overlaps'][:10]: print(f" {o['score']:3d}% {o['a']:35s} ↔ {o['b']:35s}") print(f"\n 🗑️ Archive Candidates: {len(report['tier3_archive_candidate'])}") for n in report['tier3_archive_candidate'][:10]: print(f" {n}") if not dry_run: actions = execute_cleanup(skills, dry_run=False) if actions: print(f"\n{'─'*60}") print(f"🔧 CLEANUP ACTIONS") for a in actions: print(f" {a}") else: print(f"\n ✅ No cleanup actions needed") print(f"\n{'─'*60}") print(f" Report saved: {REPORT_PATH}") print(f" Done.") # Return report as JSON for programmatic use return report if __name__ == "__main__": report = main()