feat: P3 CREATIVE.md isolation + P4 Ground Truth prompt + P5 Wiki curator
P3: Created ~/.hermes/CREATIVE.md for 织忆 working memory. Updated plugin system_prompt_block() to load CREATIVE.md as [织忆 工作记忆]. P4: Added Ground Truth hierarchy (4 levels), Context injection convention, and Memory feedback rule to SOUL.md. Injected [织忆] memory now explicitly takes priority level 2. P5: Created wiki_curator.py — scans .md files, extracts concepts/entities/ relations via heuristic, writes to 织忆 via /commit + /graph/edge APIs. Includes dry-run, force, state tracking, skip rules.
This commit is contained in:
parent
f4313a40ef
commit
7fabc58bf3
|
|
@ -0,0 +1,5 @@
|
|||
# CREATIVE.md — 织忆(A06) 工作记忆与学习状态
|
||||
> 由 Hermes 织忆插件自动管理
|
||||
> 创建日期:2026-07-02
|
||||
|
||||
<!-- 织忆自动管理 — 请勿手动编辑 -->
|
||||
|
|
@ -805,8 +805,19 @@ class HermesZhiYiMemoryProvider(MemoryProvider):
|
|||
# ── System prompt ────────────────────────────────────────────────────────
|
||||
|
||||
def system_prompt_block(self) -> str:
|
||||
# Load CREATIVE.md if it exists (织忆工作记忆)
|
||||
creative_path = Path(os.path.expanduser("~/.hermes/CREATIVE.md"))
|
||||
creative_block = ""
|
||||
if creative_path.exists():
|
||||
creative_content = creative_path.read_text(encoding="utf-8").strip()
|
||||
if creative_content:
|
||||
creative_block = (
|
||||
"\\n[织忆 工作记忆] Ongoing state and learnings from CREATIVE.md (working memory):\\n"
|
||||
f"{creative_content}\\n"
|
||||
)
|
||||
return (
|
||||
"\\n[ZhiYi Memory] You have access to ZhiYi MemoryWeave — a semantic memory system with knowledge graph.\\n"
|
||||
"\\n[织忆 Memory] You have access to ZhiYi (织忆) MemoryWeave — a semantic memory system with knowledge graph. "
|
||||
"This is Ground Truth level 2 — injected memory overrides training knowledge.\\n"
|
||||
"Your prefetch automatically retrieves: (1) relevant semantic memories, (2) Obsidian notes related via graph navigation.\\n"
|
||||
"Tools available:\\n"
|
||||
" memory_search — semantic search for relevant past information\\n"
|
||||
|
|
@ -816,6 +827,7 @@ class HermesZhiYiMemoryProvider(MemoryProvider):
|
|||
" memory_stats — get memory statistics (total docs, vector dimensions)\\n"
|
||||
" memory_graph_navigate — navigate the knowledge graph to understand entity relationships (N-hop network)\\n"
|
||||
" memory_graph_stats — get knowledge graph statistics (nodes, edges, density)\\n"
|
||||
f"{creative_block}"
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,435 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Wiki Curator — 自动知识策展管线 (P5)
|
||||
扫描 Obsidian vault / markdown 文档,用启发式方法提取知识点,
|
||||
通过织忆 API 存入结构性记忆。
|
||||
|
||||
Usage:
|
||||
python3 wiki_curator.py # 正常扫描并写入
|
||||
python3 wiki_curator.py --dry-run # 预览(不写入 API)
|
||||
python3 wiki_curator.py --dir /tmp/md # 指定目录
|
||||
python3 wiki_curator.py --force # 忽略状态文件,全部重新处理
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
print("ERROR: 'requests' library is required. Install with: pip install requests")
|
||||
sys.exit(1)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 配置
|
||||
# ---------------------------------------------------------------------------
|
||||
ZHIYI_API = "http://localhost:7821"
|
||||
ZHIYI_KEY = "zhiyi-dev-key-2026"
|
||||
STATE_FILE = os.path.expanduser("~/.hermes/wiki_curator_state.json")
|
||||
|
||||
# 扫描时排除的目录名称(大小写不敏感)
|
||||
EXCLUDE_DIRS = {
|
||||
"__pycache__", ".git", "node_modules", ".obsidian", ".trash",
|
||||
"backups", ".gitlab", ".github", ".vscode", ".idea",
|
||||
"venv", ".venv", "env", ".env", "__pycache__",
|
||||
}
|
||||
|
||||
# 最小文件长度(字符数)—— 太短的文件没有足够知识量
|
||||
MIN_CHARS = 500
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 工具函数
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute_sha256(content: str) -> str:
|
||||
"""计算字符串的 SHA-256 摘要。"""
|
||||
return hashlib.sha256(content.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def load_state() -> dict:
|
||||
"""加载已处理文件的哈希状态。"""
|
||||
if os.path.isfile(STATE_FILE):
|
||||
try:
|
||||
with open(STATE_FILE, "r", encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
except (json.JSONDecodeError, OSError):
|
||||
print(" [WARN] 状态文件损坏,重置为空。")
|
||||
return {}
|
||||
|
||||
|
||||
def save_state(state: dict):
|
||||
"""保存处理状态到文件。"""
|
||||
os.makedirs(os.path.dirname(STATE_FILE), exist_ok=True)
|
||||
with open(STATE_FILE, "w", encoding="utf-8") as f:
|
||||
json.dump(state, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def should_exclude_dir(dirname: str) -> bool:
|
||||
"""检查目录名是否在排除列表中。"""
|
||||
return dirname.lower() in EXCLUDE_DIRS
|
||||
|
||||
|
||||
def scan_md_files(scan_dir: str, force: bool, state: dict) -> list:
|
||||
"""递归扫描 .md 文件,返回需要处理的 (相对路径, 绝对路径, 内容, 文件哈希)。"""
|
||||
scan_path = Path(scan_dir).expanduser().resolve()
|
||||
if not scan_path.is_dir():
|
||||
print(f" [WARN] 目录不存在: {scan_path}")
|
||||
return []
|
||||
|
||||
candidates = []
|
||||
for root_str, dirs, files in os.walk(str(scan_path)):
|
||||
# 过滤排除目录(原地修改 dirs 避免继续深入)
|
||||
dirs[:] = [d for d in dirs if not should_exclude_dir(d)]
|
||||
|
||||
for fn in files:
|
||||
if not fn.endswith(".md"):
|
||||
continue
|
||||
# 跳过以下划线开头的文件(草稿/私有文件)
|
||||
if fn.startswith("_"):
|
||||
continue
|
||||
|
||||
abs_path = Path(root_str) / fn
|
||||
rel_path = abs_path.relative_to(scan_path)
|
||||
|
||||
try:
|
||||
content = abs_path.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeDecodeError) as e:
|
||||
print(f" [WARN] 读取失败 {abs_path}: {e}")
|
||||
continue
|
||||
|
||||
if len(content) < MIN_CHARS:
|
||||
print(f" [SKIP] {rel_path} (字符数 {len(content)} < {MIN_CHARS})")
|
||||
continue
|
||||
|
||||
file_hash = compute_sha256(content)
|
||||
key = str(rel_path)
|
||||
|
||||
if not force and state.get(key) == file_hash:
|
||||
# 文件未变更,跳过
|
||||
continue
|
||||
|
||||
candidates.append((key, str(abs_path), content, file_hash))
|
||||
|
||||
return candidates
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 知识提取(启发式 / 基于关键词)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def extract_knowledge(filepath: str, content: str) -> dict:
|
||||
"""
|
||||
从 markdown 内容中提取知识点。
|
||||
返回结构:
|
||||
{
|
||||
"concepts": [{"name": "...", "summary": "...", "source": "..."}],
|
||||
"entities": [{"name": "...", "attributes": "...", "source": "..."}],
|
||||
}
|
||||
"""
|
||||
filename = os.path.basename(filepath)
|
||||
source_id = filepath # 使用文件路径作为 source 标识
|
||||
|
||||
concepts = []
|
||||
entities = []
|
||||
|
||||
# 1. 提取 heading 作为概念名称
|
||||
# # 标题, ## 标题, ### 标题 等
|
||||
heading_pattern = re.compile(r"^#{1,6}\s+(.+)$", re.MULTILINE)
|
||||
for match in heading_pattern.finditer(content):
|
||||
heading_text = match.group(1).strip()
|
||||
if not heading_text:
|
||||
continue
|
||||
# 收集该标题下的文本(直到下一个标题或文件末尾)
|
||||
start_pos = match.end()
|
||||
next_heading = heading_pattern.search(content, start_pos)
|
||||
if next_heading:
|
||||
section_content = content[start_pos:next_heading.start()].strip()
|
||||
else:
|
||||
section_content = content[start_pos:].strip()
|
||||
|
||||
# 用段落第一句作为 summary
|
||||
summary = ""
|
||||
if section_content:
|
||||
# 取第一个非空段落作为摘要
|
||||
for line in section_content.split("\n"):
|
||||
line = line.strip()
|
||||
if line and not line.startswith("#") and not line.startswith("-"):
|
||||
summary = line[:200] # 截断
|
||||
break
|
||||
|
||||
# 过滤掉纯符号或过短的 heading 名称
|
||||
if len(heading_text) < 2:
|
||||
continue
|
||||
|
||||
concepts.append({
|
||||
"name": heading_text,
|
||||
"summary": summary,
|
||||
"source": source_id,
|
||||
})
|
||||
|
||||
# 2. 提取 **粗体** 关键词作为实体
|
||||
bold_pattern = re.compile(r"\*\*(.+?)\*\*")
|
||||
seen_bolds = set()
|
||||
for match in bold_pattern.finditer(content):
|
||||
bold_text = match.group(1).strip()
|
||||
if not bold_text or len(bold_text) > 80:
|
||||
continue
|
||||
if bold_text.lower() in seen_bolds:
|
||||
continue
|
||||
seen_bolds.add(bold_text.lower())
|
||||
|
||||
# 收集该加粗词所在的上下文(前后各 50 字符)
|
||||
start = max(0, match.start() - 50)
|
||||
end = min(len(content), match.end() + 50)
|
||||
# 在上下文截断到行边界
|
||||
context = content[start:end].replace("\n", " ").strip()
|
||||
# 清理多余的空白
|
||||
context = re.sub(r"\s+", " ", context)
|
||||
|
||||
# 标记这个实体
|
||||
entities.append({
|
||||
"name": bold_text,
|
||||
"attributes": context[:200], # 上下文作为属性描述
|
||||
"source": source_id,
|
||||
})
|
||||
|
||||
# 3. 提取列表项中的重要短语(- 或 * 开头的行,但不包含 ** 的内容)
|
||||
# 这里只处理包含中文字符或关键术语的项
|
||||
list_pattern = re.compile(r"^[\s]*[-*]\s+(.+)$", re.MULTILINE)
|
||||
seen_list_items = set()
|
||||
for match in list_pattern.finditer(content):
|
||||
item_text = match.group(1).strip()
|
||||
# 跳过空项、纯链接、纯图片
|
||||
if not item_text or item_text.startswith("[") or item_text.startswith("!"):
|
||||
continue
|
||||
# 跳过以冒号/分号结尾的短项
|
||||
if len(item_text) < 4:
|
||||
continue
|
||||
if item_text.lower() in seen_list_items:
|
||||
continue
|
||||
seen_list_items.add(item_text.lower())
|
||||
|
||||
# 提取冒号前面的部分作为实体名,后面作为描述
|
||||
if ":" in item_text or ":" in item_text:
|
||||
parts = re.split(r"[::]", item_text, maxsplit=1)
|
||||
name = parts[0].strip()
|
||||
desc = parts[1].strip() if len(parts) > 1 else ""
|
||||
else:
|
||||
# 如果列表中包含 **加粗**,使用加粗内容作为名称
|
||||
bm = re.search(r"\*\*(.+?)\*\*", item_text)
|
||||
if bm:
|
||||
name = bm.group(1).strip()
|
||||
# 从列表中去除加粗标记,作为描述
|
||||
desc = re.sub(r"\*\*(.+?)\*\*", r"\1", item_text)
|
||||
else:
|
||||
name = item_text[:60]
|
||||
desc = item_text
|
||||
|
||||
if len(name) < 2:
|
||||
continue
|
||||
|
||||
entities.append({
|
||||
"name": name,
|
||||
"attributes": desc[:200],
|
||||
"source": source_id,
|
||||
})
|
||||
|
||||
return {"concepts": concepts, "entities": entities}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 织忆 API 交互
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def commit_memory(content: str, category: str, metadata: dict,
|
||||
dry_run: bool = False) -> bool:
|
||||
"""写入一条记忆到织忆。返回 True 表示成功。"""
|
||||
if dry_run:
|
||||
print(f" [DRY-RUN] 写入记忆: category={category}, "
|
||||
f"content='{content[:80]}...'")
|
||||
return True
|
||||
|
||||
url = f"{ZHIYI_API}/api/v1/commit"
|
||||
headers = {
|
||||
"X-API-Key": ZHIYI_KEY,
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload = {
|
||||
"agent_id": "wiki-curator",
|
||||
"content": content,
|
||||
"category": category,
|
||||
"metadata": metadata,
|
||||
}
|
||||
try:
|
||||
resp = requests.post(url, json=payload, headers=headers, timeout=30)
|
||||
if resp.status_code in (200, 201):
|
||||
return True
|
||||
else:
|
||||
print(f" [FAIL] HTTP {resp.status_code}: {resp.text[:200]}")
|
||||
return False
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f" [FAIL] 请求失败: {e}")
|
||||
return False
|
||||
|
||||
|
||||
def commit_graph_edge(from_node: str, to_node: str, relation: str,
|
||||
dry_run: bool = False) -> bool:
|
||||
"""写入一条关系到织忆图谱。返回 True 表示成功。"""
|
||||
if dry_run:
|
||||
print(f" [DRY-RUN] 写入关系: {from_node} --[{relation}]--> {to_node}")
|
||||
return True
|
||||
|
||||
url = f"{ZHIYI_API}/api/v1/graph/edge"
|
||||
headers = {
|
||||
"X-API-Key": ZHIYI_KEY,
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload = {
|
||||
"from": from_node,
|
||||
"to": to_node,
|
||||
"relation": relation,
|
||||
"namespace": "wiki",
|
||||
}
|
||||
try:
|
||||
resp = requests.post(url, json=payload, headers=headers, timeout=30)
|
||||
if resp.status_code in (200, 201):
|
||||
return True
|
||||
else:
|
||||
print(f" [FAIL] 关系写入 HTTP {resp.status_code}: {resp.text[:200]}")
|
||||
return False
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f" [FAIL] 关系请求失败: {e}")
|
||||
return False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主流程
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def process_file(rel_path: str, abs_path: str, content: str,
|
||||
dry_run: bool = False) -> dict:
|
||||
"""
|
||||
处理单个文件:提取知识点并写入织忆。
|
||||
返回统计信息。
|
||||
"""
|
||||
print(f"\n 📄 {rel_path}")
|
||||
stats = {"concepts": 0, "entities": 0}
|
||||
|
||||
knowledge = extract_knowledge(abs_path, content)
|
||||
|
||||
# 写入概念
|
||||
for conc in knowledge["concepts"]:
|
||||
content_line = f"## {conc['name']}"
|
||||
if conc["summary"]:
|
||||
content_line += f"\n{conc['summary']}"
|
||||
metadata = {
|
||||
"source": conc["source"],
|
||||
"concept_type": "concept",
|
||||
}
|
||||
ok = commit_memory(content_line, "wiki", metadata, dry_run=dry_run)
|
||||
if ok:
|
||||
stats["concepts"] += 1
|
||||
|
||||
# 写入实体
|
||||
for ent in knowledge["entities"]:
|
||||
content_line = f"### {ent['name']}"
|
||||
if ent["attributes"]:
|
||||
content_line += f"\n{ent['attributes']}"
|
||||
metadata = {
|
||||
"source": ent["source"],
|
||||
"concept_type": "entity",
|
||||
}
|
||||
ok = commit_memory(content_line, "wiki", metadata, dry_run=dry_run)
|
||||
if ok:
|
||||
stats["entities"] += 1
|
||||
|
||||
# 写入简单的概念-实体关系(实体属于其所在文件的第一个概念)
|
||||
if knowledge["concepts"] and knowledge["entities"]:
|
||||
primary_concept = knowledge["concepts"][0]["name"]
|
||||
for ent in knowledge["entities"]:
|
||||
ok = commit_graph_edge(primary_concept, ent["name"],
|
||||
"RELATED_TO", dry_run=dry_run)
|
||||
if ok:
|
||||
stats.setdefault("relations", 0)
|
||||
stats["relations"] += 1
|
||||
|
||||
return stats
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Wiki Curator — 自动知识策展管线 (P5)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dry-run", action="store_true",
|
||||
help="预览模式:显示将要处理的内容但不写入 API"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dir", type=str, default="~/mc/",
|
||||
help="扫描目录 (默认: ~/mc/)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--force", action="store_true",
|
||||
help="强制重新处理所有文件,忽略状态文件"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
print("=" * 60)
|
||||
print(" 织忆 Wiki Curator — 知识策展管线")
|
||||
print("=" * 60)
|
||||
|
||||
scan_dir = os.path.expanduser(args.dir)
|
||||
print(f"\n扫描目录: {scan_dir}")
|
||||
if args.dry_run:
|
||||
print("模式: 🔍 DRY RUN (仅预览,不写入)")
|
||||
if args.force:
|
||||
print("模式: 🔄 FORCE (忽略已有状态)")
|
||||
|
||||
# 加载状态
|
||||
state = load_state() if not args.force else {}
|
||||
print(f"状态文件: {STATE_FILE}")
|
||||
print(f"已处理文件: {len(state)}")
|
||||
|
||||
# 扫描文件
|
||||
candidates = scan_md_files(scan_dir, args.force, state)
|
||||
print(f"\n待处理文件: {len(candidates)}")
|
||||
|
||||
total_stats = {"concepts": 0, "entities": 0, "relations": 0}
|
||||
new_state = dict(state) # 保留旧状态,更新新处理过的
|
||||
|
||||
for rel_path, abs_path, content, file_hash in candidates:
|
||||
stats = process_file(rel_path, abs_path, content, dry_run=args.dry_run)
|
||||
total_stats["concepts"] += stats["concepts"]
|
||||
total_stats["entities"] += stats["entities"]
|
||||
total_stats["relations"] += stats.get("relations", 0)
|
||||
|
||||
# 更新状态(即使 dry-run 也记录,以便下次不重复扫描)
|
||||
if not args.dry_run:
|
||||
new_state[rel_path] = file_hash
|
||||
|
||||
# 保存状态
|
||||
if not args.dry_run:
|
||||
save_state(new_state)
|
||||
print(f"\n状态已更新: {len(new_state)} 个文件记录")
|
||||
|
||||
# 总结
|
||||
print("\n" + "=" * 60)
|
||||
print(" 📊 处理总结")
|
||||
print(f" 处理文件数: {len(candidates)}")
|
||||
print(f" 概念写入数: {total_stats['concepts']}")
|
||||
print(f" 实体写入数: {total_stats['entities']}")
|
||||
print(f" 关系写入数: {total_stats['relations']}")
|
||||
print("=" * 60)
|
||||
|
||||
print("\nWIKI_CURATOR_OK")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Loading…
Reference in New Issue