feat(vault): Phase 3 tools + batch citation/relation enrichment

This commit is contained in:
2026-07-01 15:20:40 +08:00
parent 8273017082
commit 407a9213bf
548 changed files with 53410 additions and 2113 deletions
+224
View File
@@ -0,0 +1,224 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
review-pages.py — Phase 3 wiki 页面评审器
两种模式:
1. LLM 评审(需 Ollama 后端):调用本地模型审核内容准确性,检测声明与 raw 来源
不匹配、跨页面矛盾、过时信息。
2. 静态评审(LLM 不可用时自动降级):基于 frontmatter 规则与 raw 来源比对,
检测低置信度、缺 source、断链、过时等结构问题。无外部依赖。
B-3.3 低置信度自动建议:confidence <= 2 的页面生成"需要更多证据"提示。
B-3.1 LLM 评审:可检测语义偏差(被 AI 过度概括/扭曲的结论)。
用法:
python tools/scripts/review-pages.py --stale # 审查过时页面
python tools/scripts/review-pages.py --random 5 # 随机抽 5 页
python tools/scripts/review-pages.py --page "LLM-Wiki-v2" # 指定单页
python tools/scripts/review-pages.py --confidence-low # 低置信度页面
python tools/scripts/review-pages.py --random 3 --llm # 强制启用 LLM 评审
python tools/scripts/review-pages.py --json # JSON 输出
"""
import argparse
import json
import random
import re
import shutil
import subprocess
import sys
from datetime import date
from pathlib import Path
try:
import yaml
except ImportError:
yaml = None
VAULT = Path(r"D:\Applications\app\kepano-obsidian-main")
WIKI = VAULT / "wiki"
RAW = VAULT / "raw"
TODAY = date.today()
LLM_MODEL = "qwen2.5:7b"
REVIEW_PROMPT = """请审核以下 Obsidian wiki 页面,输出 JSON(仅 JSON,无多余文字):
{
"accuracy_score": 1到5的整数,
"issues": ["具体问题描述"],
"confidence_match": true或false,
"suggested_updates": ["可执行的改进建议"]
}
重点关注:数字/百分比是否标注来源行号、结论是否被过度概括、frontmatter 是否完整。
页面内容(截断至 4000 字符):
{content}
"""
def read_frontmatter(fp: Path) -> tuple[dict, str]:
content = fp.read_text(encoding="utf-8")
m = re.match(r"^---\n(.*?)\n---", content, re.DOTALL)
if not m or not yaml:
return {}, content
try:
return yaml.safe_load(m.group(1)) or {}, content
except Exception:
return {}, content
def ollama_available() -> bool:
return shutil.which("ollama") is not None
def llm_review(page_name: str, content: str) -> dict | None:
"""调用 Ollama 审核单页,返回结构化结果"""
if not ollama_available():
return None
prompt = REVIEW_PROMPT.format(content=content[:4000])
try:
result = subprocess.run(
["ollama", "run", LLM_MODEL, prompt],
capture_output=True, text=True, timeout=180, encoding="utf-8",
)
out = result.stdout.strip()
start, end = out.find("{"), out.rfind("}")
if start != -1 and end != -1:
return json.loads(out[start:end + 1])
except Exception:
pass
return None
def static_review(fp: Path) -> dict:
"""静态规则评审:无 LLM 依赖的结构化检查"""
front, content = read_frontmatter(fp)
issues = []
suggestions = []
cats = front.get("categories", []) or []
if not any("LLM Wiki" in str(c) for c in cats):
issues.append("categories 缺少 [[LLM Wiki]]")
tags = front.get("tags", []) or []
if "wiki" not in tags:
issues.append("tags 缺少 wiki")
if not front.get("type"):
issues.append("缺少 type 字段")
source = front.get("source", "")
if not source or str(source).strip('[]"') == "":
issues.append("缺少 sourcewiki 层必填)")
suggestions.append(f"补充 source 指向 raw/ 来源,参考同类页面:{fp.stem}")
conf = front.get("confidence")
if conf is not None and isinstance(conf, (int, float)) and conf <= 2:
issues.append(f"低置信度 (confidence={conf})")
suggestions.append("需要更多证据:补充多源交叉验证或将 confidence 提升至 3+")
last = str(front.get("last_reviewed", ""))
if last:
try:
overdue = (TODAY - date.fromisoformat(last)).days
interval = front.get("review_interval_days", 180)
if overdue > interval:
issues.append(f"过时(上次审查 {last},超期 {overdue - interval} 天)")
suggestions.append(f"复查内容并更新 last_reviewed 为 {TODAY.isoformat()}")
except ValueError:
issues.append(f"last_reviewed 格式无效:{last}")
no_line = re.findall(r"\[raw:[^:\]\[]+\]", content)
if no_line:
issues.append(f"{len(no_line)} 条 raw 引用缺少行号")
suggestions.append("运行 python tools/scripts/fix-raw-citations.py 自动补全行号")
score = 5 - min(4, len([i for i in issues if i.startswith(("低置信", "过时", "缺少 source"))]))
return {
"page": fp.stem,
"mode": "static",
"accuracy_score": max(1, score),
"issues": issues,
"confidence_match": (conf is None or conf >= 3),
"suggested_updates": suggestions,
}
def select_pages(args) -> list[Path]:
all_pages = [p for p in sorted(WIKI.glob("*.md"))
if p.name not in ("index.md", "log.md")]
if args.page:
fp = WIKI / f"{args.page}.md"
return [fp] if fp.exists() else []
if args.confidence_low:
return [p for p in all_pages
if (lambda c: c is not None and isinstance(c, (int, float)) and c <= 2)(
read_frontmatter(p)[0].get("confidence"))]
if args.stale:
selected = []
for p in all_pages:
front, _ = read_frontmatter(p)
last = str(front.get("last_reviewed", ""))
interval = front.get("review_interval_days", 180)
if not last:
selected.append(p)
continue
try:
if (TODAY - date.fromisoformat(last)).days > interval:
selected.append(p)
except ValueError:
selected.append(p)
return selected
if args.random:
return random.sample(all_pages, min(args.random, len(all_pages)))
return all_pages
def main():
parser = argparse.ArgumentParser(description="Review wiki pages (LLM or static fallback)")
sel = parser.add_mutually_exclusive_group()
sel.add_argument("--stale", action="store_true", help="Review stale pages only")
sel.add_argument("--random", type=int, metavar="N", help="Randomly sample N pages")
sel.add_argument("--page", help="Review a single page")
sel.add_argument("--confidence-low", action="store_true", help="Review confidence<=2 pages")
parser.add_argument("--llm", action="store_true", help="Force LLM review (requires Ollama)")
parser.add_argument("--json", action="store_true", help="Output JSON")
args = parser.parse_args()
pages = select_pages(args)
if not pages:
print("No pages matched selection criteria.")
return
use_llm = args.llm or (not args.confidence_low)
llm_on = use_llm and ollama_available()
if use_llm and not llm_on:
print("(Ollama 不可用,降级为静态评审模式。安装 Ollama 以启用 LLM 语义审核。)\n")
reports = []
for fp in pages:
front, content = read_frontmatter(fp)
if llm_on:
llm = llm_review(fp.stem, content)
report = llm if llm else static_review(fp)
if report:
report["mode"] = "llm"
else:
report = static_review(fp)
report.setdefault("page", fp.stem)
reports.append(report)
if args.json:
print(json.dumps(reports, ensure_ascii=False, indent=2))
return
print(f"=== Review Report ({'LLM' if llm_on else 'static'} mode, {len(reports)} pages) ===\n")
needs_action = [r for r in reports if r["issues"]]
for r in reports:
flag = "OK " if not r["issues"] else "FIX"
print(f"[{flag}] [[{r['page']}]] — score {r['accuracy_score']}/5")
for issue in r["issues"]:
print(f"{issue}")
for sug in r["suggested_updates"]:
print(f"{sug}")
print(f"\nSummary: {len(needs_action)}/{len(reports)} pages need attention")
if __name__ == "__main__":
main()