Files
llm_wiki/tools/scripts/enrich-citations.py

208 lines
6.9 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
enrich-citations.py — Phase 3 P0: 批量为零引用页面补溯源链接
策略:
1. **来源节补全**:无 `## 来源` 节的页面,自动添加,包含指向 raw source 的 wikilink
2. **首段引用注入**:若 raw source 文件存在,提取页面标题关键词,
在 raw 文件中定位匹配行,注入首条 `[raw:source:行号]` 作为溯源种子
3. **存根页标记**body < 50 字的页面添加 `> [!warning] 存根页面` callout
安全措施:
- 幂等:已有 `## 来源` 节或 `[raw:` 引用则跳过
- --dry-run / --apply
- 仅处理 wiki/*.md(排除 index/log/archive
用法:
python tools/scripts/enrich-citations.py --dry-run
python tools/scripts/enrich-citations.py --apply
"""
import argparse
import re
import sys
from pathlib import Path
try:
import yaml
except ImportError:
yaml = None
VAULT = Path(r"D:\Applications\app\kepano-obsidian-main")
WIKI = VAULT / "wiki"
RAW = VAULT / "raw"
EXCLUDE = {"index.md", "log.md"}
def read_frontmatter(fp: Path) -> tuple[dict, str, str]:
content = fp.read_text(encoding="utf-8")
m = re.match(r"^---\n(.*?)\n---\n?(.*)", content, re.DOTALL)
if not m:
return {}, content, ""
fm_text = m.group(1)
body = m.group(2)
if yaml:
try:
front = yaml.safe_load(fm_text) or {}
except Exception:
front = {}
else:
front = {}
return front, body, fm_text
def get_source_filename(front: dict) -> str:
"""Extract raw filename from source field"""
source = front.get("source", "")
if not source:
return ""
s = str(source).strip().strip('"').strip("'")
s = re.sub(r"\[\[(?:raw/)?(.+?)(?:\|.*)?\]\]", r"\1", s)
s = re.sub(r"\.md$", "", s)
return s
def find_raw_file(source_name: str) -> Path | None:
"""Try to locate the raw file"""
candidates = [
RAW / f"{source_name}.md",
RAW / source_name,
]
# Also search subdirectories
for pattern in [f"**/{source_name}.md", f"**/{source_name}/index.md"]:
matches = list(RAW.glob(pattern))
if matches:
candidates.append(matches[0])
for c in candidates:
if c.exists() and c.is_file():
return c
return None
def find_keyword_line(raw_path: Path, page_name: str) -> str:
"""Find a line in raw file matching the page name or key keyword"""
try:
raw_text = raw_path.read_text(encoding="utf-8")
except Exception:
return ""
lines = raw_text.splitlines()
# Strategy: search for page name, then aliases
keywords = [page_name]
# Also try shorter keywords
if len(page_name) > 4:
keywords.append(page_name[:4])
for kw in keywords:
for i, line in enumerate(lines, 1):
if kw in line and len(line.strip()) > 5:
return f"{i}"
return ""
def has_source_section(body: str) -> bool:
return bool(re.search(r"^##\s*来源", body, re.MULTILINE))
def has_raw_citations(body: str) -> bool:
return "[raw:" in body
def is_stub(body: str) -> bool:
clean = re.sub(r"!\[\[.*?\]\]", "", body)
clean = re.sub(r"```.*?```", "", clean, flags=re.DOTALL)
clean = re.sub(r"\|.*?\|", "", clean)
clean = re.sub(r"\s+", "", clean)
return len(clean) < 150
def main():
parser = argparse.ArgumentParser(description="Enrich wiki pages with source citations")
parser.add_argument("--dry-run", action="store_true", help="Preview without writing")
parser.add_argument("--apply", action="store_true", help="Write changes to files")
args = parser.parse_args()
if not args.dry_run and not args.apply:
args.dry_run = True
stats = {"source_section": 0, "citation_seed": 0, "stub_flag": 0, "skip_has_citation": 0, "skip_no_source": 0}
for fp in sorted(WIKI.glob("*.md")):
if fp.name in EXCLUDE:
continue
front, body, fm_text = read_frontmatter(fp)
page_name = fp.stem
if has_raw_citations(body):
stats["skip_has_citation"] += 1
continue
source_name = get_source_filename(front)
raw_path = find_raw_file(source_name) if source_name else None
modified = False
additions = []
# 1. Add citation seed if raw file exists
if raw_path:
line_num = find_keyword_line(raw_path, page_name)
if line_num:
# Find first paragraph in body and append citation
first_para_match = re.search(r"^(.+?)(?:\n\n|\n#)", body, re.DOTALL)
if first_para_match:
insert_point = first_para_match.end()
cite = f"[raw:{source_name}:{line_num}]"
if cite not in body:
body = body[:insert_point] + f" {cite}" + body[insert_point:]
modified = True
stats["citation_seed"] += 1
additions.append(f"citation seed [raw:{source_name}:{line_num}]")
# 2. Add/fix source section
if not has_source_section(body):
source_link = front.get("source", "")
if source_link:
source_section = f"\n\n## 来源\n\n> **溯源规则**:所有数字/百分比/具体结论必须标注 `[raw:{{文件名}}:{{行号}}]` 格式。\n\n- {source_link}"
body = body.rstrip() + source_section
modified = True
stats["source_section"] += 1
additions.append("source section")
# 3. Flag stub pages
if is_stub(body) and not args.dry_run:
if "> [!warning] 存根页面" not in body:
stub_callout = "\n\n> [!warning] 存根页面\n> 本页内容不足,待扩充。参考同类黄金标准页面(如 [[DSpark]]、[[何长工]])补充来源引用和正文内容。\n"
body = body.rstrip() + stub_callout
modified = True
stats["stub_flag"] += 1
additions.append("stub callout")
if not modified:
if not source_name:
stats["skip_no_source"] += 1
continue
if args.dry_run:
print(f" [[{page_name}]]: {', '.join(additions)}")
continue
if args.apply:
# Reconstruct file
if yaml:
new_yaml = yaml.dump(front, allow_unicode=True, default_flow_style=False, sort_keys=False)
else:
new_yaml = fm_text
fp.write_text(f"---\n{new_yaml}---\n{body}", encoding="utf-8")
mode = "DRY RUN" if args.dry_run else "APPLIED"
print(f"\n=== {mode} ===")
print(f" Citation seeds added: {stats['citation_seed']}")
print(f" Source sections added: {stats['source_section']}")
print(f" Stub flags added: {stats['stub_flag']}")
print(f" Skipped (has citation): {stats['skip_has_citation']}")
print(f" Skipped (no source): {stats['skip_no_source']}")
if __name__ == "__main__":
main()