208 lines
6.9 KiB
Python
208 lines
6.9 KiB
Python
#!/usr/bin/env python
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
enrich-citations.py — Phase 3 P0: 批量为零引用页面补溯源链接
|
||
|
||
策略:
|
||
1. **来源节补全**:无 `## 来源` 节的页面,自动添加,包含指向 raw source 的 wikilink
|
||
2. **首段引用注入**:若 raw source 文件存在,提取页面标题关键词,
|
||
在 raw 文件中定位匹配行,注入首条 `[raw:source:行号]` 作为溯源种子
|
||
3. **存根页标记**:body < 50 字的页面添加 `> [!warning] 存根页面` callout
|
||
|
||
安全措施:
|
||
- 幂等:已有 `## 来源` 节或 `[raw:` 引用则跳过
|
||
- --dry-run / --apply
|
||
- 仅处理 wiki/*.md(排除 index/log/archive)
|
||
|
||
用法:
|
||
python tools/scripts/enrich-citations.py --dry-run
|
||
python tools/scripts/enrich-citations.py --apply
|
||
"""
|
||
import argparse
|
||
import re
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
try:
|
||
import yaml
|
||
except ImportError:
|
||
yaml = None
|
||
|
||
VAULT = Path(r"D:\Applications\app\kepano-obsidian-main")
|
||
WIKI = VAULT / "wiki"
|
||
RAW = VAULT / "raw"
|
||
EXCLUDE = {"index.md", "log.md"}
|
||
|
||
|
||
def read_frontmatter(fp: Path) -> tuple[dict, str, str]:
|
||
content = fp.read_text(encoding="utf-8")
|
||
m = re.match(r"^---\n(.*?)\n---\n?(.*)", content, re.DOTALL)
|
||
if not m:
|
||
return {}, content, ""
|
||
fm_text = m.group(1)
|
||
body = m.group(2)
|
||
if yaml:
|
||
try:
|
||
front = yaml.safe_load(fm_text) or {}
|
||
except Exception:
|
||
front = {}
|
||
else:
|
||
front = {}
|
||
return front, body, fm_text
|
||
|
||
|
||
def get_source_filename(front: dict) -> str:
|
||
"""Extract raw filename from source field"""
|
||
source = front.get("source", "")
|
||
if not source:
|
||
return ""
|
||
s = str(source).strip().strip('"').strip("'")
|
||
s = re.sub(r"\[\[(?:raw/)?(.+?)(?:\|.*)?\]\]", r"\1", s)
|
||
s = re.sub(r"\.md$", "", s)
|
||
return s
|
||
|
||
|
||
def find_raw_file(source_name: str) -> Path | None:
|
||
"""Try to locate the raw file"""
|
||
candidates = [
|
||
RAW / f"{source_name}.md",
|
||
RAW / source_name,
|
||
]
|
||
# Also search subdirectories
|
||
for pattern in [f"**/{source_name}.md", f"**/{source_name}/index.md"]:
|
||
matches = list(RAW.glob(pattern))
|
||
if matches:
|
||
candidates.append(matches[0])
|
||
for c in candidates:
|
||
if c.exists() and c.is_file():
|
||
return c
|
||
return None
|
||
|
||
|
||
def find_keyword_line(raw_path: Path, page_name: str) -> str:
|
||
"""Find a line in raw file matching the page name or key keyword"""
|
||
try:
|
||
raw_text = raw_path.read_text(encoding="utf-8")
|
||
except Exception:
|
||
return ""
|
||
lines = raw_text.splitlines()
|
||
|
||
# Strategy: search for page name, then aliases
|
||
keywords = [page_name]
|
||
# Also try shorter keywords
|
||
if len(page_name) > 4:
|
||
keywords.append(page_name[:4])
|
||
|
||
for kw in keywords:
|
||
for i, line in enumerate(lines, 1):
|
||
if kw in line and len(line.strip()) > 5:
|
||
return f"{i}"
|
||
return ""
|
||
|
||
|
||
def has_source_section(body: str) -> bool:
|
||
return bool(re.search(r"^##\s*来源", body, re.MULTILINE))
|
||
|
||
|
||
def has_raw_citations(body: str) -> bool:
|
||
return "[raw:" in body
|
||
|
||
|
||
def is_stub(body: str) -> bool:
|
||
clean = re.sub(r"!\[\[.*?\]\]", "", body)
|
||
clean = re.sub(r"```.*?```", "", clean, flags=re.DOTALL)
|
||
clean = re.sub(r"\|.*?\|", "", clean)
|
||
clean = re.sub(r"\s+", "", clean)
|
||
return len(clean) < 150
|
||
|
||
|
||
def main():
|
||
parser = argparse.ArgumentParser(description="Enrich wiki pages with source citations")
|
||
parser.add_argument("--dry-run", action="store_true", help="Preview without writing")
|
||
parser.add_argument("--apply", action="store_true", help="Write changes to files")
|
||
args = parser.parse_args()
|
||
|
||
if not args.dry_run and not args.apply:
|
||
args.dry_run = True
|
||
|
||
stats = {"source_section": 0, "citation_seed": 0, "stub_flag": 0, "skip_has_citation": 0, "skip_no_source": 0}
|
||
|
||
for fp in sorted(WIKI.glob("*.md")):
|
||
if fp.name in EXCLUDE:
|
||
continue
|
||
front, body, fm_text = read_frontmatter(fp)
|
||
page_name = fp.stem
|
||
|
||
if has_raw_citations(body):
|
||
stats["skip_has_citation"] += 1
|
||
continue
|
||
|
||
source_name = get_source_filename(front)
|
||
raw_path = find_raw_file(source_name) if source_name else None
|
||
|
||
modified = False
|
||
additions = []
|
||
|
||
# 1. Add citation seed if raw file exists
|
||
if raw_path:
|
||
line_num = find_keyword_line(raw_path, page_name)
|
||
if line_num:
|
||
# Find first paragraph in body and append citation
|
||
first_para_match = re.search(r"^(.+?)(?:\n\n|\n#)", body, re.DOTALL)
|
||
if first_para_match:
|
||
insert_point = first_para_match.end()
|
||
cite = f"[raw:{source_name}:{line_num}]"
|
||
if cite not in body:
|
||
body = body[:insert_point] + f" {cite}" + body[insert_point:]
|
||
modified = True
|
||
stats["citation_seed"] += 1
|
||
additions.append(f"citation seed [raw:{source_name}:{line_num}]")
|
||
|
||
# 2. Add/fix source section
|
||
if not has_source_section(body):
|
||
source_link = front.get("source", "")
|
||
if source_link:
|
||
source_section = f"\n\n## 来源\n\n> **溯源规则**:所有数字/百分比/具体结论必须标注 `[raw:{{文件名}}:{{行号}}]` 格式。\n\n- {source_link}"
|
||
body = body.rstrip() + source_section
|
||
modified = True
|
||
stats["source_section"] += 1
|
||
additions.append("source section")
|
||
|
||
# 3. Flag stub pages
|
||
if is_stub(body) and not args.dry_run:
|
||
if "> [!warning] 存根页面" not in body:
|
||
stub_callout = "\n\n> [!warning] 存根页面\n> 本页内容不足,待扩充。参考同类黄金标准页面(如 [[DSpark]]、[[何长工]])补充来源引用和正文内容。\n"
|
||
body = body.rstrip() + stub_callout
|
||
modified = True
|
||
stats["stub_flag"] += 1
|
||
additions.append("stub callout")
|
||
|
||
if not modified:
|
||
if not source_name:
|
||
stats["skip_no_source"] += 1
|
||
continue
|
||
|
||
if args.dry_run:
|
||
print(f" [[{page_name}]]: {', '.join(additions)}")
|
||
continue
|
||
|
||
if args.apply:
|
||
# Reconstruct file
|
||
if yaml:
|
||
new_yaml = yaml.dump(front, allow_unicode=True, default_flow_style=False, sort_keys=False)
|
||
else:
|
||
new_yaml = fm_text
|
||
fp.write_text(f"---\n{new_yaml}---\n{body}", encoding="utf-8")
|
||
|
||
mode = "DRY RUN" if args.dry_run else "APPLIED"
|
||
print(f"\n=== {mode} ===")
|
||
print(f" Citation seeds added: {stats['citation_seed']}")
|
||
print(f" Source sections added: {stats['source_section']}")
|
||
print(f" Stub flags added: {stats['stub_flag']}")
|
||
print(f" Skipped (has citation): {stats['skip_has_citation']}")
|
||
print(f" Skipped (no source): {stats['skip_no_source']}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|