Phase 0-2: Schema cleanup, typed relations, event-driven automation

- Phase 0: AGENTS.md cleanup (dedup quotes, renumber sections, merge qmd)
- Phase 1: typed relations (manage-relations.py, graph-search.py, check-staleness.py, detect-conflicts.py)
- Phase 2: frontmatter validator, weekly lint, knowledge promotion, git hooks
- Fix .gitignore to track tools/ and .githooks/
- Fix git remote URL (remove plaintext token)
- New wiki pages: 504 pages, 34 raw sources
This commit is contained in:
hehaiguang1123
2026-07-01 08:05:43 +08:00
parent e544d6e04a
commit a6f05ab2d5
1067 changed files with 522992 additions and 819 deletions
+206
View File
@@ -0,0 +1,206 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
fix-raw-citations.py — 为 wiki/ 页面中缺失行号的 [raw:filename] 引用
自动从原始 raw/ 文件查找匹配文本并补上行号范围。
用法:
python tools/scripts/fix-raw-citations.py # 补行号
python tools/scripts/fix-raw-citations.py --stats # 仅统计不修改
python tools/scripts/fix-raw-citations.py --dry-run # 预览修改
退出码:0 成功;1 有错误。
"""
import argparse
import re
import sys
from pathlib import Path
VAULT = Path(r"D:\Applications\app\kepano-obsidian-main")
WIKI = VAULT / "wiki"
RAW = VAULT / "raw"
def collect_wiki_files():
"""收集 wiki/ 中需要处理的 .md 文件(排除 index.md, log.md"""
files = []
for f in sorted(WIKI.glob("*.md")):
if f.name not in ("index.md", "log.md"):
files.append(f)
return files
def collect_aw_references(content: str):
"""提取所有 [raw:...] 引用,返回 (match_obj, raw_filename, has_line_number)"""
pattern = r'\[raw:([^:\]]+)(?::(\d+(?:-\d+)?))?\]'
refs = []
for m in re.finditer(pattern, content):
fname = m.group(1)
has_ln = m.group(2) is not None
refs.append((m, fname, has_ln))
return refs
def extract_keyword(content: str, pos: int, max_chars: int = 60) -> str:
"""
从 content 中 pos 位置向前提取关键词(中文/英文/数字)。
返回用于在 raw 文件中匹配的文本片段。
"""
start = max(0, pos - max_chars)
before = content[start:pos]
tokens = re.findall(r'[\u4e00-\u9fff\w]+', before)
# 取最后 3-6 个 token 作为关键词
return " ".join(tokens[-6:]) if len(tokens) >= 3 else " ".join(tokens)
def find_line_range(raw_text: str, keyword: str, context: int = 3) -> str | None:
"""
在 raw_text 中搜索 keyword,返回匹配行所在的行号范围。
格式: "start-end""line"(单行匹配)。
返回 None 表示未找到。
"""
if not keyword:
return None
lines = raw_text.splitlines()
matched_lines = set()
for i, line in enumerate(lines, 1):
if keyword in line:
matched_lines.add(i)
if not matched_lines:
# fallback: try individual tokens
tokens = keyword.split()
for token in tokens:
if len(token) < 2:
continue
for i, line in enumerate(lines, 1):
if token in line:
matched_lines.add(i)
if not matched_lines:
return None
start = max(1, min(matched_lines) - context)
end = min(len(lines), max(matched_lines) + context)
if start == end:
return str(start)
return f"{start}-{end}"
def process_file(filepath: Path, dry_run: bool = False) -> tuple[str, int, int]:
"""
处理单个文件。
返回: (修改后的内容, 补行号数, 总引用数)
"""
content = filepath.read_text(encoding="utf-8")
refs = collect_aw_references(content)
total = len(refs)
fixed = 0
if total == 0:
return content, 0, 0
raw_cache = {}
# 从后往前替换以保持 offsets
for m, fname, has_ln in reversed(refs):
if has_ln:
continue # 已有行号,跳过
# 加载 raw 文件
if fname not in raw_cache:
rpath = RAW / f"{fname}.md"
if rpath.exists():
raw_cache[fname] = rpath.read_text(encoding="utf-8")
else:
# 尝试模糊匹配(取文件名最后一段)
candidates = list(RAW.glob(f"*{fname}*.md"))
if candidates:
raw_cache[fname] = candidates[0].read_text(encoding="utf-8")
else:
raw_cache[fname] = None
raw_text = raw_cache.get(fname)
if raw_text is None:
continue
# 提取关键词
keyword = extract_keyword(content, m.start())
line_range = find_line_range(raw_text, keyword)
if line_range:
old = m.group(0)
new = f"[raw:{fname}:{line_range}]"
content = content[:m.start()] + new + content[m.end():]
fixed += 1
return content, fixed, total
def stats_only():
"""仅统计行号覆盖率"""
files = collect_wiki_files()
total_refs = 0
total_with_ln = 0
total_missing_ln = 0
per_file = []
for fp in files:
content = fp.read_text(encoding="utf-8")
refs = collect_aw_references(content)
total_refs += len(refs)
with_ln = sum(1 for _, _, has_ln in refs if has_ln)
missing = len(refs) - with_ln
total_with_ln += with_ln
total_missing_ln += missing
if missing > 0:
per_file.append((fp.name, missing, with_ln))
print(f"\n=== 行号标注覆盖率统计 ===")
print(f" 文件数: {len(files)}")
print(f" 总引用: {total_refs}")
print(f" 有行号: {total_with_ln} ({total_with_ln / total_refs * 100:.1f}%)")
print(f" 缺行号: {total_missing_ln} ({total_missing_ln / total_refs * 100:.1f}%)")
if per_file:
print(f"\n 缺行号的文件 (top 20):")
for name, miss, have in sorted(per_file, key=lambda x: -x[1])[:20]:
print(f" {name}: 缺 {miss} / 共 {miss + have}")
return total_missing_ln
def main():
parser = argparse.ArgumentParser(description="为 wiki/ 的 raw 引用补上行号")
parser.add_argument("--stats", action="store_true", help="仅统计不修改")
parser.add_argument("--dry-run", action="store_true", help="预览修改但不写入")
args = parser.parse_args()
if args.stats:
stats_only()
return
files = collect_wiki_files()
total_fixed = 0
total_refs = 0
changed_files = []
for fp in files:
content, fixed, refs = process_file(fp, dry_run=args.dry_run)
total_fixed += fixed
total_refs += refs
if fixed > 0:
changed_files.append((fp.name, fixed, refs))
if not args.dry_run:
fp.write_text(content, encoding="utf-8")
print(f"\n=== 处理结果 ===")
print(f" 处理文件: {len(files)}")
print(f" 总引用数: {total_refs}")
print(f" 补行号数: {total_fixed}")
if total_refs > 0:
print(f" 覆盖率: {(total_refs - total_fixed + total_fixed) / total_refs * 100:.1f}% → "
f"{total_refs / total_refs * 100:.1f}% (理论上限,行号仅补匹配到的)")
if changed_files:
print(f"\n 更新文件 ({len(changed_files)}):")
for name, fixed, refs in sorted(changed_files, key=lambda x: -x[1]):
print(f" {name}: +{fixed} 行号")
if args.dry_run and changed_files:
print(f"\n 以上为预览,未写入文件(--dry-run)")
print(f"\n Total: {total_fixed} citations updated")
if __name__ == "__main__":
main()