chore(vault): backup 2026-07-03 21:40:41
This commit is contained in:
@@ -0,0 +1,199 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
真实 A 组实验(Two-Step 模式)- 使用真实 OpenAI API
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import requests
|
||||
from datetime import datetime
|
||||
import re
|
||||
|
||||
# 配置
|
||||
base_dir = os.getcwd()
|
||||
experiment_dir = os.path.join(base_dir, "tools", "experiments", "wiki-generation-compare")
|
||||
source_file = os.path.join(base_dir, "raw", "呼吸之间_李谨伯", "第一编 从身体入手.md")
|
||||
output_dir = os.path.join(experiment_dir, "output", "group-a-real")
|
||||
|
||||
# 创建输出目录
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
os.makedirs(os.path.join(output_dir, "wiki", "concepts"), exist_ok=True)
|
||||
os.makedirs(os.path.join(output_dir, "wiki", "methods"), exist_ok=True)
|
||||
os.makedirs(os.path.join(output_dir, "wiki", "entities"), exist_ok=True)
|
||||
|
||||
# 记录开始时间
|
||||
start_time = datetime.now()
|
||||
start_time_str = start_time.strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
print("=== A 组真实实验开始:Two-Step 模式 ===")
|
||||
print(f"开始时间: {start_time_str}")
|
||||
|
||||
# 检查 API Key
|
||||
api_key = os.environ.get('OPENAI_API_KEY')
|
||||
if not api_key:
|
||||
print("[ERROR] 未找到 OPENAI_API_KEY 环境变量")
|
||||
exit(1)
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
# Step 1: Analysis
|
||||
print("\nStep 1: Analysis 阶段...")
|
||||
|
||||
# 读取源文件
|
||||
with open(source_file, 'r', encoding='utf-8') as f:
|
||||
source_content = f.read()
|
||||
|
||||
# 读取分析 prompt
|
||||
analysis_prompt_file = os.path.join(experiment_dir, "prompts", "twostep-analysis.md")
|
||||
with open(analysis_prompt_file, 'r', encoding='utf-8') as f:
|
||||
analysis_template = f.read()
|
||||
|
||||
analysis_full_prompt = analysis_template.replace("{SOURCE_CONTENT}", source_content)
|
||||
|
||||
print(" 调用 OpenAI API 进行分析...")
|
||||
analysis_body = {
|
||||
"model": "gpt-4o-mini",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "你是一位知识库分析专家。请分析源文件并提取结构化信息,仅输出 JSON 格式。"
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": analysis_full_prompt
|
||||
}
|
||||
],
|
||||
"temperature": 0.3,
|
||||
"max_tokens": 8000
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post("https://api.openai.com/v1/chat/completions",
|
||||
headers=headers,
|
||||
json=analysis_body,
|
||||
timeout=180)
|
||||
response.raise_for_status()
|
||||
analysis_result = response.json()['choices'][0]['message']['content']
|
||||
|
||||
# 保存分析结果
|
||||
analysis_output = os.path.join(output_dir, "analysis.json")
|
||||
with open(analysis_output, 'w', encoding='utf-8') as f:
|
||||
f.write(analysis_result)
|
||||
|
||||
print(f" [OK] Analysis 完成")
|
||||
|
||||
# 解析 JSON
|
||||
try:
|
||||
gold_standard = json.loads(analysis_result)
|
||||
recommended_count = len(gold_standard.get('recommended_pages', []))
|
||||
print(f" [OK] 识别到 {recommended_count} 个推荐页面")
|
||||
except json.JSONDecodeError as e:
|
||||
print(f" [ERROR] JSON 解析失败: {e}")
|
||||
print(f" 分析结果预览: {analysis_result[:200]}")
|
||||
exit(1)
|
||||
|
||||
except Exception as e:
|
||||
print(f" [ERROR] LLM API 调用失败: {e}")
|
||||
print(" 退出实验")
|
||||
exit(1)
|
||||
|
||||
# Step 2: Generation
|
||||
print("\nStep 2: Generation 阶段...")
|
||||
|
||||
# 读取生成 prompt
|
||||
generation_prompt_file = os.path.join(experiment_dir, "prompts", "twostep-generation.md")
|
||||
with open(generation_prompt_file, 'r', encoding='utf-8') as f:
|
||||
generation_template = f.read()
|
||||
|
||||
generation_full_prompt = generation_template.replace("{ANALYSIS_RESULT}", json.dumps(gold_standard, ensure_ascii=False, indent=2))
|
||||
|
||||
print(" 调用 OpenAI API 生成 Wiki 页面...")
|
||||
generation_body = {
|
||||
"model": "gpt-4o-mini",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "你是一位知识库构建专家。请基于分析结果生成 Wiki 页面。"
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": generation_full_prompt
|
||||
}
|
||||
],
|
||||
"temperature": 0.5,
|
||||
"max_tokens": 16000
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post("https://api.openai.com/v1/chat/completions",
|
||||
headers=headers,
|
||||
json=generation_body,
|
||||
timeout=300)
|
||||
response.raise_for_status()
|
||||
generation_result = response.json()['choices'][0]['message']['content']
|
||||
|
||||
print(" [OK] LLM 生成完成")
|
||||
|
||||
except Exception as e:
|
||||
print(f" [ERROR] LLM API 调用失败: {e}")
|
||||
exit(1)
|
||||
|
||||
# 解析生成结果
|
||||
sample_pages = []
|
||||
|
||||
# 解析 ---FILE: ... ---END FILE--- 块
|
||||
file_blocks = re.findall(r'---FILE: (.*?)---(.*?)---END FILE---', generation_result, re.DOTALL)
|
||||
|
||||
if not file_blocks:
|
||||
# 尝试宽松匹配
|
||||
file_blocks = re.findall(r'FILE: (.*?)\n(.*?)(?=(FILE:|$))', generation_result, re.DOTALL)
|
||||
|
||||
for block in file_blocks:
|
||||
file_path = block[0].strip()
|
||||
file_content = block[1].strip()
|
||||
|
||||
# 创建目录(如果需要)
|
||||
full_path = os.path.join(output_dir, file_path)
|
||||
file_dir = os.path.dirname(full_path)
|
||||
os.makedirs(file_dir, exist_ok=True)
|
||||
|
||||
# 写入文件
|
||||
with open(full_path, 'w', encoding='utf-8') as f:
|
||||
f.write(file_content)
|
||||
sample_pages.append(full_path)
|
||||
print(f" [OK] 生成: {file_path}")
|
||||
|
||||
# 记录结束时间
|
||||
end_time = datetime.now()
|
||||
end_time_str = end_time.strftime("%Y-%m-%d %H:%M:%S")
|
||||
duration = (end_time - start_time).total_seconds() / 60
|
||||
|
||||
print(f"\n=== A 组实验完成 ===")
|
||||
print(f"结束时间: {end_time_str}")
|
||||
print(f"总耗时: {round(duration, 2)} 分钟")
|
||||
print(f"生成文件数: {len(sample_pages)}")
|
||||
|
||||
# 保存元数据
|
||||
metadata = {
|
||||
"group": "A-real",
|
||||
"mode": "Two-Step (Real API)",
|
||||
"sourceFile": "raw/呼吸之间_李谨伯/第一编 从身体入手.md",
|
||||
"startTime": start_time_str,
|
||||
"endTime": end_time_str,
|
||||
"durationMinutes": round(duration, 2),
|
||||
"step1Status": "completed",
|
||||
"step2Status": "completed",
|
||||
"pagesGenerated": len(sample_pages),
|
||||
"apiModel": "gpt-4o-mini",
|
||||
"apiCost": f"Analysis (8K tokens) + Generation (16K tokens) ≈ ${round(0.15 * 24 / 1000000, 4)}"
|
||||
}
|
||||
|
||||
metadata_file = os.path.join(output_dir, "metadata.json")
|
||||
with open(metadata_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(metadata, f, ensure_ascii=False, indent=2)
|
||||
|
||||
print(f"[OK] 元数据已保存到 {metadata_file}")
|
||||
Reference in New Issue
Block a user