292 lines
9.8 KiB
Python
Executable File
292 lines
9.8 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""
|
||
daily_report_ai.py — AI 驱动的微信读书每日阅读报告
|
||
|
||
调用 DeepSeek API 对当日划线+批注进行深度分析,
|
||
生成有洞察的阅读报告,推送到 Gitea。
|
||
|
||
定时:每天 22:00(配合 sync.py 同步)
|
||
|
||
依赖:pip install requests
|
||
"""
|
||
|
||
import os
|
||
import re
|
||
import sys
|
||
import json
|
||
import subprocess
|
||
from datetime import datetime
|
||
|
||
REPO_DIR = os.path.expanduser("~/weread-notes")
|
||
NOTES_DIR = os.path.join(REPO_DIR, "notes")
|
||
DAILY_DIR = os.path.join(REPO_DIR, "daily")
|
||
SYNC_SCRIPT = os.path.expanduser("~/weread-sync/sync.py")
|
||
API_KEY_FILE = os.path.expanduser("~/.keys/deepseek_api_key")
|
||
|
||
DEEPSEEK_URL = "https://api.deepseek.com/chat/completions"
|
||
DEEPSEEK_MODEL = "deepseek-chat"
|
||
|
||
|
||
# ─── 读取今日数据 ────────────────────────────────────
|
||
|
||
def parse_time_from_line(line):
|
||
m = re.search(r'🕐\s*(\d{4}-\d{2}-\d{2})', line)
|
||
return m.group(1) if m else None
|
||
|
||
|
||
def load_today_data(date_str=None):
|
||
"""读取当日所有划线+批注"""
|
||
if not date_str:
|
||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||
|
||
today_entries = []
|
||
|
||
for fname in os.listdir(NOTES_DIR):
|
||
if not fname.endswith(".md"):
|
||
continue
|
||
filepath = os.path.join(NOTES_DIR, fname)
|
||
book_title = fname[:-3]
|
||
with open(filepath, "r", encoding="utf-8") as f:
|
||
lines = f.readlines()
|
||
|
||
# 提取书籍元信息
|
||
meta = {"author": "", "progress": "", "category": ""}
|
||
for line in lines[:35]:
|
||
m = re.match(r'\|\s*\*\*作者\*\*\s*\|\s*(.+?)\s*\|', line)
|
||
if m: meta["author"] = m.group(1).strip()
|
||
m = re.match(r'\|\s*\*\*阅读进度\*\*\s*\|\s*(\d+)%\s*\|', line)
|
||
if m: meta["progress"] = f"{m.group(1)}%"
|
||
m = re.match(r'\|\s*\*\*分类\*\*\s*\|\s*(.+?)\s*\|', line)
|
||
if m: meta["category"] = m.group(1).strip()
|
||
|
||
# 解析时间戳行
|
||
for i, line in enumerate(lines):
|
||
d = parse_time_from_line(line)
|
||
if d != date_str:
|
||
continue
|
||
|
||
entry = {
|
||
"book": book_title,
|
||
"author": meta["author"],
|
||
"progress": meta["progress"],
|
||
"category": meta["category"],
|
||
"time": line.strip(),
|
||
}
|
||
|
||
# 上一行是原文
|
||
if i > 0:
|
||
prev = lines[i - 1].strip()
|
||
entry["text"] = re.sub(r'^>\s*', '', prev).strip()
|
||
|
||
# 后续行找批注
|
||
for j in range(i + 1, min(i + 5, len(lines))):
|
||
t = lines[j].strip()
|
||
if t.startswith("💬"):
|
||
entry["thought"] = re.sub(r'^💬\s*\*{0,2}(.*?)\*{0,2}$', r'\1', t).strip()
|
||
break
|
||
|
||
today_entries.append(entry)
|
||
|
||
return today_entries
|
||
|
||
|
||
def format_entries_for_prompt(entries):
|
||
"""将今日条目格式化为 AI 可读的文本"""
|
||
if not entries:
|
||
return "今日暂无阅读记录。"
|
||
|
||
# 按书名分组
|
||
from collections import defaultdict
|
||
by_book = defaultdict(list)
|
||
for e in entries:
|
||
by_book[e["book"]].append(e)
|
||
|
||
parts = []
|
||
for book_name in sorted(by_book.keys()):
|
||
items = by_book[book_name]
|
||
info = []
|
||
if items[0]["author"]:
|
||
info.append(f"作者:{items[0]['author']}")
|
||
if items[0]["category"]:
|
||
info.append(f"分类:{items[0]['category']}")
|
||
if items[0]["progress"]:
|
||
info.append(f"进度:{items[0]['progress']}")
|
||
|
||
parts.append(f"## 《{book_name}》")
|
||
if info:
|
||
parts.append("(" + " | ".join(info) + ")")
|
||
parts.append("")
|
||
|
||
for e in items:
|
||
parts.append(f"### 划线")
|
||
parts.append(f"{e['text']}")
|
||
parts.append(f"时间:{e['time']}")
|
||
if e.get("thought"):
|
||
parts.append(f"")
|
||
parts.append(f"我的批注:{e['thought']}")
|
||
parts.append("")
|
||
|
||
return "\n".join(parts)
|
||
|
||
|
||
# ─── AI 分析 ──────────────────────────────────────────
|
||
|
||
def call_deepseek(prompt, system_prompt):
|
||
"""调用 DeepSeek API"""
|
||
api_key_file = os.path.expanduser(API_KEY_FILE)
|
||
if os.path.exists(api_key_file):
|
||
with open(api_key_file) as f:
|
||
api_key = f.read().strip()
|
||
else:
|
||
print("[ERROR] DeepSeek API Key not found")
|
||
print(f" Please save your key to: {API_KEY_FILE}")
|
||
return None
|
||
|
||
import urllib.request
|
||
|
||
body = json.dumps({
|
||
"model": DEEPSEEK_MODEL,
|
||
"messages": [
|
||
{"role": "system", "content": system_prompt},
|
||
{"role": "user", "content": prompt},
|
||
],
|
||
"temperature": 0.7,
|
||
"max_tokens": 4096,
|
||
}).encode()
|
||
|
||
req = urllib.request.Request(
|
||
DEEPSEEK_URL,
|
||
data=body,
|
||
headers={
|
||
"Authorization": f"Bearer {api_key}",
|
||
"Content-Type": "application/json",
|
||
}
|
||
)
|
||
try:
|
||
resp = urllib.request.urlopen(req, timeout=120)
|
||
result = json.loads(resp.read())
|
||
return result["choices"][0]["message"]["content"]
|
||
except Exception as e:
|
||
print(f"[ERROR] DeepSeek API call failed: {e}")
|
||
return None
|
||
|
||
|
||
# ─── 报告生成 ────────────────────────────────────────
|
||
|
||
SYSTEM_PROMPT = """你是一位深度阅读分析助手,不是摘要工具。
|
||
|
||
你的任务是分析用户今日在微信读书上的划线和批注,生成一份高质量的阅读报告。
|
||
|
||
## 写作要求
|
||
- 每一条划线都要展开分析:补充背景知识、指出其深层含义、联系用户的实际场景
|
||
- 如果有批注,要围绕批注展开讨论,追问关键问题
|
||
- 找出当日阅读内容之间的内在联系(主题、方法、思想上的呼应)
|
||
- 像一位私人思考伙伴一样,提供有启发的见解和追问
|
||
- 报告要有深度、有观点,不要写成流水账或摘要
|
||
- 使用平实、客观的语言,不要抒情、不要煽情
|
||
- 适当引用原文,但不要大段堆砌
|
||
|
||
## 报告结构
|
||
1. 今日阅读概览(总览性描述当天阅读的特点)
|
||
2. 分书深度分析(每本书独立小节,有洞察的分析)
|
||
3. 主题串联(找出跨书的联系和启发)
|
||
4. 明日阅读建议(基于当天阅读给出具体建议)"""
|
||
|
||
|
||
def generate_report_md(entries, ai_analysis, date_str):
|
||
"""组装最终报告 Markdown"""
|
||
lines = []
|
||
lines.append(f"# 今日阅读报告 · {date_str}")
|
||
lines.append("")
|
||
lines.append("---")
|
||
lines.append("")
|
||
lines.append(ai_analysis)
|
||
lines.append("")
|
||
lines.append("---")
|
||
lines.append("")
|
||
lines.append(f"*报告生成:{datetime.now().strftime('%Y-%m-%d %H:%M')} | AI 模型:DeepSeek*")
|
||
return "\n".join(lines)
|
||
|
||
|
||
# ─── Git 操作 ────────────────────────────────────────
|
||
|
||
def save_and_push(content, date_str):
|
||
"""保存报告并推送到 Gitea"""
|
||
os.makedirs(DAILY_DIR, exist_ok=True)
|
||
filename = f"每日阅读_{date_str}.md"
|
||
filepath = os.path.join(DAILY_DIR, filename)
|
||
with open(filepath, "w", encoding="utf-8") as f:
|
||
f.write(content)
|
||
print(f"[OUTPUT] {filepath}")
|
||
|
||
try:
|
||
subprocess.run(["git", "add", "daily/"], cwd=REPO_DIR,
|
||
capture_output=True, timeout=30)
|
||
result = subprocess.run(
|
||
["git", "commit", "-m", f"report: 每日阅读报告 {date_str} (AI分析)", "--allow-empty"],
|
||
cwd=REPO_DIR, capture_output=True, timeout=30
|
||
)
|
||
if result.returncode == 0:
|
||
subprocess.run(["git", "push"], cwd=REPO_DIR,
|
||
capture_output=True, timeout=60)
|
||
print(f"[GIT] Pushed: 每日阅读报告 {date_str}")
|
||
else:
|
||
print(f"[GIT] No changes")
|
||
except subprocess.CalledProcessError as e:
|
||
print(f"[GIT] Error: {e.stderr.decode()[:200]}")
|
||
|
||
|
||
# ─── 主流程 ───────────────────────────────────────────
|
||
|
||
def main():
|
||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||
print(f"\n{'='*50}")
|
||
print(f"每日阅读报告 (AI) | {datetime.now().strftime('%Y-%m-%d %H:%M')}")
|
||
print(f"{'='*50}")
|
||
|
||
# 1. 同步
|
||
print("\n[1/4] 同步微信读书笔记...")
|
||
if os.path.exists(SYNC_SCRIPT):
|
||
subprocess.run(["python3", SYNC_SCRIPT], capture_output=True, timeout=300)
|
||
print(" 同步完成")
|
||
else:
|
||
print(f" [SKIP] {SYNC_SCRIPT} not found")
|
||
|
||
# 2. 解析今日数据
|
||
print(f"\n[2/4] 提取今日数据 ({date_str})...")
|
||
entries = load_today_data(date_str)
|
||
print(f" 找到 {len(entries)} 条记录")
|
||
|
||
if not entries:
|
||
md = f"# 今日阅读报告 · {date_str}\n\n> 暂无阅读记录\n"
|
||
save_and_push(md, date_str)
|
||
print("\n今日无阅读数据,已生成空报告")
|
||
return
|
||
|
||
# 3. AI 分析
|
||
print(f"\n[3/4] AI 深度分析 ({len(entries)} 条)...")
|
||
prompt = format_entries_for_prompt(entries)
|
||
full_prompt = f"""以下是用户今日在微信读书上的划线和批注数据:
|
||
|
||
{prompt}
|
||
|
||
请根据这些数据生成一份高质量的今日阅读报告。"""
|
||
|
||
ai_content = call_deepseek(full_prompt, SYSTEM_PROMPT)
|
||
if not ai_content:
|
||
print("[ERROR] AI 分析失败,使用备用模板")
|
||
ai_content = "> AI 分析暂时不可用,请稍后重试。"
|
||
|
||
# 4. 输出
|
||
print(f"\n[4/4] 生成报告并推送...")
|
||
md = generate_report_md(entries, ai_content, date_str)
|
||
save_and_push(md, date_str)
|
||
|
||
print(f"\n{'='*50}")
|
||
print(f"每日阅读报告完成 ✅")
|
||
print(f"{'='*50}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|