feat: AI-powered daily reading report\n\n- Calls DeepSeek API for deep analysis of daily highlights+thoughts\n- Detailed per-book analysis + cross-book theme connection\n- Cron: 22:00 daily (auto sync + analysis + push)\n- Sync frequency: every hour (was every 4h)\n- API key: ~/.keys/deepseek_api_key
This commit is contained in:
291
weread-sync/daily_report_ai.py
Executable file
291
weread-sync/daily_report_ai.py
Executable file
@ -0,0 +1,291 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
daily_report_ai.py — AI 驱动的微信读书每日阅读报告
|
||||
|
||||
调用 DeepSeek API 对当日划线+批注进行深度分析,
|
||||
生成有洞察的阅读报告,推送到 Gitea。
|
||||
|
||||
定时:每天 22:00(配合 sync.py 同步)
|
||||
|
||||
依赖:pip install requests
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import json
|
||||
import subprocess
|
||||
from datetime import datetime
|
||||
|
||||
REPO_DIR = os.path.expanduser("~/weread-notes")
|
||||
NOTES_DIR = os.path.join(REPO_DIR, "notes")
|
||||
DAILY_DIR = os.path.join(REPO_DIR, "daily")
|
||||
SYNC_SCRIPT = os.path.expanduser("~/weread-sync/sync.py")
|
||||
API_KEY_FILE = os.path.expanduser("~/.keys/deepseek_api_key")
|
||||
|
||||
DEEPSEEK_URL = "https://api.deepseek.com/chat/completions"
|
||||
DEEPSEEK_MODEL = "deepseek-chat"
|
||||
|
||||
|
||||
# ─── 读取今日数据 ────────────────────────────────────
|
||||
|
||||
def parse_time_from_line(line):
|
||||
m = re.search(r'🕐\s*(\d{4}-\d{2}-\d{2})', line)
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def load_today_data(date_str=None):
|
||||
"""读取当日所有划线+批注"""
|
||||
if not date_str:
|
||||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||||
|
||||
today_entries = []
|
||||
|
||||
for fname in os.listdir(NOTES_DIR):
|
||||
if not fname.endswith(".md"):
|
||||
continue
|
||||
filepath = os.path.join(NOTES_DIR, fname)
|
||||
book_title = fname[:-3]
|
||||
with open(filepath, "r", encoding="utf-8") as f:
|
||||
lines = f.readlines()
|
||||
|
||||
# 提取书籍元信息
|
||||
meta = {"author": "", "progress": "", "category": ""}
|
||||
for line in lines[:35]:
|
||||
m = re.match(r'\|\s*\*\*作者\*\*\s*\|\s*(.+?)\s*\|', line)
|
||||
if m: meta["author"] = m.group(1).strip()
|
||||
m = re.match(r'\|\s*\*\*阅读进度\*\*\s*\|\s*(\d+)%\s*\|', line)
|
||||
if m: meta["progress"] = f"{m.group(1)}%"
|
||||
m = re.match(r'\|\s*\*\*分类\*\*\s*\|\s*(.+?)\s*\|', line)
|
||||
if m: meta["category"] = m.group(1).strip()
|
||||
|
||||
# 解析时间戳行
|
||||
for i, line in enumerate(lines):
|
||||
d = parse_time_from_line(line)
|
||||
if d != date_str:
|
||||
continue
|
||||
|
||||
entry = {
|
||||
"book": book_title,
|
||||
"author": meta["author"],
|
||||
"progress": meta["progress"],
|
||||
"category": meta["category"],
|
||||
"time": line.strip(),
|
||||
}
|
||||
|
||||
# 上一行是原文
|
||||
if i > 0:
|
||||
prev = lines[i - 1].strip()
|
||||
entry["text"] = re.sub(r'^>\s*', '', prev).strip()
|
||||
|
||||
# 后续行找批注
|
||||
for j in range(i + 1, min(i + 5, len(lines))):
|
||||
t = lines[j].strip()
|
||||
if t.startswith("💬"):
|
||||
entry["thought"] = re.sub(r'^💬\s*\*{0,2}(.*?)\*{0,2}$', r'\1', t).strip()
|
||||
break
|
||||
|
||||
today_entries.append(entry)
|
||||
|
||||
return today_entries
|
||||
|
||||
|
||||
def format_entries_for_prompt(entries):
|
||||
"""将今日条目格式化为 AI 可读的文本"""
|
||||
if not entries:
|
||||
return "今日暂无阅读记录。"
|
||||
|
||||
# 按书名分组
|
||||
from collections import defaultdict
|
||||
by_book = defaultdict(list)
|
||||
for e in entries:
|
||||
by_book[e["book"]].append(e)
|
||||
|
||||
parts = []
|
||||
for book_name in sorted(by_book.keys()):
|
||||
items = by_book[book_name]
|
||||
info = []
|
||||
if items[0]["author"]:
|
||||
info.append(f"作者:{items[0]['author']}")
|
||||
if items[0]["category"]:
|
||||
info.append(f"分类:{items[0]['category']}")
|
||||
if items[0]["progress"]:
|
||||
info.append(f"进度:{items[0]['progress']}")
|
||||
|
||||
parts.append(f"## 《{book_name}》")
|
||||
if info:
|
||||
parts.append("(" + " | ".join(info) + ")")
|
||||
parts.append("")
|
||||
|
||||
for e in items:
|
||||
parts.append(f"### 划线")
|
||||
parts.append(f"{e['text']}")
|
||||
parts.append(f"时间:{e['time']}")
|
||||
if e.get("thought"):
|
||||
parts.append(f"")
|
||||
parts.append(f"我的批注:{e['thought']}")
|
||||
parts.append("")
|
||||
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
# ─── AI 分析 ──────────────────────────────────────────
|
||||
|
||||
def call_deepseek(prompt, system_prompt):
|
||||
"""调用 DeepSeek API"""
|
||||
api_key_file = os.path.expanduser(API_KEY_FILE)
|
||||
if os.path.exists(api_key_file):
|
||||
with open(api_key_file) as f:
|
||||
api_key = f.read().strip()
|
||||
else:
|
||||
print("[ERROR] DeepSeek API Key not found")
|
||||
print(f" Please save your key to: {API_KEY_FILE}")
|
||||
return None
|
||||
|
||||
import urllib.request
|
||||
|
||||
body = json.dumps({
|
||||
"model": DEEPSEEK_MODEL,
|
||||
"messages": [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
"temperature": 0.7,
|
||||
"max_tokens": 4096,
|
||||
}).encode()
|
||||
|
||||
req = urllib.request.Request(
|
||||
DEEPSEEK_URL,
|
||||
data=body,
|
||||
headers={
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
)
|
||||
try:
|
||||
resp = urllib.request.urlopen(req, timeout=120)
|
||||
result = json.loads(resp.read())
|
||||
return result["choices"][0]["message"]["content"]
|
||||
except Exception as e:
|
||||
print(f"[ERROR] DeepSeek API call failed: {e}")
|
||||
return None
|
||||
|
||||
|
||||
# ─── 报告生成 ────────────────────────────────────────
|
||||
|
||||
SYSTEM_PROMPT = """你是一位深度阅读分析助手,不是摘要工具。
|
||||
|
||||
你的任务是分析用户今日在微信读书上的划线和批注,生成一份高质量的阅读报告。
|
||||
|
||||
## 写作要求
|
||||
- 每一条划线都要展开分析:补充背景知识、指出其深层含义、联系用户的实际场景
|
||||
- 如果有批注,要围绕批注展开讨论,追问关键问题
|
||||
- 找出当日阅读内容之间的内在联系(主题、方法、思想上的呼应)
|
||||
- 像一位私人思考伙伴一样,提供有启发的见解和追问
|
||||
- 报告要有深度、有观点,不要写成流水账或摘要
|
||||
- 使用平实、客观的语言,不要抒情、不要煽情
|
||||
- 适当引用原文,但不要大段堆砌
|
||||
|
||||
## 报告结构
|
||||
1. 今日阅读概览(总览性描述当天阅读的特点)
|
||||
2. 分书深度分析(每本书独立小节,有洞察的分析)
|
||||
3. 主题串联(找出跨书的联系和启发)
|
||||
4. 明日阅读建议(基于当天阅读给出具体建议)"""
|
||||
|
||||
|
||||
def generate_report_md(entries, ai_analysis, date_str):
|
||||
"""组装最终报告 Markdown"""
|
||||
lines = []
|
||||
lines.append(f"# 今日阅读报告 · {date_str}")
|
||||
lines.append("")
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
lines.append(ai_analysis)
|
||||
lines.append("")
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
lines.append(f"*报告生成:{datetime.now().strftime('%Y-%m-%d %H:%M')} | AI 模型:DeepSeek*")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ─── Git 操作 ────────────────────────────────────────
|
||||
|
||||
def save_and_push(content, date_str):
|
||||
"""保存报告并推送到 Gitea"""
|
||||
os.makedirs(DAILY_DIR, exist_ok=True)
|
||||
filename = f"每日阅读_{date_str}.md"
|
||||
filepath = os.path.join(DAILY_DIR, filename)
|
||||
with open(filepath, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
print(f"[OUTPUT] {filepath}")
|
||||
|
||||
try:
|
||||
subprocess.run(["git", "add", "daily/"], cwd=REPO_DIR,
|
||||
capture_output=True, timeout=30)
|
||||
result = subprocess.run(
|
||||
["git", "commit", "-m", f"report: 每日阅读报告 {date_str} (AI分析)", "--allow-empty"],
|
||||
cwd=REPO_DIR, capture_output=True, timeout=30
|
||||
)
|
||||
if result.returncode == 0:
|
||||
subprocess.run(["git", "push"], cwd=REPO_DIR,
|
||||
capture_output=True, timeout=60)
|
||||
print(f"[GIT] Pushed: 每日阅读报告 {date_str}")
|
||||
else:
|
||||
print(f"[GIT] No changes")
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"[GIT] Error: {e.stderr.decode()[:200]}")
|
||||
|
||||
|
||||
# ─── 主流程 ───────────────────────────────────────────
|
||||
|
||||
def main():
|
||||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||||
print(f"\n{'='*50}")
|
||||
print(f"每日阅读报告 (AI) | {datetime.now().strftime('%Y-%m-%d %H:%M')}")
|
||||
print(f"{'='*50}")
|
||||
|
||||
# 1. 同步
|
||||
print("\n[1/4] 同步微信读书笔记...")
|
||||
if os.path.exists(SYNC_SCRIPT):
|
||||
subprocess.run(["python3", SYNC_SCRIPT], capture_output=True, timeout=300)
|
||||
print(" 同步完成")
|
||||
else:
|
||||
print(f" [SKIP] {SYNC_SCRIPT} not found")
|
||||
|
||||
# 2. 解析今日数据
|
||||
print(f"\n[2/4] 提取今日数据 ({date_str})...")
|
||||
entries = load_today_data(date_str)
|
||||
print(f" 找到 {len(entries)} 条记录")
|
||||
|
||||
if not entries:
|
||||
md = f"# 今日阅读报告 · {date_str}\n\n> 暂无阅读记录\n"
|
||||
save_and_push(md, date_str)
|
||||
print("\n今日无阅读数据,已生成空报告")
|
||||
return
|
||||
|
||||
# 3. AI 分析
|
||||
print(f"\n[3/4] AI 深度分析 ({len(entries)} 条)...")
|
||||
prompt = format_entries_for_prompt(entries)
|
||||
full_prompt = f"""以下是用户今日在微信读书上的划线和批注数据:
|
||||
|
||||
{prompt}
|
||||
|
||||
请根据这些数据生成一份高质量的今日阅读报告。"""
|
||||
|
||||
ai_content = call_deepseek(full_prompt, SYSTEM_PROMPT)
|
||||
if not ai_content:
|
||||
print("[ERROR] AI 分析失败,使用备用模板")
|
||||
ai_content = "> AI 分析暂时不可用,请稍后重试。"
|
||||
|
||||
# 4. 输出
|
||||
print(f"\n[4/4] 生成报告并推送...")
|
||||
md = generate_report_md(entries, ai_content, date_str)
|
||||
save_and_push(md, date_str)
|
||||
|
||||
print(f"\n{'='*50}")
|
||||
print(f"每日阅读报告完成 ✅")
|
||||
print(f"{'='*50}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user