#!/usr/bin/env python3 """ daily_report.py — 微信读书每日/每周/每月阅读报告 从 weread-notes 笔记中提取当日/当周/当月的划线+批注, 生成结构化阅读报告,推送到 Gitea。 用法: python3 daily_report.py # 今日报告(默认) python3 daily_report.py --period week # 本周报告 python3 daily_report.py --period month # 本月报告 定时(crontab): 0 22 * * * cd ~/weread-sync && python3 daily_report.py >> daily_report.log 2>&1 """ import os import re import sys import json import subprocess from datetime import datetime, timedelta from collections import defaultdict REPO_DIR = os.path.expanduser("~/weread-notes") NOTES_DIR = os.path.join(REPO_DIR, "notes") DAILY_DIR = os.path.join(REPO_DIR, "daily") SYNC_SCRIPT = os.path.expanduser("~/weread-sync/sync.py") # ─── 时间解析 ──────────────────────────────────────── def parse_time_from_line(line): """从 '> 🕐 2026-06-15 09:10' 中提取日期对象""" m = re.search(r'🕐\s*(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2})', line) if m: return datetime.strptime(f"{m.group(1)} {m.group(2)}", "%Y-%m-%d %H:%M") return None def get_period_range(period="day"): """获取时间范围(起始日期 00:00, 结束日期 23:59)""" now = datetime.now() today = now.replace(hour=0, minute=0, second=0, microsecond=0) if period == "day": return today, today + timedelta(days=1) elif period == "week": start = today - timedelta(days=today.weekday()) return start, start + timedelta(days=7) elif period == "month": start = today.replace(day=1) if start.month == 12: end = start.replace(year=start.year + 1, month=1) else: end = start.replace(month=start.month + 1) return start, end return today, today + timedelta(days=1) def period_label(period): labels = {"day": "今日", "week": "本周", "month": "本月"} return labels.get(period, "今日") # ─── 笔记解析 ──────────────────────────────────────── def parse_notes(): """解析所有笔记文件,返回结构化数据""" entries = [] for fname in os.listdir(NOTES_DIR): if not fname.endswith(".md"): continue filepath = os.path.join(NOTES_DIR, fname) book_title = fname[:-3] # 去掉 .md with open(filepath, "r", encoding="utf-8") as f: lines = f.readlines() # 提取书籍信息头部 author = "" progress = "" for line in lines[:30]: m = re.match(r'\|\s*\*\*作者\*\*\s*\|\s*(.+?)\s*\|', line) if m: author = m.group(1).strip() m = re.match(r'\|\s*\*\*阅读进度\*\*\s*\|\s*(\d+)%\s*\|', line) if m: progress = f"{m.group(1)}%" # 解析划线 + 想法 i = 0 while i < len(lines): line = lines[i] # 找时间戳行 ts = parse_time_from_line(line) if ts: # 划线:时间戳上一行是内容(> 开头) if i > 0: prev = lines[i - 1].strip() content_text = re.sub(r'^>\s*', '', prev).strip() # 找紧跟的 💬 想法 thought = "" for j in range(i + 1, min(i + 5, len(lines))): t_line = lines[j].strip() if t_line.startswith("💬"): thought = re.sub(r'^💬\s*\*{0,2}(.*?)\*{0,2}$', r'\1', t_line).strip() break entries.append({ "book": book_title, "author": author, "progress": progress, "time": ts, "type": "bookmark", "text": content_text, "thought": thought, }) i += 1 return entries # ─── 报告生成 ──────────────────────────────────────── def generate_report(entries, period="day"): """生成阅读报告 Markdown""" now = datetime.now() start, end = get_period_range(period) label = period_label(period) date_str = now.strftime("%Y-%m-%d") # 筛选时间范围内的条目 filtered = [e for e in entries if start <= e["time"] < end] if not filtered: return f"# {label}阅读报告 · {date_str}\n\n> 暂无阅读记录\n" # 按书籍分组 by_book = defaultdict(list) for e in filtered: by_book[e["book"]].append(e) # 按时间排序 filtered.sort(key=lambda x: x["time"], reverse=True) lines = [] lines.append(f"# {label}阅读报告 · {date_str}") lines.append("") # ── 概览 ── total_bookmarks = sum(1 for e in filtered if e["type"] == "bookmark" and not e["thought"]) total_thoughts = sum(1 for e in filtered if e["thought"]) finished_books = set(e["book"] for e in filtered) lines.append("## 📊 阅读概览") lines.append("") lines.append(f"| 项目 | 数量 |") lines.append(f"|:---|:---:|") lines.append(f"| 阅读书籍 | {len(finished_books)} 本 |") lines.append(f"| 新增划线 | {total_bookmarks} 条 |") lines.append(f"| 新增批注 | {total_thoughts} 条 |") # 各书进度 lines.append("") progress_lines = [] for book_name in sorted(by_book.keys()): entries_list = by_book[book_name] p = entries_list[0]["progress"] if entries_list[0]["progress"] else "—" bm_count = sum(1 for e in entries_list if e["type"] == "bookmark" and not e["thought"]) th_count = sum(1 for e in entries_list if e["thought"]) progress_lines.append(f"- **{book_name}**({p})— {bm_count}条划线, {th_count}条批注") if progress_lines: lines.append("\n".join(progress_lines)) lines.append("") # ── 划线精选 ── bookmarks_only = [e for e in filtered if e["type"] == "bookmark"] if bookmarks_only: lines.append("---") lines.append("## 📖 划线精选") lines.append("") for book_name in sorted(by_book.keys()): bm_list = [e for e in by_book[book_name] if e["type"] == "bookmark"] if not bm_list: continue lines.append(f"### 《{book_name}》") lines.append("") for e in bm_list: lines.append(f"> {e['text']}") lines.append(f"> 🕐 {e['time'].strftime('%Y-%m-%d %H:%M')}") lines.append("") lines.append("") # ── 批注精选 ── thoughts = [e for e in filtered if e["thought"]] if thoughts: lines.append("---") lines.append("## 💬 批注精选") lines.append("") for book_name in sorted(by_book.keys()): th_list = [e for e in by_book[book_name] if e["thought"]] if not th_list: continue lines.append(f"### 《{book_name}》") lines.append("") for e in th_list: if e["text"]: lines.append(f"> {e['text']}") lines.append(f"> 🕐 {e['time'].strftime('%Y-%m-%d %H:%M')}") lines.append("") lines.append(f"💬 **{e['thought']}**") lines.append("") if e["text"]: lines.append("---") lines.append("") return "\n".join(lines) # ─── 输出 ──────────────────────────────────────────── def save_report(md_content, period="day"): """保存报告文件并推送到 Gitea""" os.makedirs(DAILY_DIR, exist_ok=True) now = datetime.now() date_str = now.strftime("%Y-%m-%d") period_map = {"day": "每日", "week": "每周", "month": "每月"} period_cn = period_map.get(period, "每日") filename = f"{period_cn}阅读_{date_str}.md" filepath = os.path.join(DAILY_DIR, filename) with open(filepath, "w", encoding="utf-8") as f: f.write(md_content) print(f"[OUTPUT] {filepath}") # Git 推送 try: subprocess.run(["git", "add", "daily/"], cwd=REPO_DIR, capture_output=True, timeout=30) result = subprocess.run( ["git", "commit", "-m", f"report: {period_cn}阅读 {date_str}", "--allow-empty"], cwd=REPO_DIR, capture_output=True, timeout=30 ) if result.returncode == 0: subprocess.run(["git", "push"], cwd=REPO_DIR, capture_output=True, timeout=60) print(f"[GIT] Pushed: {period_cn}阅读报告 {date_str}") else: print(f"[GIT] No changes (already up to date)") except subprocess.CalledProcessError as e: print(f"[GIT] Error: {e.stderr.decode()[:200]}") # ─── 主流程 ─────────────────────────────────────────── def main(): # 解析参数 period = "day" if "--period" in sys.argv: idx = sys.argv.index("--period") if idx + 1 < len(sys.argv): period = sys.argv[idx + 1] label = period_label(period) now = datetime.now() print(f"\n{'='*50}") print(f"{label}阅读报告 | {now.strftime('%Y-%m-%d %H:%M')}") print(f"{'='*50}") # 先同步最新笔记 print("\n[1/4] 同步微信读书笔记...") if os.path.exists(SYNC_SCRIPT): result = subprocess.run(["python3", SYNC_SCRIPT], capture_output=True, timeout=300) print(result.stdout.decode()[-200:] if result.stdout else "") else: print(f" [SKIP] Sync script not found: {SYNC_SCRIPT}") # 解析笔记 print(f"\n[2/4] 解析笔记文件...") entries = parse_notes() print(f" 共解析 {len(entries)} 条记录") # 生成报告 print(f"\n[3/4] 生成报告...") md = generate_report(entries, period) # 保存 + 推送 print(f"\n[4/4] 推送到 Gitea...") save_report(md, period) print(f"\n{'='*50}") print(f"{label}阅读报告完成 ✅") print(f"{'='*50}") if __name__ == "__main__": main()