Files
server-ops/weread-sync/daily_report.py

297 lines
10 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""
daily_report.py — 微信读书每日/每周/每月阅读报告
从 weread-notes 笔记中提取当日/当周/当月的划线+批注,
生成结构化阅读报告,推送到 Gitea。
用法:
python3 daily_report.py # 今日报告(默认)
python3 daily_report.py --period week # 本周报告
python3 daily_report.py --period month # 本月报告
定时crontab
0 22 * * * cd ~/weread-sync && python3 daily_report.py >> daily_report.log 2>&1
"""
import os
import re
import sys
import json
import subprocess
from datetime import datetime, timedelta
from collections import defaultdict
REPO_DIR = os.path.expanduser("~/weread-notes")
NOTES_DIR = os.path.join(REPO_DIR, "notes")
DAILY_DIR = os.path.join(REPO_DIR, "daily")
SYNC_SCRIPT = os.path.expanduser("~/weread-sync/sync.py")
# ─── 时间解析 ────────────────────────────────────────
def parse_time_from_line(line):
"""'> 🕐 2026-06-15 09:10' 中提取日期对象"""
m = re.search(r'🕐\s*(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2})', line)
if m:
return datetime.strptime(f"{m.group(1)} {m.group(2)}", "%Y-%m-%d %H:%M")
return None
def get_period_range(period="day"):
"""获取时间范围(起始日期 00:00, 结束日期 23:59"""
now = datetime.now()
today = now.replace(hour=0, minute=0, second=0, microsecond=0)
if period == "day":
return today, today + timedelta(days=1)
elif period == "week":
start = today - timedelta(days=today.weekday())
return start, start + timedelta(days=7)
elif period == "month":
start = today.replace(day=1)
if start.month == 12:
end = start.replace(year=start.year + 1, month=1)
else:
end = start.replace(month=start.month + 1)
return start, end
return today, today + timedelta(days=1)
def period_label(period):
labels = {"day": "今日", "week": "本周", "month": "本月"}
return labels.get(period, "今日")
# ─── 笔记解析 ────────────────────────────────────────
def parse_notes():
"""解析所有笔记文件,返回结构化数据"""
entries = []
for fname in os.listdir(NOTES_DIR):
if not fname.endswith(".md"):
continue
filepath = os.path.join(NOTES_DIR, fname)
book_title = fname[:-3] # 去掉 .md
with open(filepath, "r", encoding="utf-8") as f:
lines = f.readlines()
# 提取书籍信息头部
author = ""
progress = ""
for line in lines[:30]:
m = re.match(r'\|\s*\*\*作者\*\*\s*\|\s*(.+?)\s*\|', line)
if m:
author = m.group(1).strip()
m = re.match(r'\|\s*\*\*阅读进度\*\*\s*\|\s*(\d+)%\s*\|', line)
if m:
progress = f"{m.group(1)}%"
# 解析划线 + 想法
i = 0
while i < len(lines):
line = lines[i]
# 找时间戳行
ts = parse_time_from_line(line)
if ts:
# 划线:时间戳上一行是内容(> 开头)
if i > 0:
prev = lines[i - 1].strip()
content_text = re.sub(r'^>\s*', '', prev).strip()
# 找紧跟的 💬 想法
thought = ""
for j in range(i + 1, min(i + 5, len(lines))):
t_line = lines[j].strip()
if t_line.startswith("💬"):
thought = re.sub(r'^💬\s*\*{0,2}(.*?)\*{0,2}$', r'\1', t_line).strip()
break
entries.append({
"book": book_title,
"author": author,
"progress": progress,
"time": ts,
"type": "bookmark",
"text": content_text,
"thought": thought,
})
i += 1
return entries
# ─── 报告生成 ────────────────────────────────────────
def generate_report(entries, period="day"):
"""生成阅读报告 Markdown"""
now = datetime.now()
start, end = get_period_range(period)
label = period_label(period)
date_str = now.strftime("%Y-%m-%d")
# 筛选时间范围内的条目
filtered = [e for e in entries if start <= e["time"] < end]
if not filtered:
return f"# {label}阅读报告 · {date_str}\n\n> 暂无阅读记录\n"
# 按书籍分组
by_book = defaultdict(list)
for e in filtered:
by_book[e["book"]].append(e)
# 按时间排序
filtered.sort(key=lambda x: x["time"], reverse=True)
lines = []
lines.append(f"# {label}阅读报告 · {date_str}")
lines.append("")
# ── 概览 ──
total_bookmarks = sum(1 for e in filtered if e["type"] == "bookmark" and not e["thought"])
total_thoughts = sum(1 for e in filtered if e["thought"])
finished_books = set(e["book"] for e in filtered)
lines.append("## 📊 阅读概览")
lines.append("")
lines.append(f"| 项目 | 数量 |")
lines.append(f"|:---|:---:|")
lines.append(f"| 阅读书籍 | {len(finished_books)} 本 |")
lines.append(f"| 新增划线 | {total_bookmarks} 条 |")
lines.append(f"| 新增批注 | {total_thoughts} 条 |")
# 各书进度
lines.append("")
progress_lines = []
for book_name in sorted(by_book.keys()):
entries_list = by_book[book_name]
p = entries_list[0]["progress"] if entries_list[0]["progress"] else ""
bm_count = sum(1 for e in entries_list if e["type"] == "bookmark" and not e["thought"])
th_count = sum(1 for e in entries_list if e["thought"])
progress_lines.append(f"- **{book_name}**{p})— {bm_count}条划线, {th_count}条批注")
if progress_lines:
lines.append("\n".join(progress_lines))
lines.append("")
# ── 划线精选 ──
bookmarks_only = [e for e in filtered if e["type"] == "bookmark"]
if bookmarks_only:
lines.append("---")
lines.append("## 📖 划线精选")
lines.append("")
for book_name in sorted(by_book.keys()):
bm_list = [e for e in by_book[book_name] if e["type"] == "bookmark"]
if not bm_list:
continue
lines.append(f"### 《{book_name}")
lines.append("")
for e in bm_list:
lines.append(f"> {e['text']}")
lines.append(f"> 🕐 {e['time'].strftime('%Y-%m-%d %H:%M')}")
lines.append("")
lines.append("")
# ── 批注精选 ──
thoughts = [e for e in filtered if e["thought"]]
if thoughts:
lines.append("---")
lines.append("## 💬 批注精选")
lines.append("")
for book_name in sorted(by_book.keys()):
th_list = [e for e in by_book[book_name] if e["thought"]]
if not th_list:
continue
lines.append(f"### 《{book_name}")
lines.append("")
for e in th_list:
if e["text"]:
lines.append(f"> {e['text']}")
lines.append(f"> 🕐 {e['time'].strftime('%Y-%m-%d %H:%M')}")
lines.append("")
lines.append(f"💬 **{e['thought']}**")
lines.append("")
if e["text"]:
lines.append("---")
lines.append("")
return "\n".join(lines)
# ─── 输出 ────────────────────────────────────────────
def save_report(md_content, period="day"):
"""保存报告文件并推送到 Gitea"""
os.makedirs(DAILY_DIR, exist_ok=True)
now = datetime.now()
date_str = now.strftime("%Y-%m-%d")
period_map = {"day": "每日", "week": "每周", "month": "每月"}
period_cn = period_map.get(period, "每日")
filename = f"{period_cn}阅读_{date_str}.md"
filepath = os.path.join(DAILY_DIR, filename)
with open(filepath, "w", encoding="utf-8") as f:
f.write(md_content)
print(f"[OUTPUT] {filepath}")
# Git 推送
try:
subprocess.run(["git", "add", "daily/"], cwd=REPO_DIR,
capture_output=True, timeout=30)
result = subprocess.run(
["git", "commit", "-m", f"report: {period_cn}阅读 {date_str}", "--allow-empty"],
cwd=REPO_DIR, capture_output=True, timeout=30
)
if result.returncode == 0:
subprocess.run(["git", "push"], cwd=REPO_DIR,
capture_output=True, timeout=60)
print(f"[GIT] Pushed: {period_cn}阅读报告 {date_str}")
else:
print(f"[GIT] No changes (already up to date)")
except subprocess.CalledProcessError as e:
print(f"[GIT] Error: {e.stderr.decode()[:200]}")
# ─── 主流程 ───────────────────────────────────────────
def main():
# 解析参数
period = "day"
if "--period" in sys.argv:
idx = sys.argv.index("--period")
if idx + 1 < len(sys.argv):
period = sys.argv[idx + 1]
label = period_label(period)
now = datetime.now()
print(f"\n{'='*50}")
print(f"{label}阅读报告 | {now.strftime('%Y-%m-%d %H:%M')}")
print(f"{'='*50}")
# 先同步最新笔记
print("\n[1/4] 同步微信读书笔记...")
if os.path.exists(SYNC_SCRIPT):
result = subprocess.run(["python3", SYNC_SCRIPT],
capture_output=True, timeout=300)
print(result.stdout.decode()[-200:] if result.stdout else "")
else:
print(f" [SKIP] Sync script not found: {SYNC_SCRIPT}")
# 解析笔记
print(f"\n[2/4] 解析笔记文件...")
entries = parse_notes()
print(f" 共解析 {len(entries)} 条记录")
# 生成报告
print(f"\n[3/4] 生成报告...")
md = generate_report(entries, period)
# 保存 + 推送
print(f"\n[4/4] 推送到 Gitea...")
save_report(md, period)
print(f"\n{'='*50}")
print(f"{label}阅读报告完成 ✅")
print(f"{'='*50}")
if __name__ == "__main__":
main()