Files
inspiration-collector/analyzers/daily_digest.py

379 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Daily digest analyzer - fetches today's memos and generates AI analysis."""
import json
import logging
import os
import sys
import subprocess
from collections import Counter
from datetime import datetime, timezone, timedelta
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from tools.config import load_secrets, get_output_dir
from tools.llm import DeepSeekClient
from tools.memos_client import MemosClient
from tools.formatter import format_daily_digest
logger = logging.getLogger(__name__)
TZ_BEIJING = timezone(timedelta(hours=8))
PROJECT_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def utc_to_beijing(ts_str):
"""Convert UTC timestamp string (ISO 8601 with Z) to Beijing time string."""
try:
dt = datetime.fromisoformat(ts_str.replace("Z", "+00:00"))
dt_bj = dt.astimezone(TZ_BEIJING)
return dt_bj.strftime("%Y-%m-%d %H:%M")
except (ValueError, AttributeError):
return ts_str[:16].replace("T", " ") + " (UTC?)"
def read_previous_digest(date):
"""Read the most recent previous daily digest (if exists) for continuity."""
daily_root = get_output_dir("daily")
for days_back in range(1, 8):
prev_date = date - timedelta(days=days_back)
prev_dir = os.path.join(daily_root, prev_date.strftime("%Y-%m-%d"))
if not os.path.isdir(prev_dir):
continue
files = sorted([f for f in os.listdir(prev_dir) if f.endswith(".md")], reverse=True)
if not files:
continue
prev_file = os.path.join(prev_dir, files[0])
with open(prev_file, "r", encoding="utf-8") as f:
content = f.read()
ai_section_start = content.find("## AI 分析")
if ai_section_start != -1:
annotation_start = content.find("## 我的批注", ai_section_start)
if annotation_start != -1:
ai_section = content[ai_section_start:annotation_start].strip()
else:
ai_section = content[ai_section_start:].strip()
logger.info(
"Found previous digest from %s (%d chars)",
prev_date.strftime("%Y-%m-%d"), len(ai_section)
)
return prev_date, ai_section
else:
logger.info(
"Found previous digest from %s (no AI section found, using full)",
prev_date.strftime("%Y-%m-%d")
)
return prev_date, content
logger.info("No previous digest found in the last 7 days")
return None, None
# ============================================================
# System prompt for daily analysis (v2 - tag-aware)
# ============================================================
DAILY_SYSTEM_PROMPT = """你是一个私人思考伙伴。你的任务是认真阅读用户今天的每一条灵感记录,结合之前的分析,写一篇有深度的分析文章。
注意:所有时间戳都是**北京时间**UTC+8不是 UTC。用户在中国记录和活动时间均以北京时间为准。
## 关于标签
用户的每条记录可能带有标签(如 #培训 #交易 #生活 等)。标签是用户自己对记录的分类和上下文标注,对理解记录非常重要:
1. **标签揭示场景**:同一条灵感,标注 #培训 和标注 #交易 的解读角度完全不同。标签告诉你用户在什么情境下产生的这个想法。
2. **标签揭示延续性**:如果多天都有 #90后四级副职培训班 标签,说明这是一个持续性的活动/项目,分析时应体现跨日的连续思考。
3. **同标签分组**:当多条记录共享同一标签时,它们通常是同一主题下的不同侧面,应该融会贯通地分析,而不是孤立看待。
4. **不要忽略无标签记录**:没有标签的记录可能是随想、零散灵感,但也有其价值。
核心原则:不做缩写,不做分类表,不写空话。
具体要求:
1. **引用原文**:分析每一条灵感时,必须先引用(或精炼复述)用户的原文,让用户一眼就知道"哦我在说这个"。引用要压缩但不失原意,不能断章取义。
2. **融会贯通,而非逐条罗列**:不要"第一条...第二条...第三条..."机械堆砌。要把所有灵感当作一个整体来思考——哪几条在说同一个主题?哪几条看似无关实则互补?把它们串起来写。文章是流淌的整体,不是并列的零件。**注意利用标签来识别主题聚类**。
3. **承接延续之前的分析**:你会看到上一次的分析内容。今天的分析不能只写"今天的事",要把今天的灵感和昨天的分析结合起来——昨天讨论了什么?今天有什么进展?哪些问题有了答案?哪些问题还在延续?要让文章有跨日的时间纵深感。
4. **逐渐深入,有节奏感**:从表面现象往下挖。先指出用户说了什么,再追问为什么,再展开你的洞察。每一段都要比上一段深一层。
5. **有实质内容**:不要写"这是一个有价值的思考"这种废话。要说清楚为什么有价值,值在哪里,用户可以从这个方向挖到什么。
6. **关联要真实**:如果多条灵感确实指向同一个主题,就写一段贯通的分析。不要硬凑关联。
7. **待办要锋利**:待办事项不是"调研X""优化Y"这种万金油。要写到"具体做什么、什么时机做、做到什么程度"的颗粒度。
输出格式:纯 Markdown不要代码块包裹不要 JSON。
文章结构参考(不是模板,不必严格遵守):
### 1. 今日概览
一两句话点出今天思考的主旋律,并与之前的内容形成呼应。
### 2. 深入分析
融会贯通地写。引用原文要自然嵌入行文。注意与上次分析的承接关系。
### 3. 待办事项
具体、可执行的事项列表。
---
注意即便只有1条灵感也要写出深度。篇幅不设上限内容完整即可。"""
def build_user_prompt(memos, date):
"""Build user prompt for AI, with tag-aware formatting and grouping.
Memos are grouped by shared tags so the AI can see thematic clusters.
Each memo includes its tags in the text block.
"""
# Step 1: Collect all tags across today's memos
tag_freq = MemosClient.aggregate_tags(memos)
groups = MemosClient.group_memos_by_tag(memos)
prompt_parts = []
# Header: date + tag overview
date_str = date.strftime("%Y-%m-%d")
prompt_parts.append(f"以下是我 {date_str} 的灵感记录(共 {len(memos)} 条)。")
if tag_freq:
tag_summary = "".join(f"#{t}{c}条)" for t, c in tag_freq.items())
prompt_parts.append(f"\n今天的标签分布:{tag_summary}")
prompt_parts.append("\n标签说明:用户通过标签自行分类记录。同标签的记录通常属于同一主题或场景,分析时应关注标签揭示的上下文。")
# Step 2: Group memos by tags for thematic presentation
prompt_parts.append("\n---\n")
# Priority: present tagged memos first (grouped by tag), then untagged
sorted_tags = sorted(groups.keys(), key=lambda t: (t == "无标签", -len(groups[t])))
for tag in sorted_tags:
group_memos = groups[tag]
if tag == "无标签":
prompt_parts.append(f"\n## 无标签记录({len(group_memos)}条)")
else:
prompt_parts.append(f"\n## 标签 #{tag}{len(group_memos)}条)")
for m in group_memos:
time_bj = utc_to_beijing(m["created_at"])
content = m["content"].strip()
memo_tags = m.get("tags", [])
# Show all tags on this memo (for cross-tagged memos)
tag_line = ""
if memo_tags:
tag_line = " [" + ", ".join("#" + t for t in memo_tags) + "]"
prompt_parts.append(f"\n- **[{time_bj}]**{tag_line}\n{content}")
return "\n".join(prompt_parts)
def git_push(digest_file):
"""Commit and push the digest file to Gitea."""
try:
result = subprocess.run(
["git", "add", digest_file],
cwd=PROJECT_DIR,
capture_output=True, text=True, timeout=15
)
if result.returncode != 0:
logger.warning("git add failed: %s", result.stderr.strip())
return False
result = subprocess.run(
["git", "status", "--porcelain", "ai-insights/"],
cwd=PROJECT_DIR,
capture_output=True, text=True, timeout=10
)
if not result.stdout.strip():
logger.info("No changes to commit")
return True
date_str = datetime.now(TZ_BEIJING).strftime("%Y-%m-%d %H:%M")
result = subprocess.run(
["git", "commit", "-m", "daily digest " + date_str],
cwd=PROJECT_DIR,
capture_output=True, text=True, timeout=15
)
if result.returncode != 0 and "nothing to commit" not in result.stdout:
logger.warning("git commit failed: %s", result.stderr.strip())
return False
result = subprocess.run(
["git", "push", "origin", "main"],
cwd=PROJECT_DIR,
capture_output=True, text=True, timeout=30
)
if result.returncode != 0:
logger.warning("git push failed: %s", result.stderr.strip())
return False
for line in result.stdout.split("\n"):
if "->" in line or "remote:" in line:
logger.info("Gitea push: %s", line.strip())
logger.info("Pushed to Gitea successfully")
return True
except subprocess.TimeoutExpired:
logger.error("git push timed out")
return False
except Exception as e:
logger.error("git push error: %s", e)
return False
def run(memos_client, llm_client, date=None):
"""Run daily digest analysis."""
date = date or datetime.now(TZ_BEIJING)
logger.info("Daily digest started for %s", date.strftime("%Y-%m-%d"))
# Step 1: Fetch today's memos
memos = memos_client.list_memos(days=1)
if not memos:
logger.info("No memos today, skipping")
content = format_daily_digest(date, ai_body="今天没有记录灵感。", tags=["灵感收集器", "每日总结", "静默"], doc_type="daily-digest")
date_str = date.strftime("%Y-%m-%d")
daily_dir = os.path.join(get_output_dir("daily"), date_str)
os.makedirs(daily_dir, exist_ok=True)
now_str = datetime.now(TZ_BEIJING).strftime("%H%M%S")
filename = date_str + "_digest_" + now_str + ".md"
filepath = os.path.join(daily_dir, filename)
with open(filepath, "w", encoding="utf-8") as f:
f.write(content)
logger.info("Empty digest written to %s", filepath)
git_push(filepath)
return filepath, 0
# Step 2: Build tag-aware user prompt
user_prompt = build_user_prompt(memos, date)
# Step 3: Include previous analysis for continuity
prev_date, prev_analysis = read_previous_digest(date)
if prev_analysis:
user_prompt += (
"\n\n---\n\n"
"以下是我上一次的分析内容("
+ prev_date.strftime("%Y-%m-%d") + ""
"请结合今天的灵感一起思考,保持连续性:\n\n"
+ prev_analysis
)
# Step 4: Check for cross-day tag continuity
# If today has tags that appeared in previous digest, note this
tag_freq = MemosClient.aggregate_tags(memos)
today_tags = set(tag_freq.keys())
if prev_date and today_tags:
try:
prev_tags_line = None
# Extract tags from previous digest frontmatter
daily_root = get_output_dir("daily")
prev_dir = os.path.join(daily_root, prev_date.strftime("%Y-%m-%d"))
if os.path.isdir(prev_dir):
prev_files = sorted([f for f in os.listdir(prev_dir) if f.endswith(".md")], reverse=True)
if prev_files:
with open(os.path.join(prev_dir, prev_files[0]), "r", encoding="utf-8") as f:
prev_content = f.read(500)
if "tags:" in prev_content:
prev_tags_line = prev_content.split("tags:")[1].split("\n")[0]
if prev_tags_line:
import re
prev_tags = set(re.findall(r"'([^']+)'", prev_tags_line))
cross_tags = today_tags & prev_tags
if cross_tags:
user_prompt += (
"\n\n---\n\n"
"跨标签延续提示:以下标签与上一次分析("
+ prev_date.strftime("%Y-%m-%d")
+ ")共享:"
+ "".join("#" + t for t in sorted(cross_tags))
+ "。这些标签代表持续性的主题,请特别关注跨日的思维连续性。"
)
except Exception as e:
logger.debug("Cross-tag check failed: %s", e)
user_prompt += "\n\n---\n\n请基于以上所有素材,写一篇有深度的分析文章。"
# Step 5: Call DeepSeek API
raw_response = llm_client.ask(
system_prompt=DAILY_SYSTEM_PROMPT,
user_prompt=user_prompt,
temperature=0.5
)
# Step 6: Build frontmatter tags (default tags + today's user tags)
default_tags = ["灵感收集器", "每日总结", "AI分析"]
user_tags = sorted(tag_freq.keys())
all_tags = default_tags + user_tags
# Extract #知识 tagged memos for "今日新知" section
knowledge_items = [m for m in memos if "知识" in m.get("tags", [])]
knowledge_section = ""
if knowledge_items:
knowledge_lines = []
for m in knowledge_items:
time_bj = utc_to_beijing(m["created_at"])
content_clean = m["content"].strip()
knowledge_lines.append("- [%s] %s" % (time_bj, content_clean))
if knowledge_lines:
knowledge_section = "\n\n---\n\n### 今日新知\n\n你今天学到了以下新知识:\n\n" + "\n".join(knowledge_lines)
else:
knowledge_section = ""
content = format_daily_digest(
date, ai_body=raw_response + knowledge_section,
tags=all_tags, doc_type="daily-digest"
)
date_str = date.strftime("%Y-%m-%d")
daily_dir = os.path.join(get_output_dir("daily"), date_str)
os.makedirs(daily_dir, exist_ok=True)
now_str = datetime.now(TZ_BEIJING).strftime("%H%M%S")
filename = date_str + "_digest_" + now_str + ".md"
filepath = os.path.join(daily_dir, filename)
with open(filepath, "w", encoding="utf-8") as f:
f.write(content)
logger.info(
"Daily digest written to %s | %d memos | %d chars analysis | tags: %s",
filepath, len(memos), len(raw_response), ", ".join(user_tags) if user_tags else "(none)"
)
# Step 7: Auto-push to Gitea
git_push(filepath)
return filepath, len(memos)
def main():
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(name)s: %(message)s"
)
try:
secrets = load_secrets()
except FileNotFoundError as e:
print(e)
sys.exit(1)
memos_client = MemosClient(
base_url=secrets.get("memos_url", "http://localhost:5230"),
access_token=secrets["memos_token"]
)
llm_client = DeepSeekClient(
api_key=secrets["deepseek_api_key"],
model=secrets.get("deepseek_model", "deepseek-chat")
)
filepath, count = run(memos_client, llm_client)
print("Done: " + filepath + " (" + str(count) + " memos)")
if __name__ == "__main__":
main()