fix: extract_todos 适配新版 digest 格式\n\n- 递归搜索日期子目录(daily/YYYY-MM-DD/)\n- 文件名匹配从 _digest.md 改为包含 _digest 的 .md\n- 新增数字列表格式匹配(1. **bold**: desc)\n- 新增 dash 列表格式匹配(- **bold**: desc)\n- 过滤模板标题(【xxx】短标题)\n- 只从待办事项 section 开始提取

This commit is contained in:
Beast
2026-06-15 16:27:22 +08:00
parent 068ef58d56
commit 1a3a2117c4

View File

@ -1,4 +1,9 @@
"""Extract todos from the latest inspiration digest + sync to todo API.""" """Extract todos from the latest inspiration digest + sync to todo API.
Supports multiple formats:
- [ ] checkbox style
- 1. **bold**: description (numbered list with bold prefix)
- - **bold**: description (dash list with bold prefix)
"""
import json import json
import os import os
import re import re
@ -9,33 +14,88 @@ TODO_API = "http://127.0.0.1:9001/api/todos"
def get_latest_digest(): def get_latest_digest():
"""Find the most recent digest file.""" """Find the most recent digest file (checks date subdirectories)."""
if not os.path.isdir(DIGEST_DIR): if not os.path.isdir(DIGEST_DIR):
return None return None
files = [f for f in os.listdir(DIGEST_DIR) if f.endswith("_digest.md")] files = []
for root, dirs, filenames in os.walk(DIGEST_DIR):
for f in filenames:
if "_digest" in f and f.endswith(".md"):
files.append(os.path.join(root, f))
if not files: if not files:
return None return None
files.sort(reverse=True) files.sort(key=lambda x: os.path.getmtime(x), reverse=True)
return os.path.join(DIGEST_DIR, files[0]) return files[0]
def extract_todos(filepath): def extract_todos(filepath):
"""Extract [ ] todo items from the digest markdown.""" """Extract todo items from the digest markdown.
Matches:
1. [ ] checkbox items
2. Numbered items like "1. **bold title**description"
3. Dash items like "- **bold title**description"
Only extracts from the "待办事项" section onwards.
"""
with open(filepath) as f: with open(filepath) as f:
content = f.read() content = f.read()
todos = [] todos = []
# Match markdown checkboxes: - [ ] something in_todo_section = False
for line in content.split("\n"): for line in content.split("\n"):
stripped = line.strip()
# Detect start of todo section
if re.search(r"待办事项|待办|TODO|Todo", stripped) and not in_todo_section:
in_todo_section = True
continue
# End of todo section: next major heading or empty line after content
if in_todo_section and stripped.startswith("## ") and "待办" not in stripped:
break
if in_todo_section and stripped == "" and todos:
# Allow one blank line, but stop at second
if todos and todos[-1] == "__blank__":
todos.pop() # remove previous blank marker
break
todos.append("__blank__")
continue
if not in_todo_section:
continue
# Pattern 1: [ ] checkbox
m = re.match(r"\s*[-*]\s*\[\s*[ ]\s*\]\s*(.+)", line) m = re.match(r"\s*[-*]\s*\[\s*[ ]\s*\]\s*(.+)", line)
if m: if m:
todos.append(m.group(1).strip()) text = re.sub(r"\*\*", "", m.group(1).strip()) # strip bold markers
if text:
todos.append(text)
continue
# Pattern 2: numbered list "1. **bold**description" or "1. description"
m = re.match(r"\s*\d+\.\s+(.+)", line)
if m:
text = re.sub(r"\*\*", "", m.group(1).strip()) # strip bold markers
if text:
todos.append(text)
continue
# Pattern 3: dash list "- **bold**description"
m = re.match(r"\s*[-*]\s+(.+)", line)
if m:
text = re.sub(r"\*\*", "", m.group(1).strip())
# Skip template headers like 【今日输入】
if text and len(text) > 8 and not re.match(r"^【.+】$", text):
todos.append(text)
continue
# Clean up blank markers
todos = [t for t in todos if t != "__blank__"]
return todos return todos
def push_todos(todos): def push_todos(todos):
"""Send extracted todos to the API (skip if already exists).""" """Send extracted todos to the API (skip if already exists)."""
# Get existing todos
try: try:
resp = requests.get(TODO_API, timeout=5) resp = requests.get(TODO_API, timeout=5)
existing = [t["text"] for t in resp.json().get("todos", [])] existing = [t["text"] for t in resp.json().get("todos", [])]
@ -59,5 +119,7 @@ if __name__ == "__main__":
todos = extract_todos(filepath) todos = extract_todos(filepath)
print(f"Found {len(todos)} todos in {os.path.basename(filepath)}") print(f"Found {len(todos)} todos in {os.path.basename(filepath)}")
if todos: if todos:
for t in todos:
print(f" - {t[:60]}")
pushed = push_todos(todos) pushed = push_todos(todos)
print(f"Pushed {pushed} new todos to API") print(f"Pushed {pushed} new todos to API")