Files
openclaw-config/workspace-resume/weekly-reports/sync_reports_db.py
T

363 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
周报数据库同步工具(SQLite 版本)
将 Markdown 周报的元数据同步到 SQLite 数据库
用法:
python3 sync_reports_db.py --all # 同步所有周报
python3 sync_reports_db.py --file xxx.md # 同步单个文件
python3 sync_reports_db.py --week 2026-W35 # 同步指定周
python3 sync_reports_db.py --stats # 查看统计信息
"""
import os
import re
import sys
import argparse
import json
import sqlite3
from pathlib import Path
from datetime import datetime, timedelta
REPORTS_DIR = Path(__file__).parent
DB_PATH = REPORTS_DIR / 'weekly_reports.db'
def init_database():
"""初始化 SQLite 数据库"""
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
# 创建周报元数据表
cursor.execute('''
CREATE TABLE IF NOT EXISTS weekly_reports (
id INTEGER PRIMARY KEY AUTOINCREMENT,
year INTEGER NOT NULL,
week_number INTEGER NOT NULL,
start_date TEXT NOT NULL,
end_date TEXT NOT NULL,
project TEXT NOT NULL,
main_task TEXT,
file_path TEXT NOT NULL UNIQUE,
has_problems INTEGER DEFAULT 0,
word_count INTEGER DEFAULT 0,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
)
''')
# 创建每日工作明细表
cursor.execute('''
CREATE TABLE IF NOT EXISTS weekly_report_daily (
id INTEGER PRIMARY KEY AUTOINCREMENT,
report_id INTEGER NOT NULL,
day_of_week INTEGER NOT NULL,
work_date TEXT NOT NULL,
work_items TEXT,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
FOREIGN KEY (report_id) REFERENCES weekly_reports(id) ON DELETE CASCADE
)
''')
# 创建问题记录表
cursor.execute('''
CREATE TABLE IF NOT EXISTS weekly_report_problems (
id INTEGER PRIMARY KEY AUTOINCREMENT,
report_id INTEGER NOT NULL,
problem_description TEXT NOT NULL,
status TEXT DEFAULT 'open',
resolved_at TEXT,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
FOREIGN KEY (report_id) REFERENCES weekly_reports(id) ON DELETE CASCADE
)
''')
# 创建索引
cursor.execute('CREATE INDEX IF NOT EXISTS idx_date ON weekly_reports(start_date, end_date)')
cursor.execute('CREATE INDEX IF NOT EXISTS idx_project ON weekly_reports(project)')
cursor.execute('CREATE UNIQUE INDEX IF NOT EXISTS uk_year_week ON weekly_reports(year, week_number, project)')
conn.commit()
return conn
def parse_markdown_report(filepath):
"""解析 Markdown 周报文件,提取元数据"""
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
data = {
'file_path': str(filepath),
'year': None,
'week_number': None,
'start_date': None,
'end_date': None,
'project': None,
'main_task': None,
'daily_work': {},
'problems': [],
'word_count': len(content)
}
# 提取年份和周数(从文件名)
filename = filepath.name
match = re.match(r'(\d{4})-W(\d{2})', filename)
if match:
data['year'] = int(match.group(1))
data['week_number'] = int(match.group(2))
# 提取日期范围(多种格式兼容)
# 格式 1: **日期范围:** 2026-08-25 ~ 2026-08-29
date_match = re.search(r'\*\*日期范围:\*\*\s*(\d{4}-\d{2}-\d{2})\s*~\s*(\d{4}-\d{2}-\d{2})', content)
if date_match:
data['start_date'] = date_match.group(1)
data['end_date'] = date_match.group(2)
else:
# 格式 2: 2026-07-27 ~ 2026-07-31(在文件末尾)
date_match2 = re.search(r'(\d{4}-\d{2}-\d{2})\s*~\s*(\d{4}-\d{2}-\d{2})', content)
if date_match2:
data['start_date'] = date_match2.group(1)
data['end_date'] = date_match2.group(2)
else:
# 格式 3: 从文件名推算(2026-W31
if data['year'] and data['week_number']:
# 计算该周的周一和周五
jan_first = datetime(data['year'], 1, 1)
week_start = jan_first + timedelta(weeks=data['week_number'] - 1)
week_start = week_start - timedelta(days=week_start.weekday())
week_end = week_start + timedelta(days=4)
data['start_date'] = week_start.strftime('%Y-%m-%d')
data['end_date'] = week_end.strftime('%Y-%m-%d')
# 提取项目名称(多种格式兼容)
# 格式 1: 项目名称:维云智造 G5
project_match = re.search(r'项目名称:\s*(维云智造\s*(G\d+)|G\d+)', content)
if project_match:
project_text = project_match.group(1)
g_match = re.search(r'G(\d+)', project_text)
if g_match:
data['project'] = f"G{g_match.group(1)}"
else:
data['project'] = 'G5'
else:
# 格式 2: **项目:** 维云智造 G6
project_match2 = re.search(r'\*\*项目:\*\*\s*(维云智造\s*(G\d+)|G\d+)', content)
if project_match2:
project_text = project_match2.group(1)
g_match = re.search(r'G(\d+)', project_text)
if g_match:
data['project'] = f"G{g_match.group(1)}"
else:
data['project'] = 'G5'
else:
# 默认 G5
data['project'] = 'G5'
# 提取主要任务
task_match = re.search(r'主要任务:\s*(.+?)(?:\n|$)', content)
if task_match:
data['main_task'] = task_match.group(1).strip()
# 提取每日工作
daily_section = re.search(r'## 本周工作内容\s*\n([\s\S]*?)(?:\n##|\Z)', content)
if daily_section:
daily_text = daily_section.group(1)
days = re.findall(r'###?\s*(?:周 [一二三四五六日]|(\d{4}-\d{2}-\d{2}))\s*\n([\s\S]*?)(?=###?|$)', daily_text)
for date_str, work_text in days:
if date_str:
items = re.findall(r'^\s*[-•*]\s*(.+)$', work_text, re.MULTILINE)
data['daily_work'][date_str] = items
# 提取存在问题
problem_section = re.search(r'## 存在问题\s*\n([\s\S]*?)(?:\n##|\Z)', content)
if problem_section:
problem_text = problem_section.group(1).strip()
if problem_text and problem_text != '无':
data['problems'] = [line.strip() for line in problem_text.split('\n') if line.strip()]
return data
def sync_report(conn, report_data):
"""同步单个周报到数据库"""
cursor = conn.cursor()
try:
# 插入或更新周报元数据
cursor.execute('''
INSERT INTO weekly_reports
(year, week_number, start_date, end_date, project, main_task, file_path, has_problems, word_count)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(year, week_number, project) DO UPDATE SET
main_task = excluded.main_task,
file_path = excluded.file_path,
has_problems = excluded.has_problems,
word_count = excluded.word_count,
updated_at = CURRENT_TIMESTAMP
''', (
report_data['year'],
report_data['week_number'],
report_data['start_date'],
report_data['end_date'],
report_data['project'],
report_data['main_task'],
report_data['file_path'],
1 if report_data['problems'] else 0,
report_data['word_count']
))
report_id = cursor.lastrowid
if report_id is None:
# 更新操作,需要获取现有 ID
cursor.execute('SELECT id FROM weekly_reports WHERE file_path = ?', (report_data['file_path'],))
row = cursor.fetchone()
if row:
report_id = row[0]
# 删除旧的每日工作记录
cursor.execute('DELETE FROM weekly_report_daily WHERE report_id = ?', (report_id,))
# 插入新的每日工作记录
for work_date, items in report_data['daily_work'].items():
day_of_week = datetime.strptime(work_date, '%Y-%m-%d').weekday() + 1
items_json = json.dumps(items, ensure_ascii=False)
cursor.execute('''
INSERT INTO weekly_report_daily
(report_id, day_of_week, work_date, work_items)
VALUES (?, ?, ?, ?)
''', (report_id, day_of_week, work_date, items_json))
# 删除旧的问题记录
cursor.execute('DELETE FROM weekly_report_problems WHERE report_id = ?', (report_id,))
# 插入新的问题记录
for problem in report_data['problems']:
cursor.execute('''
INSERT INTO weekly_report_problems
(report_id, problem_description)
VALUES (?, ?)
''', (report_id, problem))
conn.commit()
return True
except sqlite3.Error as e:
print(f"❌ 数据库错误:{e}")
conn.rollback()
return False
def show_stats(conn):
"""显示统计信息"""
cursor = conn.cursor()
print("\n📊 周报统计信息\n")
print("=" * 60)
# 总览
cursor.execute('SELECT COUNT(*) FROM weekly_reports')
total = cursor.fetchone()[0]
print(f"📁 周报总数:{total} 周")
# 按项目统计
cursor.execute('''
SELECT project, COUNT(*) as count,
SUM(has_problems) as problem_weeks,
AVG(word_count) as avg_words
FROM weekly_reports
GROUP BY project
''')
print("\n📋 按项目统计:")
for row in cursor.fetchall():
print(f" {row[0]}: {row[1]} 周,问题 {row[2]} 周,平均字数 {int(row[3])}")
# 最近 5 周
cursor.execute('''
SELECT year, week_number, start_date, end_date, project, main_task, has_problems
FROM weekly_reports
ORDER BY year DESC, week_number DESC
LIMIT 5
''')
print("\n📅 最近 5 周:")
for row in cursor.fetchall():
problem_flag = "⚠️" if row[6] else "✅"
print(f" {problem_flag} {row[0]}-W{row[1]:02d} ({row[2]}~{row[3]}) {row[4]} - {row[5][:30]}")
# 存在问题统计
cursor.execute('SELECT COUNT(*) FROM weekly_reports WHERE has_problems = 1')
problem_count = cursor.fetchone()[0]
print(f"\n⚠️ 存在问题周报:{problem_count} 周")
print("=" * 60)
def main():
parser = argparse.ArgumentParser(description='周报数据库同步工具(SQLite')
group = parser.add_mutually_exclusive_group(required=True)
group.add_argument('--all', action='store_true', help='同步所有周报')
group.add_argument('--file', type=str, help='同步单个文件')
group.add_argument('--week', type=str, help='同步指定周,如 2026-W35')
group.add_argument('--stats', action='store_true', help='查看统计信息')
args = parser.parse_args()
# 初始化数据库
conn = init_database()
print(f"✅ 数据库已初始化:{DB_PATH}")
if args.stats:
show_stats(conn)
conn.close()
return 0
files_to_sync = []
if args.file:
files_to_sync = [Path(args.file)]
elif args.week:
match = re.match(r'(\d{4})-W(\d{2})', args.week)
if match:
year, week = match.groups()
pattern = f"{year}-W{week}-*.md"
files_to_sync = list(REPORTS_DIR.glob(f"{year}/{pattern}"))
elif args.all:
for year_dir in REPORTS_DIR.iterdir():
if year_dir.is_dir() and year_dir.name.isdigit():
files_to_sync.extend(year_dir.glob("*.md"))
if not files_to_sync:
print("⚠️ 未找到需要同步的文件")
conn.close()
return 1
print(f"📂 找到 {len(files_to_sync)} 个文件待同步\n")
success_count = 0
for filepath in sorted(files_to_sync):
if not filepath.exists():
print(f"⚠️ 文件不存在:{filepath}")
continue
print(f"📄 处理:{filepath.name}")
report_data = parse_markdown_report(filepath)
if not report_data['year'] or not report_data['week_number']:
print(f" ⚠️ 无法解析年份/周数,跳过")
continue
if sync_report(conn, report_data):
print(f" ✅ 同步成功")
success_count += 1
else:
print(f" ❌ 同步失败")
print(f"\n{'✅' if success_count == len(files_to_sync) else '⚠️'} 完成:{success_count}/{len(files_to_sync)} 个文件")
conn.close()
return 0 if success_count == len(files_to_sync) else 1
if __name__ == '__main__':
exit(main())