397 lines
13 KiB
Python
397 lines
13 KiB
Python
"""
|
|
Markdown生成器模块
|
|
生成公告的Markdown格式输出文件
|
|
"""
|
|
|
|
import os
|
|
from pathlib import Path
|
|
from typing import List, Dict, Any, Optional
|
|
from datetime import datetime
|
|
from collections import defaultdict
|
|
|
|
try:
|
|
from ..core.models import Announcement
|
|
from ..core.config_manager import get_config
|
|
from ..core.logger import get_logger
|
|
except ImportError:
|
|
from core.models import Announcement
|
|
from core.config_manager import get_config
|
|
from core.logger import get_logger
|
|
|
|
|
|
logger = get_logger(__name__)
|
|
|
|
|
|
class MarkdownGenerator:
|
|
"""Markdown生成器"""
|
|
|
|
def __init__(self, output_file: Optional[str] = None):
|
|
"""
|
|
初始化Markdown生成器
|
|
|
|
Args:
|
|
output_file: 输出文件路径
|
|
"""
|
|
self.config = get_config()
|
|
self.output_file = output_file or self.config.markdown.output_file
|
|
self.max_entries = self.config.markdown.max_entries
|
|
self.include_today_highlight = self.config.markdown.include_today_highlight
|
|
|
|
# 确保输出目录存在
|
|
output_path = Path(self.output_file)
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
def generate_markdown(self, announcements: List[Announcement],
|
|
title: str = "广西政府采购网公告监控",
|
|
time_period: str = None) -> str:
|
|
"""
|
|
生成Markdown内容
|
|
|
|
Args:
|
|
announcements: 公告列表
|
|
title: 文档标题
|
|
|
|
Returns:
|
|
str: Markdown格式的文本
|
|
"""
|
|
if not announcements:
|
|
return self._generate_empty_markdown(title, time_period)
|
|
|
|
# 按来源分组
|
|
grouped_announcements = self._group_announcements_by_source(announcements)
|
|
|
|
# 生成Markdown
|
|
lines = []
|
|
lines.append("# 搜索完成")
|
|
lines.append("")
|
|
|
|
# 解析标题中的关键词
|
|
keyword = "未知"
|
|
if "关键词:" in title:
|
|
keyword_part = title.split("关键词:")[-1].strip()
|
|
keyword = keyword_part.split()[0] if keyword_part else "未知"
|
|
|
|
lines.append(f"📋 关键词搜索: `{keyword}` - 总公告数: `{len(announcements)}`")
|
|
lines.append("")
|
|
|
|
# 使用传入的时间段或默认的更新时间
|
|
if time_period:
|
|
lines.append(f"**时间段: {time_period}**")
|
|
else:
|
|
lines.append(f"**更新时间: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}**")
|
|
lines.append("")
|
|
lines.append("")
|
|
|
|
# 生成各来源的公告
|
|
for source_name, source_announcements in grouped_announcements.items():
|
|
lines.extend(self._generate_source_section(source_name, source_announcements))
|
|
lines.append("")
|
|
|
|
return "\n".join(lines)
|
|
|
|
def _group_announcements_by_source(self, announcements: List[Announcement]) -> Dict[str, List[Announcement]]:
|
|
"""按来源分组公告"""
|
|
grouped = defaultdict(list)
|
|
|
|
for announcement in announcements:
|
|
grouped[announcement.source_name].append(announcement)
|
|
|
|
# 对每个组内的公告按时间倒序排列
|
|
for source_name in grouped:
|
|
grouped[source_name].sort(key=lambda x: x.publish_date, reverse=True)
|
|
|
|
return dict(grouped)
|
|
|
|
def _generate_toc(self, grouped_announcements: Dict[str, List[Announcement]]) -> List[str]:
|
|
"""生成目录"""
|
|
lines = ["## 目录", ""]
|
|
|
|
for source_name, announcements in grouped_announcements.items():
|
|
# 创建锚点链接
|
|
anchor = self._create_anchor(source_name)
|
|
count = len(announcements)
|
|
lines.append(f"- [{source_name}](#{anchor}) ({count}条)")
|
|
|
|
return lines
|
|
|
|
def _generate_source_section(self, source_name: str, announcements: List[Announcement]) -> List[str]:
|
|
"""生成来源章节"""
|
|
lines = []
|
|
|
|
lines.append(f"## {source_name} - **共 {len(announcements)} 条**")
|
|
lines.append("")
|
|
|
|
# 生成公告列表
|
|
for i, announcement in enumerate(announcements, 1):
|
|
lines.extend(self._generate_announcement_item(announcement, i))
|
|
|
|
return lines
|
|
|
|
def _generate_announcement_item(self, announcement: Announcement, index: int) -> List[str]:
|
|
"""生成单个公告项"""
|
|
lines = []
|
|
|
|
# 公告标题(包含超链接)
|
|
title_line = f"### {index}. [{announcement.title}]({announcement.content_url})"
|
|
lines.append(title_line)
|
|
lines.append("")
|
|
|
|
# 公告信息 - 简洁格式
|
|
info_parts = []
|
|
|
|
if announcement.publish_date:
|
|
publish_date = announcement.publish_date.strftime("%Y-%m-%d")
|
|
info_parts.append(publish_date)
|
|
|
|
if announcement.purchase_name:
|
|
info_parts.append(announcement.purchase_name)
|
|
|
|
info_parts.append(announcement.source_name)
|
|
|
|
if info_parts:
|
|
info_line = " | ".join(info_parts)
|
|
lines.append(info_line)
|
|
lines.append("")
|
|
lines.append("")
|
|
|
|
return lines
|
|
|
|
def _generate_empty_markdown(self, title: str, time_period: str = None) -> str:
|
|
"""生成空内容的Markdown"""
|
|
# 解析标题中的关键词
|
|
keyword = "未知"
|
|
if "关键词:" in title:
|
|
keyword_part = title.split("关键词:")[-1].strip()
|
|
keyword = keyword_part.split()[0] if keyword_part else "未知"
|
|
|
|
lines = [
|
|
f"📋 关键词搜索: `{keyword}` - 总公告数: `0`",
|
|
"",
|
|
f"**时间段: {time_period}**" if time_period else f"**更新时间: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}**",
|
|
"",
|
|
"",
|
|
"## 无匹配公告",
|
|
"",
|
|
"在指定时间范围内没有找到符合条件的公告。",
|
|
""
|
|
]
|
|
|
|
return "\n".join(lines)
|
|
|
|
def _create_anchor(self, text: str) -> str:
|
|
"""创建锚点链接"""
|
|
# 移除特殊字符,替换空格为连字符,转为小写
|
|
import re
|
|
anchor = re.sub(r'[^\w\s-]', '', text)
|
|
anchor = re.sub(r'[-\s]+', '-', anchor)
|
|
return anchor.lower().strip('-')
|
|
|
|
def save_to_file(self, announcements: List[Announcement],
|
|
title: Optional[str] = None) -> bool:
|
|
"""
|
|
保存Markdown到文件
|
|
|
|
Args:
|
|
announcements: 公告列表
|
|
title: 文档标题
|
|
|
|
Returns:
|
|
bool: 保存是否成功
|
|
"""
|
|
try:
|
|
markdown_content = self.generate_markdown(announcements, title)
|
|
|
|
with open(self.output_file, 'w', encoding='utf-8') as f:
|
|
f.write(markdown_content)
|
|
|
|
logger.info(f"Markdown文件已保存到: {self.output_file} (共 {len(announcements)} 条公告)")
|
|
return True
|
|
|
|
except Exception as e:
|
|
logger.error(f"保存Markdown文件失败: {str(e)}")
|
|
return False
|
|
|
|
def append_to_file(self, new_announcements: List[Announcement]) -> bool:
|
|
"""
|
|
追加新公告到现有文件
|
|
|
|
Args:
|
|
new_announcements: 新公告列表
|
|
|
|
Returns:
|
|
bool: 追加是否成功
|
|
"""
|
|
if not new_announcements:
|
|
return True
|
|
|
|
try:
|
|
# 读取现有文件
|
|
existing_content = ""
|
|
if os.path.exists(self.output_file):
|
|
with open(self.output_file, 'r', encoding='utf-8') as f:
|
|
existing_content = f.read()
|
|
|
|
# 如果文件不存在或为空,创建新文件
|
|
if not existing_content.strip():
|
|
return self.save_to_file(new_announcements)
|
|
|
|
# 解析现有公告(这里简化处理,实际可能需要更复杂的解析)
|
|
# 为简单起见,我们重新生成完整文件
|
|
logger.info("重新生成完整Markdown文件")
|
|
return self.save_to_file(new_announcements)
|
|
|
|
except Exception as e:
|
|
logger.error(f"追加公告到Markdown文件失败: {str(e)}")
|
|
return False
|
|
|
|
def get_file_stats(self) -> Dict[str, Any]:
|
|
"""获取文件统计信息"""
|
|
stats = {
|
|
"file_exists": False,
|
|
"file_size": 0,
|
|
"last_modified": None,
|
|
"announcement_count": 0
|
|
}
|
|
|
|
try:
|
|
if os.path.exists(self.output_file):
|
|
file_stat = os.stat(self.output_file)
|
|
stats["file_exists"] = True
|
|
stats["file_size"] = file_stat.st_size
|
|
stats["last_modified"] = datetime.fromtimestamp(file_stat.st_mtime).isoformat()
|
|
|
|
# 尝试统计公告数量(简单计数)
|
|
with open(self.output_file, 'r', encoding='utf-8') as f:
|
|
content = f.read()
|
|
# 统计###开头的行(每个公告的标题行)
|
|
stats["announcement_count"] = content.count("### ")
|
|
|
|
except Exception as e:
|
|
logger.warning(f"获取文件统计信息失败: {str(e)}")
|
|
|
|
return stats
|
|
|
|
|
|
class AnnouncementMarkdownFormatter:
|
|
"""公告Markdown格式化器"""
|
|
|
|
@staticmethod
|
|
def format_announcement_card(announcement: Announcement) -> str:
|
|
"""格式化单个公告为卡片样式"""
|
|
lines = []
|
|
|
|
# 标题
|
|
emoji = "🆕" if announcement.is_today else "📄"
|
|
lines.append(f"### {emoji} {announcement.title}")
|
|
lines.append("")
|
|
|
|
# 链接
|
|
lines.append(f"[查看详情]({announcement.content_url})")
|
|
lines.append("")
|
|
|
|
# 信息表格
|
|
lines.append("| 属性 | 值 |")
|
|
lines.append("|------|-----|")
|
|
|
|
if announcement.publish_date:
|
|
lines.append(f"| 发布时间 | {announcement.publish_date.strftime('%Y-%m-%d %H:%M')} |")
|
|
|
|
lines.append(f"| 发布单位 | {announcement.purchase_name or 'N/A'} |")
|
|
lines.append(f"| 来源 | {announcement.source_name} |")
|
|
lines.append(f"| 公告类型 | {announcement.announcement_type.value} |")
|
|
|
|
if announcement.crawled_at:
|
|
lines.append(f"| 爬取时间 | {announcement.crawled_at.strftime('%m-%d %H:%M')} |")
|
|
|
|
lines.append("")
|
|
|
|
return "\n".join(lines)
|
|
|
|
@staticmethod
|
|
def format_announcement_list(announcements: List[Announcement]) -> str:
|
|
"""格式化公告列表"""
|
|
if not announcements:
|
|
return "*暂无公告*"
|
|
|
|
lines = []
|
|
for announcement in announcements:
|
|
emoji = "🆕" if announcement.is_today else "•"
|
|
publish_date = announcement.publish_date.strftime("%m-%d") if announcement.publish_date else "N/A"
|
|
line = f"{emoji} [{announcement.title}]({announcement.content_url}) - {publish_date}"
|
|
lines.append(line)
|
|
|
|
return "\n".join(lines)
|
|
|
|
@staticmethod
|
|
def format_notification_message(announcements: List[Announcement],
|
|
max_count: int = 10) -> str:
|
|
"""格式化为通知消息"""
|
|
if not announcements:
|
|
return "暂无新公告"
|
|
|
|
# 只显示前N条
|
|
display_announcements = announcements[:max_count]
|
|
remaining_count = len(announcements) - max_count
|
|
|
|
lines = [f"🔔 发现 {len(announcements)} 条新公告:", ""]
|
|
|
|
for announcement in display_announcements:
|
|
title = announcement.title[:50] + "..." if len(announcement.title) > 50 else announcement.title
|
|
publish_date = announcement.publish_date.strftime("%m-%d") if announcement.publish_date else "N/A"
|
|
lines.append(f"• {title} ({publish_date})")
|
|
|
|
if remaining_count > 0:
|
|
lines.append(f"... 还有 {remaining_count} 条公告")
|
|
|
|
lines.append("")
|
|
lines.append("*点击公告标题查看详情*")
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
def create_markdown_generator(output_file: Optional[str] = None) -> MarkdownGenerator:
|
|
"""
|
|
创建Markdown生成器实例
|
|
|
|
Args:
|
|
output_file: 输出文件路径
|
|
|
|
Returns:
|
|
MarkdownGenerator: 生成器实例
|
|
"""
|
|
return MarkdownGenerator(output_file)
|
|
|
|
|
|
def generate_onu_md(announcements: List[Announcement]) -> bool:
|
|
"""
|
|
生成onu.md文件
|
|
|
|
Args:
|
|
announcements: 公告列表
|
|
|
|
Returns:
|
|
bool: 生成是否成功
|
|
"""
|
|
generator = create_markdown_generator()
|
|
return generator.save_to_file(announcements, "广西政府采购网公告监控")
|
|
|
|
|
|
def update_onu_md(new_announcements: List[Announcement]) -> bool:
|
|
"""
|
|
更新onu.md文件,追加新公告
|
|
|
|
Args:
|
|
new_announcements: 新公告列表
|
|
|
|
Returns:
|
|
bool: 更新是否成功
|
|
"""
|
|
generator = create_markdown_generator()
|
|
|
|
# 如果文件不存在,创建新文件
|
|
if not os.path.exists(generator.output_file):
|
|
return generator.save_to_file(new_announcements)
|
|
|
|
# 否则追加新公告
|
|
return generator.append_to_file(new_announcements)
|