754214692e
- 新增 ai_enabled/ai_api_key/ai_base_url/ai_model 等配置项,通过 .env 管理 - 新增 app/services/ai_analyzer.py — DeepSeek API 调用 + 详情页正文提取 - 新增 extract_page_content() 从详情页抓取正文供 AI 分析 - Announcement 模型新增 ai_relevant / ai_analysis 字段 - 流水线集成 AI 分析步骤(关键词匹配后、通知前) - AI 标记为可承接的项目额外发送 markdown 着重通知 - 创建 alembic 迁移版本 6e8f4c2d1b0a
209 lines
7.1 KiB
Python
209 lines
7.1 KiB
Python
import hashlib
|
|
from datetime import datetime
|
|
from typing import Any
|
|
from urllib.parse import urljoin
|
|
from zoneinfo import ZoneInfo
|
|
|
|
_TZ = ZoneInfo("Asia/Shanghai")
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
|
|
def parse_gxgp_api_response(
|
|
response_data: dict[str, Any],
|
|
source_code: str,
|
|
source_name: str,
|
|
crawled_at: datetime,
|
|
category_id: int,
|
|
) -> list[dict[str, Any]]:
|
|
if not response_data.get("success"):
|
|
return []
|
|
data = response_data.get("result", {}).get("data", {})
|
|
records = data.get("data", [])
|
|
if not records:
|
|
return []
|
|
|
|
results = []
|
|
for record in records:
|
|
title = str(record.get("title", "")).strip()
|
|
if not title:
|
|
continue
|
|
|
|
timestamp = record.get("publishDate")
|
|
if not timestamp:
|
|
continue
|
|
try:
|
|
publish_date = datetime.fromtimestamp(int(timestamp) / 1000, tz=_TZ).replace(tzinfo=None)
|
|
except (ValueError, TypeError):
|
|
continue
|
|
|
|
purchase_name = str(record.get("purchaseName", "")).strip()
|
|
article_id = record.get("articleId")
|
|
if not article_id:
|
|
continue
|
|
|
|
content_url = (
|
|
f"https://zfcg.gxzf.gov.cn/site/detail?"
|
|
f"parentId={category_id}&articleId={article_id}"
|
|
)
|
|
|
|
announce = {
|
|
"title": title,
|
|
"publish_date": publish_date,
|
|
"purchase_name": purchase_name,
|
|
"content_url": content_url,
|
|
"source_code": source_code,
|
|
"source_name": source_name,
|
|
"announcement_type": "purchase",
|
|
"crawl_mode": "auto",
|
|
"is_new": True,
|
|
"is_today": publish_date.date() == datetime.now().date(),
|
|
}
|
|
announce["content_hash"] = _generate_hash(announce)
|
|
results.append(announce)
|
|
|
|
return results
|
|
|
|
|
|
def extract_pagination(response_data: dict[str, Any]) -> dict[str, Any]:
|
|
data = response_data.get("result", {}).get("data", {})
|
|
return {
|
|
"total": data.get("total", 0),
|
|
"page_no": data.get("pageNo", 1),
|
|
"page_size": data.get("pageSize", 100),
|
|
"pages": data.get("pages", 0),
|
|
"empty": data.get("empty", True),
|
|
"has_next": data.get("hasNext", False),
|
|
"has_previous": data.get("hasPrevious", False),
|
|
}
|
|
|
|
|
|
def parse_dahuagov_html(html: str, crawled_at: datetime) -> list[dict[str, Any]]:
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
lists = soup.find_all("ul", class_="more-list")
|
|
if not lists:
|
|
return []
|
|
|
|
results = []
|
|
base_url = "http://www.gxdh.gov.cn"
|
|
base_path = "/xxgk/zdlyxxgk/ggzypzly/zfcgly/cggg/"
|
|
|
|
for ul in lists:
|
|
for li in ul.find_all("li"):
|
|
date_span = li.find("span")
|
|
if not date_span:
|
|
continue
|
|
date_text = date_span.get_text(strip=True)
|
|
try:
|
|
publish_date = datetime.strptime(date_text, "%Y-%m-%d")
|
|
except ValueError:
|
|
continue
|
|
|
|
link_tag = li.find("a")
|
|
if not link_tag:
|
|
continue
|
|
title = link_tag.get("title", "") or link_tag.get_text(strip=True)
|
|
href = link_tag.get("href", "")
|
|
if not title or not href:
|
|
continue
|
|
|
|
if href.startswith("./") or href.startswith("../"):
|
|
content_url = urljoin(base_url + base_path, href)
|
|
elif href.startswith("/"):
|
|
content_url = base_url + href
|
|
elif href.startswith("http"):
|
|
content_url = href
|
|
else:
|
|
content_url = urljoin(base_url + base_path, href)
|
|
|
|
announce = {
|
|
"title": title,
|
|
"publish_date": publish_date,
|
|
"purchase_name": "大化瑶族自治县",
|
|
"content_url": content_url,
|
|
"source_code": "dahuagov",
|
|
"source_name": "大化县政府网采购公告",
|
|
"announcement_type": "purchase",
|
|
"crawl_mode": "auto",
|
|
"is_new": True,
|
|
"is_today": publish_date.date() == datetime.now().date(),
|
|
}
|
|
announce["content_hash"] = _generate_hash(announce)
|
|
results.append(announce)
|
|
|
|
return results
|
|
|
|
|
|
def parse_dahuagov_detail_pubdate(html: str) -> datetime | None:
|
|
"""从详情页 <meta name="PubDate"> 解析精确发布时间"""
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
meta = soup.find("meta", attrs={"name": "PubDate"})
|
|
if not meta:
|
|
return None
|
|
content = meta.get("content", "").strip()
|
|
for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%d %H:%M", "%Y-%m-%d"):
|
|
try:
|
|
return datetime.strptime(content, fmt)
|
|
except ValueError:
|
|
continue
|
|
return None
|
|
|
|
|
|
async def extract_page_content(url: str, timeout: int = 30) -> str | None:
|
|
"""抓取详情页并用 BeautifulSoup 提取正文纯文本"""
|
|
import httpx
|
|
|
|
try:
|
|
headers = {
|
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
|
"Accept": "text/html,application/xhtml+xml",
|
|
"Accept-Language": "zh-CN,zh;q=0.9",
|
|
}
|
|
async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client:
|
|
resp = await client.get(url, headers=headers)
|
|
if resp.status_code != 200:
|
|
return None
|
|
|
|
soup = BeautifulSoup(resp.text, "html.parser")
|
|
|
|
# 尝试多种常见正文容器选择器
|
|
selectors = [
|
|
"div.article-content", "div.content", "div.TRS_Editor",
|
|
"div.Custom_UnionStyle", "div.pages_content", "div#content",
|
|
"div.main-content", "article", ".article", ".detail-content",
|
|
".text-content", ".news-content", ".detail-article",
|
|
]
|
|
for selector in selectors:
|
|
container = soup.select_one(selector)
|
|
if container:
|
|
# 移除脚本和样式
|
|
for tag in container.find_all(["script", "style"]):
|
|
tag.decompose()
|
|
text = container.get_text(separator="\n", strip=True)
|
|
if len(text) > 50: # 至少50字才算有效正文
|
|
return text
|
|
|
|
# 兜底:取 body 内所有文本
|
|
body = soup.find("body")
|
|
if body:
|
|
for tag in body.find_all(["script", "style", "nav", "footer", "header"]):
|
|
tag.decompose()
|
|
text = body.get_text(separator="\n", strip=True)
|
|
# 移除过长的空白行
|
|
lines = [l.strip() for l in text.split("\n") if l.strip()]
|
|
text = "\n".join(lines[:200]) # 最多取前200行
|
|
if len(text) > 50:
|
|
return text
|
|
|
|
return None
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def _generate_hash(ann: dict[str, Any]) -> str:
|
|
content = (
|
|
f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"
|
|
f"|{ann['purchase_name']}|{ann['content_url']}|{ann['source_code']}"
|
|
)
|
|
return hashlib.sha256(content.encode("utf-8")).hexdigest()
|