fix: 大化县政府网公告抓取详情页获取精确发布时间
列表页只有日期(00:00),改为并发抓取详情页解析 <meta name="PubDate"> 获取精确时分秒,失败时保留列表页日期作为 fallback。 Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -6,7 +6,7 @@ import httpx
|
|||||||
|
|
||||||
from app.config import settings
|
from app.config import settings
|
||||||
from app.crawler.base import BaseSpider, CrawlResult, PipelineConfig
|
from app.crawler.base import BaseSpider, CrawlResult, PipelineConfig
|
||||||
from app.crawler.parsers import parse_dahuagov_html
|
from app.crawler.parsers import parse_dahuagov_html, parse_dahuagov_detail_pubdate
|
||||||
|
|
||||||
|
|
||||||
class DahuagovSpider(BaseSpider):
|
class DahuagovSpider(BaseSpider):
|
||||||
@@ -56,7 +56,9 @@ class DahuagovSpider(BaseSpider):
|
|||||||
crawled_at=start_time,
|
crawled_at=start_time,
|
||||||
)
|
)
|
||||||
|
|
||||||
announcements = parse_dahuagov_html(html, start_time)
|
announcements = parse_dahuagov_html(html, start_time)
|
||||||
|
await self._fetch_detail_dates(client, announcements, headers)
|
||||||
|
|
||||||
duration = (datetime.now() - start_time).total_seconds()
|
duration = (datetime.now() - start_time).total_seconds()
|
||||||
|
|
||||||
return CrawlResult(
|
return CrawlResult(
|
||||||
@@ -69,6 +71,31 @@ class DahuagovSpider(BaseSpider):
|
|||||||
duration=duration,
|
duration=duration,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
async def _fetch_detail_dates(
|
||||||
|
self,
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
announcements: list[dict],
|
||||||
|
headers: dict,
|
||||||
|
) -> None:
|
||||||
|
"""并发抓取详情页,用 PubDate meta 更新精确发布时间"""
|
||||||
|
sem = asyncio.Semaphore(5)
|
||||||
|
|
||||||
|
async def fetch_one(ann: dict) -> None:
|
||||||
|
async with sem:
|
||||||
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
||||||
|
try:
|
||||||
|
resp = await client.get(ann["content_url"], headers=headers)
|
||||||
|
if resp.status_code == 200:
|
||||||
|
pub = parse_dahuagov_detail_pubdate(resp.text)
|
||||||
|
if pub:
|
||||||
|
ann["publish_date"] = pub
|
||||||
|
ann["is_today"] = pub.date() == datetime.now().date()
|
||||||
|
except Exception:
|
||||||
|
pass # 保留列表页日期作为 fallback
|
||||||
|
|
||||||
|
await asyncio.gather(*[fetch_one(ann) for ann in announcements])
|
||||||
|
|
||||||
async def _delay(self):
|
async def _delay(self):
|
||||||
delay = random.uniform(1.0, 3.0)
|
delay = random.uniform(1.0, 3.0)
|
||||||
await asyncio.sleep(delay)
|
await asyncio.sleep(delay)
|
||||||
|
|
||||||
|
|||||||
@@ -134,6 +134,21 @@ def parse_dahuagov_html(html: str, crawled_at: datetime) -> list[dict[str, Any]]
|
|||||||
return results
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
def parse_dahuagov_detail_pubdate(html: str) -> datetime | None:
|
||||||
|
"""从详情页 <meta name="PubDate"> 解析精确发布时间"""
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
meta = soup.find("meta", attrs={"name": "PubDate"})
|
||||||
|
if not meta:
|
||||||
|
return None
|
||||||
|
content = meta.get("content", "").strip()
|
||||||
|
for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%d %H:%M", "%Y-%m-%d"):
|
||||||
|
try:
|
||||||
|
return datetime.strptime(content, fmt)
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _generate_hash(ann: dict[str, Any]) -> str:
|
def _generate_hash(ann: dict[str, Any]) -> str:
|
||||||
content = (
|
content = (
|
||||||
f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"
|
f"{ann['title']}|{ann['publish_date'].strftime('%Y-%m-%d')}"
|
||||||
|
|||||||
Reference in New Issue
Block a user