#!/usr/bin/env python3
"""
crawl_ycq.py — 峄城区人民政府 /ywdt/tzgg/ 通知公告
CMS: 静态HTML分页 (index_N.html)
列表: <span class="newstxt"><a href="./YYYYMM/tXXXXXXXXX.html">title</a></span>
      <span class="date">YYYY-MM-DD</span>
详情: <meta ArticleTitle / PubDate | <div class="zwnr">正文HTML</div>
页数: 25页（首页+index_1~24），~20条/页
"""

import sys, sqlite3, re, os, time, random
from datetime import datetime, timedelta
from urllib.parse import urljoin
import urllib.request

DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")
SITE_NAME = "峄城区通知公告"
SOURCE = "峄城区政府"
BASE_URL = "http://www.ycq.gov.cn/ywdt/tzgg/"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}
CUTOFF = (datetime.now() - timedelta(days=3 * 365)).strftime("%Y-%m-%d")
MAX_PAGES = 25  # 首页 + index_1 ~ index_24
seen_urls = set()

def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=25)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        log(f"请求失败: {url[-60:]}: {e}")
        return None

def extract_list(html, page_url):
    items = []
    for m in re.finditer(
        r'<span class="newstxt"><a href="([^"]+)"[^>]*>(.*?)</a></span><span class="date">([^<]+)</span>',
        html,
    ):
        rel_url = m.group(1).strip()
        title = re.sub(r"<[^>]+>", "", m.group(2)).strip()
        title = re.sub(r"^[·\s]*&middot;\s*", "", title).strip()
        date_str = m.group(3).strip()
        full_url = urljoin(page_url, rel_url)
        # Only keep internal links
        if not full_url.startswith("http://www.ycq.gov.cn") and not full_url.startswith("http://ycq.gov.cn"):
            continue
        items.append((full_url, title, date_str))
    return items

def extract_detail(html):
    title = ""
    content = ""
    pub_date = ""
    m = re.search(r'ArticleTitle"\s*content="([^"]+)"', html)
    if m:
        title = m.group(1).strip()
    m = re.search(r'PubDate"\s*content="([^"]+)"', html)
    if m:
        pub_date = m.group(1).strip()
    m = re.search(r'class="zwnr">(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    return title, content, pub_date

def save_to_db(articles):
    conn = sqlite3.connect(DB_PATH)
    cursor = conn.cursor()
    inserted = 0
    skipped = 0
    for url, title, date, content in articles:
        try:
            cursor.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, title, page_url, publish_date, source_url, content)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (SITE_NAME, title, url, date, SOURCE, content or ""),
            )
            if cursor.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"入库失败: {url[-50:]}: {e}")
    conn.commit()
    conn.close()
    return inserted, skipped

def main():
    pages = [BASE_URL] + [f"{BASE_URL}index_{i}.html" for i in range(1, MAX_PAGES)]
    total_articles = []

    for page_url in pages:
        log(f"取列表: {page_url}")
        html = fetch(page_url)
        if not html:
            continue
        items = extract_list(html, page_url)
        if not items:
            log(f"  无内容，停止分页")
            break
        log(f"  共{len(items)}条")

        for url, title, date_str in items:
            if date_str < CUTOFF:
                continue
            if url in seen_urls:
                continue
            seen_urls.add(url)

            # 抓详情
            time.sleep(random.uniform(0.3, 0.8))
            detail_html = fetch(url)
            if not detail_html:
                continue
            d_title, content, d_date = extract_detail(detail_html)
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date_str
            if not content:
                log(f"  无正文: {d_title[:40]}")
                continue

            total_articles.append((url, d_title, d_date, content))

        time.sleep(0.3)

    inserted, skipped = save_to_db(total_articles)
    log(f"完成！新增{inserted}条，跳过{skipped}条（总计{len(total_articles)}条）")

if __name__ == "__main__":
    main()
