#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
肥城市人民政府 — 通知公告 爬虫 (col/col173747, TAD51)
带重试机制
"""
import re, sys, os, time, sqlite3, urllib.request, ssl, urllib.parse

SITE_NAME = "肥城市人民政府-通知公告"
BASE_URL = "http://www.feicheng.gov.cn"
LIST_URL = BASE_URL + "/module/xxgk/search.jsp?standardXxgk=1&infotypeId=TAD51&vc_title=&vc_number=&area="
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CTX = ssl._create_unverified_context()

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36"
}
LIST_HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "http://www.feicheng.gov.cn/col/col173747/index.html?number=TAD51&vc_xxgkarea=FC127154154545401-1",
    "Content-Type": "application/x-www-form-urlencoded",
}
LIST_DATA_TPL = {
    "infotypeId":"TAD51","jdid":"1","area":"","divid":"div4",
    "vc_title":"","vc_number":"","sortfield":"top:0,createdatetime:0,orderid:0",
    "fields":"","fieldConfigId":"","hasNoPages":"","infoCount":"",
    "vc_filenumber":"","vc_all":"","texttype":"","fbtime":""
}

def retry_request(url, method="GET", data=None, headers=None, max_retries=3):
    for attempt in range(1, max_retries + 1):
        try:
            if method == "POST":
                req = urllib.request.Request(url, data=urllib.parse.urlencode(data).encode(), headers=headers)
            else:
                req = urllib.request.Request(url, headers=headers or HEADERS)
            resp = urllib.request.urlopen(req, timeout=60, context=CTX)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if attempt < max_retries:
                wait = attempt * 3
                log(f"请求失败(第{attempt}次): {str(e)[:50]}，{wait}s后重试")
                time.sleep(wait)
            else:
                log(f"请求失败(已放弃): {str(e)[:60]}")
    return None

def log(msg):
    print(f"[{time.strftime('%H:%M:%S')}] {msg}")

def fetch_list_page(page):
    data = dict(LIST_DATA_TPL)
    data["currpage"] = str(page)
    return retry_request(LIST_URL, "POST", data, LIST_HEADERS, max_retries=3)

def parse_articles(html):
    arts = []
    for item in re.findall(r"<li>(.*?)</li>", html, re.DOTALL):
        m = re.search(r'<a\s+title="([^"]*)"[^>]*href="([^"]*)"', item)
        dm = re.search(r"<b>\s*(\d{4}-\d{2}-\d{2})\s*</b>", item)
        if m:
            title = m.group(1).strip()
            url = m.group(2).strip()
            if not url.startswith("http"):
                url = BASE_URL + url
            pub_date = dm.group(1).strip() if dm else ""
            arts.append({"title": title, "url": url, "pub_date": pub_date})
    return arts

def fetch_detail(url):
    html = retry_request(url, max_retries=2)
    if not html:
        return "", []
    lines = []
    for p in re.findall(r"<p[^>]*>(.*?)</p>", html, re.DOTALL):
        t = re.sub(r"<[^>]+>", "", p).strip().replace("&nbsp;"," ").replace("\u3000"," ").strip()
        if t:
            lines.append(t)
    content = "\n".join(lines)
    atts = []
    for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]*)</a>', html, re.I):
        link = m.group(1)
        if not link.startswith("http"):
            link = BASE_URL + link if link.startswith("/") else BASE_URL + "/" + link
        atts.append({"url": link, "name": m.group(2).strip() or link.rsplit("/",1)[-1]})
    return content, atts

def get_total_pages(html):
    m = re.search(r"\u5171(?:&nbsp;|\s)*(\d+)(?:&nbsp;|\s)*\u9875", html)
    if m:
        return int(m.group(1))
    m2 = re.search(r"\u5171(?:&nbsp;|\s)*(\d+)(?:&nbsp;|\s)*\u6761", html)
    if m2:
        return (int(m2.group(1)) + 14) // 15
    return None

def main():
    log("开始爬取肥城市通知公告...")

    # 命令行参数：最大页数
    max_pages = None
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
            log(f"参数: 最大页数={max_pages}")
        except ValueError:
            pass

    # DB连接（提前打开，用于增量断点查询）
    db = sqlite3.connect(SEARCH_DB, timeout=30)
    db.execute("PRAGMA journal_mode=WAL")
    existing = set()
    latest_in_db = None
    try:
        cur = db.execute("SELECT MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        row = cur.fetchone()
        if row and row[0]:
            latest_in_db = row[0]
        cur = db.execute("SELECT title FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        for row in cur.fetchall():
            existing.add(row[0])
        log(f"已有记录: {len(existing)} 条, 最新日期: {latest_in_db or '无'}")
    except Exception as e:
        log(f"读取已有记录失败: {e}")

    html = fetch_list_page(1)
    if not html:
        log("ERROR: 无法获取第1页")
        return
    total = get_total_pages(html)
    if not total:
        log("ERROR: 无法获取总页数")
        return
    if max_pages and total > max_pages:
        total = max_pages
    log(f"将爬取: {total} 页")

    all_arts = []
    for p in range(1, total + 1):
        h = fetch_list_page(p)
        if not h:
            log(f"第{p}页失败，跳过继续下一页")
            continue
        arts = parse_articles(h)
        if not arts:
            log(f"第{p}页无数据，停止")
            break

        all_arts.extend(arts)
        if p % 20 == 0 or p == 1:
            log(f"第{p}页: {len(arts)} 条 (累计{len(all_arts)})")

        # 增量断点：如果该页所有日期 <= DB最新日期，说明新数据已爬完
        if latest_in_db and all(a.get("pub_date","") <= latest_in_db for a in arts):
            log(f"第{p}页所有日期 <= {latest_in_db}，新数据已爬完，停止")
            break

        time.sleep(0.5)

    log(f"列表共 {len(all_arts)} 条")

    saved, skipped = 0, 0
    for art in all_arts:
        if art["title"] in existing:
            skipped += 1
            continue
        try:
            content, atts = fetch_detail(art["url"])
        except Exception as e:
            log(f"详情失败: {art['title'][:30]} {e}")
            content, atts = "", []
        body = ""
        if content:
            body = "<p>" + content.replace("\\n", "</p><p>") + "</p>"
        for a in atts:
            body += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'
        try:
            db.execute("INSERT INTO gov_raw (title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                       (art["title"], body, art["url"], art["pub_date"], SITE_NAME))
            db.commit()
            saved += 1
        except Exception as e:
            log(f"DB写入失败: {art['title'][:30]} {e}")
        if saved % 100 == 0 and saved > 0:
            log(f"进度: 已存 {saved} 条, 跳过 {skipped} 条")
        time.sleep(0.3)
    db.close()
    log(f"完成! 新增 {saved} 条, 跳过 {skipped} 条")

if __name__ == "__main__":
    main()
