#!/usr/bin/env python3
"""无棣县人民政府-通知公告-WDX017 (search.jsp API版)
Hanweb xxgk search.jsp API
POST /module/xxgk/search.jsp?standardXxgk=1&infotypeId=WDX017&...
参数: pageindex=1..N, pagesize=20
详情: meta ArticleTitle + meta PubDate + div.xxgk_content / div.TRS_UEDITOR
"""

import requests, re, sys, os, time, sqlite3
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE_URL = "http://www.wudi.gov.cn"
SITE_NAME = "无棣县人民政府-通知公告-WDX017"
GROUP = "山东滨州"
INDUSTRY = "政府公告"
SCRIPT_NAME = "crawl_wudi_wdx017.py"
MAX_PAGES = 1

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest",
}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def fetch_list(page):
    """通过 search.jsp API 获取一页列表"""
    # 先拿 session cookie
    session.get(f"{BASE_URL}/col/col225684/index.html?number=WDX017&vc_xxgkarea=11371623004384523M-1", timeout=10)
    
    search_url = f"{BASE_URL}/module/xxgk/search.jsp?standardXxgk=1&infotypeId=WDX017&vc_title=&vc_number=&area="
    data = {
        "pageindex": page,
        "pagesize": 20,
        "divid": "div4",
        "infotypeId": "WDX017",
        "area": "",
        "vc_title": "",
        "vc_number": "",
        "standardXxgk": "1"
    }
    for retry in range(3):
        try:
            r = session.post(search_url, data=data, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if retry < 2:
                time.sleep(2)
    return None


def parse_list(html):
    """从API返回的HTML提取列表项"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.find_all("li"):
        a = li.find("a", title=True)
        b = li.find("b")
        if a and b:
            href = a.get("href", "")
            title = a.get("title", "").strip() or a.get_text(strip=True)
            date = b.get_text(strip=True)
            full_url = href if href.startswith("http") else BASE_URL + href
            items.append((title, full_url, date))
    return items


def fetch(url):
    for i in range(3):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if i < 2:
                time.sleep(2)
    return None


def parse_detail(html):
    """提取标题、日期、正文"""
    soup = BeautifulSoup(html, "html.parser")
    
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True).replace("无棣县人民政府 通知公告 ", "").strip()
    
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html[:2000])
        if m:
            date_str = m.group(1)
    
    content = ""
    body = (
        soup.find("div", class_="news-content")
        or soup.find("div", class_="article-content")
        or soup.find("div", class_="TRS_UEDITOR")
        or soup.find("div", class_="content")
        or soup.find("div", id="zoom")
    )
    if body:
        content = str(body)
        content = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', content, flags=re.DOTALL|re.I)
        content = content.strip()
    if not content:
        for div in soup.find_all("div"):
            txt = div.get_text(strip=True)
            if len(txt) > 200:
                paras = [p.get_text(strip=True) for p in div.find_all("p") if p.get_text(strip=True)]
                if paras:
                    content = "\n\n".join(paras)
                    break
    
    return title, date_str, content


def crawl():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cursor = conn.cursor()
    
    new_count = 0
    skip_count = 0
    
    for pg in range(1, MAX_PAGES + 1):
        html = fetch_list(pg)
        if not html:
            print(f"[{SCRIPT_NAME}] Page {pg}: fetch failed")
            break
        
        items = parse_list(html)
        if not items:
            print(f"[{SCRIPT_NAME}] Page {pg}: empty, done")
            break
        
        print(f"[{SCRIPT_NAME}] Page {pg}: {len(items)} items")
        
        for title, item_url, date_str in items:
            existing = cursor.execute(
                "SELECT id FROM gov_raw WHERE page_url = ?", (item_url,)
            ).fetchone()
            if existing:
                skip_count += 1
                continue
            
            detail_html = fetch(item_url)
            if not detail_html:
                skip_count += 1
                continue
            
            real_title, real_date, content = parse_detail(detail_html)
            if not real_title:
                real_title = title
            if not real_date:
                real_date = date_str
            if not content or len(content.strip()) < 50:
                content = "正文为空"
            
            try:
                cursor.execute("""
                    INSERT INTO gov_raw (source_url, page_url, title, publish_date, site_name, content, group_name, industry)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                """, (item_url, item_url, real_title, real_date, SITE_NAME, content, GROUP, INDUSTRY))
                conn.commit()
                new_count += 1
            except Exception as e:
                conn.rollback()
                skip_count += 1
            
            time.sleep(0.5)
        
        time.sleep(0.5)
    
    conn.close()
    print(f"\n[{SCRIPT_NAME}] Done. New: {new_count}, Skipped: {skip_count}")


if __name__ == "__main__":
    crawl()
