#!/usr/bin/env python3
"""
山东省人民政府 - 环评许可审批 爬虫
CMS: Hanweb (山东省政府门户 xxgk 模块)
列表: POST /module/xxgk/search.jsp?infotypeId=SD341206
      body: infotypeId=SD341206&jdid=410&area=&divid=div4&currpage=N&sortfield=top:0,createdatetime:0,orderid:0
      20条/页, 共1180条/59页
详情: /art/{y}/{m}/{d}/art_305272_{id}.html?xxgkhide=1
      标题: 页面 title 提取 (去掉"山东省人民政府 环评许可审批 "前缀)
      正文: div.wip_art_con
用法: python3 crawl_shandong_hpjxsp.py [--pages=N]  (默认 5 页)
"""
import os, sys, re, time, sqlite3, requests
from datetime import date
from urllib.parse import urljoin
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "山东省政府-环评许可审批"
BASE = "http://www.shandong.gov.cn"
SEARCH_URL = "http://www.shandong.gov.cn/module/xxgk/search.jsp?infotypeId=SD341206"
INFOTYPE_ID = "SD341206"
JDID = "410"
CUTOFF_DATE = date(2023, 8, 15)
MAX_PAGES_DEFAULT = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
    "Referer": "http://www.shandong.gov.cn/col/col305272/index.html?vc_xxgkarea=113700000045022274&number=SD341206",
    "X-Requested-With": "XMLHttpRequest",
    "Content-Type": "application/x-www-form-urlencoded",
}


def fetch_list_page(pn):
    """POST search.jsp page N. Returns list of dicts {url, title, pub_date}."""
    body = {
        "infotypeId": INFOTYPE_ID,
        "jdid": JDID,
        "area": "",
        "divid": "div4",
        "currpage": str(pn),
        "vc_title": "",
        "vc_number": "",
        "vc_filenumber": "",
        "vc_all": "",
        "texttype": "",
        "fbtime": "",
    }
    r = requests.post(SEARCH_URL, data=body, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "lxml")
    items = []
    # 列表结构: table > tr.tr_main_value_odd/even > td > a.bt_link
    for tr in soup.find_all("tr", class_=re.compile(r"tr_main_value")):
        a = tr.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if "art_" not in href:
            continue
        full = href if href.startswith("http") else urljoin(BASE, href)
        title = a.get("title", "") or a.get_text(strip=True)
        pub_date = ""
        m = re.search(r"(\d{4}-\d{2}-\d{2})", tr.get_text())
        if m:
            pub_date = m.group(1)
        if full and title:
            items.append({"url": full, "title": title.strip(), "pub_date": pub_date})
    # 兜底: 某些响应可能是 li 结构
    if not items:
        for li in soup.find_all("li"):
            a = li.find("a", href=True)
            if not a or "art_" not in a["href"]:
                continue
            href = a["href"]
            full = href if href.startswith("http") else urljoin(BASE, href)
            title = a.get("title", "") or a.get_text(strip=True)
            pub_date = ""
            b = li.find("b")
            if b:
                m = re.search(r"(\d{4}-\d{2}-\d{2})", b.get_text())
                if m:
                    pub_date = m.group(1)
            if full and title:
                items.append({"url": full, "title": title.strip(), "pub_date": pub_date})
    return items


def fetch_detail(url):
    """Fetch detail page: return (full_title, content_html)."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        # 标题: title 标签去掉前缀
        title = ""
        t = soup.find("title")
        if t:
            raw = t.get_text(strip=True)
            raw = re.sub(r"^山东省人民政府\s*(环评许可审批)?\s*", "", raw)
            title = raw.strip()
        # 正文容器
        content_div = soup.find("div", class_="wip_art_con")
        if not content_div:
            return title, ""
        # 去脚本/样式
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()
        # 链接/图片绝对化
        for a in content_div.find_all("a", href=True):
            href = a["href"]
            if href and not href.startswith("http"):
                a["href"] = urljoin(BASE, href)
        for img in content_div.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("http"):
                img["src"] = urljoin(BASE, src)
        return title, str(content_div)
    except Exception as e:
        print(f"  [detail] {url} ERR {e}", file=sys.stderr)
        return "", ""


def main():
    max_pages = MAX_PAGES_DEFAULT
    args = sys.argv[1:]
    for i, a in enumerate(args):
        if a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
        elif a.startswith("--pages="):
            max_pages = int(a.split("=")[1])

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    all_new = 0
    all_skip = 0
    reached_cutoff = False

    for page in range(1, max_pages + 1):
        items = fetch_list_page(page)
        if not items:
            print(f"  第{page}页: 空 -> 结束", file=sys.stderr)
            break
        print(f"  第{page}页: {len(items)} 条", file=sys.stderr)
        for it in items:
            pub = it["pub_date"]
            if pub and pub < CUTOFF_DATE.strftime("%Y-%m-%d"):
                reached_cutoff = True
                print(f"  已达截断日 {CUTOFF_DATE} (当前 {pub})，停止", file=sys.stderr)
                break
            url = it["url"]
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if c.fetchone():
                all_skip += 1
                continue
            detail_title, content = fetch_detail(url)
            if not content or len(content) < 20:
                print(f"    ! 空正文: {it['title'][:40]}", file=sys.stderr)
                all_skip += 1
                continue
            title = detail_title or it["title"]
            pub_date = pub or ""
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, script_name) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                (SITE_NAME, url, title[:500], content, pub_date[:20], url, "done", "crawl_shandong_hpjxsp.py"),
            )
            if c.rowcount > 0:
                all_new += 1
                print(f"    + {title[:50]} ({pub_date}) body:{len(content)}B", file=sys.stderr)
            else:
                all_skip += 1
            conn.commit()
            time.sleep(0.3)
        if reached_cutoff:
            break

    print(f"\n[{SITE_NAME}] 完成! 新增 {all_new} 条, 跳过 {all_skip} 条", file=sys.stderr)
    if all_new > 0:
        c.execute("""INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
                      SELECT rowid, title, site_name, substr(content,1,500) FROM gov_raw
                      WHERE site_name=? AND rowid NOT IN (SELECT rowid FROM gov_search)""", (SITE_NAME,))
        conn.commit()
    conn.close()


if __name__ == "__main__":
    main()
