#!/usr/bin/env python3
"""爬虫：随州市人民政府 — 建设项目环评公示
www.suizhou.gov.cn/zwgk/xxgk/shgysyjs/hjbh/hpgs/
"""

import os
import re
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
BASE_URL = "http://www.suizhou.gov.cn/zwgk/xxgk/shgysyjs/hjbh/hpgs/"
SITE_NAME = "随州市建设项目环评公示"
GROUP_NAME = "湖北"
SCRIPT_NAME = os.path.basename(__file__)
MAX_PAGES = 3  # createPageHTML(18, 0, ...) 实际18页352条, 日增量取前3页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def get_page_url(page):
    """第1页=index.shtml, 第N页(N>=2)=index_{N-1}.shtml (0索引)"""
    if page == 1:
        return BASE_URL
    return f"{BASE_URL}index_{page-1}.shtml"

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # 标题
    title_div = soup.select_one(".news-detl-title")
    title = title_div.get_text(strip=True) if title_div else ""

    # 日期 + 来源
    pub_date = ""
    source_text = ""
    sub = soup.select_one(".news-detl-sub")
    if sub:
        text = sub.get_text(strip=True)
        # 兼容 发布日期：/发布时间： 等 (发布+0~2个日期字+冒号)
        m = re.search(r"发布[日日期期]*[：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        m2 = re.search(r"信息[来源来源：:]\s*([^\s<li<]+)", text)
        if m2:
            source_text = m2.group(1).strip()
    if not pub_date:
        m = soup.find("meta", attrs={"name": "PubDate"})
        if m and m.get("content"):
            pm = re.search(r"(\d{4}-\d{2}-\d{2})", m["content"])
            if pm:
                pub_date = pm.group(1)

    # 正文
    con = soup.select_one(".view.TRS_UEDITOR")
    if not con:
        return title, pub_date, source_text, "", ""

    # 附件
    attachments = []
    fj = soup.select_one("#fj")
    if fj:
        for a_tag in fj.find_all("a"):
            href = a_tag.get("href", "")
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
                text = a_tag.get_text(strip=True)
                attachments.append(f"[{text}]({href})")

    # 正文段落
    parts = []
    has_table = bool(con.find("table"))
    for tag in con.find_all(["p", "table"]):
        if tag.name == "p" and tag.find_parent("table"):
            continue
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        text = tag.get_text(strip=True)
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)

    if not parts:
        text = re.sub(r"\n+", "", con.get_text(strip=True))
        if text:
            parts.append(text)

    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""

    # 图片回退
    if not content.strip() and not attachments:
        imgs = con.find_all("img")
        if imgs:
            img_parts = []
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src:
                    full_src = urljoin(detail_url, src)
                    img_parts.append(f"![{alt}]({full_src})")
            if img_parts:
                content = f'<p><a href="{detail_url}">{title}</a></p>\n\n' + "\n\n".join(img_parts)

    if not content.strip() and attachments:
        content = f'<p><a href="{detail_url}">{title}</a></p>\n\n{attachments_str}'

    return title, pub_date, source_text, content, attachments_str

def crawl(pages=None):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    total = 0
    new_total = 0
    skip_total = 0
    max_pages = pages or MAX_PAGES

    for page in range(1, max_pages + 1):
        url = get_page_url(page)
        print(f"  第{page}页: {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}")
            time.sleep(2)
            continue

        soup = BeautifulSoup(r.text, "html.parser")
        ul = soup.select_one(".xxgk-list2 ul")
        if not ul:
            print(f"  [WARN] 未找到列表ul")
            break

        items = []
        for li in ul.find_all("li"):
            a_tag = li.find("a")
            if not a_tag:
                continue
            href = a_tag.get("href", "")
            title = a_tag.get_text(strip=True)
            # 日期在<a>内部的<span>中
            span = a_tag.find("span")
            list_date = span.get_text(strip=True) if span else ""
            # 从title中去掉日期后缀
            if span:
                title = title.replace(span.get_text(strip=True), "").strip()
            if not href:
                continue
            detail_url = href if href.startswith("http") else urljoin(url, href)
            items.append((title, list_date, detail_url))

        if not items:
            print(f"  第{page}页无数据，终止")
            break

        for title, list_date, detail_url in items:
            print(f"    提取: {title[:40]}...")
            full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)
            if not full_title:
                full_title = title
            if not pub_date:
                pub_date = list_date

            date_rank = 0
            if pub_date:
                try:
                    date_rank = 0 - int(pub_date.replace("-", "") + "0000")
                except ValueError:
                    date_rank = 0

            try:
                cur.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (page_url, site_name, title, publish_date, content, summary, attachments, date_rank,
                     group_name, script_name, has_table, status)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
                """, (detail_url, SITE_NAME, full_title, pub_date, content, source_text, attachments_str, date_rank,
                      GROUP_NAME, SCRIPT_NAME, 1 if (content and "<table" in content) else 0, "published"))
                if cur.rowcount > 0:
                    row_id = cur.lastrowid
                    try:
                        # gov_search FTS5 同步 (trigram): rowid 必须与 gov_raw.id 一致
                        import re as _re
                        plain = _re.sub(r"<[^>]+>", " ", content or "")
                        plain = _re.sub(r"\s+", " ", plain).strip()[:200] or full_title
                        cur.execute(
                            "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                            (row_id, full_title, SITE_NAME, plain)
                        )
                    except Exception:
                        pass
                    new_total += 1
                else:
                    skip_total += 1
                total += 1
            except Exception as e:
                print(f"    [DB ERROR] {e}")

            time.sleep(0.3)

        conn.commit()
        print(f"  第{page}页完成，累计{total}条")
        time.sleep(0.5)

    conn.close()
    print(f"\n总计: {total}条 | 新增: {new_total} | 跳过: {skip_total}")

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description=f"{SITE_NAME}爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="爬取页数(默认%d)" % MAX_PAGES)
    args = parser.parse_args()
    crawl(args.pages)
