#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
洋浦经济开发区 - 公示公告 爬虫
CMS: UCAP (统一搜索平台)
WAF: 无
列表: /common/search/b22e1388d017400e9218980187b93732?sort=publishedTime&_isAgg=false&_pageSize=12&_template=yangpu&_channelName=&page={n}
分页: 12条/页, 682页
详情: /yangpu/0400/{YYYYMM}/{uuid}.shtml
详情标题: <title> (去掉 "_公示公告_洋浦经济开发区" 后缀)
详情正文: div.con_cen > ucapcontent > p
详情日期: 发布日期：YYYY-MM-DD

用法:
  python3 crawl_yangpu_gg.py
  python3 crawl_yangpu_gg.py --pages 10
"""

import re
import sys
import time
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://yangpu.hainan.gov.cn"
LIST_URL = (BASE_URL + "/common/search/b22e1388d017400e9218980187b93732"
            "?sort=publishedTime&_isAgg=false&_pageSize=12&_template=yangpu"
            "&_channelName=&page=%d")
DB_PATH = "/root/search.db"
SITE_NAME = "洋浦经济开发区-公示公告"
CATEGORY = GROUP = "公示公告"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
MAX_PAGES = 10
TIMEOUT = 30


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if attempt == 2:
                print("  [WARN] 获取失败 (%d/3): %s - %s" % (attempt + 1, url, e), file=sys.stderr)
                return ""
            time.sleep(2)


def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for list_div in soup.find_all("div", class_="list_div"):
        title_div = list_div.find("div", class_="list-right_title")
        if not title_div:
            continue
        a_tag = title_div.find("a")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get_text(strip=True)
        if not href or not title:
            continue
        full_url = urljoin(BASE_URL, href)

        # 日期：表格中的 "发布时间：YYYY-MM-DD"
        date = ""
        td = list_div.find("td", align="left")
        if td:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", td.get_text())
            if m:
                date = m.group(1)

        items.append((full_url, title, date))
    return items


def extract_detail(html):
    soup = BeautifulSoup(html, "html.parser")

    # 标题：<title> 去掉后缀
    title = ""
    title_tag = soup.find("title")
    if title_tag and title_tag.string:
        title = title_tag.string.strip()
        title = re.sub(r"_公示公告_洋浦经济开发区$", "", title).strip()

    # 日期：发布日期：YYYY-MM-DD
    date = ""
    m = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        date = m.group(1)

    # 正文：div.con_cen > ucapcontent > p
    content = ""
    con_cen = soup.find("div", class_="con_cen")
    if con_cen:
        # Try ucapcontent first, then direct p tags
        ucap = con_cen.find("ucapcontent")
        container = ucap if ucap else con_cen
        parts = []
        for p in container.find_all("p"):
            text = p.get_text(strip=True)
            if text:
                parts.append(text)
        content = "\n\n".join(parts)

    # 附件
    attachments = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if ".pdf" in href.lower() or ".doc" in href.lower() or ".xls" in href.lower():
            name = a.get_text(strip=True) or href.split("/")[-1]
            attachments.append({"name": name, "url": urljoin(BASE_URL, href)})

    return title, date, content, attachments


def push_to_db(conn, items):
    saved = 0
    skipped = 0
    for url, title, date, content, attachments in items:
        import json
        attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
        summary = content[:200].replace("\n", " ") if content else title
        if not content:
            if attachments:
                content = "\n".join('<p><a href="%s">%s</a></p>' % (a["url"], a["name"]) for a in attachments)
            else:
                content = title
        try:
            cur = conn.execute(
                "INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_yangpu_gg.py')",
                (url, title, SITE_NAME, date, content, date, summary, attachments_json)
            )
            if cur.rowcount > 0:
                saved += 1
            else:
                skipped += 1
        except Exception as e:
            print("  [ERR] DB: %s - %s" % (url, e), file=sys.stderr)
            skipped += 1
    return saved, skipped


def main():
    pages = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
            break

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    all_items = []

    print("洋浦经济开发区-公示公告: 爬取 %d 页" % pages)

    for idx in range(1, pages + 1):
        list_url = LIST_URL % idx
        print("  [列表页 %d/%d] %s" % (idx, pages, list_url[:80]))
        html = fetch(list_url)
        if not html:
            continue

        items = extract_list_items(html)
        if not items:
            print("    -> 无数据，停止")
            break

        print("    -> 找到 %d 条" % len(items))

        for item_url, item_title, item_date in items:
            time.sleep(0.3)
            print("    [%d] %s..." % (len(all_items) + 1, item_title[:40]), end=" ", flush=True)
            detail_html = fetch(item_url)
            if not detail_html:
                print("fail")
                continue

            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date

            all_items.append((item_url, title, date, content, attachments))
            status = "ok" if content else "empty"
            attach_msg = ""
            if attachments:
                attach_msg = " +%d附" % len(attachments)
            print("(%s, %d字%s)" % (status, len(content), attach_msg))

    if all_items:
        saved, skipped = push_to_db(conn, all_items)
        conn.commit()
        print("\n完成! 新增: %d, 跳过: %d, 总共: %d" % (saved, skipped, len(all_items)))

    conn.close()
    return 0


if __name__ == "__main__":
    sys.exit(main())
