#!/usr/bin/env python3
"""
shayang.gov.cn 沙洋县政府 - 通知公告爬虫
==========================================
TRS CMS / jPage API 分页，详情页 requests 直连。

栏目: col5449（通知公告）

用法:
  python3 crawl_shayang.py               # 增量第1页
  python3 crawl_shayang.py 2             # 爬2页
  python3 crawl_shayang.py --full        # 全量4页
"""

import os, re, sys, time, json, sqlite3, html
import requests
from bs4 import BeautifulSoup

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "沙洋县人民政府"
DOMAIN = "www.shayang.gov.cn"
COLUMN_ID = "5449"
WEB_ID = "36"
UNIT_ID = "56733"
CATEGORY_NAME = "通知公告"
GROUP = "湖北-荆门"
INDUSTRY = "政府公告"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

API_URL = "https://www.shayang.gov.cn/module/web/jpage/dataproxy.jsp"
DETAIL_BASE = "https://www.shayang.gov.cn"
TOTAL_PAGES = 4
DEFAULT_PAGES = 1
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch_list_page(page_num):
    """通过jPage API获取列表页"""
    data = {
        "page": str(page_num),
        "webid": WEB_ID,
        "path": "/",
        "columnid": COLUMN_ID,
        "unitid": UNIT_ID,
        "webname": "沙洋县人民政府",
        "permissiontype": "0",
    }
    r = requests.post(API_URL, data=data, headers=HEADERS, timeout=60)
    r.encoding = "utf-8"
    return r.text


def parse_items(xml_text):
    """解析jPage XML返回 (title, url, date)"""
    items = []
    # 提取所有record中的CDATA
    records = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", xml_text, re.DOTALL)
    for rec in records:
        soup = BeautifulSoup(rec, "lxml")
        a_tag = soup.find("a")
        span = soup.find("span")
        if a_tag:
            href = a_tag.get("href", "").strip()
            title = a_tag.get_text(strip=True)
            full_url = href if href.startswith("http") else DETAIL_BASE + href
            date = span.get_text(strip=True) if span else ""
            items.append((title, full_url, date))
    return items


def fetch_detail(url):
    """获取详情页内容"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")

        # 标题
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()

        # 日期
        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            raw = meta_d["content"].strip()
            m = re.search(r"(\d{4})-(\d{2})-(\d{2})", raw)
            if m:
                date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))

        # 正文
        body_html = ""
        summary = ""
        content_div = soup.select_one("#content")
        if content_div:
            for tag in content_div.find_all(["script", "style"]):
                tag.decompose()
            body_html = str(content_div)
            summary = content_div.get_text(strip=True)[:200]

        return title, date, body_html, summary
    except Exception as e:
        print("    detail error: %s" % e, file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, body_html="", summary=""):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title.strip(), page_url, DOMAIN, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, body_html, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0

        if not is_new and body_html:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body_html, summary, row[0]))
                conn.commit()
                is_new = True

        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                _c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    _c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name) VALUES (?,?,?)",
                                (row[0], title.strip(), SITE_NAME))
                    _c2.commit()
                except:
                    pass
                finally:
                    _c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description="shayang 通知公告爬虫")
    parser.add_argument("--full", action="store_true", help="全量4页")
    parser.add_argument("--pages", type=int, default=None, help="页数")
    parser.add_argument("--limit", type=int, default=0, help="每页限制条数(0=不限)")
    parser.add_argument("n", nargs="?", type=int, default=None, help="页数(简写)")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    if args.full:
        pages_to_crawl = TOTAL_PAGES
    elif args.pages is not None:
        pages_to_crawl = args.pages
    elif args.n is not None:
        pages_to_crawl = args.n
    else:
        pages_to_crawl = DEFAULT_PAGES

    pages_to_crawl = min(pages_to_crawl, TOTAL_PAGES)
    print("页数: %d (总共%d页)" % (pages_to_crawl, TOTAL_PAGES))

    for p in range(1, pages_to_crawl + 1):
        try:
            xml = fetch_list_page(p)
            items = parse_items(xml)
            if not items:
                print("  第%d页: 空" % p)
                break
            print("  第%d页: %d条" % (p, len(items)))
            for idx, (title, url, date_from_list) in enumerate(items):
                if args.limit > 0 and idx >= args.limit:
                    print("    limit=%d, 停止" % args.limit)
                    break
                dt_title, dt_date, body, summary = fetch_detail(url)
                final_title = dt_title or title
                final_date = dt_date or date_from_list
                store_record(final_title, url, final_date, body, summary)
                time.sleep(0.3)
        except Exception as e:
            print("  第%d页错误: %s" % (p, e))

    print("\n完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))
