#!/usr/bin/env python3
"""
广西百色田阳区-部门公告
====================
CMS: 广西政府网站通用 CMS
列表: ul.ty-common-ul > li > a[title] + span (日期)
分页: index.shtml (page 1), index_N.shtml
详情: div.article > h1 + div.article-inf-left + div.article-con
"""
import re, sys, os, time
from datetime import datetime, timedelta, timezone
import requests, urllib3
from bs4 import BeautifulSoup
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "田阳区-部门公告"
BASE_URL = "http://www.gxty.gov.cn/zfxxgk/fdzdgknr_1_1_1/gsgg_1/bmgg/"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def fetch_page(page):
    url = BASE_URL + "index.shtml" if page == 1 else BASE_URL + "index_{}.shtml".format(page)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("  ! page {} fail: {}".format(page, str(e)[:60]))
        return None


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.select_one("ul.ty-common-ul")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        pub_date = span.get_text(strip=True)
        if href.startswith("./"):
            href = BASE_URL + href[2:]
        elif not href.startswith("http"):
            href = BASE_URL + href
        items.append({"title": title, "url": href, "pub_date": pub_date})
    return items


def get_total_pages(html):
    m = re.search(r"createPageHTML\((\d+)", html)
    if m:
        return int(m.group(1))
    return 1


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        con = soup.select_one("div.article-con")
        if con:
            # Strip scripts and styles
            for t in con.find_all(["script", "style"]):
                t.decompose()
            return str(con)
        return ""
    except Exception as e:
        print("  ! detail fail: {}".format(str(e)[:60]))
        return ""


def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "部门公告",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    db.commit()
    db.close()
    return ok, fail


def main():
    print("=== {} ===".format(SITE_NAME))
    print("日期阈值: {}".format(CUTOFF))

    html = fetch_page(1)
    if not html:
        print("无法获取首页")
        return

    total_pages = get_total_pages(html)
    print("总页数: {}".format(total_pages))

    # 前5页 (增量日跑5页足矣)
    all_items = []
    pages_to_fetch = min(total_pages, 5)
    for page in range(1, pages_to_fetch + 1):
        h = html if page == 1 else fetch_page(page)
        if not h:
            break
        items = parse_list(h)
        hit_old = False
        for item in items:
            if item["pub_date"] < CUTOFF:
                hit_old = True
                continue
            # Dedup by URL
            if any(i["url"] == item["url"] for i in all_items):
                continue
            all_items.append(item)
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        print("  [Page {}] {} items ({} ~ {}), new: {}".format(
            page, len(items), f, l,
            len([x for x in items if x["pub_date"] >= CUTOFF])))
        if hit_old:
            print("  检测到超期数据，停止翻页")
            break

    print("\n共{}条待抓取详情".format(len(all_items)))

    if not all_items:
        print("无需抓取")
        return

    for i, item in enumerate(all_items):
        print("  [{}/{}] {}... ".format(i+1, len(all_items), item["title"][:40]),
              end="", flush=True)
        content = fetch_detail(item["url"])
        item["content"] = content
        print("{}B".format(len(content)))

    ok, fail = to_db(all_items)
    print("\n=== 完成 ===")
    print("新增: {}, 跳过: {}".format(ok, fail))


if __name__ == "__main__":
    main()
