#!/usr/bin/env python3
"""
海城市-环评公示 (www.haicheng.gov.cn)
==================================
CMS: 鞍山政府网站 CMS
列表: glist.html (page1) / glistN.html (page N+1)
详情: /html/HCS/YYYYMM/...html, div.hwq-info-article-center
"""
import re, sys, os, time
from datetime import datetime, timedelta, timezone
import requests, urllib3
from bs4 import BeautifulSoup
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME  = "海城市-环评公示"
BASE_URL   = "http://www.haicheng.gov.cn/hcs/zwgkzdgz/zdlyxxgk/wrfz/hpgs/"
DETAIL_DOMAIN = "http://www.haicheng.gov.cn"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def fetch_list_page(page):
    if page == 1:
        url = BASE_URL + "glist.html"
    else:
        url = BASE_URL + "glist{}.html".format(page - 1)  # page 2 = glist1.html
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("  ! page {} fail: {}".format(page, str(e)[:60]))
        return None


def get_page_info(html):
    """Get total pages and current page from pagination text"""
    m = re.search(r'当前第\s*(\d+)\s*/\s*(\d+)\s*页', html)
    if m:
        return int(m.group(1)), int(m.group(2))
    return 1, 1


def parse_list(html):
    """Parse list items - date format is MM-DD, need to derive year from detail"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    tab = soup.find("div", class_="hwq-tab")
    if not tab:
        return items
    for li in tab.find_all("li"):
        time_div = li.find("div", class_="time")
        name_div = li.find("div", class_="name")
        a = name_div.find("a") if name_div else None
        if not time_div or not a:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        mmdd = time_div.get_text(strip=True)  # "MM-DD"
        items.append({"title": title, "url": href, "mmdd": mmdd})
    return items


def fetch_detail(item):
    """Fetch detail page, extract full date and content"""
    try:
        r = requests.get(item["url"], headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")

        # Extract date from detail page
        pub_date = ""
        time_div = soup.find("div", class_="hwq-info-article-time")
        if time_div:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', time_div.get_text())
            if m:
                pub_date = m.group(1)

        # If date not found, derive from URL path /YYYYMM/
        if not pub_date:
            m = re.search(r'/HCS/(\d{4})(\d{2})/', item["url"])
            if m:
                # Use the MM-DD from list with year from URL
                pub_date = "{}-{}-{}".format(m.group(1), item["mmdd"][:2], item["mmdd"][3:5])

        # Extract content
        content = ""
        center = soup.find("div", class_="hwq-info-article-center")
        if center:
            # Remove scripts and styles
            for t in center.find_all(["script", "style"]):
                t.decompose()
            content = str(center)

        item["pub_date"] = pub_date
        item["content"] = content
        return pub_date, len(content)
    except Exception as e:
        print("  ! detail fail: {}".format(str(e)[:60]))
        item["pub_date"] = ""
        item["content"] = ""
        return "", 0


def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "环评公示",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    db.commit()
    db.close()
    return ok, fail


def main():
    print("=== {} ===".format(SITE_NAME))
    print("日期阈值: {}".format(CUTOFF))
    this_year = str(datetime.now(timezone.utc).year)

    html = fetch_list_page(1)
    if not html:
        print("无法获取首页")
        return

    cur_page, total_pages = get_page_info(html)
    print("总页数: {}, 当前: {}".format(total_pages, cur_page))

    # 前5页
    all_items = []
    pages_to_fetch = min(total_pages, 5)
    for page in range(1, pages_to_fetch + 1):
        h = html if page == 1 else fetch_list_page(page)
        if not h:
            break
        items = parse_list(h)
        if not items:
            print("  [Page {}] empty".format(page))
            break

        # No full date on list page - we check after fetching detail
        for item in items:
            # Dedup by URL
            if any(i["url"] == item["url"] for i in all_items):
                continue
            all_items.append(item)

        cur, total = get_page_info(h)
        print("  [Page {}] {} items (total pages: {})".format(page, len(items), total))

    print("\n共{}条待抓取详情".format(len(all_items)))

    if not all_items:
        print("无需抓取")
        return

    # Fetch details
    new_count = 0
    skip_count = 0
    for i, item in enumerate(all_items):
        print("  [{}/{}] {}... ".format(i+1, len(all_items), item["title"][:40]),
              end="", flush=True)
        pub_date, length = fetch_detail(item)
        if pub_date < CUTOFF:
            skip_count += 1
            print("超期({}) {}B".format(pub_date, length))
        else:
            new_count += 1
            print("{} {}B".format(pub_date, length))

    # Filter out items before cutoff
    valid_items = [it for it in all_items if it.get("pub_date", "") >= CUTOFF]
    print("\n阈值内: {}, 超期跳过: {}".format(len(valid_items), skip_count))

    if not valid_items:
        print("无有效记录")
        return

    ok, fail = to_db(valid_items)
    print("\n=== 完成 ===")
    print("新增: {}, 跳过: {}".format(ok, fail))


if __name__ == "__main__":
    main()
