#!/usr/bin/env python3
"""
青阳县人民政府 - 建设项目环境影响评价审批
CMS: 自定义政府CMS (OpennessContent + AJAX列表)
列表AJAX: /OpennessTarget/672/100097/page_N.html (N=1~29)
详情: /OpennessContent/show/ID.html
总计: 567条, 29页, 20条/页
"""

import os

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
import re
import sys
import time
import requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "ahqy.gov.cn-环评审批"
BASE_URL = "https://www.ahqy.gov.cn"
API_TPL = "https://www.ahqy.gov.cn/OpennessTarget/672/100097/page_{}.html"
TOTAL_PAGES = 29
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
print(f"Cutoff date: {CUTOFF}")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERR] fetch failed: {url} - {e}")
        return None


def parse_list(html):
    """
    Parse the AJAX list HTML and return list of (url, title, date_str).
    Items from table rows with OpennessContent/show links and cwrq dates.
    """
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    rows = soup.select("table tr")
    for tr in rows:
        tds = tr.find_all("td")
        if len(tds) < 3:
            continue
        # First td: 序号, ignore
        # Second td: title with link
        a_tag = tds[1].find("a") if len(tds) > 1 else None
        if not a_tag or not a_tag.get("href"):
            continue
        href = a_tag["href"].strip()
        title = a_tag.get_text(strip=True)
        if not href.startswith("http"):
            href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
        # Third td: date
        date_str = ""
        if len(tds) > 2:
            date_str = tds[2].get_text(strip=True)
            # Validate date format
            if not re.match(r"\d{4}-\d{2}-\d{2}", date_str):
                date_str = ""
        if title and href:
            items.append((href, title, date_str))
    return items


def parse_detail(html, url):
    """Parse detail page, return (title, publish_date, content_html)"""
    soup = BeautifulSoup(html, 'html.parser')

    # Title from meta tag
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # Fallback to div.u-title
    if not title:
        title_el = soup.select_one("div.u-title")
        if title_el:
            title = title_el.get_text(strip=True)

    if not title:
        print(f"  [ERR] no title for {url}")
        return None, None, None

    # Date from meta PubDate
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
        if m:
            pub_date = m.group(1)

    # Content from div#zoom
    content_div = soup.select_one("div#zoom")
    if not content_div:
        print(f"  [WARN] no content div for {url}, saving title only")
        content_html = ""
    else:
        content_html = str(content_div).strip()

    return title, pub_date, content_html


def main():
    import sqlite3

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    cur = conn.cursor()

    total_new = 0
    total_skip = 0
    total_dup = 0

    for page_num in range(1, min(TOTAL_PAGES, _MAX_PG or TOTAL_PAGES) + 1):
        page_url = API_TPL.format(page_num)

        print(f"\n--- Page {page_num}/{TOTAL_PAGES} ---")
        html = fetch(page_url)
        if not html:
            print(f"  [SKIP] page {page_num} fetch failed")
            continue

        items = parse_list(html)
        if not items:
            print(f"  [SKIP] page {page_num} no items")
            break

        print(f"  Found {len(items)} items")

        page_new = 0
        page_skip = 0
        for url, list_title, list_date in items:
            # Quick filter by list date - skip if clearly too old
            if list_date and list_date < CUTOFF:
                page_skip += 1
                total_skip += 1
                continue

            # Fetch detail
            detail_html = fetch(url)
            if not detail_html:
                print(f"  [SKIP] detail fetch: {list_title[:40]}")
                continue

            title, date_str, content = parse_detail(detail_html, url)
            if not title:
                continue

            # Check cutoff (re-check with authoritative date from detail page)
            if date_str and date_str < CUTOFF:
                page_skip += 1
                total_skip += 1
                print(f"  跳过(超3年): {date_str} {title[:40]}")
                continue

            date_rank = 0
            if date_str:
                date_rank = int(date_str.replace("-", ""))

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, title, url, date_str, content, date_rank, 'hjxx')
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
                    if page_new % 10 == 0:
                        conn.commit()
                else:
                    total_dup += 1
            except Exception as e:
                print(f"  [ERR] insert: {e}")

            time.sleep(0.3)

        conn.commit()
        print(f"  Page {page_num}: +{page_new} new, {page_skip} skipped (total +{total_new} new, {total_skip} skip)")
        time.sleep(0.3)

    print(f"\n{'='*50}")
    print(f"Final: {total_new} new, {total_dup} dup, {total_skip} skipped")

    if total_new > 0:
        print("\nRebuilding FTS index...")
        try:
            cur.execute(f"DELETE FROM gov_search WHERE site_name='{SITE_NAME}'")
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("  FTS index rebuilt.")
        except Exception as e:
            print(f"  [ERR] FTS rebuild: {e}")

    conn.close()
    print("Done.")


if __name__ == "__main__":
    main()
