#!/usr/bin/env python3
"""
crawl_xingbin_hjyp.py -- Xingbin district EIA approval crawler (TRS)
"""

import sys, re, time, os
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

def extract_plain_text(html_text):
    if not html_text:
        return ""
    soup = BeautifulSoup(html_text, "html.parser")
    for tag in soup(["script", "style"]):
        tag.decompose()
    text = soup.get_text(separator=" ", strip=True)
    text = re.sub(r"\s+", " ", text)
    return text[:500]

LIST_BASE = "http://www.xingbin.gov.cn/zfxxgk_1/fdzdgknr/zdlyxxgk/shgysyjsly/hjbhly/jsxmhjyxpjsp/index"
LIST_EXT = ".shtml"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "兴宾区-建设项目环境影响评价审批"
TOTAL_PAGES = 3
MAX_PAGES = 3

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            resp = session.get(url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return resp.text
        except Exception:
            if attempt < max_retries - 1:
                time.sleep(2)
    return None


def parse_list_page(html_text):
    records = []
    soup = BeautifulSoup(html_text, 'html.parser')
    ul = soup.find('ul', class_=lambda c: c and 'more-list' in str(c))
    if not ul:
        return records

    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not href or not title:
            continue

        if href.startswith('./'):
            href = 'http://www.xingbin.gov.cn/zfxxgk_1/fdzdgknr/zdlyxxgk/shgysyjsly/hjbhly/jsxmhjyxpjsp/' + href[2:]
        elif href.startswith('/'):
            href = 'http://www.xingbin.gov.cn' + href
        elif not href.startswith('http'):
            href = 'http://www.xingbin.gov.cn/zfxxgk_1/fdzdgknr/zdlyxxgk/shgysyjsly/hjbhly/jsxmhjyxpjsp/' + href

        date_span = li.find('span')
        date_str = date_span.get_text(strip=True) if date_span else ''

        records.append({
            "title": title,
            "url": href,
            "date": date_str,
        })

    return records


def fetch_detail(url):
    html_text = fetch(url)
    if not html_text:
        return "", ""
    soup = BeautifulSoup(html_text, 'html.parser')

    meta_date = soup.find('meta', attrs={"name": "PubDate"})
    date_str = ''
    if meta_date:
        raw_date = meta_date.get('content', '')
        date_str = raw_date[:10] if raw_date else ''

    content_div = soup.find('div', class_=lambda c: c and 'TRS_UEDITOR' in str(c))
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'trs_editor_view' in str(c))
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'article-con' in str(c))

    content_html = ""
    if content_div:
        content_html = str(content_div)

    return content_html, date_str


def store_item(title, url, content, date_str):
    import sqlite3
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        summary = extract_plain_text(content)
        sql = "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, source_url, summary, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
        c.execute(sql,
                  (SITE_NAME, title, url, date_str, content, url, summary, "hjyp"))
        if c.rowcount > 0:
            stats["new"] += 1
        else:
            stats["skip"] += 1
        conn.commit()
        conn.close()
    except Exception as e:
        stats["errors"] += 1


def crawl(full=False, test_n=None):
    print("=" * 60)
    print("Xingbin EIA approval crawler")
    print("=" * 60)

    page_count = min(TOTAL_PAGES, MAX_PAGES)
    print(f"Total pages: {page_count}")

    seen = 0
    for page in range(page_count):
        if page == 0:
            url = LIST_BASE + LIST_EXT
        else:
            url = f"{LIST_BASE}_{page}{LIST_EXT}"

        print(f"\n--- Page {page+1}/{page_count}: {url}")
        html_text = fetch(url)
        if not html_text:
            print("  [ERROR] Fetch failed")
            stats["errors"] += 1
            continue

        records = parse_list_page(html_text)
        if not records:
            print("  [ERROR] No records found")
            break

        print(f"  Found {len(records)} items")

        for idx, item in enumerate(records):
            if item['date'] and item['date'] < THREE_YEARS_AGO:
                print(f"  [SKIP] Too old: {item['date']} {item['title'][:30]}")
                stats["skip"] += 1
                continue

            print(f"  [{idx+1}] {item['title'][:40]} | {item['date']}")

            if test_n and seen >= test_n:
                print(f"\n  Test mode: {test_n} items done")
                return

            content, detail_date = fetch_detail(item['url'])
            date_str = detail_date or item['date']
            store_item(item['title'], item['url'], content, date_str)
            seen += 1
            time.sleep(0.5)

    print(f"\nDone! New: {stats['new']}, Skip: {stats['skip']}, Errors: {stats['errors']}")


if __name__ == '__main__':
    full = '--full' in sys.argv
    test_n = None
    for arg in sys.argv:
        if arg.startswith('--test='):
            test_n = int(arg.split('=')[1])
    crawl(full=full, test_n=test_n)
