#!/usr/bin/env python3
"""通城县人民政府 - 生态环境

通城县政府信息公开平台 > 法定主动公开内容 > 公益事业建设 > 生态环境

List: ul.info-list > li > a href="detail_url"
Pagination: createPageHTML(25, 0, "index", "shtml", "pages clearfix", 518)
  P1: http://www.zgtc.gov.cn/xxgk/xxgkml/gysyjs/hjbh/
  P2+: http://www.zgtc.gov.cn/xxgk/xxgkml/gysyjs/hjbh/index_{N-1}.shtml
  20 items/page, max 25 pages
Detail: /xxgk/xxgkml/gysyjs/hjbh/YYYYMM/tYYYYMMDD_ID.shtml
  Title: Meta ArticleTitle (complete)
  Date: Meta PubDate (YYYY-MM-DD HH:MM)
  Content: div.mlxl-content-box-two > div.view.TRS_UEDITOR > p paragraphs + tables
  Attachments: div.mlxl-content-box-two > div#appendix or a[href$=.pdf|.doc|...]
"""
import json, re, sys, time, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "通城县人民政府-生态环境"
CATEGORY = "湖北"
BASE_URL = "http://www.zgtc.gov.cn/xxgk/xxgkml/gysyjs/hjbh"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

INSERT_SQL = """INSERT OR IGNORE INTO gov_raw
    (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"""


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def init_session():
    """Create a session with proper cookies for WAF bypass"""
    session = requests.Session()
    session.headers.update(HEADERS)
    # Visit home page to get cookies
    try:
        session.get("http://www.zgtc.gov.cn/", timeout=30)
        session.get(BASE_URL + "/", timeout=30)
    except Exception as e:
        print(f"[WARN] Session init: {e}")
    return session


def extract_list_page(session, page_num):
    """Extract article items from list page"""
    if page_num == 1:
        url = BASE_URL + "/"
    else:
        url = f"{BASE_URL}/index_{page_num - 1}.shtml"

    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            print(f"[WARN] Page {page_num}: HTTP {r.status_code}")
            return []
    except Exception as e:
        print(f"[ERROR] Page {page_num}: {e}")
        return []

    soup = BeautifulSoup(r.text, 'html.parser')
    ul = soup.find('ul', class_=re.compile(r'info-list', re.I))
    if not ul:
        print(f"[WARN] Page {page_num}: no info-list found")
        return []

    items = []
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        if not href:
            continue
        # Build full URL
        detail_url = urljoin(url, href)

        # Title from a text (check if complete)
        title = a.get_text(strip=True)
        if not title:
            continue

        # Title may be truncated on list page ('...') - will use detail page title

        # Date
        date_span = li.find('span')
        date_str = ""
        if date_span:
            date_str = date_span.get_text(strip=True)
        if not date_str:
            dates = re.findall(r'20\d{2}[-/]\d{1,2}[-/]\d{1,2}', li.get_text())
            if dates:
                date_str = dates[0]

        items.append({"title": title, "url": detail_url, "date": date_str})

    print(f"  Page {page_num}: {len(items)} items")
    return items


def extract_detail(session, url, list_title):
    """Extract full detail from article page"""
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None
    except Exception as e:
        print(f"    [ERROR] {url[:80]}: {e}")
        return None

    soup = BeautifulSoup(r.text, 'html.parser')

    # Title: Meta ArticleTitle (most complete)
    title = ""
    meta_title = soup.find('meta', attrs={'name': re.compile(r'ArticleTitle', re.I)})
    if meta_title and meta_title.get('content', '').strip():
        title = meta_title['content'].strip()
    else:
        # Fallback to <title> minus suffix
        t_tag = soup.find('title')
        if t_tag:
            t = t_tag.get_text(strip=True)
            t = re.sub(r'[-_|\s]*通城县政府网\s*$', '', t).strip()
            if t:
                title = t

    # Date
    date_str = ""
    meta_date = soup.find('meta', attrs={'name': re.compile(r'PubDate', re.I)})
    if meta_date and meta_date.get('content', '').strip():
        date_str = meta_date['content'].strip()

    # Content: from div.mlxl-content-box-two > div.view.TRS_UEDITOR
    content = ""
    attachments = []

    box2 = soup.find('div', class_=re.compile(r'mlxl-content-box-two', re.I))
    if not box2:
        # Fallback: look for content div
        box2 = soup.find('div', class_=re.compile(r'mlxl-content', re.I))

    if box2:
        # Find the actual editor/content div
        editor = box2.find('div', class_=re.compile(r'view|TRS|TRS_UEDITOR|trs_paper', re.I))
        if not editor:
            editor = box2

        # Extract paragraphs
        content_parts = []
        for child in editor.find_all(['p', 'table'], recursive=True):
            if child.name == 'p':
                txt = child.get_text(" ", strip=True)
                if txt:
                    content_parts.append(txt)
            elif child.name == 'table':
                md = table_to_markdown(child)
                if md:
                    content_parts.append(md)

        # If no <p> or <table>, fallback to all text
        if not content_parts:
            txt = editor.get_text(" ", strip=True)
            if txt:
                content_parts.append(txt)

        content = "\n\n".join(content_parts)

        # Attachments
        for a_tag in box2.find_all('a'):
            href = a_tag.get('href', '')
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|rar|zip)$', href, re.I):
                att_name = a_tag.get_text(strip=True) or "附件"
                att_url = urljoin(url, href)
                attachments.append({"name": att_name, "url": att_url})

    # Check PDF-only content (empty text content)
    if not content or len(content.strip()) < 20:
        title_from_content = title or list_title
        content = f'<p><a href="{url}">{title_from_content}</a></p>'
        if attachments:
            for att in attachments:
                content += f"\n[{att['name']}]({att['url']})"

    return {
        "title": title or list_title,
        "date": date_str,
        "content": content,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else ""
    }


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def insert_to_db(conn, cur, rec):
    try:
        summary = (rec.get("content") or "")[:200]
        cur.execute(INSERT_SQL, (
            SITE_NAME,
            rec.get("source_url", ""),
            rec.get("page_url"),
            rec.get("title"),
            rec.get("publish_date", ""),
            summary,
            rec.get("content", ""),
            CATEGORY,
            rec.get("attachments", ""),
            ""
        ))
        conn.commit()
        return True
    except Exception as e:
        print(f"    [DB ERROR] {e}")
        conn.rollback()
        return False


def crawl(test_mode=False, max_pages=MAX_PAGES):
    session = init_session()
    conn = init_db()
    cur = conn.cursor()

    total = 0
    errors = 0
    total_listed = 0

    for pn in range(1, max_pages + 1):
        items = extract_list_page(session, pn)
        if not items:
            print(f"  [STOP] Page {pn} is empty, stopping")
            break

        total_listed += len(items)

        for item in items:
            # Check if already exists
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
            if cur.fetchone():
                print(f"    [SKIP] already exists: {item['title'][:50]}...")
                continue

            # Extract detail
            detail = extract_detail(session, item["url"], item["title"])
            if not detail:
                errors += 1
                print(f"    [ERROR] detail unreachable: {item['url'][:60]}")
                continue

            rec = {
                "page_url": item["url"],
                "source_url": BASE_URL + "/",
                "title": detail["title"],
                "publish_date": detail["date"],
                "content": detail["content"],
                "attachments": detail["attachments"],
            }

            if insert_to_db(conn, cur, rec):
                total += 1
                print(f"    + {detail['title'][:60]} ({detail['date']})")

        # Short delay between pages
        time.sleep(1)

    conn.close()
    print(f"\n[DONE] Total: {total} articles, Errors: {errors}")
    if errors:
        print(f"  (列表页共 {total_listed} 条, 其中 {errors} 条详情不可达)")
    return total


if __name__ == "__main__":
    mp = MAX_PAGES
    if "--max-pages" in sys.argv:
        idx = sys.argv.index("--max-pages")
        if idx + 1 < len(sys.argv):
            mp = int(sys.argv[idx + 1])
    crawl(max_pages=mp)
