#!/usr/bin/env python3
"""Crawler for 新会区-民意征集 (xinhui.gov.cn)
   https://www.xinhui.gov.cn/zmhd/dhzf/zjdc/myzj/index.html
   CMS: Guangdong unified survey system (AJAX content via question_data JSON)
   Pagination: index_N.html
"""
import urllib.request, ssl, re, json, sqlite3, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "https://www.xinhui.gov.cn"
LIST_DIR = "/zmhd/dhzf/zjdc/myzj/"
SITE_NAME = "新会区-征集调查"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = "/root/search.db"
MAX_PAGES = 5

incremental = "--incremental" in sys.argv


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def get_list_items(html):
    """Extract (page_url, title, date) from list page.
    Uses simple regex to find answer links and then looks for dates nearby.
    """
    items = []
    # Find all answer links with title
    links = re.findall(r'<a[^>]*href="(http[^"]*answer/\d+)"[^>]*title="([^"]*)"', html)
    # Find all dates in the page
    dates = re.findall(r'(\d{4}-\d{2}-\d{2})', html)
    
    # Pair them: the first 20 dates are for the first 20 answer links
    for i, (href, title) in enumerate(links):
        date = dates[i] if i < len(dates) else ""
        items.append((href.strip(), title.strip(), date.strip()))
    return items


def get_detail(url):
    """Extract content from question_data JSON embedded in page"""
    html = fetch(url)
    
    # Find question_data JSON using manual brace matching
    idx = html.find("question_data:")
    if idx < 0:
        return "", "", ""
    
    start = html.find("{", idx)
    if start < 0:
        return "", "", ""
    
    # Match braces to find the complete JSON
    depth = 0
    end = start
    for i in range(start, len(html)):
        ch = html[i]
        if ch == "{":
            depth += 1
        elif ch == "}":
            depth -= 1
            if depth == 0:
                end = i + 1
                break
    
    if end == start:
        return "", "", ""
    
    try:
        data = json.loads(html[start:end])
    except json.JSONDecodeError:
        return "", "", ""
    
    article = data.get("article", {})
    
    title = article.get("title", "") or data.get("title", "")
    
    pub_date = ""
    published = article.get("published_at", "")
    if published:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', str(published))
        if m:
            pub_date = m.group(1)
    
    content_html = article.get("content", "")
    if content_html:
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'\s+style="[^"]*"', '', content_html)
        content_html = re.sub(r"\s+class=\"[^\"]*\"", '', content_html)
        content_html = re.sub(r"\s+id=\"[^\"]*\"", '', content_html)
        content_html = re.sub(r"\s+(noextractcontent|indenttext|data-\w+)=\"[^\"]*\"", '', content_html)
        content_html = re.sub(r'<p\s[^>]*>', '<p>', content_html)
        content_html = content_html.strip()
        
        # Append attachments
        attachments = article.get("attachments", [])
        if attachments:
            ans_id = data.get("id", "")
            att_html = '\n<p><strong>附件：</strong></p>\n'
            for att_id in attachments:
                att_url = f"http://www.xinhui.gov.cn/hdjlpt/yjzj/answer/{ans_id}/attachment/{att_id}"
                att_html += f'<p><a href="{att_url}">附件</a></p>\n'
            content_html += att_html
    
    return title, pub_date, content_html


def main():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    max_pages = MAX_PAGES if incremental else 999
    
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE + LIST_DIR + "index.html"
        else:
            url = f"{BASE}{LIST_DIR}index_{page}.html"
        
        print(f"[{datetime.now().strftime('%H:%M:%S')}] Page {page}...", end=" ", flush=True)
        
        try:
            html = fetch(url)
        except Exception as e:
            print(f"ERROR: {e}")
            time.sleep(1)
            continue
        
        items = get_list_items(html)
        if not items:
            print("No items, stopping")
            break
        
        print(f"{len(items)} items")
        
        for page_url, list_title, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            try:
                detail_title, pub_date, content_html = get_detail(page_url)
                if not content_html or len(content_html) < 50:
                    total_skip += 1
                    continue
                
                title = detail_title or list_title
                date = pub_date or list_date
                
                summary = re.sub(r'<[^>]+>', ' ', content_html)
                summary = re.sub(r'\s+', ' ', summary).strip()[:300]
                
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, BASE, title, date, content_html, summary, SITE_NAME, "民意征集")
                )
                if c.rowcount > 0:
                    total_new += 1
                    if total_new <= 100:
                        print(f"  + {title[:50]}")
                conn.commit()
                time.sleep(0.3)
            except Exception as e:
                print(f"  ERROR: {list_title[:30]} - {e}")
                conn.rollback()
                time.sleep(1)
    
    conn.close()
    print(f"\n=== Done: New={total_new}, Skip={total_skip} ===")


if __name__ == "__main__":
    main()
