#!/usr/bin/env python3
"""
清徐县人民政府 - 信息公开 爬虫
JEECMS, 20页×20条=400条
附件为主（PDF/DOCX），正文内容以链接形式嵌入
"""

import sys, re, time, sqlite3, subprocess, hashlib, argparse
from datetime import datetime
from urllib.parse import urljoin
import requests

BASE_URL = "https://www.qx.gov.cn/xxgk2222222222222222222222222222222222222.html"
BASE_DIR = "https://www.qx.gov.cn/"
SESSION = requests.Session()
SESSION.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
})
SITE_NAME = "清徐县信息公开"
GROUP = "县区"
DB_PATH = "/mnt/data/search.db"

def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            resp = SESSION.get(url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return resp.text
        except Exception as e:
            pass
        time.sleep(2)
    return None

def parse_list_page(html, base_url):
    """Extract (title, url, date) from list page."""
    items = []
    # Find the list UL before pagination
    idx = html.find('list_page')
    if idx == -1:
        return items
    section_start = html.rfind('<ul', 0, idx)
    if section_start == -1:
        return items
    section = html[section_start:idx]

    for m in re.finditer(r'<li>\s*<a href="([^"]+)" title="([^"]*)"[^>]*>.*?</a>\s*<span class="riqi">([^<]+)</span>\s*</li>', section, re.DOTALL):
        href = m.group(1).strip()
        title = m.group(2).replace('&nbsp;', ' ').replace('\u00a0', ' ').strip()
        date = m.group(3).strip()
        if title and href:
            items.append((title, urljoin(base_url, href), date))
    return items

def parse_detail(html, page_url, list_title=""):
    """Extract title, date, content from detail page."""
    # Title from <title> tag (strip site suffix)
    title = ""
    m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    if m:
        t = m.group(1).strip()
        # Strip "-太原市清徐县人民政府门户网站" suffix
        for suffix in ['-太原市清徐县人民政府门户网站', '-清徐县人民政府']:
            if t.endswith(suffix):
                t = t[:-len(suffix)]
                break
        title = t
    # Fallback: h1 tag (but avoid "清徐县人民政府门户网站" placeholder)
    if not title or title == '清徐县人民政府门户网站' or '门户网站' in title:
        m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if m:
            t = re.sub(r'<[^>]+>', '', m.group(1)).strip()
            if t and '门户网站' not in t:
                title = t
    # Last resort: use list title
    if not title or '门户网站' in title or len(title) < 5:
        title = list_title
    title = title.replace('&nbsp;', ' ').replace('\u00a0', ' ').strip()

    # Date from <em> after 时间：
    publish_date = ""
    m = re.search(r'时间[：:]\s*<em>(\d{4}[-/]\d{1,2}[-/]\d{1,2})', html)
    if m:
        publish_date = m.group(1).replace('/', '-')

    # Content from Zoom div
    content_html = ""
    zoom_start = html.find('id="Zoom"')
    if zoom_start == -1:
        zoom_start = html.find('id=\"Zoom\"')
    if zoom_start != -1:
        zoom_close = html.find('</div>', zoom_start)
        if zoom_close != -1:
            content_html = html[zoom_start:zoom_close + 6]

    # Extract content between markers
    content_body = ""
    if content_html:
        m = re.search(r'<!--\s*<\$\[CONTENT\]>start-->([\s\S]*?)<!--\s*<\$\[CONTENT\]>end-->', content_html)
        if m:
            content_body = m.group(1).strip()

    # Attachments inside content
    attachments = []
    if content_body:
        for m in re.finditer(r'<a\s+href="([^"]+)"[^>]*>([^<]+)</a>', content_body):
            href = m.group(1).strip()
            name = m.group(2).strip()
            # Clean weird span in name
            name = re.sub(r'<span[^>]*>.*?</span>', '', name).strip()
            attachments.append((name, urljoin(page_url, href)))

    # Render content
    content_text = content_body
    if attachments:
        # For PDF-only or doc-only pages, embed as links
        lines = []
        # Check if content has actual text (not just attachment links)
        text_only = re.sub(r'<[^>]+>', '', content_body).strip()
        if text_only and any(c for c in ['告', '示', '通', '知', '公', '项', '目'] if c in text_only):
            lines.append(text_only)
        lines.append('\n**相关文件：**')
        for name, href in attachments:
            lines.append(f'[{name}]({href})')
        content_text = '\n\n'.join(lines)
    elif not content_text.strip():
        content_text = title  # fallback to title

    return title, publish_date, content_text, attachments

def main(max_pages=5):
    html = fetch(BASE_URL)
    if not html:
        print("ERROR: Failed to fetch list page", file=sys.stderr)
        sys.exit(1)

    # Get total pages from HTML
    total_pages = 20  # fixed: 1/20 pages shown
    pages = min(max_pages, total_pages)
    print(f"Total: {total_pages} pages, fetching: {pages}")

    # Parse all list pages
    all_items = []
    urls = [BASE_URL] + [f"https://www.qx.gov.cn/xxgk2222222222222222222222222222222222222_{p}.html" for p in range(2, pages + 1)]

    for i, url in enumerate(urls):
        if i == 0:
            page_html = html
        else:
            page_html = fetch(url)
            if not page_html:
                continue
        items = parse_list_page(page_html, url)
        all_items.extend(items)
        print(f"  Page {i+1}: {len(items)} items")
        time.sleep(1)

    # Deduplicate
    seen = set()
    unique = []
    for t, u, d in all_items:
        if u not in seen:
            seen.add(u)
            unique.append((t, u, d))

    print(f"\nTotal unique: {len(unique)}")

    # Check existing
    existing = set()
    try:
        res = subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH,
            "SELECT page_url FROM gov_raw WHERE page_url LIKE '%qx.gov.cn/xxgk222%'"],
            capture_output=True, text=True, timeout=10)
        if res.stdout.strip():
            existing = set(res.stdout.strip().split('\n'))
    except Exception:
        pass

    new_items = [(t, u, d) for t, u, d in unique if u not in existing] if existing else unique
    print(f"New: {len(new_items)}")

    if not new_items:
        print("Nothing new.")
        return

    # Fetch details and insert
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    inserted = 0
    errors = 0
    fts_sqls = []

    for title, url, list_date in new_items:
        detail_html = fetch(url)
        if not detail_html or '<title>404页面</title>' in detail_html:
            # Detail page not available - use list-level data only
            content = title
            final_title = title
            final_date = list_date
            summary = content[:200].replace('\n', ' ').strip() if len(content) > 200 else content.replace('\n', ' ')
            attachments_str = ''
            source_url = url
            try:
                cur = conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments, category, group_name)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
                ''', (final_title, content, summary, SITE_NAME,
                      url, final_date, source_url, attachments_str, '信息公开', GROUP))
                if cur.rowcount > 0:
                    inserted += 1
                    rid = cur.lastrowid
                    fts_sqls.append(
                        f"INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES ({rid},"
                        f"'{final_title.replace(chr(39), chr(39)+chr(39))}',"
                        f"'{SITE_NAME.replace(chr(39), chr(39)+chr(39))}',"
                        f"'{summary.replace(chr(39), chr(39)+chr(39))}');"
                    )
                    print(f"  [{inserted}] {final_title[:40]}...")
            except Exception as e:
                print(f"  ERROR: {e}", file=sys.stderr)
                errors += 1
            continue

        if not detail_html:
            errors += 1
            continue

        det_title, det_date, content, attachments = parse_detail(detail_html, url, title)
        final_title = (det_title or title).replace('\xa0', ' ').strip()
        final_date = det_date or list_date

        summary = content[:200].replace('\n', ' ').strip() if len(content) > 200 else content.replace('\n', ' ')
        attachments_str = '; '.join([f'{name}|{href}' for name, href in attachments]) if attachments else ''
        source_url = url

        try:
            cur = conn.execute('''
                INSERT INTO gov_raw (title, content, summary, site_name,
                page_url, publish_date, source_url, attachments, category, group_name)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
            ''', (final_title, content, summary, SITE_NAME,
                  url, final_date, source_url, attachments_str, '信息公开', GROUP))
            if cur.rowcount > 0:
                inserted += 1
                rid = cur.lastrowid
                fts_sqls.append(
                    f"INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES ({rid},"
                    f"'{final_title.replace(chr(39), chr(39)+chr(39))}',"
                    f"'{SITE_NAME.replace(chr(39), chr(39)+chr(39))}',"
                    f"'{summary.replace(chr(39), chr(39)+chr(39))}');"
                )
                print(f"  [{inserted}] {final_title[:40]}...")
        except Exception as e:
            print(f"  ERROR: {e}", file=sys.stderr)
            errors += 1

        time.sleep(0.3)

    conn.commit()

    # FTS sync - write to v4 too
    if fts_sqls:
        try:
            fts_sqls.append("INSERT INTO gov_search(gov_search) VALUES('rebuild');")
            subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH],
                input='\n'.join(fts_sqls), capture_output=True, text=True, timeout=120)
            print(f"FTS: {len(fts_sqls)-1} rows")
        except Exception as e:
            print(f"FTS error: {e}", file=sys.stderr)

    conn.close()
    print(f"\nDone. Inserted: {inserted}, Errors: {errors}")

if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='清徐县信息公开爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数')
    args = parser.parse_args()
    main(max_pages=args.max_pages)
