#!/usr/bin/env python3
"""通州区人民政府 - 公告公示 爬虫
站点: http://www.tongzhou.gov.cn/tzqrmzf/gsgg/gsgg.html
CMS: truecms，子栏目共7个，每页30条嵌入页面，client-side分页(3页*10条)
"""
import re, sys, os, time, json, urllib.request, urllib.error, sqlite3
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.tongzhou.gov.cn"
SUBS = {
    "zygs": "重要公示",
    "lhgg": "两会公告",
    "zpzk": "招聘招考",
    "spgs": "审批公示",
    "ghgs": "规划公示",
    "sjgs": "审计公示",
    "qtgg": "其他公告",
}

def fetch(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers={
                "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
            })
            resp = urllib.request.urlopen(req, timeout=30)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None

def parse_list(html, sub_name):
    """Extract articles from initData div"""
    items = []
    init = re.search(r'id="initData"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not init:
        return items
    content = init.group(1)
    uls = re.findall(r'<ul class="list-ul">(.*?)</ul>', content, re.DOTALL)
    for ul in uls:
        href = re.search(r'href=["\']([^"\']+)["\']', ul)
        txt_all = re.sub(r'<[^>]+>', '', ul).strip()
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})', txt_all)
        title = re.sub(r'\d{4}-\d{2}-\d{2}', '', txt_all).strip()
        if href:
            page_url = href.group(1)
            if not page_url.startswith("http"):
                page_url = BASE_URL + page_url
            items.append({
                "page_url": page_url,
                "title": title,
                "date": date_m.group(1) if date_m else "",
            })
    return items

def parse_detail(html, url):
    """Extract content from detail page"""
    # Title
    title_m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    title = title_m.group(1).replace("-南通市通州区人民政府", "").strip() if title_m else ""

    # Publish date
    pub_m = re.search(r'发布时间[：:]\s*<span[^>]*>\s*([^<]+)\s*</span>', html)
    publish_date = ""
    if pub_m:
        date_str = pub_m.group(1).strip()
        date_match = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', date_str)
        if date_match:
            publish_date = date_match.group(1)

    # Content - find the page-con div inside nr cf
    content_html = ""
    nrcf = re.search(r'class="nr cf">(.*?)</div>\s*<div[^>]*class="[^"]*modal', html, re.DOTALL)
    if nrcf:
        body = nrcf.group(1)
        page_con = re.search(r'class="page-con"[^>]*>(.*?)</div>', body, re.DOTALL)
        if page_con:
            content_html = page_con.group(1).strip()

    # If no page-con found, fallback to nr cf itself
    if not content_html:
        nrcf2 = re.search(r'class="nr cf">(.*?)</div>', html, re.DOTALL)
        if nrcf2:
            body2 = nrcf2.group(1)
            # Skip the page-tit and fb-info parts
            pc = re.search(r'</div>\s*</div>\s*<div[^>]*class="[^"]*page-con"[^>]*>(.*?)</div>', body2, re.DOTALL)
            if pc:
                content_html = pc.group(1).strip()

    # Extract text from paragraphs and tables (no duplicates)
    parts = []
    # p tags - first strip out table sections to avoid extracting table cell text
    clean_html = re.sub(r'<table[^>]*>.*?</table>', '', content_html, flags=re.DOTALL)
    for p in re.findall(r'<p[^>]*>(.*?)</p>', clean_html, re.DOTALL):
        # Skip paragraphs that only contain attachment links (icon img + download link)
        stripped = re.sub(r'<img[^>]*>', '', p)
        stripped = re.sub(r'<a[^>]*>.*?</a>', '', stripped)
        stripped = re.sub(r'<[^>]+>', '', stripped).strip()
        if not stripped:
            continue  # this p was just an attachment link, skip
        text = re.sub(r'<[^>]+>', '', p).strip()
        if text:
            parts.append(text)
    # table tags (raw HTML for proper rendering)
    for t in re.findall(r'<table[^>]*>.*?</table>', content_html, re.DOTALL):
        parts.append(str(t))

    # Attachment links
    attachments = []
    # Direct file URLs
    for m in re.finditer(r'href=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))["\']', html, re.I):
        att_url = m.group(1)
        if not att_url.startswith("http"):
            att_url = BASE_URL + att_url
        attachments.append(att_url)
    # download.do URLs (truecms attachment proxy, detect by filename in title)
    for m in re.finditer(r'<a[^>]+href=["\']([^"\']*download\.do[^"\']*)["\'][^>]*title=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))["\']', html, re.I):
        att_url = m.group(1)
        if not att_url.startswith("http"):
            att_url = BASE_URL + att_url
        attachments.append(att_url)

    # Real content images (skip icon/decorative images)
    content_images = []
    for m in re.finditer(r'<img[^>]+src=["\']([^"\']+(?:jpg|jpeg|png|gif|webp))["\']', html, re.I):
        img_url = m.group(1)
        # Skip decorative/icon images
        if any(x in img_url.lower() for x in ['icon', 'filetype', 'ewm', 'logo', 'banner']):
            continue
        if not img_url.startswith("http"):
            img_url = BASE_URL + img_url
        img_alt = re.search(r'alt=["\']([^"\']*)["\']', m.group(0))
        alt_text = img_alt.group(1) if img_alt else ""
        content_images.append({"src": img_url, "alt": alt_text})

    # Build rendered text
    rendered = "\n\n".join(parts)
    if content_images:
        for img in content_images:
            alt = img["alt"] if img["alt"] else "图片"
            rendered += f'\n\n![{alt}]({img["src"]})'
    if attachments:
        # Extract attachment titles from the HTML for better links
        att_markdown = []
        for att_url in attachments:
            # Try to find the link text
            att_title = att_url.split("/")[-1]
            if att_title.startswith("download.do") or "download.do" in att_url:
                # Try to find title attribute
                title_match = re.search(r'href=["\']' + re.escape(att_url) + r'["\'][^>]*title=["\']([^"\']+)["\']', html)
                if title_match:
                    att_title = title_match.group(1).strip()
            else:
                title_match = re.search(r'href=["\']' + re.escape(att_url) + r'["\'][^>]*>([^<]+)<', html)
                if title_match:
                    att_title = title_match.group(1).strip()
            att_markdown.append(f"- [{att_title}]({att_url})")
        rendered += f"\n\n**附件：**\n" + "\n".join(att_markdown)

    return {
        "title": title or url.split("/")[-1].replace(".html", ""),
        "publish_date": publish_date,
        "content_html": content_html,
        "content_rendered": rendered,
        "attachments": json.dumps(attachments, ensure_ascii=False),
    }

def save_to_db(items, details):
    """Upsert into gov_raw + sync to FTS"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    site_name = "南通市通州区人民政府-公告公示"
    now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
    inserted = 0

    for item in items:
        pu = item["page_url"]
        det = details.get(pu, {})
        # Check if exists
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND site_name=?", (pu, site_name))
        if c.fetchone()[0] > 0:
            continue
        content = det.get("content_rendered", "")
        c.execute("""INSERT OR REPLACE INTO gov_raw 
            (page_url, title, site_name, publish_date, content, attachments, category, date_rank)
            VALUES (?,?,?,?,?,?,?,?)""", (
            pu,
            det.get("title", item.get("title", "")),
            site_name,
            det.get("publish_date", item.get("date", "")),
            content,
            det.get("attachments", "[]"),
            "公示公告",
            int(datetime.now().timestamp()),
        ))
        inserted += 1

    conn.commit()
    conn.close()
    return inserted

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--incremental", action="store_true")
    args = parser.parse_args()

    # Step 1: parse all sub-column list pages
    all_items = []
    for sub_key, sub_label in SUBS.items():
        url = f"{BASE_URL}/tzqrmzf/{sub_key}/{sub_key}.html"
        print(f"[LIST] {sub_label} ({sub_key})...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("FAILED")
            continue
        items = parse_list(html, sub_key)
        print(f"{len(items)} items")
        all_items.extend(items)

    print(f"\nTotal items: {len(all_items)}")

    if args.incremental:
        # Deduplicate by URL before fetching details
        print(f"Incremental mode: will skip existing URLs")

    # Step 2: fetch detail pages
    details = {}
    for i, item in enumerate(all_items):
        pu = item["page_url"]
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}...", end=" ", flush=True)
        html = fetch(pu)
        if not html:
            print("SKIP")
            continue
        det = parse_detail(html, pu)
        details[pu] = det
        print(f"OK ({len(det.get('content_rendered',''))} chars)")

    # Step 3: save to DB
    inserted = save_to_db(all_items, details)
    print(f"\nInserted: {inserted} new items")
    print(f"Skipped (existing): {len(all_items) - inserted}")

if __name__ == "__main__":
    main()
