#!/usr/bin/env python3
"""托克逊县公告通知爬虫
站点: http://www.tkx.gov.cn/tkxx/c106471/list.shtml
分页: list_{n}.shtml, 共33页, 每页33条
"""
import re, sys, os, time, json, urllib.request, urllib.error, sqlite3
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.tkx.gov.cn"
MAX_PAGES = 33

def fetch(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers={
                "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
            })
            resp = urllib.request.urlopen(req, timeout=30)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None

def parse_list(html):
    """Extract articles from the list page"""
    items = []
    idx = html.find("id=\"list_data\"")
    if idx < 0:
        return items
    end_idx = html.find("<div class=\"page_num\">", idx)
    if end_idx < 0:
        end_idx = idx + 10000
    list_html = html[idx:end_idx]

    # Find each li - only those with coninfo inside (article items)
    lis = re.findall(r'<li>\s*<a[^>]+href=["\']([^"\']+)["\'][^>]*>.*?</a>\s*<div class="coninfo">(.*?)</div>\s*</li>', list_html, re.DOTALL)
    for href, con_html in lis:
        # Title from info info01
        title_m = re.search(r'title=["\']([^"\']+)["\']', con_html)
        title = title_m.group(1).strip() if title_m else ""
        if not title:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href

        # Date
        date_m = re.search(r'<span class="time">(\d{4})年(\d{1,2})月(\d{1,2})日', con_html)
        publish_date = ""
        if date_m:
            publish_date = f"{date_m.group(1)}-{date_m.group(2).zfill(2)}-{date_m.group(3).zfill(2)}"

        items.append({
            "page_url": href,
            "title": title,
            "date": publish_date,
        })
    return items

def parse_detail(html, url):
    """Extract content from detail page"""
    # Title from h1 inside detail div
    title_m = re.search(r'<h1>(.*?)</h1>', html, re.DOTALL)
    title = title_m.group(1).strip() if title_m else ""

    # Publish date
    pub_m = re.search(r'发布时间[：:]\s*(\d{4})年(\d{1,2})月(\d{1,2})日', html)
    publish_date = ""
    if pub_m:
        publish_date = f"{pub_m.group(1)}-{pub_m.group(2).zfill(2)}-{pub_m.group(3).zfill(2)}"

    # Content - find NewsContent div
    content_html = ""
    c = re.search(r'id="NewsContent"[^>]*>(.*?)</div>', html, re.DOTALL)
    if c:
        content_html = c.group(1).strip()

    # Extract text from paragraphs and tables
    parts = []
    clean_html = re.sub(r'<table[^>]*>.*?</table>', '', content_html, flags=re.DOTALL)
    for p in re.findall(r'<p[^>]*>(.*?)</p>', clean_html, re.DOTALL):
        # Skip empty paragraphs or attachment-only paragraphs
        text = re.sub(r'<[^>]+>', '', p).strip()
        if not text:
            continue
        # Decode HTML entities
        text = text.replace('&nbsp;', ' ').replace('&lt;', '<').replace('&gt;', '>')
        text = text.replace('&amp;', '&').replace('&quot;', '"')
        if text.strip():
            parts.append(text.strip())
    # Table tags (raw HTML)
    for t in re.findall(r'<table[^>]*>.*?</table>', content_html, re.DOTALL):
        parts.append(str(t))

    # Images (only from content area)
    content_images = []
    for m in re.finditer(r'<img[^>]+src=["\']([^"\']+)["\']', content_html, re.I):
        img_url = m.group(1)
        if any(x in img_url.lower() for x in ['icon', 'filetype', 'ewm', 'logo', 'banner', 'bottom', 'thumb', 'nopic']):
            continue
        if not img_url.startswith("http"):
            # Check if it starts with /
            if img_url.startswith('/'):
                img_url = BASE_URL + img_url
            else:
                img_url = BASE_URL + '/' + img_url
        content_images.append(img_url)

    # Attachments
    attachments = []
    for m in re.finditer(r'href=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))["\']', html, re.I):
        att_url = m.group(1)
        if not att_url.startswith("http"):
            if att_url.startswith('/'):
                att_url = BASE_URL + att_url
            else:
                att_url = BASE_URL + '/' + att_url
        attachments.append(att_url)

    # Build rendered text
    rendered = "\n\n".join(parts)
    if content_images:
        for img_url in content_images:
            rendered += f'\n\n![图片]({img_url})'
    if attachments:
        att_lines = []
        for att_url in attachments:
            att_title = att_url.split("/")[-1]
            att_lines.append(f"- [{att_title}]({att_url})")
        rendered += "\n\n**附件：**\n" + "\n".join(att_lines)

    return {
        "title": title or url.split("/")[-1].replace(".shtml", ""),
        "publish_date": publish_date,
        "content_rendered": rendered,
        "attachments": json.dumps(attachments, ensure_ascii=False),
    }

def save_to_db(items, details):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    site_name = "托克逊县-公告通知"
    inserted = 0

    for item in items:
        pu = item["page_url"]
        det = details.get(pu, {})
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND site_name=?", (pu, site_name))
        if c.fetchone()[0] > 0:
            continue
        content = det.get("content_rendered", "")
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, attachments, category, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?, 'crawl_tkx.py')""", (
            pu,
            det.get("title", item.get("title", "")),
            site_name,
            det.get("publish_date", item.get("date", "")),
            content,
            det.get("attachments", "[]"),
            "公告通知",
            int(datetime.now().timestamp()),
        ))
        inserted += 1

    conn.commit()
    conn.close()
    return inserted

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--incremental", action="store_true")
    parser.add_argument("--max-pages", type=int, default=5, help="Max pages to crawl")
    args = parser.parse_args()

    pages_to_crawl = min(args.max_pages, MAX_PAGES)
    all_items = []
    for page_num in range(1, pages_to_crawl + 1):
        if page_num == 1:
            url = f"{BASE_URL}/tkxx/c106471/list.shtml"
        else:
            url = f"{BASE_URL}/tkxx/c106471/list_{page_num}.shtml"
        print(f"[LIST] Page {page_num}/{pages_to_crawl}...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("FAILED")
            continue
        items = parse_list(html)
        print(f"{len(items)} items")
        all_items.extend(items)

    print(f"\nTotal items: {len(all_items)}")

    details = {}
    for i, item in enumerate(all_items):
        pu = item["page_url"]
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}...", end=" ", flush=True)
        html = fetch(pu)
        if not html:
            print("SKIP")
            continue
        det = parse_detail(html, pu)
        details[pu] = det
        print(f"OK ({len(det.get('content_rendered',''))} chars)")

    inserted = save_to_db(all_items, details)
    print(f"\nInserted: {inserted} new items")
    print(f"Skipped (existing): {len(all_items) - inserted}")

if __name__ == "__main__":
    main()
