#!/usr/bin/env python3
"""天元区-生态环境保护爬虫
站点: http://www.tianyuan.gov.cn/c19445/index.html
栏目: 生态环境保护 (c19445)
分页: /c19445/pages/N.html (第1页: index.html), 10页×15条
CMS: ZZCMS
"""
import re, sys, os, time, json, urllib.request, urllib.error, sqlite3
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.tianyuan.gov.cn"
COLUMN_ID = "c19445"
MAX_PAGES = 10
SITE_NAME = "天元区-生态环境保护"

def fetch(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers={
                "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
                "Accept-Language": "zh-CN,zh;q=0.9",
            })
            resp = urllib.request.urlopen(req, timeout=30)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None

def parse_list(html):
    items = []
    # Find the xxgk-list section
    idx = html.find('class="xxgk-list"')
    if idx < 0:
        return items
    end = html.find("</div>", html.find("</ul>", idx))
    section = html[idx:end] if end > idx else html[idx:]
    
    lis = re.findall(r'<li[^>]*>(.*?)</li>', section, re.DOTALL)
    for li in lis:
        href_m = re.search(r'href=[\"\']([^\"\']+)[\"\']', li)
        if not href_m:
            continue
        url = href_m.group(1)
        if not url.startswith("http"):
            url = BASE_URL + url
        
        # Title from <a>
        a_text = re.sub(r'<[^>]+>', '', re.search(r'<a[^>]*>(.*?)</a>', li, re.DOTALL).group(1) if re.search(r'<a[^>]*>(.*?)</a>', li, re.DOTALL) else "").strip()
        
        # Date from <span>
        date_m = re.search(r'<span>(.*?)</span>', li)
        date_str = date_m.group(1).strip() if date_m else ""
        # Normalize date format YYYY-MM-DD
        dm = re.search(r'(\d{4})-(\d{1,2})-(\d{1,2})', date_str)
        publish_date = dm.group(0) if dm else date_str
        
        items.append({
            "page_url": url,
            "title": a_text,
            "date": publish_date,
        })
    return items

def parse_detail(html, url):
    title = ""
    title_m = re.search(r'<h2>(.*?)</h2>', html, re.DOTALL)
    if title_m:
        title = title_m.group(1).strip()
    
    # Publish date
    publish_date = ""
    date_m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
    if date_m:
        publish_date = date_m.group(1)
    if not publish_date:
        dm = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', url)
        if dm:
            publish_date = dm.group(1)
    
    # Content
    content_html = ""
    c = re.search(r'class="content-main"[^>]*>(.*?)</div>', html, re.DOTALL)
    if c:
        content_html = c.group(1).strip()
        # Clean empty <p><br/></p> leading/trailing
        content_html = re.sub(r'^<p>\s*<br\s*/?>\s*</p>', '', content_html)
        content_html = re.sub(r'<p>\s*<br\s*/?>\s*</p>$', '', content_html)
    
    # Build content: tables as raw HTML, p as text
    # First extract and remove tables from content_html to avoid double-extraction
    tables = re.findall(r'<table[^>]*>.*?</table>', content_html, re.DOTALL)
    no_table_html = re.sub(r'<table[^>]*>.*?</table>', '', content_html, flags=re.DOTALL)
    # Also remove <tbody> leftovers
    no_table_html = re.sub(r'</?tbody[^>]*>', '', no_table_html)
    
    # Extract <p> text from non-table content only
    parts = []
    for p in re.findall(r'<p[^>]*>(.*?)</p>', no_table_html, re.DOTALL):
        text = re.sub(r'<[^>]+>', '', p).strip()
        if text:
            parts.append(text)
    
    for t in tables:
        parts.append(t)
    
    # Images
    for img in re.findall(r'<img[^>]+src=[\"\']([^\"\']+)[\"\'][^>]*>', content_html):
        img_url = img
        if img_url.startswith("//"):
            img_url = "https:" + img_url
        elif not img_url.startswith("http"):
            img_url = BASE_URL + ("" if img_url.startswith("/") else "/") + img_url
        alt_m = re.search(r'alt=[\"\']([^\"\']*)[\"\']', content_html[content_html.find(img)-100:content_html.find(img)+len(img)+50] if content_html.find(img) >= 100 else html)
        alt = alt_m.group(1) if alt_m else ""
        parts.append(f"![{alt}]({img_url})")
    
    # Attachments - look for protocol-relative (//), absolute, and relative URLs to PDF/doc/etc
    attachments = []
    for m in re.finditer(r'href=[\"\']([^\"\']*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))[\"\']', html, re.I):
        att_url = m.group(1)
        if att_url.startswith("//"):
            att_url = "https:" + att_url
        elif not att_url.startswith("http"):
            att_url = BASE_URL + ("" if att_url.startswith("/") else "/") + att_url
        attachments.append(att_url)
    # Deduplicate
    attachments = list(dict.fromkeys(attachments))
    
    rendered = "\n\n".join(parts)
    
    if attachments:
        att_lines = []
        for att_url in attachments:
            att_title = att_url.split("/")[-1]
            att_lines.append(f"- [{att_title}]({att_url})")
        rendered += "\n\n**附件：**\n" + "\n".join(att_lines)
    
    return {
        "title": title or url.split("/")[-1].replace(".html", ""),
        "publish_date": publish_date,
        "content_rendered": rendered,
        "attachments": json.dumps(attachments, ensure_ascii=False),
    }

def save_to_db(items, details):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for item in items:
        pu = item["page_url"]
        det = details.get(pu, {})
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND site_name=?", (pu, SITE_NAME))
        if c.fetchone()[0] > 0:
            continue
        content = det.get("content_rendered", "")
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, attachments, category, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?, 'crawl_tianyuan.py')""", (
            pu,
            det.get("title", item.get("title", "")),
            SITE_NAME,
            det.get("publish_date", item.get("date", "")),
            content,
            det.get("attachments", "[]"),
            "公示公告",
            int(datetime.now().timestamp()),
        ))
        inserted += 1
    conn.commit()
    conn.close()
    return inserted

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--incremental", action="store_true")
    parser.add_argument("--max-pages", type=int, default=5, help="Max pages to crawl")
    args = parser.parse_args()
    
    pages_to_crawl = min(args.max_pages, MAX_PAGES)
    all_items = []
    for page_num in range(1, pages_to_crawl + 1):
        if page_num == 1:
            url = f"{BASE_URL}/{COLUMN_ID}/index.html"
        else:
            url = f"{BASE_URL}/{COLUMN_ID}/pages/{page_num}.html"
        print(f"[LIST] Page {page_num}/{pages_to_crawl}...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("FAILED")
            continue
        items = parse_list(html)
        print(f"{len(items)} items")
        all_items.extend(items)
    
    print(f"\nTotal items: {len(all_items)}")
    
    details = {}
    for i, item in enumerate(all_items):
        pu = item["page_url"]
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}...", end=" ", flush=True)
        html = fetch(pu)
        if not html:
            print("SKIP")
            continue
        det = parse_detail(html, pu)
        details[pu] = det
        print(f"OK ({len(det.get('content_rendered',''))} chars)")
    
    inserted = save_to_db(all_items, details)
    print(f"\nInserted: {inserted} new items")
    print(f"Skipped (existing): {len(all_items) - inserted}")

if __name__ == "__main__":
    main()
