#!/usr/bin/env python3
"""Crawler for 惠水县-通知公告 (gzhs.gov.cn)
   https://www.gzhs.gov.cn/xwdt/tzgg/
   CMS: TRS, pagination: index_{N}.html
   Detail: div#Zoom > .trs_editor_view or raw content
"""
import urllib.request, urllib.parse, ssl, re, sqlite3, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "https://www.gzhs.gov.cn"
LIST_DIR = "/xwdt/tzgg/"
SITE_NAME = "惠水县人民政府"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = "/root/search.db"
MAX_PAGES = 5

incremental = "--incremental" in sys.argv


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def get_list_items(html):
    """Extract (page_url, title, date) from list page"""
    items = []
    for m in re.finditer(
        r'<li>\s*<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date = m.group(3).strip()
        full_url = href if href.startswith("http") else urllib.parse.urljoin(BASE, href)
        items.append((full_url, title, date))
    return items


def clean_content(raw_html):
    """Clean HTML content preserving <p>, <table>, <img>, <br>"""
    content = raw_html
    # Remove scripts/styles
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
    # Strip inline formatting
    content = re.sub(r'</?(?:span|font|strong|b|em|i|u|a|s|o:p)\b[^>]*>', '', content, flags=re.I)
    # Strip headings -> <p>
    content = re.sub(r'<h[1-6][^>]*>', '<p>', content)
    content = re.sub(r'</h[1-6]>', '</p>', content)
    # Strip attributes
    content = re.sub(r'\s+style="[^"]*"', '', content)
    content = re.sub(r"\s+style='[^']*'", '', content)
    content = re.sub(r'<p\s[^>]*>', '<p>', content)
    # Clean whitespace in <p>
    def collapse_p(m):
        inner = re.sub(r'\s+', ' ', m.group(1))
        return '<p>' + inner.strip() + '</p>'
    content = re.sub(r'<p>(.*?)</p>', collapse_p, content, flags=re.DOTALL)
    # Remove empty <p> tags
    content = re.sub(r'<p>\s*(?:<br>\s*)*</p>', '', content)
    content = re.sub(r'<br\s*/?>', '<br>', content)
    content = re.sub(r'&nbsp;', ' ', content)
    content = re.sub(r'>\s+<', '>\n<', content)
    # Remove data-* and other custom attributes
    content = re.sub(r'\s+(data-\w+|class|id|width|height|border|alt|srcset|sizes)="[^"]*"', '', content)
    return content.strip()


def get_detail(url):
    """Fetch and extract detail page content"""
    html = fetch(url)
    
    # Title from <h1>
    h1 = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    title = h1.group(1).strip() if h1 else ""
    
    # Date from <span class="fbsj">
    d = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    pub_date = d.group(1) if d else ""
    if not pub_date:
        d2 = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
        pub_date = d2.group(1) if d2 else ""
    
    # Content: div#Zoom > .trs_editor_view, or div#Zoom itself
    content_html = ""
    c = re.search(r'<div[^>]*id="Zoom"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if not c:
        c = re.search(r'<div[^>]*id="Zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if c:
        content_html = c.group(1)
        content_html = clean_content(content_html)
    
    return title, pub_date, content_html


def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    max_pages = MAX_PAGES if incremental else 999
    
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE + LIST_DIR
        else:
            url = f"{BASE}{LIST_DIR}index_{page-1}.html"
        
        print(f"[{datetime.now().strftime('%H:%M:%S')}] Page {page}...", end=" ", flush=True)
        
        try:
            html = fetch(url)
        except Exception as e:
            print(f"ERROR: {e}")
            time.sleep(2)
            continue
        
        items = get_list_items(html)
        if not items:
            print("No items, stopping")
            break
        
        print(f"{len(items)} items")
        
        for page_url, list_title, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            try:
                detail_title, pub_date, content_html = get_detail(page_url)
                if not content_html or len(content_html) < 50:
                    print(f"  SKIP (short): {list_title[:40]}")
                    total_skip += 1
                    continue
                
                title = detail_title or list_title
                date = pub_date or list_date
                
                summary = re.sub(r'<[^>]+>', ' ', content_html)
                summary = re.sub(r'\s+', ' ', summary).strip()[:300]
                
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, BASE, title, date, content_html, summary, SITE_NAME, "通知公告")
                )
                if c.rowcount > 0:
                    total_new += 1
                    if total_new <= 3 or total_new % 10 == 0:
                        print(f"  + {title[:50]}")
                conn.commit()
                time.sleep(0.3)
            except Exception as e:
                print(f"  ERROR: {list_title[:30]} - {e}")
                conn.rollback()
                time.sleep(1)
    
    conn.close()
    print(f"\n=== Done: New={total_new}, Skip={total_skip} ===")


if __name__ == "__main__":
    main()
