#!/usr/bin/env python3
"""Crawler for dancheng.gov.cn - 郸城县生态环境"""
import urllib.request, urllib.error, ssl, re, sqlite3, os, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE_DOMAIN = "www.dancheng.gov.cn"
SITE_NAME = "dancheng"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

DB_PATH = "/root/search.db"
MAX_PAGES = 5  # ~50 records for daily run

def fetch(url):
    req = urllib.request.Request("https://" + BASE_DOMAIN + url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    html = resp.read().decode("utf-8", errors="replace")
    return html

def extract_detail(detail_url):
    """Fetch and extract detail page content"""
    html = fetch(detail_url)
    # Title from meta
    t = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = t.group(1).strip() if t else ""
    if not title:
        t2 = re.search(r'<div class="cms-article-tit">(.*?)</div>', html, re.DOTALL)
        if t2:
            title = t2.group(1).strip()
    if not title:
        t3 = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if t3:
            title = t3.group(1).strip()
            title = re.sub(r'\s*_郸城县人民政府\s*$', '', title).strip()
    # Date from meta
    d = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    pub_date = d.group(1) if d else ""
    if not pub_date:
        d2 = re.search(r'时间：\s*(\d{4}-\d{2}-\d{2})', html)
        pub_date = d2.group(1) if d2 else ""
    # Source from meta
    s = re.search(r'<meta name="ContentSource" content="([^"]*)"', html)
    source = s.group(1).strip() if s else ""
    if not source:
        s2 = re.search(r'来源：\s*([^<]+)', html)
        source = s2.group(1).strip() if s2 else "郸城县人民政府"
    # Content - div#articleDetail
    c = re.search(r'<div class="article-detail" id="articleDetail">(.*?)</div>\s*</div>', html, re.DOTALL)
    content_html = c.group(1) if c else ""
    if not content_html:
        c2 = re.search(r'id="articleDetail"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        content_html = c2.group(1) if c2 else ""
    # Clean content - preserve table HTML tags, strip formatting tags
    # First, protect table structure
    content_text = content_html
    
    # Replace structural tags with newlines
    content_text = re.sub(r'<p[^>]*>', '\n', content_text)
    content_text = re.sub(r'</p>', '\n', content_text)
    content_text = re.sub(r'<br\s*/?>', '\n', content_text)
    content_text = re.sub(r'<div[^>]*>', '\n', content_text)
    content_text = re.sub(r'</div>', '\n', content_text)
    
    # Strip inline formatting tags but keep table tags
    # Remove spans, fonts, strong, b, em, etc.
    content_text = re.sub(r'</?(?:span|font|strong|b|em|i|u|a|s|style|link|meta|o:p|h[1-6])\b[^>]*>', '', content_text, flags=re.I)
    
    # Remove attributes from table tags (keep the tags clean)
    content_text = re.sub(r'<(table|tr|td|th|tbody|thead|tfoot)\b[^>]*>', r'<\1>', content_text, flags=re.I)
    content_text = re.sub(r'</(table|tr|td|th|tbody|thead|tfoot)>', r'</\1>', content_text, flags=re.I)
    
    # CRITICAL: Un-indent table tags so markdown doesn't treat them as code blocks
    # Remove leading whitespace from lines that are pure HTML tags
    lines = content_text.split('\n')
    result_lines = []
    for line in lines:
        stripped = line.strip()
        # If line is a pure HTML tag (e.g., <table>, </td>, <tr>)
        if re.match(r'^</?(?:table|tbody|thead|tfoot|tr|td|th)>$', stripped, re.I):
            result_lines.append(stripped)
        else:
            result_lines.append(line)
    content_text = '\n'.join(result_lines)
    
    # Clean up
    content_text = re.sub(r'&nbsp;', ' ', content_text)
    content_text = re.sub(r'\n\s*\n+', '\n\n', content_text)
    content_text = content_text.strip()
    
    # Add CSS for table rendering in the detail page
    if '<table>' in content_text or '<table ' in content_text:
        # Ensure tables have basic styling via inline style or just keep them as HTML
        # The search app's markdown renderer allows HTML
        pass
    
    return title, pub_date, source, content_text, content_html

def main():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    
    for page in range(1, MAX_PAGES + 1):
        list_url = "/sitesources/dcx/page_pc/zwgk/zdly/hjbhxxgk/list%d.html" % page
        print("[%s] Fetching page %d..." % (datetime.now().strftime("%H:%M:%S"), page))
        
        try:
            html = fetch(list_url)
        except Exception as e:
            print("  ERROR fetching page %d: %s" % (page, e))
            time.sleep(2)
            continue
        
        # Extract items
        items = re.findall(
            r'<div class="colRightOne">\s*<a href="(/sitesources/dcx/page_pc/zwgk/zdly/hjbhxxgk/article[^"]+\.html)"[^>]*title="([^"]*)"[^>]*>.*?<div class="pictext">(.*?)</div>.*?<div class="artpub">\s*<font>(\d{4}-\d{2}-\d{2})</font>',
            html, re.DOTALL
        )
        print("  Found %d items" % len(items))
        if not items:
            # Might have reached the end
            print("  No more items, stopping")
            break
        
        for href, title_attr, pictext, date_str in items:
            page_url = "https://" + BASE_DOMAIN + href
            title = title_attr.strip()
            
            # Check if exists
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            try:
                detail_title, pub_date, source, content_text, _ = extract_detail(href)
                if not detail_title:
                    detail_title = title
                if not pub_date:
                    pub_date = date_str
                if not content_text or len(content_text) < 50:
                    print("  SKIP %s: empty content" % href[-20:])
                    total_skip += 1
                    continue
                
                summary = content_text[:300] if len(content_text) > 300 else content_text
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, page_url, detail_title, pub_date, content_text, summary, SITE_NAME, "环评公示")
                )
                if c.rowcount > 0:
                    total_new += 1
                    print("  + %s: %s" % (href[-20:], detail_title[:50]))
                conn.commit()
                time.sleep(0.3)
            except Exception as e:
                print("  ERROR detail %s: %s" % (href[-20:], e))
                conn.rollback()
                time.sleep(1)
                continue
    
    conn.close()
    print("\n=== Done ===")
    print("New: %d, Skipped: %d" % (total_new, total_skip))

if __name__ == "__main__":
    main()
