#!/usr/bin/env python3
"""Crawler for 东坡区-生态环境 (dp.gov.cn)
   https://www.dp.gov.cn/zwxxgk/fdzdgknr/sthj.htm
   CMS: VSB9, list: ul.content-list li, detail: div.v_news_content
"""
import urllib.request, urllib.parse, urllib.error, ssl, re, sqlite3, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "https://www.dp.gov.cn"
LIST_PATH = "/zwxxgk/fdzdgknr/sthj.htm"
SITE_NAME = "东坡区政府"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = "/root/search.db"
MAX_PAGES = 5

incremental = "--incremental" in sys.argv


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def get_list_items(html):
    """Extract (page_url, title, date) from list page"""
    items = []
    # Match li items with a[href] and span[date]
    for m in re.finditer(r'<li[^>]*>\s*<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>', html, re.DOTALL):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date = m.group(3).strip()
        full_url = urllib.parse.urljoin(BASE, href)
        items.append((full_url, title, date))
    return items


def get_detail(url):
    """Fetch detail page and extract content"""
    html = fetch(url)
    
    # Title from <title>
    t = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    title = t.group(1).strip() if t else ""
    title = re.sub(r'\s*[-–—]\s*东坡区委区政府网\s*$', '', title).strip()
    
    # Date
    d = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    pub_date = d.group(1) if d else ""
    if not pub_date:
        d2 = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if d2:
            pub_date = d2.group(1)
    if not pub_date:
        d3 = re.search(r'(\d{4}-\d{2}-\d{2})\s*\d{2}:\d{2}', html)
        if d3:
            pub_date = d3.group(1)
    
    # Content from div.v_news_content
    content_html = ""
    c = re.search(r'<div[^>]*class="v_news_content"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not c:
        c = re.search(r'id="vsb_content[\w]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if c:
        content_html = c.group(1)
        # Remove scripts
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        # Clean style attributes
        content_html = re.sub(r'\s+style="[^"]*"', '', content_html)
        content_html = re.sub(r"\s+style='[^']*'", '', content_html)
        content_html = content_html.strip()
    
    return title, pub_date, content_html


def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    max_pages = MAX_PAGES if incremental else 999
    
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE + LIST_PATH
        else:
            url = f"{BASE}/zwxxgk/fdzdgknr/sthj/{page-1}.htm"
        
        print(f"[{datetime.now().strftime('%H:%M:%S')}] Page {page}...", end=" ", flush=True)
        
        try:
            html = fetch(url)
        except Exception as e:
            print(f"ERROR: {e}")
            time.sleep(2)
            continue
        
        items = get_list_items(html)
        if not items:
            print(f"No items found, stopping")
            break
        
        print(f"{len(items)} items")
        
        for page_url, list_title, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            try:
                detail_title, pub_date, content_html = get_detail(page_url)
                if not content_html or len(content_html.strip()) < 50:
                    print(f"  SKIP (short content): {list_title[:40]}")
                    total_skip += 1
                    continue
                
                title = detail_title or list_title
                date = pub_date or list_date
                
                # Summary: strip HTML tags
                summary = re.sub(r'<[^>]+>', ' ', content_html)
                summary = re.sub(r'\s+', ' ', summary).strip()[:300]
                
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, BASE, title, date, content_html, summary, SITE_NAME, "环评公示")
                )
                if c.rowcount > 0:
                    total_new += 1
                    if total_new <= 3 or total_new % 10 == 0:
                        print(f"  + {title[:50]}")
                conn.commit()
                time.sleep(0.3)
            except Exception as e:
                print(f"  ERROR: {list_title[:30]} - {e}")
                conn.rollback()
                time.sleep(1)
    
    conn.close()
    print(f"\n=== Done: New={total_new}, Skip={total_skip} ===")


if __name__ == "__main__":
    main()
