#!/usr/bin/env python3
"""Crawler for yayc.gov.cn - 雨城区人民政府 公示公告"""
import urllib.request, urllib.error, ssl, re, sqlite3, os, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE_DOMAIN = "www.yayc.gov.cn"
SERVER_IP = "124.161.245.169"
SITE_NAME = "yayc"
HEADERS = {
    "Host": BASE_DOMAIN,
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

DB_PATH = "/root/search.db"
MAX_PAGES = 5  # Latest ~100 records for daily run
PAGE_URL_BASE = "http://" + BASE_DOMAIN

def fetch(url):
    req = urllib.request.Request("http://" + SERVER_IP + url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    html = resp.read().decode("utf-8", errors="replace")
    return html

def extract_detail(detail_url):
    """Fetch and extract detail page content"""
    html = fetch(detail_url)
    # Title from meta
    t = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = t.group(1).strip() if t else ""
    if not title:
        t2 = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if t2:
            title = t2.group(1).strip()
            # Remove site suffix
            title = re.sub(r'\s*-\s*雨城区人民政府\s*$', '', title).strip()
    # Date from meta
    d = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    pub_date = d.group(1) if d else ""
    # Source from meta
    s = re.search(r'<meta name="ContentSource" content="([^"]*)"', html)
    source = s.group(1).strip() if s else ""
    # Content - div.mt30.xqing-web-box
    c = re.search(r'<div class="mt30 xqing-web-box">(.*?)</div>\s*<div class="xqing-web-file"', html, re.DOTALL)
    content_html = c.group(1) if c else ""
    if not content_html:
        c2 = re.search(r'<div class="mt30 xqing-web-box">(.*?)</div>', html, re.DOTALL)
        content_html = c2.group(1) if c2 else ""
    
    # Check for attachments
    attachments = re.findall(r'<a[^>]*href="([^"]+)"[^>]*>([^<]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))</a>', html, re.I)
    if attachments:
        for attach_url, attach_name in attachments:
            content_html += '<br><br>附件：<a href="%s">%s</a>' % (attach_url, attach_name)
    
    # Clean content
    content_text = re.sub(r'<p[^>]*>', '\n', content_html)
    content_text = re.sub(r'</p>', '\n', content_text)
    content_text = re.sub(r'<br\s*/?>', '\n', content_text)
    content_text = re.sub(r'<[^>]+>', '', content_text)
    content_text = re.sub(r'&nbsp;', ' ', content_text)
    content_text = re.sub(r'\n\s*\n+', '\n\n', content_text)
    content_text = content_text.strip()
    
    return title, pub_date, source, content_text, content_html

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    total_new = 0
    total_skip = 0
    
    for page in range(1, MAX_PAGES + 1):
        list_url = "/gongkai/jichu/40016.html?page=%d" % page
        print("[%s] Fetching list page %d..." % (datetime.now().strftime("%H:%M:%S"), page))
        
        try:
            html = fetch(list_url)
        except Exception as e:
            print("  ERROR fetching list page %d: %s" % (page, e))
            time.sleep(2)
            continue
        
        # Extract items
        items = re.findall(r'<li>.*?<a href="(/gongkai/show/([^"]+\.html))"[^>]*title="([^"]*)"[^>]*>(.*?)</a>.*?<span>([^<]+)</span>.*?</li>', html, re.DOTALL)
        print("  Found %d items" % len(items))
        
        for href, item_id, title_attr, link_text, date_str in items:
            page_url = PAGE_URL_BASE + href
            title = title_attr.strip()
            if not title:
                title = re.sub(r'<[^>]+>', '', link_text).strip()
            
            # Check if already exists
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            # Fetch detail page
            try:
                detail_title, pub_date, source, content_text, content_html = extract_detail(href)
                if not detail_title:
                    detail_title = title
                if not pub_date:
                    pub_date = date_str.strip()
                if not content_text or len(content_text) < 50:
                    print("  SKIP %s: empty or too short content" % item_id[:20])
                    total_skip += 1
                    continue
                
                # Summary
                summary = content_text[:300] if len(content_text) > 300 else content_text
                
                # Insert into gov_raw
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, PAGE_URL_BASE + href, detail_title, pub_date, content_text, summary, SITE_NAME, "公示公告")
                )
                if c.rowcount > 0:
                    total_new += 1
                    print("  + %s: %s" % (item_id[:20], detail_title[:50]))
                conn.commit()
                time.sleep(0.3)
                
            except Exception as e:
                print("  ERROR detail %s: %s" % (item_id[:20], e))
                conn.rollback()
                time.sleep(1)
                continue
    
    conn.close()
    print("\n=== Done ===")
    print("New: %d, Skipped: %d" % (total_new, total_skip))

if __name__ == "__main__":
    main()
