#!/usr/bin/env python3
"""Crawler for xrd.yingkou.gov.cn - 营口仙人岛经济开发区 通知公告
   Fixed: preserve <p> tags for proper paragraph rendering
"""
import urllib.request, urllib.error, ssl, re, sqlite3, os, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE_DOMAIN = "xrd.yingkou.gov.cn"
SITE_NAME = "yingkou_xrd"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

DB_PATH = "/root/search.db"
MAX_PAGES = 5  # Latest ~75 records for daily run


def fetch(url):
    req = urllib.request.Request("https://" + BASE_DOMAIN + url, headers=HEADERS)
    resp = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    html = resp.read().decode("utf-8", errors="replace")
    return html


def clean_content(raw_html):
    """Convert the ewb-article-content HTML to clean HTML with <p> tags preserved.
    Strip inline formatting but keep structural tags (p, table, img, br).
    """
    content = raw_html

    # Remove <div class="ewebeditor_doc"> wrapper but keep inner content
    content = re.sub(r'<div class="ewebeditor_doc"[^>]*>', '', content)
    content = re.sub(r'</div>\s*$', '', content)

    # Strip inline formatting: span, font, strong, b, em, i, u, a, s, style, link, meta, o:p
    content = re.sub(r'</?(?:span|font|strong|b|em|i|u|a|s|style|link|meta|o:p)\b[^>]*>', '', content, flags=re.I)

    # Strip heading tags (h1-h6) - replace with <p>
    content = re.sub(r'<h[1-6][^>]*>', '<p>', content)
    content = re.sub(r'</h[1-6]>', '</p>', content)

    # Clean <br> tags - keep as <br>
    content = re.sub(r'<br\s*/?>', '<br>', content)

    # Strip 'style' attributes from remaining tags
    content = re.sub(r'\s+style="[^"]*"', '', content)
    content = re.sub(r"\s+style='[^']*'", '', content)

    # Strip attributes from <p> tags - keep only <p> and </p>
    content = re.sub(r'<p\s[^>]*>', '<p>', content)

    # Clean up &nbsp; etc
    content = re.sub(r'&nbsp;', ' ', content)
    content = re.sub(r'&ensp;', ' ', content)
    content = re.sub(r'&emsp;', ' ', content)

    # Remove empty <p> tags (just <br> or whitespace)
    content = re.sub(r'<p>\s*(?:<br>\s*)*</p>', '', content)

    # Collapse whitespace inside <p> tags: newlines/tabs -> single space
    def collapse_p(m):
        inner = m.group(1)
        inner = re.sub(r'\s+', ' ', inner)
        return '<p>' + inner.strip() + '</p>'
    content = re.sub(r'<p>(.*?)</p>', collapse_p, content, flags=re.DOTALL)

    # Normalize inter-tag whitespace
    content = re.sub(r'>\s+<', '>\n<', content)
    content = content.strip()

    return content


def extract_detail(detail_url):
    """Fetch and extract detail page content"""
    html = fetch(detail_url)
    
    # Title
    t = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = t.group(1).strip() if t else ""
    if not title:
        t2 = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if t2:
            title = t2.group(1).strip()
            title = re.sub(r'\s*[-–—]\s*营口仙人岛经济开发区管理委员会\s*$', '', title).strip()
    
    # Date
    d = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    pub_date = d.group(1) if d else ""
    if not pub_date:
        d2 = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        pub_date = d2.group(1) if d2 else ""
    
    # Content - extract from article content div (single closing </div>)
    c = re.search(r'class="ewb-article-content"[^>]*>(.*?)</div>', html, re.DOTALL)
    content_html = c.group(1) if c else ""
    if not content_html:
        c2 = re.search(r'id="ivs_content"[^>]*>(.*?)</div>', html, re.DOTALL)
        content_html = c2.group(1) if c2 else ""

    # Also extract attachments from <ul class="ewb-attach">
    attach_links = []
    a_matches = re.findall(r'<ul class="ewb-attach">.*?(<a[^>]+>.*?</a>).*?</ul>', html, re.DOTALL)
    if not a_matches:
        a_matches = re.findall(r'<li class="ewb-attach-tip">附件：</li>\s*<li><a[^>]+>([^<]+)</a>', html)
    
    content_text = ""
    if content_html:
        content_text = clean_content(content_html)
        # Append attachments if any
        if a_matches:
            attach_html = '\n<p><strong>附件：</strong></p>\n'
            for a_match in re.finditer(r'<a\s[^>]*href="([^"]*)"[^>]*>([^<]+)</a>', html[html.find('ewb-attach'):], re.DOTALL):
                attach_url = a_match.group(1)
                attach_name = a_match.group(2).strip()
                if attach_url.startswith('/'):
                    attach_url = 'https://' + BASE_DOMAIN + attach_url
                attach_html += f'<p><a href="{attach_url}">{attach_name}</a></p>\n'
            content_text += '\n' + attach_html

    return title, pub_date, content_text


def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            list_url = "/001/001005/about.html"
        else:
            list_url = "/001/001005/%d.html" % page
        
        print("[%s] Fetching page %d..." % (datetime.now().strftime("%H:%M:%S"), page))
        
        try:
            html = fetch(list_url)
        except Exception as e:
            print("  ERROR fetching page %d: %s" % (page, e))
            time.sleep(2)
            continue
        
        # Extract items - the href format is: /001/001005/YYYYMMDD/uuid.html
        items = re.findall(r'href="(/001/001005/\d+/[a-f0-9-]+\.html)"[^>]*>', html)
        dates = re.findall(r'class="ewb-list-date"[^>]*>(\d{4}-\d{2}-\d{2})', html)
        print("  Found %d items" % len(items))
        if not items:
            print("  No more items, stopping")
            break
        
        for idx, href in enumerate(items):
            page_url = "https://" + BASE_DOMAIN + href
            date_str = dates[idx] if idx < len(dates) else ""
            
            # Check if exists
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            try:
                detail_title, pub_date, content_text = extract_detail(href)
                if not pub_date:
                    pub_date = date_str
                if not content_text or len(content_text) < 50:
                    print("  SKIP %s: empty content" % href[-30:])
                    total_skip += 1
                    continue
                
                # Title: prefer detail page title
                if not detail_title:
                    detail_title = re.sub(r'<[^>]+>', '', items[idx]).strip() if idx < len(items) else ""
                
                summary = re.sub(r'<[^>]+>', ' ', content_text)
                summary = re.sub(r'\s+', ' ', summary).strip()[:300]
                
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, summary, site_name, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (page_url, page_url, detail_title, pub_date, content_text, summary, SITE_NAME, "环评公示")
                )
                if c.rowcount > 0:
                    total_new += 1
                    if total_new <= 3 or total_new % 10 == 0:
                        print("  + %s: %s" % (href[-30:], detail_title[:50]))
                conn.commit()
                time.sleep(0.3)
            except Exception as e:
                print("  ERROR detail %s: %s" % (href[-30:], e))
                conn.rollback()
                time.sleep(1)
                continue
    
    conn.close()
    print("\n=== Done ===")
    print("New: %d, Skipped: %d" % (total_new, total_skip))


if __name__ == "__main__":
    main()
