#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
营口仙人岛能源化工区 - 项目环保公示
http://xrd.yingkou.gov.cn/009/009004/about.html
"""
import re, sys, os, json, time, requests

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
SITE_NAME = "营口仙人岛能源化工区-项目环保公示"
GROUP = "营口"
BASE_URL = "http://xrd.yingkou.gov.cn/009/009004"
PER_PAGE = 15
MAX_PAGES = 5


def get_list_url(page):
    if page == 1:
        return BASE_URL + "/about.html"
    else:
        return BASE_URL + "/about.html?pageIndex=%d" % page


def fetch_page(url):
    r = requests.get(url, headers=HEADERS, timeout=30, allow_redirects=True)
    r.encoding = "utf-8"
    return r.text


def extract_list_items(html):
    items = []
    m = re.search(r'<ul[^>]*id="infolist"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items
    ul_html = m.group(1)
    # Find all li elements
    for li in re.finditer(r'<li[^>]*>(.*?)</li>', ul_html, re.DOTALL):
        li_html = li.group(1)
        a = re.search(r'<a[^>]*href="([^"]*)"[^>]*>', li_html)
        if not a:
            continue
        url = a.group(1)
        # Title - try title attribute first (single or double quotes)
        title_m = re.search(r'title=[\'"]([^\'"]*)[\'"]', li_html)
        title = title_m.group(1).strip() if title_m else ""
        if not title:
            # fallback to link text
            txt_m = re.search(r'<a[^>]*>(.*?)</a>', li_html, re.DOTALL)
            if txt_m:
                title = re.sub(r'<[^>]+>', '', txt_m.group(1)).strip()
        if not title or not url:
            continue
        # Date
        date_m = re.search(r'<span[^>]*class="ewb-list-date"[^>]*>(.*?)</span>', li_html)
        date_str = date_m.group(1).strip() if date_m else ""
        # Normalize URL
        if url.startswith("/"):
            url = "http://xrd.yingkou.gov.cn" + url
        elif not url.startswith("http"):
            url = "http://xrd.yingkou.gov.cn/" + url
        if not url.startswith("http://xrd.yingkou.gov.cn"):
            continue
        items.append({"url": url, "title": title, "date": date_str})
    return items


def extract_content(html, source_url):
    title = ""
    m = re.search(r'<h3[^>]*id="ivs_title"[^>]*>(.*?)</h3>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()
    
    content_parts = []
    attachments = []
    
    m = re.search(r'<div[^>]*class="ewb-article-content"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content_html = m.group(1)
        for p in re.finditer(r'<p[^>]*>(.*?)</p>', content_html, re.DOTALL):
            txt = re.sub(r'<[^>]+>', '', p.group(1)).strip()
            txt = re.sub(r'\s+', ' ', txt).strip()
            if txt:
                content_parts.append(txt)
        for table in re.finditer(r'<table[^>]*>(.*?)</table>', content_html, re.DOTALL):
            rows = []
            for tr in re.finditer(r'<tr[^>]*>(.*?)</tr>', table.group(1), re.DOTALL):
                cells = []
                for td in re.finditer(r'<t[dh][^>]*>(.*?)</t[dh]>', tr.group(1), re.DOTALL):
                    cell_text = re.sub(r'<[^>]+>', '', td.group(1)).strip()
                    cells.append(cell_text)
                if cells:
                    rows.append("| " + " | ".join(cells) + " |")
            if rows:
                content_parts.append("\n".join(rows))
        for a in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>(.*?)</a>', content_html, re.DOTALL):
            furl = a.group(1)
            fname = re.sub(r'<[^>]+>', '', a.group(2)).strip()
            if not fname:
                fname = os.path.basename(furl)
            if not furl.startswith("http"):
                furl = ("http://xrd.yingkou.gov.cn" + furl) if furl.startswith("/") else "http://xrd.yingkou.gov.cn/" + furl
            attachments.append({"name": fname, "url": furl})
    
    content = "\n\n".join(content_parts)
    if len(content.strip()) < 20:
        content = '<p><a href="{}">{}</a></p>'.format(source_url, title or "文件")
        if attachments:
            for att in attachments:
                content += '\n📎 <p><a href="{}">{}</a></p>'.format(att['url'], att['name'])
    
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
    return {"title": title, "content": content, "attachments": attachments_json}


def main():
    import sqlite3
    
    is_incremental = "--max-pages" in sys.argv or "incremental" in sys.argv
    max_pages = 1 if is_incremental else MAX_PAGES
    
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=15000")
    c = conn.cursor()
    
    all_items = []
    for page in range(1, max_pages + 1):
        url = get_list_url(page)
        print(f"[Page {page}/{max_pages}] {url}")
        html = fetch_page(url)
        items = extract_list_items(html)
        if not items:
            print("  No items found, stopping")
            break
        print(f"  Found {len(items)} items")
        all_items.extend(items)
        time.sleep(1)
    
    print(f"\nTotal items: {len(all_items)}")
    
    inserted = 0
    existing = 0
    for i, item in enumerate(all_items):
        if i % 10 == 0:
            print(f"  {i}/{len(all_items)}", flush=True)
        
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
        if c.fetchone():
            existing += 1
            continue
        
        html = fetch_page(item["url"])
        result = extract_content(html, item["url"])
        
        summary = (result["title"] + " " + SITE_NAME + " " + (result["content"][:200] if result["content"] else ""))[:500]
        
        c.execute(
            """INSERT INTO gov_raw (title, content, summary, publish_date, page_url, site_name, attachments, group_name, source_url)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (
                result["title"] or item["title"],
                result["content"],
                summary,
                item.get("date", ""),
                item["url"],
                SITE_NAME,
                result["attachments"],
                GROUP,
                item["url"],
            ),
        )
        inserted += 1
        if inserted % 20 == 0:
            conn.commit()
        time.sleep(0.5)
    
    conn.commit()
    conn.close()
    print(f"\nDone. Inserted: {inserted}, Existing: {existing}")


if __name__ == "__main__":
    main()
