#!/usr/bin/env python3
"""Crawl sthjj.huainan.gov.cn - 淮南市生态环境局 环评公示"""
import sys, re, sqlite3, urllib.request, urllib.error, traceback
from datetime import datetime

BASE = "https://sthjj.huainan.gov.cn"
SITE_NAME = "hn_sthj_hpgs"
DB_PATH = "/root/search.db"
COLUMN_ID = "18125225"
PER_PAGE = 20
MAX_PAGES = 22
MAX_ARTICLES = 435

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE + "/hbyw/xmgl/hpgs/index.html",
}

def fetch(url):
    req = urllib.request.Request(url, headers=headers)
    resp = urllib.request.urlopen(req, timeout=15)
    return resp.read().decode("utf-8", errors="replace")

def extract_items(html):
    """Extract (title, url, date) from list page HTML."""
    items = []
    # Lonsun format: <li><a href=... title=TITLE>...<span class=date>DATE</span></li>
    pattern = re.compile(
        r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span[^>]*class="[^"]*date[^"]*"[^>]*>(\d{4}-\d{2}-\d{2})',
        re.S
    )
    for m in pattern.finditer(html):
        href = m.group(1)
        title = m.group(2).strip()
        date = m.group(3)
        if href and title and len(title) > 5:
            if not href.startswith("http"):
                href = BASE + href
            items.append((title, href, date))
    return items


def extract_clean_text(html_chunk):
    """Extract text from HTML, preserving paragraph breaks."""
    text = re.sub(r"</p>", "\n\n", html_chunk, flags=re.I)
    text = re.sub(r"<br\s*/?>", "\n", text, flags=re.I)
    text = re.sub(r"</div>", "\n\n", text, flags=re.I)
    text = re.sub(r"</?(?:b|span|font|strong|em|u|i|a)\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"<p\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"\n{3,}", "\n\n", text)
    lines = [l.strip() for l in text.split("\n")]
    text = "\n".join(lines)
    text = re.sub(r"\n{3,}", "\n\n", text)
    return text.strip()


def extract_attachments(html):
    """Extract attachment links, return as markdown."""
    parts = []
    links = re.findall(
        r'<a\s[^>]*href="([^"]*download[^"]*)"[^>]*>([^<]+)</a>',
        html, re.I
    )
    for href, name in links:
        full_url = href if href.startswith("http") else BASE + href
        parts.append(f"[{name.strip()}]({full_url})")
    if not parts:
        # Also check for file extensions
        links2 = re.findall(
            r'<a\s[^>]*href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]+)</a>',
            html, re.I
        )
        for href, name in links2:
            full_url = href if href.startswith("http") else BASE + href
            parts.append(f"[{name.strip()}]({full_url})")
    return "\n".join(parts)


def get_detail(url):
    """Get (title, date, content, source) from detail page."""
    html = fetch(url)

    # Title
    title = ""
    tm = re.search(r"<title>([^<]+)</title>", html)
    if tm:
        title = re.sub(r"\s*[-–—|_].*$", "", tm.group(1)).strip()

    # Date
    date = ""
    dm = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if dm:
        date = dm.group(1)
    if not date:
        dm2 = re.search(r"(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}", html)
        if dm2:
            date = dm2.group(1)

    # Source
    source = ""
    sm = re.search(r"来源[：:]\s*([^<]+)", html)
    if sm:
        source = sm.group(1).strip()

    # Content from div.wzcon
    content = ""
    idx = html.find('class="wzcon')
    if idx < 0:
        idx = html.find('class="j-fontContent')
    if idx > 0:
        start = html.rfind("<div", 0, idx)
        if start < 0:
            start = idx
        pos = html.find(">", idx) + 1

        depth = 1
        end = pos
        while depth > 0 and end < len(html):
            no = html.find("<div", end)
            nc = html.find("</div>", end)
            if nc < 0:
                break
            if no >= 0 and no < nc:
                depth += 1
                end = html.find(">", no) + 1
            else:
                depth -= 1
                end = nc + 6

        chunk = html[pos:end-6] if depth == 0 else html[pos:]
        content = extract_clean_text(chunk)

    # Attachments
    att = extract_attachments(html)
    if att:
        if content:
            content += "\n\n"
        content += att

    return title, date, content, source


def save_to_db(records):
    if not records:
        return 0
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()
    count = 0
    for r in records:
        existing = c.execute("SELECT id FROM gov_raw WHERE source_url=?", (r["url"],)).fetchone()
        if existing:
            print(f"  SKIP: {r['title'][:40]}")
            continue
        content = (r["content"] or "")[:5000]
        c.execute("""INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, content)
            VALUES (?,?,?,?,?,?,?,?)""",
            (SITE_NAME, r["url"], r["url"], r["title"], r["date"],
             int(datetime.strptime(r["date"], "%Y-%m-%d").timestamp()) if r["date"] else 0,
             content[:200], content))
        conn.commit()
        count += 1
        print(f"  OK: {r['title'][:50]} | {r['date']}")
    conn.close()
    return count


def main():
    print(f"=== Crawling {SITE_NAME} ===")

    all_items = []

    # Page 1: index.html
    html1 = fetch(BASE + "/hbyw/xmgl/hpgs/index.html")
    items = extract_items(html1)
    print(f"Page 1: {len(items)} items")
    all_items.extend(items)

    # Pages 2-22 via column API
    for page in range(2, MAX_PAGES + 1):
        url = f"{BASE}/content/column/{COLUMN_ID}?pageIndex={page}"
        try:
            html = fetch(url)
            items = extract_items(html)
            if not items:
                print(f"Page {page}: no items, stopping")
                break
            print(f"Page {page}: {len(items)} items")
            all_items.extend(items)
            if len(all_items) >= MAX_ARTICLES:
                break
        except Exception as e:
            print(f"Page {page}: error - {e}")
            break

    print(f"\n=== Total: {len(all_items)} items ===")

    # Deduplicate by URL
    seen = set()
    unique = []
    for title, url, date in all_items:
        if url not in seen:
            seen.add(url)
            unique.append((title, url, date))
    print(f"Unique: {len(unique)} items")

    # Fetch details
    records = []
    for i, (title, url, date) in enumerate(unique):
        print(f"\n[{i+1}/{len(unique)}] {title[:50]}...")
        try:
            d_title, d_date, content, source = get_detail(url)
            records.append({
                "title": d_title or title,
                "url": url,
                "date": d_date or date,
                "content": content,
                "source": source,
            })
        except Exception as e:
            print(f"  ERROR: {e}")

    saved = save_to_db(records)
    print(f"\n=== Complete: {saved}/{len(records)} new records ===")


if __name__ == "__main__":
    main()
