#!/usr/bin/env python3
"""
crawl_linzi.py - 临淄生态环境分局-拟受理（环评受理公示）
淄博市政务公开平台 - 通过CDN IP访问
"""
import requests
from bs4 import BeautifulSoup
import re
import sqlite3
import sys
import time
import urllib3
import urllib.parse

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

BASE_URL = "http://www.linzi.gov.cn"
LIST_PATH = "/gongkai/site_lzqsthjj/channel_c_5f9f6cdc1ebfe2f7fcddefe1_n_1605681498.3499/"
CDN_IP = "221.194.141.170"
TOTAL_ITEMS = 716
DB_PATH = "/root/search.db"
SITE_NAME = "临淄生态环境分局-拟受理"
SLEEP = 0.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def get_page(url, params=None):
    """Fetch via CDN IP with Host header"""
    real_url = url.replace("www.linzi.gov.cn", CDN_IP)
    headers = {"Host": "www.linzi.gov.cn"}
    try:
        resp = session.get(real_url, headers=headers, params=params, timeout=30)
        resp.encoding = "utf-8"
        return resp
    except Exception as e:
        print(f"  [NET ERR] {e}")
        return None


def parse_list(html):
    """Parse article list from HTML"""
    items = []
    for m in re.finditer(
        r'<a[^>]*href="([^"]*doc_[a-f0-9]+\.html)"[^>]*>(.*?)</a>',
        html,
        re.DOTALL,
    ):
        href = m.group(1)
        title = re.sub(r"<[^>]+>", "", m.group(2)).strip()
        title = re.sub(r"\s+", " ", title).strip()

        ctx_after = html[m.end() : m.end() + 200]
        date_m = re.search(r"zfxxgk-list-time[^>]*>(\d{4}-\d{2}-\d{2})", ctx_after)
        date_str = date_m.group(1) if date_m else ""

        # Make absolute URL
        if href.startswith("/"):
            url = BASE_URL + href
        elif href.startswith("http"):
            if "linzi.gov.cn" not in href:
                continue
            url = href
        else:
            url = BASE_URL + "/" + href

        items.append({"title": title, "url": url, "date": date_str})

    return items


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(html):
    """Parse article detail - handles p, table, img in order"""
    result = {"title": "", "date": "", "content": ""}

    # Title and date from meta
    m_title = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    if m_title:
        result["title"] = m_title.group(1).strip()

    m_date = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m_date:
        raw = m_date.group(1).strip()
        dm = re.search(r"\d{4}-\d{2}-\d{2}", raw)
        if dm:
            result["date"] = dm.group()

    # Get the inner content div
    match = re.search(
        r'<div[^>]*id="details-content"[^>]*>(.*?)</div>', html, re.DOTALL
    )
    if not match:
        return result

    content_html = match.group(1)
    parts = []

    # Walk through top-level children in order: p, table (with wrapper div), img
    # Strategy: find all elements in order
    pos = 0
    while pos < len(content_html):
        # Check for <p> at current position
        p_match = re.match(r"<p[^>]*>(.*?)</p>", content_html[pos:], re.DOTALL)
        if p_match:
            p_text = re.sub(r"<[^>]+>", "", p_match.group(1)).strip()
            p_text = re.sub(r"\s+", " ", p_text).strip()
            p_text = p_text.replace("&nbsp;", " ").replace("&mdash;", "—")
            if p_text:
                parts.append(p_text)
            pos += p_match.end()
            continue

        # Check for <table> (possibly inside a wrapper div)
        table_match = re.match(
            r"(?:<div[^>]*>\s*)?<table[^>]*>(.*?)</table>(?:\s*</div>)?",
            content_html[pos:],
            re.DOTALL,
        )
        if table_match:
            table_html = table_match.group(0)
            md_table = html_table_to_html(table_html)
            if md_table:
                parts.append(md_table)
            pos += table_match.end()
            continue

        # Check for <img>
        img_match = re.match(
            r'<img[^>]*src="([^"]+)"[^>]*>', content_html[pos:], re.DOTALL
        )
        if img_match:
            src = img_match.group(1)
            if not src.startswith("http"):
                src = BASE_URL + src
            parts.append(f"![图片]({src})")
            pos += img_match.end()
            continue

        # Check for <h2> or other headers
        h_match = re.match(
            r"<(h[23])[^>]*>(.*?)</\1>", content_html[pos:], re.DOTALL
        )
        if h_match:
            h_text = re.sub(r"<[^>]+>", "", h_match.group(2)).strip()
            h_text = h_text.replace("&nbsp;", " ").strip()
            if h_text:
                parts.append(h_text)
            pos += h_match.end()
            continue

        # Skip other elements (div wrappers, br, etc.)
        tag_match = re.match(r"<[^>]+>", content_html[pos:])
        if tag_match:
            pos += tag_match.end()
            continue

        # Skip whitespace/newlines
        ws_match = re.match(r"\s+", content_html[pos:])
        if ws_match:
            pos += ws_match.end()
            continue

        # If nothing matched, move forward one char to avoid infinite loop
        pos += 1

    result["content"] = "\n\n".join(parts).strip()
    return result


def save_to_db(url, title, date, content):
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    try:
        c.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (url, SITE_NAME),
        )
        if c.fetchone():
            conn.close()
            return "skip"

        summary = content[:200]
        has_table = 1 if "| ---" in content else 0
        c.execute(
            """INSERT OR REPLACE INTO gov_raw
            (page_url, title, publish_date, site_name, content, attachments, summary, has_table)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
            (url, title, date, SITE_NAME, content, "", summary, has_table),
        )
        conn.commit()
        conn.close()
        return "new"
    except Exception as e:
        conn.close()
        return f"error:{e}"


def crawl(pages=3):
    max_pages = (TOTAL_ITEMS + 9) // 10
    total_new = 0
    total_skip = 0

    for page in range(1, min(pages + 1, max_pages + 1)):
        params = {"open": "fdzdgknr", "nr_page": str(page), "nr_per_page": "10"}
        resp = get_page(BASE_URL + LIST_PATH, params=params)
        if resp is None or resp.status_code != 200:
            print(f"[WARN] Page {page} failed")
            continue

        items = parse_list(resp.text)
        if not items:
            print(f"[WARN] Page {page} has no items")
            continue

        print(f"[LIST] Page {page}: {len(items)} items")
        has_content = False

        for item in items:
            if "linzi.gov.cn" not in item["url"]:
                total_skip += 1
                continue

            resp2 = get_page(item["url"])
            if resp2 is None or resp2.status_code != 200:
                print(f"  [FAIL] HTTP ?: {item['title'][:50]}")
                total_skip += 1
                continue

            detail = parse_detail(resp2.text)
            content = detail.get("content", "")
            if not content:
                print(f"  [EMPTY] {item['title'][:50]}")
                total_skip += 1
                continue

            title = detail.get("title") or item["title"]
            date = detail.get("date") or item.get("date", "")

            result = save_to_db(item["url"], title, date, content)
            if result == "new":
                total_new += 1
                has_content = True
                print(f"  [OK] {title[:60]}")
            elif result == "skip":
                total_skip += 1
                print(f"  [SKIP] {title[:50]}")
            else:
                total_skip += 1
                print(f"  [ERR] {title[:50]} - {result}")

            time.sleep(SLEEP)

        if not has_content and page > 1:
            print(f"[INFO] Page {page} no new content, stopping")
            break

        time.sleep(SLEEP)

    return total_new, total_skip


if __name__ == "__main__":
    import argparse

    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=3)
    args = parser.parse_args()

    new, skip = crawl(pages=args.pages)
    print(f"\n[RESULT] {SITE_NAME}: {new} new, {skip} skip")
