#!/usr/bin/env python3
"""华容县人民政府 - 业务管理 (生态环境局华容分局)"""
import re, sys, os, json
import urllib.request, urllib.error
from urllib.parse import urljoin
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://www.huarong.gov.cn"
LIST_URL = BASE_URL + "/33159/37006/37008/37035/37247/index.htm"
SITE_NAME = "华容县-业务管理"
SITE_DISPLAY = "华容县人民政府"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
GROUP_NAME = "环评平台"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

ATTACH_RE = re.compile(r"\.(docx?|pdf|xlsx?|rar|zip)(?:[?#]|$)", re.I)

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("gb2312", errors="replace")
    except Exception as e:
        print(f"[WARN] fetch failed: {url} - {e}", file=sys.stderr)
        return ""

def to_abs(u, base):
    """相对路径 -> 绝对 URL（urljoin 正确处理 ../）"""
    u = (u or "").strip()
    if not u:
        return ""
    if u.startswith("//"):
        return "https:" + u
    if u.startswith(("http://", "https://")):
        return u
    return urljoin(base, u)

def get_list_url(page):
    """Get the list URL for a given page number (1-indexed)"""
    if page == 1:
        return LIST_URL
    elif 2 <= page <= 5:
        return f"{BASE_URL}/33159/37006/37008/37035/37247/index_{page-1}.htm"
    else:
        offset = (page - 1) * 20
        return f"{BASE_URL}/hrx/33159/37006/37008/37035/37247/index.jsp?pager.offset={offset}&pager.desc=false"

def extract_items(html):
    """Extract items from a list page"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for entry in soup.select("div.news-list-entry"):
        a = entry.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "") or a.get_text(strip=True)
        date_span = entry.find("span", class_="news-list-date")
        date_text = date_span.get_text(strip=True) if date_span else ""
        # Make absolute URL
        href = to_abs(href, LIST_URL)
        items.append({"title": title.strip(), "url": href, "date": date_text})
    return items

def _clean_fragment(fragment, base):
    """清洗 HTML 片段：unwrap 样式包装(span/font)，清 p/td/tr 内联 style，保留 a/img 并绝对化 href/src"""
    fsoup = BeautifulSoup(fragment, "html.parser")
    for tag in fsoup.find_all(["span", "font"]):
        tag.unwrap()
    for tag in fsoup.find_all("div"):
        tag.unwrap()
    for tag in fsoup.find_all(["p", "td", "tr", "th", "tbody", "table", "b", "strong", "u", "i", "em", "h1", "h2", "h3"]):
        if tag.has_attr("style"):
            del tag["style"]
        if tag.has_attr("class") and tag.name == "p":
            del tag["class"]
    for a in fsoup.find_all("a", href=True):
        a["href"] = to_abs(a["href"], base)
    for img in fsoup.find_all("img", src=True):
        src = img.get("src", "")
        if "icon_" in src or "/sysimage/" in src or "/filetypeimages/" in src:
            img.decompose()
            continue
        img["src"] = to_abs(img["src"], base)
    return str(fsoup)

def parse_detail(html, url):
    """Parse detail page content"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from h1 or meta
    title = ""
    title_div = soup.select_one("div.title")
    if title_div:
        title = title_div.get_text(strip=True)
    if not title and soup.title:
        t = soup.title.string.strip()
        title = re.sub(r"-\s*华容县人民政府\s*", "", t).strip()

    # Date from meta
    date_text = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
        if m:
            date_text = m.group(1)
    if not date_text:
        date_li = soup.select_one("li span:contains('发布日期')")
        if date_li:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", date_li.get_text())
            if m:
                date_text = m.group(1)

    # Content from div.content-wrapper
    content_div = soup.select_one("div.content-wrapper")
    if not content_div:
        print(f"[WARN] No content found: {url}", file=sys.stderr)
        return {"title": title, "date": date_text, "content": "", "attachments": []}

    attachments = []
    seen_att = set()

    def add_att(a):
        href = to_abs(a.get("href", ""), url)
        if not ATTACH_RE.search(href):
            return
        atitle = a.get_text(strip=True) or a.get("download", "") or href.rsplit("/", 1)[-1]
        if href not in seen_att:
            seen_att.add(href)
            attachments.append({"title": atitle, "url": href})

    parts = []
    # 遍历 content-wrapper 直接子节点，保留段落/表格结构
    for el in content_div.find_all(recursive=False):
        if el.name == "p":
            for a in el.find_all("a", href=True):
                add_att(a)
            inner = _clean_fragment(el.decode_contents(), url)
            txt = re.sub(r"<[^>]+>", "", inner).strip()
            if txt or "<a " in inner:
                parts.append(f"<p>{inner}</p>")
        elif el.name == "table":
            for a in el.find_all("a", href=True):
                add_att(a)
            t_html = _clean_fragment(el.decode_contents(), url)
            parts.append(f"<table>{t_html}</table>")
        elif el.name == "div":
            # div 内可能包裹表格
            t = el.find("table")
            if t:
                for a in t.find_all("a", href=True):
                    add_att(a)
                t_html = _clean_fragment(t.decode_contents(), url)
                parts.append(f"<table>{t_html}</table>")
            else:
                # 其他 div：提取附件后保留文本
                for a in el.find_all("a", href=True):
                    add_att(a)
                txt = el.get_text(" ", strip=True)
                if txt:
                    parts.append(f"<p>{txt}</p>")
        elif el.name == "img":
            src = el.get("src", "")
            if "icon_" in src or "/sysimage/" in src or "/filetypeimages/" in src:
                continue
            parts.append(f'<img src="{to_abs(src, url)}" alt="{el.get("alt", "")}">')
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(f"<p>{txt}</p>")

    # 附件区（wjxx）独立成段
    wjxx = soup.select_one("div.wjxx")
    if wjxx:
        for a in wjxx.find_all("a", href=True):
            add_att(a)
            ahref = to_abs(a.get("href", ""), url)
            if ATTACH_RE.search(ahref):
                atitle = a.get_text(strip=True) or a.get("download", "") or ahref.rsplit("/", 1)[-1]
                parts.append(f'<p><a href="{ahref}">{atitle}</a></p>')

    content = "\n".join(parts)
    return {"title": title, "date": date_text, "content": content, "attachments": attachments}


def main():
    # First, determine total pages from page 1
    html = fetch(LIST_URL)
    if not html:
        print("[ERROR] Cannot fetch list page", file=sys.stderr)
        sys.exit(1)

    total_pages = 1
    soup = BeautifulSoup(html, "html.parser")
    pages_div = soup.select_one("div.pages-l")
    if pages_div:
        max_page = 1
        for a in pages_div.find_all("a", href=True):
            # Check for the "57" link that goes to JSP
            text = a.get_text(strip=True)
            if text.isdigit():
                p = int(text)
                if p > max_page:
                    max_page = p
            # Also check index_{N}.htm links
            m = re.search(r"index_(\d+)\.htm", a["href"])
            if m:
                p = int(m.group(1)) + 1  # index_1.htm = page 2
                if p > max_page:
                    max_page = p
        # Check for JSP page link
        for a in pages_div.find_all("a", href=True):
            m = re.search(r"pager\.offset=(\d+)", a["href"])
            if m:
                p = int(m.group(1)) // 20 + 1
                if p > max_page:
                    max_page = p
        total_pages = max_page
        # Check page count text
        pages_text = soup.select_one("div.pages-r")
        if pages_text:
            m = re.search(r"共(\d+)页", pages_text.get_text())
            if m:
                total_pages = int(m.group(1))

    print(f"[INFO] Total pages: {total_pages}", file=sys.stderr)

    if _MAX_PG and total_pages > _MAX_PG:
        print(f"[INFO] Limiting to {_MAX_PG} pages (--pages={_MAX_PG})", file=sys.stderr)
        total_pages = _MAX_PG

    all_items = []
    for pg in range(1, total_pages + 1):
        url = get_list_url(pg)
        print(f"[INFO] Fetching page {pg}/{total_pages}: {url}", file=sys.stderr)
        html = fetch(url)
        if not html:
            print(f"[WARN] Empty response for page {pg}", file=sys.stderr)
            continue
        items = extract_items(html)
        print(f"[INFO] Found {len(items)} items on page {pg}", file=sys.stderr)
        all_items.extend(items)

    print(f"[INFO] Total items from list: {len(all_items)}", file=sys.stderr)

    # Connect to DB
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()

    new_count = 0
    empty_content = 0
    total_attachments = 0

    for item in all_items:
        # Check if exists by page_url
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            continue

        print(f"[INFO] Fetching detail: {item['title'][:50]}...", file=sys.stderr)
        html = fetch(item["url"])
        if not html:
            print(f"[WARN] Cannot fetch detail, skipping: {item['url']}", file=sys.stderr)
            continue

        detail = parse_detail(html, item["url"])

        title = detail["title"] or item["title"]
        date_text = detail["date"] or item["date"]
        content = detail["content"]
        attachments = detail["attachments"]

        # Clean title: remove site prefix patterns
        title = re.sub(r"^\s*(华容县人民政府|岳阳市生态环境局华容分局|生态环境局华容分局)\s+", "", title).strip()

        # Summary
        plain = re.sub(r"<[^>]+>", "", content)
        plain = re.sub(r"\s+", " ", plain).strip()
        summary = plain[:200] if plain else ""

        if not content:
            empty_content += 1
            print(f"[WARN] Empty content: {title}", file=sys.stderr)

        # Date rank
        date_rank = 0
        if date_text:
            m = re.search(r"(\d{4})-(\d{2})-(\d{2})", date_text)
            if m:
                date_rank = int(m.group(1) + m.group(2) + m.group(3))

        att_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"

        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank) VALUES (?,?,?,?,?,?,?,?,?)",
                (title, SITE_NAME, GROUP_NAME, item["url"], date_text, content, summary, att_json, date_rank)
            )
            if c.rowcount > 0:
                new_count += 1
                total_attachments += len(attachments)
        except Exception as e:
            print(f"[ERROR] DB insert failed: {title[:30]} - {e}", file=sys.stderr)

    conn.commit()
    conn.close()

    print(f"\n[RESULT] 华容县-业务管理: {new_count} new, {empty_content} empty content, {total_attachments} attachments")


if __name__ == "__main__":
    main()
