#!/usr/bin/env python3
"""沾益区人民政府 — 环评管理

华海CMS
列表: https://www.zhanyi.gov.cn/pub/special/378.html
分页: ?cate=378&page=N (共15页, 约13条/页)
详情: /news/1hpgl/{id}.html
正文: div.web_con.wow_fadeInUp (p/table/img)
标题: h5 (详情页完整标题, 列表页标题有省略号)
"""

import re, sys, os, json, time
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://www.zhanyi.gov.cn"
LIST_URL = "/pub/special/378.html"
SITE_NAME = "沾益区-环评管理"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
    except Exception as e:
        print("[WARN] fetch failed: %s - %s" % (url, e), file=sys.stderr)
        return ""


def extract_list_items(html):
    """解析列表页, 返回 (title, date, url) 列表"""
    items = []
    soup = BeautifulSoup(html, "html.parser")

    ol = soup.find("ol", class_="news_list")
    if not ol:
        print("[WARN] 未找到 ol.news_list", file=sys.stderr)
        return items

    for li in ol.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get_text(strip=True)
        if not title or len(title) < 5:
            continue

        # 列表页标题有省略号, 标记但保留, 后面用详情页标题覆盖
        has_ellipsis = "..." in title

        # 日期在li文本中 (a标签后面的文本)
        date_text = ""
        li_text = li.get_text(strip=True)
        # 去掉标题部分获取日期
        a_text = a.get_text(strip=True)
        after = li_text[len(a_text):].strip()
        m = re.search(r'(\d{4}-\d{2}-\d{2})', after)
        if m:
            date_text = m.group(1)

        # 补全URL
        if href.startswith("http"):
            full_url = href
        elif href.startswith("/"):
            full_url = BASE_URL + href
        else:
            full_url = BASE_URL + "/" + href.lstrip("./")

        items.append((title, date_text, full_url, has_ellipsis))

    return items


def parse_detail(html, url):
    """解析详情页, 返回 (title, date, content, attachments_str)"""
    soup = BeautifulSoup(html, "html.parser")

    # === Title ===
    title = ""
    h5 = soup.find("h5")
    if h5:
        title = h5.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            raw = title_tag.get_text(strip=True)
            # 去站点后缀
            title = re.sub(r'\s*[-_—].*$', '', raw).strip()

    # === Date ===
    date_text = ""
    date_p = soup.find("p", string=re.compile(r'发布时间'))
    if date_p:
        txt = date_p.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m:
            date_text = m.group(1)

    # === Content ===
    content_parts = []

    # 正文在 div.web_con.wow_fadeInUp
    web_con = soup.find("div", class_="web_con")
    if not web_con:
        # fallback: 找常见正文容器
        web_con = soup.find("div", class_=re.compile(r'web_con|content|article|detail|zoom', re.I))
    if not web_con:
        web_con = soup.find("div", class_=lambda c: c and "web" in c)

    if web_con:
        for el in web_con.find_all(["p", "div", "table", "img"], recursive=True):
            if el.name == "img":
                src = el.get("src", "")
                alt = el.get("alt", "")
                if src:
                    if src.startswith("http"):
                        full_src = src
                    elif src.startswith("/"):
                        full_src = BASE_URL + src
                    elif src.startswith("./"):
                        base = url[:url.rfind("/")] if "/" in url else url
                        full_src = base + "/" + src[2:]
                    elif "/" in url[:url.rfind("/")]:
                        base = url[:url.rfind("/")]
                        full_src = base + "/" + src
                    else:
                        full_src = src
                    content_parts.append("![%s](%s)" % (alt, full_src))
                continue

            if el.name == "table":
                rows = []
                for tr in el.find_all("tr"):
                    cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                    if any(cells):
                        rows.append(" | ".join(cells))
                if rows:
                    content_parts.append("\n".join(rows))
                continue

            # p 或 div
            txt = el.get_text(strip=True)
            if txt and len(txt) > 2:
                # 过滤无效文本
                if re.match(r'^[\[（(]?\s*(大|中|小)\s*[\]）)]?\s*$', txt):
                    continue
                if "打印" in txt and "关闭" in txt:
                    continue
                if "字号" in txt and "[" in txt:
                    continue
                if txt.startswith("附件") and len(txt) < 15:
                    continue
                content_parts.append(txt)

    content = "\n\n".join(content_parts)

    # === Attachments ===
    attachments = []
    seen_hrefs = set()
    if web_con:
        for a in web_con.find_all("a", href=True):
            h = a["href"].lower().strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ppt|pptx)$', h):
                if h in seen_hrefs:
                    continue
                seen_hrefs.add(h)
                fname = a.get_text(strip=True) or os.path.basename(a["href"])
                if h.startswith("http"):
                    full_url = a["href"]
                elif h.startswith("/"):
                    full_url = BASE_URL + a["href"]
                elif h.startswith("./"):
                    base = url[:url.rfind("/")] if "/" in url else url
                    full_url = base + "/" + h[2:]
                elif "/" in url[:url.rfind("/")]:
                    base = url[:url.rfind("/")]
                    full_url = base + "/" + h
                else:
                    full_url = a["href"]
                attachments.append("[%s](%s)" % (fname, full_url))

    attach_str = "|".join(attachments)

    if attachments:
        attach_section = "\n\n---\n**附件：**\n" + "\n".join(attachments)
        if not content:
            content = attach_section.lstrip()
        else:
            content += attach_section

    return title, date_text, content, attach_str


def store_item(title, date_str, content, page_url, attachments):
    max_retries = 5
    for attempt in range(max_retries):
        try:
            conn = sqlite3.connect(DB, timeout=30)
            conn.execute("PRAGMA busy_timeout=30000")
            c = conn.cursor()
            summary = (content.strip()[:200] if content.strip() else "")
            date_rank = 0
            if date_str:
                try:
                    from datetime import datetime
                    date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
                except:
                    pass
            c.execute("""INSERT OR IGNORE INTO gov_raw
                         (title, publish_date, content, page_url, source_url, site_name, summary, date_rank, attachments)
                         VALUES (?,?,?,?,?,?,?,?,?)""",
                      (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(),
                       SITE_NAME, summary, date_rank, attachments))
            a = c.rowcount
            conn.commit()
            conn.close()
            return a
        except sqlite3.OperationalError as e:
            if "locked" in str(e) and attempt < max_retries - 1:
                time.sleep(1)
                continue
            print("[ERROR] 写入失败: %s" % e, file=sys.stderr)
            return 0
        except Exception as e:
            print("[ERROR] 写入失败: %s" % e, file=sys.stderr)
            return 0
    return 0


def main():
    print("[START] %s" % SITE_NAME, file=sys.stderr)

    # 先获取第1页看总页数
    first_url = BASE_URL + LIST_URL
    first_html = fetch(first_url)
    if not first_html:
        print("[FATAL] 首页获取失败", file=sys.stderr)
        return

    # 解析总页数: 从分页链接找最大页码
    soup = BeautifulSoup(first_html, "html.parser")
    total_pages = 1
    for a in soup.find_all("a", href=True):
        m = re.search(r'page=(\d+)', a["href"])
        if m:
            pg = int(m.group(1))
            if pg > total_pages:
                total_pages = pg
    print("[INFO] 共 %d 页" % total_pages, file=sys.stderr)

    max_pg = total_pages
    if _MAX_PG and _MAX_PG < max_pg:
        max_pg = _MAX_PG
        print("[INFO] 限制为前 %d 页" % max_pg, file=sys.stderr)

    total_new = 0
    total_skip = 0
    total_err = 0

    for pg in range(1, max_pg + 1):
        if pg == 1:
            page_url = first_url
        else:
            page_url = first_url + "?cate=378&page=%d" % pg

        print("[PAGE] %d/%d: %s" % (pg, max_pg, page_url), file=sys.stderr)

        if pg == 1:
            html = first_html
        else:
            html = fetch(page_url)
            if not html:
                print("[ERR] 第%d页获取失败" % pg, file=sys.stderr)
                continue

        items = extract_list_items(html)
        print("[INFO] 第%d页 %d 条" % (pg, len(items)), file=sys.stderr)

        for title, date_str, detail_url, has_ellipsis in items:
            detail_html = fetch(detail_url)
            if not detail_html:
                print("  [ERR] 详情页获取失败: %s" % detail_url, file=sys.stderr)
                total_err += 1
                continue

            d_title, d_date, d_content, attachments = parse_detail(detail_html, detail_url)
            # 优先用详情页标题（完整无省略号）
            final_title = d_title or title
            final_date = d_date or date_str

            if final_title and d_content:
                st = store_item(final_title, final_date, d_content, detail_url, attachments)
                if st:
                    total_new += 1
                    print("  [OK] %s %s" % (final_date, final_title[:60]), file=sys.stderr)
                else:
                    total_skip += 1
                    print("  [DUP] %s %s" % (final_date, final_title[:60]), file=sys.stderr)
            else:
                total_err += 1
                print("  [ERR] 解析失败: %s: title=%s, content=%s" %
                      (detail_url, bool(d_title), bool(d_content)), file=sys.stderr)

            time.sleep(0.3)

        time.sleep(0.5)

    print("[DONE] 新增: %d, 跳过: %d, 失败: %d" % (total_new, total_skip, total_err), file=sys.stderr)
    print("OK: new=%d skip=%d err=%d" % (total_new, total_skip, total_err))


if __name__ == "__main__":
    main()
