#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
应城市生态环境局 - 业务工作 爬虫 (v2 修复版)
CMS: 孝感市统一平台（Hanweb/JPAAS htmledit_views）
问题修复:
  1. 域名: www.yingcheng.gov.cn 已空壳(984B) → 改用 zwgk.yingcheng.gov.cn
  2. 正文: 嵌套span字体标注致 get_text(' ') 空格污染 → 保内部HTML+unwrap, 附件独立段
List: /c/ycssthjj/ywgz.jhtml (P1), ywgz_{N}.jhtml (PN, 1-indexed)
Detail: /c/ycssthjj/ywgz/{article_id}.jhtml
每页15条
用法:
  python3 crawl_yingcheng_ywgz.py --pages=5
"""

import requests, re, json, sqlite3, time, sys
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "应城市生态环境局-业务工作"
CATEGORY = "业务工作"
GROUP = "湖北"
BASE_URL = "http://zwgk.yingcheng.gov.cn"
LIST_PATH = "/c/ycssthjj/ywgz"
MAX_PAGES = 5
CUTOFF = (time.strftime("%Y-%m-%d", time.localtime(time.time() - 3*365*86400)))

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

SKIP_TEXTS = [
    "【打印本页】", "【关闭窗口】", "打印本页", "关闭窗口",
    "转载分享：", "浏览量：", "相关附件：", "相关稿件：",
    "扫一扫在手机打开", "附件：",
]


def fetch(url, retries=4):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=25)
            r.encoding = "utf-8"
            if r.status_code == 200 and len(r.text) > 2000:
                return r.text
        except Exception:
            pass
        time.sleep(2 * (i + 1))
    return None


def abs_url(href):
    if not href:
        return ""
    if href.startswith("http"):
        return href
    return urljoin(BASE_URL, href)


def table_to_html(table_tag):
    """表格保留为 HTML"""
    for t in table_tag.find_all(["script", "style"]):
        t.decompose()
    return str(table_tag)


def extract_block(content_div, detail_url):
    """正文: p/pre 保内部HTML(span unwrap), table保留, 附件独立段, 图片查看链接"""
    parts = []
    for s in content_div.find_all(["script", "style"]):
        s.decompose()

    # 附件: 先收集 (仅文件类链接)
    attach_links = []
    for a in content_div.find_all("a", href=True):
        href = a.get("href", "")
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt|wps|et|dwg)$", href, re.I):
            text = a.get_text(strip=True) or href.split("/")[-1].split("?")[0]
            attach_links.append((text, abs_url(href)))

    # 处理 a/img: 非附件a保留为内联链接, img转查看链接
    for a in content_div.find_all("a", href=True):
        href = a.get("href", "")
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt|wps|et|dwg)$", href, re.I):
            continue  # 附件已收集, 删除原链接
    for a in content_div.find_all("a", href=True):
        href = a.get("href", "")
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt|wps|et|dwg)$", href, re.I):
            a.decompose()

    for img in content_div.find_all("img"):
        src = img.get("src", "")
        if not src or "icon_" in src or "fileTypeImages" in src:
            img.decompose()
            continue
        full = abs_url(src)
        alt = img.get("alt", "") or "查看图片"
        img.replace_with(f'<p><a href="{full}">{alt}</a></p>')

    for el in content_div.find_all(["p", "pre"]):
        if el.find_parent("table"):
            continue
        # span 等 inline 标签 unwrap (去字体标注)
        for tag in el.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
            tag.unwrap()
        txt = el.get_text(separator="", strip=True)
        if not txt:
            continue
        # 段落内剩余链接保 HTML
        inner = "".join(str(c) for c in el.contents if getattr(c, "name", None) in ("a", "br") or isinstance(c, str))
        if inner.strip():
            inner = re.sub(r"\s+", " ", inner)
            parts.append(f"<p>{inner}</p>")
        else:
            parts.append(f"<p>{txt}</p>")

    for table in content_div.find_all("table"):
        if len(table.get_text(strip=True)) >= 10:
            parts.append(table_to_html(table))

    # 附件独立段 (放末尾)
    seen = set()
    for text, url in attach_links:
        if url and url not in seen:
            seen.add(url)
            parts.append(f'<p><a href="{url}">{text}</a></p>')

    # 过滤噪声
    filtered = []
    for p in parts:
        if any(sk in p for sk in SKIP_TEXTS):
            continue
        if re.match(r"^<p>(来源|发布日期)[：:]\s*", p):
            continue
        filtered.append(p)

    return "\n\n".join(filtered)


def fetch_detail(url, list_title=""):
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    if not title:
        nt = soup.find("div", class_="news-title")
        if nt:
            h2 = nt.find("h2")
            if h2:
                title = h2.get_text(strip=True)
    if not title:
        title = list_title
    if not title:
        ttag = soup.find("title")
        if ttag:
            title = ttag.get_text(strip=True)
            title = re.sub(r"\s*-\s*应城市生态环境局\s*$", "", title).strip()

    date_str = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        date_str = md["content"].strip()[:10]
    if not date_str:
        dm = re.search(r"发布日期[：:](\d{4}-\d{1,2}-\d{1,2})", html)
        if dm:
            date_str = dm.group(1)

    content_div = soup.find("article", class_="htmledit_views")
    if not content_div:
        content_div = soup.find("div", class_="htmledit_views")
    if not content_div:
        content_div = soup.find("div", id="js_contentBox")
    if not content_div:
        content_div = soup.find("div", class_="news-box")

    content = ""
    if content_div:
        content = extract_block(content_div, url)

    # Fallback
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title or list_title}</a></p>'

    return title, date_str, content


def extract_list_page(page_num):
    if page_num == 1:
        url = f"{BASE_URL}{LIST_PATH}.jhtml"
    else:
        url = f"{BASE_URL}{LIST_PATH}_{page_num}.jhtml"

    html = fetch(url)
    if not html:
        print(f"  Page {page_num} fetch fail")
        return []

    soup = BeautifulSoup(html, "html.parser")
    articles = []
    for card in soup.find_all("div", class_="card"):
        body = card.find("div", class_="card-body")
        if not body:
            continue
        ul = body.find("ul", class_="news-list")
        if not ul:
            continue
        for li in ul.find_all("li"):
            a = li.find("a", class_="news-title")
            if not a:
                continue
            href = a.get("href", "")
            if not href:
                continue
            title = a.get("title", "") or a.get_text(strip=True)
            if not title or len(title) < 5:
                continue
            full_url = abs_url(href)
            date_span = li.find("span", class_="news-time")
            date = date_span.get_text(strip=True) if date_span else ""
            articles.append((title, full_url, date))
        break
    return articles


def main():
    max_pages = MAX_PAGES
    args = sys.argv[1:]
    if "--pages=" in " ".join(args):
        max_pages = int(re.search(r"--pages=(\d+)", " ".join(args)).group(1))
    elif args and args[0].isdigit():
        max_pages = int(args[0])

    print(f"Scraping {SITE_NAME}, max pages: {max_pages}", flush=True)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=15000")
    total_new = total_found = 0

    for pn in range(1, max_pages + 1):
        articles = extract_list_page(pn)
        if not articles:
            print(f"  Page {pn}: 0 articles (end)", flush=True)
            break
        total_found += len(articles)
        for title, page_url, date in articles:
            if date and date < CUTOFF:
                continue
            cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if cur.fetchone():
                continue
            det_title, det_date, det_content = fetch_detail(page_url, title)
            final_title = det_title or title
            final_date = det_date or date
            final_content = det_content or f'<p><a href="{page_url}">{final_title}</a></p>'
            summary = re.sub(r"\s+", " ", final_content[:200]).strip()
            try:
                cur = conn.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (site_name, source_url, page_url, title, publish_date, summary, content, category, group_name, script_name)
                       VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, page_url, page_url, final_title.strip(), final_date,
                     summary, final_content, CATEGORY, GROUP, "crawl_yingcheng_ywgz.py"),
                )
                conn.commit()
                if cur.rowcount > 0:
                    total_new += 1
            except sqlite3.OperationalError as e:
                print(f"  DB error: {e}", file=sys.stderr)
            time.sleep(0.3)
        print(f"  Page {pn}: {len(articles)} found, {total_new} new so far", flush=True)

    conn.commit()
    conn.close()
    print(f"\nDone! Total found: {total_found}, New: {total_new}", flush=True)


if __name__ == "__main__":
    main()
