#!/usr/bin/env python3
"""天长市-建设项目环境影响评价 爬虫 (滁州信息公开平台/PowerEasy变体)"""
import sys, os, re, json, time, argparse
from datetime import datetime
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
import urllib.parse

BASE_URL = "https://www.tianchang.gov.cn"
SITE_NAME = "天长市-建设项目环境影响评价"
ORGAN_ID = "161054628"
CAT_ID = "170002338"
SITE_ID = "2653861"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE_URL}/public/column/{ORGAN_ID}?type=4&catId={CAT_ID}&action=list&nav=3",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
    "X-Requested-With": "XMLHttpRequest",
}

FETCH_DELAY = 0.5
MAX_RETRIES = 3

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list(page: int) -> list:
    """通过 AJAX API 获取列表数据 (JSON)"""
    url = f"{BASE_URL}/czxxgk/site/label/8888"
    data = {
        "labelName": "publicInfoList",
        "siteId": SITE_ID,
        "organId": ORGAN_ID,
        "pageSize": "20",
        "pageIndex": str(page),
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "50",
        "type": "4",
        "action": "list",
        "result": "",
        "isJson": "true",
        "keyWords": "",
        "isSetValue": "true",
        "catId": CAT_ID,
        "catIds": "",
    }
    for attempt in range(MAX_RETRIES):
        try:
            r = session.post(url, data=data, timeout=30)
            if r.status_code != 200:
                print(f"  ⚠️ 列表第{page}页 HTTP {r.status_code}, 重试 {attempt+1}/{MAX_RETRIES}")
                time.sleep(2)
                continue
            j = r.json()
            items = j.get("data", [])
            if not items:
                print(f"  ⚠️ 第{page}页无数据")
                return []
            print(f"  ✓ 第{page}页: {len(items)}条 (共{j.get('total','?')}条, {j.get('pageCount','?')}页)")
            return items
        except Exception as e:
            print(f"  ⚠️ 第{page}页出错: {e}, 重试 {attempt+1}/{MAX_RETRIES}")
            time.sleep(3)
    return []


def fetch_detail(content_id: str) -> tuple:
    """获取详情页：标题、日期、正文、附件"""
    url = f"{BASE_URL}/public/{ORGAN_ID}/{content_id}.html"
    for attempt in range(MAX_RETRIES):
        try:
            r = session.get(url, timeout=30)
            if r.status_code != 200:
                print(f"    ⚠️ 详情 {content_id} HTTP {r.status_code}, 重试 {attempt+1}/{MAX_RETRIES}")
                time.sleep(2)
                continue
            # 修复：页面无charset头，requests默认ISO-8859-1，实际为UTF-8
            r.encoding = r.apparent_encoding or 'utf-8'
            soup = BeautifulSoup(r.text, "html.parser")

            # 标题：meta[ArticleTitle] 或 h1.newstitle
            title = ""
            meta_title = soup.select_one("meta[ArticleTitle]")
            if meta_title and meta_title.get("content", "").strip():
                title = meta_title.get("content", "").strip()
            if not title:
                h1 = soup.select_one("h1.newstitle")
                if h1:
                    title = h1.get_text(strip=True)
            if not title:
                h1 = soup.select_one("h1")
                if h1:
                    title = h1.get_text(strip=True)
            title = re.sub(r"\s+", " ", title).strip()

            # 日期：meta[PubDate] 或 页面信息
            pub_date = ""
            meta_date = soup.select_one("meta[PubDate]")
            if meta_date:
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", meta_date.get("content", ""))
                if m:
                    pub_date = m.group(1)
            if not pub_date:
                date_span = soup.select_one("span.sp[tabindex]")
                if date_span:
                    m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_span.get_text())
                    if m:
                        pub_date = m.group(1)

            # 正文容器
            content_div = (
                soup.select_one("div.gkwz_contnet.j-fontContent")
                or soup.select_one("div.j-fontContent.clearfix")
                or soup.select_one("div.wzcon.clearfix")
            )
            if not content_div:
                print(f"    ⚠️ 详情 {content_id}: 未找到正文容器")
                return title, pub_date, "", []

            paragraphs = []
            attachments = []

            for elem in content_div.find_all(["p", "table", "img"], recursive=True):
                style = elem.get("style", "")
                if "display:none" in style.replace(" ", ""):
                    continue
                if elem.name == "p" and elem.find_parent("table"):
                    continue

                if elem.name == "p":
                    p_text = elem.get_text(strip=True)
                    if p_text:
                        # 检查附件链接
                        for a in elem.find_all("a", href=True):
                            href = a.get("href", "")
                            if any(ext in href.lower()
                                   for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar",
                                               "/group5/", "/upload/", "/down"]):
                                full_url = urljoin(BASE_URL, href)
                                a_text = a.get_text(strip=True)
                                if a_text:
                                    attachments.append({"name": a_text, "url": full_url})
                        paragraphs.append(p_text)

                elif elem.name == "table":
                    table_md = html_table_to_html(elem)
                    if table_md:
                        paragraphs.append(table_md)

                elif elem.name == "img":
                    src = elem.get("src", "")
                    alt = elem.get("alt", "")
                    if src and not src.endswith(".gif"):
                        full_src = urljoin(BASE_URL, src)
                        paragraphs.append(f"![{alt}]({full_src})")

            content = "\n\n".join(paragraphs)
            content = content.replace("\u3000", " ").replace("\xa0", " ")
            content = re.sub(r"[ \t]+", " ", content)

            return title, pub_date, content, attachments

        except Exception as e:
            print(f"    ⚠️ 详情 {content_id} 出错: {e}, 重试 {attempt+1}/{MAX_RETRIES}")
            time.sleep(3)
    return "", "", "", []


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def main():
    parser = argparse.ArgumentParser(description=f"爬取{SITE_NAME}")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (默认5)")
    args = parser.parse_args()

    max_pages = args.pages
    all_items = []

    print(f"开始爬取 {SITE_NAME}")
    print(f"目标: 前{max_pages}页 (每页20条, 最多{max_pages*20}条)")

    for page in range(1, max_pages + 1):
        items = fetch_list(page)
        if not items:
            print(f"  第{page}页无数据，停止翻页")
            break
        for item in items:
            all_items.append(item)
        time.sleep(FETCH_DELAY)

    print(f"\n列表合计: {len(all_items)} 条")
    print("开始提取详情...")

    results = []
    for i, item in enumerate(all_items, 1):
        content_id = str(item.get("contentId", ""))
        title = item.get("title", "").strip()
        pub_date = item.get("publishDate", "")[:10]
        link = item.get("link", "")

        if not content_id or not link:
            print(f"  [{i}/{len(all_items)}] 跳过: 缺少ID或链接")
            continue

        print(f"  [{i}/{len(all_items)}] {title[:40]}...", end="")

        detail_title, detail_date, content, attachments = fetch_detail(content_id)

        final_title = detail_title or title
        final_title = re.sub(r"\s+", " ", final_title).strip()
        final_date = detail_date or pub_date

        # 附件嵌入正文
        if attachments:
            attach_lines = []
            for att in attachments:
                attach_lines.append(f"[附件：{att['name']}]({att['url']})")
            if attach_lines:
                if content:
                    content += "\n\n" + "\n".join(attach_lines)
                else:
                    content = "\n".join(attach_lines)

        results.append({
            "url": link,
            "source_url": link,
            "title": final_title,
            "content": content,
            "summary": content[:500],
            "pub_date": final_date,
            "site_name": SITE_NAME,
            "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        })

        print(f" ✓ ({len(content)}字)")
        time.sleep(FETCH_DELAY)

    # 批量入库
    print(f"\n入库 {len(results)} 条...")
    try:
        import sqlite3
        db = sqlite3.connect("/root/search.db", timeout=60)
        db.execute("PRAGMA journal_mode=WAL")
        db.execute("PRAGMA busy_timeout=30000")

        inserted = 0
        skipped = 0
        for rec in results:
            try:
                db.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (title, page_url, content, summary, publish_date, site_name, date_rank, attachments, source_url)
                       VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                    (
                        rec["title"],
                        rec["url"],
                        rec["content"],
                        rec["summary"],
                        rec["pub_date"],
                        rec["site_name"],
                        int(datetime.now().timestamp()),
                        rec["attachments"],
                        rec["source_url"],
                    ),
                )
                if db.total_changes > 0:
                    inserted += 1
                else:
                    skipped += 1
            except Exception as e:
                print(f"  ⚠️ 入库失败: {rec['title'][:30]}... {e}")
                skipped += 1

        db.commit()
        count_after = db.execute(
            "SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)
        ).fetchone()[0]
        db.close()

        print(f"  ✅ 入库: {inserted} 条新增, {skipped} 条跳过/重复")
        print(f"  ✅ DB中 {SITE_NAME} 总量: {count_after} 条")

    except Exception as e:
        print(f"  ❌ 数据库错误: {e}")

    print("完成!")


if __name__ == "__main__":
    main()
