#!/usr/bin/env python3
"""
彭泽县人民政府-通知公告 /zx/03/
列表: div.thc > ul.thlist > li > a + span
分页: createPage(11, 0, "index", "html") -> index.html ~ index_10.html
详情: div#mart > div.trs_editor_view > p(/table/img)
"""

import requests, re, sqlite3, time, os, sys
from bs4 import BeautifulSoup
from requests.adapters import HTTPAdapter
from urllib3.util.retry import Retry

BASE_URL = "https://www.pengze.gov.cn/zx/03/"
SITE_NAME = "彭泽县-通知公告"
DB_PATH = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

session = requests.Session()
session.headers.update(HEADERS)
retry_strategy = Retry(total=5, backoff_factor=2, allowed_methods=["GET"])
adapter = HTTPAdapter(max_retries=retry_strategy)
session.mount("https://", adapter)
session.mount("http://", adapter)


def get_total_pages(list_html):
    m = re.search(r'createPage\s*\(\s*(\d+)\s*,\s*(\d+)', list_html)
    if m:
        return int(m.group(1)), int(m.group(2))
    return 1, 0


def parse_list_page(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for ul in soup.select("ul.thlist"):
        a = ul.find("a")
        span = ul.find("span")
        if not a or not a.get("href"):
            continue
        href = a.get("href", "").strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_text = span.get_text(strip=True) if span else ""
        date_match = re.match(r"(\d{4})\.(\d{2})\.(\d{2})", date_text)
        date_str = f"{date_match.group(1)}-{date_match.group(2)}-{date_match.group(3)}" if date_match else ""
        items.append((href, title, date_str))
    return items


def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"ERR")
        return None


def extract_editor_content(editor, base_url):
    """从 editor 容器提取 p/table/img 内容，处理嵌套div包裹"""
    parts = []
    has_table = False

    # Try direct children first
    for child in editor.children:
        if child.name == "p":
            text = child.get_text(strip=True)
            if text and text not in ("\xa0", ""):
                if any(kw in text for kw in ["主办单位", "承办单位", "ICP备案", "网站标识码", "扫一扫", "关闭"]):
                    continue
                parts.append(text)
        elif child.name == "table":
            has_table = True
            md = table_to_markdown(child)
            if md:
                parts.append(md)
        elif child.name == "div":
            # Check for wrapped table
            if "ue_table" in child.get("class", []):
                tbl = child.find("table")
                if tbl:
                    has_table = True
                    md = table_to_markdown(tbl)
                    if md:
                        parts.append(md)
                continue
            # Recurse into wrapper div
            sub_parts, sub_has_table = extract_editor_content(child, base_url)
            parts.extend(sub_parts)
            if sub_has_table:
                has_table = True
        elif child.name == "img":
            src = child.get("src", "")
            alt = child.get("alt", "")
            if src:
                full_src = resolve_url(src, base_url)
                parts.append(f"![{alt}]({full_src})")

    return parts, has_table


def resolve_url(url_str, base_url):
    if url_str.startswith("http"):
        return url_str
    if url_str.startswith("//"):
        return "https:" + url_str
    if url_str.startswith("/"):
        return "https://www.pengze.gov.cn" + url_str
    if url_str.startswith("./"):
        return "https://www.pengze.gov.cn" + url_str[1:]
    if url_str.startswith(".."):
        return "https://www.pengze.gov.cn" + url_str[2:]
    return base_url.rstrip("/") + "/" + url_str.lstrip("/")


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    base = "https://www.pengze.gov.cn"

    # Title
    title = ""
    meta_title = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).replace(" - 彭泽县人民政府网站", "").strip()

    # Date
    date_str = ""
    for name in ["PublishDate", "PubDate", "publishdate", "pubdate"]:
        meta_date = soup.find("meta", attrs={"name": re.compile(name, re.I)})
        if meta_date and meta_date.get("content"):
            raw = meta_date["content"].strip()
            m = re.match(r"(\d{4}-\d{2}-\d{2})", raw)
            if m:
                date_str = m.group(1)
                break

    # Content
    content_parts = []
    has_table = False
    mart = soup.find("div", id="mart")
    if mart:
        editor = mart.find("div", class_=re.compile(r"trs_editor_view", re.I))
        if not editor:
            editor = mart.find("div", class_=re.compile(r"trs", re.I))
        if editor:
            parts, ht = extract_editor_content(editor, base)
            content_parts = parts
            has_table = ht

    if not content_parts:
        # Fallback: any trs_editor_view in page
        editor = soup.find("div", class_=re.compile(r"trs_editor_view", re.I))
        if editor:
            parts, ht = extract_editor_content(editor, base)
            content_parts = parts
            has_table = ht

    if not content_parts:
        # Last resort: all p elements inside #mart
        if mart:
            for p in mart.find_all("p", recursive=True):
                text = p.get_text(strip=True)
                if text and text not in ("\xa0", ""):
                    if any(kw in text for kw in ["主办单位", "承办单位", "ICP备案", "网站标识码", "扫一扫", "关闭"]):
                        continue
                    content_parts.append(text)

    content = "\n\n".join(content_parts)

    # Attachments
    attachments = []
    for a_tag in soup.find_all("a", href=re.compile(r"\.(doc|pdf|xls|docx|xlsx|rar|zip)$", re.I)):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True)
        if href:
            full_href = resolve_url(href, "https://www.pengze.gov.cn")
            attachments.append({"name": text or os.path.basename(full_href), "url": full_href})
    attachments_json = str(attachments) if attachments else ""

    return title, date_str, content, attachments_json, has_table


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def save_to_db(items_data):
    conn = sqlite3.connect(DB_PATH, timeout=10)
    c = conn.cursor()
    inserted = 0
    for title, date_str, content, attachments_json, has_table, page_url in items_data:
        try:
            c.execute("""
                INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, summary, attachments, has_table, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_pengze_gsgg.py')
            """, (
                page_url, title, SITE_NAME, date_str, content,
                content[:200] if content else "", attachments_json, 1 if has_table else 0
            ))
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
    conn.commit()
    conn.close()
    return inserted


def crawl():
    total_saved = 0
    total_empty = 0
    total_segmented = 0
    total_attachments = 0
    empty_titles = []

    print(f"[{time.strftime('%H:%M:%S')}] 获取列表页...")
    try:
        resp = session.get(BASE_URL + "index.html", timeout=30)
        resp.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] 列表页请求失败: {e}")
        return

    total_pages, _ = get_total_pages(resp.text)
    print(f"  总页数: {total_pages}")

    all_items = []
    for page_idx in range(total_pages):
        if page_idx == 0:
            page_url = BASE_URL + "index.html"
        else:
            page_url = BASE_URL + f"index_{page_idx}.html"
        if page_idx > 0:
            try:
                resp = session.get(page_url, timeout=30)
                resp.encoding = "utf-8"
            except Exception as e:
                print(f"  第{page_idx+1}页失败: {e}")
                continue
        items = parse_list_page(resp.text)
        print(f"  第{page_idx+1}/{total_pages}页: {len(items)}条")
        all_items.extend(items)
        time.sleep(1.0)

    print(f"\n共{len(all_items)}条待抓取")

    batch_data = []
    for i, (url, title, date_str) in enumerate(all_items):
        detail_url = resolve_url(url, BASE_URL)
        print(f"  [{i+1}/{len(all_items)}] {title[:30]}...", end=" ", flush=True)
        html = fetch_detail(detail_url)
        if not html:
            print("F")
            continue
        detail_title, detail_date, content, attachments_json, has_table = extract_detail(html, detail_url)
        final_date = detail_date or date_str
        final_title = detail_title or title
        if not content:
            total_empty += 1
            empty_titles.append(final_title)
            print("E")
        elif "\n\n" in content:
            total_segmented += 1
            print("S")
        else:
            print("U")
        if attachments_json:
            import json as _json
            try:
                total_attachments += len(_json.loads(attachments_json))
            except:
                pass
        batch_data.append((final_title, final_date, content, attachments_json, has_table, detail_url))
        time.sleep(0.5)

    if batch_data:
        saved = save_to_db(batch_data)
        total_saved = saved
        print(f"\n入库: {saved}条")
    else:
        print("\n无数据入库")

    print(f"\n===== 报告 =====")
    print(f"总抓取: {total_saved}")
    print(f"正文为空: {total_empty}")
    print(f"有分段: {total_segmented}/{total_saved} ({total_segmented/max(total_saved,1)*100:.1f}%)")
    print(f"附件: {total_attachments}")
    if empty_titles:
        print(f"\n空正文标题:")
        for t in empty_titles[:10]:
            print(f"  - {t}")


if __name__ == "__main__":
    crawl()
