#!/usr/bin/env python3
"""
宕昌县人民政府 - 通知公告 爬虫
CMS: 龙讯Lonsun
列表: /content/column/52673537?pageIndex={n} (24条/页, 共23页)
详情: /zwxx/tzgg/{id}.html
"""

import sys, os, re, time, sqlite3, subprocess, hashlib, argparse
from datetime import datetime
from urllib.parse import urljoin
import requests

BASE_URL = "https://www.tanchang.gov.cn"
LIST_API = "https://www.tanchang.gov.cn/content/column/52673537"
SITE_NAME = "宕昌县人民政府_通知公告"
GROUP = "县区"
DB_PATH = "/mnt/data/search.db"
ITEMS_PER_PAGE = 24
MAX_PAGES = 23

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

SESSION = requests.Session()
SESSION.headers.update(HEADERS)


def fetch(url):
    for attempt in range(3):
        try:
            resp = SESSION.get(url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return resp.text
        except Exception as e:
            pass
        time.sleep(2)
    return None


def parse_list(html):
    """从列表页提取 (title, url)"""
    items = []
    m = re.search(r'<ul[^>]*class="[^"]*doc_list[^"]*"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items
    ul_html = m.group(1)
    for m2 in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*>\s*(.*?)\s*</a>', ul_html, re.DOTALL):
        href = m2.group(1).strip()
        raw_title = re.sub(r'<[^>]+>', '', m2.group(2)).strip()
        raw_title = raw_title.replace('&nbsp;', ' ').replace('\u00a0', ' ')
        raw_title = re.sub(r'\s+', ' ', raw_title).strip()
        if not raw_title or not href:
            continue
        if '/zwxx/tzgg/' not in href:
            continue
        full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
        items.append((raw_title, full_url))
    return items


def parse_detail(html, page_url, list_title=""):
    """提取详情页信息"""
    # --- 标题 ---
    title = ""
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if m:
            t = m.group(1).strip()
            t = re.sub(r'_宕昌县人民政府$', '', t).strip()
            if t:
                title = t
    if not title:
        m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if m:
            t = re.sub(r'<[^>]+>', '', m.group(1)).strip()
            if t:
                title = t
    if not title:
        title = list_title
    title = title.replace('&nbsp;', ' ').replace('\u00a0', ' ').strip()

    # --- 发布日期 ---
    publish_date = ""
    m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m:
        publish_date = m.group(1).strip()[:10]
    if not publish_date:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m:
            publish_date = m.group(1)

    # --- 来源 ---
    source = ""
    m = re.search(r'<meta\s+name="ContentSource"\s+content="([^"]*)"', html)
    if m:
        source = m.group(1).strip()
    if not source:
        m = re.search(r'信息来源[：:]?\s*([^<]{2,20}?)<', html)
        if m:
            source = m.group(1).strip()

    # --- 正文 ---
    body_text = ""
    m = re.search(r'<div[^>]*class="[^"]*wzcon[^"]*j-fontContent[^"]*"[^>]*>(.*?)</div>\s*<div[^>]*class="[^"]*clear[^"]*"', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*class="[^"]*wzcon[^"]*j-fontContent[^"]*"[^>]*>(.*?)</div>\s*<div\b', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*class="[^"]*wzcon[^"]*j-fontContent[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)

    if m:
        inner = m.group(1).strip()
        text_only = re.sub(r'<[^>]+>', '', inner).strip()
        text_only = re.sub(r'如果内容不能正常显示.*?PDF文档|鼠标右键点击此处.*?|请安装pdf.*', '', text_only).strip()
        has_pdf_obj = '<object' in inner or '<embed' in inner or 'pdf.gif' in inner

        if (not text_only or len(text_only) < 20) and has_pdf_obj:
            # PDF-only页
            links = []
            for m_pdf in re.finditer(r'<a[^>]*href="([^"]+\.pdf[^"]*)"[^>]*>(.*?)</a>', inner, re.DOTALL):
                pdf_url = m_pdf.group(1).strip()
                pdf_title = re.sub(r'<[^>]+>', '', m_pdf.group(2)).strip()
                if not pdf_title:
                    pdf_title = os.path.basename(pdf_url.split('?')[0])
                full_url = pdf_url if pdf_url.startswith('http') else urljoin(BASE_URL, pdf_url)
                links.append(f'<a href="{full_url}">{pdf_title}</a>')
            body_text = '\n'.join(links)
        else:
            parts = []
            for m_p in re.finditer(r'<p[^>]*>(.*?)</p>', inner, re.DOTALL):
                p_html = m_p.group(1).strip()
                p_text = re.sub(r'<[^>]+>', '', p_html).strip()
                if not p_text:
                    continue
                if 'pdf.gif' in p_html or 'pdf.png' in p_html or '<object' in p_html or '<embed' in p_html:
                    for m_a in re.finditer(r'<a[^>]*href="([^"]+\.pdf[^"]*)"[^>]*>(.*?)</a>', p_html, re.DOTALL):
                        pdf_url = m_a.group(1).strip()
                        pdf_title = re.sub(r'<[^>]+>', '', m_a.group(2)).strip()
                        full_url = pdf_url if pdf_url.startswith('http') else urljoin(BASE_URL, pdf_url)
                        parts.append(f'<a href="{full_url}">{pdf_title}</a>')
                    continue
                cleaned = re.sub(r'\sstyle="[^"]*"', '', p_html)
                parts.append(f'<p>{cleaned}</p>')

            for m_t in re.finditer(r'(<table[^>]*>.*?</table>)', inner, re.DOTALL):
                tbl = m_t.group(1)
                parts.append(tbl)

            for m_i in re.finditer(r'<img[^>]*src="([^"]+)"[^>]*>', inner):
                src = m_i.group(1)
                if 'pdf.gif' not in src and 'pdf.png' not in src:
                    parts.append(f'<img src="{src}">')

            seen_pdf = set()
            for m_a in re.finditer(r'<a[^>]*href="([^"]+\.pdf[^"]*)"[^>]*>(.*?)</a>', inner, re.DOTALL):
                pdf_url = m_a.group(1).strip()
                if pdf_url not in seen_pdf:
                    seen_pdf.add(pdf_url)
                    pdf_title = re.sub(r'<[^>]+>', '', m_a.group(2)).strip()
                    full_url = pdf_url if pdf_url.startswith('http') else urljoin(BASE_URL, pdf_url)
                    link = f'<a href="{full_url}">{pdf_title}</a>'
                    if link not in parts:
                        parts.append(link)

            body_text = '\n\n'.join(p for p in parts if p.strip())

    return {
        'title': title,
        'publish_date': publish_date,
        'source': source or SITE_NAME,
        'body_text': body_text,
    }


def safe_date(date_str):
    if not date_str:
        return ""
    m = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', str(date_str))
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        return f"{y:04d}-{mo:02d}-{d:02d}"
    return ""


def truncate_summary(text, max_len=300):
    if not text:
        return ""
    s = re.sub(r'<[^>]+>', ' ', text)
    s = re.sub(r'\s+', ' ', s).strip()
    return s[:max_len]


def crawl(max_pages=MAX_PAGES):
    """主爬取逻辑"""
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    # 检查已有page_url
    existing_cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = set(row[0] for row in existing_cur.fetchall())

    total_new = total_skip = total_error = 0
    pages_to_crawl = min(max_pages, MAX_PAGES)
    print(f"[宕昌·通知公告] 开始爬取, 已有 {len(existing)} 条, 目标 {pages_to_crawl} 页")

    for page in range(1, pages_to_crawl + 1):
        page_url = f"{LIST_API}?pageIndex={page}"
        print(f"  第 {page}/{pages_to_crawl} 页: {page_url}", flush=True)

        html = fetch(page_url)
        if not html:
            print(f"    请求失败")
            total_error += 1
            continue

        items = parse_list(html)
        print(f"    发现 {len(items)} 条", flush=True)
        if not items:
            print("    无条目, 停止翻页")
            break

        for list_title, url in items:
            if url in existing:
                total_skip += 1
                continue

            detail_html = fetch(url)
            if not detail_html:
                print(f"    ✗ {list_title[:30]}... 请求失败")
                total_error += 1
                continue

            if '404' in detail_html[:500] and ('页面不出来' in detail_html or '未找到' in detail_html):
                print(f"    ⚠ {list_title[:30]}... 404")
                total_error += 1
                continue

            info = parse_detail(detail_html, url, list_title)
            body = info['body_text']
            pub_date = safe_date(info['publish_date'])
            source = info['source']
            summary = truncate_summary(body)

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, url, url, info['title'][:500], pub_date,
                    summary[:500], body, 'active', '', GROUP, ''
                ))
                if cur.rowcount > 0:
                    # 获取新插入行的rowid
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
                    if row:
                        rid = row[0]
                        esc_title = info['title'].replace("'", "''")
                        esc_site = SITE_NAME.replace("'", "''")
                        esc_summary = summary.replace("'", "''")
                        sql = f"INSERT INTO gov_search(rowid, title, site_name, summary) VALUES({rid},'{esc_title}','{esc_site}','{esc_summary}')"
                        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=30)
                    existing.add(url)
                    db.commit()
                    total_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"    ✗ DB error: {e}")
                total_error += 1

            time.sleep(0.5)

        time.sleep(1)

    db.close()
    print(f"\n[宕昌·通知公告] 完成！新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}")
    return total_new


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='宕昌县·通知公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES, help=f'最大页数 (默认 {MAX_PAGES})')
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
