#!/usr/bin/env python3
"""
crawl_sixian.py - 泗县人民政府 通知公告 爬虫
CMS: 龙讯科技(Lonsun), API分页 content/column/{id}?pageIndex={n}
"""
import requests

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
import re
import sqlite3
import time
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "泗县人民政府-通知公告"
BASE_URL = "https://www.sixian.gov.cn"
LIST_URL = BASE_URL + "/xwzx/tzgg/index.html"
API_URL = BASE_URL + "/content/column/11465623?pageIndex={}"
TOTAL_PAGES = 139
PAGE_SIZE = 20
DELAY = 0.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

def get_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  [ERROR] {e}")
        return ""

def parse_list(html):
    """从列表页提取 (url, title, date)"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    
    # Find the doc_list ul
    ul = soup.find("ul", class_=re.compile(r"doc_list"))
    if not ul:
        # Try the whole page for links
        for li in soup.find_all("li"):
            a = li.find("a", href=re.compile(r"/xwzx/tzgg/\d+\.html"))
            if a:
                href = a.get("href", "")
                if href and not href.startswith("http"):
                    href = BASE_URL + href
                title = a.get("title", "").strip() or a.get_text(strip=True)
                date_span = li.find("span", class_="date")
                date = date_span.get_text(strip=True) if date_span else ""
                items.append((href, title, date))
        return items
    
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or "tzgg" not in href or not href.endswith(".html"):
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_span = li.find("span", class_="date")
        date = date_span.get_text(strip=True) if date_span else ""
        items.append((href, title, date))
    
    return items

def extract_detail(html, url, list_date):
    """从详情页提取完整信息"""
    soup = BeautifulSoup(html, "html.parser")
    
    # 标题
    title = ""
    h1 = soup.find("h1", class_=re.compile(r"newstitl"))
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta = soup.find("meta", {"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            # Remove trailing site name
            if "_" in t:
                title = t.split("_")[0].strip()
            elif "-" in t:
                title = t.split("-")[0].strip()
            else:
                title = t
    
    # 发布日期
    publish_date = list_date
    # From span.fbsj
    fbsj = soup.find("span", class_="fbsj")
    if fbsj:
        txt = fbsj.get_text(strip=True)
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
        if dm:
            publish_date = dm.group(1)
    # From meta PubDate
    if not publish_date:
        meta = soup.find("meta", {"name": "PubDate"})
        if meta and meta.get("content"):
            dm = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
            if dm:
                publish_date = dm.group(1)
    
    # 正文 contentbox
    content_html = ""
    content_box = soup.find("div", class_="contentbox")
    if content_box:
        # Find the article div inside
        article_div = content_box.find("div", role="article", attrs={"aria-label": "内容"})
        if not article_div:
            article_div = content_box
        
        # 找正文容器 newscontnet
        content_area = article_div.find("div", class_="newscontnet")
        if not content_area:
            content_area = article_div
        
        # 递归提取所有 p/table/img 标签，按文档顺序保留
        parts = []
        seen_positions = set()
        
        def extract_ordered(tag):
            for child in tag.children:
                if not hasattr(child, "name") or child.name is None:
                    continue
                cname = child.name.lower()
                # 跳过标题/信息/分享栏
                cls = " ".join(child.get("class", [])) if child.get("class") else ""
                if cname == "h1" and "newstitle" in cls:
                    continue
                if cname == "div":
                    if "newsinfo" in cls or "sharebox" in cls or "wzewmbox" in cls:
                        continue
                    # 递归进div
                    extract_ordered(child)
                elif cname in ("p", "table", "img"):
                    # 跳过已被父级包含的项（通过id或position去重）
                    child_id = id(child)
                    if child_id in seen_positions:
                        continue
                    seen_positions.add(child_id)
                    parts.append(str(child))
        
        extract_ordered(content_area)
        content_html = "\n".join(parts)
    
    # 附件（HTML内嵌格式，排除图标gif链接）
    attachments = []
    for link in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|wps|et|dps|ofd)$", re.I)):
        att_url = link.get("href", "")
        if att_url and not att_url.startswith("http"):
            att_url = BASE_URL + att_url
        if "files2/" in att_url or att_url.endswith(".gif"):
            continue
        att_title = link.get_text(strip=True)
        if not att_title:
            att_title = att_url.split("/")[-1].split("?")[0]
        attachments.append((att_title, att_url))
    
    if not content_html.strip():
        content_html = str(article_div) if 'article_div' in dir() else ""
    
    return title, publish_date, content_html, attachments

def render_html(content_html, url):
    """渲染正文为HTML（保留段落/表格，图片转查看图片链接，禁markdown）"""
    soup = BeautifulSoup(content_html, "html.parser")
    parts = []
    
    # 全局图片处理：图片转 <p><a href="完整URL">查看图片</a></p>
    for img in soup.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            src = BASE_URL + src
        # 跳过图标文件
        if "files2/" in src or "icon_" in src or src.endswith(".gif"):
            img.decompose()
            continue
        # 找到图片所在p标签，替换为链接
        parent_p = img.find_parent("p")
        repl = soup.new_tag("p")
        a = soup.new_tag("a", href=src)
        a.string = "查看图片"
        repl.append(a)
        if parent_p:
            parent_p.replace_with(repl)
        else:
            img.replace_with(repl)
    
    for child in soup.children:
        if child.name in ("p", "table", "div", "ul", "ol", "blockquote"):
            txt = child.get_text("", strip=True)
            if txt:
                parts.append(str(child))
        elif child.name == "img":
            src = child.get("src", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src
            parts.append(f'<p><a href="{src}">查看图片</a></p>')
        elif child.name is None:
            text = str(child).strip()
            if text:
                parts.append(f"<p>{text}</p>")
    
    return "\n\n".join(parts)

def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    
    for page in range(1, min(TOTAL_PAGES, _MAX_PG or TOTAL_PAGES) + 1):
        url = API_URL.format(page)
        
        print(f"--- Page {page}/{TOTAL_PAGES} ---")
        html = get_page(url)
        if not html:
            continue
        
        items = parse_list(html)
        if not items:
            print(f"  [SKIP] no items")
            continue
        
        print(f"  Found {len(items)} items")
        
        for href, title, date in items:
            try:
                detail_html = get_page(href)
                if not detail_html:
                    continue
                
                full_title, pub_date, content_html, attachments = extract_detail(detail_html, href, date)
                if not full_title:
                    full_title = title
                if not pub_date:
                    pub_date = date
                
                content_html_out = render_html(content_html, href)
                # 附件内嵌到正文尾部（每个附件独立成段）
                if attachments:
                    att_blocks = []
                    for att_title, att_url in attachments:
                        att_blocks.append(f'<p><a href="{att_url}">{att_title}</a></p>')
                    content_html_out = content_html_out.rstrip() + "\n\n" + "\n\n".join(att_blocks)
                plain = re.sub(r"<[^>]+>", "", content_html).strip()
                summary = plain[:200] if len(plain) > 200 else plain
                
                att_field = "\n".join(f'<p><a href="{u}">{t}</a></p>' for t, u in attachments)
                c.execute("""
                    INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, summary, content, source_url, date_rank, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'crawl_sixian.py')
                """, (
                    href, SITE_NAME, full_title, pub_date, summary, content_html_out,
                    href, int(time.time()), att_field
                ))
                conn.commit()
                total_new += 1
                time.sleep(DELAY)
            except Exception as e:
                print(f"  [ERR] {href.split('/')[-1][:24]}: {e}")
                continue
        time.sleep(DELAY)
    
    conn.close()
    print(f"\nDone! Upserted: {total_new}")

if __name__ == "__main__":
    run()
