#!/usr/bin/env python3
"""爬虫：河津经济技术开发区 — 公示公告
www.sxhj.gov.cn/hjkfq/gsgg/
"""

import os
import re
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
BASE_URL = "http://www.sxhj.gov.cn/hjkfq/gsgg/"
SITE_NAME = "河津开发区公示公告"
MAX_PAGES = 20  # pageCount=20, pageSize=10 → ~200条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def get_page_url(page):
    if page == 1:
        return BASE_URL
    return f"{BASE_URL}index_{page}.shtml"

def extract_year_from_url(url):
    """从URL /doc/2026/07/... 提取年份"""
    m = re.search(r'/doc/(\d{4})/', url)
    return m.group(1) if m else ""

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")
    h1 = soup.select_one(".mainCont h1")
    title = h1.get_text(strip=True) if h1 else ""

    pub_date = ""
    source_text = ""
    info_p = soup.select_one(".mainCont .explain")
    if info_p:
        text = info_p.get_text(strip=True)
        m = re.search(r"发布[日期：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        m2 = re.search(r"来源[：:]\s*(.+?)(?:\s*浏览次数|$)", text)
        if m2:
            source_text = m2.group(1).strip()

    zoom = soup.select_one("#Zoom")
    if not zoom:
        return title, pub_date, source_text, "", ""

    attachments = []
    for a_tag in zoom.find_all("a"):
        href = a_tag.get("href", "")
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
            full_url = urljoin(detail_url, href)
            text = a_tag.get_text(strip=True)
            attachments.append(f'<p><a href="{full_url}">{text}</a></p>')

    parts = []
    has_table = bool(zoom.find("table"))
    for tag in zoom.find_all(["p", "table"]):
        if tag.name == "p" and tag.find_parent("table"):
            continue
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        text = tag.get_text(strip=True)
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)

    if not parts:
        text = re.sub(r"\n+", "", zoom.get_text(strip=True))
        if text:
            parts.append(text)

    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""

    # 图片内容回退
    if not content.strip() and not attachments:
        imgs = zoom.find_all("img")
        if imgs:
            img_parts = []
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src:
                    full_src = urljoin(detail_url, src)
                    img_parts.append(f'<p><a href="{full_src}">查看图片</a></p>')
            if img_parts:
                content = f'<p><a href="{detail_url}">{title}</a></p>\n\n' + "\n\n".join(img_parts)

    if not content.strip() and attachments:
        content = f'<p><a href="{detail_url}">{title}</a></p>\n\n{attachments_str}'

    return title, pub_date, source_text, content, attachments_str

def crawl():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    total = 0

    for page in range(1, MAX_PAGES + 1):
        url = get_page_url(page)
        print(f"  第{page}页: {url}")

        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}")
            time.sleep(2)
            continue

        soup = BeautifulSoup(r.text, "html.parser")
        ul = soup.select_one("ul.List_list")
        if not ul:
            print(f"  [WARN] 未找到列表ul")
            break

        items = []
        for li in ul.find_all("li"):
            a_tag = li.find("a")
            span = li.find("span")
            if not a_tag or not span:
                continue
            href = a_tag.get("href", "")
            title = a_tag.get_text(strip=True)
            # 列表日期格式: [07-06] → 无年份, 从URL提取
            list_date_tag = span.get_text(strip=True).strip("[]").strip()
            if not href:
                continue
            detail_url = urljoin(url, href)
            # 从URL提取年份补全日期
            year = extract_year_from_url(detail_url)
            if year and list_date_tag and re.match(r'\d{2}-\d{2}', list_date_tag):
                list_date_full = f"{year}-{list_date_tag}"
            else:
                list_date_full = list_date_tag
            items.append((title, list_date_full, detail_url))

        if not items:
            print(f"  第{page}页无数据，终止")
            break

        for title, list_date, detail_url in items:
            print(f"    提取: {title[:40]}...")
            full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)
            if not full_title:
                full_title = title
            if not pub_date:
                pub_date = list_date

            date_rank = 0
            if pub_date:
                try:
                    date_rank = 0 - int(pub_date.replace("-", "") + "0000")
                except ValueError:
                    date_rank = 0

            try:
                cur.execute("""
                    INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, publish_date, content, summary, attachments, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                """, (detail_url, SITE_NAME, full_title, pub_date, content, source_text, attachments_str, date_rank))
                total += 1
            except Exception as e:
                print(f"    [DB ERROR] {e}")

            time.sleep(0.3)

        conn.commit()
        print(f"  第{page}页完成，累计{total}条")
        time.sleep(0.5)

    conn.close()
    print(f"\n总计: {total}条")

if __name__ == "__main__":
    crawl()
