#!/usr/bin/env python3
import os
"""
安庆市大观区人民政府 - 通知公告
https://www.aqdgq.gov.cn/zwdt/tzgg/index.html
Lonsun CMS, 静态列表页面
列表: ul.doc_list > li > a[title] + span.right.date
详情: div.j-fontContent.newscontnet.minh300, h1.newstitle
分页: /content/column/2000004601?pageIndex=N (JS挑战，仅爬第1页)
"""
import os, sys, re, time, json, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "大观区-通知公告"
BASE_URL = "https://www.aqdgq.gov.cn"
LIST_URL = "https://www.aqdgq.gov.cn/zwdt/tzgg/index.html"
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.aqdgq.gov.cn/",
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch(url, retries=3):
    import requests
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except:
            if i < retries - 1:
                time.sleep(2)
    return None

def parse_list(html):
    """Parse list page, return list of (title, url, date)."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_=re.compile(r'doc_list'))
    if not ul:
        return items
    
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)
        href = a.get('href', '').strip()
        if not href or href.startswith('javascript'):
            continue
        url = urljoin(BASE_URL, href)
        
        span_date = li.find('span', class_='date')
        date = span_date.get_text(strip=True) if span_date else ''
        
        if title and url and date:
            items.append((title, url, date))
    
    return items

def parse_detail(html, url):
    """Parse detail page."""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from h1.newstitle
    title_el = soup.find('h1', class_='newstitle')
    title = body_text(title_el) if title_el else ''
    # Clean up newlines in title
    title = re.sub(r'\s+', ' ', title).strip()
    
    # Date and source from .newsinfo
    date = ''
    source = ''
    info = soup.find('div', class_='newsinfo')
    if info:
        for sp in info.find_all('span', class_='sp'):
            txt = sp.get_text(strip=True)
            dm = re.search(r'(\d{4}-\d{2}-\d{2}\s*\d{2}:\d{2})', txt)
            if dm:
                date = dm.group(1)
            if '来源' in txt:
                source = re.sub(r'^来源[：:]?\s*', '', txt).strip()
    
    # Content from .j-fontContent.newscontnet
    content_parts = []
    attachments = []
    
    cont = soup.find('div', class_='j-fontContent')
    if not cont:
        cont = soup.find('div', class_='newscontnet')
    
    if cont:
        for el in cont.children:
            if isinstance(el, str):
                txt = el.strip()
                if txt:
                    content_parts.append(txt)
            elif el.name in ('p', 'div', 'section'):
                txt = body_text(el)
                if txt:
                    content_parts.append(txt)
            elif el.name == 'br':
                content_parts.append('')
        
        # Attachments
        for a_tag in cont.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|xlsm)(\?|$)', re.I)):
            att_url = urljoin(BASE_URL, a_tag.get('href', ''))
            att_name = a_tag.get_text(strip=True) or os.path.basename(att_url).split('?')[0]
            if att_url and att_name:
                attachments.append({'url': att_url, 'name': att_name})
    
    content_text = '\n'.join(content_parts).strip()
    return title, content_text, date, source, attachments

def push_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    new_count = 0
    skip_count = 0
    
    for title, url, date, content, source, atts_json in items:
        if not title:
            continue
        summary = content[:200] if content else title
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            skip_count += 1
            continue
        c.execute(
            "INSERT INTO gov_raw (title, page_url, publish_date, site_name, content, source_url, attachments, category, summary) VALUES (?,?,?,?,?,?,?,?,?)",
            (title, url, date, SITE_NAME, content, source, atts_json, '政府公告', summary)
        )
        new_count += 1
    conn.commit()
    conn.close()
    return new_count, skip_count

def main():
    print(f"[{SITE_NAME}] 开始爬取")
    
    # Step 1: Parse list page (only page 1 - pagination behind JS challenge)
    html = fetch(LIST_URL)
    if not html:
        print("  ⚠️ 获取列表失败")
        return
    
    items = parse_list(html)
    print(f"  列表: {len(items)} 条")
    
    all_items = []
    for title, url, date in items:
        if date >= THRESHOLD:
            all_items.append((title, url, date))
    
    print(f"  近3年: {len(all_items)} 条")
    
    # Step 2: Fetch details
    detail_results = []
    for i, (title, url, date) in enumerate(all_items):
        if (i+1) % 10 == 0:
            print(f"  详情: {i+1}/{len(all_items)}")
        
        html = fetch(url)
        if html:
            d_title, content, d_date, source, atts = parse_detail(html, url)
            if not d_title: d_title = title
            if not d_date: d_date = date
            atts_json = json.dumps(atts, ensure_ascii=False) if atts else '[]'
            detail_results.append((d_title, url, d_date, content, source, atts_json))
        
        time.sleep(0.5)
    
    # Step 3: Push to DB
    detail_results.sort(key=lambda x: x[2], reverse=True)
    push_items = [(dt, url, dd, cont, src, atts) for dt, url, dd, cont, src, atts in detail_results]
    new_count, skip_count = push_to_db(push_items)
    
    print(f"\n[{SITE_NAME}] 完成")
    print(f"  处理: {len(detail_results)}")
    print(f"  新增: {new_count}")
    print(f"  跳过: {skip_count}")

def incremental():
    """增量 - 只爬第1页"""
    print(f"[{SITE_NAME}] 增量爬取")
    html = fetch(LIST_URL)
    if not html:
        print("  ⚠️ 获取列表失败")
        return
    
    items = parse_list(html)
    print(f"  列表: {len(items)} 条")
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    new = 0
    skip = 0
    
    for title, url, date in items:
        if date < THRESHOLD:
            continue
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            skip += 1
            continue
        
        html = fetch(url)
        if not html:
            continue
        
        d_title, content, d_date, source, atts = parse_detail(html, url)
        if not d_title: d_title = title
        if not d_date: d_date = date
        atts_json = json.dumps(atts, ensure_ascii=False) if atts else '[]'
        summary = content[:200] if content else d_title
        
        c.execute(
            "INSERT INTO gov_raw (title, page_url, publish_date, site_name, content, source_url, attachments, category, summary) VALUES (?,?,?,?,?,?,?,?,?)",
            (d_title, url, d_date, SITE_NAME, content, source, atts_json, '政府公告', summary)
        )
        new += 1
        time.sleep(0.5)
    
    conn.commit()
    conn.close()
    print(f"  新增: {new}, 跳过: {skip}")

if __name__ == '__main__':
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        incremental()
    else:
        main()
