#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
"""
全椒县人民政府 - 通知公告 爬虫
http://www.quanjiao.gov.cn/zwdt/tzgg/index.html
CMS: LsCMS (滁州政务平台), doc_list列表, 分页/content/column/160693021?pageIndex=N
"""
import requests, re, time, json, os
from bs4 import BeautifulSoup
from datetime import datetime

SITE_NAME = "全椒县人民政府-通知公告"
LIST_URL = "http://www.quanjiao.gov.cn/zwdt/tzgg/index.html"
PAGE_API = "http://www.quanjiao.gov.cn/content/column/160693021?pageIndex={}"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=20)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if i < retries-1:
                time.sleep(2)
    return None

def get_page_count():
    """获取总页数"""
    html = fetch(LIST_URL)
    if not html:
        return 1
    m = re.search(r'pageCount:(\d+)', html)
    if m:
        return int(m.group(1))
    # 备用：从total/pageSize计算
    m2 = re.search(r'total:\s*(\d+)', html)
    if m2:
        total = int(m2.group(1))
        return (total + 19) // 20  # pageSize=20
    return 1

def parse_list(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_=re.compile(r'doc_list'))
    if not ul:
        return items
    
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        span = li.find('span', class_='date')
        date = span.get_text(strip=True) if span else ''
        if href and title:
            items.append((href, title, date))
    
    return items

def parse_detail(detail_url, list_title, list_date):
    """解析详情页"""
    html = fetch(detail_url)
    if not html:
        print(f"  ERROR: 无法获取 {detail_url}")
        return {
            'title': list_title,
            'date': list_date,
            'content': '',
            'attachments': []
        }
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # 标题
    title = list_title
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    # 日期
    date = list_date
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date = meta_date['content'].strip()
    
    # 正文
    content = ''
    attachments = []
    
    content_div = soup.find('div', id='J_content')
    if not content_div:
        content_div = soup.find('div', class_='newscontnet')
    if not content_div:
        content_div = soup.find('div', class_='j-fontContent')
    
    if content_div:
        paras = []
        for el in content_div.find_all(['p', 'table', 'img']):
            if el.name == 'p':
                # 跳过嵌套p
                if el.find('p'):
                    continue
                # 提取文本和链接
                text_parts = []
                for c in el.children:
                    if isinstance(c, str):
                        t = c.strip()
                        if t:
                            text_parts.append(t)
                    elif c.name == 'a':
                        href = c.get('href', '')
                        txt = c.get_text(strip=True)
                        if href and txt:
                            full = href if href.startswith('http') else ('https://www.quanjiao.gov.cn' + href if href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + href)
                            text_parts.append(f'[{txt}]({full})')
                            if re.search(r'\.(doc|docx|pdf|xls|xlsx|rar|zip)$', href, re.I):
                                attachments.append({'name': txt, 'url': full})
                        elif txt:
                            text_parts.append(txt)
                    elif c.name in ['span', 'strong', 'em', 'b', 'br']:
                        if c.name == 'br':
                            continue
                        t = c.get_text(strip=True)
                        if t:
                            text_parts.append(t)
                    else:
                        t = c.get_text(strip=True)
                        if t:
                            text_parts.append(t)
                text = ''.join(text_parts).strip()
                if text:
                    paras.append(text)
            elif el.name == 'table':
                # HTML table
                rows = []
                for tr in el.find_all('tr'):
                    cells = []
                    for td in tr.find_all(['td', 'th']):
                        tag = 'th' if td.name == 'th' else 'td'
                        ct = td.get_text(strip=True)
                        cells.append(f'<{tag}>{ct}</{tag}>')
                    if any(c for c in cells):
                        rows.append('<tr>' + ''.join(cells) + '</tr>')
                if rows:
                    paras.append('<table>' + ''.join(rows) + '</table>')
            elif el.name == 'img':
                src = el.get('src', '')
                if src:
                    alt = el.get('alt', '')
                    full_src = src if src.startswith('http') else ('https://www.quanjiao.gov.cn' + src if src.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + src)
                    paras.append(f'![{alt}]({full_src})')
        
        content = '\n\n'.join(paras)
    else:
        for p in soup.find_all('p'):
            t = p.get_text(strip=True)
            if t and len(t) > 5:
                content += t + '\n\n'
        content = content.strip()
    
    # 全文搜附件
    if not attachments:
        for a in soup.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|rar|zip)$', re.I)):
            href = a.get('href', '')
            txt = a.get_text(strip=True) or '附件'
            full = href if href.startswith('http') else ('https://www.quanjiao.gov.cn' + href if href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + href)
            if not any(a2['url'] == full for a2 in attachments):
                attachments.append({'name': txt, 'url': full})
    
    # summary
    clean = re.sub(r'\s+', ' ', content).strip()
    summary = clean[:200]
    
    # date_rank
    date_parsed = date[:10] if date else ''
    date_rank = 0
    if date_parsed:
        try:
            dt = datetime.strptime(date_parsed, '%Y-%m-%d')
            date_rank = int(dt.timestamp())
        except:
            pass
    
    print(f"  OK: {title[:30]}... | {date_parsed} | content={len(content)}c")
    return {
        'title': title,
        'date': date_parsed,
        'content': content,
        'summary': summary,
        'attachments': json.dumps(attachments, ensure_ascii=False),
        'date_rank': date_rank,
        'source_url': detail_url
    }

def main():
    print(f"=== {SITE_NAME} 爬虫 (仅首页) ===")
    
    # 只爬首页 inline 数据，API翻页被封
    html = fetch(LIST_URL)
    items = parse_list(html) if html else []
    print(f"首页: {len(items)} 条 (API翻页被封，仅首页)")
    
    # DB
    db_path = '/root/search.db' if os.path.exists('/root/search.db') else '/mnt/data/search.db'
    import sqlite3
    conn = sqlite3.connect(db_path, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    # 2026-09-25 QC 修复：去掉整站 DELETE（同 crawl_luanxian_854.py 的理由）。
    # 本脚本用 INSERT OR REPLACE，page_url UNIQUE 索引保证幂等更新。
    print(f"已跳过整站删除（改为幂等 REPLACE）")
    
    count = 0
    for href, list_title, list_date in items:
        detail = parse_detail(href, list_title, list_date)
        
        c.execute("""
            INSERT OR REPLACE INTO gov_raw (page_url, title, publish_date, content, summary, site_name, attachments, date_rank, source_url, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'crawl_quanjiao_tzgg.py')
        """, (
            href,
            detail['title'],
            detail.get('date', ''),
            detail.get('content', ''),
            detail.get('summary', ''),
            SITE_NAME,
            detail.get('attachments', '[]'),
            detail.get('date_rank', 0),
            href
        ))
        count += 1
        # 每20条提交一次，避免超时丢失
        if count % 20 == 0:
            conn.commit()
            print(f"  [进度: {count}/{len(items)}]")
        time.sleep(1.5)
    
    conn.commit()
    print(f"入库 {count} 条")
    
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    raw_count = c.fetchone()[0]
    c.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,))
    search_count = c.fetchone()[0]
    conn.close()
    print(f"验证: gov_raw={raw_count}, gov_search={search_count}")
    
    print(f"\n配置:")
    print(json.dumps({"name": SITE_NAME, "script": "crawl_quanjiao_tzgg.py", "url": LIST_URL, "group": "安徽"}, ensure_ascii=False, indent=2))

if __name__ == '__main__':
    main()
