#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
宜城市人民政府门户网站 - 公示公告 爬虫
CMS: 政府网站（襄阳市政府子站）
列表: /xwzx/gsgg/index_N.shtml（N从1开始，N=1即/index.shtml）
页码实际从index_2.shtml开始，每页15条，共55页
详情: /xwzx/gsgg/YYYYMM/tYYYYMMDD_XXXXXXX.shtml
"""

import requests
import re
import json
import time
import sys
import os
import sqlite3
from bs4 import BeautifulSoup

BASE_URL = "http://yc.xiangyang.gov.cn"
SITE_NAME = "宜城市人民政府-公示公告"
GROUP = "政府"
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')
MAX_PAGES = 5  # 前5页 = ~75条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url, timeout=30):
    resp = session.get(url, timeout=timeout)
    resp.encoding = 'utf-8'
    return resp


def crawl_list(page_num):
    """爬取列表页"""
    if page_num == 1:
        url = f"{BASE_URL}/xwzx/gsgg/"
    else:
        url = f"{BASE_URL}/xwzx/gsgg/index_{page_num}.shtml"
    
    print(f"  Fetching page {page_num}: {url}")
    
    try:
        resp = fetch(url)
        resp.raise_for_status()
    except Exception as e:
        print(f"  Error: {e}")
        return []
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    # 列表在 <ul> 中，<li> 条目
    # 找包含 article link 的 ul
    for ul in soup.find_all('ul'):
        lis = ul.find_all('li')
        for li in lis:
            a_tag = li.find('a', href=re.compile(r'\./t\d+|\./\d{6}/'))
            if not a_tag:
                continue
            
            href = a_tag.get('href', '')
            # 标题优先从 title 属性取（完整无省略）
            title = a_tag.get('title', '') or a_tag.get_text(strip=True)
            if not title:
                continue
            
            if not href.startswith('http'):
                if href.startswith('./'):
                    href = f"{BASE_URL}/xwzx/gsgg/{href[2:]}"
                elif href.startswith('/'):
                    href = f"{BASE_URL}{href}"
                else:
                    href = f"{BASE_URL}/xwzx/gsgg/{href}"
            
            # 日期
            date = ''
            date_match = re.search(r'(\d{4}-\d{2}-\d{2})', li.get_text())
            if date_match:
                date = date_match.group(1)
            
            items.append({
                'title': title,
                'url': href,
                'date': date,
            })
    
    return items


def crawl_detail(item):
    """爬取详情页"""
    url = item['url']
    print(f"    Detail: {item['title'][:40]}...")
    
    try:
        resp = fetch(url)
        resp.raise_for_status()
    except Exception as e:
        print(f"    Error: {e}")
        return None
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    # 标题 - 从 <title> 标签（格式: "文章标题 - 宜城市人民政府门户网站"）
    title = item['title']
    title_tag = soup.find('title')
    if title_tag:
        t_text = title_tag.get_text(strip=True)
        t_text = re.sub(r'\s*-\s*宜城市人民政府门户网站$', '', t_text).strip()
        if t_text:
            title = t_text
    
    # 日期
    date = item['date']
    date_match = re.search(r'(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})', resp.text) or \
                 re.search(r'发布日期[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', resp.text)
    if date_match:
        d = date_match.group(1)
        date = d[:10] if len(d) > 10 else d
    
    # 正文
    content_parts = []
    attachments = []
    
    # 正文区域
    content_div = soup.find('div', class_='by-content-box')
    
    if content_div:
        # 标题 - 从 xqym-title > h2 取
        h2 = content_div.find('h2')
        if h2:
            t = h2.get_text(strip=True)
            if t:
                title = t
        
        # 日期 - 从 xqym-title 的 span 取
        info_span = content_div.find('div', class_='xqym-title')
        if not info_span:
            info_span = content_div.find('div', class_=re.compile(r'title|info|meta'))
        if info_span:
            date_match = re.search(r'(\d{4}年\d{1,2}月\d{1,2}日)', info_span.get_text())
            if date_match:
                date = date_match.group(1).replace('年', '-').replace('月', '-').replace('日', '')
        
        # 正文内容 - 在 xqym-p > .view.TRS_UEDITOR 中
        article_div = content_div.find('div', class_=re.compile(r'TRS_UEDITOR|UEDITOR|view'))
        if not article_div:
            article_div = content_div.find('div', class_='xqym-p')
            if article_div:
                # 跳过第一个文本（字体控制）
                for child in article_div.find_all(recursive=False):
                    if child.name == 'div' and child.get('class'):
                        article_div = child
                        break
        if not article_div:
            article_div = content_div.find('div', class_=re.compile(r'view|content|article|text|TRS'))
        
        if article_div:
            # 附件链接
            for a_tag in article_div.find_all('a', href=True):
                href = a_tag['href']
                if re.search(r'\.(doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar)$', href, re.I):
                    attach_title = a_tag.get_text(strip=True)
                    if not href.startswith('http'):
                        if href.startswith('./'):
                            href = f"{BASE_URL}/xwzx/gsgg/{href[2:]}"
                        elif href.startswith('/'):
                            href = f"{BASE_URL}{href}"
                        else:
                            href = f"{url.rstrip('/')}/{href}"
                    attachments.append({
                        'title': attach_title or os.path.basename(href),
                        'url': href
                    })
            
            # 正文段落
            for p in article_div.find_all(['p', 'div']):
                if not p.get_text(strip=True):
                    continue
                if p.find_parent(['form', 'textarea', 'script']):
                    continue
                
                text = p.get_text(strip=True)
                if re.match(r'^(责任编辑|初审|复审|终审|编辑[：:]|来源[：:]|发布日期)', text):
                    continue
                
                tables = p.find_all('table')
                if tables:
                    for table in tables:
                        content_parts.append(str(table))
                    continue
                
                content_parts.append(text)
    
    content = '\n\n'.join(part for part in content_parts if part)
    
    # 空内容回退
    if len(content.strip()) < 20:
        attach_links = '\n'.join(f"[{a['title']}]({a['url']})" for a in attachments)
        if attach_links:
            content = f"[{title}]({url})\n\n附件：\n{attach_links}"
        else:
            content = f"[{title}]({url})"
    
    summary = content[:200] if content else title
    
    return {
        'site_name': SITE_NAME,
        'source_url': '',
        'page_url': url,
        'title': title,
        'publish_date': date,
        'content': content,
        'summary': summary,
        'category': GROUP,
        'status': 'published',
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def save_to_db(records, db_path=DB_PATH):
    if not records:
        return 0
    conn = sqlite3.connect(db_path)
    c = conn.cursor()
    new_count = 0
    for i, rec in enumerate(records):
        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date,
                 content, summary, category, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
                (rec['site_name'], rec['source_url'], rec['page_url'],
                 rec['title'], rec['publish_date'], rec['content'],
                 rec['summary'], rec['category'], rec['status'],
                 rec['attachments']))
            if c.rowcount > 0:
                new_count += 1
                if new_count <= 5 or (i+1) % 20 == 0:
                    print(f"  [{i+1}/{len(records)}] {rec['title'][:40]} +{new_count}")
        except Exception as e:
            print(f"  ERROR [{rec['page_url']}]: {e}")
    conn.commit()
    conn.close()
    return new_count


def main():
    print(f"=== {SITE_NAME} - 全量爬取 ===")
    all_items = []
    seen_urls = set()
    
    for page in range(1, MAX_PAGES + 1):
        items = crawl_list(page)
        if not items:
            print(f"  -> 0 items on page {page}, stopping")
            break
        print(f"  -> {len(items)} items on page {page}")
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
    
    print(f"\nTotal unique items: {len(all_items)}")
    
    records = []
    for idx, item in enumerate(all_items):
        print(f"  [{idx+1}/{len(all_items)}] {item['title'][:30]}...")
        detail = crawl_detail(item)
        if detail:
            records.append(detail)
        time.sleep(0.3)
    
    if records:
        count = save_to_db(records)
        print(f"\nInserted {count} new records (trigger auto-syncs FTS)")
    else:
        print("\nNo records to insert")


if __name__ == '__main__':
    if '--incremental' in sys.argv:
        MAX_PAGES = 1
    main()
