#!/usr/bin/env python3
"""张掖经济技术开发区 - 通知公告爬虫
站点: www.zhangye.gov.cn
栏目: /jjkfq/gzdt/tzgg/ (通知公告)
CMS: TRS (Terton) - 标准TRS分页
"""

import requests
import sqlite3
import re
import sys
import os
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE_URL = "http://www.zhangye.gov.cn"
LIST_PATH = "/jjkfq/gzdt/tzgg/"
DB_PATH = "/root/search.db"
SITE_NAME = "www.zhangye.gov.cn-jjkfq-tzgg"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def fetch_page(url, timeout=30):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=timeout, verify=False)
        resp.encoding = 'utf-8'
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"  [ERROR] fetch failed: {e}", file=sys.stderr)
        return None

def parse_list(html, page_num):
    """解析列表页，返回 (url, title, date) 列表"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='news_list')
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = a['href'].strip()
        if not href.endswith('.html'):
            continue
        if href.startswith('./'):
            href = BASE_URL + LIST_PATH + href[2:]
        elif not href.startswith('http'):
            href = BASE_URL + LIST_PATH + href

        title_text = a.get_text(strip=True)
        date_span = a.find('span')
        pub_date = ''
        if date_span:
            pub_date = date_span.get_text(strip=True)
            title_text = title_text.replace(pub_date, '').strip()
        title_text = re.sub(r'</?TRS_DOCUMENT>', '', title_text).strip()
        title_text = re.sub(r'\s+', ' ', title_text)
        if not title_text:
            continue
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', pub_date)
        pub_date = date_match.group(1) if date_match else ''

        items.append({'url': href, 'title': title_text, 'pub_date': pub_date})
    return items

def parse_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')

    title_el = soup.find('h1', class_='details_title')
    title = title_el.get_text(strip=True) if title_el else ''
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            t = re.sub(r'^张掖经济技术开发区[-－—]?\s*', '', t)
            title = t.strip()

    pub_date = ''
    date_el = soup.find('p', class_='second_title')
    if date_el:
        date_text = date_el.get_text(strip=True)
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', date_text)
        if date_match:
            pub_date = date_match.group(1)

    content_div = soup.find('div', class_='view TRS_UEDITOR')
    if not content_div:
        content_div = soup.find('div', class_='TRS_UEDITOR')
    if not content_div:
        nc = soup.find('div', class_='news_content')
        if nc:
            content_div = nc.find('div', class_='view')

    attachments = []
    content = ''
    tables_html = ''

    if content_div:
        for a_tag in content_div.find_all('a'):
            href = a_tag.get('href', '')
            if any(href.endswith(ext) for ext in ['.doc', '.docx', '.pdf', '.xls', '.xlsx', '.zip', '.rar', '.txt', '.ppt', '.pptx']):
                if not href.startswith('http'):
                    href = requests.compat.urljoin(url, href)
                att_title = a_tag.get_text(strip=True) or os.path.basename(href)
                attachments.append({'url': href, 'title': att_title})

        for tag in content_div.find_all(['script', 'style']):
            tag.decompose()

        raw_html = str(content_div)
        raw_html = re.sub(r'</?span[^>]*>', '', raw_html, flags=re.IGNORECASE)
        raw_html = re.sub(r'</?(?:b|font|strong|em|u|i|a)[^>]*>', '', raw_html, flags=re.IGNORECASE)

        content_div_clean = BeautifulSoup(raw_html, 'html.parser')

        seen_texts = set()
        for table in content_div.find_all('table'):
            rows = table.find_all('tr')
            has_data_row = any(len(row.find_all('td')) >= 3 for row in rows)
            if not has_data_row:
                continue
            first_row = rows[0] if rows else None
            if first_row:
                cells = first_row.find_all(['td', 'th'])
                is_data_header = all(len(c.get_text(strip=True)) <= 40 for c in cells) if cells else False
                if not is_data_header:
                    continue
            content_text = table.get_text(strip=True)
            if content_text in seen_texts:
                continue
            seen_texts.add(content_text)
            tables_html += str(table) + '\n\n'

        for tag in content_div_clean.find_all('table'):
            tag.decompose()

        content = content_div_clean.get_text(separator='\n\n', strip=True)
        content = re.sub(r'\n{4,}', '\n\n', content)

        if tables_html:
            content += '\n\n[表格]\n' + tables_html

    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>\n'
        for att in attachments:
            content += f'\n附件: [{att["title"]}]({att["url"]})'

    return title, content, pub_date, attachments


def save_article(conn, page_url, title, summary_text, publish_date, attachments, site_name):
    import json
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    now = datetime.now().strftime('%Y-%m-%d %H:%M:%S')
    level = abs(hash(page_url)) % 10 + 1
    conn.execute("""
        INSERT OR REPLACE INTO gov_raw 
        (page_url, site_name, source_url, title, publish_date, summary, content, attachments, date_rank, status, category, visits)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'published', '通知公告', 0)
    """, (page_url, site_name, page_url, title, publish_date, summary_text, summary_text, attachments_json, level))
    conn.commit()


def crawl(max_pages=5, incremental=False):
    conn = init_db()
    total = 0
    errors = 0

    for page in range(max_pages):
        if page == 0:
            list_url = BASE_URL + LIST_PATH + "index.html"
        else:
            list_url = BASE_URL + LIST_PATH + f"index_{page}.html"

        print(f"[PAGE {page+1}] {list_url}")
        html = fetch_page(list_url)
        if not html:
            print(f"  [WARN] Page {page+1} unreachable, stopping")
            break

        items = parse_list(html, page)
        print(f"  Found {len(items)} articles")

        if not items:
            print(f"  No items found, stopping")
            break

        for item in items:
            url = item['url']
            title = item['title']
            list_date = item['pub_date']

            if incremental:
                existing = conn.execute(
                    "SELECT url FROM gov_raw WHERE url = ?", (url,)
                ).fetchone()
                if existing:
                    print(f"  [SKIP] already exists: {title[:40]}...")
                    continue

            print(f"  [FETCH] {title[:50]}...")
            detail_html = fetch_page(url)
            if not detail_html:
                errors += 1
                print(f"  [ERROR] detail page unreachable: {url}")
                continue

            detail_title, content, pub_date, attachments = parse_detail(detail_html, url)
            if not pub_date and list_date:
                pub_date = list_date

            if detail_title:
                save_article(conn, url, detail_title, content, pub_date, attachments, SITE_NAME)
                total += 1
                print(f"    ✓ saved [{pub_date}] {detail_title[:50]}...")
            else:
                print(f"    ✗ empty title, skipped")

    conn.close()
    print(f"\n[DONE] Total: {total} articles, Errors: {errors}")


if __name__ == '__main__':
    import argparse
    parser = argparse.ArgumentParser(description='张掖经济技术开发区通知公告爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数')
    parser.add_argument('--incremental', action='store_true', help='增量模式')
    args = parser.parse_args()
    crawl(max_pages=args.max_pages, incremental=args.incremental)
