#!/usr/bin/env python3
"""万年县人民政府 - 建设项目环境影响评价爬虫
站点: www.zgwn.gov.cn
栏目: /ZWGK_2_0/JSXMHJYXPJ/list_1.shtml (建设项目环境影响评价)
CMS: 政府信息公开系统，table列表+json分页
"""

import requests
import sqlite3
import re
import sys
import os
import json
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "http://www.zgwn.gov.cn"
LIST_PATH = "/ZWGK_2_0/JSXMHJYXPJ/"
DB_PATH = "/root/search.db"
SITE_NAME = "www.zgwn.gov.cn-hjxp"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def fetch_page(url, timeout=30):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=timeout)
        resp.encoding = 'utf-8'
        if resp.status_code == 200:
            return resp.text
        print(f"  [WARN] HTTP {resp.status_code} for {url}", file=sys.stderr)
        return None
    except Exception as e:
        print(f"  [ERROR] fetch failed: {e}", file=sys.stderr)
        return None

def parse_list(html):
    """解析列表页 - table.table3#doclist > tr"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')

    table = soup.find('table', class_='table3')
    if not table:
        table = soup.find('table', id='doclist')
    if not table:
        print("  [WARN] table#doclist not found", file=sys.stderr)
        return items

    for tr in table.find_all('tr'):
        td_link = tr.find('td', width=lambda v: v and '83%' in str(v) if v else False)
        td_date = tr.find('td', width=lambda v: v and '18%' in str(v) if v else False)
        if not td_link or not td_date:
            continue

        a = td_link.find('a')
        if not a or not a.get('href'):
            continue

        href = a['href'].strip()
        if href.startswith('/'):
            href = BASE_URL + href
        elif href.startswith('http'):
            pass
        else:
            href = BASE_URL + '/' + href

        # 标题 - 优先 title 属性(完整)
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        title = re.sub(r'\s+', ' ', title)
        title = re.sub(r'\.\.\.$', '', title).strip()
        if not title:
            continue

        # 日期
        date_span = td_date.find('span')
        pub_date = date_span.get_text(strip=True) if date_span else td_date.get_text(strip=True)
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', pub_date)
        pub_date = date_match.group(1) if date_match else ''

        items.append({'url': href, 'title': title, 'pub_date': pub_date})

    return items

def parse_detail(html, url):
    """解析详情页（同zgwn.gov.cn主站结构）"""
    soup = BeautifulSoup(html, 'html.parser')

    # ── 标题 ──
    # 方式1: div.bt_qu > b (最干净)
    title = ''
    bt_qu = soup.find('div', class_='bt_qu')
    if bt_qu:
        b_tag = bt_qu.find('b')
        if b_tag:
            title = b_tag.get_text(strip=True)
    # 方式2: 从元数据表 <th>标题</th> 下一列取
    if not title:
        for th in soup.find_all('th'):
            if '标题' in th.get_text(strip=True):
                td = th.find_next_sibling('td')
                if td:
                    title = td.get_text(strip=True)
                    break
    # 方式3: h1
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    # 方式4: ucaptitle (兼容旧模板)
    if not title:
        ucap = soup.find('ucaptitle')
        if ucap:
            title = ucap.get_text(strip=True)
    # 方式5: title标签（去掉面包屑后缀）
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            t = re.sub(r'[-_|]\s*(?:建设项目环境影响评价|政务公开专区V2|政府信息公开).*$', '', t)
            title = t.strip()

    # 发布时间 PUBLISHTIME
    pub_date = ''
    pt = soup.find('PUBLISHTIME')
    if pt:
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', pt.get_text())
        if date_match:
            pub_date = date_match.group(1)

    # 正文
    content_div = soup.find('div', class_='nr')
    if not content_div:
        content_div = soup.find('div', class_='wzcon j-fontContent')
    if not content_div:
        content_div = soup.find('div', class_='wzcon')
    if not content_div:
        content_div = soup.find('div', id='wenzhang')
    # 兜底: 直接找 UCAPCONTENT
    if not content_div:
        ucap = soup.find('UCAPCONTENT')
        if ucap:
            content_div = ucap

    # 附件（整个页面找附件，不局限于content_div）
    attachments = []
    # 先在全页找附件链接
    for a_tag in soup.find_all('a'):
        href = a_tag.get('href', '').strip()
        if any(href.lower().endswith(ext) for ext in
               ['.doc', '.docx', '.pdf', '.xls', '.xlsx', '.zip', '.rar', '.ppt', '.pptx']):
            if not href.startswith('http'):
                href = requests.compat.urljoin(url, href)
            att_title = a_tag.get_text(strip=True) or os.path.basename(href)
            # 过滤掉太短的附件名（导航栏等误匹配）
            if len(att_title) < 2:
                att_title = os.path.basename(href)
            if not any(a['url'] == href for a in attachments):
                attachments.append({'url': href, 'title': att_title})

    content = ''
    tables_html = ''
    if content_div:
        for tag in content_div.find_all(['script', 'style']):
            tag.decompose()

        seen_texts = set()
        for table in content_div.find_all('table'):
            rows = table.find_all('tr')
            # 接受多列表格(≥3列)或key-value表格(2列且有结构数据)
            has_data = False
            if rows:
                max_cols = max(len(row.find_all(['td', 'th'])) for row in rows)
                has_data = max_cols >= 2
            if not has_data:
                continue
            txt = table.get_text(strip=True)
            if txt in seen_texts:
                continue
            seen_texts.add(txt)
            tables_html += str(table) + '\n\n'

        content_div_clean = BeautifulSoup(str(content_div), 'html.parser')
        for tag in content_div_clean.find_all('table'):
            tag.decompose()

        paragraphs = content_div_clean.find_all('p')
        if paragraphs:
            para_texts = []
            for p in paragraphs:
                p_text = p.get_text(strip=True)
                if p_text:
                    para_texts.append(p_text)
            content = '\n\n'.join(para_texts)
        else:
            content = content_div_clean.get_text(separator='\n\n', strip=True)

        content = re.sub(r'\n{4,}', '\n\n', content)
        if tables_html:
            content += '\n\n[表格]\n\n' + tables_html

    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>\n'
        for att in attachments:
            content += f'\n附件: [{att["title"]}]({att["url"]})'

    return title, content, pub_date, attachments

def save_article(conn, page_url, title, summary_text, publish_date, attachments, site_name):
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    level = abs(hash(page_url)) % 10 + 1
    conn.execute("""
        INSERT OR REPLACE INTO gov_raw (page_url, site_name, source_url, title, publish_date, summary, content, attachments, date_rank, status, category, visits, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'published', '建设项目环境影响评价', 0, 'crawl_zgwn_hjxp.py')
    """, (page_url, site_name, page_url, title, publish_date, summary_text, summary_text, attachments_json, level))
    conn.commit()

def crawl(max_pages=5, incremental=False):
    conn = init_db()
    total = 0
    errors = 0
    seen_urls = set()

    for page in range(max_pages):
        if page == 0:
            list_url = BASE_URL + LIST_PATH + "list_1.shtml"
        else:
            list_url = BASE_URL + LIST_PATH + f"list_{page+1}.shtml"

        print(f"[PAGE {page+1}] {list_url}")
        html = fetch_page(list_url)
        if not html:
            print(f"  [WARN] Page {page+1} unreachable, stopping")
            break

        items = parse_list(html)
        if not items:
            print(f"  No articles found, stopping")
            break

        unique_items = [it for it in items if it['url'] not in seen_urls]
        for it in items:
            seen_urls.add(it['url'])
        print(f"  Found {len(items)} articles ({len(unique_items)} new)")

        if not unique_items:
            if page > 0:
                print(f"  No new items, stopping")
                break

        for item in unique_items:
            url = item['url']
            title = item['title']
            list_date = item['pub_date']

            if incremental:
                existing = conn.execute(
                    "SELECT page_url FROM gov_raw WHERE page_url = ?", (url,)
                ).fetchone()
                if existing:
                    continue

            print(f"  [FETCH] {title[:50]}...")
            detail_html = fetch_page(url)
            if not detail_html:
                errors += 1
                print(f"  [ERROR] detail page unreachable: {url}")
                continue

            detail_title, content, pub_date, attachments = parse_detail(detail_html, url)
            if not pub_date and list_date:
                pub_date = list_date

            if detail_title:
                save_article(conn, url, detail_title, content, pub_date, attachments, SITE_NAME)
                total += 1
                att_count = len(attachments)
                att_str = f" +{att_count}附件" if att_count else ""
                print(f"    o saved [{pub_date}] {detail_title[:50]}...{att_str}")
            else:
                print(f"    x empty title, skipped")

    conn.close()
    print(f"\n[DONE] Total: {total} articles, Errors: {errors}")

if __name__ == '__main__':
    import argparse
    parser = argparse.ArgumentParser(description='万年县建设项目环境影响评价爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数')
    parser.add_argument('--incremental', action='store_true', help='增量模式')
    # Handle bare numeric arg for _MAX_PG
    import sys as _SYS2
    _MAX_PG = int(_SYS2.argv[-1]) if len(_SYS2.argv) > 1 and _SYS2.argv[-1].isdigit() else None
    if _MAX_PG is not None:
        print('[AutoPg] max_pages=' + str(_MAX_PG))
        _SYS2.argv.pop()
    args = parser.parse_args()
    crawl(max_pages=args.max_pages, incremental=args.incremental)
