#!/usr/bin/env python3
import os
"""永修县-生态环境局-通知公告爬虫 | TRS WCM | requests直取"""

import requests, re, time, os
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://www.yongxiu.gov.cn/bmxzxxgk/bmgk/sthjj/qtfdgkxx/gggs_191364/"
SITE_NAME = "永修县-通知公告"
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

def get_soup(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            return BeautifulSoup(r.text, 'html.parser')
        except Exception as e:
            if attempt < 2:
                time.sleep(2)
            else:
                print(f"  [ERROR] {e}")
                return None

def parse_list(soup):
    items = []
    table = soup.select_one('.table_wrap table')
    if not table:
        return items
    for tr in table.select('tbody tr'):
        tds = tr.find_all('td')
        if len(tds) >= 3:
            a = tds[1].find('a')
            date_td = tds[2]
            if a and date_td:
                href = a.get('href', '')
                title = a.get('title', a.get_text(strip=True))
                date_str = date_td.get_text(strip=True)
                if href and date_str:
                    full_url = urljoin(BASE_URL, href)
                    items.append((title, full_url, date_str))
    return items

def get_detail(url):
    soup = get_soup(url)
    if not soup:
        return "", "", ""
    title_tag = soup.select_one('.mainTitle')
    title = title_tag.get_text(strip=True) if title_tag else ""
    content = soup.select_one('.trs_editor_view')
    if not content:
        return title, "", ""
    body = str(content)
    summary = re.sub(r'<[^>]+>', '', body)[:200].strip()
    return title, body, summary

def main():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = 0

    for page_idx in range(2):  # pages 0-12 (3年cutoff)
        url = BASE_URL + "index.html" if page_idx == 0 else BASE_URL + f"index_{page_idx}.html"
        print(f"\n--- 第{page_idx+1}页 ---")
        soup = get_soup(url)
        if not soup:
            continue
        items = parse_list(soup)
        if not items:
            print("  无列表，跳过")
            continue
        print(f"  {len(items)}条")
        for title, detail_url, date_str in items:
            if date_str < CUTOFF_DATE:
                print(f"  [SKIP] {date_str} 早于{CUTOFF_DATE}")
                continue
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (detail_url,))
            if c.fetchone():
                total_skip += 1
                continue
            time.sleep(0.5)
            full_title, body, summary = get_detail(detail_url)
            if not body:
                print(f"  [EMPTY] {full_title or title[:30]}")
                continue
            final_title = full_title or title
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (title, page_url, content, site_name, publish_date, source_url, category, status, summary)
                VALUES (?,?,?,?,?,?,?,?,?)""",
                (final_title, detail_url, body, SITE_NAME, date_str, detail_url, '环评', 'published', summary))
            conn.commit()
            total_new += 1
            print(f"  [NEW] {final_title[:40]}... | {date_str}")
        time.sleep(1)

    conn.close()
    print(f"\n========== 汇总 ==========")
    print(f"站点: {SITE_NAME}")
    print(f"新增: {total_new}")
    print(f"跳过: {total_skip}")
    print(f"==========================")

if __name__ == "__main__":
    main()
