#!/usr/bin/env python3
import os
"""精河县—重大建设项目批准与实施爬虫 (修复版)
URL: https://www.xjjh.gov.cn/zwgk/zfxxgk/fdzdgknr/zdjsxmpzyss.htm
"""

import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
BASE = "https://www.xjjh.gov.cn"
LIST_PAGE = "/zwgk/zfxxgk/fdzdgknr/zdjsxmpzyss.htm"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE = "精河县-重大建设项目"

def fetch_page(num):
    if num == 1:
        url = BASE + LIST_PAGE
    else:
        url = "%s/zwgk/zfxxgk/fdzdgknr/zdjsxmpzyss/%d.htm" % (BASE, num)
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for tr in soup.select("table.winstyle27637 tr"):
        a = tr.find("a")
        date_span = tr.find("span", class_="timestyle27637")
        if a and date_span:
            href = a.get("href", "")
            if href and "info/" in href:
                title = a.get("title", "") or a.text.strip()
                date_str = date_span.text.strip().replace("年", "-").replace("月", "-").replace("日", "").strip()
                items.append({"title": title, "date": date_str, "href": href})
    return items

def get_detail_and_url(href):
    full_url = urljoin(BASE + LIST_PAGE, href)
    try:
        r = requests.get(full_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except:
        return "", full_url
    soup = BeautifulSoup(r.text, "html.parser")
    div = soup.select_one("div.v_news_content") or soup.select_one("div#vsb_content_4")
    if div:
        return str(div), full_url
    return "", full_url

def main():
    today = datetime.now().strftime("%Y-%m-%d")
    count = 0
    conn = sqlite3.connect(os.getenv("SEARCH_DB", "/root/search.db"), timeout=60)
    c = conn.cursor()
    max_pages = 20

    for page in range(1, max_pages + 1):
        html = fetch_page(page)
        items = parse_list(html)
        if not items:
            print("第%d页无数据，结束" % page)
            break

        for it in items:
            if it["date"] < CUTOFF:
                if items.index(it) == 0:
                    print("第%d页无3年内数据，结束" % page)
                    return
                continue
            sys.stdout.write("  [%s] %s... " % (it["date"], it["title"][:40]))
            sys.stdout.flush()
            body, full_url = get_detail_and_url(it["href"])
            if not body:
                print("无正文")
                continue
            try:
                c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_xjjh.py')""",
                    (it["href"], it["title"], body, it["date"], SITE, full_url, "", ""))
                conn.commit()
                count += 1
                print("OK")
            except Exception as e:
                print("DB ERROR: %s" % e)

    conn.close()
    print("\n✅ 精河县：共入库 %d 条" % count)

if __name__ == "__main__":
    main()
