#!/usr/bin/env python3
"""补爬唐山海港剩余页"""
import requests, re, time, sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

DB_PATH = "/root/search.db"
BASE_URL = "https://www.tshg.gov.cn/tshaigang/tshgxmjs/"
SITE_NAME = "唐山海港经济开发区-项目建设"
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0"}

def get_soup(url):
    for _ in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            return BeautifulSoup(r.text, 'html.parser')
        except:
            time.sleep(2)
    return None

conn = sqlite3.connect(DB_PATH)
conn.execute("PRAGMA busy_timeout=10000")
c = conn.cursor()

total_new = 0
for page_idx in range(14, 17):
    url = BASE_URL + "index.html" if page_idx == 1 else BASE_URL + "index_{}.html".format(page_idx)
    print("\n--- 第{}页 ---".format(page_idx))
    soup = get_soup(url)
    if not soup:
        continue
    ul = soup.find('ul', class_='list')
    if not ul:
        continue
    items = []
    for li in ul.find_all('li'):
        a = li.find('a')
        span = li.find('span', class_='date')
        if a and span:
            href = a.get('href', '')
            title = a.get('title', a.get_text(strip=True))
            date_str = span.get_text(strip=True)
            if href and date_str:
                items.append((title, urljoin(BASE_URL, href), date_str))
    print("  {}条".format(len(items)))
    for title, detail_url, date_str in items:
        if date_str < CUTOFF_DATE:
            print("  [SKIP] {} 早于{}".format(date_str, CUTOFF_DATE))
            continue
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (detail_url,))
        if c.fetchone():
            continue
        for _ in range(3):
            try:
                r = requests.get(detail_url, headers=HEADERS, timeout=20)
                r.encoding = 'utf-8'
                soup2 = BeautifulSoup(r.text, 'html.parser')
                break
            except:
                time.sleep(2)
        else:
            continue
        title_tag = soup2.select_one('h1.top20')
        full_title = title_tag.get_text(strip=True) if title_tag else title
        content_div = soup2.select_one('.conten_box')
        if not content_div:
            print("  [EMPTY] {}".format(full_title[:30]))
            continue
        body = str(content_div)
        summary = re.sub(r'<[^>]+>', '', body)[:200].strip()
        c.execute("""INSERT OR IGNORE INTO gov_raw 
            (title, page_url, content, site_name, publish_date, source_url, category, status, summary)
            VALUES (?,?,?,?,?,?,?,?,?)""",
            (full_title, detail_url, body, SITE_NAME, date_str, detail_url, '环评', 'published', summary))
        conn.commit()
        total_new += 1
        print("  [NEW] {}... | {}".format(full_title[:40], date_str))
        time.sleep(0.5)
    time.sleep(1)

conn.close()
print("\n补爬新增: {}".format(total_new))
