#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Crawler for 安徽歙县经济开发区 - 公示公告
import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import os

LIST_URL = "http://kfq.ahshx.gov.cn/content/column/6795674?pageIndex={}"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "安徽歙县经开区公示公告"
FULL = "--full" in sys.argv

CONTENT_PAT = re.compile(r'<div\s+class="j-fontContent\s+newscontnet\s+minh500\s+clearfix"[^>]*>([\s\S]*?)</div>\s*</div>')
TITLE_META_PAT = re.compile(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"')
DATE_META_PAT = re.compile(r'<meta\s+name="PubDate"\s+content="([^"]+)"')

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
    except:
        return None, None, None
    tm = TITLE_META_PAT.search(r.text)
    dm = DATE_META_PAT.search(r.text)
    cm = CONTENT_PAT.search(r.text)
    title = tm.group(1).strip() if tm else None
    pub_date = dm.group(1)[:10] if dm else None
    content = cm.group(1).strip() if cm else None
    return title, pub_date, content

def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = total_before = 0
    pages = 5 if not FULL else 15
    print(f"Crawling {pages} pages")
    for pg in range(1, pages + 1):
        try:
            r = requests.get(LIST_URL.format(pg), headers=HEADERS, timeout=15)
            r.encoding = "utf-8"
        except:
            print(f"Page {pg}: fetch error"); continue
        soup = BeautifulSoup(r.text, "html.parser")
        items = [li for li in soup.select("ul > li") if li.find("a") and li.find("span")]
        if not items:
            print(f"Page {pg}: empty"); break
        page_new = page_skip = page_before = 0
        for li in items:
            a = li.find("a"); sp = li.find("span")
            href = a.get("href", "")
            if not href.startswith("http"):
                href = "http://kfq.ahshx.gov.cn" + href
            title = a.get("title", "") or a.get_text(strip=True)
            pub_date = sp.get_text(strip=True)
            if pub_date and pub_date < CUTOFF_DATE:
                page_before += 1; continue
            c.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                page_skip += 1; continue
            dt, dd, content = extract_detail(href)
            ft = dt or title; fd = dd or pub_date
            if content is None:
                print(f"  Skip (no content): {ft[:40]}...")
                page_skip += 1; continue
            c.execute("INSERT OR IGNORE INTO gov_raw (site_name,source_url,page_url,title,publish_date,content,summary) VALUES (?,?,?,?,?,?,?)",
                      (SITE_NAME, href, href, ft, fd, content, ft))
            page_new += 1; total_new += 1
        conn.commit()
        print(f"Page {pg}: +{page_new} new, {page_skip} skip, {page_before} pre-cutoff")
        if page_before == len(items) and pg < pages:
            print("All remaining before cutoff, stopping"); break
    conn.close()
    print(f"\nDone: {total_new} new, {total_skip} skip, {total_before} before cutoff")

if __name__ == "__main__":
    run()
