#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Crawler for 马鞍山慈湖高新区 - 通知公告
import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import os

LIST_URL = "https://chq.mas.gov.cn/content/column/4716840?pageIndex={}"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "马鞍山慈湖高新区通知公告"
FULL = "--full" in sys.argv

CONTENT_PAT = re.compile(r'<div\s+class="wzcon\s+j-fontContent"[^>]*>([\s\S]*?)</div>\s*</div>')
TITLE_META_PAT = re.compile(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"')
DATE_META_PAT = re.compile(r'<meta\s+name="PubDate"\s+content="([^"]+)"')

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
    except:
        return None, None, None
    
    tm = TITLE_META_PAT.search(html)
    dm = DATE_META_PAT.search(html)
    cm = CONTENT_PAT.search(html)
    
    title = tm.group(1).strip() if tm else None
    pub_date = dm.group(1)[:10] if dm else None
    content = cm.group(1).strip() if cm else None
    return title, pub_date, content

def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = total_before = 0
    
    pages_to_crawl = 5 if not FULL else 100
    print(f"Crawling {pages_to_crawl} pages")
    
    for pg in range(1, pages_to_crawl + 1):
        try:
            r = requests.get(LIST_URL.format(pg), headers=HEADERS, timeout=15)
            r.encoding = "utf-8"
        except:
            print(f"Page {pg}: fetch error")
            continue
        
        soup = BeautifulSoup(r.text, "html.parser")
        items = [li for li in soup.select("ul > li") 
                 if li.find("a") and li.find("span", class_="date")]
        
        if not items:
            print(f"Page {pg}: empty")
            break
        
        page_new = page_skip = page_before = 0
        for li in items:
            a = li.find("a")
            sp = li.find("span", class_="date")
            
            href = a.get("href", "")
            if not href.startswith("http"):
                href = "https://chq.mas.gov.cn" + href
            
            title = a.get("title", "") or a.get_text(strip=True)
            pub_date = sp.get_text(strip=True)
            
            if pub_date and pub_date < CUTOFF_DATE:
                page_before += 1
                continue
            
            c.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                page_skip += 1
                continue
            
            dt, dd, content = extract_detail(href)
            ft = dt or title
            fd = dd or pub_date
            
            if content is None:
                print(f"  Skip (no content): {ft[:40]}...")
                page_skip += 1
                continue
            
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary) "
                "VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, href, href, ft, fd, content, ft)
            )
            page_new += 1
            total_new += 1
        
        conn.commit()
        print(f"Page {pg}: +{page_new} new, {page_skip} skip, {page_before} pre-cutoff")
        
        if page_before == len(items) and pg < pages_to_crawl:
            print("All remaining before cutoff, stopping")
            break
    
    conn.close()
    print(f"\nDone: {total_new} new, {total_skip} skip, {total_before} before cutoff")
    return total_new

if __name__ == "__main__":
    run()
