#!/usr/bin/env python3
"""
crawl_suichang.py — 遂昌县人民政府-建设项目环境影响评价信息公示
Site: https://www.suichang.gov.cn/col/col1229856870/index.html
CMS: Hanweb JPAAS 4.5.12.1 (ZJSJYH)
List: API /api-gateway/jpaas-publish-server/front/page/build/unit
Detail: div.zfxxgk_zdgkc > <p> article body
"""
import re, os, sys, time, json, sqlite3, requests, urllib.parse
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "suichang.gov.cn-建设项目环境影响评价信息公示"
BASE_URL = "https://www.suichang.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE_URL + "/col/col1229856870/index.html",
}

session = requests.Session()
session.headers.update(HEADERS)

# Base API params (from unitbuild.js queryData)
BASE_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "3640",
    "tplSetId": "cgFjwQ1lk8DUXJTOn8cFr",
    "pageType": "column",
    "tagId": "文章列表",
    "editType": "null",
    "pageId": "1229856870",
}

PAGE_SIZE = 20


def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERR] fetch failed: {url} - {e}")
        return None


def fetch_api(page_no):
    """Fetch list page from API"""
    params = dict(BASE_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    qs = urllib.parse.urlencode(params)
    url = f"{API_URL}?{qs}"
    try:
        r = session.get(url, timeout=30)
        if r.status_code != 200:
            return None
        d = r.json()
        return d.get("data", {}).get("html", "")
    except Exception as e:
        print(f"  [ERR] API page {page_no}: {e}")
        return None


def parse_list(html):
    """Parse list HTML, return list of (url, title, date_str)"""
    items = []
    # Pattern: <li><a href="URL" title="TITLE">display</a><b>DATE</b></li>
    pattern = r'<a href="([^"]+)"\s+title="([^"]*)"[^>]*>.*?</a><b>(\d{4}-\d{2}-\d{2})</b>'
    for m in re.finditer(pattern, html):
        href = m.group(1)
        title = m.group(2).strip()
        date_str = m.group(3)
        if not href.startswith('http'):
            href = BASE_URL + href
        items.append((href, title, date_str))
    return items


def parse_detail(html, url):
    """Parse detail page, return (title, publish_date, content)"""
    # Title from meta
    title = ""
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    
    # Title from <title> as fallback
    if not title:
        m = re.search(r'<title>(.*?)</title>', html)
        if m:
            title = m.group(1).strip()
    
    # Date from meta
    date_str = ""
    m = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        date_str = m.group(1)
    
    # Content: extract from zn_contM div (after zcjd/policy section)
    content = ""
    # Find zn_contM container
    znm_start = html.find('class="zn_contM"')
    if znm_start < 0:
        znm_start = html.find('zn_contM')
    
    if znm_start >= 0:
        # Find the start of the div
        div_start = html.find('>', znm_start) + 1
        
        # Find end of container
        depth = 1
        i = div_start
        while i < len(html) and depth > 0:
            if html[i:i+4] == '<!--':
                endc = html.find('-->', i)
                i = endc + 3 if endc > i else i + 1
                continue
            if html[i] == '<':
                if html[i:i+5] == '</div':
                    depth -= 1
                    if depth == 0:
                        inner = html[div_start:i]
                        # Skip ywlj (原文链接) and zcjd (政策解读) divs
                        zcjd_end = inner.find('</div>', inner.find('class="zcjd"'))
                        if zcjd_end > 0:
                            zcjd_end += 7  # len of </div>
                            body_start = zcjd_end
                        else:
                            body_start = 0
                        
                        # End before 附件/fj or share section
                        body_end = len(inner)
                        for mk in ['class="fj"', 'class="fj2"', '<div class="ewmBox']:
                            pos = inner.find(mk, body_start)
                            if pos > 0 and pos < body_end:
                                body_end = pos
                        
                        if body_start < body_end:
                            content = inner[body_start:body_end].strip()
                        break
                elif html[i:i+4] == '<div':
                    depth += 1
                i += 1
            else:
                i += 1
    
    return title, date_str, content


def get_total_pages():
    """Get total pages from API page 1 response"""
    html = fetch_api(1)
    if not html:
        return 1
    # count="76" rows="20"
    m = re.search(r'count="(\d+)"', html)
    if m:
        total = int(m.group(1))
        pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
        print(f"  Total items: {total}, pages: {pages}")
        return pages
    return 1


def main():
    print(f"=== {SITE_NAME} ===")
    print(f"Cutoff: {CUTOFF}")
    
    total_pages = get_total_pages()
    if total_pages == 0:
        print("ERROR: No pages found")
        return
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    cur = conn.cursor()
    
    total_new = 0
    total_skip = 0
    
    for page in range(1, total_pages + 1):
        print(f"\nPage {page}/{total_pages}...")
        html = fetch_api(page)
        if not html:
            print(f"  [SKIP] page {page} API failed")
            continue
        
        items = parse_list(html)
        if not items:
            print(f"  [SKIP] page {page} no items")
            continue
        
        print(f"  Found {len(items)} items")
        
        page_new = 0
        for url, title, date_str in items:
            # Check if already in DB
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skip += 1
                continue
            
            # Filter by date
            if date_str < CUTOFF:
                total_skip += 1
                continue
            
            # Fetch detail
            detail_html = fetch(url)
            if not detail_html:
                total_skip += 1
                continue
            
            full_title, detail_date, content = parse_detail(detail_html, url)
            
            if not full_title:
                full_title = title
            
            final_date = detail_date or date_str
            if not final_date or final_date < CUTOFF:
                total_skip += 1
                continue
            
            content = content.strip()
            if not content:
                total_skip += 1
                continue
            
            # Summary
            summary = ''
            if content:
                text_soup = BeautifulSoup(content, 'html.parser')
                plain = text_soup.get_text(strip=True)
                summary = plain[:200]
            
            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, full_title, url, final_date, content,
                     int(final_date.replace("-", "")), 'tzgg')
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
            except Exception as e:
                print(f"  [ERR] insert: {e}")
        
        conn.commit()
        print(f"  Page {page}: +{page_new} new (total {total_new}, skip {total_skip})")
        time.sleep(0.3)
    
    conn.close()
    print(f"\n=== Done: {total_new} new, {total_skip} skipped ===")


if __name__ == "__main__":
    main()
