#!/usr/bin/env python3
"""crawl_changshu_gsgg.py - 常熟市人民政府-公示 (静态ul a[title]+span.time, --col 支持gongshi/c100280)"""
import sys, os, requests, re
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0'}
SITES = {
    'gongshi': {'name': '常熟市-公示', 'path': '/zgcs/gongshi'},
    'c100280': {'name': '常熟市-公示公告', 'path': '/zgcs/c100280'},
}

def get_site(col):
    return SITES.get(col, SITES['gongshi'])

def fetch_urls_from_page(site, page_num):
    BASE = 'https://www.changshu.gov.cn' + site['path']
    if page_num == 0:
        url = BASE + '/list.shtml'
    else:
        url = f'{BASE}/list{page_num + 1}.shtml'
    try:
        r = requests.get(url, timeout=(5, 15), headers=HEADERS, verify=False)
        r.encoding = 'utf-8'
    except:
        return []
    # 列表条目: <li><a href="/zgcs/{col}/202608/xxx.shtml" title="标题">标题</a><span class="time">日期</span></li>
    # 详情链接必含年月子目录 (\d{6}/), 导航链接(index.shtml等)排除
    items = re.findall(
        r'<a href="(/zgcs/[^"]+/\d{6}/[^"]+\.shtml)"[^>]*title="([^"]+)"[^>]*>.*?</a>\s*<span[^>]*class="[^"]*time[^"]*"[^>]*>\s*(\d{4}-\d{2}-\d{2})\s*</span>',
        r.text, re.DOTALL
    )
    results = []
    for href, title, pub_date in items:
        if title in ('English', '首页', '智能问答'):
            continue
        title = re.sub(r'\s+', ' ', title).strip()
        if href.startswith('/'):
            full_url = 'https://www.changshu.gov.cn' + href
        else:
            full_url = BASE + '/' + href
        results.append({'url': full_url, 'pub_date': pub_date, 'title': title})
    return results

def fetch_detail(url):
    for _attempt in range(3):
        try:
            r = requests.get(url, timeout=(5, 15), headers=HEADERS, verify=False)
            r.encoding = 'utf-8'
            break
        except:
            r = None
    if r is None:
        return '', ''
    title = ''
    # 优先取页面标题元素, fallback <title> 标签
    # h1 内可含嵌套标签(<UCAPTITLE>等), 提取全部文本再剥标签
    m = re.search(r'<h1[^>]*>(.*?)</h1>', r.text, re.DOTALL)
    if m:
        cand = re.sub(r'<[^>]+>', '', m.group(1)).strip()
        cand = re.sub(r'^begin-->|end-->$', '', cand).strip()
        if cand and '404' not in cand:
            title = cand
    if not title:
        for pat in [r'class="[^"]*title[^"]*"[^>]*>\s*([^<]{5,200})\s*<',
                    r'class="[^"]*Title[^"]*"[^>]*>\s*([^<]{5,200})\s*<']:
            m = re.search(pat, r.text, re.DOTALL)
            if m:
                cand = re.sub(r'<[^>]+>', '', m.group(1)).strip()
                cand = re.sub(r'^begin-->|end-->$', '', cand).strip()
                if cand and '404' not in cand:
                    title = cand
                    break
    if not title:
        m = re.search(r'<title>([^<]*)</title>', r.text, re.DOTALL)
        if m:
            title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
            # 显式剥离栏目名+站点名(不用字符类split, 标题内部连字符/下划线会被误分割截断)
            title = re.sub(r'\s*[-_|]\s*公示公告\s*[-_|]\s*常熟市人民政府\s*$', '', title)
            title = re.sub(r'\s*[-_|]\s*常熟市人民政府\s*$', '', title)
            title = re.sub(r'^\s*常熟市人民政府\s*[-_|]\s*', '', title)
    title = re.sub(r'\s+', ' ', title).strip()
    content = ''
    for marker in ['<div class="article_content" id="zoomcon"', '<div class="trs_editor_view TRS_UEDITOR', '<div class="content"']:
        idx = r.text.find(marker)
        if idx > 0:
            start = r.text.find('>', idx) + 1
            depth = 1
            pos = start
            while pos < len(r.text) and depth > 0:
                if r.text[pos:pos+4] == '<div':
                    depth += 1
                    pos += 4
                elif r.text[pos:pos+6] == '</div>':
                    depth -= 1
                    pos += 6
                else:
                    pos += 1
            content = r.text[start:pos-6].strip()
            if content:
                break
    if not content:
        m = re.search(r'<div class="detail"[^>]*>(.*?)</div>', r.text, re.DOTALL)
        if m:
            content = m.group(1).strip()
    return title, content

def run(col, incremental=False):
    site = get_site(col)
    print(f"  Site: {site['name']}")
    records = []
    seen = set()
    max_pages = 1 if incremental else 50
    for pg in range(max_pages):
        items = fetch_urls_from_page(site, pg)
        if not items:
            break
        print(f"  Page {pg+1}: {len(items)} items")
        for it in items:
            if it['pub_date'] < CUTOFF:
                continue
            if it['url'] in seen:
                continue
            seen.add(it['url'])
            title, content = fetch_detail(it['url'])
            if not title or not content.strip():
                continue
            records.append({
                'title': title,
                'url': it['url'],
                'pub_date': it['pub_date'],
                'site_name': site['name'],
                'content': content,
                'summary': '',
            })
    print(f"  Total: {len(records)} items")
    valid = [r for r in records if r['content'].strip()]
    if valid:
        push_to_searchdb(valid, "changshu_" + col)
    return len(valid)

if __name__ == '__main__':
    col = 'gongshi'
    for a in sys.argv[1:]:
        if a.startswith('--col='):
            col = a.split('=', 1)[1]
    inc = '--incremental' in sys.argv
    cnt = run(col, incremental=inc)
    print(f"Done: {cnt} records")
