#!/usr/bin/env python3
"""
池州经济技术开发区 - 通知公告 (czkfq.chizhou.gov.cn)
JPAAS CMS: Content/showList/2780/page_1.html
详情: /Content/show/{id}.html
正文: div#zoom 或 div.g-detailbox
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, clean_html
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "池州经开区-通知公告"
BASE_URL = "https://czkfq.chizhou.gov.cn"
LIST_TPL = "https://czkfq.chizhou.gov.cn/Content/showList/2780/page_{page}.html"
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (compatible; Googlebot/2.1)"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except: return None

def parse_list(html):
    items = []
    for m in re.finditer(r'<a[^>]*href="(/Content/show/\d+\.html)"[^>]*title="([^"]*)"[^>]*>', html, re.I):
        href = m.group(1)
        title = m.group(2).strip()
        if title and len(title) > 5:
            items.append((title, BASE_URL + href))
    if not items:
        for m in re.finditer(r'<a[^>]*href="(/Content/show/\d+\.html)"[^>]*>(.*?)</a>', html, re.I|re.S):
            href = m.group(1)
            title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
            if title and len(title) > 5:
                items.append((title, BASE_URL + href))
    return items

def fetch_detail(url):
    html = fetch(url)
    if not html: return "", "", ""
    urljoin_base = url.split("/Content/")[0] if "/Content/" in url else url
    title = ""
    m = re.search(r'<div[^>]*class="[^"]*u-title[^"]*"[^>]*>(.*?)</div>', html, re.I|re.S)
    if m: title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    pub_date = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html, re.I)
    if not m: m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
    if m: pub_date = m.group(1)
    content = ""
    m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*?)</div>\s*</div>', html, re.I|re.S)
    if not m: m = re.search(r'<div[^>]*class="g-detailbox"[^>]*>(.*?)</div>', html, re.I|re.S)
    if m and len(m.group(1)) > 50:
        content = m.group(1).strip()
        content = clean_html(content)

    # 附件区: div.m-dtdownload 里的附件链接 → 内嵌 URL 段落追加到正文末尾
    m = re.search(r'<div[^>]*class="m-dtdownload[^"]*"[^>]*>(.*?)</div>', html, re.I|re.S)
    if m:
        att_html = m.group(1)
        att_paras = []
        for am in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', att_html, re.I|re.S):
            href = am.group(1).strip()
            txt = re.sub(r'<[^>]+>', '', am.group(2)).strip()
            # 去掉文件名后的【大小】标注，只保留干净附件名
            txt = re.sub(r'【[^】]*】\s*$', '', txt).strip()
            if not txt:
                continue
            if href.startswith('/'):
                href = urljoin_base + href
            att_paras.append('<p><a href="%s">%s</a></p>' % (href, txt))
        if att_paras and content:
            content = content.rstrip() + "\n\n" + "\n".join(att_paras)
        elif att_paras:
            content = "\n".join(att_paras)
    return title, pub_date, content

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] in ('1', '--incremental', '-i')
    print(f"爬取: {SITE_NAME}" + (" [增量]" if incremental else ""))
    results = []
    pages = 1 if incremental else MAX_PAGES
    for page in range(1, pages+1):
        url = LIST_TPL.format(page=page)
        html = fetch(url)
        if not html or len(html) < 500: break
        items = parse_list(html)
        if not items: break
        print(f"  第{page}页: {len(items)} 条")
        for i, (title, url) in enumerate(items):
            print(f"  [{i+1}/{len(items)}] {title[:40]}...", end=" ", flush=True)
            dt, date, content = fetch_detail(url)
            if date and date < CUTOFF: print("过旧"); continue
            if not content or len(content) < 50: print("无正文"); continue
            final_title = dt or title
            summary = re.sub(r'<[^>]+>', '', content)[:200].strip()
            results.append({"site_name": SITE_NAME, "title": final_title, "url": url,
                           "content": content, "summary": summary, "pub_date": date or ""})
            print(f"OK ({len(content)}字)")
            time.sleep(0.3)
    if results:
        push_to_searchdb(results, "czkfq")
    print(f"完成! 共 {len(results)} 条")

if __name__ == "__main__":
    main()
