#!/usr/bin/env python3
"""
珠海市-决策预公开 (www.zhuhai.gov.cn)
列表: /zw/wgk/jcgk/jcygk/index.html -> index_N.html 分页
详情: /zw/wgk/jcgk/jcygk/content/post_id.html
正文: div.nr (UTF-8编码)
CMS: ZZZCMS/NFCMS
"""
import sys, os, re, time, json
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, clean_html
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "珠海市-决策预公开"
BASE_URL = "https://www.zhuhai.gov.cn"
LIST_DIR = "/zw/wgk/jcgk/jcygk"
LIST_URL = f"{BASE_URL}{LIST_DIR}/index.html"
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {
    "User-Agent": "Mozilla/5.0 (compatible; Googlebot/2.1)",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  请求失败: {e}")
        return None

def parse_list(html):
    items = []
    for m in re.finditer(
        r'<a[^>]*href="(https://www\.zhuhai\.gov\.cn/zw/wgk/jcgk/jcygk/content/post_\d+\.html)"[^>]*>(.*?)</a>',
        html, re.I|re.S
    ):
        href = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        if title and len(title) > 5:
            items.append((title, href))
    return items

def get_page_url(page):
    if page == 1:
        return LIST_URL
    return f"{BASE_URL}{LIST_DIR}/index_{page}.html"

def fetch_detail(url):
    html = fetch(url)
    if not html:
        return "", "", "", ""
    # 标题
    title = ""
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]+)"', html)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<h4[^>]*>(.*?)</h4>', html, re.I|re.S)
        if m:
            title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    # 日期
    pub_date = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)"', html)
    if m:
        pub_date = m.group(1)[:10]
    # 正文
    content = ""
    m = re.search(r'<div[^>]*class="nr"[^>]*>(.*?)</div>\s*<(?:div|!--)', html, re.I|re.S)
    if m and len(m.group(1)) > 50:
        content = m.group(1).strip()
    if not content:
        m = re.search(r'<div[^>]*class="nr"[^>]*>(.*?)</div>', html, re.I|re.S)
        if m and len(m.group(1)) > 50:
            content = m.group(1).strip()
    if content:
        # Extract attachment links from full HTML first
        attachments = []
        for fam in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*class="doc"[^>]*>([^<]+)', html, re.I):
            f_url = fam.group(1)
            f_name = fam.group(2).strip()
            if not f_url.startswith('http'):
                f_url = BASE_URL + f_url
            attachments.append({"name": f_name, "url": f_url, "ext": f_url.rsplit('.',1)[-1].lower()})
        attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

        # Strip attachment section from raw content BEFORE clean_html
        # The attachment area starts with: <div><br><br><script>document.write("附件...
        # Try to truncate at the attachment boundary
        attach_idx = content.find('class="m-attachment"')
        if attach_idx < 0:
            attach_idx = content.find('document.write')
        if attach_idx >= 0:
            content = content[:attach_idx]
        # Also try to remove any trailing <div>...</div> with br/whitespace
        content = re.sub(r'<div>\s*(?:<br>\s*)+</div>\s*$', '', content, flags=re.I|re.S)
        # Also strip any remaining f-APPENDIX fragments
        content = re.sub(r'<div[^>]*class="[^"]*f-APPENDIX[^"]*"[^>]*>.*?</div>', '', content, flags=re.I|re.S)
        content = re.sub(r'<p[^>]*class="fujian2"[^>]*>.*?</p>', '', content, flags=re.I|re.S)
        content = re.sub(r'<script>.*?</script>', '', content, flags=re.I|re.S)
        content = clean_html(content)
        # Remove any trailing content after last </p>
        last_p = content.rfind('</p>')
        if last_p >= 0:
            content = content[:last_p+4]
    return title, pub_date, content, attachments_json

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] in ('1', '--incremental', '-i')
    print(f"爬取: {SITE_NAME}" + (" [增量]" if incremental else ""))
    results = []
    pages = 1 if incremental else MAX_PAGES
    for page in range(1, pages+1):
        url = get_page_url(page)
        html = fetch(url)
        if not html:
            print(f"  第{page}页: 获取失败")
            break
        items = parse_list(html)
        if not items:
            print(f"  第{page}页: 无列表项")
            if page > 1:
                break
        print(f"  第{page}页: {len(items)} 条")
        for i, (title, url) in enumerate(items):
            print(f"  [{i+1}/{len(items)}] {title[:40]}...", end=" ", flush=True)
            dt, date, content, attachments_json = fetch_detail(url)
            if date and date < CUTOFF:
                print("过旧")
                continue
            if not content or len(content) < 50:
                print("无正文")
                continue
            final_title = dt or title
            summary = re.sub(r'<[^>]+>', '', content)[:200].strip()
            if not summary:
                summary = final_title[:200]
            if attachments_json:
                print(f"附件", end=" ")
            results.append({
                "site_name": SITE_NAME, "title": final_title, "url": url,
                "source_url": url,
                "content": content, "summary": summary, "pub_date": date or "",
                "attachments": attachments_json,
            })
            print(f"OK ({len(content)}字)")
            time.sleep(0.5)
    if results:
        push_to_searchdb(results, "zhuhai_jcygk")
    print(f"完成! 共 {len(results)} 条")

if __name__ == "__main__":
    main()
