#!/usr/bin/env python3
"""
临湘市 - 公示公告 (linxiang.gov.cn)
列表: /24733/24736/24738/default.htm -> default_1.htm 分页
详情: /24733/24736/24738/content_2381380.html
正文: div.content-wrapper  (GB2312编码)
"""
import sys, os, re, time, json
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, clean_html
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "临湘市-公示公告"
BASE_URL = "https://www.linxiang.gov.cn"
LIST_URL = "https://www.linxiang.gov.cn/24733/24736/24738/default.htm"
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (compatible; Googlebot/2.1)"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = r.apparent_encoding or 'gb2312'
        return r.text
    except: return None

def parse_list(html):
    items = []
    for m in re.finditer(r'<a[^>]*href="((?:/24733/24736/24738/)?content_\d+\.html)"[^>]*>(.*?)</a>', html, re.I|re.S):
        href = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        if title and len(title) > 5:
            if href.startswith('/'): href = BASE_URL + href
            elif not href.startswith('http'): href = BASE_URL + '/24733/24736/24738/' + href
            items.append((title, href))
    return items

def get_page_url(page):
    if page == 1: return LIST_URL
    if page == 2: return f"{BASE_URL}/24733/24736/24738/default_1.htm"
    return f"https://www.linxiang.gov.cn/24733/24736/24738/default_{page-1}.htm"

def fetch_detail(url):
    html = fetch(url)
    if not html: return "", "", "", []
    title = ""
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]+)"', html)
    if m: title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.I|re.S)
        if m: title = m.group(1).replace("-临湘市人民政府","").strip()
    pub_date = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)"', html)
    if m: pub_date = m.group(1)[:10]
    if not pub_date:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if not m: m = re.search(r'日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m: pub_date = m.group(1)
    content = ""
    attachments = []
    for sel in [r'<div[^>]*class="content-wrapper[^"]*"[^>]*>(.*?)</div>',
                 r'<div[^>]*class="[^"]*content-wrapper[^"]*"[^>]*>(.*?)</div>']:
        m = re.search(sel, html, re.I|re.S)
        if m and len(m.group(1)) > 50:
            content = m.group(1).strip()
            break
    if content:
        # Filter eWebEditor icon images
        content = re.sub(r'<img[^>]*sysimage/icon[^>]*>', '', content, flags=re.I|re.S)
        # Extract attachments (PDF/DOC files)
        for am in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]+)</a>', content, re.I):
            f_url = am.group(1)
            f_name = re.sub(r'<[^>]+>', '', am.group(2)).strip()
            if f_url.startswith('/'):
                f_url = BASE_URL + f_url
            attachments.append({"name": f_name or f_url.rsplit('/',1)[-1], "url": f_url, "ext": f_url.rsplit('.',1)[-1].lower()})
        content = clean_html(content)
        # Remove icon img before attachment links
        content = re.sub(r'<img[^>]*>\s*(<a[^>]*href="[^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar)"[^>]*>[^<]+</a>)', r'\1', content, flags=re.I)
    return title, pub_date, content, attachments

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] in ('1', '--incremental', '-i')
    print(f"爬取: {SITE_NAME}" + (" [增量]" if incremental else ""))
    results = []
    pages = 1 if incremental else MAX_PAGES
    for page in range(1, pages+1):
        url = get_page_url(page)
        html = fetch(url)
        if not html: break
        items = parse_list(html)
        if not items: break
        print(f"  第{page}页: {len(items)} 条")
        for i, (title, url) in enumerate(items):
            print(f"  [{i+1}/{len(items)}] {title[:40]}...", end=" ", flush=True)
            dt, date, content, attachments = fetch_detail(url)
            if date and date < CUTOFF: print("过旧"); continue
            if not content or len(content) < 50: print("无正文"); continue
            final_title = dt or title
            # Use title as summary fallback when content is image-only
            summary = re.sub(r'<[^>]+>', '', content)[:200].strip()
            if not summary:
                summary = final_title[:200]
            attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
            if attachments:
                print(f"附件:{len(attachments)}个", end=" ")
            results.append({"site_name": SITE_NAME, "title": final_title, "url": url,
                           "content": content, "summary": summary, "pub_date": date or "",
                           "attachments": attachments_json})
            print(f"OK ({len(content)}字)")
            time.sleep(0.3)
    if results:
        push_to_searchdb(results, "linxiang")
    print(f"完成! 共 {len(results)} 条")

if __name__ == "__main__":
    main()
