#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""临川区人民政府 (www.jxlc.gov.cn) 多栏目爬虫 — 大汉 jcms dataproxy
栏目:
  tzgg  col1384 通知公告   (webid=5, columnid=1384, unitid=70893, 记录格式 href='..'title='..'+span)
  jjkfq col1611 经济开发区 (webid=5, columnid=1611, unitid=46632, 记录格式 href=\"..\"title=\"..\"+b, 页面内嵌61条)
用法: python3 crawl_jxlc_tzgg.py [--col tzgg|jjkfq] [--pages N] [--db /path] [--host IP]
加速乐防护: 需 --resolve 强制 IP + Referer + X-Requested-With, 全部走 POST
"""
import argparse, json, os, re, sys, time, random
from urllib.request import Request, urlopen
from urllib.parse import urljoin

UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
BASE = 'https://www.jxlc.gov.cn'
COL_URL = BASE + '/col/col1384/index.html'
DEFAULT_HOST = '218.98.26.222'
HOST = DEFAULT_HOST

COLS = {
    'tzgg': {
        'columnid': '1384', 'unitid': '70893',
        'list_url': BASE + '/col/col1384/index.html',
        'site_name': 'jxlc_tzgg',
        'record_fmt': 'single_quote',  # href='...'title='...'+span
        'pages': 2,
    },
    'jjkfq': {
        'columnid': '1611', 'unitid': '46632',
        'list_url': BASE + '/col/col1611/index.html?number=D00002D00004',
        'site_name': 'jxlc_jjkfq',
        'record_fmt': 'double_quote',  # href="..."title="..."+b
        'pages': 1,
    },
}

sys.path.insert(0, '/root/gov_crawler')
try:
    from crawler_lib import push_to_searchdb
except ImportError:
    push_to_searchdb = None

def fetch(url, referer=COL_URL, post=None, timeout=25):
    """强制IP直连 + Referer, POST 或 GET"""
    import ssl
    ctx = ssl.create_default_context()
    ctx.check_hostname = False
    ctx.verify_mode = ssl.CERT_NONE
    headers = {
        'User-Agent': UA,
        'Referer': referer,
        'Accept': '*/*',
    }
    if post is not None:
        headers['X-Requested-With'] = 'XMLHttpRequest'
        headers['Content-Type'] = 'application/x-www-form-urlencoded'
        req = Request(url, data=post.encode(), headers=headers)
    else:
        req = Request(url, headers=headers)
    opener = urlopen(req, context=ctx, timeout=timeout)
    # 强制解析到指定 IP (需要 --resolve 等价物, 用 Host header 直连 IP)
    return opener.read().decode('utf-8', errors='ignore')

def fetch_ip(url, referer=COL_URL, post=None, timeout=25):
    """直接连 IP + Host header (--resolve 等价)"""
    import ssl, socket
    from urllib.parse import urlparse
    ctx = ssl.create_default_context()
    ctx.check_hostname = False
    ctx.verify_mode = ssl.CERT_NONE
    p = urlparse(url)
    headers = {
        'User-Agent': UA,
        'Referer': referer,
        'Host': p.netloc,
        'Accept': '*/*',
    }
    if post is not None:
        headers['X-Requested-With'] = 'XMLHttpRequest'
        headers['Content-Type'] = 'application/x-www-form-urlencoded'
        req = Request(url, data=post.encode(), headers=headers)
    else:
        req = Request(url, headers=headers)
    # 手动 socket 到 IP
    import http.client
    conn = http.client.HTTPSConnection(HOST, 443, context=ctx, timeout=timeout)
    path = p.path + ('?' + p.query if p.query else '')
    method = 'POST' if post is not None else 'GET'
    body = post.encode() if post is not None else None
    conn.request(method, path, body=body, headers=headers)
    resp = conn.getresponse()
    data = resp.read().decode('utf-8', errors='ignore')
    conn.close()
    return data

def clean_html(html):
    """白名单清洗: 保留 p/table/a, 去 inline style, unwrap span, 链接绝对化"""
    from bs4 import BeautifulSoup
    # 清内容标记注释 (大汉 jcms: 信息内容/ContentStart/ZJEG_RSS)
    html = re.sub(r'<!--<\$\[[^\]]*\]>begin-->', '', html)
    html = re.sub(r'<!--<\$\[[^\]]*\]>end-->', '', html)
    html = re.sub(r'<!--ZJEG_RSS\.content\.(begin|end)-->', '', html)
    html = re.sub(r'<meta name="Content(Start|End)"/>?', '', html)
    soup = BeautifulSoup(html, 'html.parser')
    for t in soup(['script', 'style', 'iframe', 'object']):
        t.decompose()
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(BASE, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label']):
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(BASE, href)
        elif href and href.startswith('javascript'):
            a.decompose()
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('img'):
            p.decompose()
    return str(soup)

def parse_list(html, record_fmt='single_quote'):
    """解析 dataproxy 响应 → [(title, url, date)]
    record_fmt: single_quote=col1384格式(href='..'title='..'+span), double_quote=col1611格式(href=".."title=".."+b)
    """
    items = []
    if record_fmt == 'single_quote':
        # 用 record 块配对日期
        recs = re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', html, re.S)
        for rec in recs:
            m = re.search(r"href='([^']+)'title='([^']*)'", rec)
            if not m or '/art/' not in m.group(1):
                continue
            sp = re.search(r'<span[^>]*>([^<]*)</span>', rec)
            date = sp.group(1).strip() if sp else ''
            url = m.group(1)
            if not url.startswith('http'):
                url = urljoin(BASE, url)
            items.append({'title': m.group(2).strip(), 'url': url, 'date': date})
    else:
        # double_quote: href="..."title="..." + <b>日期</b>
        recs = re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', html, re.S)
        for rec in recs:
            m = re.search(r'href="([^"]+)"[^>]*title="([^"]*)"', rec)
            if not m or '/art/' not in m.group(1):
                continue
            b = re.search(r'<b>\s*([^<]+?)\s*</b>', rec)
            date = b.group(1).strip() if b else ''
            url = m.group(1)
            if not url.startswith('http'):
                url = urljoin(BASE, url)
            items.append({'title': m.group(2).strip(), 'url': url, 'date': date})
    return items

def parse_detail(html):
    """解析详情页 → (title, publish_date, body_html)"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    # 标题: h1 或 meta
    h1 = soup.find('h1')
    title = h1.get_text(strip=True) if h1 else ''
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.S)
        if m:
            t = m.group(1).strip()
            # 去掉站点前缀
            title = t.split(' 通知公告 ')[-1].split(' 临川区人民政府')[-1]
    # 日期: meta PubDate
    m = re.search(r'<meta name="PubDate" content="([^"]+)"', html)
    pub = m.group(1).strip()[:10] if m else ''
    # 正文: div#zoom
    zoom = soup.find(id='zoom')
    if zoom:
        # 移除尾部来源行
        for div in zoom.find_all('div', style=lambda v: v and 'float:right' in v):
            div.decompose()
        body = clean_html(str(zoom))
    else:
        body = ''
    return title, pub, body

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--col', default='tzgg', choices=list(COLS.keys()))
    ap.add_argument('--pages', type=int, default=None)
    ap.add_argument('--db', default=None)
    ap.add_argument('--host', default=DEFAULT_HOST)
    ap.add_argument('--out', default=None)
    args = ap.parse_args()
    fetch_ip.__globals__['HOST'] = args.host
    col = COLS[args.col]
    pages = args.pages or col['pages']
    list_url = col['list_url']
    site_name = col['site_name']
    record_fmt = col['record_fmt']
    post_params = f'page={{pg}}&appid=1&webid=5&path=/&columnid={col["columnid"]}&unitid={col["unitid"]}&permissiontype=0'

    # 抓列表 (jjkfq: 页面内嵌 recordset; tzgg: POST dataproxy)
    all_items = []
    seen = set()
    if record_fmt == 'double_quote':
        # 直接抓列表页 HTML, 从内嵌 recordset 解析
        try:
            html = fetch_ip(list_url)
        except Exception as e:
            print(f'列表页: 失败 {str(e)[:100]}', flush=True)
            html = ''
        items = parse_list(html, record_fmt)
        for it in items:
            if it['url'] not in seen:
                seen.add(it['url'])
                all_items.append(it)
        print(f'页面内嵌: {len(items)} 条', flush=True)
    else:
        for pg in range(1, pages + 1):
            post = post_params.format(pg=pg)
            try:
                html = fetch_ip(f'{BASE}/module/web/jpage/dataproxy.jsp', post=post)
            except Exception as e:
                print(f'页{pg}: 失败 {str(e)[:100]}', flush=True)
                break
            items = parse_list(html, record_fmt)
            new = [it for it in items if it['url'] not in seen]
            for it in new:
                seen.add(it['url'])
                all_items.append(it)
            print(f'页{pg}: {len(items)} 条, 累计 {len(all_items)}', flush=True)
            if len(items) == 0:
                break
            time.sleep(random.uniform(0.8, 1.5))
    print(f'共 {len(all_items)} 条', flush=True)

    # 预去重: 已入库的 URL 不再抓详情 (避免每天重抓全量 → 超时)
    try:
        import sqlite3 as _sq
        _db = os.getenv('SEARCH_DB', '/root/search.db')
        _c = _sq.connect(_db, timeout=60)
        _have = set(r[0] for r in _c.execute(
            'SELECT page_url FROM gov_raw WHERE site_name=?', (site_name,)))
        _c.close()
        _before = len(all_items)
        all_items = [it for it in all_items if it['url'] not in _have]
        print(f'预去重: {_before} → {len(all_items)} (库中已有 {len(_have)})', flush=True)
    except Exception as e:
        print(f'预去重跳过: {str(e)[:80]}', flush=True)

    # 抓详情
    rows = []
    fail = 0
    for i, it in enumerate(all_items):
        try:
            html = fetch_ip(it['url'], referer=list_url)
        except Exception as e:
            fail += 1
            print(f'  详情{i}: 失败 {str(e)[:80]}', flush=True)
            continue
        title, pub, body = parse_detail(html)
        if not title or not body:
            fail += 1
            print(f'  详情{i}: 空 title={title[:20]!r} body={len(body)}', flush=True)
            continue
        # 清理标题前缀 (jjkfq 栏目页面标题带站点前缀)
        for pre in ['临川区人民政府 临川经济开发区 ', '临川区人民政府 信息公开制度 ', '临川区人民政府 ']:
            if title.startswith(pre):
                title = title[len(pre):]
                break
        rows.append({
            'title': title,
            'url': it['url'],
            'source_url': it['url'],
            'pub_date': pub or it['date'],
            'publish_date': pub or it['date'],
            'content': body,
            'author': '',
            'site_name': site_name,
            'attachments': '',
        })
        if (i + 1) % 20 == 0:
            print(f'详情 {i+1}/{len(all_items)}: ok={len(rows)} fail={fail}', flush=True)
        time.sleep(random.uniform(0.5, 1.0))

    print(f'完成: ok={len(rows)} fail={fail}', flush=True)
    if args.out:
        with open(args.out, 'w', encoding='utf-8') as f:
            for r in rows:
                f.write(json.dumps(r, ensure_ascii=False) + '\n')
        print(f'JSONL 已写: {args.out}', flush=True)
    if push_to_searchdb:
        ok = push_to_searchdb(rows)
        print(f'DB 导入: {ok}', flush=True)
    else:
        print('警告: push_to_searchdb 不可用, 只写 JSONL', flush=True)

if __name__ == '__main__':
    main()
