#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
吉安井冈山经开区 - 企业动态
站点: jkq.jian.gov.cn/news-list-qiyedongtai.html
CMS: 自建PHP + AJAX分页 (同 jkqjian_tzgg 公告公示模板)
列表: POST /api-ajax_list-{page}.html, ajax_type[]=70_news,22,70,news,Y-m-d,40,20,0,"" (20条/页)
详情: news-show-{id}.html (div.text-main 正文)
用法: python3 crawl_jkq_jian_qiyedongtai.py [--pages=N | N]
"""
import re, os, sys, time, json
import requests, warnings, sqlite3
warnings.filterwarnings('ignore')
from datetime import datetime, timedelta

_MAX_PG = None
for _a in sys.argv[1:]:
    if _a.startswith("--pages="):
        try:
            _MAX_PG = int(_a.split("=", 1)[1])
        except Exception:
            pass
    elif _a.isdigit():
        _MAX_PG = int(_a)
if _MAX_PG is not None:
    print(f"[AutoPg] max_pages={_MAX_PG}")

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "吉安井冈山经开区-企业动态"
BASE_URL = "http://jkq.jian.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE_URL}/news-list-qiyedongtai.html",
    "Content-Type": "application/x-www-form-urlencoded",
    "X-Requested-With": "XMLHttpRequest",
}
API_LIST = f"{BASE_URL}/api-ajax_list-"
CATID = "22"  # 企业动态栏目

def fetch_list_page(page):
    data = [
        ('ajax_type[]', '70_news'), ('ajax_type[]', CATID), ('ajax_type[]', '70'),
        ('ajax_type[]', 'news'), ('ajax_type[]', 'Y-m-d'), ('ajax_type[]', '40'),
        ('ajax_type[]', '20'), ('ajax_type[]', '0'), ('ajax_type[]', ''),
        ('is_ds', '1'),
    ]
    try:
        r = requests.post(f"{API_LIST}{page}.html", headers=HEADERS, data=data, timeout=20, verify=False)
        if r.status_code != 200:
            print(f"  [WARN] {page} HTTP {r.status_code}")
            return []
        j = r.json()
        items = []
        for d in j.get("data", []):
            url = d.get("url", "")
            title = (d.get("title") or "").strip()
            date = (d.get("inputtime") or "").strip()
            desc = (d.get("description") or "").strip()
            if not url or not title:
                continue
            if not re.search(r'news-show-\d+\.html$', url):
                sid = (d.get("id") or "").strip()
                if sid:
                    url = f"{BASE_URL}/news-show-{sid}.html"
            items.append({"url": url, "title": title, "date": date, "desc": desc})
        return items
    except Exception as e:
        print(f"  [WARN] {page} err: {e}")
        return []

def fetch(url):
    try:
        r = requests.get(url, headers={"User-Agent": HEADERS["User-Agent"], "Referer": f"{BASE_URL}/news-list-qiyedongtai.html"}, timeout=20, verify=False)
        r.encoding = "utf-8"
        if r.status_code == 200:
            return r.text
    except Exception:
        pass
    return ""

def extract_detail(html_text, page_url):
    """div.show_content / div.text-main / 微信 rich_media_content 平衡匹配, 附件/图片 urljoin 绝对化, 保留 <p>"""
    i = html_text.find('class="show_content"')
    if i < 0:
        i = html_text.find('class="text-main"')
    if i < 0:
        i = html_text.find('class="text_main"')
    if i < 0:
        i = html_text.find('class="rich_media_content')
    if i < 0:
        return ""
    ds = html_text.rfind('<div', 0, i)
    if ds < 0:
        return ""
    s = html_text[ds:]
    d = 0
    for j in range(len(s)):
        if s[j:j+4] == "<div" and (j+4 >= len(s) or s[j+4] in " >\n\r\t"):
            d += 1
        elif s[j:j+6] == "</div>":
            d -= 1
            if d == 0:
                gt = s.find(">", 0, j)
                c = s[gt+1:j] if gt > 0 else ""
                for m in re.finditer(r'(?:href|src)="([^"]+)"', c):
                    h = m.group(1)
                    if h.startswith("/"):
                        c = c.replace(h, BASE_URL + h)
                for m in re.finditer(r'<iframe[^>]*src="([^"]+\.pdf)"[^>]*>', c, re.I):
                    h = m.group(1)
                    if h.startswith("/"):
                        h = BASE_URL + h
                    c = c.replace(m.group(0), f'<p><a href="{h}">PDF原文</a></p>')
                c = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
                c = re.sub(r'<p[^>]*>', '<p>', c)
                text = re.sub(r'<[^>]+>', '', c).strip()
                if text or '<a href=' in c or '<img' in c:
                    return c.strip()
    return ""

def parse_detail(url):
    html_text = fetch(url)
    if not html_text:
        return None
    title = ""
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html_text)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<h1[^>]*>(.*?)</h1>', html_text, re.DOTALL)
        if m:
            title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html_text, re.DOTALL)
        if m:
            title = re.sub(r'[-_—]\s*.*$', '', m.group(1).strip())
    pub_date = ""
    m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html_text)
    if m:
        pub_date = m.group(1)[:10]
    if not pub_date:
        m = re.search(r'20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}', html_text[:3000])
        if m:
            pub_date = re.sub(r'[年月/]', '-', m.group(0))[:10]
    body = extract_detail(html_text, url)
    if not body:
        return {"title": title, "content": "", "pub_date": pub_date}
    text_len = len(re.sub(r'<[^>]+>', '', body).strip())
    has_img = '<img' in body
    if text_len < 5 and not has_img:
        return {"title": title, "content": "", "pub_date": pub_date}
    return {"title": title, "content": body, "pub_date": pub_date}

def _store(detail, it):
    row = {
        "site_name": SITE_NAME,
        "source_url": it["url"],
        "page_url": it["url"],
        "title": detail["title"] or it["title"],
        "publish_date": detail["pub_date"] or it["date"],
        "content": detail["content"],
        "summary": (detail["content"] or "")[:500],
        "category": "环境保护",
        "script_name": os.path.basename(__file__),
    }
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    try:
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name) VALUES (?,?,?,?,?,?,?,?,?)",
            (row["site_name"], row["source_url"], row["page_url"], row["title"], row["publish_date"], row["content"], row["summary"], row["category"], row["script_name"]))
        conn.commit()
    except Exception:
        pass
    conn.close()

def run(max_pages=5):
    print(f"\n🚀 {SITE_NAME}")
    all_items = []
    for pg in range(1, min(max_pages, _MAX_PG or max_pages) + 1):
        its = fetch_list_page(pg)
        if not its:
            print(f"  第{pg}页: 空 -> 结束")
            break
        dates = [it["date"] for it in its if it["date"]]
        print(f"  第{pg}页: {len(its)} 条 ({min(dates) if dates else '?'} ~ {max(dates) if dates else '?'})")
        all_items.extend(its)
        if dates and max(dates) < CUTOFF:
            print(f"  该页已全部早于截断日{CUTOFF}，停止翻页")
            break
        time.sleep(0.3)
    print(f"📊 列表共 {len(all_items)} 条")
    kept = [it for it in all_items if not it["date"] or it["date"] >= CUTOFF]
    print(f"  窗口内(>= {CUTOFF}): {len(kept)} 条")
    stored = skipped = 0
    for it in kept:
        detail = parse_detail(it["url"])
        if not detail:
            continue
        if not detail.get("content"):
            detail = {"title": it["title"], "publish_date": it["date"], "content": "", "pub_date": it["date"]}
        detail["title"] = detail["title"] or it["title"]
        detail["publish_date"] = detail.get("publish_date") or it["date"]
        _store(detail, it)
        if detail["content"]:
            stored += 1
        else:
            skipped += 1
        time.sleep(0.15)
    print(f"✅ Result: {stored} stored, {skipped} skipped")

if __name__ == "__main__":
    run(max_pages=_MAX_PG or 5)
