#!/usr/bin/env python3
"""
安吉县-环境影响评价 独立爬虫 (JPAAS系统)
站点: www.anji.gov.cn
栏目: 环境影响评价
"""

import requests, re, sqlite3, time, sys, json, warnings
warnings.filterwarnings('ignore')
from datetime import datetime, timedelta
import os
import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "安吉县-环境影响评价"
DOMAIN = "www.anji.gov.cn"

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

API_URL = "https://www.anji.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "3628",
    "tplSetId": "z2V6NtY1ABvxkwIRPBqVp",
    "pageType": "column",
    "tagId": "栏目一list",
    "editType": "null",
    "pageId": "8UHBe2bEEuEubNnILsQ82",
}

API_PARAMS_FALLBACK = dict(API_PARAMS)


def fetch_list(page, page_size=15):
    """获取API列表页"""
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": page_size})
    for retry in range(3):
        try:
            r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                d = r.json()
                html = d.get("data", {}).get("html", "")
                if not html or len(html) < 50:
                    # Fallback: try with shorter pageId
                    pf = dict(API_PARAMS_FALLBACK)
                    pf["paramJson"] = json.dumps({"pageNo": page, "pageSize": page_size})
                    rf = requests.get(API_URL, params=pf, headers=HEADERS, timeout=30)
                    if rf.status_code == 200:
                        df = rf.json()
                        html = df.get("data", {}).get("html", "")
                    else:
                        return [], 0
                # Parse items
                items = []
                for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
                    m = re.search(r'title="([^"]*)"[^>]*href="([^"]+)"', li)
                    if m:
                        span = re.search(r'<span class="time"[^>]*>([^<]*)</span>', li)
                        date = span.group(1).strip() if span else ''
                        items.append({
                            "url": "https://" + DOMAIN + m.group(2),
                            "title": m.group(1),
                            "date": date,
                        })
                count_m = re.search(r'count[=:]\s*["\']?(\d+)', html)
                total_count = int(count_m.group(1)) if count_m else 0
                return items, total_count
        except Exception as e:
            if retry < 2:
                time.sleep(2)
    return [], 0


def extract_content(html):
    """提取正文 - 只保留 <div class="w100 text"> 内的实际文章内容"""
    # Step 1: Find the article body container
    # Look for: class="w100 text" div which contains the actual article
    pats = ['class="w100 text"', 'class=\'w100 text\'', 'w100 text']
    text_div_start = -1
    for pat in pats:
        text_div_start = html.find(pat)
        if text_div_start >= 0:
            break
    
    if text_div_start < 0:
        # Fallback: try barrierfree container
        for pat, tag in [
            ("id='barrierfree_container'", 'div'),
            ('id="barrierfree_container', 'div'),
            ('id="zoom"', 'div'),
            ('class="nr_content"', 'div'),
        ]:
            i = html.find(pat)
            if i < 0:
                continue
            ds = html.rfind(f"<{tag}", 0, i)
            if ds < 0:
                continue
            s = html[ds:]
            d = 0
            for j in range(len(s)):
                if s[j:j+4] == f"<{tag}" and (j+4 >= len(s) or s[j+4] in " >\n\r\t"):
                    d += 1
                elif s[j:j+3+len(tag)] == f"</{tag}>":
                    d -= 1
                    if d == 0:
                        gt = s.find(">", 0, j)
                        c = s[gt+1:j] if gt > 0 else s[7:j]
                        c = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
                        text = re.sub(r'<[^>]+>', '', c).strip()
                        if len(text) > 50:
                            return c.strip()
        return ""
    
    # Find the opening <div that contains this class
    ds = html.rfind("<div", 0, text_div_start)
    if ds < 0:
        return ""
    
    # Extract the w100 text div content using depth counting
    s = html[ds:]
    d = 0
    for j in range(len(s)):
        if s[j:j+4] == "<div" and (j+4 >= len(s) or s[j+4] in " >\n\r\t"):
            d += 1
        elif s[j:j+6] == "</div>":
            d -= 1
            if d == 0:
                gt = s.find(">", 0, j)
                c = s[gt+1:j] if gt > 0 else s[7:j]
                
                # Clean up
                c = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
                
                # Remove mso-* and other MS Office cruft from style attributes
                c = re.sub(r'style="[^"]*mso-[^"]*"', '', c, flags=re.I)
                # Clean up empty/unnecessary span styles that are just MS Word artifacts
                c = re.sub(r'<span[^>]*mso-spacerun[^>]*>\s*</span>', '', c, flags=re.DOTALL|re.I)
                
                # Remove hardcoded width=1200px, width=1000px, width=888px etc.
                c = re.sub(r'width\s*[:=]\s*["\']?\d+px["\']?', 'max-width:100%', c, flags=re.I)
                
                # Remove the "关闭窗口/返回顶部/打印文章" footer
                c = re.sub(r'<div[^>]*main_5124[^>]*>.*?</div>', '', c, flags=re.DOTALL)
                c = re.sub(r'<div[^>]*height:\s*8px[^>]*>.*?</div>', '', c, flags=re.DOTALL)
                
                text = re.sub(r'<[^>]+>', '', c).strip()
                if len(text) > 50:
                    return c.strip()
    return ""


def extract_meta(html):
    title = ""
    tm = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if tm:
        title = tm.group(1)
    if not title:
        ttm = re.search(r'<title>(.*?)<', html)
        if ttm:
            title = ttm.group(1).strip()
    date = ""
    for pm in re.finditer(r'PubDate[^>]*content="([^"]*)"', html):
        d = pm.group(1).strip()
        cm = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', d)
        if cm:
            date = f"{cm.group(1)}-{cm.group(2).zfill(2)}-{cm.group(3).zfill(2)}"
            break
    if not date:
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        if dm:
            date = dm.group(1)
    return title, date


def fetch_url(url):
    for retry in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r.text
        except:
            if retry < 2:
                time.sleep(2)
    return None


def insert_article(url, title, date, content):
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    try:
        db.execute(
            "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,status,category) VALUES(?,?,?,?,?,?,?,?,?)",
            (SITE_NAME, url, url, title, date, content, title[:200], "active", "环境影响评价"))
        affected = db.total_changes
        db.commit()
        if affected:
            db.execute(
                "INSERT INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.page_url=? AND r.id NOT IN (SELECT rowid FROM gov_search)",
                (url,))
            db.commit()
        db.close()
        return affected
    except Exception as e:
        db.close()
        print(f"  DB error: {e}")
        return 0


def run(max_pages=50):
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    items, total = fetch_list(1)
    if not items:
        print("⚠  API无返回，尝试fallback参数...")
        items, total = fetch_list(1)
        if not items:
            print("❌ API仍然无返回")
            return
    print(f"  共{total}条")
    total_pages = min(max_pages, (total + 14) // 15)

    all_items = list(items)
    for pg in range(2, min(total_pages, _MAX_PG or total_pages) + 1):
        its, _ = fetch_list(pg)
        if not its:
            print(f"  第{pg}页: 空 -> 结束")
            break
        all_items.extend(its)
        dates = [it["date"] for it in its if it["date"]]
        if dates:
            print(f"  第{pg}页: {len(its)} 条 ({min(dates)} ~ {max(dates)})")
            if max(dates) < CUTOFF:
                print(f"  全部早于{CUTOFF}，停止翻页")
                break
        else:
            print(f"  第{pg}页: {len(its)} 条")
        time.sleep(0.15)

    print(f"\n📊 API共 {len(all_items)} 条")
    new = skip = no_body = err = 0
    for i, item in enumerate(all_items, 1):
        if item["date"] and item["date"] < CUTOFF:
            skip += 1
            continue

        html = fetch_url(item["url"])
        if not html:
            err += 1
            continue

        pt, pd = extract_meta(html)
        real_title = pt if pt else item["title"]
        real_date = pd if pd else item["date"]

        if real_date and real_date < CUTOFF:
            skip += 1
            continue

        content = extract_content(html)
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 10:
            no_body += 1
            continue

        if insert_article(item["url"], real_title, real_date, content):
            new += 1
        else:
            skip += 1

        if i % 5 == 0:
            print(f"  ...{i}/{len(all_items)}")
        time.sleep(0.15)

    print(f"\n📊 FTS同步...")
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
    db.execute("INSERT INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    db.commit()
    c = db.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,))
    cnt = c.fetchone()[0]
    db.close()
    print(f"  FTS: {cnt} 条")
    print(f"\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 空正文{no_body}, 错误{err}, FTS共{cnt}条")


if __name__ == "__main__":
    mp = int(sys.argv[1]) if len(sys.argv) > 1 else 50
    run(mp)
