#!/usr/bin/env python3
"""Crawl tl.gov.cn - use IDs from JSON to construct URLs"""
import sys, re, urllib.request, traceback
from datetime import datetime

BASE = "https://sthjj.tl.gov.cn"
SITE_NAME = "tlepb_hp"
DB_PATH = "/root/search.db"

def fetch(url):
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
    resp = urllib.request.urlopen(req, timeout=30)
    return resp.read().decode("utf-8", errors="replace")

def extract_articles(html):
    # Find listData area
    idx = html.find("listData = {")
    if idx < 0: return []
    al_idx = html.find("articleList:", idx)
    if al_idx < 0: return []
    bracket = html.find("[", al_idx)
    if bracket < 0: return []

    chunk = html[bracket:bracket+50000]

    # Split by title to count and extract
    titles = chunk.split('"title":"')
    ids = chunk.split('"id":"')
    dates = chunk.split('"pubDate":"')

    articles = []
    n = min(len(titles)-1, len(ids)-1, len(dates)-1)
    for i in range(1, n+1):
        title = titles[i].split('"')[0]
        id_val = ids[i].split('"')[0]
        date = dates[i].split('"')[0]
        if title and id_val and len(title) > 5:
            url = f"{BASE}/tlssthjj/hpgzcygs/pc/content/content_{id_val}.html"
            articles.append({"title": title, "pubdate": date, "url": url})

    return articles


def extract_clean_text(html_chunk):
    """Extract text from HTML chunk, preserving paragraph breaks."""
    text = re.sub(r"</p>", "\n\n", html_chunk, flags=re.I)
    text = re.sub(r"<br\s*/?>", "\n", text, flags=re.I)
    text = re.sub(r"</div>", "\n\n", text, flags=re.I)
    text = re.sub(r"</?(?:b|span|font|strong|em|u|i|a)\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"<p\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"\n{3,}", "\n\n", text)
    lines = [l.strip() for l in text.split("\n")]
    text = "\n".join(lines)
    text = re.sub(r"\n{3,}", "\n\n", text)
    text = text.strip()
    return text


def get_content(html):
    title = ""
    tm = re.search(r"<title>([^<]+)</title>", html)
    if tm:
        title = tm.group(1).split(" - ")[0].strip()

    date = ""
    dm = re.search(r"(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}", html)
    if dm:
        date = dm.group(1)

    # Extract from article-content div (primary content area)
    content = ""
    idx = html.find('class="article-content"')
    if idx < 0:
        idx = html.find('id="zoomcon"')
    if idx > 0:
        start = html.rfind("<div", 0, idx)
        if start < 0:
            start = idx
        pos = html.find(">", idx) + 1

        depth = 1
        end = pos
        while depth > 0 and end < len(html):
            no = html.find("<div", end)
            nc = html.find("</div>", end)
            if nc < 0:
                break
            if no >= 0 and no < nc:
                depth += 1
                end = html.find(">", no) + 1
            else:
                depth -= 1
                end = nc + 6

        chunk = html[pos:end-6] if depth == 0 else html[pos:]
        content = extract_clean_text(chunk)

    return title, date, content


def save(records):
    if not records:
        return 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()
    count = 0
    for r in records:
        if c.execute("SELECT id FROM gov_raw WHERE source_url=?", (r["url"],)).fetchone():
            print(f"  - {r['title'][:30]}")
            continue
        content = (r["content"] or "")[:5000]
        c.execute("""INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, content)
            VALUES (?,?,?,?,?,?,?,?)""",
            (SITE_NAME, r["url"], r["url"], r["title"], r["date"],
             int(datetime.strptime(r["date"], "%Y-%m-%d").timestamp()) if r["date"] else 0,
             content[:200], content))
        conn.commit()
        count += 1
        print(f"  + {r['title'][:40]} | {r['date']}")
    conn.close()
    return count


def main():
    url = BASE + "/tlssthjj/hpgzcygs/pc/list.html"
    html = fetch(url)
    articles = extract_articles(html)
    print(f"Page 1: {len(articles)} articles")

    if len(articles) < 41:
        html2 = fetch(url + "?page=2")
        articles2 = extract_articles(html2)
        print(f"Page 2: {len(articles2)} articles")
        articles.extend(articles2)

    print(f"Total: {len(articles)}")

    records = []
    for i, a in enumerate(articles):
        sys.stdout.write(f"[{i+1}/{len(articles)}] {a['title'][:30]}... ")
        sys.stdout.flush()
        try:
            html = fetch(a["url"])
            title, date, content = get_content(html)
            records.append({"title": title or a["title"], "url": a["url"],
                            "date": date or a["pubdate"][:10], "content": content})
            print("OK")
        except Exception as e:
            print(f"ERR: {e}")

    saved = save(records)
    print(f"\nSaved: {saved}/{len(records)}")

if __name__ == "__main__":
    main()
