#!/usr/bin/env python3
"""
crawl_linqing.py - 临清市人民政府-镇街动态
临清市政府门户网站 - 组件化CMS系统
"""
import requests
import re
import sqlite3
import sys
import time

BASE_URL = "http://www.linqing.gov.cn"
LIST_URL = BASE_URL + "/channel_t_269_15190/"
CDN_IP = "120.224.28.134"
DB_PATH = "/root/search.db"
SITE_NAME = "临清市人民政府-镇街动态"
SLEEP = 0.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)


def get_page(url, params=None):
    """Fetch via CDN IP"""
    real_url = url.replace("www.linqing.gov.cn", CDN_IP)
    headers = {"Host": "www.linqing.gov.cn"}
    try:
        resp = session.get(real_url, headers=headers, params=params, timeout=30)
        resp.encoding = "utf-8"
        return resp
    except Exception as e:
        print(f"  [NET ERR] {e}")
        return None


def parse_list(html):
    """Parse article list"""
    items = []
    for m in re.finditer(
        r'<a[^>]*title="([^"]*)"[^>]*href="([^"]*doc_[a-f0-9]+\.html)"[^>]*>',
        html,
    ):
        title = m.group(1).strip()
        href = m.group(2)

        # Find date
        ctx = html[m.start() : m.start() + 500]
        date_m = re.search(r"发布时间[：:](\d{4}-\d{2}-\d{2})", ctx)
        date_str = date_m.group(1) if date_m else ""

        # Make absolute URL
        if href.startswith("/"):
            url = BASE_URL + href
        elif href.startswith("http"):
            if "linqing.gov.cn" not in href:
                continue
            url = href
        else:
            url = BASE_URL + "/" + href

        items.append({"title": title, "url": url, "date": date_str})

    return items


def parse_detail(html):
    """Parse article detail"""
    result = {"title": "", "date": "", "content": ""}

    m_title = re.search(
        r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html
    )
    if m_title:
        result["title"] = m_title.group(1).strip()

    m_date = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m_date:
        raw = m_date.group(1).strip()
        dm = re.search(r"\d{4}-\d{2}-\d{2}", raw)
        if dm:
            result["date"] = dm.group()

    # Content from Details_content
    m_content = re.search(
        r'<div\s+class="Details_content"[^>]*>(.*?)</div>', html, re.DOTALL
    )
    if not m_content:
        return result

    content_html = m_content.group(1)
    parts = []

    pos = 0
    while pos < len(content_html):
        p_match = re.match(r"<p[^>]*>(.*?)</p>", content_html[pos:], re.DOTALL)
        if p_match:
            p_text = re.sub(r"<[^>]+>", "", p_match.group(1)).strip()
            p_text = re.sub(r"\s+", " ", p_text).strip()
            p_text = (
                p_text.replace("&nbsp;", " ")
                .replace("&mdash;", "—")
                .replace("&amp;", "&")
            )
            if p_text:
                parts.append(p_text)
            pos += p_match.end()
            continue

        img_match = re.match(
            r'<img[^>]*src="([^"]+)"[^>]*>', content_html[pos:], re.DOTALL
        )
        if img_match:
            src = img_match.group(1)
            if not src.startswith("http"):
                src = BASE_URL + src
            parts.append(f"![图片]({src})")
            pos += img_match.end()
            continue

        tag_match = re.match(r"<[^>]+>", content_html[pos:])
        if tag_match:
            pos += tag_match.end()
            continue

        ws_match = re.match(r"\s+", content_html[pos:])
        if ws_match:
            pos += ws_match.end()
            continue

        pos += 1

    result["content"] = "\n\n".join(parts).strip()
    return result


def save_to_db(url, title, date, content):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    try:
        c.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (url, SITE_NAME),
        )
        if c.fetchone():
            conn.close()
            return "skip"

        summary = content[:200]
        has_table = 1 if "| ---" in content else 0
        c.execute(
            """INSERT OR REPLACE INTO gov_raw (page_url, title, publish_date, site_name, content, attachments, summary, has_table, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_linqing.py')""",
            (url, title, date, SITE_NAME, content, "", summary, has_table),
        )
        conn.commit()
        conn.close()
        return "new"
    except Exception as e:
        conn.close()
        return f"error:{e}"


def crawl(pages=1):
    total_new = 0
    total_skip = 0

    print(f"[LIST] Fetching page...")
    resp = get_page(LIST_URL)
    if resp is None or resp.status_code != 200:
        print(f"[WARN] List page failed")
        return total_new, total_skip

    items = parse_list(resp.text)
    print(f"[LIST] {len(items)} items found")

    for item in items:
        if "linqing.gov.cn" not in item["url"]:
            continue

        resp2 = get_page(item["url"])
        if resp2 is None or resp2.status_code != 200:
            print(f"  [FAIL] HTTP ?: {item['title'][:50]}")
            total_skip += 1
            continue

        detail = parse_detail(resp2.text)
        content = detail.get("content", "")
        if not content:
            print(f"  [EMPTY] {item['title'][:50]}")
            total_skip += 1
            continue

        title = detail.get("title") or item["title"]
        date = detail.get("date") or item.get("date", "")

        result = save_to_db(item["url"], title, date, content)
        if result == "new":
            total_new += 1
            print(f"  [OK] {title[:60]}")
        elif result == "skip":
            total_skip += 1
            print(f"  [SKIP] {title[:50]}")
        else:
            total_skip += 1
            print(f"  [ERR] {title[:50]} - {result}")

        time.sleep(SLEEP)

    return total_new, total_skip


if __name__ == "__main__":
    import argparse

    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=1)
    args = parser.parse_args()

    new, skip = crawl(pages=args.pages)
    print(f"\n[RESULT] {SITE_NAME}: {new} new, {skip} skip")
