#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫: 淮安区人民政府 - 公示公告
http://www.zghaq.gov.cn/col/829_518352/index.html
CMS: 淮安区政府网站系统
"""

import requests
from bs4 import BeautifulSoup
import sqlite3
import re
import sys
import os
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.zghaq.gov.cn"
LIST_URL = "http://www.zghaq.gov.cn/col/829_518352/index.html"
SITE_NAME = "淮安区人民政府"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def get_session():
    """Get a session with cookies to bypass WAF"""
    s = requests.Session()
    s.headers.update(HEADERS)
    # Get main page first to get cookies
    s.get(LIST_URL, timeout=15)
    return s


def get_list_urls(session, max_pages=5):
    """Get all article URLs from list pages"""
    urls = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = "http://www.zghaq.gov.cn/col/829_518352/index_{}.html".format(page)

        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print("  Page {}: HTTP {}".format(page, r.status_code), file=sys.stderr)
                continue
            soup = BeautifulSoup(r.text, "html.parser")
            items = soup.select("div.list-lb ul li")
            if not items:
                print("  Page {}: no items".format(page), file=sys.stderr)
                break
            count = 0
            for li in items:
                a = li.find("a")
                if a and a.get("href"):
                    href = a["href"]
                    if not href.startswith("http"):
                        href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
                    urls.append(href)
                    count += 1
            print("  Page {}: {} items".format(page, count), file=sys.stderr)
            # If fewer items than expected, this was the last page
            if len(items) < 3:
                break
        except Exception as e:
            print("  Page {} error: {}".format(page, e), file=sys.stderr)
            break
    return urls


def parse_detail(session, url):
    """Parse a detail page"""
    try:
        r = session.get(url, timeout=15)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")

        # Check for WAF block
        if "知道创宇" in r.text or "拦截" in r.text or soup.find("meta", {"name": "ArticleTitle"}) is None:
            # Try again with fresh session
            s2 = get_session()
            r = s2.get(url, timeout=15)
            r.encoding = "utf-8"
            soup = BeautifulSoup(r.text, "html.parser")
            if "知道创宇" in r.text or "拦截" in r.text:
                return None

        # Title - from nr-bt first, then meta
        title = ""
        nr_bt = soup.select_one(".nr-bt")
        if nr_bt:
            title = nr_bt.get_text(strip=True)
        if not title:
            meta = soup.find("meta", {"name": "ArticleTitle"})
            if meta:
                title = meta.get("content", "")

        # Date
        date = ""
        nr_time = soup.select_one(".nr-time")
        if nr_time:
            txt = nr_time.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
            if m:
                date = m.group(1)
        if not date:
            meta = soup.find("meta", {"name": "PubDate"})
            if meta:
                content = meta.get("content", "")
                m = re.search(r'(\d{4}-\d{2}-\d{2})', content)
                if m:
                    date = m.group(1)

        # Content
        content = ""
        nr_zw = soup.select_one(".nr-zw")
        if nr_zw:
            parts = []
            for child in nr_zw.children:
                if child.name == "p":
                    txt = child.get_text(separator=" ", strip=True)
                    if txt:
                        parts.append(txt)
                elif child.name == 'table':
                    tbl_html = html_table_to_html(child, url)
                    if tbl_html:
                        parts.append(tbl_html)
                elif child.name in ["div", "span"]:
                    txt = child.get_text(separator=" ", strip=True)
                    if txt:
                        parts.append(txt)
            content = "\n\n".join(parts)

        # Attachments
        attachments = []
        for a in soup.find_all("a"):
            href = a.get("href", "")
            txt = a.get_text(strip=True)
            if any(x in href.lower() for x in [".pdf", ".doc", ".xls", ".zip", ".docx", ".xlsx"]):
                if not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                attachments.append({
                    "name": txt or href.split("/")[-1],
                    "url": href
                })

        # PDF content fallback
        if len(content.strip()) < 20:
            content = '<p><a href="{}">{}</a></p>'.format(url, title)
            if attachments:
                attachment_links = "\n".join([" - [{}]({})".format(a["name"], a["url"]) for a in attachments])
                content += "\n\n附件：\n" + attachment_links
            content += "\n\n（PDF文档，请点击原文链接查看）"

        result = {
            "title": title,
            "url": url,
            "date": date,
            "content": content,
            "site_name": SITE_NAME,
            "source": "",
            "attachments": attachments,
        }
        return result
    except Exception as e:
        print("  Error parsing {}: {}".format(url, e), file=sys.stderr)
        return None


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def save_to_db(articles):
    """Save articles to search.db"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for art in articles:
        if not art or not art["title"]:
            continue
        # Check if already exists
        c.execute(
            "SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?",
            (art["url"], SITE_NAME)
        )
        if c.fetchone():
            continue
        attachments_json = str(art["attachments"]) if art["attachments"] else ""
        c.execute(
            """INSERT OR REPLACE INTO gov_raw (page_url, title, content, summary, site_name, publish_date, source_url, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_zghaq.py')""",
            (
                art["url"],
                art["title"],
                art["content"],
                art["content"][:200] if art["content"] else "",
                art["site_name"],
                art["date"],
                art.get("source", ""),
                attachments_json,
            )
        )
        inserted += 1
    conn.commit()
    conn.close()
    return inserted


def main():
    max_pages = 5
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            pass

    print("Starting crawl: {} - {} ({} pages)".format(SITE_NAME, LIST_URL, max_pages), file=sys.stderr)

    session = get_session()

    # Get URLs
    print("Fetching list pages...", file=sys.stderr)
    urls = get_list_urls(session, max_pages)
    print("Total URLs: {}".format(len(urls)), file=sys.stderr)

    # Parse each URL
    print("Fetching detail pages...", file=sys.stderr)
    articles = []
    for i, url in enumerate(urls):
        print("  [{}/{}] {}".format(i + 1, len(urls), url), file=sys.stderr)
        art = parse_detail(session, url)
        if art:
            articles.append(art)

    print("Parsed: {}/{} articles".format(len(articles), len(urls)), file=sys.stderr)

    # Save to DB
    inserted = save_to_db(articles)
    print("Inserted: {} new articles".format(inserted), file=sys.stderr)
    print("Done.", file=sys.stderr)

    # Print article count for verification
    print("\nArticle count for config: {}".format(len(urls)))


if __name__ == "__main__":
    main()
