#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫: 湛江经济技术开发区 - 环保审批公示
http://www.zetdz.gov.cn/qfj/hbj/hpgs/
CMS: 湛江开发区门户
"""

import requests
from bs4 import BeautifulSoup
import sqlite3
import re
import sys

DB_PATH = "/root/search.db"
BASE_URL = "http://www.zetdz.gov.cn"
LIST_URL = "http://www.zetdz.gov.cn/qfj/hbj/hpgs/"
SITE_NAME = "湛江经济技术开发区"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def get_list_urls(max_pages=5):
    """Get all article URLs from list pages"""
    url_data = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = "http://www.zetdz.gov.cn/qfj/hbj/hpgs/index_{}.html".format(page)

        try:
            r = requests.get(url, timeout=15, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print("  Page {}: HTTP {}".format(page, r.status_code), file=sys.stderr)
                break

            soup = BeautifulSoup(r.text, "html.parser")
            items = soup.select("ul.listTit li")
            if not items:
                print("  Page {}: no items".format(page), file=sys.stderr)
                break

            count = 0
            for li in items:
                a = li.find("a")
                if a and a.get("href"):
                    href = a["href"]
                    b = a.find("b")
                    title = b.get_text(strip=True) if b else a.get_text(strip=True)
                    span = li.find("span")
                    pub_date = span.get_text(strip=True) if span else ""

                    if not href.startswith("http"):
                        href = BASE_URL + href

                    url_data.append((href, title, pub_date))
                    count += 1

            print("  Page {}: {} items".format(page, count), file=sys.stderr)
            if len(items) < 5:
                break
        except Exception as e:
            print("  Page {} error: {}".format(page, e), file=sys.stderr)
            break

    return url_data


def parse_detail(url):
    """Parse a detail page"""
    try:
        r = requests.get(url, timeout=15, headers=HEADERS)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")

        # Title from browser title
        title = ""
        if soup.title:
            raw = soup.title.string.strip()
            # Remove site suffix
            for suffix in [" - 湛江经济技术开发区门户网站", " - 湛江经济技术开发区"]:
                if raw.endswith(suffix):
                    raw = raw[:-len(suffix)]
                    break
            title = raw

        # Date
        date = ""
        full_text = soup.get_text()
        m = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", full_text)
        if m:
            date = m.group(1)

        # Content from .showCons
        content = ""
        sc = soup.select_one(".showCons")
        if sc:
            parts = []
            for child in sc.children:
                if child.name == "p":
                    txt = child.get_text(separator=" ", strip=True)
                    if txt:
                        parts.append(txt)
                elif child.name == "table":
                    md_table = table_to_markdown(child)
                    if md_table:
                        parts.append(md_table)
            content = "\n\n".join(parts)

        # Attachments
        attachments = []
        for a in soup.find_all("a", href=True):
            href = a["href"]
            txt = a.get_text(strip=True)
            if any(x in href.lower() for x in [".pdf", ".doc", ".xls", ".zip", ".docx", ".xlsx"]):
                if not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                attachments.append({
                    "name": txt or href.split("/")[-1],
                    "url": href
                })

        # PDF content fallback
        if len(content.strip()) < 20:
            content = '<p><a href="{}">{}</a></p>'.format(url, title)
            if attachments:
                al = "\n".join([" - [{}]({})".format(a["name"], a["url"]) for a in attachments])
                content += "\n\n附件：\n" + al
            content += "\n\n（原文链接查看）"

        return {
            "title": title,
            "url": url,
            "date": date,
            "content": content,
            "site_name": SITE_NAME,
            "source": "",
            "attachments": attachments,
        }
    except Exception as e:
        print("  Error parsing {}: {}".format(url, e), file=sys.stderr)
        return None


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def save_to_db(articles):
    """Save articles to search.db"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for art in articles:
        if not art or not art["title"]:
            continue
        c.execute(
            "SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?",
            (art["url"], SITE_NAME)
        )
        if c.fetchone():
            continue
        attachments_json = str(art["attachments"]) if art["attachments"] else ""
        c.execute(
            """INSERT OR REPLACE INTO gov_raw
               (page_url, title, content, summary, site_name, publish_date, source_url, attachments)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
            (
                art["url"],
                art["title"],
                art["content"],
                art["content"][:200] if art["content"] else "",
                art["site_name"],
                art["date"],
                art.get("source", ""),
                attachments_json,
            )
        )
        inserted += 1
    conn.commit()
    conn.close()
    return inserted


def main():
    max_pages = 5
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            pass

    print("Starting crawl: {} - {} ({} pages)".format(SITE_NAME, LIST_URL, max_pages), file=sys.stderr)

    # Get URLs
    print("Fetching list pages...", file=sys.stderr)
    url_data = get_list_urls(max_pages)
    print("Total URLs: {}".format(len(url_data)), file=sys.stderr)

    # Parse
    print("Fetching detail pages...", file=sys.stderr)
    articles = []
    for i, (url, title, date) in enumerate(url_data):
        print("  [{}/{}] {}".format(i + 1, len(url_data), url), file=sys.stderr)
        art = parse_detail(url)
        if art:
            if not art["title"] and title:
                art["title"] = title
            if not art["date"] and date:
                art["date"] = date
            articles.append(art)

    print("Parsed: {}/{} articles".format(len(articles), len(url_data)), file=sys.stderr)

    # Save
    inserted = save_to_db(articles)
    print("Inserted: {} new articles".format(inserted), file=sys.stderr)
    print("Done.", file=sys.stderr)

    print("\nArticle count for config: {}".format(len(url_data)))


if __name__ == "__main__":
    main()
