#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫: 平顶山尼龙新材料开发区 - 通知公告
http://www.zgnlc.gov.cn/channels/11920.html
CMS: 自定义
"""

import requests
from bs4 import BeautifulSoup
import sqlite3
import re
import sys

DB_PATH = "/root/search.db"
BASE_URL = "http://www.zgnlc.gov.cn"
LIST_URL = "http://www.zgnlc.gov.cn/channels/11920.html"
SITE_NAME = "平顶山尼龙新材料开发区"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def get_list_urls(max_pages=5):
    """Get all article URLs from list pages"""
    urls = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = "http://www.zgnlc.gov.cn/channels/11920_{}.html".format(page)

        try:
            r = requests.get(url, timeout=15, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print("  Page {}: HTTP {}".format(page, r.status_code), file=sys.stderr)
                continue

            soup = BeautifulSoup(r.text, "html.parser")
            items = soup.select("div.content-list li")
            if not items:
                print("  Page {}: no items".format(page), file=sys.stderr)
                break

            count = 0
            for li in items:
                a = li.find("a")
                if a and a.get("href"):
                    href = a["href"]
                    if not href.startswith("http"):
                        href = BASE_URL + href
                    # Store the title from the a tag's title attribute (full title)
                    full_title = a.get("title", "").strip()
                    # Also get date
                    date_span = li.find("span")
                    pub_date = date_span.get_text(strip=True) if date_span else ""
                    urls.append((href, full_title, pub_date))
                    count += 1

            print("  Page {}: {} items".format(page, count), file=sys.stderr)
            if len(items) < 5:
                break
        except Exception as e:
            print("  Page {} error: {}".format(page, e), file=sys.stderr)
            break

    return urls


def parse_detail(url):
    """Parse a detail page"""
    try:
        r = requests.get(url, timeout=15, headers=HEADERS)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None

        soup = BeautifulSoup(r.text, "html.parser")

        # Title from h4
        title = ""
        h4 = soup.find("h4")
        if h4:
            title = h4.get_text(strip=True)

        # Date from .maintext
        date = ""
        maintext = soup.select_one(".maintext")
        if maintext:
            m = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", maintext.get_text())
            if m:
                date = m.group(1)

        # Content from .mainp
        content = ""
        mainp = soup.select_one(".mainp")
        if mainp:
            parts = []
            for child in mainp.children:
                if child.name == "p":
                    txt = child.get_text(separator=" ", strip=True)
                    if txt:
                        parts.append(txt)
                elif child.name == 'table':
                    tbl_html = html_table_to_html(child, url)
                    if tbl_html:
                        parts.append(tbl_html)
            content = "\n\n".join(parts)

        # Attachments
        attachments = []
        for a in soup.find_all("a", href=True):
            href = a["href"]
            txt = a.get_text(strip=True)
            if any(x in href.lower() for x in [".pdf", ".doc", ".xls", ".zip", ".docx", ".xlsx"]):
                if not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                attachments.append({
                    "name": txt or href.split("/")[-1],
                    "url": href
                })

        # PDF content fallback
        if len(content.strip()) < 20:
            content = '<p><a href="{}">{}</a></p>'.format(url, title)
            if attachments:
                attachment_links = "\n".join([" - [{}]({})".format(a["name"], a["url"]) for a in attachments])
                content += "\n\n附件：\n" + attachment_links
            content += "\n\n（PDF文档，请点击原文链接查看）"

        result = {
            "title": title,
            "url": url,
            "date": date,
            "content": content,
            "site_name": SITE_NAME,
            "source": "",
            "attachments": attachments,
        }
        return result
    except Exception as e:
        print("  Error parsing {}: {}".format(url, e), file=sys.stderr)
        return None


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def save_to_db(articles):
    """Save articles to search.db"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for art in articles:
        if not art or not art["title"]:
            continue
        c.execute(
            "SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?",
            (art["url"], SITE_NAME)
        )
        if c.fetchone():
            continue
        attachments_json = str(art["attachments"]) if art["attachments"] else ""
        c.execute(
            """INSERT OR REPLACE INTO gov_raw (page_url, title, content, summary, site_name, publish_date, source_url, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_zgnlc.py')""",
            (
                art["url"],
                art["title"],
                art["content"],
                art["content"][:200] if art["content"] else "",
                art["site_name"],
                art["date"],
                art.get("source", ""),
                attachments_json,
            )
        )
        inserted += 1
    conn.commit()
    conn.close()
    return inserted


def main():
    max_pages = 5
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            pass

    print("Starting crawl: {} - {} ({} pages)".format(SITE_NAME, LIST_URL, max_pages), file=sys.stderr)

    # Get URLs from list pages - includes full title from title attribute
    print("Fetching list pages...", file=sys.stderr)
    url_data = get_list_urls(max_pages)
    print("Total URLs: {}".format(len(url_data)), file=sys.stderr)

    # Parse each URL
    print("Fetching detail pages...", file=sys.stderr)
    articles = []
    for i, (url, list_title, list_date) in enumerate(url_data):
        print("  [{}/{}] {}".format(i + 1, len(url_data), url), file=sys.stderr)
        art = parse_detail(url)
        if art:
            # Use list title if detail title is empty
            if not art["title"] and list_title:
                art["title"] = list_title
            if not art["date"] and list_date:
                art["date"] = list_date
            articles.append(art)

    print("Parsed: {}/{} articles".format(len(articles), len(url_data)), file=sys.stderr)

    # Save to DB
    inserted = save_to_db(articles)
    print("Inserted: {} new articles".format(inserted), file=sys.stderr)
    print("Done.", file=sys.stderr)

    print("\nArticle count for config: {}".format(len(url_data)))


if __name__ == "__main__":
    main()
