#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫: 湛江经济技术开发区 - 民意征集
http://www.zetdz.gov.cn/wz/hdjl/myzj/
CMS: 广东省统一互动交流平台
"""

import requests
from bs4 import BeautifulSoup
import sqlite3
import re
import json
import sys

DB_PATH = "/root/search.db"
BASE_URL = "http://www.zetdz.gov.cn"
LIST_BASE = "http://www.zetdz.gov.cn/wz/hdjl/myzj"
SITE_NAME = "湛江经济技术开发区"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def get_list_urls(max_pages=5):
    url_data = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_BASE + "/"
        else:
            url = LIST_BASE + "/index_{}.html".format(page)
        try:
            r = requests.get(url, timeout=15, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print("  Page {}: HTTP {}".format(page, r.status_code), file=sys.stderr)
                break
            soup = BeautifulSoup(r.text, "html.parser")
            items = soup.select("ul.listTit li")
            if not items:
                break
            count = 0
            for li in items:
                a = li.find("a")
                if a and a.get("href"):
                    href = a["href"]
                    b = a.find("b")
                    title = b.get_text(strip=True) if b else ""
                    sp = li.find("span")
                    pub_date = sp.get_text(strip=True) if sp else ""
                    if not href.startswith("http"):
                        href = BASE_URL + href
                    url_data.append((href, title, pub_date))
                    count += 1
            print("  Page {}: {} items".format(page, count), file=sys.stderr)
            if len(items) < 5:
                break
        except Exception as e:
            print("  Page {} error: {}".format(page, e), file=sys.stderr)
            break
    return url_data


def parse_detail(url):
    try:
        r = requests.get(url, timeout=15, headers=HEADERS)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None

        soup = BeautifulSoup(r.text, "html.parser")
        title = ""
        date = ""
        content = ""
        attachments = []

        # Type A: question_data JSON (/hdjlpt/yjzj/answer/...)
        m = re.search(r'question_data\s*:\s*({.*?}),\s*\n', r.text, re.S)
        is_json = False
        if m:
            try:
                qd = json.loads(m.group(1))
                art = qd.get("article", {})
                if isinstance(art, dict):
                    is_json = True
                    title = art.get("title", "") or ""
                    date_raw = art.get("published_at", "") or ""
                    if date_raw:
                        dm = re.search(r"(\d{4}-\d{2}-\d{2})", str(date_raw))
                        if dm:
                            date = dm.group(1)
                    content_html = art.get("content", "") or ""
                    if content_html:
                        csoup = BeautifulSoup(str(content_html), "html.parser")
                        parts = []
                        for child in csoup.children:
                            if child.name == "p":
                                txt = child.get_text(separator=" ", strip=True)
                                if txt:
                                    parts.append(txt)
                            elif child.name == "table":
                                md = table_to_markdown(child)
                                if md:
                                    parts.append(md)
                        content = "\n\n".join(parts)
                    atts = art.get("attachments", []) or []
                    if isinstance(atts, list):
                        for att in atts:
                            if isinstance(att, dict):
                                name = att.get("name", "") or att.get("original_name", "") or ""
                                url2 = att.get("url", "") or att.get("path", "") or ""
                                if url2:
                                    if not str(url2).startswith("http"):
                                        url2 = BASE_URL + "/" + str(url2).lstrip("/")
                                    attachments.append({"name": name, "url": str(url2)})
            except (json.JSONDecodeError, AttributeError):
                pass

        if not is_json:
            # Type B: standard CMS (/wz/hdjl/myzj/content/post_...)
            # Title from browser title
            if soup.title:
                raw = soup.title.string.strip()
                for sf in [" - 湛江经济技术开发区门户网站", " - 湛江经济技术开发区"]:
                    if raw.endswith(sf):
                        raw = raw[:-len(sf)]
                        break
                title = raw

            # Date
            m2 = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", r.text)
            if m2:
                date = m2.group(1)

            # Content from .showCons
            sc = soup.select_one(".showCons")
            if sc:
                parts = []
                for child in sc.children:
                    if child.name == "p":
                        txt = child.get_text(separator=" ", strip=True)
                        if txt:
                            parts.append(txt)
                    elif child.name == "table":
                        md = table_to_markdown(child)
                        if md:
                            parts.append(md)
                content = "\n\n".join(parts)

            # Attachments
            for a in soup.find_all("a", href=True):
                h = a["href"]
                txt = a.get_text(strip=True)
                if any(x in h.lower() for x in [".pdf", ".doc", ".xls", ".zip", ".docx", ".xlsx"]):
                    if not h.startswith("http"):
                        h = BASE_URL + "/" + h.lstrip("/")
                    attachments.append({"name": txt or h.split("/")[-1], "url": h})

        # PDF content fallback
        if len(content.strip()) < 20:
            content = '<p><a href="{}">{}</a></p>'.format(url, title)
            if attachments:
                al = "\n".join([" - [{}]({})".format(a["name"], a["url"]) for a in attachments])
                content += "\n\n附件：\n" + al
            content += "\n\n（原文链接查看）"

        return {"title": title, "url": url, "date": date, "content": content,
                "site_name": SITE_NAME, "source": "", "attachments": attachments}
    except Exception as e:
        print("  Error {}: {}".format(url, e), file=sys.stderr)
        return None


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def save_to_db(articles):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for art in articles:
        if not art or not art["title"]:
            continue
        c.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?", (art["url"], SITE_NAME))
        if c.fetchone():
            continue
        aj = str(art["attachments"]) if art["attachments"] else ""
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, summary, site_name, publish_date, source_url, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_zetdz_myzj.py')""",
            (art["url"], art["title"], art["content"], art["content"][:200],
             art["site_name"], art["date"], art.get("source", ""), aj))
        inserted += 1
    conn.commit()
    conn.close()
    return inserted


def main():
    mp = 5
    if len(sys.argv) > 1:
        try:
            mp = int(sys.argv[1])
        except ValueError:
            pass
    print("Starting: {} ({} pages)".format(LIST_BASE, mp), file=sys.stderr)
    ud = get_list_urls(mp)
    print("URLs: {}".format(len(ud)), file=sys.stderr)
    articles = []
    for i, (u, t, d) in enumerate(ud):
        print("  [{}/{}] {}".format(i+1, len(ud), u), file=sys.stderr)
        art = parse_detail(u)
        if art:
            if not art["title"] and t:
                art["title"] = t
            if not art["date"] and d:
                art["date"] = d
            articles.append(art)
    print("Parsed: {}/{}".format(len(articles), len(ud)), file=sys.stderr)
    ins = save_to_db(articles)
    print("Inserted: {}".format(ins), file=sys.stderr)
    print("Done.")
    print("\nCount: {}".format(len(ud)))


if __name__ == "__main__":
    main()
