#!/usr/bin/env python3
"""
宿豫区人民政府 - 高新区环评公示
URL: http://www.suyu.gov.cn/suyu/gxqhpgs/xxgk_list.shtml
CMS: TRS (createPageHTML), pages: xxgk_list_N.shtml, 正文 #zoomcon UCAPCONTENT
"""

import requests, re, sys, os, json
from bs4 import BeautifulSoup
from datetime import datetime
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "宿豫区-高新区环评公示"
BASE_URL = "http://www.suyu.gov.cn"
LIST_URL = BASE_URL + "/suyu/gxqhpgs/xxgk_list.shtml"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

import argparse
parser = argparse.ArgumentParser()
parser.add_argument("--pages", type=int, default=0)
parser.add_argument("--verbose", "-v", action="store_true")
args = parser.parse_args()
MAX_PAGES = args.pages if args.pages > 0 else 99


def log(msg):
    if args.verbose:
        print("[%s] %s" % (datetime.now().strftime("%H:%M:%S"), msg))


def fetch_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        return r.text
    except:
        return None


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.select_one("ul.listContent")
    if not ul:
        return []
    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if not href.endswith(".shtml") or "xxgk_list" in href:
            continue
        # 完整标题: a[title] 属性
        title = (a.get("title") or "").strip()
        if not title:
            title = a.get_text(strip=True)
        if not title:
            continue
        page_url = urljoin(BASE_URL, href)
        # 日期
        span = li.find("span")
        pub_date = span.get_text(strip=True) if span else ""
        items.append({"href": page_url, "title": title, "date": pub_date})
    return items


def get_total_pages(html):
    m = re.search(r"createPageHTML\([^,]+,\s*(\d+)", html)
    if m:
        return int(m.group(1))
    return 1


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return "", "", "", ""
    except:
        return "", "", "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    # 标题
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)

    # 日期
    pub_date = ""
    pub_meta = soup.find("meta", attrs={"name": "PubDate"})
    if pub_meta and pub_meta.get("content"):
        raw_date = pub_meta["content"].strip()
        try:
            dt = datetime.strptime(raw_date[:10], "%Y-%m-%d")
            pub_date = dt.strftime("%Y-%m-%d")
        except:
            pub_date = raw_date[:10]

    # 正文
    content_parts = []
    attachments = []

    zoomcon = soup.find(id="zoomcon")
    if zoomcon:
        # 可能大写 UCAPCONTENT 或小写 ucapcontent
        ucap = zoomcon.find("UCAPCONTENT") or zoomcon.find("ucapcontent") or zoomcon
        for child in ucap.children:
            if not hasattr(child, "name"):
                continue
            if child.name == "p":
                text = child.get_text(strip=True)
                if text:
                    table = child.find("table")
                    if table:
                        content_parts.append(str(table))
                    else:
                        content_parts.append(text)
            elif child.name == "table":
                content_parts.append(str(child))
            elif child.name in ("div", "section"):
                txt = child.get_text(strip=True)
                if txt and len(txt) > 3:
                    content_parts.append(txt)
            elif child.name == "img":
                src = child.get("src", "")
                if src:
                    if src.startswith("/"):
                        src = BASE_URL + src
                    elif not src.startswith("http"):
                        src = BASE_URL + "/" + src.lstrip("/")
                    content_parts.append("![image](%s)" % src)

    content = "\n\n".join(p for p in content_parts if p)
    content = re.sub(r"[ \t]+", " ", content).strip()

    # 附件: 全文搜索文件链接
    file_exts = (".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")
    for a_tag in soup.find_all("a", href=True):
        h = a_tag["href"]
        ext = os.path.splitext(h)[1].lower()
        if ext in file_exts:
            file_url = urljoin(url, h)
            file_name = a_tag.get_text(strip=True) or os.path.basename(h)
            attachments.append({"name": file_name, "url": file_url})
            content += '\n\n附件：<p><a href="%s">%s</a></p>' % (file_url, file_name)

    return title, content, json.dumps(attachments, ensure_ascii=False), pub_date


# Main
log("Fetching first page...")
first_html = fetch_page(LIST_URL)
if not first_html:
    print("ERROR: cannot fetch list page")
    sys.exit(1)

total_pages = get_total_pages(first_html)
pages_to_crawl = min(total_pages, MAX_PAGES)
print("总 %d 页，爬取 %d 页" % (total_pages, pages_to_crawl))

all_items = []
seen_urls = set()

for page_num in range(1, pages_to_crawl + 1):
    if page_num == 1:
        url = LIST_URL
        html = first_html
    else:
        url = "%s/suyu/gxqhpgs/xxgk_list_%d.shtml" % (BASE_URL, page_num)
        html = fetch_page(url)
    if not html:
        break
    items = parse_list(html)
    if not items:
        log("Page %d: empty" % page_num)
        if page_num > 1:
            break
        continue
    new_items = [it for it in items if it["href"] not in seen_urls]
    for it in new_items:
        seen_urls.add(it["href"])
    all_items.extend(new_items)
    log("Page %d: +%d (total: %d)" % (page_num, len(new_items), len(all_items)))

print("\n列表: %d 条" % len(all_items))

if not all_items:
    print("No items.")
    sys.exit(0)

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
try:
    from crawler_lib import push_to_searchdb, get_connection
except ImportError:
    import sqlite3
    def push_to_searchdb(items, site_name):
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        c = conn.cursor()
        new_c = 0
        skip_c = 0
        for item in items:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, source_url, page_url, title, publish_date, date_rank, summary, content, attachments) "
                    "VALUES (?,?,?,?,?,?,?,?,?)",
                    (site_name, item.get("url",""), item.get("url",""),
                     item.get("title",""), item.get("pub_date",""),
                     item.get("date_rank",0), item.get("summary",""),
                     item.get("content",""), item.get("attachments","[]"))
                )
                if c.lastrowid and c.lastrowid > 0:
                    new_c += 1
                else:
                    skip_c += 1
            except:
                skip_c += 1
        conn.commit()
        conn.close()
        return new_c, skip_c
    def get_connection():
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        return conn


print("抓取详情中 (%d 条)..." % len(all_items))
items_for_db = []

for idx, item in enumerate(all_items):
    log("[%d/%d] %s..." % (idx+1, len(all_items), item["title"][:40]))
    title, content, attachments_json, pub_date = fetch_detail(item["href"])

    if not title:
        title = item["title"]
    if not pub_date:
        pub_date = item["date"]

    summary = (BeautifulSoup(content, "html.parser").get_text(strip=True) if "<" in content else content)[:500]
    date_rank = 0
    if pub_date:
        try:
            dt = datetime.strptime(pub_date[:10], "%Y-%m-%d")
            date_rank = int(dt.strftime("%Y%m%d"))
        except:
            pass

    items_for_db.append({
        "title": title,
        "url": item["href"],
        "content": content,
        "pub_date": pub_date,
        "date_rank": date_rank,
        "summary": summary,
        "source_url": item["href"],
        "attachments": attachments_json,
        "page_url": item["href"],
    })

print("\n入库中 (%d 条)..." % len(items_for_db))
new, skip = push_to_searchdb(items_for_db, SITE_NAME)
print("入库完成: new=%d, skip=%d" % (new, skip))

conn = get_connection()
count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
conn.close()
print("DB中 %s 共 %d 条" % (SITE_NAME, count))
