#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_lylgkfq_gsgg.py - 临沂临港经济开发区-公示公告
CMS: VSB9 (Visual SiteBuilder 9)
列表: /xw/gsgg.htm (page 1) → /xw/gsgg/{N}.htm (N>=2)
      div.list > a[href] + [YYYY-MM-DD] 日期
详情: /info/2254/NNNNN.htm
标题: <title>标签 (去除站点名后缀)
日期: <meta PubDate>
正文: div.v_news_content > p
"""

import requests
import re
import sys
import os
import time
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "http://www.lylgkfq.gov.cn"
SITE_NAME = "临沂临港-公示公告"
GROUP = "山东"
INDUSTRY = "环境公示"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

SESSION = None


def get(url):
    global SESSION
    if SESSION is None:
        SESSION = requests.Session()
        SESSION.headers.update(HEADERS)
    try:
        r = SESSION.get(url, timeout=60)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"ERR:{e}")
        return None


def fetch_list_page(page_no):
    if page_no == 1:
        url = f"{BASE_URL}/xw/gsgg.htm"
    else:
        url = f"{BASE_URL}/xw/gsgg/{page_no}.htm"
    return get(url)


def parse_list(html):
    """Extract (url, title, date) from list page"""
    items = []
    soup = BeautifulSoup(html, "html.parser")

    # Find div.list which contains the news items
    list_div = soup.find("div", class_="list")
    if not list_div:
        return items

    for a in list_div.find_all("a", href=True):
        href = a.get("href", "").strip()
        if not href or "/info/2254/" not in href:
            continue
        title = a.get_text(strip=True)
        if not title:
            continue

        # Full URL
        if href.startswith("../"):
            href = f"{BASE_URL}/{href[3:]}"
        elif href.startswith("/"):
            href = f"{BASE_URL}{href}"

        # Date from after the link text - look for [YYYY-MM-DD] in parent
        parent = a.parent if a.parent else a
        parent_text = parent.get_text() if parent else ""

        # Try to find date in brackets after the title
        date_str = ""
        # First try meta in the link's parent div
        date_match = re.search(rf"{re.escape(title)}\s*\[(\d{{4}}-\d{{2}}-\d{{2}})\]", parent_text)
        if date_match:
            date_str = date_match.group(1)
        if not date_str:
            # Broader search in parent
            date_match = re.search(r"\[(\d{4}-\d{2}-\d{2})\]", parent_text)
            if date_match:
                date_str = date_match.group(1)
        # Fallback: search in whole page for this title's date
        if not date_str:
            body = list_div.get_text()
            date_match = re.search(rf"{re.escape(title)}.*?\[(\d{{4}}-\d{{2}}-\d{{2}})\]", body)
            if date_match:
                date_str = date_match.group(1)

        items.append((href, title, date_str))

    return items


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title from <title> tag
    title = ""
    title_tag = soup.find("title")
    if title_tag:
        raw = title_tag.get_text(strip=True)
        # Remove site name suffix
        raw = re.sub(r"\s*[-–—]\s*临沂临港经济开发区.*$", "", raw)
        title = raw.strip()

    # Date from <meta PubDate>
    date_str = ""
    pubdate_meta = soup.find("meta", {"name": "PubDate"})
    if pubdate_meta:
        content = pubdate_meta.get("content", "")
        m = re.search(r"(\d{4}-\d{2}-\d{2})", content)
        if m:
            date_str = m.group(1)

    # Content from div.v_news_content
    content = ""
    vc = soup.find("div", class_="v_news_content")
    if not vc:
        vc = soup.find("div", id="vsb_content_2")
    if not vc:
        # Fallback: try div.con
        vc = soup.find("div", class_="con")

    if vc:
        paragraphs = []
        # 递归找所有 p，但排除表格内的单元格 p（表格单独处理）
        table_ps = set()
        for tbl in vc.find_all("table"):
            table_ps.update(tbl.find_all("p"))
        for p in vc.find_all("p"):
            if p in table_ps:
                continue
            # 保留 HTML 结构（分段渲染正确），只做清洗：去掉 class/style、链接绝对化
            p_copy = BeautifulSoup(str(p), "html.parser").p or BeautifulSoup(str(p), "html.parser")
            for a in p_copy.find_all("a"):
                href = a.get("href", "").strip()
                if not href:
                    continue
                if href.startswith("//"):
                    href = "http:" + href
                elif href.startswith("/"):
                    href = BASE_URL + href
                elif not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                a["href"] = href.replace("&amp;", "&")
            for img in p_copy.find_all("img"):
                src = img.get("src", "").strip()
                if not src:
                    continue
                if src.startswith("//"):
                    src = "http:" + src
                elif src.startswith("/"):
                    src = BASE_URL + src
                elif not src.startswith("http"):
                    src = BASE_URL + "/" + src.lstrip("/")
                img["src"] = src.replace("&amp;", "&")
            p_html = re.sub(r'\s+(class|style|align|valign|width|height|border|cellspacing|cellpadding)="[^"]*"', '', str(p_copy))
            # 清理空 p（只含 <br> 或空白）
            p_text = re.sub(r'<[^>]+>', '', p_html).strip()
            if not p_text and '<img' not in p_html:
                continue
            paragraphs.append(p_html)
        # Also check for tables - 保留 HTML 表格结构，链接转绝对
        for tbl in vc.find_all("table"):
            tbl_copy = BeautifulSoup(str(tbl), "html.parser")
            for a in tbl_copy.find_all("a"):
                href = a.get("href", "").strip()
                if not href:
                    continue
                if href.startswith("//"):
                    href = "http:" + href
                elif href.startswith("/"):
                    href = BASE_URL + href
                elif not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                a["href"] = href.replace("&amp;", "&")
            for img in tbl_copy.find_all("img"):
                src = img.get("src", "").strip()
                if not src:
                    continue
                if src.startswith("//"):
                    src = "http:" + src
                elif src.startswith("/"):
                    src = BASE_URL + src
                elif not src.startswith("http"):
                    src = BASE_URL + "/" + src.lstrip("/")
                img["src"] = src.replace("&amp;", "&")
            tbl_html = str(tbl_copy)
            if tbl_html:
                paragraphs.append(tbl_html)
        if paragraphs:
            content = "\n\n".join(paragraphs)
        else:
            content = vc.get_text(separator="\n").strip()

    if content:
        content = re.sub(r"&nbsp;", "", content)

    summary = content[:200] if content else ""
    return title, date_str, content, summary


def save_to_db(items_data):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=10)
    c = conn.cursor()
    inserted = 0
    fts_batch = []
    for title, date_str, content, summary, page_url in items_data:
        try:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                continue
            c.execute(
                """INSERT INTO gov_raw 
                   (page_url, title, site_name, publish_date, content, summary,
                    source_url, category, industry, group_name, script_name)
                   VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (page_url, title, SITE_NAME, date_str, content, summary,
                 page_url, "环境公示", INDUSTRY, GROUP, "crawl_lylgkfq_gsgg.py"),
            )
            new_id = c.lastrowid
            if new_id:
                fts_batch.append((new_id, title, SITE_NAME, summary or ""))
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
    conn.commit()
    if fts_batch:
        for rid, title, site, summary in fts_batch:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                    (rid, title, site, summary),
                )
            except Exception as e:
                print(f"  [FTS ERROR] id={rid}: {e}")
        conn.commit()
        print(f"  FTS同步: {len(fts_batch)}条")
    conn.close()
    return inserted


def main():
    import argparse
    parser = argparse.ArgumentParser(description=f"{SITE_NAME}爬虫")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数(共141页)")
    parser.add_argument("--full", action="store_true", help="全量爬取(141页)")
    args = parser.parse_args()

    max_pages = 141 if args.full else args.pages
    total_new = 0

    for page_no in range(1, max_pages + 1):
        print(f"[{time.strftime('%H:%M:%S')}] 第{page_no}页...", end=" ", flush=True)
        html = fetch_list_page(page_no)
        if not html:
            print("请求失败，停止")
            break

        items = parse_list(html)
        if not items:
            print("无数据，停止")
            break
        print(f"{len(items)}条", flush=True)

        batch = []
        for href, title, date_str in items:
            detail_html = get(href)
            if not detail_html:
                continue
            dt, dd, content, summary = extract_detail(detail_html, href)
            batch.append((dt or title, dd or date_str, content, summary, href))
            time.sleep(0.5)

        saved = save_to_db(batch)
        total_new += saved
        print(f"  入库{saved}/{len(batch)}条 (累计{total_new})", flush=True)
        if not args.full and saved == 0 and len(batch) > 0:
            print("  全部已存在，增量停止")
            break

    print(f"\n===== 完成 =====")
    print(f"新增入库: {total_new} 条")
    print(f"站点: {SITE_NAME}")


if __name__ == "__main__":
    main()
