#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Crawler for 马鞍山经济技术开发区 - 通知公告
# 列表: ul.doc_list.list-4716728 > li, a[title]全标题, span.right.date日期
# 详情: meta ArticleTitle, meta PubDate, div.wzcon.j-fontContent正文
# 分页: https://jkq.mas.gov.cn/content/column/4716728?pageIndex=N

import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import os

LIST_URL = "https://jkq.mas.gov.cn/content/column/4716728?pageIndex={}"
LIST_URL_P1 = "https://jkq.mas.gov.cn/xxfb/tzgg/index.html"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "马鞍山经济技术开发区通知公告"
FULL = "--full" in sys.argv
MAX_PAGES = 5 if not FULL else 100

CONTENT_PAT = re.compile(
    r'<div\s+class="wzcon\s+j-fontContent"[^>]*>([\s\S]*?)</div>\s*</div>'
)
TITLE_META_PAT = re.compile(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"')
DATE_META_PAT = re.compile(r'<meta\s+name="PubDate"\s+content="([^"]+)"')
ATTACH_PAT = re.compile(
    r'<a[^>]+href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]+)</a>',
    re.I
)


def extract_detail(detail_url):
    """提取详情页标题、日期、正文HTML、附件"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  Detail fetch error: {e}")
        return None, None, None

    tm = TITLE_META_PAT.search(html)
    dm = DATE_META_PAT.search(html)
    cm = CONTENT_PAT.search(html)

    title = tm.group(1).strip() if tm else None
    pub_date = dm.group(1).strip()[:10] if dm else None
    content = cm.group(1).strip() if cm else None

    # 提取附件链接，追加到正文末尾
    attaches = ATTACH_PAT.findall(html)
    if attaches and content:
        attach_html = "\n<div class=\"attachments\">\n<p><strong>附件：</strong></p>\n"
        for url, fname in attaches:
            full_url = url if url.startswith("http") else "https://jkq.mas.gov.cn" + url
            attach_html += f'<p><a href="{full_url}">{fname}</a></p>\n'
        attach_html += "</div>"
        content += attach_html

    return title, pub_date, content


def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = total_before = 0

    print(f"Crawling {MAX_PAGES} pages of {SITE_NAME}")

    for pg in range(1, MAX_PAGES + 1):
        # Page 1 uses index.html, others use column API
        if pg == 1:
            url = LIST_URL_P1
        else:
            url = LIST_URL.format(pg)

        try:
            r = requests.get(url, headers=HEADERS, timeout=15)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"Page {pg}: fetch error - {e}")
            continue

        soup = BeautifulSoup(r.text, "html.parser")
        items = [li for li in soup.select("ul.doc_list > li")
                 if li.find("a") and li.find("span", class_="date")]

        if not items:
            print(f"Page {pg}: empty, stopping")
            break

        page_new = page_skip = page_before = 0
        for li in items:
            a = li.find("a")
            sp = li.find("span", class_="date")

            href = a.get("href", "").strip()
            if not href.startswith("http"):
                href = "https://jkq.mas.gov.cn" + href

            # 用 a[title] 获取完整标题（列表页 span 被截断带省略号）
            title = a.get("title", "").strip() or a.get_text(strip=True)
            pub_date = sp.get_text(strip=True)

            if pub_date and pub_date < CUTOFF_DATE:
                page_before += 1
                continue

            c.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                page_skip += 1
                continue

            dt, dd, content = extract_detail(href)
            ft = dt or title
            fd = dd or pub_date

            if content is None:
                print(f"  Skip (no content): {ft[:40]}...")
                page_skip += 1
                continue

            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary) "
                "VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, href, href, ft, fd, content, ft)
            )
            page_new += 1
            total_new += 1

        conn.commit()
        print(f"Page {pg}: +{page_new} new, {page_skip} skip, {page_before} pre-cutoff")

        if page_before == len(items) and pg < MAX_PAGES:
            print("All remaining before cutoff, stopping")
            break

    conn.close()
    print(f"\nDone: {total_new} new, {total_skip} skip, {total_before} before cutoff")
    return total_new


if __name__ == "__main__":
    run()
