#!/usr/bin/env python3
"""湛江市生态环境局-重点领域-环评 (www.zhanjiang.gov.cn)
列表: /zdlyxxgk/sthj/jsxmhjyx/index.html -> index_N.html
       注意: index_1.html = 404, page 1 = index.html
详情: /zdlyxxgk/sthj/jsxmhjyx/content/post_XXXXXX.html
     -> div.cont div.article (表格+PDF附件)

策略: full=前5页, full all=全量, incremental=第1页(日跑)
"""
import sys, os, re, json, time
from datetime import datetime, timedelta
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.zhanjiang.gov.cn"
LIST_PATH = "/zdlyxxgk/sthj/jsxmhjyx"
SITE = "湛江市生态环境局"
COLUMN = "重点领域-环评"
PROVINCE = "广东"
TOTAL_PAGES = 29  # index.html(1) + index_2~49(48)
DEFAULT_FULL_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)
MARKER = "\x00P\x00"

ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx',
               '.zip', '.rar', '.7z', '.tar', '.gz', '.txt',
               'downfile.jsp', 'download.jsp', 'downfile', 'download')


def log(msg):
    sys.stderr.write(msg + "\n")
    sys.stderr.flush()


def _is_attachment(href):
    if not href:
        return False
    low = href.lower()
    return any(ext in low for ext in ATTACH_EXTS)


def fetch_list(page):
    if page == 1:
        url = f"{BASE_URL}{LIST_PATH}/index.html"
    elif page == 2:
        url = f"{BASE_URL}{LIST_PATH}/index_2.html"
    else:
        url = f"{BASE_URL}{LIST_PATH}/index_{page}.html"
    try:
        r = session.get(url, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        log(f"  [ERROR] list page {page}: {e}")
        return None


def parse_list(html):
    """Extract articles from list page.
    Structure: <li><a title="..." href=".../content/post_XXXXXX.html">
                 <span class="point"></span>title</a>
                 <span class="time">YYYY-MM-DD</span></li>
    """
    soup = BeautifulSoup(html, "html.parser")
    items = []
    art_re = re.compile(r"/content/post_\d+\.html")
    for a_tag in soup.find_all("a", href=art_re):
        href = a_tag["href"]
        title = a_tag.get_text(strip=True)
        if not title:
            # Try title attribute
            title = a_tag.get("title", "")
        if not title:
            continue
        full_url = urljoin(BASE_URL, href)

        # Date from sibling span.time
        date_str = ""
        li = a_tag.parent
        if li:
            time_span = li.find("span", class_="time")
            if time_span:
                date_str = time_span.get_text(strip=True)
            if not date_str:
                # Fallback: look for any date pattern in parent
                dm = re.search(r'(\d{4}-\d{2}-\d{2})', li.get_text(strip=True))
                if dm:
                    date_str = dm.group(1)

        items.append({"title": title, "url": full_url, "date": date_str})
    return items


def extract_content(detail_url, title):
    try:
        r = session.get(detail_url, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        return f"[{title}]({detail_url})", title, "", []

    soup = BeautifulSoup(r.text, "html.parser")

    meta_title = soup.select_one("meta[name=ArticleTitle]")
    meta_date = soup.select_one("meta[name=pubdate]")
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    pub_date = ""
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # Content: div.cont div.article
    content_el = soup.select_one("div.cont div.article")
    if not content_el:
        content_el = soup.select_one("div.article")
    if not content_el:
        return f"[{title}]({detail_url})", title, pub_date, []

    attachments = []

    # Collect ALL attachments first (before any DOM modification)
    all_attachments = []
    for a_tag in content_el.find_all("a", href=True):
        href = a_tag["href"]
        text = a_tag.get_text(strip=True)
        if _is_attachment(href):
            full_url = urljoin(BASE_URL, href)
            all_attachments.append({"name": text or "附件", "url": full_url})

    # 0) Save tables FIRST (before any replace_with that creates text nodes)
    tables_html = []
    for table in content_el.find_all("table"):
        try:
            tables_html.append(str(table))
        except Exception:
            tables_html.append(table.get_text(separator=" ", strip=True))
        table.decompose()

    # 1) Extract attachment links to markdown
    for a_tag in content_el.find_all("a", href=True):
        href = a_tag["href"]
        text = a_tag.get_text(strip=True)
        if not href or href.startswith("#") or href.startswith("javascript"):
            continue
        full_url = urljoin(BASE_URL, href)
        if _is_attachment(href):
            if not href or href.startswith("#") or href.startswith("javascript"):
                continue
            full_url = urljoin(BASE_URL, href)
            if _is_attachment(href):
                md_link = f"[{text}]({full_url})" if text else f"[附件]({full_url})"
                a_tag.replace_with(md_link)
            elif text and (href.startswith("http") or href.startswith("/")):
                md_link = f"[{text}]({full_url})"
                a_tag.replace_with(md_link)

    # 2) Paragraph markers
    for tag in content_el.find_all(["p", "h1", "h2", "h3", "h4", "h5", "h6"]):
        tag.insert(0, MARKER)
        tag.append(MARKER)
    for br in content_el.find_all("br"):
        br.replace_with(MARKER)

    # 3) Unwrap inline
    for tag in content_el.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()

    # 4) Extract text
    text = content_el.get_text(separator="", strip=True)
    text = re.sub(r"\x00P\x00(\s*\x00P\x00)+", "\x00P\x00", text)
    text = text.replace("\x00P\x00", "\n\n")
    text = re.sub(r"\n{3,}", "\n\n", text)
    text = text.strip()

    for tbl_html in tables_html:
        text += f"\n\n{tbl_html}"

    if not text.strip() or len(text.strip()) < 20:
        text = f"[{title}]({detail_url})"

    return text, title, pub_date, all_attachments


def crawl_pages(start_page, end_page, cutoff, label):
    all_items = []
    seen_urls = set()
    for page in range(start_page, end_page + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            log(f"  [{label} PAGE {page}] no items, stopping")
            break
        new_count = 0
        for item in items:
            if item["url"] not in seen_urls:
                if cutoff and item["date"]:
                    try:
                        d = datetime.strptime(item["date"], "%Y-%m-%d")
                        if d < cutoff:
                            continue
                    except ValueError:
                        pass
                seen_urls.add(item["url"])
                all_items.append(item)
                new_count += 1
        log(f"  [{label} PAGE {page}/{end_page}] {len(items)} items, +{new_count} new")
        if new_count == 0 and page > 1:
            log("  [STOP] no new items")
            break
        time.sleep(1)
    return all_items


def _process_details(all_items):
    log(f"\n共 {len(all_items)} 条待爬详情")
    for idx, item in enumerate(all_items, 1):
        content, title, pub_date, attachments = extract_content(item["url"], item["title"])
        if pub_date:
            item["date"] = pub_date
        if title:
            item["title"] = title
        item["content"] = content
        item["attachments"] = attachments
        output_item(item, idx)
        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{len(all_items)}")
    log(f"\n[DONE] 共爬取 {len(all_items)} 条")


def crawl_all(months_back=36):
    page_count = DEFAULT_FULL_PAGES
    cutoff = datetime.now() - timedelta(days=months_back * 30) if months_back else None
    log(f"策略: 前{page_count}页初始入库 (全量请用 'full all')")
    all_items = crawl_pages(1, page_count, cutoff, f"FULL 1/{page_count}")
    _process_details(all_items)


def crawl_all_pages(months_back=36):
    cutoff = datetime.now() - timedelta(days=months_back * 30) if months_back else None
    log("全量模式: 爬取所有页面...")
    all_items = crawl_pages(1, TOTAL_PAGES, cutoff, "FULL ALL")
    _process_details(all_items)


def crawl_incremental():
    html = fetch_list(1)
    if not html:
        return
    items = parse_list(html)
    log(f"[INCREMENTAL] page 1 -> {len(items)} 条")
    for idx, item in enumerate(items, 1):
        content, title, pub_date, attachments = extract_content(item["url"], item["title"])
        if pub_date:
            item["date"] = pub_date
        if title:
            item["title"] = title
        item["content"] = content
        item["attachments"] = attachments
        output_item(item, idx)
    log(f"[DONE] 增量爬取 {len(items)} 条")


def output_item(item, idx):
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        arg2 = sys.argv[2] if len(sys.argv) > 2 else ""
        if arg2 == "all":
            crawl_all_pages()
        else:
            months = int(arg2) if arg2 and arg2.isdigit() else 36
            crawl_all(months)
    elif mode == "list":
        html = fetch_list(1)
        items = parse_list(html)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:50]} | {it['url']}")
    else:
        log(f"Usage: {sys.argv[0]} [incremental|full [months|all]|list]")
