#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
铅山县生态环境局 - 公告公示 (gggs)
==================================
http://www.jxyanshan.gov.cn/jxyanshan/gsgg/list.shtml
UCAP CMS (createPageHTML 分页, _N.shtml)
列表: ul.doc_list > li > span.date + a[href="/ysx[a-z]+/gggs/YYYYMM/{uuid}.shtml"] (多部门聚合)
详情: meta ArticleTitle/PubDate, 正文 div#zoomcon (p.MsoNormal + Word 表格 HTML)
附件: div.article-appendixs > ul.infoList (相对路径 urljoin 详情URL 绝对化)
"""
import sys, os, re, time, sqlite3
import warnings
warnings.filterwarnings("ignore")
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.jxyanshan.gov.cn"
LIST_URL = "http://www.jxyanshan.gov.cn/jxyanshan/gsgg/list"
SITE_NAME = "铅山县-公告公示"
GROUP_NAME = "江西"
SCRIPT_NAME = "crawl_yanshan_gsgg.py"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

# ---- args: --pages=N / --pages N / bare number ----
_MAX_PAGES = 1
argv = sys.argv[1:]
i = 0
while i < len(argv):
    a = argv[i]
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(argv):
        try:
            _MAX_PAGES = int(argv[i + 1])
        except ValueError:
            pass
        i += 1
    elif a.isdigit():
        _MAX_PAGES = int(a)
    i += 1


def fetch(url, retries=3, timeout=30):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=timeout)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            print("  [ERR] %s: %s" % (url[:90], e), file=sys.stderr)
            if attempt < retries - 1:
                time.sleep(2)
    return None


def clean_title(title):
    """Strip entity/space prefixes and ellipsis suffixes."""
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def extract_list_items(html):
    """Extract (title, url, date) from ul.doc_list > li items."""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    doclist = soup.find("ul", class_="doc_list")
    if not doclist:
        return items
    for li in doclist.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        # /ysxsthjj/zwdt/YYYYMM/{uuid}.shtml
        if not re.search(r"/ysx[a-z]+/gggs/\d{6}/[a-f0-9]+\.shtml", href):
            continue
        full_url = urljoin(BASE_URL, href)
        title = a.get("title") or a.get_text(strip=True)
        if not title:
            continue
        # date from span.right.date
        date = ""
        sp = li.find("span", class_=re.compile(r"date"))
        if sp:
            m = re.search(r"(\d{4})-(\d{2})-(\d{2})", sp.get_text())
            if m:
                date = "%s-%s-%s" % m.groups()
        items.append((clean_title(title), full_url, date))
    return items


def extract_detail(html, page_url):
    """Extract title, date, content (paragraphs \n\n, tables as HTML), attachments."""
    soup = BeautifulSoup(html, "html.parser")

    # 1. Title: meta ArticleTitle first, fallback zwtit / <title>
    title = ""
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_t and meta_t.get("content"):
        title = meta_t["content"]
    if not title:
        t_el = soup.find("div", class_="zwtit")
        if t_el:
            title = t_el.get_text(strip=True)
    if not title:
        tt = soup.find("title")
        if tt:
            title = tt.get_text(strip=True)
    title = clean_title(title)

    # 2. Date: meta PubDate first
    date = ""
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d and meta_d.get("content"):
        m = re.search(r"(\d{4})-(\d{2})-(\d{2})", meta_d["content"])
        if m:
            date = "%s-%s-%s" % m.groups()
    if not date:
        dm = re.search(r"(\d{4})[-年](\d{1,2})[-月](\d{1,2})[日]?", html)
        if dm:
            date = "%s-%02d-%02d" % (dm.group(1), int(dm.group(2)), int(dm.group(3)))

    # 3. Content: div#zoomcon (UCAP wraps body in <ucapcontent> custom tag)
    content_parts = []
    zoom = soup.find("div", id="zoomcon")
    if zoom:
        # remove template noise widgets
        for noise in zoom.find_all("div", class_=re.compile(r"info_ewm|qr_container|article-reldocuments")):
            noise.decompose()
        # UCAP CMS: descend into <ucapcontent>/<UCAPCONTENT> wrapper if present
        # (template has case variants — lowercase for text articles, UPPERCASE
        #  for image-only articles; must match case-insensitively)
        ucap = zoom.find(re.compile(r"^ucapcontent$", re.I))
        container = ucap if ucap else zoom
        for el in container.find_all(recursive=False):
            if el.name == "table":
                content_parts.append(str(el))
            elif el.name in ("p", "div", "h1", "h2", "h3", "li"):
                # paragraph containing attachment link -> keep HTML
                a_in = el.find("a", href=True)
                if a_in and any(x in a_in["href"].lower() for x in
                                [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".rar", ".zip", "/upload", "files/"]):
                    # absolutize hrefs inside the segment
                    seg_html = str(el)
                    for a2 in el.find_all("a", href=True):
                        if a2["href"] and not a2["href"].startswith(("http", "javascript")):
                            a2["href"] = urljoin(page_url, a2["href"])
                    content_parts.append(str(el))
                else:
                    txt = el.get_text(strip=True)
                    if txt:
                        content_parts.append(txt)

    # 4. Attachments: article-appendixs zone (absolute URLs, <p><a> format)
    appendix_html = ""
    app = soup.find("div", class_=re.compile(r"article-appendixs"))
    if app:
        for a in app.find_all("a", href=True):
            href = a["href"]
            if not href.startswith("http"):
                href = urljoin(page_url, href)
            name = a.get_text(strip=True) or href.split("/")[-1]
            appendix_html += '<p><a href="%s" target="_blank">%s</a></p>\n' % (href, name)

    # 5. Dedup paragraphs (by tag-stripped text), skip empty
    seen = set()
    parts = []
    for p in content_parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        parts.append(p)
    content = "\n\n".join(parts)

    # 6. Appendix zone appended at end (if not already inside content)
    if appendix_html and appendix_html.strip() not in content:
        content = content + "\n\n" + appendix_html.strip()

    has_table = 1 if "<table" in content else 0
    return title, content, date, has_table


def main():
    items_all = []
    seen_urls = set()
    for page in range(1, _MAX_PAGES + 1):
        url = "%s.shtml" % LIST_URL if page == 1 else "%s_%d.shtml" % (LIST_URL, page)
        html = fetch(url)
        if not html:
            print("  [SKIP] Page %d failed" % page)
            break
        items = extract_list_items(html)
        print("Page %d: %d items" % (page, len(items)))
        for it in items:
            if it[1] not in seen_urls:
                seen_urls.add(it[1])
                items_all.append(it)
        time.sleep(0.3)
        # end condition: last page has fewer items than normal page size
        # (15/页 for this column; use loose threshold so full pages never break)
        if page > 1 and len(items) < 5:
            break

    print("Total unique items: %d" % len(items_all))

    SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()

    inserted = 0
    skipped = 0
    for i, (list_title, page_url, list_date) in enumerate(items_all):
        html = fetch(page_url)
        if not html:
            skipped += 1
            continue
        title, content, date, has_table = extract_detail(html, page_url)
        if not title:
            title = list_title
        if not date:
            date = list_date
        if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
            # empty/pure-image article -> skip
            print("  [SKIP] empty content: %s" % title[:50])
            skipped += 1
            continue
        summary = re.sub(r"<[^>]+>", "", content).strip()[:200]
        c.execute("""INSERT OR IGNORE INTO gov_raw
                     (page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, summary)
                     VALUES (?,?,?,?,?,?,?,?,?,?)""",
                  (page_url, page_url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, summary))
        if c.rowcount > 0:
            c.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                      (c.lastrowid, title, SITE_NAME, summary))
            inserted += 1
        else:
            skipped += 1
        if (i + 1) % 10 == 0:
            print("  progress %d/%d (new=%d)" % (i + 1, len(items_all), inserted))
        time.sleep(0.2)

    conn.commit()
    conn.close()
    print("新增: %d 条, 跳过: %d" % (inserted, skipped))
    print("站点: %s | 栏目: 公告公示 | 页数: %d" % (SITE_NAME, _MAX_PAGES))


if __name__ == "__main__":
    main()
