#!/usr/bin/env python3
"""
Crawler for 岳阳县人民政府 - 公示公告
CMS: TRS-style static HTML, GBK encoding
Site: www.yyx.gov.cn/37584/38137/index.htm
Province: 湖南 (Hunan)
"""
import os, re, json, sys, subprocess, tempfile
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE_URL = "https://www.yyx.gov.cn"
LIST_PATH = "/37584/38137"
SITE_NAME = "岳阳县-公示公告"
INCREMENTAL_DAYS = 7
MAX_PAGES = 5
PER_PAGE = 20

HEADERS = [
    "-A", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
]

BASE_CURL = ["curl", "-s", "-k", "--connect-timeout", "10", "--max-time", "20"] + HEADERS


def fetch(url):
    """Fetch URL via curl."""
    cmd = BASE_CURL + [url]
    r = subprocess.run(cmd, capture_output=True, timeout=25)
    if r.returncode != 0:
        raise Exception("curl failed: {} (stderr: {})".format(r.returncode, r.stderr[:200]))
    # Decode GBK
    return r.stdout.decode("gbk")


def fetch_raw(url):
    """Fetch URL and return raw bytes."""
    cmd = BASE_CURL + [url]
    r = subprocess.run(cmd, capture_output=True, timeout=25)
    if r.returncode != 0:
        raise Exception("curl failed: {}".format(r.returncode))
    return r.stdout


def parse_list(html):
    """Parse list page, return list of (title, url, date)."""
    items = []
    # Find all anchor tags that link to content pages
    # Pattern: <a href="content_XXXX.html" ...>Title</a>
    # Some items have ../38149/content_XXXX.html (different sub-column)
    for m in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*title="?([^"]*)"?[^>]*>', html):
        href = m.group(1).strip()
        title = m.group(2).strip() if m.group(2) else ""
        
        # Only process links that go to content pages
        if not href.endswith(".html") or "content" not in href:
            continue
            
        # Skip non-content patterns
        if "javascript" in href:
            continue
        
        # Resolve URL
        if href.startswith("http"):
            full_url = href
        elif href.startswith("../"):
            # ../38149/content_XXXX.html -> /37584/38137/../38149/content_XXXX.html -> /37584/38149/content_XXXX.html
            full_url = BASE_URL + LIST_PATH + "/" + href
        elif href.startswith("/"):
            full_url = BASE_URL + href
        else:
            full_url = BASE_URL + LIST_PATH + "/" + href
        
        # Normalize the URL
        import posixpath
        parts = full_url.split("/")
        resolved = []
        for p in parts:
            if p == "..":
                if resolved:
                    resolved.pop()
            elif p != ".":
                resolved.append(p)
        full_url = "/".join(resolved)
        
        # Get date from the surrounding text (find date-like pattern near this link)
        # The list items don't have a separate date element visible
        # Title comes from title attribute
        if not title:
            title_match = re.search(r'>([^<]{10,})</a>', html[html.index(href):html.index(href)+500])
            if title_match:
                title = title_match.group(1).strip()
        
        # We'll get date from detail page
        if title and full_url:
            items.append((title.strip(), full_url, ""))
    
    # Deduplicate by URL
    seen = set()
    unique_items = []
    for item in items:
        if item[1] not in seen:
            seen.add(item[1])
            unique_items.append(item)
    
    return unique_items


def parse_detail(html, url):
    """Parse detail page content."""
    soup = BeautifulSoup(html, "html.parser")
    
    # Title from div.title
    title = ""
    title_div = soup.find("div", class_="title")
    if title_div:
        title = title_div.get_text(strip=True)
    
    # Date from div.desc span
    pub_date = ""
    desc_div = soup.find("div", class_="desc")
    if desc_div:
        spans = desc_div.find_all("span")
        for s in spans:
            text = s.get_text(strip=True)
            m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', text)
            if m:
                pub_date = m.group(1)
                break
    
    # Also check meta tags
    if not pub_date:
        for meta_name in ["PubDate", "publishdate", "dc.date"]:
            meta = soup.find("meta", attrs={"name": meta_name})
            if meta and meta.get("content"):
                m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', meta["content"])
                if m:
                    pub_date = m.group(1)
                    break
    
    # Content from div.content-wrapper
    content_html = ""
    content_wrapper = soup.find("div", class_="content-wrapper")
    if content_wrapper:
        content_html = content_wrapper.decode_contents()
    
    # Attachments: file links in content
    attachments = []
    if content_html:
        csoup = BeautifulSoup(content_html, "html.parser")
        for a_tag in csoup.find_all("a", href=True):
            ahref = a_tag["href"].strip()
            if any(ext in ahref.lower() for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
                if not ahref.startswith("http"):
                    if ahref.startswith("/"):
                        ahref = BASE_URL + ahref
                    elif ahref.startswith("../"):
                        # Normalize
                        import posixpath
                        parts = url.split("/")
                        while ahref.startswith("../"):
                            ahref = ahref[3:]
                            if parts:
                                parts.pop()
                        parts[-1] = ahref
                        ahref = "/".join(parts)
                    else:
                        ahref = url.rsplit("/", 1)[0] + "/" + ahref
                attachments.append({
                    "title": a_tag.get_text(strip=True) or "附件",
                    "url": ahref
                })
    
    return {
        "title": title,
        "date": pub_date,
        "content": content_html,
        "attachments": attachments,
        "source_url": url,
    }


def render_content(content_html, title, url):
    """Convert HTML to readable text with paragraphs and tables."""
    if not content_html or len(content_html.strip()) < 50:
        return "[{}]({})".format(title, url)
    
    soup = BeautifulSoup(content_html, "html.parser")
    parts = []
    
    for el in soup.children:
        if el.name is None:
            text = el.strip()
            if text:
                parts.append(text)
        elif el.name == "p":
            text = el.get_text(" ", strip=True)
            if text:
                parts.append(text)
        elif el.name == "table":
            # Preserve as HTML table
            table_html = str(el)
            parts.append("<table>" + table_html[table_html.index(">"):].rsplit("</table>", 1)[0] + "</table>")
        elif el.name == "div":
            text = el.get_text(" ", strip=True)
            if text:
                parts.append(text)
        elif el.name == "a":
            ahref = el.get("href", "")
            atext = el.get_text(strip=True)
            if atext and ahref and not ahref.startswith("javascript"):
                if any(ext in ahref.lower() for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
                    parts.append("[{}]({})".format(atext, ahref))
                elif ahref.startswith("http"):
                    parts.append("[{}]({})".format(atext, ahref))
                else:
                    parts.append(atext)
    
    result = "\n\n".join(p for p in parts if p).strip()
    result = re.sub(r"\n{3,}", "\n\n", result)
    
    if len(result.strip()) < 20:
        return "[{}]({})".format(title, url)
    return result


def is_incremental():
    return len(sys.argv) > 1 and sys.argv[1] == "incremental"


def main():
    incremental = is_incremental()
    print("[{}] Starting crawl (incremental={}, max_pages={})".format(SITE_NAME, incremental, MAX_PAGES))
    
    # Fetch list pages
    all_items = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            page_url = "{}{}/index.htm".format(BASE_URL, LIST_PATH)
        else:
            # Page 2 = index_1.htm, Page 3 = index_2.htm, etc.
            page_url = "{}{}/index_{}.htm".format(BASE_URL, LIST_PATH, page - 1)
        
        print("  Fetching page {}: {}".format(page, page_url))
        try:
            html = fetch(page_url)
        except Exception as e:
            print("    ERROR: {}".format(e))
            continue
        items = parse_list(html)
        print("    Found {} items".format(len(items)))
        if not items:
            break
        all_items.extend(items)
    
    print("\n  Total items from list: {}".format(len(all_items)))
    if not all_items:
        print("  No items found, exiting")
        return
    
    # Incremental filter
    if incremental:
        cutoff_date = datetime.now() - timedelta(days=INCREMENTAL_DAYS)
        print("  Incremental mode: cutoff = {}".format(cutoff_date.date()))
        # Need detail dates for filtering - fetch first
        filtered = []
        for idx, (title, url, _) in enumerate(all_items):
            try:
                html = fetch(url)
                detail = parse_detail(html, url)
                d = detail["date"]
                if d:
                    try:
                        dt = datetime.strptime(d, "%Y-%m-%d")
                        if dt >= cutoff_date:
                            filtered.append((title, url, d))
                    except ValueError:
                        filtered.append((title, url, d))
            except Exception:
                pass
        all_items = filtered
        print("  After incremental filter: {} items".format(len(all_items)))
    else:
        # Full mode: just note dates for all
        pass
    
    # Fetch details
    results = []
    for idx, (title, url, date_from_list) in enumerate(all_items):
        print("  [{}/{}] Fetching: {}...".format(idx + 1, len(all_items), title[:40]))
        try:
            html = fetch(url)
        except Exception as e:
            print("    ERROR: {}".format(e))
            results.append({
                "title": title, "date": date_from_list,
                "content": "[{}]({})".format(title, url),
                "attachments": [], "source_url": url,
            })
            continue
        
        detail = parse_detail(html, url)
        if not detail["title"]:
            detail["title"] = title
        if not detail["date"]:
            detail["date"] = date_from_list
        
        detail["content"] = render_content(detail["content"], detail["title"], url)
        results.append(detail)
    
    print("\n  Total valid items: {}".format(len(results)))
    if not results:
        print("  No items to import, exiting")
        return
    
    # Write JSONL
    tmp = tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, dir="/tmp")
    label = "{} ({}条)".format(SITE_NAME, len(results))
    for r in results:
        line = {
            "title": r["title"],
            "page_url": r["source_url"],
            "content": r["content"],
            "publish_date": r["date"],
            "site_name": SITE_NAME,
            "source_url": r["source_url"],
            "attachments": r.get("attachments", []),
        }
        tmp.write(json.dumps(line, ensure_ascii=False) + "\n")
    tmp.close()
    print("  JSONL written: {}".format(tmp.name))
    
    # Import
    ret = os.system("python3 /root/gov_crawler/import_jsonl.py '{}' '{}'".format(tmp.name, label))
    if ret == 0:
        print("  OK Import complete")
    else:
        print("  FAIL Import failed (exit={})".format(ret))
        sys.exit(1)
    
    try:
        os.unlink(tmp.name)
    except OSError:
        pass  # import_jsonl.py 已消费并删除


def _date_filter(date_str, cutoff):
    try:
        d = datetime.strptime(date_str, "%Y-%m-%d")
        return d >= cutoff
    except ValueError:
        return True


if __name__ == "__main__":
    main()
