#!/usr/bin/env python3
"""ww.gov.cn - 无为市基层政务公开 建设项目环评文件审批 爬虫"""
import re, json, sys, time, urllib.parse
from datetime import datetime
from html.parser import HTMLParser

import requests

BASE_URL = "https://www.ww.gov.cn"
LIST_URL = BASE_URL + "/grassroots/column/6603381?catId=1003011"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

session = requests.Session()
session.headers.update(HEADERS)


def get_page_soup(url, timeout=30):
    """Fetch page and return HTML text"""
    resp = session.get(url, timeout=timeout)
    resp.encoding = resp.apparent_encoding
    if resp.status_code != 200:
        print(f"WARN: {url} returned {resp.status_code}", file=sys.stderr)
        return None
    return resp.text


def parse_list_page(html):
    """Extract article links and dates from list page"""
    items = []
    # Find li.clearfix items that contain article links
    # Pattern: <li class="clearfix">...<a class="title" href=".../6603381/NUM.html">TITLE</a>...<span class="date">DATE</span>...</li>
    # Find all li.clearfix that contain article links
    # Note: href comes BEFORE class="title" in this site's markup
    for m in re.finditer(
        r'<li[^>]*class="[^"]*clearfix[^"]*"[^>]*>.*?'
        r'<a[^>]*href="([^"]*?/openness/grassroots/6603381/\d+\.html?)"[^>]*class="[^"]*title[^"]*"[^>]*>'
        r'(.*?)</a>.*?'
        r'<span[^>]*class="[^"]*date[^"]*"[^>]*>(.*?)</span>',
        html, re.DOTALL
    ):
        url = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        date = m.group(3).strip()
        if url and title:
            items.append({"url": url, "title": title, "date": date})
    return items


def parse_detail(html, page_url):
    """Extract title, date, source, content, attachments"""
    result = {"page_url": page_url, "title": "", "publish_date": "", "source": "", "content": "", "attachments": []}

    # Title from <title>
    m = re.search(r'<title>(.*?)</title>', html)
    if m:
        title = m.group(1).strip()
        # Remove suffix like " - 芜湖市政务公开平台"
        title = re.sub(r'[_\-—]\s*芜湖市政务公开平台\s*$', '', title).strip()
        result["title"] = title

    # Date from meta PubDate
    m = re.search(r'PubDate.*?content="([^"]+)"', html)
    if m:
        result["publish_date"] = m.group(1).strip()[:10]

    # Source from meta ContentSource
    m = re.search(r'ContentSource.*?content="([^"]+)"', html)
    if m:
        result["source"] = m.group(1).strip()

    # Fallback: rendered info
    if not result["publish_date"]:
        m = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{2}[-/]\d{2})', html)
        if m:
            result["publish_date"] = m.group(1).strip()
    if not result["source"]:
        m = re.search(r'信息来源[：:]\s*([^<\s][^<]{2,60}?)(?:</|&nbsp;|\s{2,})', html)
        if m:
            result["source"] = m.group(1).strip()

    # Content area
    content_block = ""
    idx = html.find('class="j-fontContent gkwz_contnet"')
    if idx < 0:
        idx = html.find("gkwz_contnet")
    if idx >= 0:
        # Find the opening div
        start = html.rfind("<", 0, idx)
        # Find matching closing div
        depth = 1
        j = html.find(">", idx)
        while depth > 0 and j < len(html):
            j += 1
            if html[j:j+4] == "<!--":
                k = html.find("-->", j)
                if k >= 0:
                    j = k + 2
            elif html[j:j+6] == "</div>":
                depth -= 1
            elif html[j:j+5] == "<div " or html[j:j+4] == "<div":
                k = html.find(">", j)
                if k >= 0 and k > j and html[k-1:k+1] != "/>" and html[k-1] != "/":
                    depth += 1
        content_block = html[start:j+6]
    elif html.find("xxgkcontent") >= 0:
        idx = html.find("xxgkcontent")
        start = html.rfind("<", 0, idx)
        depth = 1
        j = html.find(">", idx)
        while depth > 0 and j < len(html):
            j += 1
            if html[j:j+4] == "<!--":
                k = html.find("-->", j)
                if k >= 0:
                    j = k + 2
            elif html[j:j+6] == "</div>":
                depth -= 1
            elif html[j:j+5] == "<div " or html[j:j+4] == "<div":
                k = html.find(">", j)
                if k >= 0 and k > j and html[k-1:k+1] != "/>" and html[k-1] != "/":
                    depth += 1
        content_block = html[start:j+6]

    if not content_block:
        print(f"WARN: No content block found in {page_url}", file=sys.stderr)
        return result

    # Extract attachments BEFORE modifying DOM
    attachments = []
    for m in re.finditer(
        r'href="([^"]*\.(?:doc|pdf|docx|xls|xlsx|zip|rar)[^"]*)"',
        content_block, re.DOTALL
    ):
        att_url = m.group(1).strip()
        if att_url.startswith("/"):
            att_url = BASE_URL + att_url
        elif not att_url.startswith("http"):
            att_url = urllib.parse.urljoin(page_url, att_url)
        attachments.append(att_url)

    # Convert attachment links to markdown format
    text_block = content_block

    # Replace <a> that link to attachments with markdown links
    def replace_attach_link(m):
        full = m.group(0)
        a_url = m.group(1)
        a_text = m.group(2)
        a_text = re.sub(r'<[^>]+>', '', a_text).strip()
        if not a_text:
            # Extract filename from URL
            fn = a_url.rstrip("/").split("/")[-1]
            a_text = fn
        # Clean up the text
        a_text = re.sub(r'\s+', ' ', a_text)
        return f'<p><a href="{a_url}">{a_text}</a></p>'

    text_block = re.sub(
        r'<a\s[^>]*href="([^"]*\.(?:doc|pdf|docx|xls|xlsx|zip|rar)[^"]*)"[^>]*>(.*?)</a>',
        replace_attach_link, text_block, flags=re.DOTALL
    )

    # Now extract clean text with tables preserved
    # Keep <table> HTML, strip inline styling from other tags
    # First, preserve tables by marking them
    tables = []
    def save_table(m):
        tables.append(m.group(0))
        return f'[[TABLE_{len(tables)-1}]]'

    text_block = re.sub(r'<table[^>]*>.*?</table>', save_table, text_block, flags=re.DOTALL)

    # Strip all remaining HTML tags (but keep their text)
    # Remove inline styles and attributes first
    text_block = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', text_block, flags=re.DOTALL)
    # Keep line breaks from <br>, <p>, <div>
    text_block = re.sub(r'<br\s*/?>', '\n', text_block)
    text_block = re.sub(r'</p>', '\n\n', text_block)
    text_block = re.sub(r'</div>', '\n', text_block)
    text_block = re.sub(r'</tr>', '\n', text_block)
    text_block = re.sub(r'</td>', ' | ', text_block)
    text_block = re.sub(r'</th>', ' | ', text_block)
    # Strip all other tags
    text_block = re.sub(r'<[^>]+>', '', text_block)
    # Decode HTML entities
    text_block = text_block.replace('&nbsp;', ' ').replace('&lt;', '<').replace('&gt;', '>').replace('&amp;', '&')
    text_block = text_block.replace('&quot;', '"').replace('&#160;', ' ')
    # Clean up whitespace
    text_block = re.sub(r'[ \t]+', ' ', text_block)
    text_block = re.sub(r'\n{3,}', '\n\n', text_block)
    text_block = text_block.strip()

    # Restore tables as HTML
    for i, tbl in enumerate(tables):
        text_block = text_block.replace(f'[[TABLE_{i}]]', '\n' + tbl + '\n')

    # Filter template metadata lines
    filtered_lines = []
    skip_patterns = [
        r'^(责任编辑|初审|复审|终审|审核|编辑|发布|录入|校对|打印|关闭|分享|纠错|字体|大\s*中\s*小)\s*[：:【\[]',
        r'^\s*$',
        r'^\[.*?纠错.*?\]',
        r'^.*阅读次数',
        r'^.*字号.*大.*中.*小',
    ]
    for line in text_block.split('\n'):
        line_stripped = line.strip()
        if not line_stripped:
            continue
        skip = False
        for pat in skip_patterns:
            if re.search(pat, line_stripped):
                skip = True
                break
        if not skip:
            filtered_lines.append(line)

    result["content"] = '\n\n'.join(filtered_lines).strip()
    result["attachments"] = attachments

    return result


def crawl_list_pages(start_page=1, max_pages=5):
    """Crawl list pages and return all items"""
    all_items = []

    for page in range(start_page, start_page + max_pages):
        if page == 1:
            url = LIST_URL
        else:
            url = LIST_URL + f"&pageIndex={page}"

        print(f"List page {page}: {url}", file=sys.stderr)
        html = get_page_soup(url)
        if not html:
            print(f"Failed to fetch list page {page}", file=sys.stderr)
            continue

        items = parse_list_page(html)
        print(f"  Found {len(items)} items", file=sys.stderr)
        all_items.extend(items)
        time.sleep(1)

    return all_items


def crawl_detail(item):
    """Crawl a single detail page"""
    url = item["url"]
    html = get_page_soup(url)
    if not html:
        print(f"  FAILED: {url}", file=sys.stderr)
        return None

    detail = parse_detail(html, url)
    detail["title"] = detail["title"] or item["title"]
    if not detail["publish_date"]:
        detail["publish_date"] = item.get("date", "")

    return detail


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("mode", nargs="?", default="full", choices=["full", "incremental", "test"])
    parser.add_argument("--start-page", type=int, default=1)
    parser.add_argument("--max-pages", type=int, default=5)
    args = parser.parse_args()

    if args.mode == "test":
        # Test a single detail page
        test_url = "https://www.ww.gov.cn/openness/grassroots/6603381/41186323.html"
        html = get_page_soup(test_url)
        if html:
            detail = parse_detail(html, test_url)
            print(json.dumps(detail, ensure_ascii=False, indent=2)[:3000])
        return

    max_pages = args.max_pages
    if args.mode == "incremental":
        max_pages = 1

    items = crawl_list_pages(args.start_page, max_pages)
    print(f"Total items to crawl: {len(items)}", file=sys.stderr)

    for i, item in enumerate(items):
        print(f"[{i+1}/{len(items)}] {item['title'][:50]}...", file=sys.stderr)
        detail = crawl_detail(item)
        if detail:
            output = {
                "title": detail["title"],
                "page_url": detail["page_url"],
                "content": detail["content"],
                "publish_date": detail["publish_date"],
                "source": detail.get("source", ""),
                "attachments": detail["attachments"],
                "site_name": "无为市人民政府",
            }
            print(json.dumps(output, ensure_ascii=False))
        time.sleep(0.5)


if __name__ == "__main__":
    main()
