#!/usr/bin/env python3
"""
西安中地环境科技有限公司 - 文件公示 爬虫
CMS: PHPCMS
列表: index.php?m=content&c=index&a=lists&catid=24&page=N (10条/页, 24页)
详情: index.php?m=content&c=index&a=show&catid=24&id={id}
反爬: Nginx challenge cookie (先403设cookie，带cookie重请求)
"""

import sys, os, re, time, sqlite3, subprocess, hashlib, argparse
from datetime import datetime
from urllib.parse import urljoin
import requests

BASE_URL = "http://www.zdhjkj.cn"
LIST_TPL = "http://www.zdhjkj.cn/index.php?m=content&c=index&a=lists&catid=24&page={}"
SITE_NAME = "西安中地环境科技_文件公示"
GROUP = "企业"
DB_PATH = "/mnt/data/search.db"
MAX_PAGES = 24  # 共24页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def make_session():
    """创建带反爬处理的Session"""
    s = requests.Session()
    s.headers.update(HEADERS)
    # 绕过 Nginx challenge: 第一次请求拿cookie
    first = s.get(f"{BASE_URL}/index.php?m=content&c=index&a=lists&catid=24")
    # 第二次请求带cookie可用
    return s


def parse_list(html):
    """提取 (title, url)"""
    items = []
    # PHPCMS列表: ul > li > a
    for m in re.finditer(r'<a[^>]*href="([^"]*show[^"]*id=(\d+))"[^>]*>\s*([^<]{4,}?)\s*</a>', html):
        href = m.group(1).strip()
        title = m.group(3).strip()
        title = title.replace('&nbsp;', ' ').replace('\u00a0', ' ').strip()
        if title and 'show' in href:
            full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
            items.append((title, full_url))
    return items


def parse_detail(html, url):
    """提取详情"""
    # --- 标题 ---
    title = ""
    # 1. div.ar_title > h1 (正文标题)
    m = re.search(r'<div class="ar_title">\s*<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        t = re.sub(r'<[^>]+>', '', m.group(1)).strip()
        if t:
            title = t
    # 2. <title> 标签
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if m:
            t = m.group(1).strip()
            t = re.sub(r'\s*-\s*文件公示\s*-\s*西安中地环境科技有限公司_官网$', '', t).strip()
            if t:
                title = t
    # 3. 任意<h1>（排除导航类短标题）
    if not title:
        for m in re.finditer(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL):
            t = re.sub(r'<[^>]+>', '', m.group(1)).strip()
            if len(t) >= 10 and '新闻中心' not in t and '快捷通道' not in t and '文件公示' not in t:
                title = t
                break

    title = title.replace('&nbsp;', ' ').replace('\u00a0', ' ').strip()

    # --- 日期 ---
    publish_date = ""
    m = re.search(r'发布时间[：:].*?(\d{4})[\.\-](\d{1,2})[\.\-](\d{1,2})', html)
    if m:
        publish_date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

    # --- 正文 ---
    body_text = ""
    m = re.search(r'<div class="article">(.*?)(?=</div>\s*</div>|</div>\s*<div class="[^"]*(?:clear|n_|ar_))', html, re.DOTALL)
    if not m:
        m = re.search(r'<div class="article">(.*?)</div>', html, re.DOTALL)

    if m:
        inner = m.group(1).strip()
        parts = []
        # 处理所有子元素：<p>段落、<div>链接块、<table>表格、<img>图片、<a>附件
        for m_elem in re.finditer(r'(<p[^>]*>.*?</p>|<div[^>]*>.*?</div>|<table[^>]*>.*?</table>|<a[^>]*>.*?</a>|<img[^>]*/>)', inner, re.DOTALL):
            elem_html = m_elem.group(1).strip()
            if not elem_html:
                continue

            tag = re.match(r'<(\w+)', elem_html)
            tag_name = tag.group(1) if tag else ''

            if tag_name == 'p':
                # 提取<p>内部内容（去掉外层<p>标签）再重新包装
                p_inner = re.sub(r'^<p[^>]*>|</p>$', '', elem_html).strip()
                p_text = re.sub(r'<[^>]+>', '', p_inner).strip()
                if not p_text:
                    continue
                cleaned = re.sub(r'\sstyle="[^"]*"', '', p_inner)
                # 扁平化<span>/<strong>嵌套，只保留文字
                cleaned = re.sub(r'<span[^>]*>|</span>', '', cleaned)
                cleaned = re.sub(r'<strong[^>]*>|</strong>', '', cleaned)
                parts.append(f'<p>{cleaned}</p>')

            elif tag_name == 'div':
                # 检查是否包含附件链接
                a_in_div = re.findall(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', elem_html, re.DOTALL)
                for href, text in a_in_div:
                    text = re.sub(r'<[^>]+>', '', text).strip()
                    full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
                    if 'uploadfile' in href or href.endswith(('.pdf', '.doc', '.docx', '.zip', '.rar', '.xls', '.xlsx')):
                        parts.append(f'<a href="{full_url}">{text or os.path.basename(href.split("?")[0])}</a>')
                    else:
                        parts.append(f'<a href="{full_url}">{text}</a>')
                # 普通div的文字也提取
                div_text = re.sub(r'<[^>]+>', '', elem_html).strip()
                if div_text and not a_in_div:
                    parts.append(f'<p>{div_text}</p>')

            elif tag_name == 'table':
                parts.append(elem_html)

            elif tag_name == 'a':
                href = re.search(r'href="([^"]*)"', elem_html)
                text = re.sub(r'<[^>]+>', '', elem_html).strip()
                if href:
                    full_url = href.group(1) if href.group(1).startswith('http') else urljoin(BASE_URL, href.group(1))
                    parts.append(f'<a href="{full_url}">{text or os.path.basename(href.group(1).split("?")[0])}</a>')

            elif tag_name == 'img':
                src = re.search(r'src="([^"]*)"', elem_html)
                if src:
                    parts.append(f'<img src="{src.group(1)}">')

        body_text = '\n\n'.join(p for p in parts if p.strip())

    return {
        'title': title,
        'publish_date': publish_date,
        'body_text': body_text,
    }


def safe_date(date_str):
    if not date_str:
        return ""
    m = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', str(date_str))
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        return f"{y:04d}-{mo:02d}-{d:02d}"
    return ""


def truncate_summary(text, max_len=300):
    if not text:
        return ""
    s = re.sub(r'<[^>]+>', ' ', text)
    s = re.sub(r'\s+', ' ', s).strip()
    return s[:max_len]


def crawl(max_pages=MAX_PAGES):
    """主爬取逻辑"""
    s = make_session()

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    # 检查已有
    existing_cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = set(row[0] for row in existing_cur.fetchall())

    total_new = total_skip = total_error = 0
    pages_to_crawl = min(max_pages, MAX_PAGES)
    print(f"[西安中地·文件公示] 开始爬取, 已有 {len(existing)} 条, 目标 {pages_to_crawl} 页")

    for page in range(1, pages_to_crawl + 1):
        page_url = LIST_TPL.format(page)
        print(f"  第 {page}/{pages_to_crawl} 页: {page_url}", flush=True)

        try:
            resp = s.get(page_url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code != 200:
                print(f"    HTTP {resp.status_code}, 跳过")
                total_error += 1
                continue
        except Exception as e:
            print(f"    请求失败: {e}")
            total_error += 1
            time.sleep(3)
            continue

        items = parse_list(resp.text)
        print(f"    发现 {len(items)} 条", flush=True)
        if not items:
            print("    无条目, 停止翻页")
            break

        for list_title, url in items:
            if url in existing:
                total_skip += 1
                continue

            try:
                detail_resp = s.get(url, timeout=30)
                detail_resp.encoding = 'utf-8'
                if detail_resp.status_code != 200:
                    print(f"    ⚠ {list_title[:30]}... HTTP {detail_resp.status_code}")
                    total_error += 1
                    continue
            except Exception as e:
                print(f"    ✗ {list_title[:30]}... 请求失败: {e}")
                total_error += 1
                continue

            info = parse_detail(detail_resp.text, url)
            body = info['body_text']
            pub_date = safe_date(info['publish_date'])
            summary = truncate_summary(body)

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, url, url, info['title'][:500], pub_date,
                    summary[:500], body, 'active', '', GROUP, ''
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
                    if row:
                        rid = row[0]
                        esc_title = info['title'].replace("'", "''")
                        esc_site = SITE_NAME.replace("'", "''")
                        esc_summary = summary.replace("'", "''")
                        sql = f"INSERT INTO gov_search(rowid, title, site_name, summary) VALUES({rid},'{esc_title}','{esc_site}','{esc_summary}')"
                        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=30)
                    existing.add(url)
                    db.commit()
                    total_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"    ✗ DB error: {e}")
                total_error += 1

            time.sleep(0.5)

        time.sleep(1)

    db.close()
    print(f"\n[西安中地·文件公示] 完成！新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}")
    return total_new


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='西安中地环境科技·文件公示爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES, help=f'最大页数 (默认 {MAX_PAGES})')
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
