#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Crawler for 日照市岚山区人民政府 - 环境环评
CMS: TRS WCM (拓尔思), jPage AJAX
List: Inline datastore (45 records), pagination via /module/web/jpage/dataproxy.jsp
Detail: meta ArticleTitle/PubDate, content from body
"""
import requests, re, subprocess, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "日照市岚山区-环境环评"
INDUSTRY = "环评公示"
BASE_URL = "http://www.rzlanshan.gov.cn"
LIST_URL = "http://www.rzlanshan.gov.cn/col/col227795/index.html?number=c29c29c34"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def esc(s):
    return (s or "").replace("'", "''")


def extract_items_from_page(html):
    """Extract list items from the inline datastore"""
    items = []
    records = re.findall(r'<record><!\[CDATA\[(.+?)\]\]></record>', html, re.DOTALL)
    for rec in records:
        rec_soup = BeautifulSoup(rec, 'html.parser')
        a = rec_soup.find('a')
        if a:
            href = a.get('href', '')
            title = a.get('title', '') or a.get_text(strip=True)
            span = a.find('span')
            if span:
                title = span.get_text(strip=True)
            date_i = a.find('i')
            date_str = date_i.get_text(strip=True) if date_i else ''
            full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
            if title and full_url:
                items.append((title, full_url, date_str))
    return items


def parse_detail(url):
    """Parse detail page for title, date, content"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, None, None
    except Exception as e:
        print("  [ERR] %s" % e, flush=True)
        return None, None, None

    soup = BeautifulSoup(r.text, 'html.parser')

    # Title - from h1
    h1 = soup.find('h1')
    title = h1.get_text(strip=True) if h1 else ""

    # Date from meta or text
    date_str = ""
    for meta in soup.find_all('meta'):
        name = meta.get('name', '').lower()
        if 'pubdate' in name or 'publishdate' in name:
            content = meta.get('content', '')
            if content:
                date_str = content.strip()[:10]
                break
    if not date_str:
        m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', r.text)
        if m:
            date_str = m.group(1).replace('/', '-')

    # Content - find the main article area
    content = ""
    # Helper: extract text preserving paragraph breaks, no spaces between inline spans
    def extract_paras(el):
        # Get <p> tags if available
        ps = el.find_all('p')
        if ps:
            paras = []
            for p in ps:
                txt = p.get_text(separator='', strip=True)
                if txt and len(txt) > 5:
                    paras.append(txt)
            if paras:
                result = '\n\n'.join(paras)
                # Remove leftover nav/footer noise
                for noiz in ['扫一扫在手机打开当前页', '【打印本页】', '【关闭窗口】']:
                    result = result.replace(noiz, '')
                return result.strip()
        return ''

    # Try article_text class
    el = soup.find('div', class_='article_text')
    if el:
        content = extract_paras(el)

    if not content:
        # Try show_nr class
        el = soup.find('div', class_='show_nr')
        if el:
            meta_table = el.find('table')
            if meta_table:
                meta_table.decompose()
            content = extract_paras(el)

    if not content:
        # Specific content id
        for cid in ['UCAP_CONTENT', 'Zoom', 'content']:
            el = soup.find(id=cid)
            if el:
                content = extract_paras(el)
                if content:
                    break

    if not content:
        # Fallback: body text after title
        body = soup.find('body')
        if body:
            body_text = body.get_text(separator='', strip=True)
            if title and title in body_text:
                parts = body_text.split(title, 1)
                if len(parts) > 1:
                    content = parts[1].strip()

    # Final cleanup
    if content:
        for noiz in ['扫一扫在手机打开当前页', '【打印本页】', '【关闭窗口】', '字体：【大中小】']:
            content = content.replace(noiz, '')
        date_line = "发布日期：" + date_str
        if date_line in content:
            content = content.split(date_line, 1)[-1].strip()
        content = re.sub(r'信息来源：\S+', '', content)
        content = re.sub(r'浏览次数：\d+', '', content)
        content = content.strip()
    if not content:
        content = title

    return title, date_str, content


def insert_one(page_url, title, content, pub_date):
    summary = (content or "")[:200]
    sql = """INSERT INTO gov_raw (page_url, title, publish_date, content, site_name, industry, summary)
VALUES ('%(u)s','%(t)s','%(d)s','%(c)s','%(s)s','%(i)s','%(sum)s')""" % {
        'u': esc(page_url), 't': esc(title), 'd': esc(pub_date),
        'c': esc(content), 's': SITE_NAME, 'i': INDUSTRY, 'sum': esc(summary)
    }
    result = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql],
        capture_output=True, text=True, timeout=10,
    )
    if result.returncode != 0 and "UNIQUE" not in result.stderr:
        print("  [DB ERROR] %s" % result.stderr, file=sys.stderr)
        return False
    # FTS
    fts_sql = """INSERT INTO gov_search (rowid, title, site_name, summary)
SELECT r.rowid, r.title, r.site_name, r.summary
FROM gov_raw r LEFT JOIN gov_search s ON r.rowid = s.rowid
WHERE r.page_url='%(u)s' AND s.rowid IS NULL""" % {'u': esc(page_url)}
    subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, fts_sql], capture_output=True, text=True, timeout=30)
    return True


def crawl():
    max_pages = 5
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            pass

    print("[%s] Fetching list page..." % SITE_NAME, flush=True)
    r = requests.get(LIST_URL, headers=HEADERS, timeout=20)
    r.encoding = 'utf-8'
    if r.status_code != 200:
        print("[ERR] List page HTTP %d" % r.status_code, file=sys.stderr)
        return

    items = extract_items_from_page(r.text)
    print("Found %d items in datastore" % len(items), flush=True)

    # Limit to max_pages (15 per page)
    limit = max_pages * 15
    items = items[:limit]
    print("Crawling first %d items (max_pages=%d)" % (len(items), max_pages), flush=True)

    new_count = 0
    dup_count = 0
    for i, (title, url, date_str) in enumerate(items):
        print("  [%d/%d] %s..." % (i+1, len(items), title[:40]), flush=True)
        detail_title, detail_date, content = parse_detail(url)
        final_title = detail_title or title
        final_date = detail_date or date_str
        if insert_one(url, final_title, content, final_date):
            new_count += 1
        else:
            dup_count += 1
        time.sleep(0.3)

    print("\nDone! New: %d, Duplicates: %d" % (new_count, dup_count), flush=True)


if __name__ == "__main__":
    crawl()
