#!/usr/bin/env python3
import os
"""Crawl 东华能源股份有限公司 - 公司要闻"""
import sys, re, os, requests
from datetime import datetime, timedelta
import sqlite3

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "东华能源-公司要闻"
BASE_URL = "http://www.chinadhe.com"
LIST_URL = BASE_URL + "/xwzx/gsyw/list.html"
DETAIL_URL = BASE_URL + "/xwzx/gsyw/detail%d.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

_month_map = {
    '1': '01', '2': '02', '3': '03', '4': '04', '5': '05', '6': '06',
    '7': '07', '8': '08', '9': '09', '10': '10', '11': '11', '12': '12',
    '01': '01', '02': '02', '03': '03', '04': '04', '05': '05', '06': '06',
    '07': '07', '08': '08', '09': '09',
}

def parse_date_cn(s):
    """Parse Chinese date like '2026年5月21日' -> '2026-05-21'"""
    m = re.search(r'(\d{4})\s*年\s*(\d{1,2})\s*月\s*(\d{1,2})\s*日', s)
    if m:
        y, mo, d = m.group(1), _month_map.get(m.group(2), m.group(2).zfill(2)), m.group(3).zfill(2)
        return "%s-%s-%s" % (y, mo, d)
    return ""

def unescape_js(s):
    """Unescape JS string: \" -> ", \\n -> \n (newlines are literal in the HTML)"""
    s = s.replace('\\"', '"')
    s = s.replace("\\'", "'")
    return s

def get_list():
    """Fetch list page and extract newsData JS array, return list of dicts"""
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print("  List page error: %s" % e, flush=True)
        return []

    # Extract all items from JS array using regex
    # Use flexible pattern to handle escaped quotes (\" ) in titles/summaries
    pattern = r'\{\s*id:\s*(\d+),\s*title:\s*"((?:[^"\\]|\\.)*)",\s*summary:\s*"((?:[^"\\]|\\.)*)",\s*date:\s*"([^"]*)",\s*views:\s*(\d+),\s*image:\s*"([^"]*)"\s*\}'
    items = []
    for m in re.finditer(pattern, html):
        item = {
            'id': int(m.group(1)),
            'title': unescape_js(m.group(2).strip()),
            'date_raw': m.group(4).strip(),
        }
        date_str = parse_date_cn(item['date_raw'])
        if date_str and date_str >= CUTOFF:
            item['date'] = date_str
            items.append(item)
    return items

def extract_content(html):
    """Extract content from detail page"""
    # 1. Newer articles: <div class="txtinfos" id="ContentBody">...</div></div>
    m = re.search(r'<div\s+class="txtinfos"\s+id="ContentBody"[^>]*>\s*(.*?)\s*</div>\s*</div>', html, re.DOTALL)
    if m:
        return m.group(1).strip()
    # 2. Older articles: <div class="news-content">..direct <p> tags..</div></div></div>
    m2 = re.search(r'<div\s+class="news-content"[^>]*>\s*(.*?)\s*</div>\s*</div>\s*</div>', html, re.DOTALL)
    if m2:
        return m2.group(1).strip()
    return ""

def crawl_detail(item):
    """Fetch detail page and extract title + content"""
    url = DETAIL_URL % item['id']
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print("  Detail error id=%d: %s" % (item['id'], e), flush=True)
        return None, ""

    # Title from <h1 class="news-title"> (clean, no prefix)
    title = ""
    m = re.search(r'<h1\s+class="news-title"[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', title).strip()
        # Also unescape if needed (detail page titles shouldn't have JS escapes)
        title = title.replace('\\"', '"').replace("\\'", "'")

    content = extract_content(html)
    return title, content

def crawl():
    print("=== %s ===" % SITE_NAME, flush=True)
    print("Cutoff: %s" % CUTOFF, flush=True)

    items = get_list()
    print("Items in 3yr range: %d" % len(items), flush=True)

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    c = conn.cursor()

    inserted = 0
    empty = 0
    for idx, item in enumerate(items):
        title, content = crawl_detail(item)
        if not content:
            empty += 1
            print("  EMPTY id=%d: %s | %s" % (item['id'], item['title'], item['date']), flush=True)
            continue
        if not title:
            title = item['title']
        url = DETAIL_URL % item['id']
        c.execute("""INSERT OR REPLACE INTO gov_raw (source_url, page_url, site_name, title, content, summary, publish_date, category, script_name) VALUES (?,?,?,?,?,?,?,?, 'crawl_chinadhe.py')""", (
            url, url, SITE_NAME, title, content,
            item['title'] if item['title'] != title else "",
            item['date'], 'chinadhe'
        ))
        inserted += 1
        if (idx + 1) % 10 == 0:
            conn.commit()
            print("  Progress: %d/%d" % (idx + 1, len(items)), flush=True)

    conn.commit()
    conn.close()
    print("Done: %d inserted, %d empty" % (inserted, empty), flush=True)

if __name__ == "__main__":
    crawl()
