#!/usr/bin/env python3
import os
"""Crawl 山西三强新能源科技有限公司 - 新闻动态"""
import sys, re, json, time, sqlite3
from datetime import datetime, timedelta
from urllib.request import urlopen, Request
from urllib.error import HTTPError, URLError

BASE = "http://www.sqtanhei.com"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "山西三强新能源-新闻"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch(url, retries=3):
    for i in range(retries):
        try:
            with urlopen(url, timeout=15) as resp:
                return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
    return ""

def extract_list_articles(html):
    entries = re.findall(r'<a[^>]*href="(/html/94733-\d+\.html)"[^>]*>(.*?)</a>', html, re.DOTALL)
    seen = {}
    for href, link_text in entries:
        if href not in seen:
            # Try to get date from link text
            date_m = re.search(r'发布时间：(\d{4}-\d{2}-\d{2})', link_text)
            date = date_m.group(1) if date_m else ""
            seen[href] = date
    # Return as list of (url, date) pairs
    return list(seen.items())

def extract_detail(html):
    title = ""
    m = re.search(r'<h2[^>]*>(.*?)</h2>', html, re.DOTALL)
    if m: title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html)
        if m: title = m.group(1).strip()
    # Date
    date = ""
    m = re.search(r'(\d{4}-\d{2}-\d{2})\s*\d{2}:\d{2}:\d{2}', html)
    if m: date = m.group(1)
    # Content
    content = ""
    m = re.search(r'<div[^>]*class="article-content"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: content = m.group(1).strip()
    # Fix relative image URLs
    if content:
        def _fix_src(m):
            tag = m.group(0)
            s = re.search(r'src="([^"]*)"', tag)
            if s:
                src = s.group(1)
                if src.startswith("http://") or src.startswith("https://") or src.startswith("data:"):
                    return tag
                ab = (BASE.rstrip("/") + "/" + src.lstrip("/")) if not src.startswith("/") else (BASE.rstrip("/") + src)
                tag = tag.replace('src="' + src + '"', 'src="' + ab + '"')
            return tag
        content = re.sub(r'<img[^>]*>', _fix_src, content)
    return title, date, content

def main():
    db = sqlite3.connect(DB_PATH, timeout=60)
    total_inserted = 0
    total_processed = 0
    
    # Crawl pages 1-4 (known total)
    for page in range(1, 5):
        if page == 1:
            url = BASE + "/html/94722.html"
        else:
            url = BASE + f"/html/94722-{page}-43592.html"
        
        html = fetch(url)
        if not html:
            print(f"Page {page}: fetch failed")
            continue
        
        articles = extract_list_articles(html)
        if not articles:
            print(f"Page {page}: no articles found")
            continue
        
        print(f"Page {page}: {len(articles)} articles")
        
        for href, list_date in articles:
            detail_url = BASE + href
            
            # Check if already exists
            exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (detail_url,)).fetchone()
            if exists:
                total_processed += 1
                continue
            
            detail_html = fetch(detail_url)
            if not detail_html:
                print(f"  Skip {href}: fetch failed")
                total_processed += 1
                continue
            
            title, date, content = extract_detail(detail_html)
            if not date:
                date = list_date
            if not date:
                date = ""
            
            # 3-year cutoff check
            if date and date < CUTOFF:
                total_processed += 1
                continue
            
            if not title:
                title = f"Article {href}"
            if not content:
                content = f"<p>{title}</p>"
            
            summary = re.sub(r'<[^>]+>', "", content)[:200].strip()
            
            # Generate numeric id from href
            m = re.search(r'/(\d+)\.html', href)
            record_id = int(m.group(1)) if m else hash(href) % (2**31)
            
            try:
                db.execute(
                    "INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (record_id, title, content, date, detail_url, detail_url, SITE_NAME, summary)
                )
                total_inserted += 1
            except Exception as e:
                print(f"  DB Error {href}: {e}")
            
            time.sleep(0.3)
            
            total_processed += 1
        
        db.commit()
    
    db.close()
    print("Done! {total_inserted} new records inserted, {total_processed} processed")

if __name__ == "__main__":
    main()
