#!/usr/bin/env python3
"""
宜昌市生态环境局-环境影响评价 (hbj.yichang.gov.cn)
============================================
CMS: 宜昌市政府网站群（自研）
列表: list-42531-{page}.html, 每页15条
详情: content-42531-{id}-1.html, <div class="txtcontent-div"> 含正文

用法:
    python3 crawl_yichang_hbj.py          # 全量(最多55页+日期过滤)
    python3 crawl_yichang_hbj.py --test   # 测试 5 条
"""

import re, sys, os, time
from datetime import datetime, timedelta, timezone
import requests, urllib3
import os
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME  = "宜昌市生态环境局-环境影响评价"
LIST_URL   = "http://hbj.yichang.gov.cn/list-42531-{}.html"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 5  # ~3yr boundary
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch_list(page):
    url = LIST_URL.format(page)
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("  ! list page " + str(page) + " fail: " + str(e)[:60])
        return None

def parse_list(html):
    items = []
    for m in re.finditer(
        r'<div class="txtlisty1">.*?<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>.*?'
        r'<div class="txtlisty2">(\d{4}-\d{2}-\d{2})',
        html, re.DOTALL
    ):
        url = m.group(1)
        if not url.startswith("http"):
            url = "http://hbj.yichang.gov.cn" + url if url.startswith("/") else "http://hbj.yichang.gov.cn/" + url
        title = m.group(2).strip()
        pub_date = m.group(3)
        items.append({"title": title, "url": url, "pub_date": pub_date})
    return items

def fetch_detail(url):
    """Extract full title + article body from detail page"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        html = r.text

        # Extract full title from the second <h1> (first is site title)
        full_title = ""
        h1s = re.findall(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        for h1 in h1s:
            t = re.sub(r'<[^>]+>', '', h1).strip()
            if t and t != "宜昌市生态环境局":
                full_title = t
                break
        # Fallback: <title> tag
        if not full_title:
            m = re.search(r'<title>(.*?)</title>', html)
            if m:
                full_title = m.group(1).strip()
        # Find txtcontent-div opening
        idx = html.find("class=\"txtcontent-div\"")
        if idx < 0:
            return ""
        div_start = html.rfind("<div", 0, idx)
        if div_start < 0:
            return ""
        section = html[div_start:]
        # Find nry-info and count its closing </div>
        nry_pos = section.find("class=\"nry-info\"")
        if nry_pos < 0:
            return ""
        depth = 1
        after_nry = nry_pos
        for i in range(nry_pos, len(section)):
            if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                depth += 1
            elif section[i:i+6] == "</div>":
                depth -= 1
                if depth == 0:
                    after_nry = i + 6
                    break
        # Find txtcontent-div's closing </div>
        content_depth = 0
        content = ""
        for j in range(after_nry, len(section)):
            if section[j:j+4] == "<div" and (j+4 >= len(section) or section[j+4] in " >\n\r\t"):
                content_depth += 1
            elif section[j:j+6] == "</div>":
                if content_depth == 0:
                    content = section[after_nry:j].strip()
                    break
                content_depth -= 1
        # Clean
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        content = content.strip()
        return full_title, content
    except Exception as e:
        print("  ! detail fail: " + str(e)[:80])
        return "", ""

def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR REPLACE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, tags, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_yichang_hbj.py')",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "环境影响评价",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    db.execute(
        "INSERT INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    return ok, fail

def crawl(test=False):
    all_items = []
    seen_urls = set()
    for page in range(1, MAX_PAGES + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            print("  [Page " + str(page) + "] empty, reached end")
            break
        new = 0
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        print("  [Page " + str(page) + "] " + str(len(items)) + " items (" + f + " ~ " + l + "), new: " + str(new))
        if new == 0:
            break
    print("  Total: " + str(len(all_items)) + " items within 3yr")
    if test:
        all_items = all_items[:5]
        print("  TEST mode: " + str(len(all_items)) + " items")
    if not all_items:
        return 0, 0
    for i, item in enumerate(all_items):
        print("  [" + str(i+1) + "/" + str(len(all_items)) + "] " + item["title"][:40] + "... ", end="", flush=True)
        full_title, content = fetch_detail(item["url"])
        item["content"] = content
        if full_title:
            item["title"] = full_title
        print(str(len(content)) + "B" + ("  title_fixed" if full_title else ""))
    ok, fail = to_db(all_items)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    mode = "TEST" if test else "FULL"
    print()
    print("[" + SITE_NAME + "] " + mode)
    t0 = time.time()
    ok, fail = crawl(test=test)
    print("  Time: " + str(round(time.time()-t0, 1)) + "s")
    print("  New: " + str(ok) + " Skip: " + str(fail))
