#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_lubeichem.py - 山东鲁北化工股份有限公司-环评报告
URL: http://www.lubeichem.cn/index.php?m=home&c=Lists&a=index&tid=39
CMS: EyouCMS, ?page=N 分页
列表: /index.php?m=home&c=Lists&a=index&tid=39&page=N
详情: /index.php?m=home&c=View&a=index&aid={aid}
标题: <title> tag
日期: <p>发布时间：YYYY-MM-DD HH:MM:SS</p>
正文: div.ymnr > p
"""
import requests, re, sys, os, time
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.lubeichem.cn"
LIST_URL = "/index.php?m=home&c=Lists&a=index&tid=39"
SITE_NAME = "山东鲁北化工股份有限公司-环评报告"
GROUP = "企业"
INDUSTRY = "07化工"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
SESSION = requests.Session()
SESSION.trust_env = False


def fetch(url):
    try:
        r = SESSION.get(url, headers=HEADERS, timeout=60)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("ERR:%s" % e)
        return None


def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.find_all("a", href=True):
        h = a["href"]
        t = a.get_text(strip=True)
        if "aid=" in h and len(t) > 10:
            full = BASE_URL + "/" + h.lstrip("/")
            items.append((full, t))
    return items


def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title from <title>
    title = ""
    if soup.title:
        title = soup.title.get_text(strip=True)
        # Remove trailing site name
        title = re.sub(r"_山东鲁北化工股份有限公司$", "", title).strip()

    # Date from "发布时间：YYYY-MM-DD"
    date_str = ""
    m = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        date_str = m.group(1)

    # Content from div.ymnr > p (skip first few nav p's)
    content = ""
    ymnr = soup.find(class_=lambda c: c and isinstance(c, str) and "ymnr" in c)
    if ymnr:
        parts = []
        started = False
        for p in ymnr.find_all("p"):
            txt = p.get_text(strip=True)
            if not txt:
                continue
            # Skip navigation text at top
            if not started and "首页" in txt:
                continue
            started = True
            if "发布时间" not in txt:
                parts.append(txt)
        content = "\n".join(parts)

    return title, date_str, content


def push_to_db(items):
    import subprocess
    total = 0
    for url, title, pub_date, content, site_name, script_name, industry in items:
        sql = "INSERT OR IGNORE INTO gov_raw (page_url, title, publish_date, content, site_name, script_name, industry) VALUES ('%s', '%s', '%s', '%s', '%s', '%s', '%s');" % (
            url.replace("'", "''"),
            (title or "").replace("'", "''"),
            (pub_date or "").replace("'", "''"),
            (content or "").replace("'", "''"),
            (site_name or "").replace("'", "''"),
            (script_name or "").replace("'", "''"),
            (industry or "").replace("'", "''"),
        )
        r = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, text=True, timeout=10)
        if r.returncode == 0:
            total += 1
    return total


def main():
    pages = MAX_PAGES
    if len(sys.argv) > 1:
        try:
            pages = int(sys.argv[1].replace("--pages=", ""))
        except:
            pass

    script_name = os.path.basename(__file__)
    all_items = []

    for page in range(1, pages + 1):
        if page == 1:
            url = BASE_URL + LIST_URL
        else:
            url = BASE_URL + LIST_URL + "&page=%d" % page

        print("[%d] %s..." % (page, url), end=" ")
        html = fetch(url)
        if not html:
            print("FAIL")
            continue

        links = parse_list(html)
        print("%d links" % len(links))

        for link_url, link_title in links:
            print("  %s" % link_title[:40], end="... ")
            detail_html = fetch(link_url)
            if not detail_html:
                print("ERR")
                continue
            title, pub_date, content = parse_detail(detail_html, link_url)
            if not title:
                title = link_title
            print("OK (%d字)" % len(content))
            all_items.append((link_url, title, pub_date, content, SITE_NAME, script_name, INDUSTRY))
            time.sleep(0.5)

    if not all_items:
        print("未获取到任何数据")
        return

    total = 0
    for item in all_items:
        total += push_to_db([item])

    print("\n入库: %d/%d 条" % (total, len(all_items)))

    if total > 0:
        import subprocess
        for url, _, _, _, _, _, _ in all_items[:total]:
            r = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, "SELECT rowid FROM gov_raw WHERE page_url='%s';" % url.replace("'", "''")],
                              capture_output=True, text=True, timeout=10)
            rid = r.stdout.strip()
            if rid:
                sql = "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) SELECT %s, title, site_name, substr(content,1,300) FROM gov_raw WHERE rowid=%s;" % (rid, rid)
                subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=10)
        print("FTS同步完成")

    print("\n===== 完成 =====")
    print("新增入库: %d 条" % total)
    print("站点: %s" % SITE_NAME)


if __name__ == "__main__":
    main()
