#!/usr/bin/env python3
import requests, re, sys, os, subprocess, argparse
from bs4 import BeautifulSoup

SITE_NAME = "西安彩晶光电科技股份有限公司-新闻中心"
BASE_URL = "http://www.xacjoe.com"
HEADERS = {"User-Agent": "Mozilla/5.0"}
DB_PATH = "/mnt/data/search.db"

def fetch(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="news_ul")
    if not ul:
        return items, 0
    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a: continue
        href = a.get("href", "")
        if not href or href == "#": continue
        full_url = BASE_URL + href if href.startswith("/") else href
        title = a.get("title", "")
        if not title:
            h3 = a.find("h3")
            if h3: title = h3.get_text(strip=True)
        if not title: continue
        date_div = li.find("div", class_="in_new_text1")
        date_str = date_div.get("data-date", "") if date_div else ""
        items.append((title.strip(), full_url, date_str))
    max_page = 1
    pag = soup.find("div", class_="pagination")
    if pag:
        for a in pag.find_all("a", class_="page-num"):
            t = a.get_text(strip=True)
            if t.isdigit():
                max_page = max(max_page, int(t))
    return items, max_page

def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    title_tag = soup.find("div", class_="new_detail_title")
    title = ""
    date_str = ""
    if title_tag:
        h2 = title_tag.find("h2")
        if h2: title = h2.get_text(strip=True)
        h4 = title_tag.find("h4")
        if h4:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", h4.get_text())
            if m: date_str = m.group(1)
    cd = soup.find("div", class_="new_detail_text2")
    content = str(cd) if cd else ""
    if not title:
        m = re.search(r"<title>([^<]+)</title>", html)
        if m: title = m.group(1).replace(SITE_NAME, "").strip().rstrip("-").strip()
    if not date_str:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if m: date_str = m.group(1)
    return title, content, date_str

def crawl(max_pages=5):
    all_items = []
    html = fetch("http://www.xacjoe.com/news/")
    if not html: return []
    items, total_pages = parse_list(html)
    all_items.extend(items)
    print("  Page 1: {} items (total: {})".format(len(items), total_pages))
    for p in range(2, min(max_pages, total_pages) + 1):
        html = fetch("http://www.xacjoe.com/news_{}/".format(p))
        if not html: break
        items, _ = parse_list(html)
        all_items.extend(items)
        print("  Page {}: {} items".format(p, len(items)))
    return all_items

def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()
    items = crawl(args.pages)
    if not items:
        print("No items")
        sys.exit(1)
    new_count = 0
    dup_count = 0
    for title, page_url, pub_date in items:
        html = fetch(page_url)
        if not html: continue
        dt, content, dd = parse_detail(html)
        final_title = dt or title
        final_date = dd or pub_date
        summary = re.sub(r"<[^>]+>", " ", content)[:200] if content else ""
        summary = re.sub(r"\s+", " ", summary).strip()
        has_table = 1 if "<table" in content else 0

        safe_vals = {
            "site": SITE_NAME.replace("'", "''"),
            "url": page_url.replace("'", "''"),
            "title": final_title.replace("'", "''"),
            "date": final_date,
            "summary": summary.replace("'", "''"),
            "content": content.replace("'", "''"),
        }

        sql = "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,summary,content,industry,group_name,has_table) VALUES('{}','{}','{}','{}','{}','{}','{}','企业环保','企业',{})".format(
            safe_vals["site"], safe_vals["url"], safe_vals["url"],
            safe_vals["title"], safe_vals["date"], safe_vals["summary"],
            safe_vals["content"], has_table
        )

        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=30)

        result = subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, "SELECT id FROM gov_raw WHERE page_url='{}'".format(safe_vals["url"])],
            capture_output=True, text=True, timeout=30
        )
        row_id = result.stdout.strip()
        if row_id and row_id.isdigit():
            fts_sql = "INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) VALUES({},'{}','{}','{}')".format(
                row_id, safe_vals["title"], safe_vals["site"], safe_vals["summary"]
            )
            subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, fts_sql], capture_output=True, timeout=30)
            new_count += 1
            print("  + {} | {}".format(final_title[:30], final_date))
        else:
            dup_count += 1
    print("\nDone: {} new, {} dup".format(new_count, dup_count))

if __name__ == "__main__":
    main()
