#!/usr/bin/env python3
"""Retry missing items for pengze"""
import subprocess, re, sys, sqlite3
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

IP = "106.225.177.147"
RESOLVE = "www.pengze.gov.cn:443:{}".format(IP)
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
CURLOpts = ["curl", "-s", "-L", "--resolve", RESOLVE,
            "--connect-timeout", "30", "--max-time", "90",
            "-A", UA]
SITE_NAME = "彭泽县重大建设项目"
CUTOFF = datetime.now() - timedelta(days=3*365)

def fetch(url):
    r = subprocess.run(CURLOpts + [url], capture_output=True, timeout=100)
    if r.returncode != 0:
        print("  FAIL: exit {}".format(r.returncode), file=sys.stderr)
        return None
    return r.stdout.decode("utf-8", errors="replace")

conn = sqlite3.connect("/root/search.db")
existing = set()
for row in conn.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
    existing.add(row[0])
print("Existing: {}".format(len(existing)), file=sys.stderr)

# Fetch pages 8-10 slowly
for idx in [7, 8, 9]:
    url = "https://www.pengze.gov.cn/sthjj/fdzdgknr_171512/01/index_{}.html".format(idx)
    print("Page {}: ".format(idx+1), end="", file=sys.stderr, flush=True)
    html = fetch(url)
    if not html:
        print("FAILED", file=sys.stderr)
        continue
    soup = BeautifulSoup(html, "html.parser")
    count = 0
    for a in soup.select("a.xxgklist"):
        title = a.get("title", "").strip() or a.get_text(strip=True)
        href = a["href"].strip()
        if href.startswith("/"):
            full = "https://www.pengze.gov.cn" + href
        else:
            full = "https://www.pengze.gov.cn/sthjj/fdzdgknr_171512/01/" + href
        if full in existing:
            continue
        span = a.find_next("span")
        ds = span.get_text(strip=True) if span else ""
        fd = None
        try: fd = datetime.strptime(ds.strip(), "%Y-%m-%d")
        except: pass
        if fd and fd < CUTOFF:
            continue
        # Fetch detail
        print("  Fetching {}...".format(title[:40]), file=sys.stderr)
        dh = fetch(full)
        if not dh:
            continue
        soup2 = BeautifulSoup(dh, "html.parser")
        meta = soup2.find("meta", attrs={"name": "ArticleTitle"})
        dt = meta["content"].strip() if meta and meta.get("content") else ""
        if not dt:
            h1 = soup2.find("h1")
            dt = h1.get_text(strip=True) if h1 else ""
        if not dt:
            t = soup2.find("title")
            dt = t.get_text(strip=True).replace(" - 彭泽县人民政府网站", "").strip() if t else ""
        ds2 = ""
        md = soup2.find("meta", attrs={"name": "PubDate"})
        if md and md.get("content"):
            ds2 = md["content"].strip()
        if not ds2:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", soup2.get_text())
            if m: ds2 = m.group(1)
        cd = soup2.select_one("div.trs_editor_view")
        ch = str(cd) if cd else ""
        fd2 = None
        if ds2:
            try: fd2 = datetime.strptime(ds2.strip()[:19], "%Y-%m-%d")
            except: pass
        if fd2 and fd2 < CUTOFF:
            print("    Date-skip: {}".format(ds2), file=sys.stderr)
            continue
        ft = dt or title
        try:
            conn.execute("""INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, date_rank)
                VALUES (?,?,?,?,?,?,?)""",
                (SITE_NAME, full, full, ft,
                 fd2.strftime("%Y-%m-%d") if fd2 else "",
                 ch, 0 if not fd2 else int(fd2.strftime("%Y%m%d"))))
            conn.commit()
            count += 1
            print("    OK: +1 ({})".format(ft[:40]), file=sys.stderr)
        except Exception as e:
            print("    ERR: {}".format(e), file=sys.stderr)
    print("  Page {}: {} new".format(idx+1, count), file=sys.stderr)

# Rebuild FTS
conn.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
conn.execute("""INSERT INTO gov_search(rowid, site_name, title, page_url, publish_date, source_url, content, summary)
    SELECT rowid, site_name, title, page_url, publish_date, source_url, content, summary
    FROM gov_raw WHERE site_name=?""", (SITE_NAME,))
conn.commit()
conn.close()
tot = subprocess.run(["sqlite3", "/root/search.db",
    "SELECT COUNT(*) FROM gov_raw WHERE site_name='{}'".format(SITE_NAME)],
    capture_output=True)
print("Total for site: {}".format(tot.stdout.decode().strip()), file=sys.stderr)
print("DONE", file=sys.stderr)
