"""5ch 収集 (2026-09-21)。 変わるものに手を取られない作り:
  - 板は固定しない: bbsmenu (板一覧) を毎回取り、 カテゴリ語 (channels.toml の `match`) で選ぶ → 板の移転・サーバ変更に追従
  - スレ一覧は subject.txt (2ch 時代から同じ形式: `<dat>.dat<>タイトル (レス数)`)
  - 本文は read.cgi の HTML。 レス区切りは複数パターンで試し、 全部外れたら 「0 件」 を健全性として記録して止まる
  - 本文は YouTube と同じ経路 (mask → NFKC → SqliteAppender)。 板名・スレ名・URL・名前欄は保存しない (genre だけ)

  python scripts/import_5ch.py --match 実況 --limit-boards 1 --threads 3 --dry     # 動作確認
  python scripts/import_5ch.py --match VTuber --genre vtuber --threads 20         # 収集

健全性: data/health_5ch.tsv に (時刻, 板, スレ数, レス数) を追記。 レス 0 が続けば parser が壊れている
"""
from __future__ import annotations

import argparse
import html
import pathlib
import re
import sys
import time
import threading
import urllib.request

from import_youtube_live import SqliteAppender, mask, normalize

ROOT = pathlib.Path(__file__).resolve().parent.parent
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Gecko/20100101 Firefox/120.0"
BBSMENU = "https://menu.5ch.net/bbsmenu.html"
WAIT = 3.0  # 1 リクエストあたりの間隔 (秒)


_GATE = threading.Lock()
_LAST = [0.0]


def get(url: str) -> bytes:
    """全 watcher 共通の throttle (WAIT 秒に 1 リクエスト)。 2026-09-21: 板 watcher を 5 本にしたら同時起動で
    read.cgi が空ページを返し 「parse 0 posts」 が続いたので、 板ごとではなくプロセス全体で間隔を守る。"""
    with _GATE:
        wait = _LAST[0] + WAIT - time.time()
        if wait > 0:
            time.sleep(wait)
        _LAST[0] = time.time()
    req = urllib.request.Request(url, headers={"User-Agent": UA})
    with urllib.request.urlopen(req, timeout=60) as r:
        return r.read()


def boards(match: str) -> list[tuple[str, str, str]]:
    """bbsmenu → (カテゴリ, 板名, URL)。 カテゴリ名か板名に match を含むもの。"""
    t = get(BBSMENU).decode("shift_jis", errors="replace")
    out = []
    cat = ""
    for line in t.splitlines():
        m = re.search(r"<BR><BR><B>(.+?)</B>", line, re.I)
        if m:
            cat = m.group(1)
        # ホスト名は決め打ちしない (5ch.net → 5ch.io のように変わる)。 「ホスト/板/」 の形の link だけ拾う
        for m in re.finditer(r'<A HREF=(https?://[^/ >"]+/[^/ >"]+/)[^>]*>([^<]+)</A>', line, re.I):
            url, name = m.group(1), m.group(2)
            if re.search(r"/(bbsmenu|test|_)", url) or "." in url.rstrip("/").rsplit("/", 1)[-1]:
                continue
            if "headline." in url or "find." in url or "dig." in url:
                continue  # 集約ページ (板ではない)
            if match in cat or match in name:
                out.append((cat, name, url.rstrip("/") + "/"))
    return out


def threads(board_url: str, n: int) -> list[tuple[str, str, int]]:
    """subject.txt → (dat, タイトル, レス数) をレス数降順で n 件。"""
    t = get(board_url + "subject.txt").decode("shift_jis", errors="replace")
    out = []
    for line in t.splitlines():
        m = re.match(r"(\d+)\.dat<>(.*)\s\((\d+)\)\s*$", line)
        if m:
            out.append((m.group(1), html.unescape(m.group(2)), int(m.group(3))))
    out.sort(key=lambda x: -x[2])
    return out[:n]


# レス本文の切り出し。 HTML が変わっても、 どれかのパターンで取れれば OK
POST_PATTERNS = [
    re.compile(r'<div class="post-content"[^>]*>(.*?)</div>', re.S),   # 2026 の read.cgi (5ch.io)
    re.compile(r'<div class="message"[^>]*>(.*?)</div>', re.S),        # 2020〜 の read.cgi
    re.compile(r'<div class="post"[^>]*>.*?<div class="message">(.*?)</div>', re.S),
    re.compile(r"<dd>(.*?)</dd>", re.S),                                  # 旧 read.cgi
    re.compile(r'<span class="escaped">(.*?)</span>', re.S),
]


def posts(board_url: str, dat: str) -> list[str]:
    m = re.match(r"(https?://[^/]+)/([^/]+)/", board_url)
    if not m:
        return []
    host, bbs = m.group(1), m.group(2)
    b = get(f"{host}/test/read.cgi/{bbs}/{dat}/")
    # 文字コードは決め打ちしない: meta の charset を見て、 無ければ utf-8 → cp932 の順
    m_cs = re.search(rb'charset=([\w-]+)', b[:4000], re.I)
    enc = m_cs.group(1).decode() if m_cs else "utf-8"
    if enc.lower().replace("_", "-") in ("shift-jis", "shift-jis", "sjis", "x-sjis", "windows-31j", "ms932"):
        enc = "cp932"
    try:
        t = b.decode(enc, errors="replace")
    except LookupError:
        t = b.decode("utf-8", errors="replace")
    for pat in POST_PATTERNS:
        found = pat.findall(t)
        if len(found) >= 3:
            out = []
            for p in found:
                p = re.sub(r"<br\s*/?>", "\n", p, flags=re.I)
                p = re.sub(r"<a [^>]*>(.*?)</a>", r"\1", p, flags=re.S)
                p = re.sub(r"<[^>]+>", "", p)
                p = html.unescape(p)
                p = re.sub(r"^>>\d+(-\d+)?\s*", "", p.strip())  # アンカーは落とす
                if re.match(r"^(!extend:|VIPQ2_EXTDAT|https?://|youtu\.be/)", p) or not p:
                    continue  # スレのテンプレ行・URL だけの行
                out.append(p)
            return out
    return []


def collect(match: str, genre: str | None, out_dir: pathlib.Path, limit_boards: int = 3, threads_per_board: int = 10,
            dry: bool = False, log_tag: str = "", stop_event=None) -> int:
    """板を発見して勢い上位スレのレスを集める。 戻り値 = 取り込んだ行数。 daemon から定期呼び出し。"""
    prefix = f"[{log_tag}] " if log_tag else "[5ch] "
    bs = boards(match)[:limit_boards]
    print(f"{prefix}match={match!r} boards={[(c, n) for c, n, _ in bs]}", flush=True)
    writer = None if dry else SqliteAppender.shared(out_dir / "comments.sqlite")
    health = ROOT / "data" / "health_5ch.tsv"
    total = 0
    try:
        for cat, name, url in bs:
            if stop_event is not None and stop_event.is_set():
                break
            time.sleep(WAIT)
            try:
                ths = threads(url, threads_per_board)
            except Exception as e:
                print(f"{prefix}[err] subject {name}: {e!r}", flush=True)
                continue
            n_posts = 0
            for dat, title, cnt in ths:
                if stop_event is not None and stop_event.is_set():
                    break
                time.sleep(WAIT)
                try:
                    ps = posts(url, dat)
                except Exception as e:  # HTTP エラー等は記録して続行 (壊れたら健全性の数字で分かる)
                    print(f"{prefix}[err] {name} dat={dat}: {e!r}", flush=True)
                    ps = []
                for p in ps:
                    for line in p.split("\n"):
                        body = normalize(mask(line))
                        if not body or len(body) < 2:
                            continue
                        if dry:
                            if n_posts < 15:
                                print("  ", body[:80], flush=True)
                        else:
                            writer.write(int(time.time() * 1000), body, genre)
                        n_posts += 1
                if not ps:
                    print(f"{prefix}[warn] parse 0 posts: {name} dat={dat} ({title[:30]})", flush=True)
            total += n_posts
            with health.open("a", encoding="utf-8") as f:
                f.write(f"{int(time.time())}\t{name}\t{len(ths)}\t{n_posts}\n")
            print(f"{prefix}{name}: threads={len(ths)} posts={n_posts}", flush=True)
    finally:
        if writer is not None:
            writer.close()
    print(f"{prefix}done posts={total}", flush=True)
    return total


def main() -> int:
    ap = argparse.ArgumentParser()
    ap.add_argument("--match", required=True, help="bbsmenu のカテゴリ名 / 板名に含まれる語 (例: なんでも実況J, ゲーム)")
    ap.add_argument("--genre", default=None)
    ap.add_argument("--limit-boards", type=int, default=3)
    ap.add_argument("--threads", type=int, default=10, help="板ごとに勢い (レス数) 上位 N スレ")
    ap.add_argument("--dry", action="store_true")
    ap.add_argument("--out-dir", default=str(ROOT / "data"))
    a = ap.parse_args()
    collect(a.match, a.genre, pathlib.Path(a.out_dir), a.limit_boards, a.threads, a.dry)
    return 0


if __name__ == "__main__":
    sys.exit(main())
