# -*- coding: utf-8 -*-
"""VL 重跑《未命名书_20260909》(15页) → 用章节名页眉重命名 → 导出整书文本
输出最后一行 JSON 摘要便于宿主解析。
"""
import json
import re
import time
import sys
from collections import Counter

import requests

BASE = "http://127.0.0.1:8901"
BOOK = "未命名书_20260909"


def jget(u):
    r = requests.get(BASE + u, timeout=30)
    return r.json() if r.ok else {"ok": False, "error": r.text[:200]}


def jpost(u, payload):
    r = requests.post(BASE + u, json=payload, timeout=240)
    return r.json() if r.ok else {"ok": False, "error": r.text[:200]}


def pages(book=BOOK):
    import urllib.parse
    return jget("/api/books/" + urllib.parse.quote(book) + "/pages")


def main():
    # 0) 前置检查
    books = jget("/api/books").get("books", [])
    info = next((b for b in books if b["name"] == BOOK), None)
    if not info:
        print(json.dumps({"ok": False, "error": f"找不到项目《{BOOK}》", "books": [b["name"] for b in books]}, ensure_ascii=False))
        return
    print(f"[0] 项目《{BOOK}》：{info['pages']} 页（done={info['done']} todo={info['todo']} err={info['errors']}）", flush=True)

    # 1) VL 整书重跑（全部页）
    r = jpost("/api/batch/vlall", {"book_name": BOOK, "scope": "all", "retry_errors": True})
    if not r.get("ok"):
        print(json.dumps({"ok": False, "error": r}, ensure_ascii=False))
        return
    print(f"[1] 已排入 VL 队列：{r['n']} 页（正在处理的 {len(r.get('running', []))} 页跳过）", flush=True)

    # 2) 轮询直到全部处理完（单 worker 串行，每页约 10-20s；最多 25 分钟）
    t0 = time.time()
    done_all = False
    fail_round = 0
    while time.time() - t0 < 1500:
        d = pages()
        if not d.get("ok"):
            print("[poll] 读取失败：", d.get("error"), flush=True)
            time.sleep(8)
            continue
        ps = d["pages"]
        st = Counter(p["status"] for p in ps)
        remain = st.get("pending", 0) + st.get("queued", 0) + st.get("running", 0)
        errs = st.get("error", 0)
        el = int(time.time() - t0)
        print(f"[2] {el}s 状态: 完成{st.get('done',0)} 排队/识别中{remain} 失败{errs} / {len(ps)}", flush=True)
        if remain == 0:
            if errs == 0:
                done_all = True
                break
            if fail_round < 3:
                fail_round += 1
                print(f"[2] {errs} 页失败，第 {fail_round} 次自动重试失败页…", flush=True)
                jpost("/api/batch/process", {"book_name": BOOK, "retry_errors": True})
                time.sleep(5)
                continue
            else:
                break
        time.sleep(6)

    if not done_all:
        print(json.dumps({"ok": False, "error": "未能全部完成（见上）", "status": dict(Counter(p["status"] for p in pages().get("pages", [])))}, ensure_ascii=False))
        return

    # 3) 收集 VL 结果：页码/页眉/页脚 + 正文字数
    d = pages()
    ps = d["pages"]
    headers = []
    nums = []
    total_chars = 0
    for p in ps:
        h = (p.get("header") or "").strip()
        if h and h != "无":
            headers.append(h)
        pn = p.get("page_num")
        if pn is not None:
            nums.append(pn)
        total_chars += p.get("chars") or 0
    hc = Counter(headers)
    print(f"[3] 全部 {len(ps)} 页完成，总字数 {total_chars}；检测到页码 {len(nums)} 页：{sorted(nums)}", flush=True)
    print(f"[3] 页眉统计（前5）：{hc.most_common(5)}", flush=True)

    # 4) 章节名：优先含"第…章"且出现最多的页眉，其次出现最多的页眉
    def score(h):
        s = 0
        if re.search(r"第.{1,4}章", h):
            s += 10
        s += min(len(h), 24) * 0.1
        if len(h) > 40:
            s -= 5
        return s

    cands = [h for h, _ in hc.most_common() if h != "无"]
    new_name = None
    if cands:
        best = max(cands, key=score)
        clean = re.sub(r"[·•:：—-]+", " ", best).strip()
        clean = re.sub(r"\s+", " ", clean).strip()
        new_name = clean[:40]
    print(f"[4] 建议章节名：{new_name!r}（原文 '{best if cands else ''}'）", flush=True)

    # 5) 重命名
    renamed = BOOK
    if new_name:
        rr = jpost("/api/books/rename", {"book_name": BOOK, "new_name": new_name})
        if rr.get("ok"):
            renamed = rr.get("new", new_name)
            print(f"[5] 已重命名为《{renamed}》", flush=True)
        else:
            print(f"[5] 重命名失败（{rr.get('error')}），沿用《{BOOK}》", flush=True)

    # 6) 导出两版整书文本
    out = {}
    for side, tag in ((0, "正文版"), (1, "含页眉页脚版")):
        e = jpost("/api/batch/export", {"book_name": renamed, "scope": "folder", "include_side": side})
        if e.get("ok"):
            out[tag] = {"file": e["file"], "pages": e["pages"], "chars": e["chars"],
                        "url": e["url"], "ordered_by": e.get("ordered_by"), "vl_pages": e.get("vl_pages")}
            print(f"[6] {tag} 导出 OK：{e['pages']}页/{e['chars']}字 ordered_by={e.get('ordered_by')} url={e['url']}", flush=True)
        else:
            print(f"[6] {tag} 导出失败：{e}", flush=True)

    print(json.dumps({"ok": True, "renamed": renamed, "from": BOOK, "pages": len(ps),
                      "chars": total_chars, "headers": hc.most_common(6),
                      "page_nums": sorted(nums), "exports": out}, ensure_ascii=False), flush=True)


if __name__ == "__main__":
    main()
