#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""提取全部问答库题目为统一 JSON，供相似问法生成与 xlsx 合并。"""
import os, re, json

BASE = "/home/zyw/Downloads/dl-hub/01-AI课程设计项目/知识库问答库"
QB_ROOT = os.path.join(BASE, "问答库")
OUT = os.path.join(BASE, "题目全集.json")

CH_MAP = {
    "第1章 人工智能概述": "人工智能概述",
    "第2章 Python程序设计基础": "Python基础",
    "第3章 NumPy数值分析库": "NumPy分析",
    "第4章 pandas数据分析库": "pandas分析",
    "第5章 计算机视觉技术与应用": "计算机视觉",
    "第6章 智能语音处理与应用": "智能语音",
    "第7章 自然语言处理与应用": "自然语言处理",
    "第8章 生成式大模型应用": "生成式大模型",
    "实验篇": "实验篇",
}

def parse_questions(text):
    qs = []
    blocks = re.split(r'\n(?=\d+\.\s*【)', text)
    for blk in blocks:
        m = re.match(r'(\d+)\.\s*【([^】·]+)·([^】]+)】\s*(.*)', blk, re.S)
        if not m: continue
        _, qtype, diff, rest = m.groups()
        opt_re = re.compile(r'^[ \t]*([A-D])[\.、)．]\s*(.*)$', re.M)
        opt_lines = opt_re.findall(rest)
        epos = [m2.start() for m2 in [re.search(r'\n[ \t]*[A-D][\.、)．]', rest), re.search(r'\*\*答案', rest)] if m2]
        cut = min(epos) if epos else len(rest)
        qtext = re.sub(r'\s+', ' ', rest[:cut]).strip()
        opts = [f"{l}. {o.strip()}" for l, o in opt_lines]
        am = re.search(r'\*\*答案[：:]\s*([^*\n]+)', rest)
        answer = am.group(1).strip() if am else ""
        anm = re.search(r'\*\*解析[：:]\s*(.+?)(?=\n\d+\.|\Z)', rest, re.S)
        analysis = re.sub(r'\s+', ' ', anm.group(1)).strip() if anm else ""
        qs.append({"type": qtype, "diff": diff, "q": qtext, "opts": opts, "answer": answer, "analysis": analysis})
    return qs

all_items = []
for ch_dir in sorted(os.listdir(QB_ROOT)):
    ch_path = os.path.join(QB_ROOT, ch_dir)
    if not os.path.isdir(ch_path): continue
    cat1 = CH_MAP.get(ch_dir, ch_dir)
    for fn in sorted(os.listdir(ch_path)):
        if not fn.endswith(".md"): continue
        m = re.match(r'^(.+?)-\s*(.+)\.md$', fn)
        if not m: continue
        nid, title = m.group(1), m.group(2)
        text = open(os.path.join(ch_path, fn), encoding='utf-8').read()
        for q in parse_questions(text):
            all_items.append({
                "cat": f"{cat1}/{nid}",
                "title": title,
                "q": q["q"], "type": q["type"], "diff": q["diff"],
                "opts": q["opts"], "answer": q["answer"], "analysis": q["analysis"],
            })

with open(OUT, 'w', encoding='utf-8') as f:
    json.dump(all_items, f, ensure_ascii=False, indent=1)
print("题目总数:", len(all_items), "→", OUT)