feat: 初始化 bigdata 题库与 quiz-app 刷题系统

This commit is contained in:
2026-08-14 08:13:11 +08:00
commit c303426858
25 changed files with 25276 additions and 0 deletions
+87
View File
@@ -0,0 +1,87 @@
import json
import re
import fitz
def extract_questions_from_pdf(pdf_path, output_json_path):
doc = fitz.open(pdf_path)
text_content = ""
for page in doc:
text_content += page.get_text("text")
lines = [line.strip() for line in text_content.split('\n') if line.strip()]
questions = []
current_q = None
option_pattern = re.compile(r'^[A-H][\.、]\s*(.*)$')
answer_pattern = re.compile(r'^[A-H]+$')
ignore_keywords = {
"", "", "序号", "题目", "可选项", "题型", "答案", "来源文件",
"文件内", "文件内具体", "具体出处", "出处", "难度等级", "1-5 分)", "1-5分)", "文件内具体出处"
}
# 因为页码可能干扰,我们要确保 current_id 是递增的
expected_id = 1
state = "WAIT_FOR_ID"
i = 0
while i < len(lines):
line = lines[i]
if line in ignore_keywords:
i += 1
continue
if state == "WAIT_FOR_ID":
if line == str(expected_id):
current_q = {"id": expected_id, "title": "", "options": [], "type": "", "answer": ""}
state = "READING_TITLE"
elif state == "READING_TITLE":
if option_pattern.match(line) or line.startswith('A.') or line.startswith('A、'):
state = "READING_OPTIONS"
current_q["options"].append(line)
elif line.isdigit():
# 忽略多余的数字(如页码、难度等级、重复的题号)
pass
else:
if current_q["title"]:
current_q["title"] += " " + line
else:
current_q["title"] = line
elif state == "READING_OPTIONS":
if line in ['单选', '多选']:
current_q["type"] = line
state = "WAIT_FOR_ANSWER"
elif option_pattern.match(line) or line.startswith('B.') or line.startswith('C.') or line.startswith('D.') or line.startswith('E.') or line.startswith('F.'):
current_q["options"].append(line)
elif line.isdigit():
pass
else:
if len(current_q["options"]) > 0:
current_q["options"][-1] += " " + line
elif state == "WAIT_FOR_ANSWER":
if answer_pattern.match(line) or line in ['A', 'B', 'C', 'D', 'AB', 'ABC', 'ABCD', 'ABD', 'ACD', 'BCD', 'AC', 'AD', 'BC', 'BD', 'CD']:
current_q["answer"] = line
questions.append(current_q)
expected_id += 1
state = "IGNORE_REST"
elif state == "IGNORE_REST":
if line == str(expected_id):
current_q = {"id": expected_id, "title": "", "options": [], "type": "", "answer": ""}
state = "READING_TITLE"
i += 1
print(f"Extracted {len(questions)} questions.")
with open(output_json_path, 'w', encoding='utf-8') as f:
json.dump(questions, f, ensure_ascii=False, indent=2)
if __name__ == "__main__":
extract_questions_from_pdf("d:/code/bigdata/题库.pdf", "d:/code/bigdata/questions.json")