feat: 初始化 bigdata 题库与 quiz-app 刷题系统
This commit is contained in:
+87
@@ -0,0 +1,87 @@
|
||||
import json
|
||||
import re
|
||||
import fitz
|
||||
|
||||
def extract_questions_from_pdf(pdf_path, output_json_path):
|
||||
doc = fitz.open(pdf_path)
|
||||
text_content = ""
|
||||
for page in doc:
|
||||
text_content += page.get_text("text")
|
||||
|
||||
lines = [line.strip() for line in text_content.split('\n') if line.strip()]
|
||||
|
||||
questions = []
|
||||
|
||||
current_q = None
|
||||
|
||||
option_pattern = re.compile(r'^[A-H][\.、]\s*(.*)$')
|
||||
answer_pattern = re.compile(r'^[A-H]+$')
|
||||
|
||||
ignore_keywords = {
|
||||
"序", "号", "序号", "题目", "可选项", "题型", "答案", "来源文件",
|
||||
"文件内", "文件内具体", "具体出处", "出处", "难度等级", "(1-5 分)", "(1-5分)", "文件内具体出处"
|
||||
}
|
||||
|
||||
# 因为页码可能干扰,我们要确保 current_id 是递增的
|
||||
expected_id = 1
|
||||
state = "WAIT_FOR_ID"
|
||||
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
|
||||
if line in ignore_keywords:
|
||||
i += 1
|
||||
continue
|
||||
|
||||
if state == "WAIT_FOR_ID":
|
||||
if line == str(expected_id):
|
||||
current_q = {"id": expected_id, "title": "", "options": [], "type": "", "answer": ""}
|
||||
state = "READING_TITLE"
|
||||
|
||||
elif state == "READING_TITLE":
|
||||
if option_pattern.match(line) or line.startswith('A.') or line.startswith('A、'):
|
||||
state = "READING_OPTIONS"
|
||||
current_q["options"].append(line)
|
||||
elif line.isdigit():
|
||||
# 忽略多余的数字(如页码、难度等级、重复的题号)
|
||||
pass
|
||||
else:
|
||||
if current_q["title"]:
|
||||
current_q["title"] += " " + line
|
||||
else:
|
||||
current_q["title"] = line
|
||||
|
||||
elif state == "READING_OPTIONS":
|
||||
if line in ['单选', '多选']:
|
||||
current_q["type"] = line
|
||||
state = "WAIT_FOR_ANSWER"
|
||||
elif option_pattern.match(line) or line.startswith('B.') or line.startswith('C.') or line.startswith('D.') or line.startswith('E.') or line.startswith('F.'):
|
||||
current_q["options"].append(line)
|
||||
elif line.isdigit():
|
||||
pass
|
||||
else:
|
||||
if len(current_q["options"]) > 0:
|
||||
current_q["options"][-1] += " " + line
|
||||
|
||||
elif state == "WAIT_FOR_ANSWER":
|
||||
if answer_pattern.match(line) or line in ['A', 'B', 'C', 'D', 'AB', 'ABC', 'ABCD', 'ABD', 'ACD', 'BCD', 'AC', 'AD', 'BC', 'BD', 'CD']:
|
||||
current_q["answer"] = line
|
||||
questions.append(current_q)
|
||||
expected_id += 1
|
||||
state = "IGNORE_REST"
|
||||
|
||||
elif state == "IGNORE_REST":
|
||||
if line == str(expected_id):
|
||||
current_q = {"id": expected_id, "title": "", "options": [], "type": "", "answer": ""}
|
||||
state = "READING_TITLE"
|
||||
|
||||
i += 1
|
||||
|
||||
print(f"Extracted {len(questions)} questions.")
|
||||
|
||||
with open(output_json_path, 'w', encoding='utf-8') as f:
|
||||
json.dump(questions, f, ensure_ascii=False, indent=2)
|
||||
|
||||
if __name__ == "__main__":
|
||||
extract_questions_from_pdf("d:/code/bigdata/题库.pdf", "d:/code/bigdata/questions.json")
|
||||
Reference in New Issue
Block a user