""" 公式预处理模块 灵感来源于 https://github.com/GALVINLAI/formatting 在将 Markdown 交给 Pandoc 转换之前,统一各种 AI (GPT/Claude) 生成内容中 常见的非标准公式定界符,使其变成 Pandoc 可识别的标准格式。 默认启用的处理流程: 0. 修复粗体 **...** → ...(解决 CJK 标点场景下的 Pandoc 定界符问题) 1. 修复 GPT 缺失反斜杠:独立行的 [ ... ] → \[ ... \] 2. 修复 GPT 缺失反斜杠:疑似数学的 ( ... ) → \( ... \)(保守策略) 3. \( ... \) → $ ... $(行内公式) 4. \[ ... \] → $$ ... $$(行间公式) """ import re # ==================================== # 内部辅助函数 # ==================================== # 已存在的数学环境匹配(用于跳过,防止重复处理) _MATH_SPAN_RE = re.compile( r'(' r'\$\$.*?\$\$' r'|\\\[.*?\\\]' r'|\\begin\{equation\*?\}.*?\\end\{equation\*?\}' r'|\\begin\{align\*?\}.*?\\end\{align\*?\}' r'|\\\(.*?\\\)' r'|(? bool: """判断文本是否看起来像数学表达式(保守策略)""" stripped = text.strip() if not stripped: return False # 含中日韩字符的不是数学 if re.search(r'[\u4e00-\u9fff\u30a0-\u30ff\u3040-\u309f\uac00-\ud7af]', stripped): return False # 去掉 LaTeX 命令后判断 stripped_no_cmd = re.sub(r'\\[a-zA-Z]+', '', stripped) if re.search(r'[A-Za-z]{3,}', stripped_no_cmd): has_cmd = re.search(r'\\[a-zA-Z]+', stripped) is not None has_eq = re.search(r'[=<>]', stripped) is not None has_arith = re.search( r'(?:[A-Za-z0-9]{1,2}|\\[a-zA-Z]+)\s*[+*/-]\s*(?:[A-Za-z0-9]{1,2}|\\[a-zA-Z]+)', stripped ) is not None has_scripts = re.search(r'[\^_]', stripped) is not None has_abs = re.search(r'\|[^|]+\|', stripped) is not None words = re.findall(r'[A-Za-z]{2,}', stripped_no_cmd) has_sentence_punct = re.search(r'[,:;.]', stripped) is not None if has_cmd and len(words) == 0: return True if ' ' in stripped: stopwords = { 'or', 'and', 'between', 'for', 'if', 'then', 'as', 'such', 'that', 'which', 'where', 'there', 'exists', 'only', 'with', 'without', 'on', 'in', 'at', 'from', 'to', 'by', 'of', 'the', 'a', 'an' } lower_words = {w.lower() for w in words} if lower_words & stopwords and not (has_eq or has_arith): return False if len(words) >= 2 and has_scripts and not (has_eq or has_arith or has_abs): return False if ' ' in stripped and len(words) >= 2 and has_sentence_punct and not (has_eq or has_arith): return False if ' ' in stripped and len(words) >= 2 and not (has_eq or has_arith or has_scripts or has_abs): return False if ' ' in stripped and len(words) >= 1 and has_cmd and not (has_eq or has_arith or has_scripts or has_abs): return False if not (has_cmd or has_eq or has_arith or has_scripts or has_abs): return False if re.search(r'\\[a-zA-Z]+', stripped): return True if re.search(r'[\^_]', stripped): return True if re.search(r'[=<>]', stripped): return True if re.search(r'\|[^|]+\|', stripped): return True return False def _is_escaped(text: str, idx: int) -> bool: backslashes = 0 j = idx - 1 while j >= 0 and text[j] == '\\': backslashes += 1 j -= 1 return backslashes % 2 == 1 def _find_parentheses_pairs(text: str): stack = [] pairs = [] for i, ch in enumerate(text): if ch == '(' and not _is_escaped(text, i): stack.append(i) elif ch == ')' and not _is_escaped(text, i): if stack: start = stack.pop() pairs.append((start, i)) return pairs def _repair_inline_parentheses_plain(text: str) -> str: """将文本中疑似数学的 (…) 替换为 \(…\)""" pairs = _find_parentheses_pairs(text) if not pairs: return text candidates = [] for start, end in pairs: inner = text[start + 1:end] if '\n' in inner: continue if start > 0 and re.match(r'[A-Za-z0-9_\\]', text[start - 1]): continue if end + 1 < len(text) and re.match(r'[A-Za-z0-9_\\]', text[end + 1]): continue if not _is_mathish(inner): continue candidates.append((start, end)) if not candidates: return text # 选择外层优先,避免嵌套 \( \) candidates.sort(key=lambda p: (p[0], -(p[1] - p[0]))) selected = [] for start, end in candidates: overlaps = any(not (end < s_start or start > s_end) for s_start, s_end in selected) if not overlaps: selected.append((start, end)) if not selected: return text out = text for start, end in sorted(selected, key=lambda p: p[0], reverse=True): inner = out[start + 1:end] out = out[:start] + f'\\({inner}\\)' + out[end + 1:] return out # ==================================== # 核心公开处理函数 # ==================================== def repair_display_brackets(content: str) -> str: """ 将 GPT 输出中独立成行的 [ ... ] 识别为行间公式,改为 \[ ... \]。 只在 [ 和 ] 各自单独成行时才转换,以避免误伤普通方括号。 """ lines = content.splitlines(keepends=True) out = [] i = 0 while i < len(lines): line = lines[i] if _OPEN_LINE_RE.match(line): j = i + 1 while j < len(lines) and not _CLOSE_LINE_RE.match(lines[j]): j += 1 if j < len(lines): out.append(line.replace('[', r'\[', 1)) out.extend(lines[i + 1:j]) out.append(lines[j].replace(']', r'\]', 1)) i = j + 1 continue out.append(line) i += 1 return ''.join(out) def repair_inline_parentheses(content: str) -> str: """ 保守策略:将非数学环境中疑似数学的 (…) 替换为 \(…\)。 已在 $...$ 或 \[...\] 等数学环境内的括号不受影响。 """ parts = _MATH_SPAN_RE.split(content) for i in range(0, len(parts), 2): parts[i] = _repair_inline_parentheses_plain(parts[i]) return ''.join(parts) def square_brackets_to_dollars(content: str) -> str: """将 \[ ... \] 行间公式环境替换为 $$ ... $$""" return re.sub(r'\\\[|\\\]', '$$', content) def parentheses_to_single_dollar(content: str) -> str: """将 \( ... \) 行内公式环境替换为 $ ... $""" return re.sub(r'\\\(|\\\)', '$', content) # ==================================== # 粗体修复(CJK 兼容) # ==================================== # 用于分离代码块(避免修改代码块内的 ** 标记) _CODE_BLOCK_RE = re.compile(r'(```[\s\S]*?```|`[^`\n]+`)', re.DOTALL) # 会导致 ** 开头定界符失效的 Unicode 标点字符(涵盖 ASCII 和 CJK 引号、括号等) _OPEN_PUNCT_AFTER_BOLD = re.compile( r'(\*\*)' # 匹配 ** r'([\'\"' # ASCII 标点 ' " r'\u2018\u2019\u201c\u201d' # 弯引号 ' ' " " r'\u00ab\u00bb\u2039\u203a' # 《》‹› r'\u300c\u300d\u300e\u300f' # 「」『』 r'\u2014\u2013\u2026' # — – … r'\uff01\uff08\uff09\uff0c\uff1a\uff1b\uff1f' # !(),:;? r'])' ) # 会导致 ** 结尾定界符失效的 Unicode 标点字符 _CLOSE_PUNCT_BEFORE_BOLD = re.compile( r'([\'\"' # ASCII 标点 ' " r'\u2018\u2019\u201c\u201d' # 弯引号 ' ' " " r'\u00ab\u00bb\u2039\u203a' # 《》‹› r'\u300c\u300d\u300e\u300f' # 「」『』 r'\u2014\u2013\u2026' # — – … r'\uff01\uff08\uff09\uff0c\uff1a\uff1b\uff1f' # !(),:;? r'\u3002' # 句号 。 r'])' r'(\*\*)' # 匹配 ** ) def fix_bold_markers(content: str) -> str: """ 修复 ** 粗体定界符在 CJK 标点场景下失效的问题。 【问题根源】 CommonMark 对定界符侧翼规则有严格要求: 1. 左侧定界符:若 ** 后面是标点,则前面必须是空白或标点。 2. 右侧定界符:若 ** 前面是标点,则后面必须是空白或标点。 当中英文混排时(例如 `的**"文本"**。`),上述条件常常不满足,导致 Pandoc 识别失败。 【修复方案】 使用 U+200B (零宽空格),其类别为 Cf(格式字符),既非标点也非空白。 - 左侧修复:在 `**` 和后接标点之间插入 U+200B。这使得 `**` 后面紧跟的变成 Cf,不属于标点,满足左侧条件。 - 右侧修复:在先接标点和 `**` 之间插入 U+200B。这使得 `**` 前面紧跟的变成 Cf,不属于标点,满足右侧条件。 """ segments = _CODE_BLOCK_RE.split(content) result = [] ZWS = '\u200b' for i, seg in enumerate(segments): if i % 2 == 1: result.append(seg) else: # 修复右侧定界符(如 "文本"** -> "文本"[ZWS]** ) seg = _CLOSE_PUNCT_BEFORE_BOLD.sub(lambda m: m.group(1) + ZWS + m.group(2), seg) # 修复左侧定界符(如 **"文本" -> **[ZWS]"文本" ) seg = _OPEN_PUNCT_AFTER_BOLD.sub(lambda m: m.group(1) + ZWS + m.group(2), seg) result.append(seg) return ''.join(result) # ==================================== # 主入口:对 Markdown 进行公式预处理 # ==================================== def preprocess_formulas(content: str) -> str: """ 对 Markdown 内容进行格式标准化处理,再交给 Pandoc 转换。 处理顺序: 0. **粗体** → (修复 CJK 标点场景下 Pandoc 定界符失效) 1. 修复独立行的 [ ... ] → \[ ... \] 2. 修复疑似数学的 ( ... ) → \( ... \)(保守策略) 3. \( ... \) → $ ... $ 4. \[ ... \] → $$ ... $$ Args: content: 原始 Markdown 字符串 Returns: 处理后的 Markdown 字符串 """ # 步骤0:修复粗体标记(必须在公式处理之前,避免 $ 内的 ** 被误处理) content = fix_bold_markers(content) # 步骤1:修复 GPT 缺失反斜杠的行间公式 [ ... ] → \[ ... \] content = repair_display_brackets(content) # 步骤2:修复 GPT 缺失反斜杠的行内公式 ( ... ) → \( ... \) content = repair_inline_parentheses(content) # 步骤3:\( ... \) → $ ... $ content = parentheses_to_single_dollar(content) # 步骤4:\[ ... \] → $$ ... $$ content = square_brackets_to_dollars(content) return content