diff --git a/parser.py b/parser.py index 479b538..36e8e5a 100644 --- a/parser.py +++ b/parser.py @@ -254,6 +254,15 @@ def extract_era_years(text): # 标注符号([]、【】)不参与匹配,但保留在文本中——位置偏移以原始文本为准 # (含换行/空格/标注),仅在收尾时把标注符号从"原文"字段剔除 original_text = text + # 码点→UTF-16码元 映射:JS 前端字符串索引按 UTF-16 计,补充平面字符 + # (如 𫓧 U+2B4E7)在 Python 算 1 个码点、在 JS 算 2 个码元,会导致其后位置漂移。 + # 返回的"位置"统一用 UTF-16 码元,保证前端 text.slice(start, end) == 原文。 + _utf16_index = [] + _utf16_pos = 0 + for ch in text: + _utf16_index.append(_utf16_pos) + _utf16_pos += 2 if ord(ch) > 0xFFFF else 1 + _utf16_index.append(_utf16_pos) simplified_text = normalize_chars(cc.convert(text)) results = [] # 覆盖集:先匹配的段落占据位置,后续段落跳过重叠(防止同一文本被多段重复摘出) @@ -646,9 +655,10 @@ def extract_era_years(text): if result.get("公元") is not None: result.setdefault("干支", gregorian_to_ganzhi(result["公元"])) - # === 收尾:原文剔除标注符号;位置保持原始文本偏移(含换行/空格/标注) === + # === 收尾:原文剔除标注符号;位置换算为 UTF-16 码元(JS 前端索引) === for result in results: start, end = result["位置"]["起始"], result["位置"]["结束"] + result["位置"] = {"起始": _utf16_index[start], "结束": _utf16_index[end]} result["原文"] = ''.join(ch for ch in original_text[start:end] if ch not in '[]【】') result["原文(简体)"] = ''.join(ch for ch in simplified_text[start:end] if ch not in '[]【】')