修复转换后文字漏缺问题

This commit is contained in:
sansen
2026-08-07 17:56:01 +08:00
parent a850c72ab5
commit 117eab4ad5
+18 -2
View File
@@ -384,14 +384,21 @@ def extract_era_years(text):
covered.update(range(start, end))
raw_fan = original_text[start:end]
era_type = "公元纪年(中华人民共和国)" if year_num >= 1949 else "公元纪年(近代/其他)"
results.append({
result = {
"原文": raw_fan,
"原文(简体)": m.group(0),
"类型": era_type,
"公元": year_num,
"公元中文": f"{arabic_to_chinese_year(year_num)}",
"位置": {"起始": start, "结束": end},
})
}
# 公元年后紧跟的干支(如"公元一九六〇年庚子")用于校验年份一致性
ganzhi = ''.join(ch for ch in simplified_text[end:end + 4] if ch not in '【】').strip()[:2]
if ganzhi in _ganzhi_to_offset:
result["干支"] = ganzhi
if gregorian_to_ganzhi(year_num) != ganzhi:
result["干支不符"] = True
results.append(result)
# === 3. 干支纪年(如"清雍正乙巳十一月十二午时生")===
def _process_ganzhi_match(m):
@@ -628,6 +635,15 @@ def extract_era_years(text):
if result.get("公元") is not None:
result.setdefault("干支", gregorian_to_ganzhi(result["公元"]))
# === 收尾:紧邻的【】标注(用户标记的日期跨度)纳入原文 ===
for result in results:
start, end = result["位置"]["起始"], result["位置"]["结束"]
if (start > 0 and simplified_text[start - 1] == ''
and end < len(simplified_text) and simplified_text[end] == ''):
result["位置"] = {"起始": start - 1, "结束": end + 1}
result["原文"] = original_text[start - 1:end + 1]
result["原文(简体)"] = simplified_text[start - 1:end + 1]
return results