返回字段全量化:所有结果都带同一套键,无内容给空值

This commit is contained in:
sansen
2026-09-21 14:30:47 +08:00
parent 1a65c070d6
commit 2a6ce68e59
+41 -26
View File
@@ -261,15 +261,35 @@ def arabic_to_chinese_count(num: int) -> str:
return out
def _insert_after(d: dict, key: str, new_key: str, value) -> dict:
"""在 key 之后插入 new_key(保持字段相邻,便于前端展示)"""
# === 返回字段全量化 ===
# 每条结果都返回同一套键(顺序固定),无内容的字段给空值:
# 字符串 ""、数值 None、布尔 False、列表 []——前端/后端按对象固定字段接收参数时不会"找不到键"。
_RESULT_FIELDS = (
"原文", "原文(简体)", "类型",
"朝代", "年号", "庙号",
"年数", "年数中文", "干支",
"公元", "公元中文",
"位置",
"干支不符", "干支推算", "超出使用期", "多候选", "候选",
)
_RESULT_DEFAULTS = {
"原文": "", "原文(简体)": "", "类型": "",
"朝代": "", "年号": "", "庙号": "",
"年数": None, "年数中文": "", "干支": "",
"公元": None, "公元中文": "",
"位置": None,
"干支不符": False, "干支推算": None, "超出使用期": False, "多候选": False,
}
def _finalize_result(result: dict) -> dict:
"""按 _RESULT_FIELDS 固定顺序输出全字段(缺的键补空值;列表字段各自独立拷贝)"""
out = {}
for k, v in d.items():
out[k] = v
if k == key:
out[new_key] = value
if new_key not in out:
out[new_key] = value
for key in _RESULT_FIELDS:
if key == "候选":
out[key] = list(result.get("候选") or [])
else:
out[key] = result.get(key, _RESULT_DEFAULTS[key])
return out
@@ -434,7 +454,6 @@ def extract_era_years(text):
"年号": "民国纪年",
"年数": year_num,
"公元": gregorian,
"公元中文": arabic_to_chinese_year(gregorian),
"位置": {"起始": start, "结束": end},
"类型": "民国纪年",
}
@@ -480,7 +499,6 @@ def extract_era_years(text):
"年数": year_in_era,
"干支": ganzhi,
"公元": gy_year if gy_year is not None else computed,
"公元中文": arabic_to_chinese_year(gy_year if gy_year is not None else computed),
"位置": {"起始": start, "结束": end},
"类型": "民国纪年(干支)",
}
@@ -534,7 +552,6 @@ def extract_era_years(text):
"年号": era,
"年数": year_in_era,
"公元": year_num,
"公元中文": arabic_to_chinese_year(year_num),
"位置": {"起始": start, "结束": end},
"类型": "年号+公元",
}
@@ -593,7 +610,6 @@ def extract_era_years(text):
"原文(简体)": m.group(0),
"干支": ganzhi,
"公元": gy_year,
"公元中文": arabic_to_chinese_year(gy_year),
"位置": {"起始": start, "结束": end},
"类型": "干支纪年",
}
@@ -657,7 +673,6 @@ def extract_era_years(text):
"原文(简体)": simplified_text[start:end],
"类型": era_type,
"公元": year_num,
"公元中文": arabic_to_chinese_year(year_num),
"位置": {"起始": start, "结束": end},
}
if ganzhi in _ganzhi_to_offset:
@@ -714,7 +729,6 @@ def extract_era_years(text):
"年数": year_in_era,
"干支": ganzhi,
"公元": gregorian,
"公元中文": arabic_to_chinese_year(gregorian),
"位置": {"起始": start, "结束": end},
"类型": "干支纪年",
}
@@ -788,7 +802,6 @@ def extract_era_years(text):
"朝代": dynasty,
"庙号": temple,
"公元": base,
"公元中文": arabic_to_chinese_year(base),
"位置": {"起始": start, "结束": end},
"类型": "庙号纪年",
})
@@ -859,7 +872,6 @@ def extract_era_years(text):
"年号": era,
"年数": year_num,
"公元": gregorian,
"公元中文": arabic_to_chinese_year(gregorian),
"位置": {"起始": start, "结束": end},
"类型": "古代纪年",
}
@@ -910,21 +922,24 @@ def extract_era_years(text):
if result.get("公元") is not None:
result.setdefault("干支", gregorian_to_ganzhi(result["公元"]))
# 年数中文:纪年计数的中文写法(廿/卅/卌、空位作〇、首位1作元),只给数字写法本身、不带"年"单位。
# 统一在收尾处补齐,任何新增的产生"年数"的分支都会自动带上该字段。
# === 收尾 ===
# 1) 派生字段:公元中文(公元年份逐位读法)、年数中文(廿/卅/卌 计数读法)——都只给数字写法,不带"年"单位。
# 在这里统一派生,任何新增的产生"公元"/"年数"的分支都会自动带上;
# 公元纪年、庙号纪年没有年数 → 年数中文 = "";未匹配年号没有公元 → 公元中文 = ""。
# 2) 原文剔除标注符号;位置换算为 UTF-16 码元(JS 前端索引)。
# 3) 字段全量化:_finalize_result 按固定顺序补齐所有键(缺内容给空值)。
for i, result in enumerate(results):
if "年数" in result:
results[i] = _insert_after(
result, "年数", "年数中文",
arabic_to_chinese_count(result["年数"]),
)
# === 收尾:原文剔除标注符号;位置换算为 UTF-16 码元(JS 前端索引) ===
for result in results:
start, end = result["位置"]["起始"], result["位置"]["结束"]
result["年数中文"] = (
arabic_to_chinese_count(result["年数"]) if result.get("年数") is not None else ""
)
result["公元中文"] = (
arabic_to_chinese_year(result["公元"]) if result.get("公元") is not None else ""
)
result["位置"] = {"起始": _utf16_index[start], "结束": _utf16_index[end]}
result["原文"] = ''.join(ch for ch in original_text[start:end] if ch not in '[]【】')
result["原文(简体)"] = ''.join(ch for ch in simplified_text[start:end] if ch not in '[]【】')
results[i] = _finalize_result(result)
return results