响应体新增年数中文:廿/卅/卌 缩写、空位补〇、首位1作元

This commit is contained in:
sansen
2026-09-20 17:24:56 +08:00
parent 2f3830bd5d
commit d82e857841
+83
View File
@@ -192,12 +192,86 @@ _gz_era_group = r'(?P<era>' + '|'.join(
# === 数字转中文 ===
def arabic_to_chinese_year(num: int) -> str:
"""公元年份转中文(逐位体):1662 → 一六六二,-140 → -一四〇。
公元年份读作逐位数字,故不适用计数体规则(见 arabic_to_chinese_count)。"""
digits = "〇一二三四五六七八九"
if num < 0:
return '-' + ''.join(digits[int(d)] for d in str(abs(num)))
return ''.join(digits[int(d)] for d in str(num))
# === 年数(纪年计数)转中文 ===
# 规则(纪年计数体,用于"年数中文"字段):
# 1. 首位 1 → "元"(元年):民国元年、康熙元年、建元元年。
# 2. 10-19 → 十、十一…十九(不写"一十")。
# 3. 20-49 且十位即最高位 → 廿/卅/卌 缩写:20→廿、21→廿一、26→廿六、35→卅五、40→卌、49→卌九。
# 更高位用全称(120→一百二十,不写"一百廿")。
# 4. 50 以上用全称:五十、六十一、九十九、一百、二百五十、一千二百三十四。
# 5. 空位补"〇"(不用"零"):105→一百〇五、1005→一千〇五、1050→一千〇五十;
# 连续空位只补一个〇(10005→一万〇五)。
# 6. 万/亿 分级同普通话:一万、十万、一百万、一千〇五十万。
_CN_UNITS = '〇一二三四五六七八九'
_CN_TENS_ABBR = {2: '廿', 3: '卅', 4: '卌'}
def _cn_under_10000(n: int, leading: bool) -> str:
"""1 ≤ n ≤ 9999 的计数体中文;leading=True 表示该段是整个数的最高位段"""
digits = [n // 1000, n % 1000 // 100, n % 100 // 10, n % 10]
names = ['千', '百', '十', '']
out = []
zero_pending = False
for idx, (d, name) in enumerate(zip(digits, names)):
if d == 0:
if out:
zero_pending = True
continue
if zero_pending:
out.append('〇')
zero_pending = False
if name == '十' and leading and idx == 2 and not out and d < 5:
# 最高位就是十位:10-19 → 十…十九(不写"一十");20-49 → 廿/卅/卌 缩写
out.append('十' if d == 1 else _CN_TENS_ABBR[d])
else:
out.append(_CN_UNITS[d] + name)
return ''.join(out)
def arabic_to_chinese_count(num: int) -> str:
"""年数(纪年计数)转中文:1→元、20→廿、26→廿六、105→一百〇五、1050→一千〇五十。
与 arabic_to_chinese_year 的逐位体不同,本函数按计数体读法书写。"""
if num < 0:
return '-' + arabic_to_chinese_count(-num)
if num == 0:
return '〇'
if num == 1:
return '元'
if num < 10000:
return _cn_under_10000(num, leading=True)
if num < 100000000:
high, low = divmod(num, 10000)
out = _cn_under_10000(high, leading=True) + '万'
else:
high, low = divmod(num, 100000000)
out = arabic_to_chinese_count(high) + '亿'
if low:
if low < 1000: # 低位段最高位低于"千"→ 中间有空位,补一个〇
out += '〇'
out += _cn_under_10000(low, leading=False)
return out
def _insert_after(d: dict, key: str, new_key: str, value) -> dict:
"""在 key 之后插入 new_key(保持字段相邻,便于前端展示)"""
out = {}
for k, v in d.items():
out[k] = v
if k == key:
out[new_key] = value
if new_key not in out:
out[new_key] = value
return out
# === 古代年号正则 ===
# 从 era_dict 构建朝代匹配组(按长度降序排列,优先匹配长朝代名如"北宋"而非"北")
# 额外添加常见但不在CSV中的朝代简称(如"汉")
@@ -835,6 +909,15 @@ def extract_era_years(text):
if result.get("公元") is not None:
result.setdefault("干支", gregorian_to_ganzhi(result["公元"]))
# 年数中文:纪年计数的中文写法(廿/卅/卌、空位作〇、首位1作元)。
# 统一在收尾处补齐,任何新增的产生"年数"的分支都会自动带上该字段。
for i, result in enumerate(results):
if "年数" in result:
results[i] = _insert_after(
result, "年数", "年数中文",
f"{arabic_to_chinese_count(result['年数'])}年",
)
# === 收尾:原文剔除标注符号;位置换算为 UTF-16 码元(JS 前端索引) ===
for result in results:
start, end = result["位置"]["起始"], result["位置"]["结束"]