From d82e857841c12498f4e7db0c63a60bc47068716f Mon Sep 17 00:00:00 2001 From: sansen Date: Sun, 20 Sep 2026 17:24:56 +0800 Subject: [PATCH] =?UTF-8?q?=E5=93=8D=E5=BA=94=E4=BD=93=E6=96=B0=E5=A2=9E?= =?UTF-8?q?=E5=B9=B4=E6=95=B0=E4=B8=AD=E6=96=87=EF=BC=9A=E5=BB=BF/?= =?UTF-8?q?=E5=8D=85/=E5=8D=8C=20=E7=BC=A9=E5=86=99=E3=80=81=E7=A9=BA?= =?UTF-8?q?=E4=BD=8D=E8=A1=A5=E3=80=87=E3=80=81=E9=A6=96=E4=BD=8D1?= =?UTF-8?q?=E4=BD=9C=E5=85=83?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- parser.py | 83 +++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 83 insertions(+) diff --git a/parser.py b/parser.py index efc9f17..3477b1d 100644 --- a/parser.py +++ b/parser.py @@ -192,12 +192,86 @@ _gz_era_group = r'(?P' + '|'.join( # === 数字转中文 === def arabic_to_chinese_year(num: int) -> str: + """公元年份转中文(逐位体):1662 → 一六六二,-140 → -一四〇。 + 公元年份读作逐位数字,故不适用计数体规则(见 arabic_to_chinese_count)。""" digits = "〇一二三四五六七八九" if num < 0: return '-' + ''.join(digits[int(d)] for d in str(abs(num))) return ''.join(digits[int(d)] for d in str(num)) +# === 年数(纪年计数)转中文 === +# 规则(纪年计数体,用于"年数中文"字段): +# 1. 首位 1 → "元"(元年):民国元年、康熙元年、建元元年。 +# 2. 10-19 → 十、十一…十九(不写"一十")。 +# 3. 20-49 且十位即最高位 → 廿/卅/卌 缩写:20→廿、21→廿一、26→廿六、35→卅五、40→卌、49→卌九。 +# 更高位用全称(120→一百二十,不写"一百廿")。 +# 4. 50 以上用全称:五十、六十一、九十九、一百、二百五十、一千二百三十四。 +# 5. 空位补"〇"(不用"零"):105→一百〇五、1005→一千〇五、1050→一千〇五十; +# 连续空位只补一个〇(10005→一万〇五)。 +# 6. 万/亿 分级同普通话:一万、十万、一百万、一千〇五十万。 +_CN_UNITS = '〇一二三四五六七八九' +_CN_TENS_ABBR = {2: '廿', 3: '卅', 4: '卌'} + + +def _cn_under_10000(n: int, leading: bool) -> str: + """1 ≤ n ≤ 9999 的计数体中文;leading=True 表示该段是整个数的最高位段""" + digits = [n // 1000, n % 1000 // 100, n % 100 // 10, n % 10] + names = ['千', '百', '十', ''] + out = [] + zero_pending = False + for idx, (d, name) in enumerate(zip(digits, names)): + if d == 0: + if out: + zero_pending = True + continue + if zero_pending: + out.append('〇') + zero_pending = False + if name == '十' and leading and idx == 2 and not out and d < 5: + # 最高位就是十位:10-19 → 十…十九(不写"一十");20-49 → 廿/卅/卌 缩写 + out.append('十' if d == 1 else _CN_TENS_ABBR[d]) + else: + out.append(_CN_UNITS[d] + name) + return ''.join(out) + + +def arabic_to_chinese_count(num: int) -> str: + """年数(纪年计数)转中文:1→元、20→廿、26→廿六、105→一百〇五、1050→一千〇五十。 + 与 arabic_to_chinese_year 的逐位体不同,本函数按计数体读法书写。""" + if num < 0: + return '-' + arabic_to_chinese_count(-num) + if num == 0: + return '〇' + if num == 1: + return '元' + if num < 10000: + return _cn_under_10000(num, leading=True) + if num < 100000000: + high, low = divmod(num, 10000) + out = _cn_under_10000(high, leading=True) + '万' + else: + high, low = divmod(num, 100000000) + out = arabic_to_chinese_count(high) + '亿' + if low: + if low < 1000: # 低位段最高位低于"千"→ 中间有空位,补一个〇 + out += '〇' + out += _cn_under_10000(low, leading=False) + return out + + +def _insert_after(d: dict, key: str, new_key: str, value) -> dict: + """在 key 之后插入 new_key(保持字段相邻,便于前端展示)""" + out = {} + for k, v in d.items(): + out[k] = v + if k == key: + out[new_key] = value + if new_key not in out: + out[new_key] = value + return out + + # === 古代年号正则 === # 从 era_dict 构建朝代匹配组(按长度降序排列,优先匹配长朝代名如"北宋"而非"北") # 额外添加常见但不在CSV中的朝代简称(如"汉") @@ -835,6 +909,15 @@ def extract_era_years(text): if result.get("公元") is not None: result.setdefault("干支", gregorian_to_ganzhi(result["公元"])) + # 年数中文:纪年计数的中文写法(廿/卅/卌、空位作〇、首位1作元)。 + # 统一在收尾处补齐,任何新增的产生"年数"的分支都会自动带上该字段。 + for i, result in enumerate(results): + if "年数" in result: + results[i] = _insert_after( + result, "年数", "年数中文", + f"{arabic_to_chinese_count(result['年数'])}年", + ) + # === 收尾:原文剔除标注符号;位置换算为 UTF-16 码元(JS 前端索引) === for result in results: start, end = result["位置"]["起始"], result["位置"]["结束"]