japanese/scripts/analysis/analyze_phonetics.py
panli ef3df73166 refactor: 三层架构重构 + 配置文件 + 项目目录整理
清洗管线重构为严格三层架构:
- tango_analyser.py(底层:单词分析)
- task_processor.py(中层:文件 I/O、桶管理)
- workflow.py(顶层:状态机、任务推进)
- 移除旧的 batch_processor.py

新增配置文件系统:
- config.py:TOML 配置,相对路径相对配置文件目录解析
- 查找优先级 --config > cwd > 项目根 > ~ > 默认值
- count=None 语义为处理到文件末尾

项目目录整理:
- 根脚本归档到 scripts/analysis 与 scripts/legacy
- 文档归档到 docs/{design,history,analysis}
- 临时报告移到 reports/(已 gitignore)

文档质量:
- 新增 .markdownlint.json 与 scripts/mdlint.cmd
- 修复全部 14 个 md 文件的 markdownlint 警告

测试:74 passed(8 cleaner + 31 workflow + 6 tango + 15 validator + 14 config)
2026-08-19 19:40:02 +08:00

195 lines
6.8 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
日语汉字发音规律分析脚本
从 tests/data/*.txt 词表中提取「汉字 -> 中文拼音 -> 日语假名」对应样本,
统计中文读音与日语音读之间的系统性规律。
用法:
python analyze_phonetics.py [词表文件...]
不传参数时默认分析 tests/data/xinbiaori.txt
"""
from __future__ import annotations
import sys
import re
from collections import defaultdict
from pathlib import Path
# 中文声母表(按长度降序,保证 zh/ch/sh 优先于 z/c/s
INITIALS = [
"zh", "ch", "sh",
"b", "p", "m", "f", "d", "t", "n", "l",
"g", "k", "h", "j", "q", "x", "r", "z", "c", "s", "y", "w",
]
# 罗马音 -> 判断日语音读是否含长音/拨音等特征
LONG_VOWEL_TAILS = ("", "", "")
NASAL_TAIL = ""
def split_pinyin(py: str) -> tuple[str, str]:
"""把拼音拆成 (声母, 韵母)。无声母时声母为空串。"""
py = re.sub(r"[0-9]", "", py.strip().lower()) # 去掉可能的声调数字
for ini in INITIALS:
if py.startswith(ini):
return ini, py[len(ini):]
return "", py
def load_samples(paths: list[Path]) -> list[tuple[str, str, str]]:
"""返回 [(汉字, 拼音, 假名), ...],只保留三段等长且拼音非空的项。"""
samples: list[tuple[str, str, str]] = []
for path in paths:
if not path.is_file():
print(f"[跳过] 文件不存在: {path}")
continue
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split(":")
if len(parts) != 3:
continue
kanji = parts[0].split("|")
kana = parts[1].split("|")
pinyin = parts[2].split("|")
if len({len(kanji), len(kana), len(pinyin)}) != 1:
continue
for k, kn, p in zip(kanji, kana, pinyin):
if p and re.search(r"[\u4e00-\u9fff]", k):
samples.append((k, p, kn))
return samples
def analyze(samples: list[tuple[str, str, str]]) -> str:
out: list[str] = []
w = out.append
w("=" * 70)
w("日语汉字发音规律分析报告")
w("=" * 70)
w(f"总样本数(汉字-拼音-假名对应): {len(samples)}")
# ---- 1. 韵母 -> 音读韵尾特征 ----
# 统计中文韵母是否以 -ng / -n 结尾,与日语假名是否长音/拨音结尾的关系
final_tail = defaultdict(lambda: defaultdict(int)) # 中文韵尾类型 -> 日语尾音类型 -> 计数
for _, py, kana in samples:
_, final = split_pinyin(py)
if final.endswith("ng"):
cn = "-ng (后鼻音)"
elif final.endswith("n"):
cn = "-n (前鼻音)"
else:
cn = "其他(元音结尾)"
if kana.endswith(NASAL_TAIL):
jp = "假名以 ん 结尾"
elif kana.endswith(LONG_VOWEL_TAILS):
jp = "假名以 う/い 结尾(长音)"
else:
jp = "假名其他结尾"
final_tail[cn][jp] += 1
w("")
w("-" * 70)
w("【规律一】中文韵尾 → 日语音读韵尾")
w("-" * 70)
w("核心规律:中文后鼻音 -ng 多对应日语长音(う/い);前鼻音 -n 多对应 ん。")
w("")
for cn in ["-ng (后鼻音)", "-n (前鼻音)", "其他(元音结尾)"]:
row = final_tail.get(cn, {})
total = sum(row.values())
if not total:
continue
w(f"中文 {cn} (共 {total} 例):")
for jp, cnt in sorted(row.items(), key=lambda x: -x[1]):
w(f" {jp:<24} {cnt:>4} 例 ({cnt*100//total}%)")
# ---- 2. 声母 -> 日语首假名(行) ----
initial_head = defaultdict(lambda: defaultdict(int)) # 声母 -> 日语首假名 -> 计数
for _, py, kana in samples:
ini, _ = split_pinyin(py)
if not kana:
continue
head = kana[0]
initial_head[ini][head] += 1
w("")
w("-" * 70)
w("【规律二】中文声母 → 日语音读首假名")
w("-" * 70)
w("列出每个中文声母最常对应的日语首假名(取前 3)。")
w("")
for ini in sorted(initial_head, key=lambda x: -sum(initial_head[x].values())):
row = initial_head[ini]
total = sum(row.values())
if total < 3: # 样本太少略过
continue
top = sorted(row.items(), key=lambda x: -x[1])[:3]
top_str = " ".join(f"{h}({c})" for h, c in top)
label = ini if ini else "(零声母)"
w(f"声母 {label:<4}{total:>4}例 → {top_str}")
# ---- 3. 同一汉字的多音读(音读/训读现象) ----
kanji_readings: dict[str, set[str]] = defaultdict(set)
for k, _, kana in samples:
kanji_readings[k].add(kana)
multi = {k: v for k, v in kanji_readings.items() if len(v) > 1}
w("")
w("-" * 70)
w("【规律三】多音读汉字(同一汉字出现多种假名读法)")
w("-" * 70)
w(f"{len(kanji_readings)} 个不同汉字,其中 {len(multi)} 个有 2 种以上读法。")
w("这些多为「音读 vs 训读」或不同音读,是记忆难点,建议重点关注:")
w("")
for k, readings in sorted(multi.items(), key=lambda x: -len(x[1]))[:30]:
w(f" {k} : {' / '.join(sorted(readings))}")
# ---- 4. 拼音相同 -> 音读是否相同(同音字规律) ----
pinyin_kana: dict[str, set[str]] = defaultdict(set)
pinyin_kanji: dict[str, set[str]] = defaultdict(set)
for k, py, kana in samples:
base = re.sub(r"[0-9]", "", py.lower())
pinyin_kana[base].add(kana)
pinyin_kanji[base].add(k)
w("")
w("-" * 70)
w("【规律四】中文同音字 → 日语音读高度一致的例子")
w("-" * 70)
w("中文读音相同的汉字,日语音读也常相同。以下是同一拼音、")
w("多个汉字却共享同一日语读音的典型组(最能体现规律)")
w("")
shown = 0
for py in sorted(pinyin_kanji, key=lambda x: -len(pinyin_kanji[x])):
kanjis = pinyin_kanji[py]
kanas = pinyin_kana[py]
if len(kanjis) >= 3 and len(kanas) <= 2:
w(f" 拼音 {py:<6} → 汉字 {' '.join(sorted(kanjis))}")
w(f" {'':>11} 日语读音 {' '.join(sorted(kanas))}")
w("")
shown += 1
if shown >= 15:
break
return "\n".join(out)
def main() -> None:
root = Path(__file__).resolve().parent
if len(sys.argv) > 1:
paths = [Path(a) for a in sys.argv[1:]]
else:
paths = [root / "tests" / "data" / "xinbiaori.txt"]
samples = load_samples(paths)
report = analyze(samples)
out_file = root / "phonetics_report.txt"
out_file.write_text(report, encoding="utf-8")
print(f"分析完成,样本 {len(samples)} 条,报告已写入: {out_file}")
if __name__ == "__main__":
main()