Files
bililive-touhou-tts/chinese2kana.py
T
chun_qiu fd5c732d8a feat: web UI + QR login + portable build
- Add web UI panel (dark theme) at / with REST API
- Add Bilibili QR code scan login with browser header emulation
- Add wanakana for English -> katakana conversion
- Add volume control (0%-200%)
- Add message format with '说、' pause marker
- PyInstaller --onedir build for portable exe
- Rate-limit danmaku logging (1/20)
- Fix aiohttp.access log spam
- Add README.md
2026-08-08 13:22:58 +08:00

95 lines
2.9 KiB
Python

"""Chinese text → Japanese Katakana converter."""
import re
from pypinyin import pinyin, Style
from pinyin2kana_data import PINYIN2KANA
_CHINESE_CHAR_RE = re.compile(r"[\u4e00-\u9fa5]")
_CHINESE_SEGMENT_RE = re.compile(r"([\u4e00-\u9fa5]+)")
_NUMBER_RE = re.compile(r"-?\d+(\.\d+)?")
_TONE_RE = re.compile(r"\d")
_KANA_SAFE_RE = re.compile(
r"[\u3040-\u309F\u30A0-\u30FF\uFF65-\uFF9F"
r"\u30FC\u309B\u309C"
r"\u3001\u3002\uFF01\uFF1F"
r"\u300C\u300D"
r"\u30FB\u3000"
r"a-zA-Z0-9 ]"
)
_CHINESE_DIGITS = ["零", "一", "二", "三", "四", "五", "六", "七", "八", "九"]
_CHINESE_UNITS = ["", "十", "百", "千", "万"]
_CHINESE_POINT = "点"
def _number_to_chinese(num_str: str) -> str:
num_str = num_str.strip("-")
if "." in num_str:
integer_part, decimal_part = num_str.split(".", 1)
else:
integer_part, decimal_part = num_str, ""
result = ""
if integer_part == "0" or integer_part == "":
result = "零"
else:
digits = [int(ch) for ch in integer_part]
n = len(digits)
for i, d in enumerate(digits):
pos = n - i - 1
if d == 0:
if i < n - 1 and digits[i + 1] != 0:
result += _CHINESE_DIGITS[0]
else:
result += _CHINESE_DIGITS[d]
unit_idx = pos % 4
wan = pos // 4
if wan > 0 and unit_idx == 0:
result += _CHINESE_UNITS[4]
else:
result += _CHINESE_UNITS[unit_idx]
if decimal_part:
result += _CHINESE_POINT
for ch in decimal_part:
result += _CHINESE_DIGITS[int(ch)]
return result
def _strip_tone(py: str) -> str:
return _TONE_RE.sub("", py)
def chinese_to_kana(text: str, convert_numbers: bool = True) -> str:
if not text or not text.strip():
return ""
working = text
if convert_numbers:
def _replace_num(m: re.Match) -> str:
return _number_to_chinese(m.group(0))
working = _NUMBER_RE.sub(_replace_num, working)
if not _CHINESE_CHAR_RE.search(working):
return text
segments = _CHINESE_SEGMENT_RE.split(working)
result_parts: list[str] = []
for seg in segments:
if not seg:
continue
if _CHINESE_CHAR_RE.match(seg[0]):
py_list = pinyin(seg, style=Style.TONE3, heteronym=False)
for py_item in py_list:
py_raw = py_item[0]
py_plain = _strip_tone(py_raw)
kana = PINYIN2KANA.get(py_plain, py_plain)
result_parts.append(kana)
else:
result_parts.append(seg)
result = "".join(result_parts)
result = re.sub(r"(?<=\S) (?=\S)", "", result)
return result
def filter_kana(text: str) -> str:
filtered = "".join(c for c in text if _KANA_SAFE_RE.match(c))
return filtered.strip().lower()