"""Chinese text → Japanese Katakana converter.""" import re from pypinyin import pinyin, Style from pinyin2kana_data import PINYIN2KANA _CHINESE_CHAR_RE = re.compile(r"[\u4e00-\u9fa5]") _CHINESE_SEGMENT_RE = re.compile(r"([\u4e00-\u9fa5]+)") _NUMBER_RE = re.compile(r"-?\d+(\.\d+)?") _TONE_RE = re.compile(r"\d") _KANA_SAFE_RE = re.compile( r"[\u3040-\u309F\u30A0-\u30FF\uFF65-\uFF9F" r"\u30FC\u309B\u309C" r"\u3001\u3002\uFF01\uFF1F" r"\u300C\u300D" r"\u30FB\u3000" r"a-zA-Z0-9 ]" ) _CHINESE_DIGITS = ["零", "一", "二", "三", "四", "五", "六", "七", "八", "九"] _CHINESE_UNITS = ["", "十", "百", "千", "万"] _CHINESE_POINT = "点" def _number_to_chinese(num_str: str) -> str: num_str = num_str.strip("-") if "." in num_str: integer_part, decimal_part = num_str.split(".", 1) else: integer_part, decimal_part = num_str, "" result = "" if integer_part == "0" or integer_part == "": result = "零" else: digits = [int(ch) for ch in integer_part] n = len(digits) for i, d in enumerate(digits): pos = n - i - 1 if d == 0: if i < n - 1 and digits[i + 1] != 0: result += _CHINESE_DIGITS[0] else: result += _CHINESE_DIGITS[d] unit_idx = pos % 4 wan = pos // 4 if wan > 0 and unit_idx == 0: result += _CHINESE_UNITS[4] else: result += _CHINESE_UNITS[unit_idx] if decimal_part: result += _CHINESE_POINT for ch in decimal_part: result += _CHINESE_DIGITS[int(ch)] return result def _strip_tone(py: str) -> str: return _TONE_RE.sub("", py) def chinese_to_kana(text: str, convert_numbers: bool = True) -> str: if not text or not text.strip(): return "" working = text if convert_numbers: def _replace_num(m: re.Match) -> str: return _number_to_chinese(m.group(0)) working = _NUMBER_RE.sub(_replace_num, working) if not _CHINESE_CHAR_RE.search(working): return text segments = _CHINESE_SEGMENT_RE.split(working) result_parts: list[str] = [] for seg in segments: if not seg: continue if _CHINESE_CHAR_RE.match(seg[0]): py_list = pinyin(seg, style=Style.TONE3, heteronym=False) for py_item in py_list: py_raw = py_item[0] py_plain = _strip_tone(py_raw) kana = PINYIN2KANA.get(py_plain, py_plain) result_parts.append(kana) else: result_parts.append(seg) result = "".join(result_parts) result = re.sub(r"(?<=\S) (?=\S)", "", result) return result def filter_kana(text: str) -> str: filtered = "".join(c for c in text if _KANA_SAFE_RE.match(c)) return filtered.strip().lower()