Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f4c99b1692 | ||
|
|
f27bcfb055 | ||
|
|
cb22e074af | ||
|
|
0c0e63161d |
@@ -8,7 +8,7 @@ Bilibili 直播弹幕 + ゆっくり TTS 语音朗读器。
|
|||||||
|
|
||||||
- 实时接收 Bilibili 直播弹幕
|
- 实时接收 Bilibili 直播弹幕
|
||||||
- 中文 → 拼音 → 片假名自动转换
|
- 中文 → 拼音 → 片假名自动转换
|
||||||
- 英文按罗马音转片假名(wanakana)
|
- 英文转片假名(english-to-kana 49K 词库 + 字母拼读兜底)
|
||||||
- 可选扫码登录(获取未打码用户名)
|
- 可选扫码登录(获取未打码用户名)
|
||||||
- Web 控制面板(暗色主题)
|
- Web 控制面板(暗色主题)
|
||||||
- 支持 8 种ゆっくり音色 + 语速/音量调节
|
- 支持 8 种ゆっくり音色 + 语速/音量调节
|
||||||
@@ -94,7 +94,7 @@ Bilibili 弹幕 (blivedm)
|
|||||||
中文→片假名 (pypinyin + 映射表)
|
中文→片假名 (pypinyin + 映射表)
|
||||||
│
|
│
|
||||||
▼
|
▼
|
||||||
英文→片假名 (wanakana)
|
英文→片假名 (english-to-kana 49K 词库 + 字母拼读兜底)
|
||||||
│
|
│
|
||||||
▼
|
▼
|
||||||
AquesTalk TTS (aquestalk.js + v86 WASM 模拟)
|
AquesTalk TTS (aquestalk.js + v86 WASM 模拟)
|
||||||
@@ -121,3 +121,48 @@ python build.py
|
|||||||
## 许可
|
## 许可
|
||||||
|
|
||||||
MIT License
|
MIT License
|
||||||
|
|
||||||
|
```
|
||||||
|
.
|
||||||
|
-==- +
|
||||||
|
::. %*%#*:
|
||||||
|
--.. #***:*+********-. .-+*#%%%%%%%##*+-: ..::-- -.
|
||||||
|
.*---%#=.=*+*+*+******+#@%%#%%%#%%%#%#%#%%%%#%#%%#%@%- .:-+###**+***+:+**:.-#+=
|
||||||
|
::+.-+++#+####****+**%#%%#%%#%%#%%#%#%%#%%#%#%%%%#%%#%#%%@@%#+**+****+**+**=-+#*::-+:
|
||||||
|
*-. #+.=+*+***%%%%%%#%%#%%#%#%#%#%#%%#%%#%#%#%#%#%%#%#%%%@@%%##*+***+**++*:=**#-.-+.
|
||||||
|
*###**::*%%%#%#%#%#%#%%#%%#%%#%%#%%%#%#%#%%#%#%#%%#%#%%#%#%@@@@@%#*+*+**+**=-::-*#. =
|
||||||
|
+-#=%##%#%#%#%#%%%#%#%#%#%#%#%%#%%#%#%%%#%#%%%%#%%#%%%#%%#%#%#%@@@@@#########=+*+*+=:+.
|
||||||
|
+---:%#%%%#%%#%%#%#%%%#%%#%%#%#%%#%#%%#%#%%%#%#%%#%%#%#%#%%#%#%%#%@%@@@######*++##--==
|
||||||
|
=-##%%#%#%#%%#%%#%#%#%#%%#%%%#%#%%#%#%#%#%#%#%#%#%#%#%%%#%%%%#%#%#%@%@@%##**+++%#==-
|
||||||
|
.%%%#%#%%%#%%#%#%%%#%%%%#%#%#%%#%#%%%#%%%#%%%#%%#%%#%#%#%@##%%#%%%%%%@%#+%##*+#: :-:+
|
||||||
|
:==#%#%%#%%%%#%%#%#%%#%@#%%%%%#%%#%%%%%#%#%#%#%#%%#%%%%%%%%@%%#%%##+-=-=-=-%#*++##**=:+
|
||||||
|
===--=--+%%%#%@%#%%#%%@@%%%#%@%%%%%#%@%#%#%%%#%@%%#%#%#%@%#%@**=--=--=+#%@@@%*+#%%=-%*
|
||||||
|
:%#%#@%%%%==-----+*%%@@%@#%%%@%%@@#%%#@@%%%#%@%%#@%%%#*+-----=##@%#%@@@%%@%@%@%%%*-+:%@:
|
||||||
|
%%#%@%%@@#%@@%%@%@%%@%@@%%%#%@@%@%%#%#@@@%%%%%@%@%@@%%%%@%@@@@%@@@%%%@@@%@@@@@%-+%##%@%*
|
||||||
|
*%#%@@%@@#@@%@%@@@@%%@@@@%%%%@@@@#==%@%@%@@%%#%%@@@@@%%%%%%%@%@@@%@%%#%@@@@%@%@@#######%@.
|
||||||
|
.%%%%@%@@%%@%@%%@%@%@@@%@%#@@%@%@#=-:*%@%@@%#%+%%%@%@%@%@@#%@@@%@@@@@%%%%%@@@@@%@####*+*+*%.
|
||||||
|
=%#@@@@@%%@@@%@@@@+@@%@#-#%+@@@@#-=:..*@@@%@=%*=+%@@@@@@%@%%%@@@%@@%@@%%#%@%@%@@@###*+*####%*-*
|
||||||
|
*%#@%@%@%@%@%@%@%-#%@%*+=##-@%@#==:....+@@@%-#*.:=@%@@=-%@@=*@%@@@%@@@@@%%%@@@@%@%##**+*--:++*-
|
||||||
|
##@@@%%#@@@@%@@*==*@====-%--+@*-=:... ..-%@*=.#%-=*#%%+=-+@*-=%@%@@@%@@%@@%%@%@@@%*+**#+-*%*+.
|
||||||
|
*%@%@- @%@%@@@+==#+ =####%-....=-.........+-:.# :######%=*+%.-+@@@%@@@@@@%@@@@@%@%*####+=%%.
|
||||||
|
=%%%. .@@@-#@=-=*:####*##%... ..... ... ......%####*##*%..*::.:@%@@*%%#%@@@%@%@@@#-:==+@@@@.
|
||||||
|
:%# -@--%%%--:- %******#........... .... ...*********+ +..+ *.=.-++-%@@@@@@@%@@%@%#%@@%@:
|
||||||
|
: :-#-==#:::-.=%#%%%%*.. ... .. .......... #######+ .:.:..=*=+#+#%==*%@%@%@@@@%@@@@@@@@:
|
||||||
|
#***%-:---+.-................... .. .......---=+--=:-:-**+###%=*=*@@@@@%@@@@%@%@%@@:
|
||||||
|
.#***%.:...:.......... .. .... ................:.:.:.:.-****###=-==%@%@@@%@%@@@@@@%@.
|
||||||
|
.#*+##...:...:............. ...... .. .................-*+*####=-=-@@@%@@@@@@%@%@@@%.
|
||||||
|
#***#............. .-... .........* ... ..............:**+###*=-=*%@@@%@@%@@@@@@@%%
|
||||||
|
.#+**%.................#@**+========-..................:****##*=-+%@%@@@%@@@%@%@%@@%
|
||||||
|
:=+=*##:......... ... ...*++========*..... .............:+-+*+#*=#@@@@@%@@@%@@@@@@*%*
|
||||||
|
#.=*--*...... .... .... .:#====-==-#.. ..... .........=-*=#*#%+%@%@%@@@@%@@@@%@%@-@=
|
||||||
|
-+ %@%-.. .................*+===*=....... .... .. ...=.+=-+#+=@%%@@%@@%@@@%@@@@*:%:
|
||||||
|
:%%@%@@-..... .. ... ... ......... .. .................@%+@@@@%#%@%@#%@@@%=%@%@-:#
|
||||||
|
:%%@@@%%#.. .......... ....... ........ ... .. ... . ==%#@%@%%%%%@@%-@@%@#:@@@% :*
|
||||||
|
:%%%@%.%%@#:... .. ......... .... .. .... .........+- -%%%@@@%#%@%@:-@#-@+ %@%= =.
|
||||||
|
:%%@@: %%-%@@*:..... ... ................... .. :+. =#@@%=#*%%@@- =@:.@: #@#
|
||||||
|
:%@@- %#.-@@@@#%*:....... .. .. .... .. ...:=+. #%@% -+-#@%= *- .#. *%.
|
||||||
|
+%@: #: @%@- *- .=*+=-.... ... ...-++=. .%@# . .%@= : = *.
|
||||||
|
*@: + %# =@- %:
|
||||||
|
#. .*
|
||||||
|
|
||||||
|
```
|
||||||
@@ -135,13 +135,15 @@ def main() -> int:
|
|||||||
safe_copy_tree(aq_src / "node_modules", aq_dst / "node_modules",
|
safe_copy_tree(aq_src / "node_modules", aq_dst / "node_modules",
|
||||||
ignore_patterns=("bililive-touhou-tts",))
|
ignore_patterns=("bililive-touhou-tts",))
|
||||||
|
|
||||||
# ── 4. kuroshiro node_modules ──────────────────────────────────────
|
# ── 4. node_modules (if any) ───────────────────────────────────────
|
||||||
step("Copying kuroshiro node_modules")
|
if (ROOT / "node_modules").exists():
|
||||||
safe_copy_tree(ROOT / "node_modules", BUILD_DIR / "node_modules")
|
step("Copying node_modules")
|
||||||
|
safe_copy_tree(ROOT / "node_modules", BUILD_DIR / "node_modules")
|
||||||
|
|
||||||
# ── 5. Bridge + config ─────────────────────────────────────────────
|
# ── 5. Bridge + config ─────────────────────────────────────────────
|
||||||
step("Copying bridge files")
|
step("Copying bridge files")
|
||||||
shutil.copy2(ROOT / "tts_bridge.js", BUILD_DIR / "tts_bridge.js")
|
shutil.copy2(ROOT / "tts_bridge.js", BUILD_DIR / "tts_bridge.js")
|
||||||
|
shutil.copy2(ROOT / "english-kana-matcher.js", BUILD_DIR / "english-kana-matcher.js")
|
||||||
shutil.copy2(ROOT / "package.json", BUILD_DIR / "package.json")
|
shutil.copy2(ROOT / "package.json", BUILD_DIR / "package.json")
|
||||||
print(" OK")
|
print(" OK")
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -16,7 +16,7 @@ _KANA_SAFE_RE = re.compile(
|
|||||||
r"\u3001\u3002\uFF01\uFF1F"
|
r"\u3001\u3002\uFF01\uFF1F"
|
||||||
r"\u300C\u300D"
|
r"\u300C\u300D"
|
||||||
r"\u30FB\u3000"
|
r"\u30FB\u3000"
|
||||||
r"a-zA-Z0-9 ]"
|
r"a-zA-Z0-9 ,.!?\uFF0C]"
|
||||||
)
|
)
|
||||||
|
|
||||||
_CHINESE_DIGITS = ["零", "一", "二", "三", "四", "五", "六", "七", "八", "九"]
|
_CHINESE_DIGITS = ["零", "一", "二", "三", "四", "五", "六", "七", "八", "九"]
|
||||||
@@ -85,7 +85,7 @@ def chinese_to_kana(text: str, convert_numbers: bool = True) -> str:
|
|||||||
else:
|
else:
|
||||||
result_parts.append(seg)
|
result_parts.append(seg)
|
||||||
result = "".join(result_parts)
|
result = "".join(result_parts)
|
||||||
result = re.sub(r"(?<=\S) (?=\S)", "", result)
|
result = re.sub(r"(?<=[^\x00-\x7F]) (?=[^\x00-\x7F])", "", result)
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
File diff suppressed because one or more lines are too long
@@ -12,7 +12,7 @@ import signal
|
|||||||
import sys
|
import sys
|
||||||
|
|
||||||
from danmaku_handler import DanmakuClient
|
from danmaku_handler import DanmakuClient
|
||||||
from chinese2kana import chinese_to_kana
|
from chinese2kana import chinese_to_kana, filter_kana
|
||||||
from tts import TTSBridge
|
from tts import TTSBridge
|
||||||
from audio_player import play_wav_async, get_output_devices
|
from audio_player import play_wav_async, get_output_devices
|
||||||
|
|
||||||
@@ -47,6 +47,7 @@ async def tts_worker(queue: asyncio.Queue, bridge: TTSBridge,
|
|||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
kana = chinese_to_kana(text, convert_numbers=convert_numbers)
|
kana = chinese_to_kana(text, convert_numbers=convert_numbers)
|
||||||
|
kana = filter_kana(kana)
|
||||||
logger.info("Speaking: %s -> %s", text, kana)
|
logger.info("Speaking: %s -> %s", text, kana)
|
||||||
wav_data = await bridge.synthesize(kana)
|
wav_data = await bridge.synthesize(kana)
|
||||||
await play_wav_async(wav_data)
|
await play_wav_async(wav_data)
|
||||||
|
|||||||
Generated
+20
-7
@@ -6,17 +6,30 @@
|
|||||||
"": {
|
"": {
|
||||||
"name": "bililive-touhou-tts",
|
"name": "bililive-touhou-tts",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"wanakana": "^5.3.1"
|
"phonemize": "^1.2.0"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"node_modules/wanakana": {
|
"node_modules/number-to-words": {
|
||||||
"version": "5.3.1",
|
"version": "1.2.4",
|
||||||
"resolved": "https://registry.npmmirror.com/wanakana/-/wanakana-5.3.1.tgz",
|
"resolved": "https://registry.npmmirror.com/number-to-words/-/number-to-words-1.2.4.tgz",
|
||||||
"integrity": "sha512-OSDqupzTlzl2LGyqTdhcXcl6ezMiFhcUwLBP8YKaBIbMYW1wAwDvupw2T9G9oVaKT9RmaSpyTXjxddFPUcFFIw==",
|
"integrity": "sha512-/fYevVkXRcyBiZDg6yzZbm0RuaD6i0qRfn8yr+6D0KgBMOndFPxuW10qCHpzs50nN8qKuv78k8MuotZhcVX6Pw==",
|
||||||
|
"license": "MIT"
|
||||||
|
},
|
||||||
|
"node_modules/phonemize": {
|
||||||
|
"version": "1.2.0",
|
||||||
|
"resolved": "https://registry.npmmirror.com/phonemize/-/phonemize-1.2.0.tgz",
|
||||||
|
"integrity": "sha512-+zEpOPXrEaylYCIXMSDVhiAwFbVzIXRf+7vheuNxozg4hLKbQVDXCOpI0GAJw41xEgE9LcwMjLUN30aWrpwSkw==",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"engines": {
|
"dependencies": {
|
||||||
"node": ">=12"
|
"number-to-words": "^1.2.4",
|
||||||
|
"pinyin-pro": "^3.26.0"
|
||||||
}
|
}
|
||||||
|
},
|
||||||
|
"node_modules/pinyin-pro": {
|
||||||
|
"version": "3.28.2",
|
||||||
|
"resolved": "https://registry.npmmirror.com/pinyin-pro/-/pinyin-pro-3.28.2.tgz",
|
||||||
|
"integrity": "sha512-jV38yxXHLfidirMC4hrXasLDozLCSq/4DfX88GnHcSEJ2+GpSedG6I9VOiEXJu6iQ5dbJC/RjmzyMuS5h/wH5A==",
|
||||||
|
"license": "MIT"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -2,6 +2,6 @@
|
|||||||
"name": "bililive-touhou-tts",
|
"name": "bililive-touhou-tts",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"wanakana": "^5.3.1"
|
"phonemize": "^1.2.0"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -37,6 +37,7 @@ class TTSService:
|
|||||||
self._shutdown_event: asyncio.Event | None = None
|
self._shutdown_event: asyncio.Event | None = None
|
||||||
self._worker_task: asyncio.Task | None = None
|
self._worker_task: asyncio.Task | None = None
|
||||||
self._cookies: dict[str, str] | None = None
|
self._cookies: dict[str, str] | None = None
|
||||||
|
self._lifecycle_lock = asyncio.Lock()
|
||||||
self.running = False
|
self.running = False
|
||||||
self.start_time: float = 0
|
self.start_time: float = 0
|
||||||
self.messages_processed = 0
|
self.messages_processed = 0
|
||||||
@@ -48,49 +49,51 @@ class TTSService:
|
|||||||
self._cookies = cookies
|
self._cookies = cookies
|
||||||
|
|
||||||
async def start(self, config: dict) -> None:
|
async def start(self, config: dict) -> None:
|
||||||
if self.running:
|
async with self._lifecycle_lock:
|
||||||
raise RuntimeError("Already running")
|
if self.running:
|
||||||
room_id = config["room_id"]
|
raise RuntimeError("Already running")
|
||||||
voice = config.get("voice", "f1")
|
room_id = config["room_id"]
|
||||||
speed = config.get("speed", 100)
|
voice = config.get("voice", "f1")
|
||||||
self._volume = config.get("volume", 100)
|
speed = config.get("speed", 100)
|
||||||
message_format = config.get("format", "{uname}\u8bf4\u3001 {msg}")
|
self._volume = config.get("volume", 100)
|
||||||
convert_numbers = config.get("convert_numbers", True)
|
message_format = config.get("format", "{uname}\u8bf4\u3001 {msg}")
|
||||||
self.current_config = config
|
convert_numbers = config.get("convert_numbers", True)
|
||||||
self.recent_messages = []
|
self.current_config = config
|
||||||
self.messages_processed = 0
|
self.recent_messages = []
|
||||||
self.start_time = time.time()
|
self.messages_processed = 0
|
||||||
self._queue = asyncio.Queue(maxsize=256)
|
self.start_time = time.time()
|
||||||
self._shutdown_event = asyncio.Event()
|
self._queue = asyncio.Queue(maxsize=256)
|
||||||
self._bridge = TTSBridge(voice=voice, speed=speed)
|
self._shutdown_event = asyncio.Event()
|
||||||
await self._bridge.start()
|
self._bridge = TTSBridge(voice=voice, speed=speed)
|
||||||
self._danmaku_client = DanmakuClient(
|
await self._bridge.start()
|
||||||
room_id=room_id, queue=self._queue,
|
self._danmaku_client = DanmakuClient(
|
||||||
cookies=self._cookies, message_format=message_format,
|
room_id=room_id, queue=self._queue,
|
||||||
)
|
cookies=self._cookies, message_format=message_format,
|
||||||
await self._danmaku_client.start()
|
)
|
||||||
self.running = True
|
await self._danmaku_client.start()
|
||||||
self._worker_task = asyncio.create_task(self._tts_worker(convert_numbers))
|
self.running = True
|
||||||
|
self._worker_task = asyncio.create_task(self._tts_worker(convert_numbers))
|
||||||
|
|
||||||
async def stop(self) -> None:
|
async def stop(self) -> None:
|
||||||
if not self.running:
|
async with self._lifecycle_lock:
|
||||||
return
|
if not self.running:
|
||||||
self.running = False
|
return
|
||||||
if self._shutdown_event:
|
self.running = False
|
||||||
self._shutdown_event.set()
|
if self._shutdown_event:
|
||||||
if self._worker_task:
|
self._shutdown_event.set()
|
||||||
self._worker_task.cancel()
|
if self._worker_task:
|
||||||
try:
|
self._worker_task.cancel()
|
||||||
await self._worker_task
|
try:
|
||||||
except asyncio.CancelledError:
|
await self._worker_task
|
||||||
pass
|
except asyncio.CancelledError:
|
||||||
if self._danmaku_client:
|
pass
|
||||||
await self._danmaku_client.stop()
|
if self._danmaku_client:
|
||||||
if self._bridge:
|
await self._danmaku_client.stop()
|
||||||
await self._bridge.stop()
|
if self._bridge:
|
||||||
self._queue = None
|
await self._bridge.stop()
|
||||||
self._bridge = None
|
self._queue = None
|
||||||
self._danmaku_client = None
|
self._bridge = None
|
||||||
|
self._danmaku_client = None
|
||||||
|
|
||||||
async def _tts_worker(self, convert_numbers: bool) -> None:
|
async def _tts_worker(self, convert_numbers: bool) -> None:
|
||||||
total_in = 0
|
total_in = 0
|
||||||
|
|||||||
@@ -45,6 +45,8 @@ class TTSBridge:
|
|||||||
self._process: asyncio.subprocess.Process | None = None
|
self._process: asyncio.subprocess.Process | None = None
|
||||||
self._bridge_script = bridge_script or _resolve_path("tts_bridge.js")
|
self._bridge_script = bridge_script or _resolve_path("tts_bridge.js")
|
||||||
self._node_exe = _get_node_exe()
|
self._node_exe = _get_node_exe()
|
||||||
|
self._stderr_task: asyncio.Task | None = None
|
||||||
|
self._io_lock = asyncio.Lock()
|
||||||
|
|
||||||
async def start(self) -> None:
|
async def start(self) -> None:
|
||||||
if not pathlib.Path(self._bridge_script).exists():
|
if not pathlib.Path(self._bridge_script).exists():
|
||||||
@@ -61,6 +63,7 @@ class TTSBridge:
|
|||||||
stdout=asyncio.subprocess.PIPE,
|
stdout=asyncio.subprocess.PIPE,
|
||||||
stderr=asyncio.subprocess.PIPE,
|
stderr=asyncio.subprocess.PIPE,
|
||||||
)
|
)
|
||||||
|
self._stderr_task = asyncio.create_task(self._drain_stderr())
|
||||||
|
|
||||||
line = await asyncio.wait_for(self._process.stdout.readline(), timeout=60)
|
line = await asyncio.wait_for(self._process.stdout.readline(), timeout=60)
|
||||||
line_str = line.decode("utf-8").strip()
|
line_str = line.decode("utf-8").strip()
|
||||||
@@ -73,47 +76,72 @@ class TTSBridge:
|
|||||||
|
|
||||||
logger.info("Node.js TTS bridge ready.")
|
logger.info("Node.js TTS bridge ready.")
|
||||||
|
|
||||||
|
async def _drain_stderr(self) -> None:
|
||||||
|
"""Continuously read bridge stderr and forward to Python logs."""
|
||||||
|
if self._process is None or self._process.stderr is None:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
raw = await self._process.stderr.readline()
|
||||||
|
if not raw:
|
||||||
|
break
|
||||||
|
line = raw.decode("utf-8", errors="replace").rstrip()
|
||||||
|
if line:
|
||||||
|
logger.info("[bridge] %s", line)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
pass
|
||||||
|
except Exception:
|
||||||
|
logger.debug("Bridge stderr drain stopped", exc_info=True)
|
||||||
|
|
||||||
async def synthesize(self, kana_text: str) -> bytes:
|
async def synthesize(self, kana_text: str) -> bytes:
|
||||||
if self._process is None or self._process.stdin is None:
|
if self._process is None or self._process.stdin is None:
|
||||||
raise RuntimeError("Bridge not started. Call start() first.")
|
raise RuntimeError("Bridge not started. Call start() first.")
|
||||||
|
|
||||||
tmpdir = pathlib.Path(tempfile.mkdtemp(prefix="tts_"))
|
async with self._io_lock:
|
||||||
try:
|
tmpdir = pathlib.Path(tempfile.mkdtemp(prefix="tts_"))
|
||||||
input_path = tmpdir / "input.txt"
|
try:
|
||||||
output_path = tmpdir / "output.wav"
|
input_path = tmpdir / "input.txt"
|
||||||
|
output_path = tmpdir / "output.wav"
|
||||||
|
|
||||||
input_path.write_text(kana_text, encoding="utf-8")
|
input_path.write_text(kana_text, encoding="utf-8")
|
||||||
|
|
||||||
cmd_line = f"{input_path}|{output_path}\n"
|
cmd_line = f"{input_path}|{output_path}\n"
|
||||||
self._process.stdin.write(cmd_line.encode("utf-8"))
|
self._process.stdin.write(cmd_line.encode("utf-8"))
|
||||||
await self._process.stdin.drain()
|
await self._process.stdin.drain()
|
||||||
|
|
||||||
line = await asyncio.wait_for(self._process.stdout.readline(), timeout=120)
|
line = await asyncio.wait_for(self._process.stdout.readline(), timeout=120)
|
||||||
line_str = line.decode("utf-8").strip()
|
line_str = line.decode("utf-8").strip()
|
||||||
|
|
||||||
if line_str.startswith("ERR:"):
|
if line_str.startswith("ERR:"):
|
||||||
raise RuntimeError(f"TTS synthesis error: {line_str[4:]}")
|
raise RuntimeError(f"TTS synthesis error: {line_str[4:]}")
|
||||||
|
|
||||||
if not line_str.startswith("OK:") or line_str[3:] != str(output_path):
|
if not line_str.startswith("OK:") or line_str[3:] != str(output_path):
|
||||||
raise RuntimeError(f"Unexpected bridge response: {line_str}")
|
raise RuntimeError(f"Unexpected bridge response: {line_str}")
|
||||||
|
|
||||||
wav_data = output_path.read_bytes()
|
wav_data = output_path.read_bytes()
|
||||||
return wav_data
|
return wav_data
|
||||||
|
|
||||||
finally:
|
finally:
|
||||||
for f in tmpdir.iterdir():
|
for f in tmpdir.iterdir():
|
||||||
|
try:
|
||||||
|
f.unlink()
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
try:
|
try:
|
||||||
f.unlink()
|
tmpdir.rmdir()
|
||||||
except OSError:
|
except OSError:
|
||||||
pass
|
pass
|
||||||
try:
|
|
||||||
tmpdir.rmdir()
|
|
||||||
except OSError:
|
|
||||||
pass
|
|
||||||
|
|
||||||
async def stop(self) -> None:
|
async def stop(self) -> None:
|
||||||
if self._process is not None:
|
if self._process is not None:
|
||||||
logger.info("Stopping TTS bridge...")
|
logger.info("Stopping TTS bridge...")
|
||||||
|
if self._stderr_task is not None:
|
||||||
|
self._stderr_task.cancel()
|
||||||
|
try:
|
||||||
|
await self._stderr_task
|
||||||
|
except (asyncio.CancelledError, Exception):
|
||||||
|
pass
|
||||||
|
self._stderr_task = None
|
||||||
try:
|
try:
|
||||||
if self._process.stdin is not None:
|
if self._process.stdin is not None:
|
||||||
self._process.stdin.close()
|
self._process.stdin.close()
|
||||||
|
|||||||
+236
-7
@@ -1,17 +1,40 @@
|
|||||||
/** tts_bridge.js — Persistent Node.js bridge for aquestalk.js TTS synthesis.
|
/** tts_bridge.js — Persistent Node.js bridge for aquestalk.js TTS synthesis.
|
||||||
|
|
||||||
Converts English/romaji to katakana via wanakana before synthesis.
|
English word → katakana pipeline (never skips):
|
||||||
|
1. english-to-kana dictionary (49K words, MIT, vendored english-kana-matcher.js)
|
||||||
|
2. phonemize (MIT) G2P → IPA → rule-based katakana transcription
|
||||||
|
3. letter-by-letter romanized spelling (last resort)
|
||||||
|
|
||||||
|
English punctuation (,.!?) is converted to Japanese pauses (、。).
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { readFileSync, writeFileSync } from "fs";
|
import { readFileSync, writeFileSync } from "fs";
|
||||||
import { createInterface } from "readline";
|
import { createInterface } from "readline";
|
||||||
import { fileURLToPath, pathToFileURL } from "url";
|
import { fileURLToPath, pathToFileURL } from "url";
|
||||||
import { dirname, join } from "path";
|
import { dirname, join } from "path";
|
||||||
import wanakana from "wanakana";
|
import { createRequire } from "module";
|
||||||
|
|
||||||
const __filename = fileURLToPath(import.meta.url);
|
const __filename = fileURLToPath(import.meta.url);
|
||||||
const __dirname = dirname(__filename);
|
const __dirname = dirname(__filename);
|
||||||
|
|
||||||
|
// ── vendored english-to-kana dictionary (auto-generated, MIT) ──────────
|
||||||
|
const matcherMod = pathToFileURL(join(__dirname, "english-kana-matcher.js")).href;
|
||||||
|
const { lookupKana } = await import(matcherMod);
|
||||||
|
|
||||||
|
// ── phonemize (G2P, MIT) via CJS entry (avoids JSON import issue in Node ESM) ──
|
||||||
|
const require = createRequire(import.meta.url);
|
||||||
|
let phonemize = null;
|
||||||
|
try {
|
||||||
|
phonemize = require("phonemize").phonemize;
|
||||||
|
console.error("[bridge] phonemize G2P loaded.");
|
||||||
|
} catch (err) {
|
||||||
|
console.error(`[bridge] phonemize unavailable (${err.message}), using letter fallback only.`);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── aquestalk.js ──────────────────────────────────────────────────────
|
||||||
|
const aquestalkMod = pathToFileURL(join(__dirname, "aquestalk.js", "dist", "index.js")).href;
|
||||||
|
const { load } = await import(aquestalkMod);
|
||||||
|
|
||||||
const args = process.argv.slice(2);
|
const args = process.argv.slice(2);
|
||||||
let voice = "f1";
|
let voice = "f1";
|
||||||
let speed = 100;
|
let speed = 100;
|
||||||
@@ -24,9 +47,6 @@ for (let i = 0; i < args.length; i++) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const aquestalkMod = pathToFileURL(join(__dirname, "aquestalk.js", "dist", "index.js")).href;
|
|
||||||
const { load } = await import(aquestalkMod);
|
|
||||||
|
|
||||||
console.error(`[bridge] Loading aquestalk.js with voice="${voice}", speed=${speed}...`);
|
console.error(`[bridge] Loading aquestalk.js with voice="${voice}", speed=${speed}...`);
|
||||||
|
|
||||||
let aq;
|
let aq;
|
||||||
@@ -39,11 +59,220 @@ try {
|
|||||||
process.exit(1);
|
process.exit(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── IPA → Katakana transcription (rule-based) ─────────────────────────
|
||||||
|
const KANA = {
|
||||||
|
k: { a: "カ", i: "キ", u: "ク", e: "ケ", o: "コ" },
|
||||||
|
g: { a: "ガ", i: "ギ", u: "グ", e: "ゲ", o: "ゴ" },
|
||||||
|
s: { a: "サ", i: "シ", u: "ス", e: "セ", o: "ソ" },
|
||||||
|
z: { a: "ザ", i: "ジ", u: "ズ", e: "ゼ", o: "ゾ" },
|
||||||
|
t: { a: "タ", i: "チ", u: "トゥ", e: "テ", o: "ト" },
|
||||||
|
d: { a: "ダ", i: "ジ", u: "ドゥ", e: "デ", o: "ド" },
|
||||||
|
n: { a: "ナ", i: "ニ", u: "ヌ", e: "ネ", o: "ノ" },
|
||||||
|
h: { a: "ハ", i: "ヒ", u: "フ", e: "ヘ", o: "ホ" },
|
||||||
|
b: { a: "バ", i: "ビ", u: "ブ", e: "ベ", o: "ボ" },
|
||||||
|
p: { a: "パ", i: "ピ", u: "プ", e: "ペ", o: "ポ" },
|
||||||
|
m: { a: "マ", i: "ミ", u: "ム", e: "メ", o: "モ" },
|
||||||
|
r: { a: "ラ", i: "リ", u: "ル", e: "レ", o: "ロ" },
|
||||||
|
f: { a: "ファ", i: "フィ", u: "フ", e: "フェ", o: "フォ" },
|
||||||
|
v: { a: "バ", i: "ビ", u: "ブ", e: "ベ", o: "ボ" },
|
||||||
|
tʃ: { a: "チャ", i: "チ", u: "チュ", e: "チェ", o: "チョ" },
|
||||||
|
dʒ: { a: "ジャ", i: "ジ", u: "ジュ", e: "ジェ", o: "ジョ" },
|
||||||
|
ʃ: { a: "シャ", i: "シ", u: "シュ", e: "シェ", o: "ショ" },
|
||||||
|
ʒ: { a: "ジャ", i: "ジ", u: "ジュ", e: "ジェ", o: "ジョ" },
|
||||||
|
θ: { a: "サ", i: "シ", u: "ス", e: "セ", o: "ソ" },
|
||||||
|
ð: { a: "ザ", i: "ジ", u: "ズ", e: "ゼ", o: "ゾ" },
|
||||||
|
w: { a: "ワ", i: "ウィ", u: "ウ", e: "ウェ", o: "ウォ" },
|
||||||
|
j: { a: "ヤ", i: "イ", u: "ユ", e: "イェ", o: "ヨ" },
|
||||||
|
ŋ: { a: "ンガ", i: "ンギ", u: "ング", e: "ンゲ", o: "ンゴ" },
|
||||||
|
};
|
||||||
|
|
||||||
|
const CONS_ROW = {
|
||||||
|
"p": "p", "b": "b", "t": "t", "d": "d", "k": "k", "ɡ": "g", "g": "g",
|
||||||
|
"f": "f", "v": "v", "s": "s", "z": "z", "ʃ": "ʃ", "ʒ": "ʒ",
|
||||||
|
"h": "h", "tʃ": "tʃ", "dʒ": "dʒ", "θ": "θ", "ð": "ð",
|
||||||
|
"m": "m", "n": "n", "ŋ": "ŋ", "l": "r", "ɫ": "r", "ɹ": "r",
|
||||||
|
"r": "r", "j": "j", "w": "w", "ɾ": "r",
|
||||||
|
};
|
||||||
|
|
||||||
|
const VOWEL_KANA = { a: "ア", i: "イ", u: "ウ", e: "エ", o: "オ" };
|
||||||
|
const DIPH_KANA = { ai: "アイ", au: "アウ", oi: "オイ" };
|
||||||
|
|
||||||
|
function vClass(v) {
|
||||||
|
switch (v) {
|
||||||
|
case "ə": case "ɚ": case "ɝ": case "ɑ": case "ʌ": case "ɒ": case "æ": case "ɜ":
|
||||||
|
return { vowel: "a", long: false, diph: null };
|
||||||
|
case "ɪ": return { vowel: "i", long: false, diph: null };
|
||||||
|
case "i": return { vowel: "i", long: true, diph: null };
|
||||||
|
case "ʊ": return { vowel: "u", long: false, diph: null };
|
||||||
|
case "u": return { vowel: "u", long: true, diph: null };
|
||||||
|
case "ɛ": case "e": return { vowel: "e", long: false, diph: null };
|
||||||
|
case "eɪ": return { vowel: "e", long: true, diph: null };
|
||||||
|
case "ɔ": return { vowel: "o", long: false, diph: null };
|
||||||
|
case "o": case "oʊ": return { vowel: "o", long: true, diph: null };
|
||||||
|
case "aɪ": return { vowel: "a", long: false, diph: "ai" };
|
||||||
|
case "aʊ": return { vowel: "a", long: false, diph: "au" };
|
||||||
|
case "ɔɪ": return { vowel: "o", long: false, diph: "oi" };
|
||||||
|
default: return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const SONORANT = new Set(["m", "n", "ŋ", "l", "ɫ", "ɹ", "r", "j", "w", "ɾ"]);
|
||||||
|
|
||||||
|
function ipaToKana(ipaStr) {
|
||||||
|
const tokens = [];
|
||||||
|
let i = 0;
|
||||||
|
while (i < ipaStr.length) {
|
||||||
|
const two = ipaStr.slice(i, i + 2);
|
||||||
|
if (CONS_ROW[two] || vClass(two)) { tokens.push(two); i += 2; continue; }
|
||||||
|
const one = ipaStr[i];
|
||||||
|
if (CONS_ROW[one] || vClass(one)) { tokens.push(one); i += 1; continue; }
|
||||||
|
if (one === " " || one === "ː") { tokens.push(one); i += 1; continue; }
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
let out = "";
|
||||||
|
for (let i = 0; i < tokens.length; i++) {
|
||||||
|
const t = tokens[i];
|
||||||
|
if (t === " ") { out += " "; continue; }
|
||||||
|
if (t === "ː") { out += "ー"; continue; }
|
||||||
|
const vc = vClass(t);
|
||||||
|
if (vc) {
|
||||||
|
out += vc.diph ? DIPH_KANA[vc.diph] : (VOWEL_KANA[vc.vowel] + (vc.long ? "ー" : ""));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const row = CONS_ROW[t];
|
||||||
|
if (!row) continue;
|
||||||
|
|
||||||
|
const next = tokens[i + 1];
|
||||||
|
const nextVc = next !== undefined ? vClass(next) : null;
|
||||||
|
if (nextVc) {
|
||||||
|
if (nextVc.diph) {
|
||||||
|
out += KANA[row][nextVc.vowel] + DIPH_KANA[nextVc.diph].slice(1);
|
||||||
|
} else {
|
||||||
|
out += KANA[row][nextVc.vowel] + (nextVc.long ? "ー" : "");
|
||||||
|
}
|
||||||
|
i++;
|
||||||
|
} else if (next !== undefined) {
|
||||||
|
out += KANA[row]["u"];
|
||||||
|
} else {
|
||||||
|
out += (t === "n" || t === "ŋ") ? "ン" : KANA[row]["u"];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Letter-by-letter fallback (last resort, AquesTalk1-safe) ──────────
|
||||||
|
const LETTER_KANA = {
|
||||||
|
a: "エー", b: "ビー", c: "シー", d: "ディー", e: "イー",
|
||||||
|
f: "エフ", g: "ジー", h: "エイチ", i: "アイ", j: "ジェー",
|
||||||
|
k: "ケー", l: "エル", m: "エム", n: "エヌ", o: "オー",
|
||||||
|
p: "ピー", q: "キュー", r: "アール", s: "エス", t: "ティー",
|
||||||
|
u: "ユー", v: "ブイ", w: "ダブリュー", x: "エックス", y: "ワイ", z: "ゼット",
|
||||||
|
};
|
||||||
|
|
||||||
|
const DIGIT_KANA = {
|
||||||
|
"0": "ゼロ", "1": "ワン", "2": "ツー", "3": "スリー", "4": "フォー",
|
||||||
|
"5": "ファイブ", "6": "シックス", "7": "セブン", "8": "エイト", "9": "ナイン",
|
||||||
|
};
|
||||||
|
|
||||||
|
function kanaForChar(ch) {
|
||||||
|
return LETTER_KANA[ch.toLowerCase()] || DIGIT_KANA[ch] || null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function spellWord(word) {
|
||||||
|
let out = "";
|
||||||
|
for (const ch of word) {
|
||||||
|
const k = kanaForChar(ch);
|
||||||
|
out += k ? k : "";
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── English punctuation → Japanese pause equivalents ──────────────────
|
||||||
|
const PUNCT_MAP = {
|
||||||
|
",": "、", ".": "。", "!": "!", "?": "?",
|
||||||
|
",": "、", "。": "。", "!": "!", "?": "?",
|
||||||
|
" ": "、", // AquesTalk truncates on space; use 、 pause instead
|
||||||
|
};
|
||||||
|
|
||||||
|
function wordToKana(word) {
|
||||||
|
const clean = word.replace(/'/g, "").toLowerCase();
|
||||||
|
if (!clean) return "";
|
||||||
|
|
||||||
|
// 1. dictionary
|
||||||
|
const hit = lookupKana(clean);
|
||||||
|
if (hit) return hit;
|
||||||
|
|
||||||
|
// 2. phonemize G2P → IPA → katakana
|
||||||
|
if (phonemize && /^[a-z]+$/.test(clean)) {
|
||||||
|
try {
|
||||||
|
const ipa = phonemize(clean, { stripStress: true });
|
||||||
|
if (ipa && /[^a-z]/.test(ipa)) {
|
||||||
|
const kana = ipaToKana(ipa);
|
||||||
|
if (kana) return kana;
|
||||||
|
}
|
||||||
|
} catch (err) {
|
||||||
|
// fall through to letter spelling
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. letter-by-letter spelling (never skip)
|
||||||
|
return spellWord(clean);
|
||||||
|
}
|
||||||
|
|
||||||
|
function convertEnglishSegment(text) {
|
||||||
|
// Protect decimal points between digits (3.5) from being treated as periods
|
||||||
|
const protectedText = text.replace(/(\d)\.(\d)/g, "$1\u30FB$2");
|
||||||
|
const words = protectedText.split(/([\s]+|[.,!?,。!?]+)/).filter((s) => s.length > 0);
|
||||||
|
const parts = [];
|
||||||
|
|
||||||
|
for (const token of words) {
|
||||||
|
if (/^\s+$/.test(token)) {
|
||||||
|
parts.push("、"); // word gap: AquesTalk truncates on space, use 、 instead
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (PUNCT_MAP[token] !== undefined) {
|
||||||
|
parts.push(PUNCT_MAP[token]);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
parts.push(wordToKana(token));
|
||||||
|
}
|
||||||
|
|
||||||
|
return parts.join("").replace(/、+/g, "、");
|
||||||
|
}
|
||||||
|
|
||||||
function synthesize(kanaText) {
|
function synthesize(kanaText) {
|
||||||
const trimmed = kanaText.trim();
|
const trimmed = kanaText.trim();
|
||||||
if (!trimmed) throw new Error("EMPTY_TEXT");
|
if (!trimmed) throw new Error("EMPTY_TEXT");
|
||||||
const finalText = wanakana.toKatakana(trimmed, { convertLongVowelMark: true });
|
|
||||||
const wav = aq.run(finalText, speed);
|
let result = "";
|
||||||
|
let englishBuf = "";
|
||||||
|
|
||||||
|
for (const ch of trimmed) {
|
||||||
|
if ((ch >= "a" && ch <= "z") || (ch >= "A" && ch <= "Z") ||
|
||||||
|
(ch >= "0" && ch <= "9") || ch === "'" || ch === "," || ch === "." ||
|
||||||
|
ch === "!" || ch === "?") {
|
||||||
|
englishBuf += ch;
|
||||||
|
} else {
|
||||||
|
if (englishBuf) {
|
||||||
|
result += convertEnglishSegment(englishBuf);
|
||||||
|
englishBuf = "";
|
||||||
|
}
|
||||||
|
// Full-width punctuation that slipped through (e.g. ,) → Japanese pause
|
||||||
|
result += PUNCT_MAP[ch] !== undefined ? PUNCT_MAP[ch] : ch;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (englishBuf) {
|
||||||
|
result += convertEnglishSegment(englishBuf);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Collapse consecutive pauses; strip any residual spaces (AquesTalk truncates on space)
|
||||||
|
result = result.replace(/ +/g, "、").replace(/、+/g, "、").replace(/^、|、$/g, "");
|
||||||
|
|
||||||
|
console.error(`[bridge] SYNTH IN: ${trimmed}`);
|
||||||
|
console.error(`[bridge] SYNTH OUT: ${result}`);
|
||||||
|
|
||||||
|
const wav = aq.run(result, speed);
|
||||||
return Buffer.from(wav);
|
return Buffer.from(wav);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user