"""WinRT backend for tktts. This module is intended as a drop-in style backend next to tktts_pyttsx3.py. It uses Windows.Media.SpeechSynthesis via the Python `winsdk` package. Install on Windows: pip install winsdk Notes: * WinRT SpeechSynthesizer generates WAV audio streams. * `speak_rate` is interpreted as pyttsx3-like WPM when > 10 (150 -> 1.0x). Values <= 6 are interpreted as WinRT relative rate. * Narrator "natural voices" may or may not be exposed to third-party programs, depending on the Windows build and voice package. Use list_available_voices() to confirm what WinRT can actually see. """ from __future__ import annotations import asyncio import os import sys import tempfile import threading from typing import Any from tktts_base import apply_replacements, normalize_speaker, split_dialogue TTS_ENGINE_NAME = "winrt" DEFAULT_WINRT_VOICE = "Nanami" # fallback logic handles systems without Nanami DEFAULT_PYTTSX3_COMPAT_RATE = 150.0 # ----------------------------------------------------------------------------- # Lazy WinRT imports # ----------------------------------------------------------------------------- def _import_winrt(): """Import winsdk WinRT classes lazily. Importing at module import time makes this backend unusable on non-Windows machines and also makes a missing optional dependency fatal too early. """ try: from winsdk.windows.media.speechsynthesis import SpeechSynthesizer from winsdk.windows.storage.streams import Buffer, DataReader, InputStreamOptions except ImportError as e: print("Error: Missing library or unsupported platform for tktts_winrt") print(f"detail: {e}") print(" install: pip install winsdk") input("\nPress ENTER to terminate>>\n") sys.exit(1) return SpeechSynthesizer, Buffer, DataReader, InputStreamOptions def _run_async(coro): """Run an async coroutine from synchronous code. asyncio.run() is enough for command-line use. The thread fallback avoids crashing if this backend is called from code that already owns an event loop. """ try: asyncio.get_running_loop() except RuntimeError: return asyncio.run(coro) result_box: dict[str, Any] = {} error_box: dict[str, BaseException] = {} def runner(): try: result_box["result"] = asyncio.run(coro) except BaseException as e: # noqa: BLE001 - re-raised in caller thread error_box["error"] = e th = threading.Thread(target=runner, daemon=True) th.start() th.join() if "error" in error_box: raise error_box["error"] return result_box.get("result") # ----------------------------------------------------------------------------- # Voice handling # ----------------------------------------------------------------------------- def _safe_attr(obj, name: str, default: Any = "") -> Any: try: return getattr(obj, name) except Exception: return default def _voice_gender_text(voice) -> str: gender = _safe_attr(voice, "gender", "") if hasattr(gender, "name"): return str(gender.name) return str(gender) def _voice_to_dict(voice) -> dict[str, Any]: """Convert WinRT VoiceInformation to a pyttsx3-like dictionary.""" display_name = str(_safe_attr(voice, "display_name", "")) voice_id = str(_safe_attr(voice, "id", "")) language = str(_safe_attr(voice, "language", "")) description = str(_safe_attr(voice, "description", "")) return { "name": display_name or voice_id, "id": voice_id, "lang": language, "gender": _voice_gender_text(voice), "description": description, } def _voice_match_text(voice) -> str: d = _voice_to_dict(voice) return "\n".join(str(d.get(k, "")) for k in ["name", "id", "lang", "gender", "description"]).lower() def _select_voice(synthesizer, target_voice: str | None): """Select a WinRT voice by partial match. The match is intentionally broad: display_name, id, language, gender, and description are all searched. This allows names such as "Nanami", "ja-JP", or a full WinRT voice id. """ SpeechSynthesizer, _, _, _ = _import_winrt() voices = list(SpeechSynthesizer.all_voices) if not voices: return None target = (target_voice or "").strip().lower() if target: for voice in voices: if target in _voice_match_text(voice): synthesizer.voice = voice return voice # Prefer a Japanese voice if available. Otherwise keep the system default. for fallback in [DEFAULT_WINRT_VOICE.lower(), "ja-jp", "japan"]: for voice in voices: if fallback in _voice_match_text(voice): synthesizer.voice = voice return voice return _safe_attr(SpeechSynthesizer, "default_voice", None) def get_available_voices_info(): """Return available WinRT voices as dictionaries. Returns False on initialization/import failure to match the existing backend convention used by tktts_pyttsx3.py. """ try: SpeechSynthesizer, _, _, _ = _import_winrt() return [_voice_to_dict(v) for v in SpeechSynthesizer.all_voices] except SystemExit: raise except Exception as e: print(f"Error in tktts_winrt.get_available_voices_info(): {TTS_ENGINE_NAME}の初期化エラー: {e}") return False def get_available_voices(): voices = get_available_voices_info() if not voices: return False return [v["name"] for v in voices] def list_available_voices(): print(f"=== 利用可能な {TTS_ENGINE_NAME} voices ===") voices = get_available_voices_info() if not voices: return False for v in voices: print(f" Name: {v['name']}, Lang: {v['lang']}, Gender: {v['gender']}, ID: {v['id']}") if v.get("description"): print(f" Description: {v['description']}") return True # ----------------------------------------------------------------------------- # Speech synthesis # ----------------------------------------------------------------------------- def _convert_speak_rate(speak_rate: float | int | None) -> float: """Convert pyttsx3-like WPM to WinRT relative speaking_rate. WinRT speaking_rate is relative: 1.0 is normal, 0.5 is half-speed, and 6.0 is six-times speed. Existing tktts code tends to pass 150. """ if speak_rate is None: return 1.0 try: rate = float(speak_rate) except (TypeError, ValueError): return 1.0 if rate > 10.0: rate = rate / DEFAULT_PYTTSX3_COMPAT_RATE return max(0.5, min(6.0, rate)) async def _stream_to_bytes(stream) -> bytes: """Read a WinRT SpeechSynthesisStream into bytes.""" _, Buffer, DataReader, InputStreamOptions = _import_winrt() size = int(_safe_attr(stream, "size", 0)) if size <= 0: return b"" # Reset to the beginning before reading. try: stream.seek(0) except Exception: pass buffer = Buffer(size) read_buffer = await stream.read_async(buffer, size, InputStreamOptions.READ_AHEAD) length = int(_safe_attr(read_buffer, "length", 0)) or size reader = DataReader.from_buffer(read_buffer) data = bytearray(length) reader.read_bytes(data) try: reader.close() except Exception: pass return bytes(data) async def _synthesize_wav_bytes(text: str, target_voice: str | None, speak_rate: float | int | None) -> bytes: SpeechSynthesizer, _, _, _ = _import_winrt() text = text or "" synthesizer = SpeechSynthesizer() _select_voice(synthesizer, target_voice) try: synthesizer.options.speaking_rate = _convert_speak_rate(speak_rate) except Exception: # Older Windows builds may not support SpeakingRate. pass stream = await synthesizer.synthesize_text_to_stream_async(text) try: data = await _stream_to_bytes(stream) finally: for obj in [stream, synthesizer]: try: obj.close() except Exception: pass return data def _write_wav(outfile: str, text: str, target_voice: str | None, speak_rate: float | int | None): data = _run_async(_synthesize_wav_bytes(text, target_voice, speak_rate)) if not data: print(f" Error: {TTS_ENGINE_NAME} の音声生成結果が空です") return None outdir = os.path.dirname(os.path.abspath(outfile)) if outdir: os.makedirs(outdir, exist_ok=True) with open(outfile, "wb") as f: f.write(data) if not os.path.exists(outfile) or os.path.getsize(outfile) <= 0: print(f" Error: ファイル [{outfile}] の出力に失敗しました") return None return outfile def _play_wav_file(wavfile: str): if sys.platform != "win32": print(f"Error: WAV再生はWindows環境でのみ対応しています: {wavfile}") return False import winsound winsound.PlaySound(wavfile, winsound.SND_FILENAME) return True def speak(outfile, text, voice, speak_rate=None): """Speak or save one text string. Parameters are kept compatible with tktts_pyttsx3.speak(). If outfile is given, a WAV file is written. If outfile is empty/None, a temporary WAV file is generated and played with winsound. """ is_save_mode = bool(outfile) target_voice = voice or DEFAULT_WINRT_VOICE if is_save_mode: if str(outfile).lower().endswith(".wav") is False: print(" Warning: WinRT backend outputs WAV data. 拡張子は .wav を推奨します。") return _write_wav(str(outfile), text, target_voice, speak_rate) with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as f: tmpfile = f.name try: result = _write_wav(tmpfile, text, target_voice, speak_rate) if not result: return False return _play_wav_file(tmpfile) finally: try: os.remove(tmpfile) except OSError: pass def _resolve_target_voice(target_voices, speaker=None, original_speaker=None): """Resolve a selected WinRT voice without truncating its display name. ``split_dialogue()`` may normalize a voice-like speaker string such as ``"Microsoft Ayumi - Japanese (Japan)"`` to ``"Microsoft"``. GUI monologue mode, however, stores the selected voice under ``None``, ``""``, and ``0``. Prefer exact mappings and then these monologue fallback keys before falling back to ``DEFAULT_WINRT_VOICE``. """ if isinstance(target_voices, str): selected = target_voices.strip() return selected or DEFAULT_WINRT_VOICE if not isinstance(target_voices, dict): return DEFAULT_WINRT_VOICE candidates = [] for value in (original_speaker, speaker): if value not in candidates: candidates.append(value) try: normalized = normalize_speaker(value) except Exception: normalized = value if normalized not in candidates: candidates.append(normalized) # GUI monologue maps the selected voice to these keys. candidates.extend([None, "", 0]) for key in candidates: if key in target_voices: voice = target_voices.get(key) if voice is not None and str(voice).strip(): return str(voice).strip() return DEFAULT_WINRT_VOICE def _first_selected_voice(target_voices): """Return a useful voice for direct-playback monologue mode.""" return _resolve_target_voice(target_voices, speaker=None, original_speaker=None) def speak_dialogue(dialogue, replacements, target_voices, speakers={}, speak_rate=150, temp_dir=None, outfile=None, ext="wav", cfg=None): """Generate speech files for dialogue blocks. This mirrors tktts_pyttsx3.speak_dialogue(), but WinRT synthesis is performed one utterance at a time because the API itself is async stream-based. """ is_save_mode = bool(outfile) temp_dir = temp_dir or tempfile.gettempdir() os.makedirs(temp_dir, exist_ok=True) # WinRT returns WAV. Keep downstream honest even if caller passed mp3. if ext.lower() != "wav": print(" Warning: WinRT backend outputs WAV. 一時ファイル拡張子を wav に変更します。") ext = "wav" print() print("tktts_winrt.speak_dialogue(): ") print(f" 出力ファイル: {outfile}") print(f" is_save_mode: {is_save_mode}") print("target_voices:", target_voices) tmpfiles = [] text_all = "" idx = 1 is_monologue = bool(getattr(cfg, "monologue", False)) for i, _dialogue in enumerate(dialogue): print() print(f"Dialogue {i:04d}:") dialogue_list = split_dialogue( _dialogue, target_voices, speakers=speakers, default_voice=DEFAULT_WINRT_VOICE, is_monologue=is_monologue, ) # Preserve the source speaker before split_dialogue()/normalization. original_speaker = None if isinstance(_dialogue, (tuple, list)) and len(_dialogue) >= 2: original_speaker = _dialogue[0] for speaker, text in dialogue_list: text = apply_replacements(text, replacements) # speaker is only a display/dialogue label here. The actual WinRT # voice must be resolved from target_voices without converting a # full display name such as "Microsoft Ayumi ..." to "Microsoft". display_speaker = normalize_speaker(speaker) target_voice = _resolve_target_voice( target_voices, speaker=speaker, original_speaker=original_speaker, ) print(f" {idx:04d}: voice={target_voice} (speaker={display_speaker}): ", end="") print(text) print(f"{i:04d}: speaker={display_speaker}: voice={target_voice}: {text}") if is_save_mode: tmpfile = os.path.join(temp_dir, f"tmp_{idx:03d}.{ext}") result = _write_wav(tmpfile, text, target_voice, speak_rate) if not result: return False, tmpfiles tmpfiles.append(tmpfile) else: text_all += "\n" + text idx += 1 if is_save_mode: print(f"\n{TTS_ENGINE_NAME}で音声ファイルを一時ファイルに生成しました。") return True, tmpfiles print(f"\n{TTS_ENGINE_NAME}で音声ファイルを再生中...") selected_voice = _first_selected_voice(target_voices) ok = speak(None, text_all, selected_voice, speak_rate=speak_rate) return bool(ok), {} # ----------------------------------------------------------------------------- # Small standalone test CLI # ----------------------------------------------------------------------------- if __name__ == "__main__": import argparse parser = argparse.ArgumentParser(description="WinRT TTS test utility for tktts_winrt.py") parser.add_argument("--list", type=int, default=0, choices=[0, 1], help="list available voices") parser.add_argument("--voice", type=str, default=DEFAULT_WINRT_VOICE, help="voice name/id/language partial match") parser.add_argument("--text", type=str, default="こんにちは。これは WinRT 音声合成のテストです。", help="text to speak") parser.add_argument("--outfile", type=str, default="", help="output wav file; omit to play") parser.add_argument("--rate", type=float, default=150.0, help="pyttsx3-like WPM or WinRT relative rate") args = parser.parse_args() if args.list: list_available_voices() else: speak(args.outfile, args.text, args.voice, speak_rate=args.rate)