tktts_gemini.py ダウンロード/コピー
tktts_gemini.py
tktts_gemini.py
1import os
2import sys
3
4missing = []
5for lib in ["google-genai"]:
6 try:
7 __import__("google.genai")
8 except ImportError:
9 missing.append(lib)
10
11if missing:
12 print(f"Error: Missing libraries:\n{', '.join(missing)}")
13 print(" install: pip install google-genai")
14 input("\nPress ENTER to terminate>>\n")
15 sys.exit(1)
16
17from google import genai
18from google.genai import types
19from tktts_base import apply_replacements, normalize_speaker, split_dialogue
20
21TTS_ENGINE_NAME = 'gemini'
22# ご指摘のTTS専用モデルを正しく指定
23DEFAULT_TTS_MODEL = "gemini-3.8-flash-tts" # gemini-3.8-flash-lite-tts
24voices_available = ["Kore", "Puck", "Aoede", "Charon", "Fenrir"]
25DEFAULT_GEMINI_VOICE = "Kore"
26
27client = genai.Client()
28
29def get_available_voices_info():
30 voices = []
31 for v in voices_available:
32 voices.append({"name": v})
33 return voices
34
35def get_available_voices():
36 return voices_available
37
38def list_available_voices():
39 print(f"=== 利用可能な {TTS_ENGINE_NAME} voices ===")
40 voices = get_available_voices_info()
41 if not voices: return False
42
43 for v in voices:
44 print(f" Name: {v['name']}")
45 return True
46
47def speak(outfile, text, voice, tts_model=DEFAULT_TTS_MODEL, instruction="", output_format="wav"):
48 try:
49 # TTS専用モデルの場合、会話用のプロンプト(「読み上げてください」等)を付けると
50 # その指示部分まで音声合成されてしまうため、テキストを直接渡します。
51 prompt_text = text
52 if instruction:
53 prompt_text = f"[{instruction}] {text}"
54
55 response = client.models.generate_content(
56 model=tts_model,
57 contents=prompt_text,
58 config=types.GenerateContentConfig(
59 response_modalities=["AUDIO"],
60 speech_config=types.SpeechConfig(
61 voice_config=types.VoiceConfig(
62 prebuilt_voice_config=types.PrebuiltVoiceConfig(
63 voice_name=voice
64 )
65 )
66 )
67 )
68 )
69
70 audio_data = None
71 text_responses = []
72
73 # レスポンスから音声データを探索
74 if response.candidates and response.candidates[0].content.parts:
75 for part in response.candidates[0].content.parts:
76 if part.inline_data:
77 audio_data = part.inline_data.data
78 break
79 elif part.text:
80 text_responses.append(part.text)
81
82 if not audio_data:
83 returned_text = " ".join(text_responses)
84 print(f"❌ Geminiが音声データを生成しませんでした。")
85 print(f" 返されたテキスト: {returned_text}")
86 return None
87
88 # 音声データを保存
89 with open(outfile, "wb") as f:
90 f.write(audio_data)
91
92 if os.path.exists(outfile):
93 print(f" ファイル [{outfile}] を保存しました")
94 else:
95 print(f" Error: ファイル [{outfile}] の出力に失敗しました")
96 return None
97
98 except Exception as e:
99 print(f"❌ Gemini API 接続エラー: {e}")
100 return None
101
102 return outfile
103
104def speak_dialogue(dialogue, replacements, target_voices, speakers={}, instruction="",
105 temp_dir=None, outfile=None, ext="wav", tts_model=DEFAULT_TTS_MODEL, cfg=None):
106 is_save_mode = bool(outfile)
107
108 print("\ntktts_gemini.speak_dialogue(): ")
109 tmpfiles = []
110 idx = 1
111
112 for i, _dialogue in enumerate(dialogue):
113 print(f"\nDialogue {i:04d}:")
114 dialogue_list = split_dialogue(
115 _dialogue, target_voices, speakers=speakers,
116 default_voice=DEFAULT_GEMINI_VOICE, is_monologue=cfg.monologue if cfg else False
117 )
118
119 for speaker, text in dialogue_list:
120 if temp_dir:
121 tmpfile = os.path.join(temp_dir, f"tmp_{idx:03d}.{ext}")
122 else:
123 tmpfile = f"tmp_{idx:03d}.{ext}"
124
125 text = apply_replacements(text, replacements)
126
127 if isinstance(target_voices, str):
128 target_voice = target_voices
129 else:
130 target_voice = target_voices.get(speaker, DEFAULT_GEMINI_VOICE)
131
132 print(f" {idx:04d}: voice={speaker} (id={target_voice}): {text}")
133
134 _outfile = speak(tmpfile, text, target_voice, tts_model=tts_model, instruction=instruction, output_format=ext)
135 if _outfile is None:
136 return False, tmpfiles
137
138 tmpfiles.append(tmpfile)
139 idx += 1
140
141 return True, tmpfiles