tktts_gemini.py ダウンロード/コピー

tktts_gemini.py をダウンロード

tktts_gemini.py
tktts_gemini.py
  1import os
  2import sys
  3
  4missing = []
  5for lib in ["google-genai"]:
  6    try:
  7        __import__("google.genai")
  8    except ImportError:
  9        missing.append(lib)
 10
 11if missing:
 12    print(f"Error: Missing libraries:\n{', '.join(missing)}")
 13    print("  install: pip install google-genai")
 14    input("\nPress ENTER to terminate>>\n")
 15    sys.exit(1)
 16
 17from google import genai
 18from google.genai import types
 19from tktts_base import apply_replacements, normalize_speaker, split_dialogue
 20
 21TTS_ENGINE_NAME = 'gemini'
 22# ご指摘のTTS専用モデルを正しく指定
 23DEFAULT_TTS_MODEL = "gemini-3.8-flash-tts" # gemini-3.8-flash-lite-tts
 24voices_available = ["Kore", "Puck", "Aoede", "Charon", "Fenrir"] 
 25DEFAULT_GEMINI_VOICE = "Kore"
 26
 27client = genai.Client()
 28
 29def get_available_voices_info():
 30    voices = []
 31    for v in voices_available:
 32        voices.append({"name": v})
 33    return voices
 34
 35def get_available_voices():
 36    return voices_available
 37
 38def list_available_voices():
 39    print(f"=== 利用可能な {TTS_ENGINE_NAME} voices ===")
 40    voices = get_available_voices_info()
 41    if not voices: return False
 42    
 43    for v in voices:
 44        print(f"  Name: {v['name']}")
 45    return True
 46
 47def speak(outfile, text, voice, tts_model=DEFAULT_TTS_MODEL, instruction="", output_format="wav"):
 48    try:
 49        # TTS専用モデルの場合、会話用のプロンプト(「読み上げてください」等)を付けると
 50        # その指示部分まで音声合成されてしまうため、テキストを直接渡します。
 51        prompt_text = text
 52        if instruction:
 53            prompt_text = f"[{instruction}] {text}"
 54            
 55        response = client.models.generate_content(
 56            model=tts_model,
 57            contents=prompt_text,
 58            config=types.GenerateContentConfig(
 59                response_modalities=["AUDIO"],
 60                speech_config=types.SpeechConfig(
 61                    voice_config=types.VoiceConfig(
 62                        prebuilt_voice_config=types.PrebuiltVoiceConfig(
 63                            voice_name=voice
 64                        )
 65                    )
 66                )
 67            )
 68        )
 69        
 70        audio_data = None
 71        text_responses = []
 72
 73        # レスポンスから音声データを探索
 74        if response.candidates and response.candidates[0].content.parts:
 75            for part in response.candidates[0].content.parts:
 76                if part.inline_data:
 77                    audio_data = part.inline_data.data
 78                    break
 79                elif part.text:
 80                    text_responses.append(part.text)
 81        
 82        if not audio_data:
 83            returned_text = " ".join(text_responses)
 84            print(f"❌ Geminiが音声データを生成しませんでした。")
 85            print(f"  返されたテキスト: {returned_text}")
 86            return None
 87
 88        # 音声データを保存
 89        with open(outfile, "wb") as f:
 90            f.write(audio_data)
 91            
 92        if os.path.exists(outfile):
 93            print(f" ファイル [{outfile}] を保存しました")
 94        else:
 95            print(f" Error: ファイル [{outfile}] の出力に失敗しました")
 96            return None
 97            
 98    except Exception as e:
 99        print(f"❌ Gemini API 接続エラー: {e}")
100        return None
101
102    return outfile
103    
104def speak_dialogue(dialogue, replacements, target_voices, speakers={}, instruction="", 
105                temp_dir=None, outfile=None, ext="wav", tts_model=DEFAULT_TTS_MODEL, cfg=None):
106    is_save_mode = bool(outfile)
107
108    print("\ntktts_gemini.speak_dialogue(): ")
109    tmpfiles = []
110    idx = 1
111    
112    for i, _dialogue in enumerate(dialogue):
113        print(f"\nDialogue {i:04d}:")
114        dialogue_list = split_dialogue(
115            _dialogue, target_voices, speakers=speakers, 
116            default_voice=DEFAULT_GEMINI_VOICE, is_monologue=cfg.monologue if cfg else False
117        )
118        
119        for speaker, text in dialogue_list:
120            if temp_dir:
121                tmpfile = os.path.join(temp_dir, f"tmp_{idx:03d}.{ext}")
122            else:
123                tmpfile = f"tmp_{idx:03d}.{ext}"
124                
125            text = apply_replacements(text, replacements)
126            
127            if isinstance(target_voices, str): 
128                target_voice = target_voices
129            else:
130                target_voice = target_voices.get(speaker, DEFAULT_GEMINI_VOICE)
131                
132            print(f"  {idx:04d}: voice={speaker} (id={target_voice}): {text}")
133
134            _outfile = speak(tmpfile, text, target_voice, tts_model=tts_model, instruction=instruction, output_format=ext)
135            if _outfile is None: 
136                return False, tmpfiles
137
138            tmpfiles.append(tmpfile)
139            idx += 1
140
141    return True, tmpfiles