transcribe_simple.py ダウンロード/コピー

transcribe_simple.py をダウンロード

transcribe_simple.py
transcribe_simple.py
  1"""音声ファイルからWhisperモデルで文字起こしを実行するスクリプト。
  2
  3このスクリプトは、Windows環境では `faster-whisper` を、Linux/macOS環境では `openai/whisper` を使用して、
  4音声ファイルの文字起こしを行います。コマンドライン引数で入力ファイル、出力ファイル、モデル、言語などを指定できます。
  5長時間の音声に対する反復幻覚抑制機能も含まれています。
  6
  7:doc:`transcribe_simple_usage`
  8"""
  9
 10import os
 11import glob
 12import argparse
 13import traceback
 14import platform
 15import subprocess
 16import inspect
 17import re
 18import gzip
 19from collections import Counter
 20
 21# =========================
 22# グローバル設定
 23# =========================
 24USE_FASTER = False          # Windows: faster-whisper / Linux, macOS: whisper
 25whisper = None
 26WhisperModel = None
 27#COMPUTE_TYPE = "int8"       # CPU / DML では int8 が最速
 28
 29pause = 0
 30
 31
 32# =========================
 33# ユーティリティ
 34# =========================
 35def terminate():
 36    """スクリプトを終了する。必要に応じてユーザーの入力待ちを行う。
 37
 38    グローバル変数 `pause` が `0` でない場合、`ENTER` キーが押されるまでプロンプトを表示し、
 39    その後スクリプトを終了します。
 40    """
 41    if pause:
 42        input("\nPress ENTER to terminate\n")
 43    exit()
 44
 45
 46def safe_import_torch():
 47    """PyTorchライブラリを安全にインポートする。
 48
 49    PyTorchのインポートを試行し、失敗した場合はエラーメッセージとトレースバックを表示して
 50    `None` を返します。
 51
 52    :returns: module or None: PyTorchモジュール (`torch`) またはインポートに失敗した場合は `None`。
 53    """
 54    try:
 55        import torch
 56        return torch
 57    except Exception:
 58        print("torch import failed")
 59        traceback.print_exc()
 60        return None
 61
 62
 63def check_gpu_torch(torch_mod):
 64    """PyTorchがCUDA GPUを利用可能かを確認し、情報を表示する。
 65
 66    引数で渡されたtorchモジュールがCUDAを利用できる場合、デバイス名とCUDAバージョンを表示する。
 67    利用できない場合は、その旨を表示する。
 68
 69    :param torch_mod: module or None: PyTorchモジュールまたはNone。
 70    """
 71    if torch_mod is not None and torch_mod.cuda.is_available():
 72        print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
 73        print(f"Torch CUDA ver: {torch_mod.version.cuda}")
 74    else:
 75        print("No CUDA GPU available for Torch")
 76        print("   (this does not affect faster-whisper)")
 77
 78def check_gpu():
 79    """使用中のWhisper実装に応じてGPUの利用可能性を確認し、情報を表示する。
 80
 81    `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を介してCUDAデバイスをチェックする。
 82    `False` の場合 (openai/whisperを使用)、`safe_import_torch` を利用してPyTorchのCUDA GPUをチェックする。
 83    """
 84    if USE_FASTER:
 85        try:
 86            import ctranslate2
 87
 88            count = ctranslate2.get_cuda_device_count()
 89            if count > 0:
 90                print(f"CUDA GPU available for CTranslate2: {count} device(s)")
 91            else:
 92                print("No CUDA GPU available for CTranslate2")
 93        except Exception:
 94            print("Failed to check CTranslate2 CUDA support")
 95            traceback.print_exc()
 96        return
 97
 98    # openai-whisperではPyTorch基準
 99    torch_mod = safe_import_torch()
100    if torch_mod is not None and torch_mod.cuda.is_available():
101        print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
102        print(f"Torch CUDA ver: {torch_mod.version.cuda}")
103    else:
104        print("No CUDA GPU available for PyTorch")
105        
106
107def detect_device_torch():
108    """PyTorchが使用するデバイスを自動判定する。
109
110    優先順位 `CUDA -> DML -> CPU` で利用可能なデバイスを判定し、そのデバイス名を返します。
111
112    :returns: str: 検出されたデバイス名 ("cuda", "dml", "cpu" のいずれか)。
113    """
114    # CUDA
115    torch_mod = safe_import_torch()
116    if torch_mod is not None and torch_mod.cuda.is_available():
117        return "cuda"
118
119    # DirectML
120    try:
121        import ctranslate2
122        if "dml" in ctranslate2.get_supported_devices():
123            return "dml"
124    except Exception:
125        pass
126
127    # CPU
128    return "cpu"
129
130def detect_device():
131    """使用するWhisper実装に応じてデバイスを自動判定する。
132
133    `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を通じてCUDAの利用可能性を確認し、
134    優先的に `cuda` を返します。そうでなければ `cpu` を返します。
135    `USE_FASTER` が `False` の場合 (openai/whisperを使用)、PyTorchを通じてCUDAの利用可能性を確認し、
136    優先的に `cuda` を返します。そうでなければ `cpu` を返します。
137
138    :returns: str: 検出されたデバイス名 ("cuda", "cpu" のいずれか)。
139    """
140    if USE_FASTER:
141        try:
142            import ctranslate2
143
144            if ctranslate2.get_cuda_device_count() > 0:
145                return "cuda"
146        except Exception:
147            traceback.print_exc()
148
149        return "cpu"
150
151    torch_mod = safe_import_torch()
152    if torch_mod is not None and torch_mod.cuda.is_available():
153        return "cuda"
154
155    return "cpu"
156
157def get_audio_duration(infile):
158    """`ffprobe` コマンドを使用して音声ファイルの長さを取得する。
159
160    `ffprobe` を外部プロセスとして実行し、音声ファイルの長さを秒単位で解析して返します。
161    `ffprobe` の実行に失敗した場合は警告を表示し、`None` を返します。
162
163    :param infile: str: 入力音声ファイルのパス。
164    :returns: float or None: 音声の長さ(秒単位)または、取得できなかった場合はNone。
165    """
166    try:
167        cmd = [
168            "ffprobe", "-v", "error",
169            "-show_entries", "format=duration",
170            "-of", "default=noprint_wrappers=1:nokey=1",
171            infile,
172        ]
173        result = subprocess.run(
174            cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True
175        )
176        return float(result.stdout.strip())
177    except Exception:
178        print("Warning: failed to get duration via ffprobe, fallback to model info.")
179        return None
180
181
182def format_hms(sec):
183    """秒数を HH:MM:SS.s 形式の文字列に変換する。
184
185    渡された秒数を時、分、秒に分解し、指定されたフォーマットで整形した文字列を返します。
186    `None` が渡された場合は `??:??:??` を返します。
187
188    :param sec: float or None: 変換する秒数。
189    :returns: str: フォーマットされた時間文字列。
190    """
191    if sec is None:
192        return "??:??:??"
193    sec = float(sec)
194    h = int(sec // 3600)
195    m = int((sec % 3600) // 60)
196    s = sec % 60
197    return f"{h:02d}:{m:02d}:{s:04.1f}"
198
199
200def parse_temperature(value):
201    """temperatureパラメータの値を浮動小数点数または浮動小数点数のタプルに変換する。
202
203    カンマ区切りの文字列を解析し、単一の数値であれば `float`、複数の数値であれば `tuple[float, ...]` を返します。
204    空の文字列や不正な値が指定された場合は `ValueError` を発生させます。
205
206    :param value: Union[str, float]: temperatureとして指定された値。
207    :returns: Union[float, tuple[float, ...]]: 解析されたtemperatureの値。
208    :raises ValueError: temperatureが空の場合。
209    """
210    text = str(value).strip()
211    if "," not in text:
212        return float(text)
213    temps = tuple(float(x.strip()) for x in text.split(",") if x.strip() != "")
214    if not temps:
215        raise ValueError("temperature is empty")
216    return temps
217
218
219def text_compression_ratio(text):
220    """テキストの簡易的なgzip圧縮率を計算する。
221
222    入力テキストをUTF-8でエンコードし、gzipで圧縮します。
223    非圧縮バイト長を圧縮バイト長で割ることで圧縮率を算出します。
224    短いテキストや圧縮結果が空の場合は `0.0` を返します。
225
226    :param text: str: 圧縮率を計算するテキスト。
227    :returns: float: テキストの圧縮率。
228    """
229    raw = text.encode("utf-8", errors="ignore")
230    if len(raw) < 80:
231        return 0.0
232    comp = gzip.compress(raw)
233    if len(comp) == 0:
234        return 0.0
235    return len(raw) / len(comp)
236
237
238def normalize_tokens(text):
239    """反復検出のためにテキストをゆるくトークン化する。
240
241    英数字の単語、およびひらがな・カタカナ・漢字の連続を単語として抽出し、小文字に変換してリストで返します。
242
243    :param text: str: トークン化する入力テキスト。
244    :returns: list[str]: 正規化されたトークンのリスト。
245    """
246    return re.findall(r"[A-Za-z0-9]+(?:'[A-Za-z0-9]+)?|[ぁ-んァ-ヶー一-龥]+", text.lower())
247
248
249def looks_like_repetition_loop(text, max_token_repeat=12, compression_ratio_limit=3.2):
250    """Whisperモデルによって生成された反復幻覚らしいセグメントを検出する。
251
252    入力テキストの長さが80文字未満の場合は検出しない。
253    `normalize_tokens` でトークン化した後、最も頻繁に出現するトークンの繰り返し回数とその割合をチェックする。
254    また、`text_compression_ratio` を用いてテキストの圧縮率が高すぎる場合も反復と見なす。
255    これらの条件に合致する場合に `True` とその理由を返します。
256
257    判定の狙い:
258      - "6th, 6th, 6th, ..." のような単語反復
259      - 同じ語句が異常に多い高圧縮テキスト
260
261    強すぎると正しい復唱も消すので、講義音声向けにやや控えめにしている。
262
263    :param text: str: 検出対象のテキストセグメント。
264    :param max_token_repeat: int: 単語反復とみなす最小繰り返し回数。(default: 12)
265    :param compression_ratio_limit: float: 圧縮率がこの値以上の場合に反復とみなすしきい値。(default: 3.2)
266    :returns: tuple[bool, str]: 反復幻覚らしい場合は (True, 理由), そうでない場合は (False, "")。
267    """
268    stripped = text.strip()
269    if len(stripped) < 80:
270        return False, ""
271
272    tokens = normalize_tokens(stripped)
273    if len(tokens) >= max_token_repeat:
274        counts = Counter(tokens)
275        token, count = counts.most_common(1)[0]
276        if count >= max_token_repeat and count / len(tokens) >= 0.45:
277            return True, f"token '{token}' repeated {count}/{len(tokens)} times"
278
279    cr = text_compression_ratio(stripped)
280    if cr >= compression_ratio_limit:
281        # 圧縮率だけだと箇条書きや専門語の繰り返しも拾うので、長めの出力に限定する。
282        if len(tokens) >= 30 or len(stripped) >= 300:
283            return True, f"text compression ratio {cr:.2f} >= {compression_ratio_limit:.2f}"
284
285    return False, ""
286
287
288def bool_from_int(value):
289    """整数値をブール値に変換する。
290
291    整数 `0` を `False` に、それ以外の整数を `True` に変換します。
292
293    :param value: int: 変換する整数値。
294    :returns: bool: 変換されたブール値。
295    """
296    return bool(int(value))
297
298
299# =========================
300# OS 判定 & ライブラリ import
301# =========================
302os_name = platform.system()
303print(f"Detected OS: {os_name}")
304
305try:
306    if os_name == "Windows":
307        from faster_whisper import WhisperModel
308        USE_FASTER = True
309        print("Using faster-whisper (Windows)")
310    else:
311        import whisper
312        USE_FASTER = False
313        print("Using openai/whisper (Linux/macOS)")
314except Exception:
315    print("whisper / faster-whisper import failed")
316    traceback.print_exc()
317    terminate()
318
319torch_mod = safe_import_torch()
320#if torch_mod is None:
321#    terminate()
322
323
324# =========================
325# 文字起こし本体
326# =========================
327def transcribe_audio_faster(
328    infile,
329    outfile1,
330    outfile2,
331    model_name,
332    lang,
333    device_name="",
334    beam_size=5,
335    chunk_length=30,
336    temperature="0,0.2,0.4,0.6",
337    condition_on_previous_text=False,
338    compression_ratio_threshold=2.4,
339    log_prob_threshold=-1.0,
340    no_speech_threshold=0.6,
341    repetition_penalty=1.05,
342    no_repeat_ngram_size=3,
343    max_bad_segments=3,
344):
345    """faster-whisperライブラリを使用して音声ファイルを文字起こしする。
346
347    faster-whisperのモデルをロードし、指定されたパラメータで文字起こしを実行します。
348    セグメントごとに進捗を表示し、指定されたファイルに出力します。
349    長時間の音声処理において反復幻覚を抑制するための機能も含まれます。
350    デバイス、計算タイプは自動判別されます。
351
352    重要:
353      - offset / duration は faster-whisper の transcribe() には渡さない。
354      - 長時間音声では condition_on_previous_text=False をデフォルトにする。
355
356    :param infile: str: 入力音声ファイルのパス。
357    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
358    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
359    :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
360    :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
361    :param device_name: str: 使用するデバイス名 ("cuda", "cpu", "dml")。空文字の場合は自動判別。(default: "")
362    :param beam_size: int: ビームサーチのサイズ。(default: 5)
363    :param chunk_length: int: 内部チャンクの長さ(秒)。(default: 30)
364    :param temperature: Union[str, float, tuple[float, ...]]: temperatureパラメータ。文字列形式("0,0.2,0.4")または数値で指定。(default: "0,0.2,0.4,0.6")
365    :param condition_on_previous_text: bool: 前の認識結果を次のチャンクの文脈として使用するかどうか。(default: False)
366    :param compression_ratio_threshold: float: 反復テキスト検出用の圧縮率しきい値。(default: 2.4)
367    :param log_prob_threshold: float: 低信頼デコード検出のしきい値。(default: -1.0)
368    :param no_speech_threshold: float: 無音判定のしきい値。(default: 0.6)
369    :param repetition_penalty: float: 反復抑制ペナルティ。(default: 1.05)
370    :param no_repeat_ngram_size: int: n-gram反復抑制のサイズ。0で無効。(default: 3)
371    :param max_bad_segments: int: 反復幻覚らしいセグメントが連続した場合に文字起こしを停止する回数。(default: 3)
372    """
373    if device_name == "":
374        device_name = detect_device()
375
376    compute_type = (
377        "int8_float16" if device_name == "cuda"
378        else "int8"
379    )
380
381    print(
382        f"Using device [{device_name}] (faster-whisper, compute_type={compute_type})",
383        flush=True,
384    )
385
386    model = WhisperModel(
387        model_name,
388        device=device_name,
389        compute_type=compute_type,
390    )
391
392    transcribe_params = inspect.signature(model.transcribe).parameters
393
394    # language="" を指定した場合は自動判定にする。
395    language_arg = lang if str(lang).strip() else None
396
397    requested_kwargs = {
398        "language": language_arg,
399        "vad_filter": True,
400        "beam_size": beam_size,
401        "temperature": parse_temperature(temperature),
402        "condition_on_previous_text": condition_on_previous_text,
403        "compression_ratio_threshold": compression_ratio_threshold,
404        "log_prob_threshold": log_prob_threshold,
405        "no_speech_threshold": no_speech_threshold,
406        "repetition_penalty": repetition_penalty,
407        "no_repeat_ngram_size": no_repeat_ngram_size,
408        "suppress_blank": True,
409    }
410
411    if "chunk_length" in transcribe_params:
412        requested_kwargs["chunk_length"] = chunk_length
413    else:
414        print(
415            "chunk_length is not supported by this faster-whisper version; using default chunking.",
416            flush=True,
417        )
418
419    # VAD パラメータはバージョン差があるので、対応している場合だけ渡す。
420    if "vad_parameters" in transcribe_params:
421        requested_kwargs["vad_parameters"] = {
422            # 既定値より少し長い無音で分割し、短い講義中の間を無音扱いしすぎない。
423            "min_silence_duration_ms": 500,
424        }
425
426    # word_timestamps と組み合わせると hallucination_silence_threshold が効く版もある。
427    # ただし word_timestamps=True は重くなるので、ここでは使わない。
428
429    # 古い faster-whisper でも動くよう、存在する引数だけ渡す。
430    transcribe_kwargs = {
431        k: v for k, v in requested_kwargs.items() if k in transcribe_params
432    }
433
434    print(f"transcribe options: {transcribe_kwargs}", flush=True)
435
436    segments, info = model.transcribe(infile, **transcribe_kwargs)
437
438    duration = getattr(info, "duration", None)
439    if duration is None:
440        duration = get_audio_duration(infile)
441
442    if duration is not None:
443        print(f"Audio length: {duration:.1f} sec ({format_hms(duration)})", flush=True)
444
445    print("Start transcription...", flush=True)
446    print(f"Save to [{outfile1}]")
447    print(f"Save to [{outfile2}]", flush=True)
448
449    seg_count = 0
450    bad_count = 0
451    all_text = []
452
453    # ここで list(segments) にしてしまうと、全処理が終わるまで何も表示されない。
454    # generator を1件ずつ消費して、そのたびに進捗を表示する。
455    with open(outfile1, "w", encoding="utf-8") as f_time, \
456         open(outfile2, "w", encoding="utf-8") as f_text:
457        for seg in segments:
458            seg_count += 1
459            text = seg.text or ""
460
461            is_bad, reason = looks_like_repetition_loop(text)
462            if is_bad:
463                bad_count += 1
464                if duration and duration > 0:
465                    pct = min(100.0, 100.0 * seg.end / duration)
466                    progress = f"{pct:6.2f}%"
467                else:
468                    progress = "  ??.??%"
469
470                preview = text.strip().replace("\n", " ")[:160]
471                print(
472                    f"[WARN] skipped probable repetition loop "
473                    f"at {format_hms(seg.start)} - {format_hms(seg.end)} "
474                    f"({progress}): {reason}\n"
475                    f"       preview: {preview}",
476                    flush=True,
477                )
478
479                f_time.write(
480                    f"[{seg.start:.2f} - {seg.end:.2f}] "
481                    f"[SKIPPED: probable repetition loop: {reason}]\n"
482                )
483                f_time.flush()
484
485                if bad_count >= max_bad_segments:
486                    print(
487                        f"[WARN] {bad_count} suspicious segments were detected. "
488                        "Stopping transcription to avoid a long hallucination tail.",
489                        flush=True,
490                    )
491                    break
492                continue
493
494            bad_count = 0
495            all_text.append(text)
496
497            f_time.write(f"[{seg.start:.2f} - {seg.end:.2f}] {text}\n")
498            f_time.flush()
499
500            f_text.write(text)
501            f_text.flush()
502
503            if duration and duration > 0:
504                pct = min(100.0, 100.0 * seg.end / duration)
505                progress = f"{pct:6.2f}%"
506            else:
507                progress = "  ??.??%"
508
509            print(
510                f"[{seg_count:04d}] {progress} "
511                f"{format_hms(seg.start)} - {format_hms(seg.end)} "
512                f"{text}",
513                flush=True,
514            )
515
516    text = "".join(all_text)
517    print(f"\nSegments read: {seg_count}", flush=True)
518    print("\n=== Transcribed text ===", flush=True)
519    print(text, flush=True)
520
521
522def transcribe_audio_whisper(
523    infile,
524    outfile1,
525    outfile2,
526    model_name,
527    lang,
528    device_name="",
529):
530    """openai/whisperライブラリを使用して音声ファイルを文字起こしする。
531
532    openai/whisperモデルをロードし、指定されたパラメータで音声ファイルを一括で文字起こしします。
533    結果は指定された2つのファイルに出力されます。
534
535    :param infile: str: 入力音声ファイルのパス。
536    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
537    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
538    :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
539    :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
540    :param device_name: str: 使用するデバイス名 ("cuda", "cpu")。空文字の場合は自動判別。(default: "")
541    """
542    if device_name == "":
543        model = whisper.load_model(model_name)
544    else:
545        model = whisper.load_model(model_name, device=device_name)
546
547    print(f"Using device [{model.device}] (openai/whisper)")
548
549    language_arg = lang if str(lang).strip() else None
550    result = model.transcribe(
551        infile,
552        language=language_arg,
553        verbose=True,
554        condition_on_previous_text=False,
555    )
556
557    with open(outfile1, "w", encoding="utf-8") as f:
558        for seg in result["segments"]:
559            f.write(f"[{seg['start']:.2f} - {seg['end']:.2f}] {seg['text']}\n")
560
561    with open(outfile2, "w", encoding="utf-8") as f:
562        f.write(result["text"])
563
564    print(result["text"])
565
566
567def transcribe_audio(infile, outfile1, outfile2, args):
568    """検出された環境に応じて適切なWhisper実装で音声ファイルを文字起こしする。
569
570    グローバル変数 `USE_FASTER` の値に基づいて、`transcribe_audio_faster` または
571    `transcribe_audio_whisper` のいずれかを呼び出します。
572    コマンドライン引数を直接各関数に渡します。
573
574    :param infile: str: 入力音声ファイルのパス。
575    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
576    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
577    :param args: argparse.Namespace: コマンドライン引数を格納したオブジェクト。
578    """
579    if USE_FASTER:
580        transcribe_audio_faster(
581            infile,
582            outfile1,
583            outfile2,
584            args.model,
585            args.lang,
586            args.device,
587            beam_size=args.beam_size,
588            chunk_length=args.chunk_length,
589            temperature=args.temperature,
590            condition_on_previous_text=bool_from_int(args.condition_on_previous_text),
591            compression_ratio_threshold=args.compression_ratio_threshold,
592            log_prob_threshold=args.log_prob_threshold,
593            no_speech_threshold=args.no_speech_threshold,
594            repetition_penalty=args.repetition_penalty,
595            no_repeat_ngram_size=args.no_repeat_ngram_size,
596            max_bad_segments=args.max_bad_segments,
597        )
598    else:
599        transcribe_audio_whisper(
600            infile,
601            outfile1,
602            outfile2,
603            args.model,
604            args.lang,
605            args.device,
606        )
607
608
609# =========================
610# メイン
611# =========================
612def main():
613    """スクリプトのメインエントリポイント。コマンドライン引数を解析し、文字起こし処理を実行する。
614
615    `argparse` を使用してコマンドライン引数を定義・解析します。
616    入力ファイルパスに基づいて文字起こし処理をループ実行し、
617    `check_gpu_torch` と `check_gpu` でGPU情報を表示した後、`transcribe_audio` 関数を呼び出します。
618    処理結果は指定された出力ファイルに保存されます。
619    """
620    global pause
621
622    parser = argparse.ArgumentParser(description="Whisper音声文字起こしツール")
623    parser.add_argument("infile", type=str, help="入力音声ファイル名(glob可)")
624    parser.add_argument("--outfile1", type=str, default="", help="出力テキストファイル名 (時間範囲入り)")
625    parser.add_argument("--outfile2", type=str, default="", help="出力テキストファイル名 (時間範囲なし)")
626    parser.add_argument("-m", "--model", type=str, default="base", help="Whisperモデル名 (default: base)")
627    parser.add_argument("-d", "--device", type=str, default="", help="GPU/CPU/DMLデバイス名 (default: auto)")
628    parser.add_argument("-l", "--lang", type=str, default="ja", help="使用言語。空文字なら自動判定 (default: ja)")
629    parser.add_argument("--pause", type=int, default=0, help="終了時にENTERキー入力を要求するか (default: 0)")
630
631    # faster-whisper の安定化パラメータ
632    parser.add_argument("--beam-size", type=int, default=5, help="beam size (default: 5)")
633    parser.add_argument("--chunk-length", type=int, default=30, help="内部チャンク長[秒] (default: 30)")
634    parser.add_argument(
635        "--temperature",
636        type=str,
637        default="0,0.2,0.4,0.6",
638        help="temperature。'0' または '0,0.2,0.4,0.6' のように指定 (default: 0,0.2,0.4,0.6)",
639    )
640    parser.add_argument(
641        "--condition-on-previous-text",
642        type=int,
643        default=0,
644        help="前の認識結果を次チャンクの文脈に使うか。長時間音声では0推奨 (default: 0)",
645    )
646    parser.add_argument(
647        "--compression-ratio-threshold",
648        type=float,
649        default=2.4,
650        help="反復テキスト検出用しきい値。None にはしない (default: 2.4)",
651    )
652    parser.add_argument(
653        "--log-prob-threshold",
654        type=float,
655        default=-1.0,
656        help="低信頼デコード検出しきい値 (default: -1.0)",
657    )
658    parser.add_argument(
659        "--no-speech-threshold",
660        type=float,
661        default=0.6,
662        help="無音判定しきい値 (default: 0.6)",
663    )
664    parser.add_argument(
665        "--repetition-penalty",
666        type=float,
667        default=1.05,
668        help="反復抑制ペナルティ。対応版のみ有効 (default: 1.05)",
669    )
670    parser.add_argument(
671        "--no-repeat-ngram-size",
672        type=int,
673        default=3,
674        help="n-gram反復抑制。対応版のみ有効。0で無効 (default: 3)",
675    )
676    parser.add_argument(
677        "--max-bad-segments",
678        type=int,
679        default=3,
680        help="反復幻覚らしいセグメントが連続したら停止する数 (default: 3)",
681    )
682
683    args = parser.parse_args()
684    pause = args.pause
685
686    files = glob.glob(args.infile)
687    print()
688    print(f"infile={args.infile}")
689    if len(files) == 0:
690        print("\nError: No file found.\n")
691        return
692
693    print(f"  files: {files}")
694    print(f"model={args.model}")
695    print(f"lang={args.lang!r}")
696    print(f"device={args.device}")
697
698    print()
699    check_gpu_torch(torch_mod)
700    check_gpu()
701
702    for f in files:
703        if args.outfile1 == "":
704            outfile1 = os.path.splitext(os.path.basename(f))[0] + "-time.txt"
705        else:
706            outfile1 = args.outfile1
707
708        if args.outfile2 == "":
709            outfile2 = os.path.splitext(os.path.basename(f))[0] + ".txt"
710        else:
711            outfile2 = args.outfile2
712
713        print("\n=== Transcribing ===")
714        print(f"infile  = {f}")
715        print(f"outfile1= {outfile1}")
716        print(f"outfile2= {outfile2}")
717
718        transcribe_audio(f, outfile1, outfile2, args)
719
720
721if __name__ == "__main__":
722    main()
723    terminate()