transcribe_simple.py ダウンロード/コピー
transcribe_simple.py
transcribe_simple.py
1"""音声ファイルからWhisperモデルで文字起こしを実行するスクリプト。
2
3このスクリプトは、Windows環境では `faster-whisper` を、Linux/macOS環境では `openai/whisper` を使用して、
4音声ファイルの文字起こしを行います。コマンドライン引数で入力ファイル、出力ファイル、モデル、言語などを指定できます。
5長時間の音声に対する反復幻覚抑制機能も含まれています。
6
7:doc:`transcribe_simple_usage`
8"""
9
10import os
11import glob
12import argparse
13import traceback
14import platform
15import subprocess
16import inspect
17import re
18import gzip
19from collections import Counter
20
21# =========================
22# グローバル設定
23# =========================
24USE_FASTER = False # Windows: faster-whisper / Linux, macOS: whisper
25whisper = None
26WhisperModel = None
27#COMPUTE_TYPE = "int8" # CPU / DML では int8 が最速
28
29pause = 0
30
31
32# =========================
33# ユーティリティ
34# =========================
35def terminate():
36 """スクリプトを終了する。必要に応じてユーザーの入力待ちを行う。
37
38 グローバル変数 `pause` が `0` でない場合、`ENTER` キーが押されるまでプロンプトを表示し、
39 その後スクリプトを終了します。
40 """
41 if pause:
42 input("\nPress ENTER to terminate\n")
43 exit()
44
45
46def safe_import_torch():
47 """PyTorchライブラリを安全にインポートする。
48
49 PyTorchのインポートを試行し、失敗した場合はエラーメッセージとトレースバックを表示して
50 `None` を返します。
51
52 :returns: module or None: PyTorchモジュール (`torch`) またはインポートに失敗した場合は `None`。
53 """
54 try:
55 import torch
56 return torch
57 except Exception:
58 print("torch import failed")
59 traceback.print_exc()
60 return None
61
62
63def check_gpu_torch(torch_mod):
64 """PyTorchがCUDA GPUを利用可能かを確認し、情報を表示する。
65
66 引数で渡されたtorchモジュールがCUDAを利用できる場合、デバイス名とCUDAバージョンを表示する。
67 利用できない場合は、その旨を表示する。
68
69 :param torch_mod: module or None: PyTorchモジュールまたはNone。
70 """
71 if torch_mod is not None and torch_mod.cuda.is_available():
72 print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
73 print(f"Torch CUDA ver: {torch_mod.version.cuda}")
74 else:
75 print("No CUDA GPU available for Torch")
76 print(" (this does not affect faster-whisper)")
77
78def check_gpu():
79 """使用中のWhisper実装に応じてGPUの利用可能性を確認し、情報を表示する。
80
81 `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を介してCUDAデバイスをチェックする。
82 `False` の場合 (openai/whisperを使用)、`safe_import_torch` を利用してPyTorchのCUDA GPUをチェックする。
83 """
84 if USE_FASTER:
85 try:
86 import ctranslate2
87
88 count = ctranslate2.get_cuda_device_count()
89 if count > 0:
90 print(f"CUDA GPU available for CTranslate2: {count} device(s)")
91 else:
92 print("No CUDA GPU available for CTranslate2")
93 except Exception:
94 print("Failed to check CTranslate2 CUDA support")
95 traceback.print_exc()
96 return
97
98 # openai-whisperではPyTorch基準
99 torch_mod = safe_import_torch()
100 if torch_mod is not None and torch_mod.cuda.is_available():
101 print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
102 print(f"Torch CUDA ver: {torch_mod.version.cuda}")
103 else:
104 print("No CUDA GPU available for PyTorch")
105
106
107def detect_device_torch():
108 """PyTorchが使用するデバイスを自動判定する。
109
110 優先順位 `CUDA -> DML -> CPU` で利用可能なデバイスを判定し、そのデバイス名を返します。
111
112 :returns: str: 検出されたデバイス名 ("cuda", "dml", "cpu" のいずれか)。
113 """
114 # CUDA
115 torch_mod = safe_import_torch()
116 if torch_mod is not None and torch_mod.cuda.is_available():
117 return "cuda"
118
119 # DirectML
120 try:
121 import ctranslate2
122 if "dml" in ctranslate2.get_supported_devices():
123 return "dml"
124 except Exception:
125 pass
126
127 # CPU
128 return "cpu"
129
130def detect_device():
131 """使用するWhisper実装に応じてデバイスを自動判定する。
132
133 `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を通じてCUDAの利用可能性を確認し、
134 優先的に `cuda` を返します。そうでなければ `cpu` を返します。
135 `USE_FASTER` が `False` の場合 (openai/whisperを使用)、PyTorchを通じてCUDAの利用可能性を確認し、
136 優先的に `cuda` を返します。そうでなければ `cpu` を返します。
137
138 :returns: str: 検出されたデバイス名 ("cuda", "cpu" のいずれか)。
139 """
140 if USE_FASTER:
141 try:
142 import ctranslate2
143
144 if ctranslate2.get_cuda_device_count() > 0:
145 return "cuda"
146 except Exception:
147 traceback.print_exc()
148
149 return "cpu"
150
151 torch_mod = safe_import_torch()
152 if torch_mod is not None and torch_mod.cuda.is_available():
153 return "cuda"
154
155 return "cpu"
156
157def get_audio_duration(infile):
158 """`ffprobe` コマンドを使用して音声ファイルの長さを取得する。
159
160 `ffprobe` を外部プロセスとして実行し、音声ファイルの長さを秒単位で解析して返します。
161 `ffprobe` の実行に失敗した場合は警告を表示し、`None` を返します。
162
163 :param infile: str: 入力音声ファイルのパス。
164 :returns: float or None: 音声の長さ(秒単位)または、取得できなかった場合はNone。
165 """
166 try:
167 cmd = [
168 "ffprobe", "-v", "error",
169 "-show_entries", "format=duration",
170 "-of", "default=noprint_wrappers=1:nokey=1",
171 infile,
172 ]
173 result = subprocess.run(
174 cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True
175 )
176 return float(result.stdout.strip())
177 except Exception:
178 print("Warning: failed to get duration via ffprobe, fallback to model info.")
179 return None
180
181
182def format_hms(sec):
183 """秒数を HH:MM:SS.s 形式の文字列に変換する。
184
185 渡された秒数を時、分、秒に分解し、指定されたフォーマットで整形した文字列を返します。
186 `None` が渡された場合は `??:??:??` を返します。
187
188 :param sec: float or None: 変換する秒数。
189 :returns: str: フォーマットされた時間文字列。
190 """
191 if sec is None:
192 return "??:??:??"
193 sec = float(sec)
194 h = int(sec // 3600)
195 m = int((sec % 3600) // 60)
196 s = sec % 60
197 return f"{h:02d}:{m:02d}:{s:04.1f}"
198
199
200def parse_temperature(value):
201 """temperatureパラメータの値を浮動小数点数または浮動小数点数のタプルに変換する。
202
203 カンマ区切りの文字列を解析し、単一の数値であれば `float`、複数の数値であれば `tuple[float, ...]` を返します。
204 空の文字列や不正な値が指定された場合は `ValueError` を発生させます。
205
206 :param value: Union[str, float]: temperatureとして指定された値。
207 :returns: Union[float, tuple[float, ...]]: 解析されたtemperatureの値。
208 :raises ValueError: temperatureが空の場合。
209 """
210 text = str(value).strip()
211 if "," not in text:
212 return float(text)
213 temps = tuple(float(x.strip()) for x in text.split(",") if x.strip() != "")
214 if not temps:
215 raise ValueError("temperature is empty")
216 return temps
217
218
219def text_compression_ratio(text):
220 """テキストの簡易的なgzip圧縮率を計算する。
221
222 入力テキストをUTF-8でエンコードし、gzipで圧縮します。
223 非圧縮バイト長を圧縮バイト長で割ることで圧縮率を算出します。
224 短いテキストや圧縮結果が空の場合は `0.0` を返します。
225
226 :param text: str: 圧縮率を計算するテキスト。
227 :returns: float: テキストの圧縮率。
228 """
229 raw = text.encode("utf-8", errors="ignore")
230 if len(raw) < 80:
231 return 0.0
232 comp = gzip.compress(raw)
233 if len(comp) == 0:
234 return 0.0
235 return len(raw) / len(comp)
236
237
238def normalize_tokens(text):
239 """反復検出のためにテキストをゆるくトークン化する。
240
241 英数字の単語、およびひらがな・カタカナ・漢字の連続を単語として抽出し、小文字に変換してリストで返します。
242
243 :param text: str: トークン化する入力テキスト。
244 :returns: list[str]: 正規化されたトークンのリスト。
245 """
246 return re.findall(r"[A-Za-z0-9]+(?:'[A-Za-z0-9]+)?|[ぁ-んァ-ヶー一-龥]+", text.lower())
247
248
249def looks_like_repetition_loop(text, max_token_repeat=12, compression_ratio_limit=3.2):
250 """Whisperモデルによって生成された反復幻覚らしいセグメントを検出する。
251
252 入力テキストの長さが80文字未満の場合は検出しない。
253 `normalize_tokens` でトークン化した後、最も頻繁に出現するトークンの繰り返し回数とその割合をチェックする。
254 また、`text_compression_ratio` を用いてテキストの圧縮率が高すぎる場合も反復と見なす。
255 これらの条件に合致する場合に `True` とその理由を返します。
256
257 判定の狙い:
258 - "6th, 6th, 6th, ..." のような単語反復
259 - 同じ語句が異常に多い高圧縮テキスト
260
261 強すぎると正しい復唱も消すので、講義音声向けにやや控えめにしている。
262
263 :param text: str: 検出対象のテキストセグメント。
264 :param max_token_repeat: int: 単語反復とみなす最小繰り返し回数。(default: 12)
265 :param compression_ratio_limit: float: 圧縮率がこの値以上の場合に反復とみなすしきい値。(default: 3.2)
266 :returns: tuple[bool, str]: 反復幻覚らしい場合は (True, 理由), そうでない場合は (False, "")。
267 """
268 stripped = text.strip()
269 if len(stripped) < 80:
270 return False, ""
271
272 tokens = normalize_tokens(stripped)
273 if len(tokens) >= max_token_repeat:
274 counts = Counter(tokens)
275 token, count = counts.most_common(1)[0]
276 if count >= max_token_repeat and count / len(tokens) >= 0.45:
277 return True, f"token '{token}' repeated {count}/{len(tokens)} times"
278
279 cr = text_compression_ratio(stripped)
280 if cr >= compression_ratio_limit:
281 # 圧縮率だけだと箇条書きや専門語の繰り返しも拾うので、長めの出力に限定する。
282 if len(tokens) >= 30 or len(stripped) >= 300:
283 return True, f"text compression ratio {cr:.2f} >= {compression_ratio_limit:.2f}"
284
285 return False, ""
286
287
288def bool_from_int(value):
289 """整数値をブール値に変換する。
290
291 整数 `0` を `False` に、それ以外の整数を `True` に変換します。
292
293 :param value: int: 変換する整数値。
294 :returns: bool: 変換されたブール値。
295 """
296 return bool(int(value))
297
298
299# =========================
300# OS 判定 & ライブラリ import
301# =========================
302os_name = platform.system()
303print(f"Detected OS: {os_name}")
304
305try:
306 if os_name == "Windows":
307 from faster_whisper import WhisperModel
308 USE_FASTER = True
309 print("Using faster-whisper (Windows)")
310 else:
311 import whisper
312 USE_FASTER = False
313 print("Using openai/whisper (Linux/macOS)")
314except Exception:
315 print("whisper / faster-whisper import failed")
316 traceback.print_exc()
317 terminate()
318
319torch_mod = safe_import_torch()
320#if torch_mod is None:
321# terminate()
322
323
324# =========================
325# 文字起こし本体
326# =========================
327def transcribe_audio_faster(
328 infile,
329 outfile1,
330 outfile2,
331 model_name,
332 lang,
333 device_name="",
334 beam_size=5,
335 chunk_length=30,
336 temperature="0,0.2,0.4,0.6",
337 condition_on_previous_text=False,
338 compression_ratio_threshold=2.4,
339 log_prob_threshold=-1.0,
340 no_speech_threshold=0.6,
341 repetition_penalty=1.05,
342 no_repeat_ngram_size=3,
343 max_bad_segments=3,
344):
345 """faster-whisperライブラリを使用して音声ファイルを文字起こしする。
346
347 faster-whisperのモデルをロードし、指定されたパラメータで文字起こしを実行します。
348 セグメントごとに進捗を表示し、指定されたファイルに出力します。
349 長時間の音声処理において反復幻覚を抑制するための機能も含まれます。
350 デバイス、計算タイプは自動判別されます。
351
352 重要:
353 - offset / duration は faster-whisper の transcribe() には渡さない。
354 - 長時間音声では condition_on_previous_text=False をデフォルトにする。
355
356 :param infile: str: 入力音声ファイルのパス。
357 :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
358 :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
359 :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
360 :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
361 :param device_name: str: 使用するデバイス名 ("cuda", "cpu", "dml")。空文字の場合は自動判別。(default: "")
362 :param beam_size: int: ビームサーチのサイズ。(default: 5)
363 :param chunk_length: int: 内部チャンクの長さ(秒)。(default: 30)
364 :param temperature: Union[str, float, tuple[float, ...]]: temperatureパラメータ。文字列形式("0,0.2,0.4")または数値で指定。(default: "0,0.2,0.4,0.6")
365 :param condition_on_previous_text: bool: 前の認識結果を次のチャンクの文脈として使用するかどうか。(default: False)
366 :param compression_ratio_threshold: float: 反復テキスト検出用の圧縮率しきい値。(default: 2.4)
367 :param log_prob_threshold: float: 低信頼デコード検出のしきい値。(default: -1.0)
368 :param no_speech_threshold: float: 無音判定のしきい値。(default: 0.6)
369 :param repetition_penalty: float: 反復抑制ペナルティ。(default: 1.05)
370 :param no_repeat_ngram_size: int: n-gram反復抑制のサイズ。0で無効。(default: 3)
371 :param max_bad_segments: int: 反復幻覚らしいセグメントが連続した場合に文字起こしを停止する回数。(default: 3)
372 """
373 if device_name == "":
374 device_name = detect_device()
375
376 compute_type = (
377 "int8_float16" if device_name == "cuda"
378 else "int8"
379 )
380
381 print(
382 f"Using device [{device_name}] (faster-whisper, compute_type={compute_type})",
383 flush=True,
384 )
385
386 model = WhisperModel(
387 model_name,
388 device=device_name,
389 compute_type=compute_type,
390 )
391
392 transcribe_params = inspect.signature(model.transcribe).parameters
393
394 # language="" を指定した場合は自動判定にする。
395 language_arg = lang if str(lang).strip() else None
396
397 requested_kwargs = {
398 "language": language_arg,
399 "vad_filter": True,
400 "beam_size": beam_size,
401 "temperature": parse_temperature(temperature),
402 "condition_on_previous_text": condition_on_previous_text,
403 "compression_ratio_threshold": compression_ratio_threshold,
404 "log_prob_threshold": log_prob_threshold,
405 "no_speech_threshold": no_speech_threshold,
406 "repetition_penalty": repetition_penalty,
407 "no_repeat_ngram_size": no_repeat_ngram_size,
408 "suppress_blank": True,
409 }
410
411 if "chunk_length" in transcribe_params:
412 requested_kwargs["chunk_length"] = chunk_length
413 else:
414 print(
415 "chunk_length is not supported by this faster-whisper version; using default chunking.",
416 flush=True,
417 )
418
419 # VAD パラメータはバージョン差があるので、対応している場合だけ渡す。
420 if "vad_parameters" in transcribe_params:
421 requested_kwargs["vad_parameters"] = {
422 # 既定値より少し長い無音で分割し、短い講義中の間を無音扱いしすぎない。
423 "min_silence_duration_ms": 500,
424 }
425
426 # word_timestamps と組み合わせると hallucination_silence_threshold が効く版もある。
427 # ただし word_timestamps=True は重くなるので、ここでは使わない。
428
429 # 古い faster-whisper でも動くよう、存在する引数だけ渡す。
430 transcribe_kwargs = {
431 k: v for k, v in requested_kwargs.items() if k in transcribe_params
432 }
433
434 print(f"transcribe options: {transcribe_kwargs}", flush=True)
435
436 segments, info = model.transcribe(infile, **transcribe_kwargs)
437
438 duration = getattr(info, "duration", None)
439 if duration is None:
440 duration = get_audio_duration(infile)
441
442 if duration is not None:
443 print(f"Audio length: {duration:.1f} sec ({format_hms(duration)})", flush=True)
444
445 print("Start transcription...", flush=True)
446 print(f"Save to [{outfile1}]")
447 print(f"Save to [{outfile2}]", flush=True)
448
449 seg_count = 0
450 bad_count = 0
451 all_text = []
452
453 # ここで list(segments) にしてしまうと、全処理が終わるまで何も表示されない。
454 # generator を1件ずつ消費して、そのたびに進捗を表示する。
455 with open(outfile1, "w", encoding="utf-8") as f_time, \
456 open(outfile2, "w", encoding="utf-8") as f_text:
457 for seg in segments:
458 seg_count += 1
459 text = seg.text or ""
460
461 is_bad, reason = looks_like_repetition_loop(text)
462 if is_bad:
463 bad_count += 1
464 if duration and duration > 0:
465 pct = min(100.0, 100.0 * seg.end / duration)
466 progress = f"{pct:6.2f}%"
467 else:
468 progress = " ??.??%"
469
470 preview = text.strip().replace("\n", " ")[:160]
471 print(
472 f"[WARN] skipped probable repetition loop "
473 f"at {format_hms(seg.start)} - {format_hms(seg.end)} "
474 f"({progress}): {reason}\n"
475 f" preview: {preview}",
476 flush=True,
477 )
478
479 f_time.write(
480 f"[{seg.start:.2f} - {seg.end:.2f}] "
481 f"[SKIPPED: probable repetition loop: {reason}]\n"
482 )
483 f_time.flush()
484
485 if bad_count >= max_bad_segments:
486 print(
487 f"[WARN] {bad_count} suspicious segments were detected. "
488 "Stopping transcription to avoid a long hallucination tail.",
489 flush=True,
490 )
491 break
492 continue
493
494 bad_count = 0
495 all_text.append(text)
496
497 f_time.write(f"[{seg.start:.2f} - {seg.end:.2f}] {text}\n")
498 f_time.flush()
499
500 f_text.write(text)
501 f_text.flush()
502
503 if duration and duration > 0:
504 pct = min(100.0, 100.0 * seg.end / duration)
505 progress = f"{pct:6.2f}%"
506 else:
507 progress = " ??.??%"
508
509 print(
510 f"[{seg_count:04d}] {progress} "
511 f"{format_hms(seg.start)} - {format_hms(seg.end)} "
512 f"{text}",
513 flush=True,
514 )
515
516 text = "".join(all_text)
517 print(f"\nSegments read: {seg_count}", flush=True)
518 print("\n=== Transcribed text ===", flush=True)
519 print(text, flush=True)
520
521
522def transcribe_audio_whisper(
523 infile,
524 outfile1,
525 outfile2,
526 model_name,
527 lang,
528 device_name="",
529):
530 """openai/whisperライブラリを使用して音声ファイルを文字起こしする。
531
532 openai/whisperモデルをロードし、指定されたパラメータで音声ファイルを一括で文字起こしします。
533 結果は指定された2つのファイルに出力されます。
534
535 :param infile: str: 入力音声ファイルのパス。
536 :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
537 :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
538 :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
539 :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
540 :param device_name: str: 使用するデバイス名 ("cuda", "cpu")。空文字の場合は自動判別。(default: "")
541 """
542 if device_name == "":
543 model = whisper.load_model(model_name)
544 else:
545 model = whisper.load_model(model_name, device=device_name)
546
547 print(f"Using device [{model.device}] (openai/whisper)")
548
549 language_arg = lang if str(lang).strip() else None
550 result = model.transcribe(
551 infile,
552 language=language_arg,
553 verbose=True,
554 condition_on_previous_text=False,
555 )
556
557 with open(outfile1, "w", encoding="utf-8") as f:
558 for seg in result["segments"]:
559 f.write(f"[{seg['start']:.2f} - {seg['end']:.2f}] {seg['text']}\n")
560
561 with open(outfile2, "w", encoding="utf-8") as f:
562 f.write(result["text"])
563
564 print(result["text"])
565
566
567def transcribe_audio(infile, outfile1, outfile2, args):
568 """検出された環境に応じて適切なWhisper実装で音声ファイルを文字起こしする。
569
570 グローバル変数 `USE_FASTER` の値に基づいて、`transcribe_audio_faster` または
571 `transcribe_audio_whisper` のいずれかを呼び出します。
572 コマンドライン引数を直接各関数に渡します。
573
574 :param infile: str: 入力音声ファイルのパス。
575 :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
576 :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
577 :param args: argparse.Namespace: コマンドライン引数を格納したオブジェクト。
578 """
579 if USE_FASTER:
580 transcribe_audio_faster(
581 infile,
582 outfile1,
583 outfile2,
584 args.model,
585 args.lang,
586 args.device,
587 beam_size=args.beam_size,
588 chunk_length=args.chunk_length,
589 temperature=args.temperature,
590 condition_on_previous_text=bool_from_int(args.condition_on_previous_text),
591 compression_ratio_threshold=args.compression_ratio_threshold,
592 log_prob_threshold=args.log_prob_threshold,
593 no_speech_threshold=args.no_speech_threshold,
594 repetition_penalty=args.repetition_penalty,
595 no_repeat_ngram_size=args.no_repeat_ngram_size,
596 max_bad_segments=args.max_bad_segments,
597 )
598 else:
599 transcribe_audio_whisper(
600 infile,
601 outfile1,
602 outfile2,
603 args.model,
604 args.lang,
605 args.device,
606 )
607
608
609# =========================
610# メイン
611# =========================
612def main():
613 """スクリプトのメインエントリポイント。コマンドライン引数を解析し、文字起こし処理を実行する。
614
615 `argparse` を使用してコマンドライン引数を定義・解析します。
616 入力ファイルパスに基づいて文字起こし処理をループ実行し、
617 `check_gpu_torch` と `check_gpu` でGPU情報を表示した後、`transcribe_audio` 関数を呼び出します。
618 処理結果は指定された出力ファイルに保存されます。
619 """
620 global pause
621
622 parser = argparse.ArgumentParser(description="Whisper音声文字起こしツール")
623 parser.add_argument("infile", type=str, help="入力音声ファイル名(glob可)")
624 parser.add_argument("--outfile1", type=str, default="", help="出力テキストファイル名 (時間範囲入り)")
625 parser.add_argument("--outfile2", type=str, default="", help="出力テキストファイル名 (時間範囲なし)")
626 parser.add_argument("-m", "--model", type=str, default="base", help="Whisperモデル名 (default: base)")
627 parser.add_argument("-d", "--device", type=str, default="", help="GPU/CPU/DMLデバイス名 (default: auto)")
628 parser.add_argument("-l", "--lang", type=str, default="ja", help="使用言語。空文字なら自動判定 (default: ja)")
629 parser.add_argument("--pause", type=int, default=0, help="終了時にENTERキー入力を要求するか (default: 0)")
630
631 # faster-whisper の安定化パラメータ
632 parser.add_argument("--beam-size", type=int, default=5, help="beam size (default: 5)")
633 parser.add_argument("--chunk-length", type=int, default=30, help="内部チャンク長[秒] (default: 30)")
634 parser.add_argument(
635 "--temperature",
636 type=str,
637 default="0,0.2,0.4,0.6",
638 help="temperature。'0' または '0,0.2,0.4,0.6' のように指定 (default: 0,0.2,0.4,0.6)",
639 )
640 parser.add_argument(
641 "--condition-on-previous-text",
642 type=int,
643 default=0,
644 help="前の認識結果を次チャンクの文脈に使うか。長時間音声では0推奨 (default: 0)",
645 )
646 parser.add_argument(
647 "--compression-ratio-threshold",
648 type=float,
649 default=2.4,
650 help="反復テキスト検出用しきい値。None にはしない (default: 2.4)",
651 )
652 parser.add_argument(
653 "--log-prob-threshold",
654 type=float,
655 default=-1.0,
656 help="低信頼デコード検出しきい値 (default: -1.0)",
657 )
658 parser.add_argument(
659 "--no-speech-threshold",
660 type=float,
661 default=0.6,
662 help="無音判定しきい値 (default: 0.6)",
663 )
664 parser.add_argument(
665 "--repetition-penalty",
666 type=float,
667 default=1.05,
668 help="反復抑制ペナルティ。対応版のみ有効 (default: 1.05)",
669 )
670 parser.add_argument(
671 "--no-repeat-ngram-size",
672 type=int,
673 default=3,
674 help="n-gram反復抑制。対応版のみ有効。0で無効 (default: 3)",
675 )
676 parser.add_argument(
677 "--max-bad-segments",
678 type=int,
679 default=3,
680 help="反復幻覚らしいセグメントが連続したら停止する数 (default: 3)",
681 )
682
683 args = parser.parse_args()
684 pause = args.pause
685
686 files = glob.glob(args.infile)
687 print()
688 print(f"infile={args.infile}")
689 if len(files) == 0:
690 print("\nError: No file found.\n")
691 return
692
693 print(f" files: {files}")
694 print(f"model={args.model}")
695 print(f"lang={args.lang!r}")
696 print(f"device={args.device}")
697
698 print()
699 check_gpu_torch(torch_mod)
700 check_gpu()
701
702 for f in files:
703 if args.outfile1 == "":
704 outfile1 = os.path.splitext(os.path.basename(f))[0] + "-time.txt"
705 else:
706 outfile1 = args.outfile1
707
708 if args.outfile2 == "":
709 outfile2 = os.path.splitext(os.path.basename(f))[0] + ".txt"
710 else:
711 outfile2 = args.outfile2
712
713 print("\n=== Transcribing ===")
714 print(f"infile = {f}")
715 print(f"outfile1= {outfile1}")
716 print(f"outfile2= {outfile2}")
717
718 transcribe_audio(f, outfile1, outfile2, args)
719
720
721if __name__ == "__main__":
722 main()
723 terminate()