"""音声ファイルからWhisperモデルで文字起こしを実行するスクリプト。

このスクリプトは、Windows環境では `faster-whisper` を、Linux/macOS環境では `openai/whisper` を使用して、
音声ファイルの文字起こしを行います。コマンドライン引数で入力ファイル、出力ファイル、モデル、言語などを指定できます。
長時間の音声に対する反復幻覚抑制機能も含まれています。

:doc:`transcribe_simple_usage`
"""

import os
import glob
import argparse
import traceback
import platform
import subprocess
import inspect
import re
import gzip
from collections import Counter

# =========================
# グローバル設定
# =========================
USE_FASTER = False          # Windows: faster-whisper / Linux, macOS: whisper
whisper = None
WhisperModel = None
#COMPUTE_TYPE = "int8"       # CPU / DML では int8 が最速

pause = 0


# =========================
# ユーティリティ
# =========================
def terminate():
    """スクリプトを終了する。必要に応じてユーザーの入力待ちを行う。

    グローバル変数 `pause` が `0` でない場合、`ENTER` キーが押されるまでプロンプトを表示し、
    その後スクリプトを終了します。
    """
    if pause:
        input("\nPress ENTER to terminate\n")
    exit()


def safe_import_torch():
    """PyTorchライブラリを安全にインポートする。

    PyTorchのインポートを試行し、失敗した場合はエラーメッセージとトレースバックを表示して
    `None` を返します。

    :returns: module or None: PyTorchモジュール (`torch`) またはインポートに失敗した場合は `None`。
    """
    try:
        import torch
        return torch
    except Exception:
        print("torch import failed")
        traceback.print_exc()
        return None


def check_gpu_torch(torch_mod):
    """PyTorchがCUDA GPUを利用可能かを確認し、情報を表示する。

    引数で渡されたtorchモジュールがCUDAを利用できる場合、デバイス名とCUDAバージョンを表示する。
    利用できない場合は、その旨を表示する。

    :param torch_mod: module or None: PyTorchモジュールまたはNone。
    """
    if torch_mod is not None and torch_mod.cuda.is_available():
        print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
        print(f"Torch CUDA ver: {torch_mod.version.cuda}")
    else:
        print("No CUDA GPU available for Torch")
        print("   (this does not affect faster-whisper)")

def check_gpu():
    """使用中のWhisper実装に応じてGPUの利用可能性を確認し、情報を表示する。

    `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を介してCUDAデバイスをチェックする。
    `False` の場合 (openai/whisperを使用)、`safe_import_torch` を利用してPyTorchのCUDA GPUをチェックする。
    """
    if USE_FASTER:
        try:
            import ctranslate2

            count = ctranslate2.get_cuda_device_count()
            if count > 0:
                print(f"CUDA GPU available for CTranslate2: {count} device(s)")
            else:
                print("No CUDA GPU available for CTranslate2")
        except Exception:
            print("Failed to check CTranslate2 CUDA support")
            traceback.print_exc()
        return

    # openai-whisperではPyTorch基準
    torch_mod = safe_import_torch()
    if torch_mod is not None and torch_mod.cuda.is_available():
        print(f"CUDA GPU name: {torch_mod.cuda.get_device_name(0)}")
        print(f"Torch CUDA ver: {torch_mod.version.cuda}")
    else:
        print("No CUDA GPU available for PyTorch")
        

def detect_device_torch():
    """PyTorchが使用するデバイスを自動判定する。

    優先順位 `CUDA -> DML -> CPU` で利用可能なデバイスを判定し、そのデバイス名を返します。

    :returns: str: 検出されたデバイス名 ("cuda", "dml", "cpu" のいずれか)。
    """
    # CUDA
    torch_mod = safe_import_torch()
    if torch_mod is not None and torch_mod.cuda.is_available():
        return "cuda"

    # DirectML
    try:
        import ctranslate2
        if "dml" in ctranslate2.get_supported_devices():
            return "dml"
    except Exception:
        pass

    # CPU
    return "cpu"

def detect_device():
    """使用するWhisper実装に応じてデバイスを自動判定する。

    `USE_FASTER` が `True` の場合 (faster-whisperを使用)、CTranslate2を通じてCUDAの利用可能性を確認し、
    優先的に `cuda` を返します。そうでなければ `cpu` を返します。
    `USE_FASTER` が `False` の場合 (openai/whisperを使用)、PyTorchを通じてCUDAの利用可能性を確認し、
    優先的に `cuda` を返します。そうでなければ `cpu` を返します。

    :returns: str: 検出されたデバイス名 ("cuda", "cpu" のいずれか)。
    """
    if USE_FASTER:
        try:
            import ctranslate2

            if ctranslate2.get_cuda_device_count() > 0:
                return "cuda"
        except Exception:
            traceback.print_exc()

        return "cpu"

    torch_mod = safe_import_torch()
    if torch_mod is not None and torch_mod.cuda.is_available():
        return "cuda"

    return "cpu"

def get_audio_duration(infile):
    """`ffprobe` コマンドを使用して音声ファイルの長さを取得する。

    `ffprobe` を外部プロセスとして実行し、音声ファイルの長さを秒単位で解析して返します。
    `ffprobe` の実行に失敗した場合は警告を表示し、`None` を返します。

    :param infile: str: 入力音声ファイルのパス。
    :returns: float or None: 音声の長さ（秒単位）または、取得できなかった場合はNone。
    """
    try:
        cmd = [
            "ffprobe", "-v", "error",
            "-show_entries", "format=duration",
            "-of", "default=noprint_wrappers=1:nokey=1",
            infile,
        ]
        result = subprocess.run(
            cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True
        )
        return float(result.stdout.strip())
    except Exception:
        print("Warning: failed to get duration via ffprobe, fallback to model info.")
        return None


def format_hms(sec):
    """秒数を HH:MM:SS.s 形式の文字列に変換する。

    渡された秒数を時、分、秒に分解し、指定されたフォーマットで整形した文字列を返します。
    `None` が渡された場合は `??:??:??` を返します。

    :param sec: float or None: 変換する秒数。
    :returns: str: フォーマットされた時間文字列。
    """
    if sec is None:
        return "??:??:??"
    sec = float(sec)
    h = int(sec // 3600)
    m = int((sec % 3600) // 60)
    s = sec % 60
    return f"{h:02d}:{m:02d}:{s:04.1f}"


def parse_temperature(value):
    """temperatureパラメータの値を浮動小数点数または浮動小数点数のタプルに変換する。

    カンマ区切りの文字列を解析し、単一の数値であれば `float`、複数の数値であれば `tuple[float, ...]` を返します。
    空の文字列や不正な値が指定された場合は `ValueError` を発生させます。

    :param value: Union[str, float]: temperatureとして指定された値。
    :returns: Union[float, tuple[float, ...]]: 解析されたtemperatureの値。
    :raises ValueError: temperatureが空の場合。
    """
    text = str(value).strip()
    if "," not in text:
        return float(text)
    temps = tuple(float(x.strip()) for x in text.split(",") if x.strip() != "")
    if not temps:
        raise ValueError("temperature is empty")
    return temps


def text_compression_ratio(text):
    """テキストの簡易的なgzip圧縮率を計算する。

    入力テキストをUTF-8でエンコードし、gzipで圧縮します。
    非圧縮バイト長を圧縮バイト長で割ることで圧縮率を算出します。
    短いテキストや圧縮結果が空の場合は `0.0` を返します。

    :param text: str: 圧縮率を計算するテキスト。
    :returns: float: テキストの圧縮率。
    """
    raw = text.encode("utf-8", errors="ignore")
    if len(raw) < 80:
        return 0.0
    comp = gzip.compress(raw)
    if len(comp) == 0:
        return 0.0
    return len(raw) / len(comp)


def normalize_tokens(text):
    """反復検出のためにテキストをゆるくトークン化する。

    英数字の単語、およびひらがな・カタカナ・漢字の連続を単語として抽出し、小文字に変換してリストで返します。

    :param text: str: トークン化する入力テキスト。
    :returns: list[str]: 正規化されたトークンのリスト。
    """
    return re.findall(r"[A-Za-z0-9]+(?:'[A-Za-z0-9]+)?|[ぁ-んァ-ヶー一-龥]+", text.lower())


def looks_like_repetition_loop(text, max_token_repeat=12, compression_ratio_limit=3.2):
    """Whisperモデルによって生成された反復幻覚らしいセグメントを検出する。

    入力テキストの長さが80文字未満の場合は検出しない。
    `normalize_tokens` でトークン化した後、最も頻繁に出現するトークンの繰り返し回数とその割合をチェックする。
    また、`text_compression_ratio` を用いてテキストの圧縮率が高すぎる場合も反復と見なす。
    これらの条件に合致する場合に `True` とその理由を返します。

    判定の狙い:
      - "6th, 6th, 6th, ..." のような単語反復
      - 同じ語句が異常に多い高圧縮テキスト

    強すぎると正しい復唱も消すので、講義音声向けにやや控えめにしている。

    :param text: str: 検出対象のテキストセグメント。
    :param max_token_repeat: int: 単語反復とみなす最小繰り返し回数。(default: 12)
    :param compression_ratio_limit: float: 圧縮率がこの値以上の場合に反復とみなすしきい値。(default: 3.2)
    :returns: tuple[bool, str]: 反復幻覚らしい場合は (True, 理由), そうでない場合は (False, "")。
    """
    stripped = text.strip()
    if len(stripped) < 80:
        return False, ""

    tokens = normalize_tokens(stripped)
    if len(tokens) >= max_token_repeat:
        counts = Counter(tokens)
        token, count = counts.most_common(1)[0]
        if count >= max_token_repeat and count / len(tokens) >= 0.45:
            return True, f"token '{token}' repeated {count}/{len(tokens)} times"

    cr = text_compression_ratio(stripped)
    if cr >= compression_ratio_limit:
        # 圧縮率だけだと箇条書きや専門語の繰り返しも拾うので、長めの出力に限定する。
        if len(tokens) >= 30 or len(stripped) >= 300:
            return True, f"text compression ratio {cr:.2f} >= {compression_ratio_limit:.2f}"

    return False, ""


def bool_from_int(value):
    """整数値をブール値に変換する。

    整数 `0` を `False` に、それ以外の整数を `True` に変換します。

    :param value: int: 変換する整数値。
    :returns: bool: 変換されたブール値。
    """
    return bool(int(value))


# =========================
# OS 判定 & ライブラリ import
# =========================
os_name = platform.system()
print(f"Detected OS: {os_name}")

try:
    if os_name == "Windows":
        from faster_whisper import WhisperModel
        USE_FASTER = True
        print("Using faster-whisper (Windows)")
    else:
        import whisper
        USE_FASTER = False
        print("Using openai/whisper (Linux/macOS)")
except Exception:
    print("whisper / faster-whisper import failed")
    traceback.print_exc()
    terminate()

torch_mod = safe_import_torch()
#if torch_mod is None:
#    terminate()


# =========================
# 文字起こし本体
# =========================
def transcribe_audio_faster(
    infile,
    outfile1,
    outfile2,
    model_name,
    lang,
    device_name="",
    beam_size=5,
    chunk_length=30,
    temperature="0,0.2,0.4,0.6",
    condition_on_previous_text=False,
    compression_ratio_threshold=2.4,
    log_prob_threshold=-1.0,
    no_speech_threshold=0.6,
    repetition_penalty=1.05,
    no_repeat_ngram_size=3,
    max_bad_segments=3,
):
    """faster-whisperライブラリを使用して音声ファイルを文字起こしする。

    faster-whisperのモデルをロードし、指定されたパラメータで文字起こしを実行します。
    セグメントごとに進捗を表示し、指定されたファイルに出力します。
    長時間の音声処理において反復幻覚を抑制するための機能も含まれます。
    デバイス、計算タイプは自動判別されます。

    重要:
      - offset / duration は faster-whisper の transcribe() には渡さない。
      - 長時間音声では condition_on_previous_text=False をデフォルトにする。

    :param infile: str: 入力音声ファイルのパス。
    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
    :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
    :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
    :param device_name: str: 使用するデバイス名 ("cuda", "cpu", "dml")。空文字の場合は自動判別。(default: "")
    :param beam_size: int: ビームサーチのサイズ。(default: 5)
    :param chunk_length: int: 内部チャンクの長さ（秒）。(default: 30)
    :param temperature: Union[str, float, tuple[float, ...]]: temperatureパラメータ。文字列形式("0,0.2,0.4")または数値で指定。(default: "0,0.2,0.4,0.6")
    :param condition_on_previous_text: bool: 前の認識結果を次のチャンクの文脈として使用するかどうか。(default: False)
    :param compression_ratio_threshold: float: 反復テキスト検出用の圧縮率しきい値。(default: 2.4)
    :param log_prob_threshold: float: 低信頼デコード検出のしきい値。(default: -1.0)
    :param no_speech_threshold: float: 無音判定のしきい値。(default: 0.6)
    :param repetition_penalty: float: 反復抑制ペナルティ。(default: 1.05)
    :param no_repeat_ngram_size: int: n-gram反復抑制のサイズ。0で無効。(default: 3)
    :param max_bad_segments: int: 反復幻覚らしいセグメントが連続した場合に文字起こしを停止する回数。(default: 3)
    """
    if device_name == "":
        device_name = detect_device()

    compute_type = (
        "int8_float16" if device_name == "cuda"
        else "int8"
    )

    print(
        f"Using device [{device_name}] (faster-whisper, compute_type={compute_type})",
        flush=True,
    )

    model = WhisperModel(
        model_name,
        device=device_name,
        compute_type=compute_type,
    )

    transcribe_params = inspect.signature(model.transcribe).parameters

    # language="" を指定した場合は自動判定にする。
    language_arg = lang if str(lang).strip() else None

    requested_kwargs = {
        "language": language_arg,
        "vad_filter": True,
        "beam_size": beam_size,
        "temperature": parse_temperature(temperature),
        "condition_on_previous_text": condition_on_previous_text,
        "compression_ratio_threshold": compression_ratio_threshold,
        "log_prob_threshold": log_prob_threshold,
        "no_speech_threshold": no_speech_threshold,
        "repetition_penalty": repetition_penalty,
        "no_repeat_ngram_size": no_repeat_ngram_size,
        "suppress_blank": True,
    }

    if "chunk_length" in transcribe_params:
        requested_kwargs["chunk_length"] = chunk_length
    else:
        print(
            "chunk_length is not supported by this faster-whisper version; using default chunking.",
            flush=True,
        )

    # VAD パラメータはバージョン差があるので、対応している場合だけ渡す。
    if "vad_parameters" in transcribe_params:
        requested_kwargs["vad_parameters"] = {
            # 既定値より少し長い無音で分割し、短い講義中の間を無音扱いしすぎない。
            "min_silence_duration_ms": 500,
        }

    # word_timestamps と組み合わせると hallucination_silence_threshold が効く版もある。
    # ただし word_timestamps=True は重くなるので、ここでは使わない。

    # 古い faster-whisper でも動くよう、存在する引数だけ渡す。
    transcribe_kwargs = {
        k: v for k, v in requested_kwargs.items() if k in transcribe_params
    }

    print(f"transcribe options: {transcribe_kwargs}", flush=True)

    segments, info = model.transcribe(infile, **transcribe_kwargs)

    duration = getattr(info, "duration", None)
    if duration is None:
        duration = get_audio_duration(infile)

    if duration is not None:
        print(f"Audio length: {duration:.1f} sec ({format_hms(duration)})", flush=True)

    print("Start transcription...", flush=True)
    print(f"Save to [{outfile1}]")
    print(f"Save to [{outfile2}]", flush=True)

    seg_count = 0
    bad_count = 0
    all_text = []

    # ここで list(segments) にしてしまうと、全処理が終わるまで何も表示されない。
    # generator を1件ずつ消費して、そのたびに進捗を表示する。
    with open(outfile1, "w", encoding="utf-8") as f_time, \
         open(outfile2, "w", encoding="utf-8") as f_text:
        for seg in segments:
            seg_count += 1
            text = seg.text or ""

            is_bad, reason = looks_like_repetition_loop(text)
            if is_bad:
                bad_count += 1
                if duration and duration > 0:
                    pct = min(100.0, 100.0 * seg.end / duration)
                    progress = f"{pct:6.2f}%"
                else:
                    progress = "  ??.??%"

                preview = text.strip().replace("\n", " ")[:160]
                print(
                    f"[WARN] skipped probable repetition loop "
                    f"at {format_hms(seg.start)} - {format_hms(seg.end)} "
                    f"({progress}): {reason}\n"
                    f"       preview: {preview}",
                    flush=True,
                )

                f_time.write(
                    f"[{seg.start:.2f} - {seg.end:.2f}] "
                    f"[SKIPPED: probable repetition loop: {reason}]\n"
                )
                f_time.flush()

                if bad_count >= max_bad_segments:
                    print(
                        f"[WARN] {bad_count} suspicious segments were detected. "
                        "Stopping transcription to avoid a long hallucination tail.",
                        flush=True,
                    )
                    break
                continue

            bad_count = 0
            all_text.append(text)

            f_time.write(f"[{seg.start:.2f} - {seg.end:.2f}] {text}\n")
            f_time.flush()

            f_text.write(text)
            f_text.flush()

            if duration and duration > 0:
                pct = min(100.0, 100.0 * seg.end / duration)
                progress = f"{pct:6.2f}%"
            else:
                progress = "  ??.??%"

            print(
                f"[{seg_count:04d}] {progress} "
                f"{format_hms(seg.start)} - {format_hms(seg.end)} "
                f"{text}",
                flush=True,
            )

    text = "".join(all_text)
    print(f"\nSegments read: {seg_count}", flush=True)
    print("\n=== Transcribed text ===", flush=True)
    print(text, flush=True)


def transcribe_audio_whisper(
    infile,
    outfile1,
    outfile2,
    model_name,
    lang,
    device_name="",
):
    """openai/whisperライブラリを使用して音声ファイルを文字起こしする。

    openai/whisperモデルをロードし、指定されたパラメータで音声ファイルを一括で文字起こしします。
    結果は指定された2つのファイルに出力されます。

    :param infile: str: 入力音声ファイルのパス。
    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
    :param model_name: str: Whisperモデル名 (例: "base", "small", "medium")。
    :param lang: str: 文字起こし言語 (例: "ja", "en")。空文字の場合は自動判定。
    :param device_name: str: 使用するデバイス名 ("cuda", "cpu")。空文字の場合は自動判別。(default: "")
    """
    if device_name == "":
        model = whisper.load_model(model_name)
    else:
        model = whisper.load_model(model_name, device=device_name)

    print(f"Using device [{model.device}] (openai/whisper)")

    language_arg = lang if str(lang).strip() else None
    result = model.transcribe(
        infile,
        language=language_arg,
        verbose=True,
        condition_on_previous_text=False,
    )

    with open(outfile1, "w", encoding="utf-8") as f:
        for seg in result["segments"]:
            f.write(f"[{seg['start']:.2f} - {seg['end']:.2f}] {seg['text']}\n")

    with open(outfile2, "w", encoding="utf-8") as f:
        f.write(result["text"])

    print(result["text"])


def transcribe_audio(infile, outfile1, outfile2, args):
    """検出された環境に応じて適切なWhisper実装で音声ファイルを文字起こしする。

    グローバル変数 `USE_FASTER` の値に基づいて、`transcribe_audio_faster` または
    `transcribe_audio_whisper` のいずれかを呼び出します。
    コマンドライン引数を直接各関数に渡します。

    :param infile: str: 入力音声ファイルのパス。
    :param outfile1: str: 時間範囲付きの出力テキストファイルパス。
    :param outfile2: str: 時間範囲なしの出力テキストファイルパス。
    :param args: argparse.Namespace: コマンドライン引数を格納したオブジェクト。
    """
    if USE_FASTER:
        transcribe_audio_faster(
            infile,
            outfile1,
            outfile2,
            args.model,
            args.lang,
            args.device,
            beam_size=args.beam_size,
            chunk_length=args.chunk_length,
            temperature=args.temperature,
            condition_on_previous_text=bool_from_int(args.condition_on_previous_text),
            compression_ratio_threshold=args.compression_ratio_threshold,
            log_prob_threshold=args.log_prob_threshold,
            no_speech_threshold=args.no_speech_threshold,
            repetition_penalty=args.repetition_penalty,
            no_repeat_ngram_size=args.no_repeat_ngram_size,
            max_bad_segments=args.max_bad_segments,
        )
    else:
        transcribe_audio_whisper(
            infile,
            outfile1,
            outfile2,
            args.model,
            args.lang,
            args.device,
        )


# =========================
# メイン
# =========================
def main():
    """スクリプトのメインエントリポイント。コマンドライン引数を解析し、文字起こし処理を実行する。

    `argparse` を使用してコマンドライン引数を定義・解析します。
    入力ファイルパスに基づいて文字起こし処理をループ実行し、
    `check_gpu_torch` と `check_gpu` でGPU情報を表示した後、`transcribe_audio` 関数を呼び出します。
    処理結果は指定された出力ファイルに保存されます。
    """
    global pause

    parser = argparse.ArgumentParser(description="Whisper音声文字起こしツール")
    parser.add_argument("infile", type=str, help="入力音声ファイル名（glob可）")
    parser.add_argument("--outfile1", type=str, default="", help="出力テキストファイル名 (時間範囲入り)")
    parser.add_argument("--outfile2", type=str, default="", help="出力テキストファイル名 (時間範囲なし)")
    parser.add_argument("-m", "--model", type=str, default="base", help="Whisperモデル名 (default: base)")
    parser.add_argument("-d", "--device", type=str, default="", help="GPU/CPU/DMLデバイス名 (default: auto)")
    parser.add_argument("-l", "--lang", type=str, default="ja", help="使用言語。空文字なら自動判定 (default: ja)")
    parser.add_argument("--pause", type=int, default=0, help="終了時にENTERキー入力を要求するか (default: 0)")

    # faster-whisper の安定化パラメータ
    parser.add_argument("--beam-size", type=int, default=5, help="beam size (default: 5)")
    parser.add_argument("--chunk-length", type=int, default=30, help="内部チャンク長[秒] (default: 30)")
    parser.add_argument(
        "--temperature",
        type=str,
        default="0,0.2,0.4,0.6",
        help="temperature。'0' または '0,0.2,0.4,0.6' のように指定 (default: 0,0.2,0.4,0.6)",
    )
    parser.add_argument(
        "--condition-on-previous-text",
        type=int,
        default=0,
        help="前の認識結果を次チャンクの文脈に使うか。長時間音声では0推奨 (default: 0)",
    )
    parser.add_argument(
        "--compression-ratio-threshold",
        type=float,
        default=2.4,
        help="反復テキスト検出用しきい値。None にはしない (default: 2.4)",
    )
    parser.add_argument(
        "--log-prob-threshold",
        type=float,
        default=-1.0,
        help="低信頼デコード検出しきい値 (default: -1.0)",
    )
    parser.add_argument(
        "--no-speech-threshold",
        type=float,
        default=0.6,
        help="無音判定しきい値 (default: 0.6)",
    )
    parser.add_argument(
        "--repetition-penalty",
        type=float,
        default=1.05,
        help="反復抑制ペナルティ。対応版のみ有効 (default: 1.05)",
    )
    parser.add_argument(
        "--no-repeat-ngram-size",
        type=int,
        default=3,
        help="n-gram反復抑制。対応版のみ有効。0で無効 (default: 3)",
    )
    parser.add_argument(
        "--max-bad-segments",
        type=int,
        default=3,
        help="反復幻覚らしいセグメントが連続したら停止する数 (default: 3)",
    )

    args = parser.parse_args()
    pause = args.pause

    files = glob.glob(args.infile)
    print()
    print(f"infile={args.infile}")
    if len(files) == 0:
        print("\nError: No file found.\n")
        return

    print(f"  files: {files}")
    print(f"model={args.model}")
    print(f"lang={args.lang!r}")
    print(f"device={args.device}")

    print()
    check_gpu_torch(torch_mod)
    check_gpu()

    for f in files:
        if args.outfile1 == "":
            outfile1 = os.path.splitext(os.path.basename(f))[0] + "-time.txt"
        else:
            outfile1 = args.outfile1

        if args.outfile2 == "":
            outfile2 = os.path.splitext(os.path.basename(f))[0] + ".txt"
        else:
            outfile2 = args.outfile2

        print("\n=== Transcribing ===")
        print(f"infile  = {f}")
        print(f"outfile1= {outfile1}")
        print(f"outfile2= {outfile2}")

        transcribe_audio(f, outfile1, outfile2, args)


if __name__ == "__main__":
    main()
    terminate()