"""
Qwen3-ASRコマンドライン音声文字起こしツール。

このスクリプトは、Qwen3-ASRモデルを使用して音声ファイルを文字起こしするための
コマンドラインインターフェースを提供します。
入力ファイルごとに以下の3つの出力ファイルを生成します。

  1. ``<name>-time.txt`` : タイムスタンプ付きのテキスト（ForcedAlignerが必要な場合）
  2. ``<name>.txt``      : プレーンな文字起こしテキスト
  3. ``<name>-info.txt`` : 実行時間、モデル、オーディオ、および結果のメタデータ

ASRエンジンは ``ASRBackend`` クラスの背後に隔離されており、
CLI、出力ライター、環境レポートを変更することなく、
faster-whisper、openai-whisper、または他のエンジンを追加できるよう設計されています。

.. seealso::
   :doc:`transcribe_qwen3_asr_usage` （利用ガイドへの架空のリンク）
"""

from __future__ import annotations

import argparse
import glob
import importlib.metadata
import json
import os
import platform
import subprocess
import sys
import tempfile
import time
import traceback
from abc import ABC, abstractmethod
from dataclasses import asdict, dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Callable, Optional


MODEL_ALIASES = {
    "0.6B": "Qwen/Qwen3-ASR-0.6B",
    "1.7B": "Qwen/Qwen3-ASR-1.7B",
    "Qwen/Qwen3-ASR-0.6B": "Qwen/Qwen3-ASR-0.6B",
    "Qwen/Qwen3-ASR-1.7B": "Qwen/Qwen3-ASR-1.7B",
}

LANGUAGE_ALIASES = {
    "ja": "Japanese",
    "japanese": "Japanese",
    "en": "English",
    "english": "English",
    "zh": "Chinese",
    "chinese": "Chinese",
    "ko": "Korean",
    "korean": "Korean",
}


@dataclass
class Segment:
    """文字起こしされたテキストのセグメントを表すデータクラス。

    :param start: セグメントの開始時間（秒）。
    :type start: Optional[float]
    :param end: セグメントの終了時間（秒）。
    :type end: Optional[float]
    :param text: セグメント内のテキスト。
    :type text: str
    """
    start: Optional[float]
    end: Optional[float]
    text: str


@dataclass
class ASRResult:
    """ASRエンジンの文字起こし結果を表すデータクラス。

    :param text: 文字起こしされた全テキスト。
    :type text: str
    :param language: 検出された言語コードまたは名称。指定がない場合はNone。
    :type language: Optional[str]
    :param segments: タイムスタンプ付きのテキストセグメントのリスト。デフォルトは空リスト。
    :type segments: list[Segment]
    :param engine_metadata: エンジン固有の追加メタデータ。デフォルトは空の辞書。
    :type engine_metadata: dict[str, Any]
    """
    text: str
    language: Optional[str] = None
    segments: list[Segment] = field(default_factory=list)
    engine_metadata: dict[str, Any] = field(default_factory=dict)


@dataclass
class OutputPaths:
    """出力ファイルのパスを保持するデータクラス。

    :param timestamped: タイムスタンプ付きテキストの出力パス。
    :type timestamped: Path
    :param plain: プレーンテキストの出力パス。
    :type plain: Path
    :param info: 情報JSONファイルの出力パス。
    :type info: Path
    """
    timestamped: Path
    plain: Path
    info: Path


class ASRBackend(ABC):
    """ローカルASRエンジン用の共通インターフェース。

    この抽象基底クラスは、さまざまなASRバックエンド（例: Qwen3-ASR、Whisperなど）が
    共通のインターフェースを通じて使用できるようにするための基本的なメソッドを定義します。
    新しいASRエンジンを追加する際は、このクラスを継承し、抽象メソッドを実装する必要があります。
    """

    backend_name = "unknown"

    @abstractmethod
    def load(self) -> None:
        """1つ以上のファイルを処理する前に、一度モデルをロードします。

        このメソッドは、モデルの初期化やリソースの準備を行います。
        """

    @abstractmethod
    def transcribe(self, audio_path: str, language: Optional[str]) -> ASRResult:
        """1つの音声ファイルを文字起こしし、エンジンに依存しない結果を返します。

        :param audio_path: 文字起こしする音声ファイルへのパス。
        :type audio_path: str
        :param language: 音声の言語。指定がない場合は自動検出されます。
        :type language: Optional[str]
        :returns: 文字起こし結果を含む :class:`ASRResult` オブジェクト。
        :rtype: ASRResult
        """

    @abstractmethod
    def describe(self) -> dict[str, Any]:
        """このバックエンドで使用されている設定を辞書形式で返します。

        :returns: バックエンドの設定を含む辞書。
        :rtype: dict[str, Any]
        """


class Qwen3ASRBackend(ASRBackend):
    """Qwen3-ASRモデルを使用するASRバックエンドの実装。

    このクラスは、Qwen3-ASRライブラリをラップし、
    ``ASRBackend`` インターフェースに準拠した文字起こし機能を提供します。

    :param model_name: 使用するQwen3-ASRモデルの名前またはエイリアス。
    :type model_name: str
    :param device: モデルを実行するデバイス（例: "cpu", "cuda", "cuda:0"）。
    :type device: str
    :param dtype_name: モデルのデータ型（例: "float32", "float16", "bfloat16"）。
    :type dtype_name: str
    :param timestamps: タイムスタンプ付きセグメントを生成するかどうか（True/False）。
    :type timestamps: bool
    :param aligner_name: タイムスタンプ生成に使用するアライナーモデルの名前。
    :type aligner_name: str
    :param max_inference_batch_size: 推論時の最大バッチサイズ。
    :type max_inference_batch_size: int
    :param max_new_tokens: 生成される新しいトークンの最大数。
    :type max_new_tokens: int
    :param attention: Attentionの実装タイプ（例: "auto", "sdpa", "eager", "flash_attention_2"）。
    :type attention: str
    """
    backend_name = "qwen3-asr"

    def __init__(
        self,
        model_name: str,
        device: str,
        dtype_name: str,
        timestamps: bool,
        aligner_name: str,
        max_inference_batch_size: int,
        max_new_tokens: int,
        attention: str,
    ) -> None:
        """Qwen3ASRBackendを初期化します。

        :param model_name: 使用するQwen3-ASRモデルの名前またはエイリアス。
        :type model_name: str
        :param device: モデルを実行するデバイス（例: "cpu", "cuda", "cuda:0"）。
        :type device: str
        :param dtype_name: モデルのデータ型（例: "float32", "float16", "bfloat16"）。
        :type dtype_name: str
        :param timestamps: タイムスタンプ付きセグメントを生成するかどうか（True/False）。
        :type timestamps: bool
        :param aligner_name: タイムスタンプ生成に使用するアライナーモデルの名前。
        :type aligner_name: str
        :param max_inference_batch_size: 推論時の最大バッチサイズ。
        :type max_inference_batch_size: int
    :param max_new_tokens: 生成される新しいトークンの最大数。
        :type max_new_tokens: int
        :param attention: Attentionの実装タイプ（例: "auto", "sdpa", "eager", "flash_attention_2"）。
        :type attention: str
        """
        self.model_name = MODEL_ALIASES.get(model_name, model_name)
        self.requested_device = device
        self.requested_dtype = dtype_name
        self.timestamps = timestamps
        self.aligner_name = aligner_name
        self.max_inference_batch_size = max_inference_batch_size
        self.max_new_tokens = max_new_tokens
        self.attention = attention
        self.model = None
        self.torch = None
        self.device_map = ""
        self.dtype = None
        self.dtype_label = ""

    def _resolve_device_and_dtype(self) -> None:
        """要求されたデバイスとデータ型をPyTorchの実際の値に解決します。

        このメソッドは、指定された ``device`` と ``dtype_name`` の文字列値を
        PyTorchが認識するデバイス文字列とデータ型オブジェクトに変換します。
        GPUの利用可能性やbfloat16のサポート状況に応じて最適な値を自動選択します。

        :raises RuntimeError: CUDAが要求されたにもかかわらず利用できない場合。
        :raises ValueError: サポートされていないdtypeが指定された場合、またはCPU上でfloat16が指定された場合。
        """
        import torch

        self.torch = torch
        device = self.requested_device.strip().lower()
        if not device or device == "auto":
            device = "cuda:0" if torch.cuda.is_available() else "cpu"
        elif device == "cuda":
            device = "cuda:0"

        if device.startswith("cuda") and not torch.cuda.is_available():
            raise RuntimeError("CUDAが要求されましたが、torch.cuda.is_available()がFalseです")

        dtype_name = self.requested_dtype.strip().lower()
        if dtype_name == "auto":
            if device.startswith("cuda"):
                dtype_name = "bfloat16" if torch.cuda.is_bf16_supported() else "float16"
            else:
                dtype_name = "float32"

        dtype_table = {
            "bfloat16": torch.bfloat16,
            "bf16": torch.bfloat16,
            "float16": torch.float16,
            "fp16": torch.float16,
            "float32": torch.float32,
            "fp32": torch.float32,
        }
        if dtype_name not in dtype_table:
            raise ValueError(f"サポートされていないdtype: {self.requested_dtype}")
        if device == "cpu" and dtype_name in {"float16", "fp16"}:
            raise ValueError("CPU上でのfloat16は推奨されません。--dtype float32を使用してください")

        self.device_map = device
        self.dtype = dtype_table[dtype_name]
        self.dtype_label = str(self.dtype).replace("torch.", "")

    def load(self) -> None:
        """Qwen3-ASRモデルと必要に応じてForcedAlignerをロードします。

        デバイスとデータ型を解決した後、指定された設定に基づいて
        Qwen3ASRModelをHugging Face Hubからロードします。
        タイムスタンプが有効な場合は、ForcedAlignerも初期化されます。
        """
        self._resolve_device_and_dtype()
        from qwen_asr import Qwen3ASRModel

        kwargs: dict[str, Any] = {
            "dtype": self.dtype,
            "device_map": self.device_map,
            "max_inference_batch_size": self.max_inference_batch_size,
            "max_new_tokens": self.max_new_tokens,
        }
        if self.attention != "auto":
            kwargs["attn_implementation"] = self.attention

        if self.timestamps:
            kwargs["forced_aligner"] = self.aligner_name
            aligner_kwargs: dict[str, Any] = {
                "dtype": self.dtype,
                "device_map": self.device_map,
            }
            if self.attention != "auto":
                aligner_kwargs["attn_implementation"] = self.attention
            kwargs["forced_aligner_kwargs"] = aligner_kwargs

        print(f"Loading ASR model : {self.model_name}", flush=True)
        if self.timestamps:
            print(f"Loading aligner   : {self.aligner_name}", flush=True)
        print(f"Device / dtype   : {self.device_map} / {self.dtype_label}", flush=True)
        self.model = Qwen3ASRModel.from_pretrained(self.model_name, **kwargs)

    @staticmethod
    def _timestamp_item_to_segment(item: Any) -> Segment:
        """タイムスタンプ付きのアイテムを :class:`Segment` オブジェクトに変換します。

        Qwen3-ASRモデルが返すタイムスタンプ形式から、
        一貫した :class:`Segment` データクラスに変換します。

        :param item: タイムスタンプ情報を含む辞書またはオブジェクト。
        :type item: Any
        :returns: 変換された :class:`Segment` オブジェクト。
        :rtype: Segment
        """
        if isinstance(item, dict):
            text = str(item.get("text", ""))
            start = item.get("start_time", item.get("start"))
            end = item.get("end_time", item.get("end"))
        else:
            text = str(getattr(item, "text", ""))
            start = getattr(item, "start_time", getattr(item, "start", None))
            end = getattr(item, "end_time", getattr(item, "end", None))
        return Segment(
            start=float(start) if start is not None else None,
            end=float(end) if end is not None else None,
            text=text,
        )

    @classmethod
    def _convert_timestamps(cls, raw: Any) -> list[Segment]:
        """Qwen3-ASRの生のタイムスタンプ出力をSegmentのリストに変換します。

        モデルの出力が複数のリストレベルを持つ場合があるため、それを平坦化し、
        各タイムスタンプアイテムを :class:`Segment` オブジェクトに変換します。

        :param raw: Qwen3-ASRから返された生のタイムスタンプデータ。
        :type raw: Any
        :returns: :class:`Segment` オブジェクトのリスト。
        :rtype: list[Segment]
        """
        if raw is None:
            return []
        # 一部のバージョンでは、単一の入力に対して余分なリストレベルを返す場合があります。
        while isinstance(raw, (list, tuple)) and len(raw) == 1 and isinstance(raw[0], (list, tuple)):
            raw = raw[0]
        if not isinstance(raw, (list, tuple)):
            raw = [raw]
        return [cls._timestamp_item_to_segment(item) for item in raw]

    def transcribe(self, audio_path: str, language: Optional[str]) -> ASRResult:
        """Qwen3-ASRモデルを使用して音声ファイルを文字起こしします。

        :param audio_path: 文字起こしする音声ファイルへのパス。
        :type audio_path: str
        :param language: 音声の言語。指定がない場合は自動検出されます。
        :type language: Optional[str]
        :returns: 文字起こし結果を含む :class:`ASRResult` オブジェクト。
        :rtype: ASRResult
        :raises RuntimeError: モデルがロードされていない場合、またはQwen3-ASRが結果を返さなかった場合。
        """
        if self.model is None:
            raise RuntimeError("モデルがロードされていません")
        results = self.model.transcribe(
            audio=audio_path,
            language=language,
            return_time_stamps=self.timestamps,
        )
        if not results:
            raise RuntimeError("Qwen3-ASRが結果を返しませんでした")
        raw = results[0]
        text = str(getattr(raw, "text", ""))
        detected_language = getattr(raw, "language", None)
        raw_timestamps = getattr(raw, "time_stamps", None)
        segments = self._convert_timestamps(raw_timestamps)
        return ASRResult(
            text=text,
            language=str(detected_language) if detected_language is not None else None,
            segments=segments,
            engine_metadata={"timestamp_items": len(segments)},
        )

    def describe(self) -> dict[str, Any]:
        """このバックエンドで使用されている設定を辞書形式で返します。

        :returns: バックエンドの設定を含む辞書。
        :rtype: dict[str, Any]
        """
        return {
            "backend": self.backend_name,
            "model": self.model_name,
            "device": self.device_map,
            "dtype": self.dtype_label,
            "timestamps": self.timestamps,
            "aligner": self.aligner_name if self.timestamps else None,
            "attention": self.attention,
            "max_inference_batch_size": self.max_inference_batch_size,
            "max_new_tokens": self.max_new_tokens,
        }


def normalize_language(value: str) -> Optional[str]:
    """指定された言語の文字列を標準形式に正規化します。

    "auto"や空文字列はNoneに変換され、一般的なエイリアスは正式な名称に変換されます。

    :param value: 正規化する言語文字列。
    :type value: str
    :returns: 正規化された言語文字列、または自動検出を意味するNone。
    :rtype: Optional[str]
    """
    value = value.strip()
    if not value or value.lower() == "auto":
        return None
    return LANGUAGE_ALIASES.get(value.lower(), value)


def package_version(name: str) -> Optional[str]:
    """指定されたパッケージのバージョンを取得します。

    :param name: バージョンを取得するパッケージの名前。
    :type name: str
    :returns: パッケージのバージョン文字列、またはパッケージが見つからない場合はNone。
    :rtype: Optional[str]
    """
    try:
        return importlib.metadata.version(name)
    except importlib.metadata.PackageNotFoundError:
        return None


def command_output(command: list[str]) -> Optional[str]:
    """指定されたコマンドを実行し、その標準出力を取得します。

    コマンドの実行に失敗した場合や標準出力が空の場合はNoneを返します。

    :param command: 実行するコマンドとその引数のリスト。
    :type command: list[str]
    :returns: コマンドの標準出力文字列、またはエラー/空出力の場合はNone。
    :rtype: Optional[str]
    """
    try:
        completed = subprocess.run(
            command,
            stdout=subprocess.PIPE,
            stderr=subprocess.PIPE,
            text=True,
            check=False,
        )
        value = completed.stdout.strip()
        return value or None
    except (OSError, ValueError):
        return None


def get_audio_duration(path: str) -> Optional[float]:
    """指定された音声ファイルの長さを秒単位で取得します。

    ffprobeコマンドを使用して音声ファイルのduration情報を解析します。

    :param path: 音声ファイルへのパス。
    :type path: str
    :returns: 音声ファイルの長さ（秒）、または取得できない場合はNone。
    :rtype: Optional[float]
    """
    value = command_output([
        "ffprobe", "-v", "error", "-show_entries", "format=duration",
        "-of", "default=noprint_wrappers=1:nokey=1", path,
    ])
    try:
        return float(value) if value is not None else None
    except ValueError:
        return None


def format_hms(seconds: Optional[float]) -> str:
    """秒数をHH:MM:SS.ss形式の文字列にフォーマットします。

    :param seconds: フォーマットする秒数。Noneの場合は"??:??:??.??"を返します。
    :type seconds: Optional[float]
    :returns: フォーマットされた時間文字列。
    :rtype: str
    """
    if seconds === None:
        return "??:??:??.??"
    seconds = max(0.0, float(seconds))
    hours = int(seconds // 3600)
    minutes = int((seconds % 3600) // 60)
    secs = seconds % 60
    return f"{hours:02d}:{minutes:02d}:{secs:05.2f}"


def collect_environment(torch_module: Any) -> dict[str, Any]:
    """現在の実行環境に関する情報を収集します。

    OS、Python、インストールされている主要なパッケージのバージョン、
    FFmpegのバージョン、CUDAの利用可能性とGPUの詳細などを取得します。

    :param torch_module: PyTorchモジュールへの参照。
    :type torch_module: Any
    :returns: 環境情報を格納した辞書。
    :rtype: dict[str, Any]
    """
    info: dict[str, Any] = {
        "timestamp_utc": datetime.now(timezone.utc).isoformat(),
        "platform": platform.platform(),
        "os": platform.system(),
        "os_release": platform.release(),
        "machine": platform.machine(),
        "processor": platform.processor(),
        "python": platform.python_version(),
        "python_executable": sys.executable,
        "packages": {
            name: package_version(name)
            for name in (
                "qwen-asr", "torch", "transformers", "accelerate",
                "huggingface-hub", "soundfile", "librosa",
            )
        },
        "ffmpeg": command_output(["ffmpeg", "-version"]),
    }
    if info["ffmpeg"]:
        info["ffmpeg"] = str(info["ffmpeg"]).splitlines()[0]

    cuda_available = bool(torch_module.cuda.is_available())
    cuda: dict[str, Any] = {
        "available": cuda_available,
        "torch_cuda_version": torch_module.version.cuda,
        "cudnn_version": torch_module.backends.cudnn.version() if cuda_available else None,
        "device_count": torch_module.cuda.device_count() if cuda_available else 0,
        "devices": [],
    }
    if cuda_available:
        for index in range(torch_module.cuda.device_count()):
            prop = torch_module.cuda.get_device_properties(index)
            cuda["devices"].append({
                "index": index,
                "name": prop.name,
                "total_memory_gib": round(prop.total_memory / 1024**3, 3),
                "compute_capability": f"{prop.major}.{prop.minor}",
            })
    info["cuda"] = cuda
    return info


def print_environment(info: dict[str, Any]) -> None:
    """収集した環境情報をコンソールに表示します。

    :param info: :func:`collect_environment` によって収集された環境情報の辞書。
    :type info: dict[str, Any]
    """
    print("\n=== ランタイム環境 ===")
    print(f"OS              : {info['platform']}")
    print(f"Python          : {info['python']}")
    print(f"実行可能パス      : {info['python_executable']}")
    print(f"qwen-asr        : {info['packages']['qwen-asr']}")
    print(f"PyTorch         : {info['packages']['torch']}")
    print(f"Transformers    : {info['packages']['transformers']}")
    print(f"CUDA利用可能     : {info['cuda']['available']}")
    print(f"Torch CUDA      : {info['cuda']['torch_cuda_version']}")
    for gpu in info["cuda"]["devices"]:
        print(
            f"GPU {gpu['index']}           : {gpu['name']} "
            f"({gpu['total_memory_gib']:.2f} GiB, CC {gpu['compute_capability']})"
        )


def make_output_paths(input_path: str, args: argparse.Namespace) -> OutputPaths:
    """入力ファイルに基づいて出力ファイルのパスを生成します。

    CLI引数で特定の出力ファイル名が指定されている場合はそれを使用し、
    そうでない場合は入力ファイルのステム名に基づいてデフォルトのパスを作成します。

    :param input_path: 入力音声ファイルへのパス。
    :type input_path: str
    :param args: コマンドライン引数をパースしたオブジェクト。
    :type args: argparse.Namespace
    :returns: 生成された出力パスを含む :class:`OutputPaths` オブジェクト。
    :rtype: OutputPaths
    """
    stem = Path(input_path).stem
    return OutputPaths(
        timestamped=Path(args.outfile1) if args.outfile1 else Path(f"{stem}-time.txt"),
        plain=Path(args.outfile2) if args.outfile2 else Path(f"{stem}.txt"),
        info=Path(args.outfile3) if args.outfile3 else Path(f"{stem}-info.txt"),
    )


def write_transcript_outputs(
    paths: OutputPaths,
    result: ASRResult,
) -> None:
    """文字起こしされたテキストファイルを書き込みます。

    この関数は、タイムスタンプ付きのテキストファイルとプレーンテキストファイルの両方を生成します。
    長時間の処理が中断された場合でも、中間結果がディスクに残るように、
    チャンク処理の各完了後にも呼び出されます。

    :param paths: 出力ファイルのパスを含む :class:`OutputPaths` オブジェクト。
    :type paths: OutputPaths
    :param result: 文字起こし結果を含む :class:`ASRResult` オブジェクト。
    :type result: ASRResult
    """
    with paths.timestamped.open("w", encoding="utf-8") as handle:
        if result.segments:
            for seg in result.segments:
                if seg.start is None and seg.end is None:
                    handle.write(f"{seg.text}\n")
                else:
                    handle.write(
                        f"[{format_hms(seg.start)} - {format_hms(seg.end)}] {seg.text}\n"
                    )
        else:
            handle.write("[タイムスタンプ利用不可]\n")
            handle.write(result.text)
            if result.text and not result.text.endswith("\n"):
                handle.write("\n")

    with paths.plain.open("w", encoding="utf-8") as handle:
        handle.write(result.text)
        if result.text and not result.text.endswith("\n"):
            handle.write("\n")


def write_outputs(
    paths: OutputPaths,
    result: ASRResult,
    report: dict[str, Any],
) -> None:
    """すべての出力ファイル（文字起こしテキストと情報JSON）を書き込みます。

    :param paths: 出力ファイルのパスを含む :class:`OutputPaths` オブジェクト。
    :type paths: OutputPaths
    :param result: 文字起こし結果を含む :class:`ASRResult` オブジェクト。
    :type result: ASRResult
    :param report: 実行レポートデータを含む辞書。
    :type report: dict[str, Any]
    """
    write_transcript_outputs(paths, result)

    with paths.info.open("w", encoding="utf-8") as handle:
        json.dump(report, handle, ensure_ascii=False, indent=2, default=str)
        handle.write("\n")


def transcribe_with_optional_chunks(
    backend: ASRBackend,
    audio_path: str,
    language: Optional[str],
    duration: Optional[float],
    chunk_seconds: float,
    chunk_overlap: float,
    use_chunks: bool,
    progress_callback: Optional[Callable[[ASRResult, dict[str, Any]], None]] = None,
) -> ASRResult:
    """アライナーの入力長に制限がある場合に、長い音声を分割して文字起こしします。

    この関数は、指定された ``chunk_seconds`` に基づいて音声を小さなチャンクに分割し、
    各チャンクをASRバックエンドで文字起こしします。
    チャンク処理はバックエンドの外で行われるため、さまざまなASRエンジンで再利用可能です。
    タイムスタンプは元のファイルのタイムラインにシフトバックされます。
    チャンクの境界には、後の処理（例: LLMによるテキスト結合）で役立つマーカーが挿入されます。

    :param backend: 使用するASRバックエンドの実装。
    :type backend: ASRBackend
    :param audio_path: 文字起こしする音声ファイルへのパス。
    :type audio_path: str
    :param language: 音声の言語。指定がない場合は自動検出されます。
    :type language: Optional[str]
    :param duration: 音声ファイルの全体の長さ（秒）。
    :type duration: Optional[float]
    :param chunk_seconds: 音声を分割するチャンクの秒数。0以下の場合、分割は行われません。
    :type chunk_seconds: float
    :param chunk_overlap: 各チャンクの前後で重複させる秒数。重複により境界での情報損失を防ぎます。
    :type chunk_overlap: float
    :param use_chunks: チャンク分割を使用するかどうか。Falseの場合、全音声を一度に処理します。
    :type use_chunks: bool
    :param progress_callback: チャンクが完了するたびに呼び出されるコールバック関数。
                              部分的なASRResultと進捗情報を引数に取ります。
    :type progress_callback: Optional[Callable[[ASRResult, dict[str, Any]], None]]
    :returns: 全てのチャンクを結合した最終的な文字起こし結果。
    :rtype: ASRResult
    :raises RuntimeError: FFmpegがインストールされていない場合、またはFFmpegによるチャンク作成に失敗した場合。
    :raises ValueError: ``--chunk-overlap`` が無効な値の場合。
    """
    if not use_chunks or chunk_seconds <= 0 or duration is None or duration <= chunk_seconds:
        return backend.transcribe(audio_path, language)

    if command_output(["ffmpeg", "-version"]) is None:
        raise RuntimeError("タイムスタンプアライメントのために長い音声を分割するにはffmpegが必要です")

    if chunk_overlap < 0:
        raise ValueError("--chunk-overlapは0以上である必要があります")
    if chunk_overlap * 2 >= chunk_seconds:
        raise ValueError("--chunk-overlapは--chunk-secondsの半分未満である必要があります")

    parts: list[str] = []
    segments: list[Segment] = []
    detected_languages: list[str] = []
    core_start = 0.0
    chunk_index = 0

    with tempfile.TemporaryDirectory(prefix="qwen3_asr_") as temp_dir:
        while core_start < duration:
            chunk_index += 1
            core_end = min(duration, core_start + chunk_seconds)
            actual_start = max(0.0, core_start - (chunk_overlap if chunk_index > 1 else 0.0))
            actual_end = min(
                duration,
                core_end + (chunk_overlap if core_end < duration else 0.0),
            )
            length = actual_end - actual_start
            chunk_path = os.path.join(temp_dir, f"chunk_{chunk_index:04d}.wav")
            command = [
                "ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
                "-ss", str(actual_start), "-t", str(length), "-i", audio_path,
                "-vn", "-ac", "1", "-ar", "16000", chunk_path,
            ]
            completed = subprocess.run(command, check=False)
            if completed.returncode != 0:
                raise RuntimeError(f"チャンク {chunk_index} の作成中にffmpegが失敗しました")

            print(
                f"チャンク {chunk_index}: 実際の時間 {format_hms(actual_start)} - "
                f"{format_hms(actual_end)}、コア時間 {format_hms(core_start)} - "
                f"{format_hms(core_end)}",
                flush=True,
            )
            chunk_started = time.perf_counter()
            chunk_result = backend.transcribe(chunk_path, language)
            chunk_elapsed = time.perf_counter() - chunk_started
            marker = (
                f"[[ASR_CHUNK {chunk_index:04d} "
                f"ACTUAL={format_hms(actual_start)}..{format_hms(actual_end)} "
                f"CORE={format_hms(core_start)}..{format_hms(core_end)} "
                f"OVERLAP={chunk_overlap:.1f}s]]"
            )
            if chunk_index > 1:
                boundary = (
                    f"[[ASR_BOUNDARY {chunk_index - 1:04d}|{chunk_index:04d} "
                    f"OVERLAP={chunk_overlap:.1f}s; "
                    "LLM: この境界をまたがる重複または不完全なテキストを調整してください]]"
                )
                parts.append(boundary)
                segments.append(Segment(None, None, boundary))
            parts.append(marker)
            segments.append(Segment(None, None, marker))
            if chunk_result.text.strip():
                parts.append(chunk_result.text.strip())
            parts.append(f"[[/ASR_CHUNK {chunk_index:04d}]]")
            if chunk_result.language:
                detected_languages.append(chunk_result.language)
            for seg in chunk_result.segments:
                segments.append(Segment(
                    start=(seg.start + actual_start) if seg.start is not None else None,
                    end=(seg.end + actual_start) if seg.end is not None else None,
                    text=seg.text,
                ))
            segments.append(Segment(None, None, f"[[/ASR_CHUNK {chunk_index:04d}]]"))
            core_start = core_end

            progress_percent = 100.0 * core_end / duration
            print(
                f"\n=== チャンク {chunk_index} 完了: "
                f"{progress_percent:.1f}% ({format_hms(core_end)} / {format_hms(duration)}) ==="
            )
            print(f"チャンク処理時間: {chunk_elapsed:.2f} 秒")
            print(chunk_result.text.strip() or "[テキストなし]")
            print("=== チャンク結果の終わり ===\n", flush=True)

            if progress_callback is not None:
                partial_result = ASRResult(
                    text="\n".join(parts),
                    language=detected_languages[0] if detected_languages else None,
                    segments=list(segments),
                    engine_metadata={
                        "timestamp_items": len(segments),
                        "audio_chunks_completed": chunk_index,
                        "processed_seconds": core_end,
                        "progress_percent": progress_percent,
                        "chunk_seconds": chunk_seconds,
                        "chunk_overlap_seconds": chunk_overlap,
                        "boundary_markers": True,
                    },
                )
                progress_callback(
                    partial_result,
                    {
                        "chunk_index": chunk_index,
                        "processed_seconds": core_end,
                        "duration_seconds": duration,
                        "progress_percent": progress_percent,
                        "chunk_elapsed_seconds": chunk_elapsed,
                    },
                )

    detected_language = detected_languages[0] if detected_languages else None
    return ASRResult(
        text="\n".join(parts),
        language=detected_language,
        segments=segments,
        engine_metadata={
            "timestamp_items": len(segments),
            "audio_chunks": chunk_index,
            "chunk_seconds": chunk_seconds,
            "chunk_overlap_seconds": chunk_overlap,
            "boundary_markers": True,
        },
    )


def create_backend(args: argparse.Namespace) -> ASRBackend:
    """コマンドライン引数に基づいてASRバックエンドインスタンスを作成します。

    :param args: コマンドライン引数をパースしたオブジェクト。
    :type args: argparse.Namespace
    :returns: 初期化されたASRバックエンドのインスタンス。
    :rtype: ASRBackend
    """
    factories = {
        "qwen3-asr": lambda: Qwen3ASRBackend(
            model_name=args.model,
            device=args.device,
            dtype_name=args.dtype,
            timestamps=bool(args.timestamps),
            aligner_name=args.aligner,
            max_inference_batch_size=args.max_inference_batch_size,
            max_new_tokens=args.max_new_tokens,
            attention=args.attention,
        )
    }
    return factories[args.backend]()


def build_parser() -> argparse.ArgumentParser:
    """コマンドライン引数をパースするためのArgumentParserを作成します。

    利用可能なASRバックエンド、モデル、デバイス、データ型、チャンク設定など、
    様々なオプションを定義します。

    :returns: 設定済みの :class:`argparse.ArgumentParser` オブジェクト。
    :rtype: argparse.ArgumentParser
    """
    parser = argparse.ArgumentParser(
        description="Qwen3-ASR音声文字起こしツール（ASRバックエンド拡張対応）"
    )
    parser.add_argument("infile", help="入力音声ファイル名（glob可）")
    parser.add_argument(
        "--backend", choices=["qwen3-asr"], default="qwen3-asr",
        help="ASRバックエンド (default: qwen3-asr)",
    )
    parser.add_argument(
        "-m", "--model", default="0.6B",
        choices=list(MODEL_ALIASES),
        help="Qwen3-ASRモデル (default: 0.6B)",
    )
    parser.add_argument(
        "-l", "--lang", default="ja",
        help="言語コード/名称。autoまたは空文字で自動判定 (default: ja)",
    )
    parser.add_argument(
        "-d", "--device", default="auto",
        help="auto, cpu, cuda, cuda:0など (default: auto)",
    )
    parser.add_argument(
        "--dtype", choices=["auto", "bfloat16", "float16", "float32"],
        default="auto", help="推論精度 (default: auto)",
    )
    parser.add_argument(
        "--timestamps", type=int, choices=[0, 1], default=1,
        help="ForcedAlignerで時刻を付けるか (default: 1)",
    )
    parser.add_argument(
        "--aligner", default="Qwen/Qwen3-ForcedAligner-0.6B",
        help="時刻推定モデル",
    )
    parser.add_argument(
        "--attention", choices=["auto", "sdpa", "eager", "flash_attention_2"],
        default="auto", help="Attention実装 (default: auto)",
    )
    parser.add_argument(
        "--max-inference-batch-size", type=int, default=1,
        help="推論バッチ上限。小さい値はVRAMを節約 (default: 1)",
    )
    parser.add_argument(
        "--max-new-tokens", type=int, default=4096,
        help="生成トークン上限。長時間音声では大きくする (default: 4096)",
    )
    parser.add_argument(
        "--chunk-seconds", type=float, default=240.0,
        help="時刻付け時の長時間音声分割秒数。0で分割しない (default: 240)",
    )
    parser.add_argument(
        "--chunk-overlap", type=float, default=10.0,
        help="チャンク前後の重複秒数。境界は出力に明示 (default: 10)",
    )
    parser.add_argument("--outfile1", default="", help="時刻付き出力ファイル")
    parser.add_argument("--outfile2", default="", help="本文出力ファイル")
    parser.add_argument("--outfile3", default="", help="環境・設定情報JSONファイル")
    parser.add_argument(
        "--pause", type=int, choices=[0, 1], default=0,
        help="終了時にENTERを待つか (default: 0)",
    )
    return parser


def main() -> int:
    """Qwen3-ASR音声文字起こしツールのメインエントリポイント。

    コマンドライン引数をパースし、指定された音声ファイルを文字起こしします。
    環境情報を収集・表示し、ASRモデルをロードして文字起こしを実行し、
    結果を複数のファイルに出力します。
    複数の入力ファイルが指定された場合、または長い音声ファイルがチャンク分割される場合、
    それぞれ個別に処理されます。

    :returns: 終了コード。成功時は0、エラー時は1、PyTorch未インストールの場合は2、
              ユーザーによる中断時は130を返します。
    :rtype: int
    """
    parser = build_parser()
    args = parser.parse_args()
    files = sorted(glob.glob(args.infile))
    if not files:
        parser.error(f"ファイルが見つかりません: {args.infile}")
    if len(files) > 1 and (args.outfile1 or args.outfile2 or args.outfile3):
        parser.error("複数の入力ファイルで明示的な出力ファイル名は使用できません")

    try:
        import torch
    except ImportError:
        print("エラー: PyTorchがこのPython環境にインストールされていません。", file=sys.stderr)
        return 2

    environment = collect_environment(torch)
    print_environment(environment)
    language = normalize_language(args.lang)
    backend = create_backend(args)

    load_started = time.perf_counter()
    backend.load()
    load_seconds = time.perf_counter() - load_started
    print(f"モデルロード時間: {load_seconds:.2f} 秒", flush=True)

    for input_path in files:
        paths = make_output_paths(input_path, args)
        duration = get_audio_duration(input_path)
        file_size = os.path.getsize(input_path)

        print("\n=== 文字起こし ===")
        print(f"入力ファイル     : {input_path}")
        print(f"音声長さ       : {format_hms(duration)}")
        print(f"言語           : {language or '自動検出'}")
        print(f"出力1 (時刻付き) : {paths.timestamped}")
        print(f"出力2 (テキスト) : {paths.plain}")
        print(f"出力3 (情報)    : {paths.info}")

        def save_partial_result(
            partial_result: ASRResult,
            progress: dict[str, Any],
        ) -> None:
            """チャンク処理中に中間結果を保存するためのコールバック関数。

            :param partial_result: 部分的な文字起こし結果。
            :type partial_result: ASRResult
            :param progress: 進捗情報を含む辞書。
            :type progress: dict[str, Any]
            """
            write_transcript_outputs(paths, partial_result)
            print(
                f"部分結果を保存しました: チャンク {progress['chunk_index']} "
                f"({progress['progress_percent']:.1f}%) -> "
                f"{paths.timestamped}, {paths.plain}",
                flush=True,
            )

        started = time.perf_counter()
        result = transcribe_with_optional_chunks(
            backend=backend,
            audio_path=input_path,
            language=language,
            duration=duration,
            chunk_seconds=args.chunk_seconds,
            chunk_overlap=args.chunk_overlap,
            use_chunks=bool(args.timestamps),
            progress_callback=save_partial_result,
        )
        elapsed = time.perf_counter() - started
        realtime_factor = elapsed / duration if duration and duration > 0 else None

        report = {
            "environment": environment,
            "backend": backend.describe(),
            "input": {
                "path": os.path.abspath(input_path),
                "size_bytes": file_size,
                "duration_seconds": duration,
                "requested_language": language,
            },
            "result": {
                "detected_language": result.language,
                "characters": len(result.text),
                "segments": len(result.segments),
                "elapsed_seconds": elapsed,
                "realtime_factor": realtime_factor,
                **result.engine_metadata,
            },
            "outputs": {key: str(value.resolve()) for key, value in asdict(paths).items()},
            "model_load_seconds": load_seconds,
        }
        write_outputs(paths, result, report)

        print(f"検出言語       : {result.language}")
        print(f"経過時間       : {elapsed:.2f} 秒")
        if realtime_factor is not None:
            print(f"リアルタイム係数 : {realtime_factor:.4f}x")
        print("\n=== 文字起こしテキスト ===")
        print(result.text, flush=True)

    return 0


if __name__ == "__main__":
    exit_code = 1
    try:
        exit_code = main()
    except KeyboardInterrupt:
        print("\n中断されました。", file=sys.stderr)
        exit_code = 130
    except Exception:
        traceback.print_exc()
        exit_code = 1
    finally:
        if "--pause" in sys.argv:
            try:
                # --pause 1 が指定されている場合のみ待機
                index = sys.argv.index("--pause")
                if index + 1 < len(sys.argv) and sys.argv[index + 1] == "1":
                    input("\nENTERキーを押して終了します。\n")
            except (ValueError, EOFError):
                pass
    raise SystemExit(exit_code)