#!/usr/bin/env python3
# -*- coding: utf-8 -*-

r"""
概要:
    PowerPointのpptxファイルとノートファイル間のノートテキストを変換します。
詳細説明:
    word2pptxモードでは、docxファイルやtxtファイル、mdファイルからスライドセクションを読み込み、
    対応するPowerPointスライドのノートを上書きします。
    pptx2wordモードでは、PowerPointスライドのすべてのノートを抽出し、
    docxファイルやtxtファイル、mdファイルに保存します。
    スライドのヘッダーは特定のパターンで認識され、スライド 3 や Slide 3 のような書式に対応します。
    依存関係としてpython-pptxが必要で、docxファイルを扱う場合はpython-docxが必要です。
"""

from __future__ import annotations

import argparse
from copy import deepcopy
import re
import shutil
import sys
from pathlib import Path

from pptx import Presentation


SLIDE_HEADER_RE = re.compile(
    r"^\s*#\s*(?:スライド|Slides?)?\s*(\d+)\s*$",
    re.IGNORECASE,
)

NARRATION_SEPARATOR = "---\n((\n{narration}\n))"


def warning(message: str) -> None:
    """
    概要:
        警告メッセージを標準エラー出力に出力します。
    引数:
        :param message: 警告として出力するメッセージテキスト。
        :type message: str
    戻り値:
        :returns: なし。
        :rtype: None
    """
    print(f"Warning: {message}", file=sys.stderr)


def _parse_note_lines(lines: list[str], source_name: str) -> dict[int, str]:
    """
    概要:
        テキスト行のリストからスライド番号ごとのノートセクションを解析します。
    引数:
        :param lines: 解析対象となるテキスト行のリスト。
        :type lines: list[str]
        :param source_name: 警告メッセージに使用する入力元の名前。
        :type source_name: str
    戻り値:
        :returns: スライド番号をキー、ノートテキストを値とする辞書。
        :rtype: dict[int, str]
    """
    sections: dict[int, str] = {}
    current_slide: int | None = None
    current_lines: list[str] = []
    preamble_lines: list[str] = []

    def store_current() -> None:
        nonlocal current_slide, current_lines
        if current_slide is None:
            return

        note_text = "\n".join(current_lines).strip("\n")
        if current_slide in sections:
            warning(
                f"slide {current_slide} appears more than once in {source_name}; "
                "the last section is used."
            )
        sections[current_slide] = note_text

    for text in lines:
        match = SLIDE_HEADER_RE.match(text)

        if match:
            store_current()
            current_slide = int(match.group(1))
            current_lines = []
            continue

        if current_slide is None:
            preamble_lines.append(text)
        else:
            current_lines.append(text)

    store_current()

    if any(line.strip() for line in preamble_lines):
        warning("text before the first slide header was ignored.")

    if not sections:
        warning(f"no slide headers were found in {source_name}.")

    return sections


def read_notes_sections(notes_path: Path) -> dict[int, str]:
    """
    概要:
        docxやtxt、mdファイルからスライド番号ごとのノートセクションを読み込みます。
    引数:
        :param notes_path: 読み込むノートファイルのパス。
        :type notes_path: pathlib.Path
    戻り値:
        :returns: スライド番号をキー、ノートテキストを値とする辞書。
        :rtype: dict[int, str]
    例外:
        :raises RuntimeError: python-docxパッケージがインストールされていない場合。
        :raises ValueError: サポートされていないファイルの拡張子の場合。
    """
    suffix = notes_path.suffix.lower()

    if suffix == ".docx":
        try:
            from docx import Document
        except ImportError:
            raise RuntimeError(
                "python-docx is required for .docx files. "
                "Install it with: pip install python-docx"
            ) from None

        doc = Document(notes_path)
        lines = [paragraph.text for paragraph in doc.paragraphs]
        return _parse_note_lines(lines, "Word file")

    if suffix in (".txt", ".md"):
        text = notes_path.read_text(encoding="utf-8-sig")
        return _parse_note_lines(text.splitlines(), f"{suffix} file")

    raise ValueError(
        f"unsupported notes file extension: {notes_path.suffix or '(none)'}; "
        "use .docx, .txt, or .md"
    )


def get_slide_notes(slide) -> str:
    """
    概要:
        ノートスライドを作成せずに1つのスライドのノートテキストを返します。
    引数:
        :param slide: 読み込み対象のスライドオブジェクト。
        :type slide: pptx.slide.Slide
    戻り値:
        :returns: ノートのテキスト文字列。
        :rtype: str
    """
    if not slide.has_notes_slide:
        return ""

    text_frame = slide.notes_slide.notes_text_frame
    if text_frame is None:
        return ""

    return text_frame.text or ""


def set_slide_notes(slide, text: str) -> bool:
    """
    概要:
        1つのスライドのノートテキストを上書きします。
    詳細説明:
        ノートのプレースホルダーが利用できない場合のみFalseを返します。
    引数:
        :param slide: 書き込み対象のスライドオブジェクト。
        :type slide: pptx.slide.Slide
        :param text: 上書きするテキスト。
        :type text: str
    戻り値:
        :returns: 成功した場合はTrue、プレースホルダーがない場合はFalse。
        :rtype: bool
    """
    text_frame = slide.notes_slide.notes_text_frame
    if text_frame is None:
        return False

    text_frame.text = text
    return True


def compose_notes(manuscript: str, narration: str | None) -> str:
    """
    概要:
        指定された区切り文字を使用して原稿とナレーションを結合します。
    引数:
        :param manuscript: 原稿のテキスト。
        :type manuscript: str
        :param narration: ナレーションのテキスト。
        :type narration: str | None
    戻り値:
        :returns: 結合されたノートテキスト。
        :rtype: str
    """
    manuscript = manuscript.strip("\n")
    if narration is None or not narration.strip():
        return manuscript

    narration_block = NARRATION_SEPARATOR.format(narration=narration.strip("\n"))
    return f"{manuscript}\n{narration_block}" if manuscript else narration_block


def nonempty_lines(text: str) -> list[str]:
    """
    概要:
        空行を無視してテキストを行のリストに分割します。
    引数:
        :param text: 分割対象のテキスト。
        :type text: str
    戻り値:
        :returns: 空でない行のリスト。
        :rtype: list[str]
    """
    return [line for line in text.splitlines() if line.strip()]


def duplicate_slide(prs: Presentation, source_slide):
    """
    概要:
        指定されたスライドのコピーを追加し、新しいスライドを返します。
    詳細説明:
        python-pptxにはスライドをコピーする公開APIがないため、
        図形と外部関係を同じレイアウトの新しいスライドにコピーします。
    引数:
        :param prs: プレゼンテーションオブジェクト。
        :type prs: pptx.presentation.Presentation
        :param source_slide: コピー元のスライド。
        :type source_slide: pptx.slide.Slide
    戻り値:
        :returns: コピーして作成された新しいスライドオブジェクト。
        :rtype: pptx.slide.Slide
    """
    new_slide = prs.slides.add_slide(source_slide.slide_layout)

    # Remove placeholders automatically supplied by the layout before copying.
    for shape in list(new_slide.shapes):
        new_slide.shapes._spTree.remove(shape.element)

    for shape in source_slide.shapes:
        new_slide.shapes._spTree.insert_element_before(
            deepcopy(shape.element), "p:extLst"
        )

    # Copy relationships used by pictures, hyperlinks, charts, and other
    # slide content. Layout and notes relationships belong to the new slide.
    excluded = {
        "http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide",
        "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slideLayout",
    }
    relationship_ids: dict[str, str] = {}
    for rel in source_slide.part.rels.values():
        if rel.reltype in excluded:
            continue
        relationship_ids[rel.rId] = new_slide.part.rels._add_relationship(
            rel.reltype, rel._target, rel.is_external
        )

    # Copied shape XML still contains the source slide's relationship IDs.
    for element in new_slide.shapes._spTree.iter():
        for attribute, value in list(element.attrib.items()):
            if value in relationship_ids:
                element.set(attribute, relationship_ids[value])

    return new_slide


def move_slide_after(prs: Presentation, slide, after_slide) -> None:
    """
    概要:
        スライドをプレゼンテーション内の別のスライドの直後に移動します。
    引数:
        :param prs: プレゼンテーションオブジェクト。
        :type prs: pptx.presentation.Presentation
        :param slide: 移動するスライドオブジェクト。
        :type slide: pptx.slide.Slide
        :param after_slide: このスライドの直後に移動します。
        :type after_slide: pptx.slide.Slide
    戻り値:
        :returns: なし。
        :rtype: None
    例外:
        :raises RuntimeError: スライドの順序を特定できなかった場合。
    """
    slide_ids = prs.slides._sldIdLst
    # Locate IDs through their related slide parts; rId values cannot be
    # inferred from part names.
    target_id = None
    after_id = None
    for item in slide_ids:
        related_part = prs.part.related_part(item.rId)
        if related_part is slide.part:
            target_id = item
        if related_part is after_slide.part:
            after_id = item
    if target_id is None or after_id is None:
        raise RuntimeError("could not determine slide order while copying a slide")
    slide_ids.remove(target_id)
    slide_ids.insert(slide_ids.index(after_id) + 1, target_id)


def notes_to_pptx(
    pptx_path: Path,
    manuscript_path: Path,
    narration_path: Path | None,
    output_path: Path | None,
    split_lines: bool,
) -> int:
    """
    概要:
        原稿とナレーションのノートを読み込み、pptxファイルのノートに反映させます。
    引数:
        :param pptx_path: 対象のpptxファイルのパス。
        :type pptx_path: pathlib.Path
        :param manuscript_path: 原稿ファイルのパス。
        :type manuscript_path: pathlib.Path
        :param narration_path: ナレーションファイルのパス、またはNone。
        :type narration_path: pathlib.Path | None
        :param output_path: 出力するpptxファイルのパス、またはNone。
        :type output_path: pathlib.Path | None
        :param split_lines: 行ごとにスライドを分割するかどうかのフラグ。
        :type split_lines: bool
    戻り値:
        :returns: 終了ステータスコード。
        :rtype: int
    """
    manuscript_sections = read_notes_sections(manuscript_path)
    narration_sections = (
        read_notes_sections(narration_path) if narration_path is not None else {}
    )
    prs = Presentation(pptx_path)
    nslides = len(prs.slides)

    updated = 0
    skipped = 0

    original_slides = list(prs.slides)
    all_slide_numbers = list(manuscript_sections)
    for slide_no in narration_sections:
        if slide_no not in manuscript_sections:
            all_slide_numbers.append(slide_no)

    for slide_no in all_slide_numbers:
        if slide_no < 1 or slide_no > nslides:
            warning(
                f"slide {slide_no} does not exist in PPTX "
                f"(valid range: 1-{nslides}); skipped."
            )
            skipped += 1
            continue

        manuscript = manuscript_sections.get(slide_no, "")
        narration = narration_sections.get(slide_no)
        source_slide = original_slides[slide_no - 1]

        if split_lines:
            manuscript_lines = nonempty_lines(manuscript)
            narration_lines = nonempty_lines(narration or "")
            line_count = max(len(manuscript_lines), len(narration_lines))
            if not line_count:
                warning(f"slide {slide_no} contains no nonblank lines; skipped.")
                skipped += 1
                continue
            if narration_path is not None and len(manuscript_lines) != len(narration_lines):
                warning(
                    f"slide {slide_no}: manuscript has {len(manuscript_lines)} "
                    f"nonblank lines, narration has {len(narration_lines)}; "
                    "missing lines are left empty."
                )

            current_slide = source_slide
            for line_index in range(line_count):
                if line_index:
                    copied_slide = duplicate_slide(prs, source_slide)
                    move_slide_after(prs, copied_slide, current_slide)
                    current_slide = copied_slide
                manuscript_line = (
                    manuscript_lines[line_index]
                    if line_index < len(manuscript_lines) else ""
                )
                narration_line = (
                    narration_lines[line_index]
                    if line_index < len(narration_lines) else None
                )
                if not set_slide_notes(
                    current_slide, compose_notes(manuscript_line, narration_line)
                ):
                    warning(
                        f"slide {slide_no}, line {line_index + 1} has no usable "
                        "notes placeholder; skipped."
                    )
                    skipped += 1
                    continue
                print(f"Updated: slide {slide_no}, line {line_index + 1}")
                updated += 1
        else:
            note_text = compose_notes(manuscript, narration)
            if not set_slide_notes(source_slide, note_text):
                warning(f"slide {slide_no} has no usable notes placeholder; skipped.")
                skipped += 1
                continue
            print(f"Updated: slide {slide_no}")
            updated += 1

    if output_path is None:
        output_path = pptx_path

    # python-pptx writes a complete package. Saving directly over the input
    # file is supported, but a temporary file makes replacement safer.
    if output_path.resolve() == pptx_path.resolve():
        tmp_path = pptx_path.with_name(pptx_path.stem + ".__notes_tmp__.pptx")
        prs.save(tmp_path)
        shutil.move(tmp_path, pptx_path)
    else:
        output_path.parent.mkdir(parents=True, exist_ok=True)
        prs.save(output_path)

    print(f"Saved: {output_path}")
    print(f"Updated slides: {updated}")
    if skipped:
        print(f"Skipped sections: {skipped}")

    return 0


def pptx_to_notes(pptx_path: Path, notes_path: Path) -> int:
    """
    概要:
        pptxファイルからノートを抽出し、docxやtxt、mdファイルにエクスポートします。
    引数:
        :param pptx_path: 入力元のpptxファイルのパス。
        :type pptx_path: pathlib.Path
        :param notes_path: 出力先のノートファイルのパス。
        :type notes_path: pathlib.Path
    戻り値:
        :returns: 終了ステータスコード。
        :rtype: int
    例外:
        :raises RuntimeError: docx出力でpython-docxがない場合。
        :raises ValueError: サポートされていないファイルの拡張子の場合。
    """
    prs = Presentation(pptx_path)
    suffix = notes_path.suffix.lower()

    if suffix == ".docx":
        try:
            from docx import Document
        except ImportError:
            raise RuntimeError(
                "python-docx is required for .docx files. "
                "Install it with: pip install python-docx"
            ) from None

        doc = Document()

        for slide_no, slide in enumerate(prs.slides, start=1):
            # Keep the literal marker in the paragraph text so the generated
            # file can be fed back to notes2pptx/word2pptx unchanged.
            p = doc.add_paragraph(f"# Slide {slide_no}")
            try:
                p.style = "Heading 1"
            except KeyError:
                pass

            note_text = get_slide_notes(slide)
            if note_text:
                for line in note_text.split("\n"):
                    doc.add_paragraph(line)
            else:
                doc.add_paragraph("")

            doc.add_paragraph("")

        notes_path.parent.mkdir(parents=True, exist_ok=True)
        doc.save(notes_path)

    elif suffix in (".txt", ".md"):
        blocks: list[str] = []
        for slide_no, slide in enumerate(prs.slides, start=1):
            note_text = get_slide_notes(slide).strip("\n")
            block = f"# Slide {slide_no}\n"
            if note_text:
                block += note_text + "\n"
            blocks.append(block)

        notes_path.parent.mkdir(parents=True, exist_ok=True)
        notes_path.write_text("\n".join(blocks), encoding="utf-8")

    else:
        raise ValueError(
            f"unsupported notes file extension: {notes_path.suffix or '(none)'}; "
            "use .docx, .txt, or .md"
        )

    print(f"Saved: {notes_path}")
    print(f"Slides exported: {len(prs.slides)}")
    return 0


def build_parser() -> argparse.ArgumentParser:
    """
    概要:
        コマンドライン引数のパーサーを構築します。
    戻り値:
        :returns: 設定済みの引数パーサーオブジェクト。
        :rtype: argparse.ArgumentParser
    """
    parser = argparse.ArgumentParser(
        description="Convert PowerPoint speaker notes to/from .docx/.txt/.md files."
    )
    parser.add_argument(
        "--mode",
        required=True,
        choices=("word2pptx", "pptx2word", "notes2pptx", "pptx2notes"),
        help="Conversion direction.",
    )
    parser.add_argument(
        "files",
        nargs="*",
        metavar="NOTES_FILE",
        help=(
            "Manuscript file followed optionally by a narration file "
            "(.docx, .txt, or .md)."
        ),
    )
    parser.add_argument(
        "-p",
        "--pptx",
        required=True,
        help="PowerPoint .pptx file.",
    )
    parser.add_argument(
        "-w",
        "--word",
        "--notes",
        dest="notes",
        help="Legacy alternative to the manuscript positional argument.",
    )
    parser.add_argument(
        "--split_lines",
        type=lambda value: value.lower() in ("true", "1", "yes", "on"),
        default=False,
        metavar="BOOL",
        help="For word2pptx, copy each slide once per nonblank line (default: False).",
    )
    parser.add_argument(
        "-o",
        "--output",
        help=(
            "Output .pptx for mode=word2pptx/notes2pptx. "
            "If omitted, the input PPTX is overwritten. "
            "Ignored in mode=pptx2word/pptx2notes."
        ),
    )
    return parser


def main() -> int:
    """
    概要:
        コマンドラインからの実行を処理するメイン関数です。
    戻り値:
        :returns: 終了ステータスコード。
        :rtype: int
    """
    args = build_parser().parse_args()

    pptx_path = Path(args.pptx).expanduser()
    if len(args.files) > 2:
        print("Error: specify at most two positional files.", file=sys.stderr)
        return 2
    if args.notes and args.files:
        print("Error: do not combine --word/--notes with positional files.", file=sys.stderr)
        return 2
    if not args.files and not args.notes:
        print("Error: a manuscript/notes file is required.", file=sys.stderr)
        return 2

    manuscript_path = Path(args.files[0] if args.files else args.notes).expanduser()
    narration_path = (
        Path(args.files[1]).expanduser() if len(args.files) == 2 else None
    )

    if args.mode in ("word2pptx", "notes2pptx"):
        if not pptx_path.is_file():
            print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
            return 2
        for label, path in (("manuscript", manuscript_path), ("narration", narration_path)):
            if path is None:
                continue
            if not path.is_file():
                print(f"Error: {label} file not found: {path}", file=sys.stderr)
                return 2
            if path.suffix.lower() not in (".docx", ".txt", ".md"):
                print(f"Error: {label} file must have .docx, .txt, or .md extension.", file=sys.stderr)
                return 2

        output_path = Path(args.output).expanduser() if args.output else None
        if output_path is not None and output_path.suffix.lower() != ".pptx":
            print("Error: output file must have .pptx extension.", file=sys.stderr)
            return 2

        try:
            return notes_to_pptx(
                pptx_path, manuscript_path, narration_path, output_path,
                args.split_lines,
            )
        except (ValueError, RuntimeError, UnicodeError) as exc:
            print(f"Error: {exc}", file=sys.stderr)
            return 2

    if args.output:
        warning("--output is ignored in mode=pptx2word/pptx2notes; --word/--notes is the output file.")
    if narration_path is not None:
        print("Error: pptx2word/pptx2notes accepts only one output notes file.", file=sys.stderr)
        return 2
    if args.split_lines:
        warning("--split_lines is ignored in mode=pptx2word/pptx2notes.")

    if not pptx_path.is_file():
        print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
        return 2

    if manuscript_path.suffix.lower() not in (".docx", ".txt", ".md"):
        print("Error: notes file must have .docx, .txt, or .md extension.", file=sys.stderr)
        return 2

    try:
        return pptx_to_notes(pptx_path, manuscript_path)
    except (ValueError, RuntimeError, UnicodeError) as exc:
        print(f"Error: {exc}", file=sys.stderr)
        return 2


if __name__ == "__main__":
    raise SystemExit(main())