#!/usr/bin/env python3
# -*- coding: utf-8 -*-

r"""Convert speaker notes between PowerPoint (.pptx) and notes files (.docx/.txt/.md).

【COM API版】
PowerPoint本体を直接操作(win32com)するため、数式や複雑なグラフが含まれていても
ファイルが破損することなく安全に処理できます。

Dependencies::

    pip install pywin32
    pip install python-docx   # only needed for .docx files
"""

from __future__ import annotations

import argparse
import re
import sys
from pathlib import Path
import win32com.client

SLIDE_HEADER_RE = re.compile(r"^\s*#\s*(?:スライド|Slides?)?\s*(\d+).*$", re.IGNORECASE)
NARRATION_SEPARATOR = "---\n((\n{narration}\n))"
SUBTITLE_SHAPE_NAME = "pptx_notes_word subtitle"
FIXED_SPEAKERS = ("四国めたん", "ずんだもん")

# COM Constants
msoTextOrientationHorizontal = 1
msoTrue = -1
msoFalse = 0
ppAlignLeft = 1
msoAnchorMiddle = 3
msoPlaceholder = 14
ppPlaceholderBody = 2

def warning(message: str) -> None:
    print(f"Warning: {message}", file=sys.stderr)

def _parse_note_lines(lines: list[str], source_name: str) -> dict[int, str]:
    sections: dict[int, str] = {}
    current_slide: int | None = None
    current_lines: list[str] = []
    preamble_lines: list[str] = []

    def store_current() -> None:
        nonlocal current_slide, current_lines
        if current_slide is None:
            return
        note_text = "\n".join(current_lines).strip("\n")
        if current_slide in sections:
            warning(f"slide {current_slide} appears more than once in {source_name}; the last section is used.")
        sections[current_slide] = note_text

    for text in lines:
        match = SLIDE_HEADER_RE.match(text)
        if match:
            store_current()
            current_slide = int(match.group(1))
            current_lines = []
            continue
        if current_slide is None:
            preamble_lines.append(text)
        else:
            current_lines.append(text)

    store_current()
    if any(line.strip() for line in preamble_lines):
        warning("text before the first slide header was ignored.")
    if not sections:
        warning(f"no slide headers were found in {source_name}.")
    return sections

def read_notes_sections(notes_path: Path) -> dict[int, str]:
    suffix = notes_path.suffix.lower()
    if suffix == ".docx":
        try:
            from docx import Document
        except ImportError:
            raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
        doc = Document(notes_path)
        lines = [paragraph.text for paragraph in doc.paragraphs]
        return _parse_note_lines(lines, "Word file")
    if suffix in (".txt", ".md"):
        text = notes_path.read_text(encoding="utf-8-sig")
        return _parse_note_lines(text.splitlines(), f"{suffix} file")
    raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")

def get_slide_notes(slide) -> str:
    notes_page = slide.NotesPage
    for i in range(1, notes_page.Shapes.Count + 1):
        shape = notes_page.Shapes.Item(i)
        if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
            if shape.HasTextFrame and shape.TextFrame.HasText:
                return shape.TextFrame.TextRange.Text
    return ""

def set_slide_notes(slide, text: str) -> bool:
    notes_page = slide.NotesPage
    for i in range(1, notes_page.Shapes.Count + 1):
        shape = notes_page.Shapes.Item(i)
        if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
            shape.TextFrame.TextRange.Text = text
            return True
    return False

def compose_notes(manuscript: str, narration: str | None) -> str:
    manuscript = manuscript.strip("\n")
    if narration is None:
        return manuscript

    # --- 追加: 指定された特定の行だけを除外する処理 ---
    filtered_lines = []
    for line in narration.splitlines():
        stripped = line.strip()
        # '-'のみ、'='のみ、または '#' で始まる行はスキップ
        if set(stripped) in ({"-"}, {"="}) or stripped.startswith("#"):
            continue
        filtered_lines.append(line)
    
    # フィルタリング後のテキストを結合
    filtered_narration = "\n".join(filtered_lines).strip("\n")
    
    # 有効な読み上げテキストが残らなかった場合は原稿のみを返す
    if not filtered_narration:
        return manuscript
    # ---------------------------------------------------

    narration_block = NARRATION_SEPARATOR.format(narration=filtered_narration)
    return f"{manuscript}\n{narration_block}" if manuscript else narration_block
    
"""
def compose_notes(manuscript: str, narration: str | None) -> str:
    manuscript = manuscript.strip("\n")
    if narration is None or not narration.strip():
        return manuscript
    narration_block = NARRATION_SEPARATOR.format(narration=narration.strip("\n"))
    return f"{manuscript}\n{narration_block}" if manuscript else narration_block
"""

def split_speaker_line(line: str) -> tuple[str | None, str]:
    speaker_text, separator, body = line.partition(",")
    speaker = speaker_text.strip()
    if separator and speaker in FIXED_SPEAKERS:
        return speaker, body.strip()
    return None, line.strip()

def subtitle_text(text: str) -> str:
    formatted: list[str] = []
    for line in text.splitlines():
        speaker, body = split_speaker_line(line)
        formatted.append(f"{speaker}：{body}" if speaker else line)
    return "\n".join(formatted)

def warn_speaker_mismatch(manuscript_line: str, narration_line: str | None, slide_no: int, line_no: int) -> None:
    if narration_line is None: return
    manuscript_speaker, _ = split_speaker_line(manuscript_line)
    narration_speaker, _ = split_speaker_line(narration_line)
    if manuscript_speaker and narration_speaker and manuscript_speaker != narration_speaker:
        warning(f"slide {slide_no}, line {line_no}: manuscript speaker '{manuscript_speaker}' differs from narration speaker '{narration_speaker}'; narration speaker is used for TTS.")

def is_ignored_split_line(line: str) -> bool:
    stripped = line.strip()
    if not stripped: return True
    if set(stripped) in ({"-"}, {"="}): return True
    if stripped.startswith("#"): return True
    if stripped in ("((", "))"): return True
    return False

def nonempty_lines(text: str) -> list[str]:
    return [line for line in text.splitlines() if not is_ignored_split_line(line)]

def hex_to_bgr(hex_str: str) -> int:
    """16進数RGB文字列をCOM API用のBGR整数値に変換する"""
    if not re.fullmatch(r"[0-9A-Fa-f]{6}", hex_str):
        raise ValueError(f"invalid RGB color: {hex_str}; use six hexadecimal digits")
    r = int(hex_str[0:2], 16)
    g = int(hex_str[2:4], 16)
    b = int(hex_str[4:6], 16)
    return r + (g << 8) + (b << 16)

def add_subtitle(slide, prs, text: str, args) -> None:
    text = text.strip("\n")
    if not text:
        return
    
    # 既存の字幕シェイプを削除
    shapes_to_delete = []
    for i in range(1, slide.Shapes.Count + 1):
        shape = slide.Shapes.Item(i)
        if shape.Name == SUBTITLE_SHAPE_NAME:
            shapes_to_delete.append(shape)
    for shape in shapes_to_delete:
        shape.Delete()

    margin = args.subtitle_box_margin
    height = args.subtitle_box_height
    bottom_margin = args.subtitle_bottom_margin
    
    slide_width = prs.PageSetup.SlideWidth
    slide_height = prs.PageSetup.SlideHeight

    left = margin
    top = slide_height - height - bottom_margin
    width = slide_width - 2 * margin

    shape = slide.Shapes.AddTextbox(msoTextOrientationHorizontal, left, top, width, height)
    shape.Name = SUBTITLE_SHAPE_NAME

    shape.Fill.Visible = msoTrue
    shape.Fill.Solid()
    shape.Fill.ForeColor.RGB = hex_to_bgr(args.subtitle_bgcolor)
    shape.Fill.Transparency = args.subtitle_bg_transparency / 100.0
    shape.Line.Visible = msoFalse

    text_frame = shape.TextFrame
    text_frame.WordWrap = msoTrue
    text_frame.VerticalAnchor = msoAnchorMiddle

    text_range = text_frame.TextRange
    text_range.Text = subtitle_text(text)
    text_range.ParagraphFormat.Alignment = ppAlignLeft
    text_range.Font.Name = args.subtitle_font_name
    text_range.Font.Size = args.subtitle_font_size
    text_range.Font.Color.RGB = hex_to_bgr(args.subtitle_font_color)

def notes_to_pptx(pptx_path: Path, manuscript_path: Path, narration_path: Path | None, output_path: Path | None, split_lines: bool, add_subtitles: bool, args) -> int:
    manuscript_sections = read_notes_sections(manuscript_path)
    narration_sections = read_notes_sections(narration_path) if narration_path is not None else {}
    
    all_slide_numbers = sorted(list(set(list(manuscript_sections) + list(narration_sections))), reverse=True)
    if not all_slide_numbers:
        return 0

    print("PowerPointを起動中...")
    Application = win32com.client.Dispatch("PowerPoint.Application")
    # PPTを背後で処理するために開く
    prs = Application.Presentations.Open(str(pptx_path.absolute()))
    
    updated = 0
    skipped = 0

    try:
        nslides = prs.Slides.Count

        # 後ろのスライドから順に処理（スライドを複製しても前のスライドのインデックスがずれないようにするため）
        for slide_no in all_slide_numbers:
            if slide_no < 1 or slide_no > nslides:
                warning(f"slide {slide_no} does not exist in PPTX (valid range: 1-{nslides}); skipped.")
                skipped += 1
                continue

            manuscript = manuscript_sections.get(slide_no, "")
            narration = narration_sections.get(slide_no)
            source_slide = prs.Slides.Item(slide_no)

            if split_lines:
                manuscript_lines = nonempty_lines(manuscript)
                narration_lines = nonempty_lines(narration or "")
                line_count = max(len(manuscript_lines), len(narration_lines))
                if not line_count:
                    warning(f"slide {slide_no} contains no content lines; skipped.")
                    skipped += 1
                    continue
                
                current_slide = source_slide
                for line_index in range(line_count):
                    # 2行目以降はスライドを複製（COMのDuplicateは自動的に直後に挿入されます）
                    if line_index > 0:
                        current_slide = current_slide.Duplicate().Item(1)

                    manuscript_line = manuscript_lines[line_index] if line_index < len(manuscript_lines) else ""
                    narration_line = narration_lines[line_index] if line_index < len(narration_lines) else None
                    warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index + 1)
                    
                    note_text = compose_notes(manuscript_line, narration_line)
                    if not set_slide_notes(current_slide, note_text):
                        warning(f"slide {slide_no}, line {line_index + 1} has no usable notes placeholder; skipped.")
                        skipped += 1
                        continue
                    if add_subtitles:
                        add_subtitle(current_slide, prs, manuscript_line, args)
                    print(f"Updated: slide {slide_no}, line {line_index + 1}")
                    updated += 1
            else:
                manuscript_lines = nonempty_lines(manuscript)
                narration_lines = nonempty_lines(narration or "")
                for line_index, (manuscript_line, narration_line) in enumerate(zip(manuscript_lines, narration_lines), start=1):
                    warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index)
                
                note_text = compose_notes(manuscript, narration)
                if not set_slide_notes(source_slide, note_text):
                    warning(f"slide {slide_no} has no usable notes placeholder; skipped.")
                    skipped += 1
                    continue
                if add_subtitles:
                    add_subtitle(source_slide, prs, manuscript, args)
                print(f"Updated: slide {slide_no}")
                updated += 1

        if output_path is None:
            prs.Save()
            saved_path = pptx_path
        else:
            out_abs = str(output_path.absolute())
            prs.SaveCopyAs(out_abs)
            saved_path = output_path

        print(f"Saved: {saved_path}")
        print(f"Updated slides: {updated}")
        if skipped:
            print(f"Skipped sections: {skipped}")

    finally:
        prs.Close()
        # 他に開いているプレゼンテーションがなければPowerPointを終了
        if Application.Presentations.Count == 0:
            Application.Quit()

    return 0

def pptx_to_notes(pptx_path: Path, notes_path: Path) -> int:
    suffix = notes_path.suffix.lower()

    print("PowerPointを起動中...")
    Application = win32com.client.Dispatch("PowerPoint.Application")
    prs = Application.Presentations.Open(str(pptx_path.absolute()), WithWindow=msoFalse)
    
    try:
        nslides = prs.Slides.Count

        if suffix == ".docx":
            try:
                from docx import Document
            except ImportError:
                raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
            doc = Document()
            for slide_no in range(1, nslides + 1):
                slide = prs.Slides.Item(slide_no)
                p = doc.add_paragraph(f"# Slide {slide_no}")
                try: p.style = "Heading 1"
                except KeyError: pass
                note_text = get_slide_notes(slide)
                if note_text:
                    for line in note_text.split("\n"):
                        doc.add_paragraph(line)
                else:
                    doc.add_paragraph("")
                doc.add_paragraph("")
            notes_path.parent.mkdir(parents=True, exist_ok=True)
            doc.save(notes_path)
            
        elif suffix in (".txt", ".md"):
            blocks: list[str] = []
            for slide_no in range(1, nslides + 1):
                slide = prs.Slides.Item(slide_no)
                note_text = get_slide_notes(slide).strip("\n")
                block = f"# Slide {slide_no}\n"
                if note_text:
                    block += note_text + "\n"
                blocks.append(block)
            notes_path.parent.mkdir(parents=True, exist_ok=True)
            notes_path.write_text("\n".join(blocks), encoding="utf-8")
        else:
            raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")

        print(f"Saved: {notes_path}")
        print(f"Slides exported: {nslides}")
        
    finally:
        prs.Close()
        if Application.Presentations.Count == 0:
            Application.Quit()
            
    return 0

def build_parser() -> argparse.ArgumentParser:
    parser = argparse.ArgumentParser(description="Convert PowerPoint speaker notes to/from .docx/.txt/.md files (COM API ver).")
    parser.add_argument("--mode", required=True, choices=("word2pptx", "pptx2word", "notes2pptx", "pptx2notes"))
    parser.add_argument("files", nargs="*", metavar="NOTES_FILE")
    parser.add_argument("-p", "--pptx", required=True)
    parser.add_argument("-w", "--word", "--notes", dest="notes")
    parser.add_argument("--split_lines", type=int, choices=(0, 1), default=0, metavar="0|1")
    parser.add_argument("--add_subtitles", type=int, choices=(0, 1), default=0, metavar="0|1")
    parser.add_argument("--subtitle_bottom_margin", type=float, default=20)
    parser.add_argument("--subtitle_box_margin", type=float, default=50)
    parser.add_argument("--subtitle_box_height", type=float, default=60)
    parser.add_argument("--subtitle_font_name", default="メイリオ")
    parser.add_argument("--subtitle_font_size", type=float, default=12)
    parser.add_argument("--subtitle_font_color", default="884444")
    parser.add_argument("--subtitle_bgcolor", default="00DDDD")
    parser.add_argument("--subtitle_bg_transparency", type=float, default=0, metavar="0..100")
    parser.add_argument("-o", "--output")
    return parser

def main() -> int:
    args = build_parser().parse_args()
    pptx_path = Path(args.pptx).expanduser()
    if len(args.files) > 2:
        print("Error: specify at most two positional files.", file=sys.stderr)
        return 2
    if args.notes and args.files:
        print("Error: do not combine --word/--notes with positional files.", file=sys.stderr)
        return 2
    if not args.files and not args.notes:
        print("Error: a manuscript/notes file is required.", file=sys.stderr)
        return 2

    manuscript_path = Path(args.files[0] if args.files else args.notes).expanduser()
    narration_path = Path(args.files[1]).expanduser() if len(args.files) == 2 else None

    if args.mode in ("word2pptx", "notes2pptx"):
        if not pptx_path.is_file():
            print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
            return 2
        for label, path in (("manuscript", manuscript_path), ("narration", narration_path)):
            if path is None: continue
            if not path.is_file():
                print(f"Error: {label} file not found: {path}", file=sys.stderr)
                return 2
            if path.suffix.lower() not in (".docx", ".txt", ".md"):
                print(f"Error: {label} file must have .docx, .txt, or .md extension.", file=sys.stderr)
                return 2

        output_path = Path(args.output).expanduser() if args.output else None
        try:
            return notes_to_pptx(pptx_path, manuscript_path, narration_path, output_path, bool(args.split_lines), bool(args.add_subtitles), args)
        except (ValueError, RuntimeError, UnicodeError) as exc:
            print(f"Error: {exc}", file=sys.stderr)
            return 2

    if args.output: warning("--output is ignored in mode=pptx2word/pptx2notes; --word/--notes is the output file.")
    if narration_path is not None:
        print("Error: pptx2word/pptx2notes accepts only one output notes file.", file=sys.stderr)
        return 2

    if not pptx_path.is_file():
        print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
        return 2

    try:
        return pptx_to_notes(pptx_path, manuscript_path)
    except (ValueError, RuntimeError, UnicodeError) as exc:
        print(f"Error: {exc}", file=sys.stderr)
        return 2

if __name__ == "__main__":
    raise SystemExit(main())