pptx_notes_word2.py ダウンロード/コピー

pptx_notes_word2.py をダウンロード

pptx_notes_word2.py
pptx_notes_word2.py
  1#!/usr/bin/env python3
  2# -*- coding: utf-8 -*-
  3
  4r"""Convert speaker notes between PowerPoint (.pptx) and notes files (.docx/.txt/.md).
  5
  6【COM API版】
  7PowerPoint本体を直接操作(win32com)するため、数式や複雑なグラフが含まれていても
  8ファイルが破損することなく安全に処理できます。
  9
 10Dependencies::
 11
 12    pip install pywin32
 13    pip install python-docx   # only needed for .docx files
 14"""
 15
 16from __future__ import annotations
 17
 18import argparse
 19import re
 20import sys
 21from pathlib import Path
 22import win32com.client
 23
 24SLIDE_HEADER_RE = re.compile(r"^\s*#\s*(?:スライド|Slides?)?\s*(\d+).*$", re.IGNORECASE)
 25NARRATION_SEPARATOR = "---\n((\n{narration}\n))"
 26SUBTITLE_SHAPE_NAME = "pptx_notes_word subtitle"
 27FIXED_SPEAKERS = ("四国めたん", "ずんだもん")
 28
 29# COM Constants
 30msoTextOrientationHorizontal = 1
 31msoTrue = -1
 32msoFalse = 0
 33ppAlignLeft = 1
 34msoAnchorMiddle = 3
 35msoPlaceholder = 14
 36ppPlaceholderBody = 2
 37
 38def warning(message: str) -> None:
 39    print(f"Warning: {message}", file=sys.stderr)
 40
 41def _parse_note_lines(lines: list[str], source_name: str) -> dict[int, str]:
 42    sections: dict[int, str] = {}
 43    current_slide: int | None = None
 44    current_lines: list[str] = []
 45    preamble_lines: list[str] = []
 46
 47    def store_current() -> None:
 48        nonlocal current_slide, current_lines
 49        if current_slide is None:
 50            return
 51        note_text = "\n".join(current_lines).strip("\n")
 52        if current_slide in sections:
 53            warning(f"slide {current_slide} appears more than once in {source_name}; the last section is used.")
 54        sections[current_slide] = note_text
 55
 56    for text in lines:
 57        match = SLIDE_HEADER_RE.match(text)
 58        if match:
 59            store_current()
 60            current_slide = int(match.group(1))
 61            current_lines = []
 62            continue
 63        if current_slide is None:
 64            preamble_lines.append(text)
 65        else:
 66            current_lines.append(text)
 67
 68    store_current()
 69    if any(line.strip() for line in preamble_lines):
 70        warning("text before the first slide header was ignored.")
 71    if not sections:
 72        warning(f"no slide headers were found in {source_name}.")
 73    return sections
 74
 75def read_notes_sections(notes_path: Path) -> dict[int, str]:
 76    suffix = notes_path.suffix.lower()
 77    if suffix == ".docx":
 78        try:
 79            from docx import Document
 80        except ImportError:
 81            raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
 82        doc = Document(notes_path)
 83        lines = [paragraph.text for paragraph in doc.paragraphs]
 84        return _parse_note_lines(lines, "Word file")
 85    if suffix in (".txt", ".md"):
 86        text = notes_path.read_text(encoding="utf-8-sig")
 87        return _parse_note_lines(text.splitlines(), f"{suffix} file")
 88    raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")
 89
 90def get_slide_notes(slide) -> str:
 91    notes_page = slide.NotesPage
 92    for i in range(1, notes_page.Shapes.Count + 1):
 93        shape = notes_page.Shapes.Item(i)
 94        if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
 95            if shape.HasTextFrame and shape.TextFrame.HasText:
 96                return shape.TextFrame.TextRange.Text
 97    return ""
 98
 99def set_slide_notes(slide, text: str) -> bool:
100    notes_page = slide.NotesPage
101    for i in range(1, notes_page.Shapes.Count + 1):
102        shape = notes_page.Shapes.Item(i)
103        if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
104            shape.TextFrame.TextRange.Text = text
105            return True
106    return False
107
108def compose_notes(manuscript: str, narration: str | None) -> str:
109    manuscript = manuscript.strip("\n")
110    if narration is None:
111        return manuscript
112
113    # --- 追加: 指定された特定の行だけを除外する処理 ---
114    filtered_lines = []
115    for line in narration.splitlines():
116        stripped = line.strip()
117        # '-'のみ、'='のみ、または '#' で始まる行はスキップ
118        if set(stripped) in ({"-"}, {"="}) or stripped.startswith("#"):
119            continue
120        filtered_lines.append(line)
121    
122    # フィルタリング後のテキストを結合
123    filtered_narration = "\n".join(filtered_lines).strip("\n")
124    
125    # 有効な読み上げテキストが残らなかった場合は原稿のみを返す
126    if not filtered_narration:
127        return manuscript
128    # ---------------------------------------------------
129
130    narration_block = NARRATION_SEPARATOR.format(narration=filtered_narration)
131    return f"{manuscript}\n{narration_block}" if manuscript else narration_block
132    
133"""
134def compose_notes(manuscript: str, narration: str | None) -> str:
135    manuscript = manuscript.strip("\n")
136    if narration is None or not narration.strip():
137        return manuscript
138    narration_block = NARRATION_SEPARATOR.format(narration=narration.strip("\n"))
139    return f"{manuscript}\n{narration_block}" if manuscript else narration_block
140"""
141
142def split_speaker_line(line: str) -> tuple[str | None, str]:
143    speaker_text, separator, body = line.partition(",")
144    speaker = speaker_text.strip()
145    if separator and speaker in FIXED_SPEAKERS:
146        return speaker, body.strip()
147    return None, line.strip()
148
149def subtitle_text(text: str) -> str:
150    formatted: list[str] = []
151    for line in text.splitlines():
152        speaker, body = split_speaker_line(line)
153        formatted.append(f"{speaker}:{body}" if speaker else line)
154    return "\n".join(formatted)
155
156def warn_speaker_mismatch(manuscript_line: str, narration_line: str | None, slide_no: int, line_no: int) -> None:
157    if narration_line is None: return
158    manuscript_speaker, _ = split_speaker_line(manuscript_line)
159    narration_speaker, _ = split_speaker_line(narration_line)
160    if manuscript_speaker and narration_speaker and manuscript_speaker != narration_speaker:
161        warning(f"slide {slide_no}, line {line_no}: manuscript speaker '{manuscript_speaker}' differs from narration speaker '{narration_speaker}'; narration speaker is used for TTS.")
162
163def is_ignored_split_line(line: str) -> bool:
164    stripped = line.strip()
165    if not stripped: return True
166    if set(stripped) in ({"-"}, {"="}): return True
167    if stripped.startswith("#"): return True
168    if stripped in ("((", "))"): return True
169    return False
170
171def nonempty_lines(text: str) -> list[str]:
172    return [line for line in text.splitlines() if not is_ignored_split_line(line)]
173
174def hex_to_bgr(hex_str: str) -> int:
175    """16進数RGB文字列をCOM API用のBGR整数値に変換する"""
176    if not re.fullmatch(r"[0-9A-Fa-f]{6}", hex_str):
177        raise ValueError(f"invalid RGB color: {hex_str}; use six hexadecimal digits")
178    r = int(hex_str[0:2], 16)
179    g = int(hex_str[2:4], 16)
180    b = int(hex_str[4:6], 16)
181    return r + (g << 8) + (b << 16)
182
183def add_subtitle(slide, prs, text: str, args) -> None:
184    text = text.strip("\n")
185    if not text:
186        return
187    
188    # 既存の字幕シェイプを削除
189    shapes_to_delete = []
190    for i in range(1, slide.Shapes.Count + 1):
191        shape = slide.Shapes.Item(i)
192        if shape.Name == SUBTITLE_SHAPE_NAME:
193            shapes_to_delete.append(shape)
194    for shape in shapes_to_delete:
195        shape.Delete()
196
197    margin = args.subtitle_box_margin
198    height = args.subtitle_box_height
199    bottom_margin = args.subtitle_bottom_margin
200    
201    slide_width = prs.PageSetup.SlideWidth
202    slide_height = prs.PageSetup.SlideHeight
203
204    left = margin
205    top = slide_height - height - bottom_margin
206    width = slide_width - 2 * margin
207
208    shape = slide.Shapes.AddTextbox(msoTextOrientationHorizontal, left, top, width, height)
209    shape.Name = SUBTITLE_SHAPE_NAME
210
211    shape.Fill.Visible = msoTrue
212    shape.Fill.Solid()
213    shape.Fill.ForeColor.RGB = hex_to_bgr(args.subtitle_bgcolor)
214    shape.Fill.Transparency = args.subtitle_bg_transparency / 100.0
215    shape.Line.Visible = msoFalse
216
217    text_frame = shape.TextFrame
218    text_frame.WordWrap = msoTrue
219    text_frame.VerticalAnchor = msoAnchorMiddle
220
221    text_range = text_frame.TextRange
222    text_range.Text = subtitle_text(text)
223    text_range.ParagraphFormat.Alignment = ppAlignLeft
224    text_range.Font.Name = args.subtitle_font_name
225    text_range.Font.Size = args.subtitle_font_size
226    text_range.Font.Color.RGB = hex_to_bgr(args.subtitle_font_color)
227
228def notes_to_pptx(pptx_path: Path, manuscript_path: Path, narration_path: Path | None, output_path: Path | None, split_lines: bool, add_subtitles: bool, args) -> int:
229    manuscript_sections = read_notes_sections(manuscript_path)
230    narration_sections = read_notes_sections(narration_path) if narration_path is not None else {}
231    
232    all_slide_numbers = sorted(list(set(list(manuscript_sections) + list(narration_sections))), reverse=True)
233    if not all_slide_numbers:
234        return 0
235
236    print("PowerPointを起動中...")
237    Application = win32com.client.Dispatch("PowerPoint.Application")
238    # PPTを背後で処理するために開く
239    prs = Application.Presentations.Open(str(pptx_path.absolute()))
240    
241    updated = 0
242    skipped = 0
243
244    try:
245        nslides = prs.Slides.Count
246
247        # 後ろのスライドから順に処理(スライドを複製しても前のスライドのインデックスがずれないようにするため)
248        for slide_no in all_slide_numbers:
249            if slide_no < 1 or slide_no > nslides:
250                warning(f"slide {slide_no} does not exist in PPTX (valid range: 1-{nslides}); skipped.")
251                skipped += 1
252                continue
253
254            manuscript = manuscript_sections.get(slide_no, "")
255            narration = narration_sections.get(slide_no)
256            source_slide = prs.Slides.Item(slide_no)
257
258            if split_lines:
259                manuscript_lines = nonempty_lines(manuscript)
260                narration_lines = nonempty_lines(narration or "")
261                line_count = max(len(manuscript_lines), len(narration_lines))
262                if not line_count:
263                    warning(f"slide {slide_no} contains no content lines; skipped.")
264                    skipped += 1
265                    continue
266                
267                current_slide = source_slide
268                for line_index in range(line_count):
269                    # 2行目以降はスライドを複製(COMのDuplicateは自動的に直後に挿入されます)
270                    if line_index > 0:
271                        current_slide = current_slide.Duplicate().Item(1)
272
273                    manuscript_line = manuscript_lines[line_index] if line_index < len(manuscript_lines) else ""
274                    narration_line = narration_lines[line_index] if line_index < len(narration_lines) else None
275                    warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index + 1)
276                    
277                    note_text = compose_notes(manuscript_line, narration_line)
278                    if not set_slide_notes(current_slide, note_text):
279                        warning(f"slide {slide_no}, line {line_index + 1} has no usable notes placeholder; skipped.")
280                        skipped += 1
281                        continue
282                    if add_subtitles:
283                        add_subtitle(current_slide, prs, manuscript_line, args)
284                    print(f"Updated: slide {slide_no}, line {line_index + 1}")
285                    updated += 1
286            else:
287                manuscript_lines = nonempty_lines(manuscript)
288                narration_lines = nonempty_lines(narration or "")
289                for line_index, (manuscript_line, narration_line) in enumerate(zip(manuscript_lines, narration_lines), start=1):
290                    warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index)
291                
292                note_text = compose_notes(manuscript, narration)
293                if not set_slide_notes(source_slide, note_text):
294                    warning(f"slide {slide_no} has no usable notes placeholder; skipped.")
295                    skipped += 1
296                    continue
297                if add_subtitles:
298                    add_subtitle(source_slide, prs, manuscript, args)
299                print(f"Updated: slide {slide_no}")
300                updated += 1
301
302        if output_path is None:
303            prs.Save()
304            saved_path = pptx_path
305        else:
306            out_abs = str(output_path.absolute())
307            prs.SaveCopyAs(out_abs)
308            saved_path = output_path
309
310        print(f"Saved: {saved_path}")
311        print(f"Updated slides: {updated}")
312        if skipped:
313            print(f"Skipped sections: {skipped}")
314
315    finally:
316        prs.Close()
317        # 他に開いているプレゼンテーションがなければPowerPointを終了
318        if Application.Presentations.Count == 0:
319            Application.Quit()
320
321    return 0
322
323def pptx_to_notes(pptx_path: Path, notes_path: Path) -> int:
324    suffix = notes_path.suffix.lower()
325
326    print("PowerPointを起動中...")
327    Application = win32com.client.Dispatch("PowerPoint.Application")
328    prs = Application.Presentations.Open(str(pptx_path.absolute()), WithWindow=msoFalse)
329    
330    try:
331        nslides = prs.Slides.Count
332
333        if suffix == ".docx":
334            try:
335                from docx import Document
336            except ImportError:
337                raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
338            doc = Document()
339            for slide_no in range(1, nslides + 1):
340                slide = prs.Slides.Item(slide_no)
341                p = doc.add_paragraph(f"# Slide {slide_no}")
342                try: p.style = "Heading 1"
343                except KeyError: pass
344                note_text = get_slide_notes(slide)
345                if note_text:
346                    for line in note_text.split("\n"):
347                        doc.add_paragraph(line)
348                else:
349                    doc.add_paragraph("")
350                doc.add_paragraph("")
351            notes_path.parent.mkdir(parents=True, exist_ok=True)
352            doc.save(notes_path)
353            
354        elif suffix in (".txt", ".md"):
355            blocks: list[str] = []
356            for slide_no in range(1, nslides + 1):
357                slide = prs.Slides.Item(slide_no)
358                note_text = get_slide_notes(slide).strip("\n")
359                block = f"# Slide {slide_no}\n"
360                if note_text:
361                    block += note_text + "\n"
362                blocks.append(block)
363            notes_path.parent.mkdir(parents=True, exist_ok=True)
364            notes_path.write_text("\n".join(blocks), encoding="utf-8")
365        else:
366            raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")
367
368        print(f"Saved: {notes_path}")
369        print(f"Slides exported: {nslides}")
370        
371    finally:
372        prs.Close()
373        if Application.Presentations.Count == 0:
374            Application.Quit()
375            
376    return 0
377
378def build_parser() -> argparse.ArgumentParser:
379    parser = argparse.ArgumentParser(description="Convert PowerPoint speaker notes to/from .docx/.txt/.md files (COM API ver).")
380    parser.add_argument("--mode", required=True, choices=("word2pptx", "pptx2word", "notes2pptx", "pptx2notes"))
381    parser.add_argument("files", nargs="*", metavar="NOTES_FILE")
382    parser.add_argument("-p", "--pptx", required=True)
383    parser.add_argument("-w", "--word", "--notes", dest="notes")
384    parser.add_argument("--split_lines", type=int, choices=(0, 1), default=0, metavar="0|1")
385    parser.add_argument("--add_subtitles", type=int, choices=(0, 1), default=0, metavar="0|1")
386    parser.add_argument("--subtitle_bottom_margin", type=float, default=20)
387    parser.add_argument("--subtitle_box_margin", type=float, default=50)
388    parser.add_argument("--subtitle_box_height", type=float, default=60)
389    parser.add_argument("--subtitle_font_name", default="メイリオ")
390    parser.add_argument("--subtitle_font_size", type=float, default=12)
391    parser.add_argument("--subtitle_font_color", default="884444")
392    parser.add_argument("--subtitle_bgcolor", default="00DDDD")
393    parser.add_argument("--subtitle_bg_transparency", type=float, default=0, metavar="0..100")
394    parser.add_argument("-o", "--output")
395    return parser
396
397def main() -> int:
398    args = build_parser().parse_args()
399    pptx_path = Path(args.pptx).expanduser()
400    if len(args.files) > 2:
401        print("Error: specify at most two positional files.", file=sys.stderr)
402        return 2
403    if args.notes and args.files:
404        print("Error: do not combine --word/--notes with positional files.", file=sys.stderr)
405        return 2
406    if not args.files and not args.notes:
407        print("Error: a manuscript/notes file is required.", file=sys.stderr)
408        return 2
409
410    manuscript_path = Path(args.files[0] if args.files else args.notes).expanduser()
411    narration_path = Path(args.files[1]).expanduser() if len(args.files) == 2 else None
412
413    if args.mode in ("word2pptx", "notes2pptx"):
414        if not pptx_path.is_file():
415            print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
416            return 2
417        for label, path in (("manuscript", manuscript_path), ("narration", narration_path)):
418            if path is None: continue
419            if not path.is_file():
420                print(f"Error: {label} file not found: {path}", file=sys.stderr)
421                return 2
422            if path.suffix.lower() not in (".docx", ".txt", ".md"):
423                print(f"Error: {label} file must have .docx, .txt, or .md extension.", file=sys.stderr)
424                return 2
425
426        output_path = Path(args.output).expanduser() if args.output else None
427        try:
428            return notes_to_pptx(pptx_path, manuscript_path, narration_path, output_path, bool(args.split_lines), bool(args.add_subtitles), args)
429        except (ValueError, RuntimeError, UnicodeError) as exc:
430            print(f"Error: {exc}", file=sys.stderr)
431            return 2
432
433    if args.output: warning("--output is ignored in mode=pptx2word/pptx2notes; --word/--notes is the output file.")
434    if narration_path is not None:
435        print("Error: pptx2word/pptx2notes accepts only one output notes file.", file=sys.stderr)
436        return 2
437
438    if not pptx_path.is_file():
439        print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
440        return 2
441
442    try:
443        return pptx_to_notes(pptx_path, manuscript_path)
444    except (ValueError, RuntimeError, UnicodeError) as exc:
445        print(f"Error: {exc}", file=sys.stderr)
446        return 2
447
448if __name__ == "__main__":
449    raise SystemExit(main())