pptx_notes_word.py ダウンロード/コピー

pptx_notes_word.py をダウンロード

pptx_notes_word.py
pptx_notes_word.py
  1#!/usr/bin/env python3
  2# -*- coding: utf-8 -*-
  3
  4r"""
  5概要:
  6    PowerPointのpptxファイルとノートファイル間のノートテキストを変換します。
  7詳細説明:
  8    word2pptxモードでは、docxファイルやtxtファイル、mdファイルからスライドセクションを読み込み、
  9    対応するPowerPointスライドのノートを上書きします。
 10    pptx2wordモードでは、PowerPointスライドのすべてのノートを抽出し、
 11    docxファイルやtxtファイル、mdファイルに保存します。
 12    スライドのヘッダーは特定のパターンで認識され、スライド 3 や Slide 3 のような書式に対応します。
 13    依存関係としてpython-pptxが必要で、docxファイルを扱う場合はpython-docxが必要です。
 14"""
 15
 16from __future__ import annotations
 17
 18import argparse
 19from copy import deepcopy
 20import re
 21import shutil
 22import sys
 23from pathlib import Path
 24
 25from pptx import Presentation
 26
 27
 28SLIDE_HEADER_RE = re.compile(
 29    r"^\s*#\s*(?:スライド|Slides?)?\s*(\d+)\s*$",
 30    re.IGNORECASE,
 31)
 32
 33NARRATION_SEPARATOR = "---\n((\n{narration}\n))"
 34
 35
 36def warning(message: str) -> None:
 37    """
 38    概要:
 39        警告メッセージを標準エラー出力に出力します。
 40    引数:
 41        :param message: 警告として出力するメッセージテキスト。
 42        :type message: str
 43    戻り値:
 44        :returns: なし。
 45        :rtype: None
 46    """
 47    print(f"Warning: {message}", file=sys.stderr)
 48
 49
 50def _parse_note_lines(lines: list[str], source_name: str) -> dict[int, str]:
 51    """
 52    概要:
 53        テキスト行のリストからスライド番号ごとのノートセクションを解析します。
 54    引数:
 55        :param lines: 解析対象となるテキスト行のリスト。
 56        :type lines: list[str]
 57        :param source_name: 警告メッセージに使用する入力元の名前。
 58        :type source_name: str
 59    戻り値:
 60        :returns: スライド番号をキー、ノートテキストを値とする辞書。
 61        :rtype: dict[int, str]
 62    """
 63    sections: dict[int, str] = {}
 64    current_slide: int | None = None
 65    current_lines: list[str] = []
 66    preamble_lines: list[str] = []
 67
 68    def store_current() -> None:
 69        nonlocal current_slide, current_lines
 70        if current_slide is None:
 71            return
 72
 73        note_text = "\n".join(current_lines).strip("\n")
 74        if current_slide in sections:
 75            warning(
 76                f"slide {current_slide} appears more than once in {source_name}; "
 77                "the last section is used."
 78            )
 79        sections[current_slide] = note_text
 80
 81    for text in lines:
 82        match = SLIDE_HEADER_RE.match(text)
 83
 84        if match:
 85            store_current()
 86            current_slide = int(match.group(1))
 87            current_lines = []
 88            continue
 89
 90        if current_slide is None:
 91            preamble_lines.append(text)
 92        else:
 93            current_lines.append(text)
 94
 95    store_current()
 96
 97    if any(line.strip() for line in preamble_lines):
 98        warning("text before the first slide header was ignored.")
 99
100    if not sections:
101        warning(f"no slide headers were found in {source_name}.")
102
103    return sections
104
105
106def read_notes_sections(notes_path: Path) -> dict[int, str]:
107    """
108    概要:
109        docxやtxt、mdファイルからスライド番号ごとのノートセクションを読み込みます。
110    引数:
111        :param notes_path: 読み込むノートファイルのパス。
112        :type notes_path: pathlib.Path
113    戻り値:
114        :returns: スライド番号をキー、ノートテキストを値とする辞書。
115        :rtype: dict[int, str]
116    例外:
117        :raises RuntimeError: python-docxパッケージがインストールされていない場合。
118        :raises ValueError: サポートされていないファイルの拡張子の場合。
119    """
120    suffix = notes_path.suffix.lower()
121
122    if suffix == ".docx":
123        try:
124            from docx import Document
125        except ImportError:
126            raise RuntimeError(
127                "python-docx is required for .docx files. "
128                "Install it with: pip install python-docx"
129            ) from None
130
131        doc = Document(notes_path)
132        lines = [paragraph.text for paragraph in doc.paragraphs]
133        return _parse_note_lines(lines, "Word file")
134
135    if suffix in (".txt", ".md"):
136        text = notes_path.read_text(encoding="utf-8-sig")
137        return _parse_note_lines(text.splitlines(), f"{suffix} file")
138
139    raise ValueError(
140        f"unsupported notes file extension: {notes_path.suffix or '(none)'}; "
141        "use .docx, .txt, or .md"
142    )
143
144
145def get_slide_notes(slide) -> str:
146    """
147    概要:
148        ノートスライドを作成せずに1つのスライドのノートテキストを返します。
149    引数:
150        :param slide: 読み込み対象のスライドオブジェクト。
151        :type slide: pptx.slide.Slide
152    戻り値:
153        :returns: ノートのテキスト文字列。
154        :rtype: str
155    """
156    if not slide.has_notes_slide:
157        return ""
158
159    text_frame = slide.notes_slide.notes_text_frame
160    if text_frame is None:
161        return ""
162
163    return text_frame.text or ""
164
165
166def set_slide_notes(slide, text: str) -> bool:
167    """
168    概要:
169        1つのスライドのノートテキストを上書きします。
170    詳細説明:
171        ノートのプレースホルダーが利用できない場合のみFalseを返します。
172    引数:
173        :param slide: 書き込み対象のスライドオブジェクト。
174        :type slide: pptx.slide.Slide
175        :param text: 上書きするテキスト。
176        :type text: str
177    戻り値:
178        :returns: 成功した場合はTrue、プレースホルダーがない場合はFalse。
179        :rtype: bool
180    """
181    text_frame = slide.notes_slide.notes_text_frame
182    if text_frame is None:
183        return False
184
185    text_frame.text = text
186    return True
187
188
189def compose_notes(manuscript: str, narration: str | None) -> str:
190    """
191    概要:
192        指定された区切り文字を使用して原稿とナレーションを結合します。
193    引数:
194        :param manuscript: 原稿のテキスト。
195        :type manuscript: str
196        :param narration: ナレーションのテキスト。
197        :type narration: str | None
198    戻り値:
199        :returns: 結合されたノートテキスト。
200        :rtype: str
201    """
202    manuscript = manuscript.strip("\n")
203    if narration is None or not narration.strip():
204        return manuscript
205
206    narration_block = NARRATION_SEPARATOR.format(narration=narration.strip("\n"))
207    return f"{manuscript}\n{narration_block}" if manuscript else narration_block
208
209
210def nonempty_lines(text: str) -> list[str]:
211    """
212    概要:
213        空行を無視してテキストを行のリストに分割します。
214    引数:
215        :param text: 分割対象のテキスト。
216        :type text: str
217    戻り値:
218        :returns: 空でない行のリスト。
219        :rtype: list[str]
220    """
221    return [line for line in text.splitlines() if line.strip()]
222
223
224def duplicate_slide(prs: Presentation, source_slide):
225    """
226    概要:
227        指定されたスライドのコピーを追加し、新しいスライドを返します。
228    詳細説明:
229        python-pptxにはスライドをコピーする公開APIがないため、
230        図形と外部関係を同じレイアウトの新しいスライドにコピーします。
231    引数:
232        :param prs: プレゼンテーションオブジェクト。
233        :type prs: pptx.presentation.Presentation
234        :param source_slide: コピー元のスライド。
235        :type source_slide: pptx.slide.Slide
236    戻り値:
237        :returns: コピーして作成された新しいスライドオブジェクト。
238        :rtype: pptx.slide.Slide
239    """
240    new_slide = prs.slides.add_slide(source_slide.slide_layout)
241
242    # Remove placeholders automatically supplied by the layout before copying.
243    for shape in list(new_slide.shapes):
244        new_slide.shapes._spTree.remove(shape.element)
245
246    for shape in source_slide.shapes:
247        new_slide.shapes._spTree.insert_element_before(
248            deepcopy(shape.element), "p:extLst"
249        )
250
251    # Copy relationships used by pictures, hyperlinks, charts, and other
252    # slide content. Layout and notes relationships belong to the new slide.
253    excluded = {
254        "http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide",
255        "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slideLayout",
256    }
257    relationship_ids: dict[str, str] = {}
258    for rel in source_slide.part.rels.values():
259        if rel.reltype in excluded:
260            continue
261        relationship_ids[rel.rId] = new_slide.part.rels._add_relationship(
262            rel.reltype, rel._target, rel.is_external
263        )
264
265    # Copied shape XML still contains the source slide's relationship IDs.
266    for element in new_slide.shapes._spTree.iter():
267        for attribute, value in list(element.attrib.items()):
268            if value in relationship_ids:
269                element.set(attribute, relationship_ids[value])
270
271    return new_slide
272
273
274def move_slide_after(prs: Presentation, slide, after_slide) -> None:
275    """
276    概要:
277        スライドをプレゼンテーション内の別のスライドの直後に移動します。
278    引数:
279        :param prs: プレゼンテーションオブジェクト。
280        :type prs: pptx.presentation.Presentation
281        :param slide: 移動するスライドオブジェクト。
282        :type slide: pptx.slide.Slide
283        :param after_slide: このスライドの直後に移動します。
284        :type after_slide: pptx.slide.Slide
285    戻り値:
286        :returns: なし。
287        :rtype: None
288    例外:
289        :raises RuntimeError: スライドの順序を特定できなかった場合。
290    """
291    slide_ids = prs.slides._sldIdLst
292    # Locate IDs through their related slide parts; rId values cannot be
293    # inferred from part names.
294    target_id = None
295    after_id = None
296    for item in slide_ids:
297        related_part = prs.part.related_part(item.rId)
298        if related_part is slide.part:
299            target_id = item
300        if related_part is after_slide.part:
301            after_id = item
302    if target_id is None or after_id is None:
303        raise RuntimeError("could not determine slide order while copying a slide")
304    slide_ids.remove(target_id)
305    slide_ids.insert(slide_ids.index(after_id) + 1, target_id)
306
307
308def notes_to_pptx(
309    pptx_path: Path,
310    manuscript_path: Path,
311    narration_path: Path | None,
312    output_path: Path | None,
313    split_lines: bool,
314) -> int:
315    """
316    概要:
317        原稿とナレーションのノートを読み込み、pptxファイルのノートに反映させます。
318    引数:
319        :param pptx_path: 対象のpptxファイルのパス。
320        :type pptx_path: pathlib.Path
321        :param manuscript_path: 原稿ファイルのパス。
322        :type manuscript_path: pathlib.Path
323        :param narration_path: ナレーションファイルのパス、またはNone。
324        :type narration_path: pathlib.Path | None
325        :param output_path: 出力するpptxファイルのパス、またはNone。
326        :type output_path: pathlib.Path | None
327        :param split_lines: 行ごとにスライドを分割するかどうかのフラグ。
328        :type split_lines: bool
329    戻り値:
330        :returns: 終了ステータスコード。
331        :rtype: int
332    """
333    manuscript_sections = read_notes_sections(manuscript_path)
334    narration_sections = (
335        read_notes_sections(narration_path) if narration_path is not None else {}
336    )
337    prs = Presentation(pptx_path)
338    nslides = len(prs.slides)
339
340    updated = 0
341    skipped = 0
342
343    original_slides = list(prs.slides)
344    all_slide_numbers = list(manuscript_sections)
345    for slide_no in narration_sections:
346        if slide_no not in manuscript_sections:
347            all_slide_numbers.append(slide_no)
348
349    for slide_no in all_slide_numbers:
350        if slide_no < 1 or slide_no > nslides:
351            warning(
352                f"slide {slide_no} does not exist in PPTX "
353                f"(valid range: 1-{nslides}); skipped."
354            )
355            skipped += 1
356            continue
357
358        manuscript = manuscript_sections.get(slide_no, "")
359        narration = narration_sections.get(slide_no)
360        source_slide = original_slides[slide_no - 1]
361
362        if split_lines:
363            manuscript_lines = nonempty_lines(manuscript)
364            narration_lines = nonempty_lines(narration or "")
365            line_count = max(len(manuscript_lines), len(narration_lines))
366            if not line_count:
367                warning(f"slide {slide_no} contains no nonblank lines; skipped.")
368                skipped += 1
369                continue
370            if narration_path is not None and len(manuscript_lines) != len(narration_lines):
371                warning(
372                    f"slide {slide_no}: manuscript has {len(manuscript_lines)} "
373                    f"nonblank lines, narration has {len(narration_lines)}; "
374                    "missing lines are left empty."
375                )
376
377            current_slide = source_slide
378            for line_index in range(line_count):
379                if line_index:
380                    copied_slide = duplicate_slide(prs, source_slide)
381                    move_slide_after(prs, copied_slide, current_slide)
382                    current_slide = copied_slide
383                manuscript_line = (
384                    manuscript_lines[line_index]
385                    if line_index < len(manuscript_lines) else ""
386                )
387                narration_line = (
388                    narration_lines[line_index]
389                    if line_index < len(narration_lines) else None
390                )
391                if not set_slide_notes(
392                    current_slide, compose_notes(manuscript_line, narration_line)
393                ):
394                    warning(
395                        f"slide {slide_no}, line {line_index + 1} has no usable "
396                        "notes placeholder; skipped."
397                    )
398                    skipped += 1
399                    continue
400                print(f"Updated: slide {slide_no}, line {line_index + 1}")
401                updated += 1
402        else:
403            note_text = compose_notes(manuscript, narration)
404            if not set_slide_notes(source_slide, note_text):
405                warning(f"slide {slide_no} has no usable notes placeholder; skipped.")
406                skipped += 1
407                continue
408            print(f"Updated: slide {slide_no}")
409            updated += 1
410
411    if output_path is None:
412        output_path = pptx_path
413
414    # python-pptx writes a complete package. Saving directly over the input
415    # file is supported, but a temporary file makes replacement safer.
416    if output_path.resolve() == pptx_path.resolve():
417        tmp_path = pptx_path.with_name(pptx_path.stem + ".__notes_tmp__.pptx")
418        prs.save(tmp_path)
419        shutil.move(tmp_path, pptx_path)
420    else:
421        output_path.parent.mkdir(parents=True, exist_ok=True)
422        prs.save(output_path)
423
424    print(f"Saved: {output_path}")
425    print(f"Updated slides: {updated}")
426    if skipped:
427        print(f"Skipped sections: {skipped}")
428
429    return 0
430
431
432def pptx_to_notes(pptx_path: Path, notes_path: Path) -> int:
433    """
434    概要:
435        pptxファイルからノートを抽出し、docxやtxt、mdファイルにエクスポートします。
436    引数:
437        :param pptx_path: 入力元のpptxファイルのパス。
438        :type pptx_path: pathlib.Path
439        :param notes_path: 出力先のノートファイルのパス。
440        :type notes_path: pathlib.Path
441    戻り値:
442        :returns: 終了ステータスコード。
443        :rtype: int
444    例外:
445        :raises RuntimeError: docx出力でpython-docxがない場合。
446        :raises ValueError: サポートされていないファイルの拡張子の場合。
447    """
448    prs = Presentation(pptx_path)
449    suffix = notes_path.suffix.lower()
450
451    if suffix == ".docx":
452        try:
453            from docx import Document
454        except ImportError:
455            raise RuntimeError(
456                "python-docx is required for .docx files. "
457                "Install it with: pip install python-docx"
458            ) from None
459
460        doc = Document()
461
462        for slide_no, slide in enumerate(prs.slides, start=1):
463            # Keep the literal marker in the paragraph text so the generated
464            # file can be fed back to notes2pptx/word2pptx unchanged.
465            p = doc.add_paragraph(f"# Slide {slide_no}")
466            try:
467                p.style = "Heading 1"
468            except KeyError:
469                pass
470
471            note_text = get_slide_notes(slide)
472            if note_text:
473                for line in note_text.split("\n"):
474                    doc.add_paragraph(line)
475            else:
476                doc.add_paragraph("")
477
478            doc.add_paragraph("")
479
480        notes_path.parent.mkdir(parents=True, exist_ok=True)
481        doc.save(notes_path)
482
483    elif suffix in (".txt", ".md"):
484        blocks: list[str] = []
485        for slide_no, slide in enumerate(prs.slides, start=1):
486            note_text = get_slide_notes(slide).strip("\n")
487            block = f"# Slide {slide_no}\n"
488            if note_text:
489                block += note_text + "\n"
490            blocks.append(block)
491
492        notes_path.parent.mkdir(parents=True, exist_ok=True)
493        notes_path.write_text("\n".join(blocks), encoding="utf-8")
494
495    else:
496        raise ValueError(
497            f"unsupported notes file extension: {notes_path.suffix or '(none)'}; "
498            "use .docx, .txt, or .md"
499        )
500
501    print(f"Saved: {notes_path}")
502    print(f"Slides exported: {len(prs.slides)}")
503    return 0
504
505
506def build_parser() -> argparse.ArgumentParser:
507    """
508    概要:
509        コマンドライン引数のパーサーを構築します。
510    戻り値:
511        :returns: 設定済みの引数パーサーオブジェクト。
512        :rtype: argparse.ArgumentParser
513    """
514    parser = argparse.ArgumentParser(
515        description="Convert PowerPoint speaker notes to/from .docx/.txt/.md files."
516    )
517    parser.add_argument(
518        "--mode",
519        required=True,
520        choices=("word2pptx", "pptx2word", "notes2pptx", "pptx2notes"),
521        help="Conversion direction.",
522    )
523    parser.add_argument(
524        "files",
525        nargs="*",
526        metavar="NOTES_FILE",
527        help=(
528            "Manuscript file followed optionally by a narration file "
529            "(.docx, .txt, or .md)."
530        ),
531    )
532    parser.add_argument(
533        "-p",
534        "--pptx",
535        required=True,
536        help="PowerPoint .pptx file.",
537    )
538    parser.add_argument(
539        "-w",
540        "--word",
541        "--notes",
542        dest="notes",
543        help="Legacy alternative to the manuscript positional argument.",
544    )
545    parser.add_argument(
546        "--split_lines",
547        type=lambda value: value.lower() in ("true", "1", "yes", "on"),
548        default=False,
549        metavar="BOOL",
550        help="For word2pptx, copy each slide once per nonblank line (default: False).",
551    )
552    parser.add_argument(
553        "-o",
554        "--output",
555        help=(
556            "Output .pptx for mode=word2pptx/notes2pptx. "
557            "If omitted, the input PPTX is overwritten. "
558            "Ignored in mode=pptx2word/pptx2notes."
559        ),
560    )
561    return parser
562
563
564def main() -> int:
565    """
566    概要:
567        コマンドラインからの実行を処理するメイン関数です。
568    戻り値:
569        :returns: 終了ステータスコード。
570        :rtype: int
571    """
572    args = build_parser().parse_args()
573
574    pptx_path = Path(args.pptx).expanduser()
575    if len(args.files) > 2:
576        print("Error: specify at most two positional files.", file=sys.stderr)
577        return 2
578    if args.notes and args.files:
579        print("Error: do not combine --word/--notes with positional files.", file=sys.stderr)
580        return 2
581    if not args.files and not args.notes:
582        print("Error: a manuscript/notes file is required.", file=sys.stderr)
583        return 2
584
585    manuscript_path = Path(args.files[0] if args.files else args.notes).expanduser()
586    narration_path = (
587        Path(args.files[1]).expanduser() if len(args.files) == 2 else None
588    )
589
590    if args.mode in ("word2pptx", "notes2pptx"):
591        if not pptx_path.is_file():
592            print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
593            return 2
594        for label, path in (("manuscript", manuscript_path), ("narration", narration_path)):
595            if path is None:
596                continue
597            if not path.is_file():
598                print(f"Error: {label} file not found: {path}", file=sys.stderr)
599                return 2
600            if path.suffix.lower() not in (".docx", ".txt", ".md"):
601                print(f"Error: {label} file must have .docx, .txt, or .md extension.", file=sys.stderr)
602                return 2
603
604        output_path = Path(args.output).expanduser() if args.output else None
605        if output_path is not None and output_path.suffix.lower() != ".pptx":
606            print("Error: output file must have .pptx extension.", file=sys.stderr)
607            return 2
608
609        try:
610            return notes_to_pptx(
611                pptx_path, manuscript_path, narration_path, output_path,
612                args.split_lines,
613            )
614        except (ValueError, RuntimeError, UnicodeError) as exc:
615            print(f"Error: {exc}", file=sys.stderr)
616            return 2
617
618    if args.output:
619        warning("--output is ignored in mode=pptx2word/pptx2notes; --word/--notes is the output file.")
620    if narration_path is not None:
621        print("Error: pptx2word/pptx2notes accepts only one output notes file.", file=sys.stderr)
622        return 2
623    if args.split_lines:
624        warning("--split_lines is ignored in mode=pptx2word/pptx2notes.")
625
626    if not pptx_path.is_file():
627        print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
628        return 2
629
630    if manuscript_path.suffix.lower() not in (".docx", ".txt", ".md"):
631        print("Error: notes file must have .docx, .txt, or .md extension.", file=sys.stderr)
632        return 2
633
634    try:
635        return pptx_to_notes(pptx_path, manuscript_path)
636    except (ValueError, RuntimeError, UnicodeError) as exc:
637        print(f"Error: {exc}", file=sys.stderr)
638        return 2
639
640
641if __name__ == "__main__":
642    raise SystemExit(main())