pptx_notes_word2.py ダウンロード/コピー
pptx_notes_word2.py
pptx_notes_word2.py
1#!/usr/bin/env python3
2# -*- coding: utf-8 -*-
3
4r"""Convert speaker notes between PowerPoint (.pptx) and notes files (.docx/.txt/.md).
5
6【COM API版】
7PowerPoint本体を直接操作(win32com)するため、数式や複雑なグラフが含まれていても
8ファイルが破損することなく安全に処理できます。
9
10Dependencies::
11
12 pip install pywin32
13 pip install python-docx # only needed for .docx files
14"""
15
16from __future__ import annotations
17
18import argparse
19import re
20import sys
21from pathlib import Path
22import win32com.client
23
24SLIDE_HEADER_RE = re.compile(r"^\s*#\s*(?:スライド|Slides?)?\s*(\d+).*$", re.IGNORECASE)
25NARRATION_SEPARATOR = "---\n((\n{narration}\n))"
26SUBTITLE_SHAPE_NAME = "pptx_notes_word subtitle"
27FIXED_SPEAKERS = ("四国めたん", "ずんだもん")
28
29# COM Constants
30msoTextOrientationHorizontal = 1
31msoTrue = -1
32msoFalse = 0
33ppAlignLeft = 1
34msoAnchorMiddle = 3
35msoPlaceholder = 14
36ppPlaceholderBody = 2
37
38def warning(message: str) -> None:
39 print(f"Warning: {message}", file=sys.stderr)
40
41def _parse_note_lines(lines: list[str], source_name: str) -> dict[int, str]:
42 sections: dict[int, str] = {}
43 current_slide: int | None = None
44 current_lines: list[str] = []
45 preamble_lines: list[str] = []
46
47 def store_current() -> None:
48 nonlocal current_slide, current_lines
49 if current_slide is None:
50 return
51 note_text = "\n".join(current_lines).strip("\n")
52 if current_slide in sections:
53 warning(f"slide {current_slide} appears more than once in {source_name}; the last section is used.")
54 sections[current_slide] = note_text
55
56 for text in lines:
57 match = SLIDE_HEADER_RE.match(text)
58 if match:
59 store_current()
60 current_slide = int(match.group(1))
61 current_lines = []
62 continue
63 if current_slide is None:
64 preamble_lines.append(text)
65 else:
66 current_lines.append(text)
67
68 store_current()
69 if any(line.strip() for line in preamble_lines):
70 warning("text before the first slide header was ignored.")
71 if not sections:
72 warning(f"no slide headers were found in {source_name}.")
73 return sections
74
75def read_notes_sections(notes_path: Path) -> dict[int, str]:
76 suffix = notes_path.suffix.lower()
77 if suffix == ".docx":
78 try:
79 from docx import Document
80 except ImportError:
81 raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
82 doc = Document(notes_path)
83 lines = [paragraph.text for paragraph in doc.paragraphs]
84 return _parse_note_lines(lines, "Word file")
85 if suffix in (".txt", ".md"):
86 text = notes_path.read_text(encoding="utf-8-sig")
87 return _parse_note_lines(text.splitlines(), f"{suffix} file")
88 raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")
89
90def get_slide_notes(slide) -> str:
91 notes_page = slide.NotesPage
92 for i in range(1, notes_page.Shapes.Count + 1):
93 shape = notes_page.Shapes.Item(i)
94 if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
95 if shape.HasTextFrame and shape.TextFrame.HasText:
96 return shape.TextFrame.TextRange.Text
97 return ""
98
99def set_slide_notes(slide, text: str) -> bool:
100 notes_page = slide.NotesPage
101 for i in range(1, notes_page.Shapes.Count + 1):
102 shape = notes_page.Shapes.Item(i)
103 if shape.Type == msoPlaceholder and shape.PlaceholderFormat.Type == ppPlaceholderBody:
104 shape.TextFrame.TextRange.Text = text
105 return True
106 return False
107
108def compose_notes(manuscript: str, narration: str | None) -> str:
109 manuscript = manuscript.strip("\n")
110 if narration is None:
111 return manuscript
112
113 # --- 追加: 指定された特定の行だけを除外する処理 ---
114 filtered_lines = []
115 for line in narration.splitlines():
116 stripped = line.strip()
117 # '-'のみ、'='のみ、または '#' で始まる行はスキップ
118 if set(stripped) in ({"-"}, {"="}) or stripped.startswith("#"):
119 continue
120 filtered_lines.append(line)
121
122 # フィルタリング後のテキストを結合
123 filtered_narration = "\n".join(filtered_lines).strip("\n")
124
125 # 有効な読み上げテキストが残らなかった場合は原稿のみを返す
126 if not filtered_narration:
127 return manuscript
128 # ---------------------------------------------------
129
130 narration_block = NARRATION_SEPARATOR.format(narration=filtered_narration)
131 return f"{manuscript}\n{narration_block}" if manuscript else narration_block
132
133"""
134def compose_notes(manuscript: str, narration: str | None) -> str:
135 manuscript = manuscript.strip("\n")
136 if narration is None or not narration.strip():
137 return manuscript
138 narration_block = NARRATION_SEPARATOR.format(narration=narration.strip("\n"))
139 return f"{manuscript}\n{narration_block}" if manuscript else narration_block
140"""
141
142def split_speaker_line(line: str) -> tuple[str | None, str]:
143 speaker_text, separator, body = line.partition(",")
144 speaker = speaker_text.strip()
145 if separator and speaker in FIXED_SPEAKERS:
146 return speaker, body.strip()
147 return None, line.strip()
148
149def subtitle_text(text: str) -> str:
150 formatted: list[str] = []
151 for line in text.splitlines():
152 speaker, body = split_speaker_line(line)
153 formatted.append(f"{speaker}:{body}" if speaker else line)
154 return "\n".join(formatted)
155
156def warn_speaker_mismatch(manuscript_line: str, narration_line: str | None, slide_no: int, line_no: int) -> None:
157 if narration_line is None: return
158 manuscript_speaker, _ = split_speaker_line(manuscript_line)
159 narration_speaker, _ = split_speaker_line(narration_line)
160 if manuscript_speaker and narration_speaker and manuscript_speaker != narration_speaker:
161 warning(f"slide {slide_no}, line {line_no}: manuscript speaker '{manuscript_speaker}' differs from narration speaker '{narration_speaker}'; narration speaker is used for TTS.")
162
163def is_ignored_split_line(line: str) -> bool:
164 stripped = line.strip()
165 if not stripped: return True
166 if set(stripped) in ({"-"}, {"="}): return True
167 if stripped.startswith("#"): return True
168 if stripped in ("((", "))"): return True
169 return False
170
171def nonempty_lines(text: str) -> list[str]:
172 return [line for line in text.splitlines() if not is_ignored_split_line(line)]
173
174def hex_to_bgr(hex_str: str) -> int:
175 """16進数RGB文字列をCOM API用のBGR整数値に変換する"""
176 if not re.fullmatch(r"[0-9A-Fa-f]{6}", hex_str):
177 raise ValueError(f"invalid RGB color: {hex_str}; use six hexadecimal digits")
178 r = int(hex_str[0:2], 16)
179 g = int(hex_str[2:4], 16)
180 b = int(hex_str[4:6], 16)
181 return r + (g << 8) + (b << 16)
182
183def add_subtitle(slide, prs, text: str, args) -> None:
184 text = text.strip("\n")
185 if not text:
186 return
187
188 # 既存の字幕シェイプを削除
189 shapes_to_delete = []
190 for i in range(1, slide.Shapes.Count + 1):
191 shape = slide.Shapes.Item(i)
192 if shape.Name == SUBTITLE_SHAPE_NAME:
193 shapes_to_delete.append(shape)
194 for shape in shapes_to_delete:
195 shape.Delete()
196
197 margin = args.subtitle_box_margin
198 height = args.subtitle_box_height
199 bottom_margin = args.subtitle_bottom_margin
200
201 slide_width = prs.PageSetup.SlideWidth
202 slide_height = prs.PageSetup.SlideHeight
203
204 left = margin
205 top = slide_height - height - bottom_margin
206 width = slide_width - 2 * margin
207
208 shape = slide.Shapes.AddTextbox(msoTextOrientationHorizontal, left, top, width, height)
209 shape.Name = SUBTITLE_SHAPE_NAME
210
211 shape.Fill.Visible = msoTrue
212 shape.Fill.Solid()
213 shape.Fill.ForeColor.RGB = hex_to_bgr(args.subtitle_bgcolor)
214 shape.Fill.Transparency = args.subtitle_bg_transparency / 100.0
215 shape.Line.Visible = msoFalse
216
217 text_frame = shape.TextFrame
218 text_frame.WordWrap = msoTrue
219 text_frame.VerticalAnchor = msoAnchorMiddle
220
221 text_range = text_frame.TextRange
222 text_range.Text = subtitle_text(text)
223 text_range.ParagraphFormat.Alignment = ppAlignLeft
224 text_range.Font.Name = args.subtitle_font_name
225 text_range.Font.Size = args.subtitle_font_size
226 text_range.Font.Color.RGB = hex_to_bgr(args.subtitle_font_color)
227
228def notes_to_pptx(pptx_path: Path, manuscript_path: Path, narration_path: Path | None, output_path: Path | None, split_lines: bool, add_subtitles: bool, args) -> int:
229 manuscript_sections = read_notes_sections(manuscript_path)
230 narration_sections = read_notes_sections(narration_path) if narration_path is not None else {}
231
232 all_slide_numbers = sorted(list(set(list(manuscript_sections) + list(narration_sections))), reverse=True)
233 if not all_slide_numbers:
234 return 0
235
236 print("PowerPointを起動中...")
237 Application = win32com.client.Dispatch("PowerPoint.Application")
238 # PPTを背後で処理するために開く
239 prs = Application.Presentations.Open(str(pptx_path.absolute()))
240
241 updated = 0
242 skipped = 0
243
244 try:
245 nslides = prs.Slides.Count
246
247 # 後ろのスライドから順に処理(スライドを複製しても前のスライドのインデックスがずれないようにするため)
248 for slide_no in all_slide_numbers:
249 if slide_no < 1 or slide_no > nslides:
250 warning(f"slide {slide_no} does not exist in PPTX (valid range: 1-{nslides}); skipped.")
251 skipped += 1
252 continue
253
254 manuscript = manuscript_sections.get(slide_no, "")
255 narration = narration_sections.get(slide_no)
256 source_slide = prs.Slides.Item(slide_no)
257
258 if split_lines:
259 manuscript_lines = nonempty_lines(manuscript)
260 narration_lines = nonempty_lines(narration or "")
261 line_count = max(len(manuscript_lines), len(narration_lines))
262 if not line_count:
263 warning(f"slide {slide_no} contains no content lines; skipped.")
264 skipped += 1
265 continue
266
267 current_slide = source_slide
268 for line_index in range(line_count):
269 # 2行目以降はスライドを複製(COMのDuplicateは自動的に直後に挿入されます)
270 if line_index > 0:
271 current_slide = current_slide.Duplicate().Item(1)
272
273 manuscript_line = manuscript_lines[line_index] if line_index < len(manuscript_lines) else ""
274 narration_line = narration_lines[line_index] if line_index < len(narration_lines) else None
275 warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index + 1)
276
277 note_text = compose_notes(manuscript_line, narration_line)
278 if not set_slide_notes(current_slide, note_text):
279 warning(f"slide {slide_no}, line {line_index + 1} has no usable notes placeholder; skipped.")
280 skipped += 1
281 continue
282 if add_subtitles:
283 add_subtitle(current_slide, prs, manuscript_line, args)
284 print(f"Updated: slide {slide_no}, line {line_index + 1}")
285 updated += 1
286 else:
287 manuscript_lines = nonempty_lines(manuscript)
288 narration_lines = nonempty_lines(narration or "")
289 for line_index, (manuscript_line, narration_line) in enumerate(zip(manuscript_lines, narration_lines), start=1):
290 warn_speaker_mismatch(manuscript_line, narration_line, slide_no, line_index)
291
292 note_text = compose_notes(manuscript, narration)
293 if not set_slide_notes(source_slide, note_text):
294 warning(f"slide {slide_no} has no usable notes placeholder; skipped.")
295 skipped += 1
296 continue
297 if add_subtitles:
298 add_subtitle(source_slide, prs, manuscript, args)
299 print(f"Updated: slide {slide_no}")
300 updated += 1
301
302 if output_path is None:
303 prs.Save()
304 saved_path = pptx_path
305 else:
306 out_abs = str(output_path.absolute())
307 prs.SaveCopyAs(out_abs)
308 saved_path = output_path
309
310 print(f"Saved: {saved_path}")
311 print(f"Updated slides: {updated}")
312 if skipped:
313 print(f"Skipped sections: {skipped}")
314
315 finally:
316 prs.Close()
317 # 他に開いているプレゼンテーションがなければPowerPointを終了
318 if Application.Presentations.Count == 0:
319 Application.Quit()
320
321 return 0
322
323def pptx_to_notes(pptx_path: Path, notes_path: Path) -> int:
324 suffix = notes_path.suffix.lower()
325
326 print("PowerPointを起動中...")
327 Application = win32com.client.Dispatch("PowerPoint.Application")
328 prs = Application.Presentations.Open(str(pptx_path.absolute()), WithWindow=msoFalse)
329
330 try:
331 nslides = prs.Slides.Count
332
333 if suffix == ".docx":
334 try:
335 from docx import Document
336 except ImportError:
337 raise RuntimeError("python-docx is required for .docx files. Install it with: pip install python-docx") from None
338 doc = Document()
339 for slide_no in range(1, nslides + 1):
340 slide = prs.Slides.Item(slide_no)
341 p = doc.add_paragraph(f"# Slide {slide_no}")
342 try: p.style = "Heading 1"
343 except KeyError: pass
344 note_text = get_slide_notes(slide)
345 if note_text:
346 for line in note_text.split("\n"):
347 doc.add_paragraph(line)
348 else:
349 doc.add_paragraph("")
350 doc.add_paragraph("")
351 notes_path.parent.mkdir(parents=True, exist_ok=True)
352 doc.save(notes_path)
353
354 elif suffix in (".txt", ".md"):
355 blocks: list[str] = []
356 for slide_no in range(1, nslides + 1):
357 slide = prs.Slides.Item(slide_no)
358 note_text = get_slide_notes(slide).strip("\n")
359 block = f"# Slide {slide_no}\n"
360 if note_text:
361 block += note_text + "\n"
362 blocks.append(block)
363 notes_path.parent.mkdir(parents=True, exist_ok=True)
364 notes_path.write_text("\n".join(blocks), encoding="utf-8")
365 else:
366 raise ValueError(f"unsupported notes file extension: {notes_path.suffix or '(none)'}; use .docx, .txt, or .md")
367
368 print(f"Saved: {notes_path}")
369 print(f"Slides exported: {nslides}")
370
371 finally:
372 prs.Close()
373 if Application.Presentations.Count == 0:
374 Application.Quit()
375
376 return 0
377
378def build_parser() -> argparse.ArgumentParser:
379 parser = argparse.ArgumentParser(description="Convert PowerPoint speaker notes to/from .docx/.txt/.md files (COM API ver).")
380 parser.add_argument("--mode", required=True, choices=("word2pptx", "pptx2word", "notes2pptx", "pptx2notes"))
381 parser.add_argument("files", nargs="*", metavar="NOTES_FILE")
382 parser.add_argument("-p", "--pptx", required=True)
383 parser.add_argument("-w", "--word", "--notes", dest="notes")
384 parser.add_argument("--split_lines", type=int, choices=(0, 1), default=0, metavar="0|1")
385 parser.add_argument("--add_subtitles", type=int, choices=(0, 1), default=0, metavar="0|1")
386 parser.add_argument("--subtitle_bottom_margin", type=float, default=20)
387 parser.add_argument("--subtitle_box_margin", type=float, default=50)
388 parser.add_argument("--subtitle_box_height", type=float, default=60)
389 parser.add_argument("--subtitle_font_name", default="メイリオ")
390 parser.add_argument("--subtitle_font_size", type=float, default=12)
391 parser.add_argument("--subtitle_font_color", default="884444")
392 parser.add_argument("--subtitle_bgcolor", default="00DDDD")
393 parser.add_argument("--subtitle_bg_transparency", type=float, default=0, metavar="0..100")
394 parser.add_argument("-o", "--output")
395 return parser
396
397def main() -> int:
398 args = build_parser().parse_args()
399 pptx_path = Path(args.pptx).expanduser()
400 if len(args.files) > 2:
401 print("Error: specify at most two positional files.", file=sys.stderr)
402 return 2
403 if args.notes and args.files:
404 print("Error: do not combine --word/--notes with positional files.", file=sys.stderr)
405 return 2
406 if not args.files and not args.notes:
407 print("Error: a manuscript/notes file is required.", file=sys.stderr)
408 return 2
409
410 manuscript_path = Path(args.files[0] if args.files else args.notes).expanduser()
411 narration_path = Path(args.files[1]).expanduser() if len(args.files) == 2 else None
412
413 if args.mode in ("word2pptx", "notes2pptx"):
414 if not pptx_path.is_file():
415 print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
416 return 2
417 for label, path in (("manuscript", manuscript_path), ("narration", narration_path)):
418 if path is None: continue
419 if not path.is_file():
420 print(f"Error: {label} file not found: {path}", file=sys.stderr)
421 return 2
422 if path.suffix.lower() not in (".docx", ".txt", ".md"):
423 print(f"Error: {label} file must have .docx, .txt, or .md extension.", file=sys.stderr)
424 return 2
425
426 output_path = Path(args.output).expanduser() if args.output else None
427 try:
428 return notes_to_pptx(pptx_path, manuscript_path, narration_path, output_path, bool(args.split_lines), bool(args.add_subtitles), args)
429 except (ValueError, RuntimeError, UnicodeError) as exc:
430 print(f"Error: {exc}", file=sys.stderr)
431 return 2
432
433 if args.output: warning("--output is ignored in mode=pptx2word/pptx2notes; --word/--notes is the output file.")
434 if narration_path is not None:
435 print("Error: pptx2word/pptx2notes accepts only one output notes file.", file=sys.stderr)
436 return 2
437
438 if not pptx_path.is_file():
439 print(f"Error: PPTX file not found: {pptx_path}", file=sys.stderr)
440 return 2
441
442 try:
443 return pptx_to_notes(pptx_path, manuscript_path)
444 except (ValueError, RuntimeError, UnicodeError) as exc:
445 print(f"Error: {exc}", file=sys.stderr)
446 return 2
447
448if __name__ == "__main__":
449 raise SystemExit(main())