make_programs_index.py ダウンロード/コピー

make_programs_index.py をダウンロード

make_programs_index.py
make_programs_index.py
  1#!/usr/bin/env python3
  2"""
  3概要:
  4    Sphinxのusage.mdファイルからコンパクトなプログラムインデックスを作成します。
  5詳細説明:
  6    各入力ドキュメントから最初のレベル2見出しの本文を抽出します。
  7    プログラムのパスは、最初の見出し1にconvert2md.pyのようなファイル名が含まれている場合はそこから取得し、
  8    それ以外の場合はusageファイルの相対パスを安全なフォールバックとして使用します。
  9"""
 10
 11from __future__ import annotations
 12
 13import argparse
 14import fnmatch
 15import glob
 16import html
 17import os
 18import re
 19import sys
 20from pathlib import Path
 21
 22
 23USAGE_GLOB = "*_usage.md"
 24HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$", re.MULTILINE)
 25# Sphinx/MyST's generated heading-anchor link, e.g. [](... "Link to this heading")
 26SPHINX_ANCHOR_RE = re.compile(r"\s*\[\]\([^)]*(?:\"Link to this heading\")?\)")
 27FILENAME_RE = re.compile(r"`([^`/\\]+\.(?:py|pl|sh|bat|ps1|exe))`|\b([^\s`/\\]+\.(?:py|pl|sh|bat|ps1|exe))\b", re.I)
 28LAUNCHER_PROGRAM_RE = re.compile(r"^\s*([^\s/\\]+\.py)\s*$", re.IGNORECASE)
 29LAUNCHER_CONTEXT_RE = re.compile(r"^\s{2,}(.+?)\s*$")
 30
 31
 32def clean_heading(text: str) -> str:
 33    """
 34    概要:
 35        見出しからSphinxのパーマリンク装飾を削除します。
 36    引数:
 37        :param text: 処理対象の見出しテキスト
 38        :type text: str
 39    戻り値:
 40        :returns: 装飾が削除されたテキスト
 41        :rtype: str
 42    """
 43    return SPHINX_ANCHOR_RE.sub("", text).strip()
 44
 45
 46def first_h1_filename(markdown: str) -> str | None:
 47    """
 48    概要:
 49        最初のレベル1見出しからプログラムのファイル名を抽出して返します。
 50    引数:
 51        :param markdown: 解析対象のマークダウンテキスト
 52        :type markdown: str
 53    戻り値:
 54        :returns: 抽出されたファイル名文字列、見つからない場合はNone
 55        :rtype: str | None
 56    """
 57    for match in HEADING_RE.finditer(markdown):
 58        if len(match.group(1)) != 1:
 59            continue
 60        found = FILENAME_RE.search(clean_heading(match.group(2)))
 61        if found:
 62            return next(part for part in found.groups() if part is not None)
 63        return None
 64    return None
 65
 66
 67def first_level2_body(markdown: str) -> str | None:
 68    """
 69    概要:
 70        最初のレベル2見出しのセクションを抽出し、そのサブセクションも保持して返します。
 71    引数:
 72        :param markdown: 解析対象のマークダウンテキスト
 73        :type markdown: str
 74    戻り値:
 75        :returns: 抽出されたセクションのテキスト、見つからない場合はNone
 76        :rtype: str | None
 77    """
 78    headings = list(HEADING_RE.finditer(markdown))
 79    start = next((i for i, item in enumerate(headings) if len(item.group(1)) == 2), None)
 80    if start is None:
 81        return None
 82
 83    section_start = headings[start].end()
 84    section_end = len(markdown)
 85    for item in headings[start + 1 :]:
 86        # ``###`` etc. belong to this section; another ``#`` or ``##`` does not.
 87        if len(item.group(1)) <= 2:
 88            section_end = item.start()
 89            break
 90    return markdown[section_start:section_end].strip()
 91
 92
 93def program_label(usage_path: Path, markdown: str, display_root: Path) -> str:
 94    """
 95    概要:
 96        display_rootを基準とした表示用のプログラムパスを構築します。
 97    引数:
 98        :param usage_path: usageファイルのパス
 99        :type usage_path: pathlib.Path
100        :param markdown: 解析対象のマークダウンテキスト
101        :type markdown: str
102        :param display_root: 相対パスの基準となるディレクトリ
103        :type display_root: pathlib.Path
104    戻り値:
105        :returns: 構築されたプログラムパスの文字列表現
106        :rtype: str
107    """
108    filename = first_h1_filename(markdown)
109    if filename:
110        candidate = usage_path.parent / filename
111    else:
112        candidate = usage_path
113    return Path(os.path.relpath(candidate, start=display_root)).as_posix()
114
115
116def inline_html(text: str) -> str:
117    """
118    概要:
119        依存関係なしに説明文中で使用されるインラインのマークダウンをHTMLに変換します。
120    引数:
121        :param text: 変換対象のテキスト
122        :type text: str
123    戻り値:
124        :returns: HTMLに変換されたテキスト
125        :rtype: str
126    """
127    escaped = html.escape(text)
128    escaped = re.sub(r"`([^`]+)`", r"<code>\1</code>", escaped)
129    escaped = re.sub(r"\*\*([^*]+)\*\*", r"<strong>\1</strong>", escaped)
130    return escaped
131
132
133def markdown_body_to_html(markdown: str) -> str:
134    """
135    概要:
136        一般的なマークダウンの説明文を自己完結型の小さなHTMLフラグメントに変換します。
137    引数:
138        :param markdown: 変換対象のマークダウンテキスト
139        :type markdown: str
140    戻り値:
141        :returns: 変換されたHTML文字列
142        :rtype: str
143    """
144    output: list[str] = []
145    paragraph: list[str] = []
146    list_tag: str | None = None
147
148    def flush_paragraph() -> None:
149        """
150        概要:
151            蓄積された段落の内容をHTMLとして出力リストに追加し、段落をクリアします。
152        """
153        if paragraph:
154            rendered_lines = "<br>\n".join(inline_html(line) for line in paragraph)
155            output.append(f"<p>{rendered_lines}</p>")
156            paragraph.clear()
157
158    def close_list() -> None:
159        """
160        概要:
161            開いているリストタグがあれば閉じます。
162        """
163        nonlocal list_tag
164        if list_tag:
165            output.append(f"</{list_tag}>")
166            list_tag = None
167
168    for line in markdown.splitlines():
169        heading = re.match(r"^(#{3,6})\s+(.*)$", line)
170        unordered = re.match(r"^\s*[-*+]\s+(.*)$", line)
171        ordered = re.match(r"^\s*\d+[.)]\s+(.*)$", line)
172        if heading:
173            flush_paragraph()
174            close_list()
175            level = len(heading.group(1))
176            output.append(f"<h{level}>{inline_html(clean_heading(heading.group(2)))}</h{level}>")
177        elif unordered or ordered:
178            flush_paragraph()
179            tag = "ul" if unordered else "ol"
180            if list_tag != tag:
181                close_list()
182                output.append(f"<{tag}>")
183                list_tag = tag
184            output.append(f"<li>{inline_html((unordered or ordered).group(1))}</li>")
185        elif not line.strip():
186            flush_paragraph()
187            close_list()
188        else:
189            close_list()
190            paragraph.append(line.strip())
191    flush_paragraph()
192    close_list()
193    return "\n".join(output)
194
195
196def render_markdown(entries: list[tuple[str, str, list[str]]]) -> str:
197    """
198    概要:
199        プログラムのエントリリストからマークダウンを生成して返します。
200    引数:
201        :param entries: ラベル、本文、ランチャーエントリのリストを格納したタプルのリスト
202        :type entries: list
203    戻り値:
204        :returns: 生成されたマークダウン文字列
205        :rtype: str
206    """
207    blocks = ["# Programs"]
208    for label, body, launcher_entries in entries:
209        blocks.extend(("", "---", "", f"## {label}", "", body))
210        if launcher_entries:
211            blocks.extend(("", "### Launcher実装", ""))
212            blocks.extend(f"- {entry}" for entry in launcher_entries)
213    return "\n".join(blocks).rstrip() + "\n"
214
215
216def split_summary_line(markdown: str) -> tuple[str, str]:
217    """
218    概要:
219        セクションの最初の空でない行を要約として取得し、残りのテキストと分割します。
220    引数:
221        :param markdown: 分割対象のマークダウンテキスト
222        :type markdown: str
223    戻り値:
224        :returns: 要約文字列と残りのテキスト文字列のタプル
225        :rtype: tuple
226    """
227    lines = markdown.splitlines()
228    for index, line in enumerate(lines):
229        if line.strip():
230            return line.strip(), "\n".join(lines[index + 1 :]).strip()
231    return "", ""
232
233
234def render_html(entries: list[tuple[str, str, list[str]]]) -> str:
235    """
236    概要:
237        マークダウンのプログラムエントリを折りたたみ可能なHTMLブロックとしてレンダリングします。
238    引数:
239        :param entries: ラベル、本文、ランチャーエントリのリストを格納したタプルのリスト
240        :type entries: list
241    戻り値:
242        :returns: レンダリングされたHTML文字列
243        :rtype: str
244    """
245    blocks = [
246        "<!doctype html>", "<html lang=\"ja\">", "<head>",
247        "  <meta charset=\"utf-8\">", "  <title>Programs</title>",
248        "  <style>",
249        "    body { max-width: 1000px; margin: 2rem auto; padding: 0 1rem; font-family: sans-serif; }",
250        "    #keyword { width: min(32rem, 100%); padding: .45rem .6rem; font-size: 1rem; }",
251        "    #search-status { margin-left: .5rem; color: #555; }",
252        "    details { margin: .7rem 0; padding: .45rem .7rem; border: 1px solid #ccc; border-radius: .3rem; }",
253        "    summary { cursor: pointer; font-weight: bold; }",
254        "  </style>", "</head>", "<body>",
255        "<h1>Programs</h1>",
256        "<label for=\"keyword\">キーワード検索:</label>",
257        "<input id=\"keyword\" type=\"search\" placeholder=\"プログラム名・説明を検索\" autocomplete=\"off\">",
258        "<span id=\"search-status\"></span>",
259    ]
260    for label, body, launcher_entries in entries:
261        summary, remainder = split_summary_line(body)
262        blocks.extend((
263            "<section class=\"program\">",
264            "<hr>",
265            f"<h3>{html.escape(label)}</h3>",
266            "<details>",
267            f"  <summary>{inline_html(summary)}</summary>",
268            markdown_body_to_html(remainder),
269            "</details>",
270        ))
271        if launcher_entries:
272            blocks.extend((
273                "<details>",
274                "  <summary>Launcher実装</summary>",
275                *[f"  <div>{html.escape(entry)}</div>" for entry in launcher_entries],
276                "</details>",
277            ))
278        blocks.append("</section>")
279    blocks.extend((
280        "<script>",
281        "const keyword = document.getElementById('keyword');",
282        "const status = document.getElementById('search-status');",
283        "const items = [...document.querySelectorAll('.program')];",
284        "keyword.addEventListener('input', () => {",
285        "  const terms = keyword.value.trim().toLocaleLowerCase().split(/\\s+/).filter(Boolean);",
286        "  let visible = 0;",
287        "  for (const item of items) {",
288        "    const text = item.textContent.toLocaleLowerCase();",
289        "    const matched = terms.every(term => text.includes(term));",
290        "    item.hidden = !matched;",
291        "    if (matched) {",
292        "      visible += 1;",
293        "      if (terms.length) item.querySelector('details').open = true;",
294        "    }",
295        "  }",
296        "  status.textContent = terms.length ? `${visible} 件` : '';",
297        "});",
298        "</script>", "</body>", "</html>",
299    ))
300    return "\n".join(blocks) + "\n"
301
302
303def launcher_log_paths(pattern_specs: list[str]) -> list[Path]:
304    """
305    概要:
306        カレントディレクトリからセミコロン区切りのログファイルのワイルドカードを展開します。
307    引数:
308        :param pattern_specs: パターン指定文字列のリスト
309        :type pattern_specs: list
310    戻り値:
311        :returns: 解決されたファイルパスのリスト
312        :rtype: list
313    """
314    paths: dict[Path, None] = {}
315    for spec in pattern_specs:
316        for pattern in spec.split(";"):
317            pattern = pattern.strip()
318            if not pattern:
319                continue
320            for found in glob.glob(pattern):
321                path = Path(found)
322                if path.is_file():
323                    paths[path.resolve()] = None
324    return sorted(paths, key=lambda path: str(path).lower())
325
326
327def read_launcher_logs(log_paths: list[Path]) -> dict[str, list[str]]:
328    """
329    概要:
330        ランチャーのログを読み込み、Pythonのベース名と表示行のリストの辞書として返します。
331    引数:
332        :param log_paths: ログファイルのパスのリスト
333        :type log_paths: list
334    戻り値:
335        :returns: プログラム名をキー、表示用文字列リストを値とする辞書
336        :rtype: dict
337    """
338    mapping: dict[str, list[str]] = {}
339    for log_path in log_paths:
340        try:
341            lines = log_path.read_text(encoding="utf-8", errors="replace").splitlines()
342        except OSError as exc:
343            print(f"Warning: cannot read Launcher log {log_path}: {exc}", file=sys.stderr)
344            continue
345
346        in_program_references = False
347        current_program: str | None = None
348        for line in lines:
349            if line.startswith("Python program references"):
350                in_program_references = True
351                current_program = None
352                continue
353            if not in_program_references:
354                continue
355
356            program_match = LAUNCHER_PROGRAM_RE.match(line)
357            if program_match:
358                current_program = program_match.group(1)
359                mapping.setdefault(current_program, [])
360                continue
361
362            context_match = LAUNCHER_CONTEXT_RE.match(line)
363            if current_program and context_match:
364                display = f"{log_path.name}: {context_match.group(1)}"
365                if display not in mapping[current_program]:
366                    mapping[current_program].append(display)
367    return mapping
368
369
370def build_index(root_dir: Path, display_root: Path, program_patterns: list[str], output_format: str,
371                launcher_references: dict[str, list[str]]) -> tuple[str, int, list[str]]:
372    """
373    概要:
374        対象ディレクトリ内のマークダウンファイルを再帰的に検索してプログラムインデックスを構築します。
375    引数:
376        :param root_dir: 検索対象のルートディレクトリ
377        :type root_dir: pathlib.Path
378        :param display_root: 表示用の相対パスの基準となるディレクトリ
379        :type display_root: pathlib.Path
380        :param program_patterns: 対象プログラムを絞り込むためのパターンリスト
381        :type program_patterns: list
382        :param output_format: 出力フォーマット名、htmlまたはmarkdown
383        :type output_format: str
384        :param launcher_references: ランチャーの参照情報の辞書
385        :type launcher_references: dict
386    戻り値:
387        :returns: 生成された出力文字列、処理されたエントリ数、警告メッセージリストのタプル
388        :rtype: tuple
389    """
390    entries: list[tuple[str, str, list[str]]] = []
391    warnings: list[str] = []
392
393    usage_paths = sorted(path for path in root_dir.rglob(USAGE_GLOB) if path.is_file())
394    for usage_path in usage_paths:
395        try:
396            markdown = usage_path.read_text(encoding="utf-8")
397        except UnicodeDecodeError:
398            warnings.append(f"skip (not UTF-8): {usage_path}")
399            continue
400        body = first_level2_body(markdown)
401        if not body:
402            warnings.append(f"skip (no ## section): {usage_path}")
403            continue
404        label = program_label(usage_path, markdown, display_root)
405        if not any(fnmatch.fnmatchcase(label, pattern) for pattern in program_patterns):
406            continue
407        launcher_entries = launcher_references.get(Path(label).name, [])
408        entries.append((label, body, launcher_entries))
409
410    output = render_html(entries) if output_format == "html" else render_markdown(entries)
411    return output, len(entries), warnings
412
413
414def main() -> int:
415    """
416    概要:
417        コマンドライン引数を解析し、プログラムインデックスの構築処理を実行します。
418    戻り値:
419        :returns: 終了コード
420        :rtype: int
421    """
422    parser = argparse.ArgumentParser(
423        description="Create a Programs index from recursively searched Markdown files."
424    )
425    parser.add_argument("root_dir", type=Path, help="directory to search recursively")
426    parser.add_argument(
427        "root_dir_output",
428        type=Path,
429        nargs="?",
430        default=None,
431        help="base directory for displayed relative program paths (default: root_dir)",
432    )
433    parser.add_argument("-o", "--output", type=Path, default=None, help="output file (default: stdout)")
434    parser.add_argument("-p", "--pattern", action="append", default=None, metavar="GLOB", help="program-path wildcard(s), separated by ; and/or specified repeatedly (default: *)")
435    parser.add_argument("--script_list", action="append", default=None, metavar="GLOB", help="Launcher_extract_python.py log-file wildcard(s), separated by ; and/or specified repeatedly")
436    parser.add_argument("-f", "--format", choices=("auto", "markdown", "html"), default="auto", help="output format (default: infer from --output suffix)")
437    args = parser.parse_args()
438
439    root_dir = args.root_dir.resolve()
440    if not root_dir.is_dir():
441        parser.error(f"root_dir is not a directory: {root_dir}")
442    display_root = (args.root_dir_output if args.root_dir_output is not None else root_dir).resolve()
443
444    output_format = args.format
445    if output_format == "auto":
446        output_format = "html" if args.output and args.output.suffix.lower() in {".html", ".htm"} else "markdown"
447    pattern_specs = args.pattern or ["*"]
448    patterns = [pattern.strip() for spec in pattern_specs for pattern in spec.split(";") if pattern.strip()]
449    launcher_logs = launcher_log_paths(args.script_list or [])
450    launcher_references = read_launcher_logs(launcher_logs)
451    output, count, warnings = build_index(root_dir, display_root, patterns, output_format, launcher_references)
452    if args.output:
453        args.output.write_text(output, encoding="utf-8")
454    else:
455        sys.stdout.write(output)
456    for warning in warnings:
457        print(f"Warning: {warning}", file=sys.stderr)
458    print(f"Collected {count} program description(s).", file=sys.stderr)
459    if launcher_logs:
460        print(f"Read {len(launcher_logs)} Launcher log file(s).", file=sys.stderr)
461    return 0
462
463
464if __name__ == "__main__":
465    raise SystemExit(main())