make_title_html.py ダウンロード/コピー

make_title_html.py をダウンロード

make_title_html.py
make_title_html.py
  1"""
  2概要:
  3    タイトル一覧からリンク付きHTMLファイルを生成するスクリプトです。
  4詳細説明:
  5    カレントディレクトリのPDFファイルリストからCSVを生成し、そのCSVを基にして
  6    テンプレートHTMLにリンク情報を埋め込んだHTMLファイルを生成します。
  7    コマンドライン引数で create または html を指定して実行します。
  8"""
  9import os
 10import csv
 11import glob
 12import quopri
 13import re
 14
 15
 16fmask = "*.pdf"
 17f = sorted(glob.glob(fmask))
 18CSVPath = "Title.csv"
 19TemplatePath = 'template.html'
 20OutPath = 'output.html'
 21
 22if not os.path.exists(TemplatePath):
 23    TemplatePath = "Title-source.html"
 24    OutPath = "Title.html"
 25if not os.path.exists(TemplatePath):
 26    TemplatePath = "/doc4.eBook2.etc2/Title-source.html"
 27    OutPath = "Title.html"
 28
 29DirName = os.getcwd()
 30
 31
 32def usage():
 33    """
 34    概要:
 35        コマンドライン引数の使用方法を表示します。
 36    引数:
 37        なし
 38    戻り値:
 39        なし
 40    """
 41    print("usage: python MakeTitleHTML.py [create|html|imagehtml]")
 42
 43def create_source_csv():
 44    """
 45    概要:
 46        PDFファイルの一覧からソースCSVファイルを作成します。
 47    詳細説明:
 48        カレントディレクトリのPDFファイルを検索し、ファイル名からタイトルと
 49        インデックスを抽出して Title-source.csv に書き出します。
 50    引数:
 51        なし
 52    戻り値:
 53        なし
 54    """
 55    CSVPath = "Title-source.csv"
 56    with open(CSVPath, "w", newline='', encoding='utf-8') as csvfile:
 57        writer = csv.writer(csvfile)
 58        for fn in f:
 59            fn_base = os.path.splitext(fn)[0]
 60            idx = ''
 61            if ' ' in fn_base:
 62                parts = fn_base.rsplit(' ', 1)
 63                if parts[1].isdigit():
 64                    idx = parts[1]
 65                    fn_base = parts[0]
 66            writer.writerow([fn_base, fn, idx])
 67
 68def make_html():
 69    """
 70    概要:
 71        CSVファイルとテンプレートからリンク集HTMLを生成します。
 72    詳細説明:
 73        CSVファイルからタイトルとファイル情報を読み込み、リンクタグを作成します。
 74        次にテンプレートHTMLを読み込み、指定されたタイトル部分をリンクに置換して
 75        新しいHTMLファイルとして出力します。
 76        日本語文字化けを防ぐため、Quoted-Printableエンコードを利用して置換処理を行います。
 77    引数:
 78        なし
 79    戻り値:
 80        なし
 81    """
 82    db = {}
 83    key = []
 84    BaseDirectory = ''
 85    PageOffset = 0
 86    iPrint = 0
 87    c = 0
 88
 89    with open(CSVPath, "r", encoding = 'cp932') as csvfile:
 90#    with open(CSVPath, "r", encoding='utf-8') as csvfile:
 91        reader = csv.reader(csvfile)
 92
 93        prev = []
 94        for row in reader:
 95            if not row: continue
 96
 97            print("row: ", row)
 98            _aa = row
 99            na = len(_aa)
100            title = _aa[0] if na >= 1 else None
101            file  = _aa[1] if na >= 2 else None
102            index = _aa[2] if na >= 3 else None
103            page  = _aa[3] if na >= 4 else None
104            if title.startswith('#!'):
105                key_cmd = title[2:]
106                if key_cmd.lower() == 'end':
107                    break
108                elif key_cmd.lower() == 'basedirectory':
109                    BaseDirectory = file
110                elif key_cmd.lower() == 'pageoffset':
111                    PageOffset = int(file)
112                elif key_cmd.lower() == 'print':
113                    file = file.replace('\\n', '\n').replace('\\t', '\t')
114                    db[f"#!Print_{iPrint}"] = file
115                    key.append(f"#!Print_{iPrint}")
116                    iPrint += 1
117                prev = []
118                continue
119
120            if title is None: continue
121            if file is None: file = prev[1]
122            if index is None: index = prev[2]
123
124            if title not in db: key.append(title)
125            if index is None: index = 'no index'
126
127            file_path = os.path.join(BaseDirectory, file) if BaseDirectory else file
128            try:
129                page = int(page) + PageOffset
130            except:
131                pass
132
133            opt = f"#page__equal__{page}" if page else ''
134
135            if title in db:
136                db[title] += f" (<a href=\"{file_path}{opt}\" target=\"_blank\">{index}</a>)"
137            else:
138                db[title] = f"<a href=\"{file_path}{opt}\" target=\"_blank\">{title}</a> (<a href=\"{file_path}{opt}\" target=\"_blank\">{index}</a>)"
139            prev = row
140            c += 1
141
142    with open(TemplatePath, "r", encoding='cp932') as template_file:
143#    with open(TemplatePath, "r", encoding='utf-8') as template_file:
144        content = template_file.read()
145#    content = Utils.QPEncode(content)
146    content = quopri.encodestring(content.encode('cp932'))
147
148    OtherLinks = ''
149    for i, key in enumerate(key):
150        k = key
151        v = db[key]
152#        k = Utils.QPEncode(k)
153        k = quopri.encodestring(k.encode('cp932'))
154        k = re.escape(k)
155        if re.search(k, content, re.IGNORECASE):
156            content = re.sub(k, v, content, flags=re.IGNORECASE)
157        else:
158            OtherLinks += f"{v}<br>\n"
159
160    '''
161#    dl = Utils.QPEncode('{DirName}')
162    dl = quopri.encodestring('{DirName}'.encode('cp932'))
163    content = re.sub(dl, DirName, content, flags=re.IGNORECASE)
164#    ol = Utils.QPEncode('{OtherLinks}')
165    ol = quopri.encodestring('{OtherLinks}'.encode('cp932'))
166    content = re.sub(ol, OtherLinks, content, flags=re.IGNORECASE)
167    '''
168
169#    content = Utils.QPDecode(content)
170    content = quopri.decodestring(content)
171    content = content.decode('cp932').replace('__equal__', '=')
172
173    with open(OutPath, "w", encoding='cp932') as out_file:
174#    with open(OutPath, "w", encoding='utf-8') as out_file:
175        out_file.write("<H1>Sorted by title</H1>\n")
176        out_file.write(content)
177        out_file.write("<hr>\n")
178        out_file.write("<H1>Sorted by volume</H1>\n")
179
180    with open(CSVPath, "r", encoding='cp932') as csvfile:
181#    with open(CSVPath, "r", encoding='utf-8') as csvfile:
182        reader = csv.reader(csvfile)
183        currentfile = ''
184        prev = []
185        for row in reader:
186            if not row:
187                continue
188            if row[0].startswith('#!End'):
189                break
190            if row[0].startswith('#!'):
191                continue
192
193            _aa = row
194            na = len(_aa)
195            title = _aa[0] if na >= 1 else None
196            file  = _aa[1] if na >= 2 else None
197            index = _aa[2] if na >= 3 else None
198            page  = _aa[3] if na >= 4 else None
199 
200            if title is None: continue
201            if file is None: file = prev[1]
202            if index is None: index = prev[2]
203
204            if currentfile != file:
205                with open(OutPath, "a", encoding='cp932') as out_file:
206#                with open(OutPath, "a", encoding='utf-8') as out_file:
207                    out_file.write("<hr>\n")
208                    out_file.write(f"<H2>{file}</H2>\n")
209                currentfile = file
210
211            db[title] = db[title].replace('__equal__', '=')
212            with open(OutPath, "a", encoding='cp932') as out_file:
213#            with open(OutPath, "a", encoding='utf-8') as out_file:
214                out_file.write(f"{db[title]}<br>\n")
215            prev = row
216
217
218if __name__ == "__main__":
219    import sys
220    if len(sys.argv) < 2:
221        usage()
222        sys.exit()
223
224    command = sys.argv[1]
225    if command == 'create':
226        create_source_csv()
227    elif command == 'html':
228        make_html()
229    else:
230        usage()