make_title_html.py ダウンロード/コピー
make_title_html.py
make_title_html.py
1"""
2概要:
3 タイトル一覧からリンク付きHTMLファイルを生成するスクリプトです。
4詳細説明:
5 カレントディレクトリのPDFファイルリストからCSVを生成し、そのCSVを基にして
6 テンプレートHTMLにリンク情報を埋め込んだHTMLファイルを生成します。
7 コマンドライン引数で create または html を指定して実行します。
8"""
9import os
10import csv
11import glob
12import quopri
13import re
14
15
16fmask = "*.pdf"
17f = sorted(glob.glob(fmask))
18CSVPath = "Title.csv"
19TemplatePath = 'template.html'
20OutPath = 'output.html'
21
22if not os.path.exists(TemplatePath):
23 TemplatePath = "Title-source.html"
24 OutPath = "Title.html"
25if not os.path.exists(TemplatePath):
26 TemplatePath = "/doc4.eBook2.etc2/Title-source.html"
27 OutPath = "Title.html"
28
29DirName = os.getcwd()
30
31
32def usage():
33 """
34 概要:
35 コマンドライン引数の使用方法を表示します。
36 引数:
37 なし
38 戻り値:
39 なし
40 """
41 print("usage: python MakeTitleHTML.py [create|html|imagehtml]")
42
43def create_source_csv():
44 """
45 概要:
46 PDFファイルの一覧からソースCSVファイルを作成します。
47 詳細説明:
48 カレントディレクトリのPDFファイルを検索し、ファイル名からタイトルと
49 インデックスを抽出して Title-source.csv に書き出します。
50 引数:
51 なし
52 戻り値:
53 なし
54 """
55 CSVPath = "Title-source.csv"
56 with open(CSVPath, "w", newline='', encoding='utf-8') as csvfile:
57 writer = csv.writer(csvfile)
58 for fn in f:
59 fn_base = os.path.splitext(fn)[0]
60 idx = ''
61 if ' ' in fn_base:
62 parts = fn_base.rsplit(' ', 1)
63 if parts[1].isdigit():
64 idx = parts[1]
65 fn_base = parts[0]
66 writer.writerow([fn_base, fn, idx])
67
68def make_html():
69 """
70 概要:
71 CSVファイルとテンプレートからリンク集HTMLを生成します。
72 詳細説明:
73 CSVファイルからタイトルとファイル情報を読み込み、リンクタグを作成します。
74 次にテンプレートHTMLを読み込み、指定されたタイトル部分をリンクに置換して
75 新しいHTMLファイルとして出力します。
76 日本語文字化けを防ぐため、Quoted-Printableエンコードを利用して置換処理を行います。
77 引数:
78 なし
79 戻り値:
80 なし
81 """
82 db = {}
83 key = []
84 BaseDirectory = ''
85 PageOffset = 0
86 iPrint = 0
87 c = 0
88
89 with open(CSVPath, "r", encoding = 'cp932') as csvfile:
90# with open(CSVPath, "r", encoding='utf-8') as csvfile:
91 reader = csv.reader(csvfile)
92
93 prev = []
94 for row in reader:
95 if not row: continue
96
97 print("row: ", row)
98 _aa = row
99 na = len(_aa)
100 title = _aa[0] if na >= 1 else None
101 file = _aa[1] if na >= 2 else None
102 index = _aa[2] if na >= 3 else None
103 page = _aa[3] if na >= 4 else None
104 if title.startswith('#!'):
105 key_cmd = title[2:]
106 if key_cmd.lower() == 'end':
107 break
108 elif key_cmd.lower() == 'basedirectory':
109 BaseDirectory = file
110 elif key_cmd.lower() == 'pageoffset':
111 PageOffset = int(file)
112 elif key_cmd.lower() == 'print':
113 file = file.replace('\\n', '\n').replace('\\t', '\t')
114 db[f"#!Print_{iPrint}"] = file
115 key.append(f"#!Print_{iPrint}")
116 iPrint += 1
117 prev = []
118 continue
119
120 if title is None: continue
121 if file is None: file = prev[1]
122 if index is None: index = prev[2]
123
124 if title not in db: key.append(title)
125 if index is None: index = 'no index'
126
127 file_path = os.path.join(BaseDirectory, file) if BaseDirectory else file
128 try:
129 page = int(page) + PageOffset
130 except:
131 pass
132
133 opt = f"#page__equal__{page}" if page else ''
134
135 if title in db:
136 db[title] += f" (<a href=\"{file_path}{opt}\" target=\"_blank\">{index}</a>)"
137 else:
138 db[title] = f"<a href=\"{file_path}{opt}\" target=\"_blank\">{title}</a> (<a href=\"{file_path}{opt}\" target=\"_blank\">{index}</a>)"
139 prev = row
140 c += 1
141
142 with open(TemplatePath, "r", encoding='cp932') as template_file:
143# with open(TemplatePath, "r", encoding='utf-8') as template_file:
144 content = template_file.read()
145# content = Utils.QPEncode(content)
146 content = quopri.encodestring(content.encode('cp932'))
147
148 OtherLinks = ''
149 for i, key in enumerate(key):
150 k = key
151 v = db[key]
152# k = Utils.QPEncode(k)
153 k = quopri.encodestring(k.encode('cp932'))
154 k = re.escape(k)
155 if re.search(k, content, re.IGNORECASE):
156 content = re.sub(k, v, content, flags=re.IGNORECASE)
157 else:
158 OtherLinks += f"{v}<br>\n"
159
160 '''
161# dl = Utils.QPEncode('{DirName}')
162 dl = quopri.encodestring('{DirName}'.encode('cp932'))
163 content = re.sub(dl, DirName, content, flags=re.IGNORECASE)
164# ol = Utils.QPEncode('{OtherLinks}')
165 ol = quopri.encodestring('{OtherLinks}'.encode('cp932'))
166 content = re.sub(ol, OtherLinks, content, flags=re.IGNORECASE)
167 '''
168
169# content = Utils.QPDecode(content)
170 content = quopri.decodestring(content)
171 content = content.decode('cp932').replace('__equal__', '=')
172
173 with open(OutPath, "w", encoding='cp932') as out_file:
174# with open(OutPath, "w", encoding='utf-8') as out_file:
175 out_file.write("<H1>Sorted by title</H1>\n")
176 out_file.write(content)
177 out_file.write("<hr>\n")
178 out_file.write("<H1>Sorted by volume</H1>\n")
179
180 with open(CSVPath, "r", encoding='cp932') as csvfile:
181# with open(CSVPath, "r", encoding='utf-8') as csvfile:
182 reader = csv.reader(csvfile)
183 currentfile = ''
184 prev = []
185 for row in reader:
186 if not row:
187 continue
188 if row[0].startswith('#!End'):
189 break
190 if row[0].startswith('#!'):
191 continue
192
193 _aa = row
194 na = len(_aa)
195 title = _aa[0] if na >= 1 else None
196 file = _aa[1] if na >= 2 else None
197 index = _aa[2] if na >= 3 else None
198 page = _aa[3] if na >= 4 else None
199
200 if title is None: continue
201 if file is None: file = prev[1]
202 if index is None: index = prev[2]
203
204 if currentfile != file:
205 with open(OutPath, "a", encoding='cp932') as out_file:
206# with open(OutPath, "a", encoding='utf-8') as out_file:
207 out_file.write("<hr>\n")
208 out_file.write(f"<H2>{file}</H2>\n")
209 currentfile = file
210
211 db[title] = db[title].replace('__equal__', '=')
212 with open(OutPath, "a", encoding='cp932') as out_file:
213# with open(OutPath, "a", encoding='utf-8') as out_file:
214 out_file.write(f"{db[title]}<br>\n")
215 prev = row
216
217
218if __name__ == "__main__":
219 import sys
220 if len(sys.argv) < 2:
221 usage()
222 sys.exit()
223
224 command = sys.argv[1]
225 if command == 'create':
226 create_source_csv()
227 elif command == 'html':
228 make_html()
229 else:
230 usage()