Repository navigation
Expand file tree
/
Copy pathmht_extractor.py
More file actions
95 lines (72 loc) · 2.57 KB
/
Copy pathmht_extractor.py
File metadata and controls
95 lines (72 loc) · 2.57 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
import os
import email
import argparse
from urllib.parse import urlparse
class Helpers:
@staticmethod
def partname_or_filename(fp, url):
return os.path.basename(urlparse(url).path) or '{}.html'.format(os.path.basename(os.path.splitext(fp.name)[0]))
class Extractor:
def __init__(self, fp, output_dir, verbose=False):
self.fp = fp
self.output_dir = output_dir
self.verbose = verbose
self.fdir = '{}.html_files'.format(os.path.basename(os.path.splitext(self.fp.name)[0]))
@classmethod
def extract_files(cls, files, output_dir, verbose=False):
for file in files:
with open(file, 'r') as fp:
yield cls(fp, output_dir, verbose)
def extract(self):
message = email.message_from_file(self.fp)
self._print(f'Extracting {self.fp.name}', verbose_only=True)
for part in message.walk():
content_type = part.get_content_type()
content_location = part['Content-Location']
name = Helpers.partname_or_filename(self.fp, content_location)
if content_type.startswith('multipart'):
continue
#TODO: using legacy api
self._write(name, part.get_payload(decode=True))
def _write(self, name, content):
newpath = ''
if name.endswith('html'):
newpath = os.path.join(self.output_dir, name)
else:
if not os.path.exists(os.path.join(self.output_dir, self.fdir)):
os.mkdir(os.path.join(self.output_dir, self.fdir))
newpath = os.path.join(self.output_dir, self.fdir, name)
self._print(f'{name} -> {newpath}')
with open(newpath, 'wb') as fp:
fp.write(content)
def _print(self, *args, verbose_only=False, **kwargs):
if verbose_only:
if self.verbose:
print(*args, **kwargs)
else:
print(*args, **kwargs)
if __name__ == '__main__':
parser = argparse.ArgumentParser(
prog='mht_extractor',
description='info: https://github.com/aicantar/MhtExtractor'
)
parser.add_argument(
'FILE',
nargs='+',
help='files to extract'
)
parser.add_argument(
'-o',
default='.',
dest='output_dir',
help='output directory'
)
parser.add_argument(
'-v',
action='store_true',
dest='verbose',
help='enable verbose mode'
)
args = parser.parse_args()
for ex in Extractor.extract_files(args.FILE, args.output_dir, args.verbose):
ex.extract()