Fri, 19 Oct 2007 22:46:20 -0400
[svn] Change all the license notices from GPLv2 to GPLv3.
Instead of checking the archive contents, figuring out what to do, and
doing it, instead we now always extract the archive to a private directory,
and then shuffle around the contents appropriately. I expected this to be
a bigger win than my benchmarks have borne out, but I'm sticking with this
strategy because it provides a cleaner separation of responsibilities
between the extractors and the archive type handlers, and also I have to
believe it's a much better way to handle bigger archives -- since we're now
reading it once and not twice.
1 | 1 | #!/usr/bin/env python |
2 | # | |
19 | 3 | # dtrx -- Intelligently extract various archive types. |
23
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
4 | # Copyright (c) 2006, 2007 Brett Smith <brettcsmith@brettcsmith.org>. |
1 | 5 | # |
6 | # This program is free software; you can redistribute it and/or modify it | |
7 | # under the terms of the GNU General Public License as published by the | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
8 | # Free Software Foundation; either version 3 of the License, or (at your |
1 | 9 | # option) any later version. |
10 | # | |
11 | # This program is distributed in the hope that it will be useful, but | |
12 | # WITHOUT ANY WARRANTY; without even the implied warranty of | |
13 | # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General | |
14 | # Public License for more details. | |
15 | # | |
16 | # You should have received a copy of the GNU General Public License along | |
17 | # with this program; if not, write to the Free Software Foundation, Inc., | |
18 | # 51 Franklin Street, 5th Floor, Boston, MA, 02111. | |
19 | ||
5 | 20 | import errno |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
21 | import logging |
1 | 22 | import mimetypes |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
23 | import optparse |
1 | 24 | import os |
15
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
25 | import stat |
1 | 26 | import subprocess |
27 | import sys | |
28 | import tempfile | |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
29 | import textwrap |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
30 | import traceback |
1 | 31 | |
32 | from cStringIO import StringIO | |
33 | ||
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
34 | VERSION = "5.0" |
19 | 35 | VERSION_BANNER = """dtrx version %s |
23
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
36 | Copyright (c) 2006, 2007 Brett Smith <brettcsmith@brettcsmith.org> |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
37 | |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
38 | This program is free software; you can redistribute it and/or modify it |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
39 | under the terms of the GNU General Public License as published by the |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
40 | Free Software Foundation; either version 3 of the License, or (at your |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
41 | option) any later version. |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
42 | |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
43 | This program is distributed in the hope that it will be useful, but |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
44 | WITHOUT ANY WARRANTY; without even the implied warranty of |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
45 | MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
46 | Public License for more details.""" % (VERSION,) |
1 | 47 | |
48 | MATCHING_DIRECTORY = 1 | |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
49 | ONE_ENTRY = 2 |
1 | 50 | BOMB = 3 |
51 | EMPTY = 4 | |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
52 | ONE_ENTRY_KNOWN = 5 |
1 | 53 | |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
54 | EXTRACT_HERE = 1 |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
55 | EXTRACT_WRAP = 2 |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
56 | EXTRACT_RENAME = 3 |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
57 | |
23
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
58 | RECURSE_ALWAYS = 1 |
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
59 | RECURSE_ONCE = 2 |
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
60 | RECURSE_NOT_NOW = 3 |
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
61 | RECURSE_NEVER = 4 |
039dd321a7d0
[svn] If an archive contains other archives, and the user didn't specify that
brett
parents:
22
diff
changeset
|
62 | |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
63 | mimetypes.encodings_map.setdefault('.bz2', 'bzip2') |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
64 | mimetypes.types_map['.exe'] = 'application/x-msdos-program' |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
65 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
66 | def run_command(command, description, stdout=None, stderr=None, stdin=None): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
67 | process = subprocess.Popen(command, stdin=stdin, stdout=stdout, |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
68 | stderr=stderr) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
69 | status = process.wait() |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
70 | for pipe in (process.stdout, process.stderr): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
71 | try: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
72 | pipe.close() |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
73 | except AttributeError: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
74 | pass |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
75 | if status != 0: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
76 | return ("%s error: '%s' returned status code %s" % |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
77 | (description, ' '.join(command), status)) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
78 | return None |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
79 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
80 | class FilenameChecker(object): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
81 | def __init__(self, original_name): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
82 | self.original_name = original_name |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
83 | |
17
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
84 | def is_free(self, filename): |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
85 | return not os.path.exists(filename) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
86 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
87 | def check(self): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
88 | for suffix in [''] + ['.%s' % (x,) for x in range(1, 10)]: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
89 | filename = '%s%s' % (self.original_name, suffix) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
90 | if self.is_free(filename): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
91 | return filename |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
92 | raise ValueError("all alternatives for name %s taken" % |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
93 | (self.original_name,)) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
94 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
95 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
96 | class DirectoryChecker(FilenameChecker): |
17
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
97 | def is_free(self, filename): |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
98 | try: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
99 | os.mkdir(filename) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
100 | except OSError, error: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
101 | if error.errno == errno.EEXIST: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
102 | return False |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
103 | raise |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
104 | return True |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
105 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
106 | |
1 | 107 | class ExtractorError(Exception): |
108 | pass | |
109 | ||
110 | ||
111 | class BaseExtractor(object): | |
112 | decoders = {'bzip2': 'bzcat', 'gzip': 'zcat', 'compress': 'zcat'} | |
113 | ||
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
114 | name_checker = DirectoryChecker |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
115 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
116 | def __init__(self, filename, encoding): |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
117 | if encoding and (not self.decoders.has_key(encoding)): |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
118 | raise ValueError("unrecognized encoding %s" % (encoding,)) |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
119 | self.filename = os.path.realpath(filename) |
1 | 120 | self.encoding = encoding |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
121 | self.included_archives = [] |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
122 | self.target = None |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
123 | self.content_type = None |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
124 | self.content_name = None |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
125 | self.pipes = [] |
5 | 126 | try: |
127 | self.archive = open(filename, 'r') | |
128 | except (IOError, OSError), error: | |
129 | raise ExtractorError("could not open %s: %s" % | |
130 | (filename, error.strerror)) | |
1 | 131 | if encoding: |
132 | self.pipe([self.decoders[encoding]], "decoding") | |
133 | self.prepare() | |
134 | ||
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
135 | def pipe(self, command, description="extraction"): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
136 | self.pipes.append((command, description)) |
1 | 137 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
138 | def run_pipes(self, final_stdout=None): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
139 | if final_stdout is None: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
140 | # FIXME: Buffering this might be dumb. |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
141 | final_stdout = tempfile.TemporaryFile() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
142 | if not self.pipes: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
143 | return |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
144 | num_pipes = len(self.pipes) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
145 | last_pipe = num_pipes - 1 |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
146 | processes = [] |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
147 | for index, command in enumerate([pipe[0] for pipe in self.pipes]): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
148 | if index == 0: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
149 | stdin = self.archive |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
150 | else: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
151 | stdin = processes[-1].stdout |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
152 | if index == last_pipe: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
153 | stdout = final_stdout |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
154 | else: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
155 | stdout = subprocess.PIPE |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
156 | processes.append(subprocess.Popen(command, stdin=stdin, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
157 | stdout=stdout, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
158 | stderr=subprocess.PIPE)) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
159 | exit_codes = [pipe.wait() for pipe in processes] |
1 | 160 | self.archive.close() |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
161 | for index in range(last_pipe): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
162 | processes[index].stdout.close() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
163 | processes[index].stderr.close() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
164 | for index, status in enumerate(exit_codes): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
165 | if status != 0: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
166 | raise ExtractorError("%s error: '%s' returned status code %s" % |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
167 | (self.pipes[index][1], |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
168 | ' '.join(self.pipes[index][0]), status)) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
169 | self.archive = final_stdout |
1 | 170 | |
171 | def prepare(self): | |
172 | pass | |
173 | ||
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
174 | def check_included_archives(self, filenames): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
175 | for filename in filenames: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
176 | if extractor_map.has_key(mimetypes.guess_type(filename)[0]): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
177 | self.included_archives.append(filename) |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
178 | |
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
179 | def check_contents(self): |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
180 | filenames = os.listdir('.') |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
181 | if not filenames: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
182 | self.content_type = EMPTY |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
183 | elif len(filenames) == 1: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
184 | if self.basename() == filenames[0]: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
185 | self.content_type = MATCHING_DIRECTORY |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
186 | else: |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
187 | self.content_type = ONE_ENTRY |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
188 | self.content_name = filenames[0] |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
189 | if os.path.isdir(filenames[0]): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
190 | self.content_name += '/' |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
191 | else: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
192 | self.content_type = BOMB |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
193 | self.check_included_archives(filenames) |
1 | 194 | |
195 | def basename(self): | |
5 | 196 | pieces = os.path.basename(self.filename).split('.') |
2
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
197 | extension = '.' + pieces[-1] |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
198 | if mimetypes.encodings_map.has_key(extension): |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
199 | pieces.pop() |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
200 | extension = '.' + pieces[-1] |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
201 | if (mimetypes.types_map.has_key(extension) or |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
202 | mimetypes.common_types.has_key(extension) or |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
203 | mimetypes.suffix_map.has_key(extension)): |
1 | 204 | pieces.pop() |
205 | return '.'.join(pieces) | |
206 | ||
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
207 | def extract(self): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
208 | self.target = tempfile.mkdtemp(prefix='.dtrx-', dir='.') |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
209 | old_path = os.path.realpath(os.curdir) |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
210 | os.chdir(self.target) |
1 | 211 | self.archive.seek(0, 0) |
212 | self.extract_archive() | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
213 | self.check_contents() |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
214 | os.chdir(old_path) |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
215 | |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
216 | def get_filenames(self): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
217 | self.run_pipes() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
218 | self.archive.seek(0, 0) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
219 | while True: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
220 | line = self.archive.readline() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
221 | if not line: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
222 | self.archive.close() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
223 | return |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
224 | yield line.rstrip('\n') |
1 | 225 | |
226 | ||
227 | class TarExtractor(BaseExtractor): | |
228 | def get_filenames(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
229 | self.pipe(['tar', '-t'], "listing") |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
230 | return BaseExtractor.get_filenames(self) |
1 | 231 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
232 | def extract_archive(self): |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
233 | self.pipe(['tar', '-x']) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
234 | self.run_pipes() |
1 | 235 | |
236 | ||
237 | class ZipExtractor(BaseExtractor): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
238 | def __init__(self, filename, encoding): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
239 | BaseExtractor.__init__(self, '/dev/null', None) |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
240 | self.filename = os.path.realpath(filename) |
1 | 241 | |
242 | def get_filenames(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
243 | self.pipe(['zipinfo', '-1', self.filename], "listing") |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
244 | return BaseExtractor.get_filenames(self) |
1 | 245 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
246 | def extract_archive(self): |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
247 | self.pipe(['unzip', '-q', self.filename]) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
248 | self.run_pipes() |
1 | 249 | |
250 | ||
251 | class CpioExtractor(BaseExtractor): | |
252 | def get_filenames(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
253 | self.pipe(['cpio', '-t'], "listing") |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
254 | return BaseExtractor.get_filenames(self) |
1 | 255 | |
256 | def extract_archive(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
257 | self.pipe(['cpio', '-i', '--make-directories', |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
258 | '--no-absolute-filenames']) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
259 | self.run_pipes() |
1 | 260 | |
261 | ||
262 | class RPMExtractor(CpioExtractor): | |
263 | def prepare(self): | |
264 | self.pipe(['rpm2cpio', '-'], "rpm2cpio") | |
265 | ||
266 | def basename(self): | |
9
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
267 | pieces = os.path.basename(self.filename).split('.') |
2
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
268 | if len(pieces) == 1: |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
269 | return pieces[0] |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
270 | elif pieces[-1] != 'rpm': |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
271 | return BaseExtractor.basename(self) |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
272 | pieces.pop() |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
273 | if len(pieces) == 1: |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
274 | return pieces[0] |
9
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
275 | elif len(pieces[-1]) < 8: |
2
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
276 | pieces.pop() |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
277 | return '.'.join(pieces) |
1 | 278 | |
279 | def check_contents(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
280 | self.check_included_archives(os.listdir('.')) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
281 | self.content_type = BOMB |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
282 | |
1 | 283 | |
284 | class DebExtractor(TarExtractor): | |
285 | def prepare(self): | |
286 | self.pipe(['ar', 'p', self.filename, 'data.tar.gz'], | |
287 | "data.tar.gz extraction") | |
288 | self.pipe(['zcat'], "data.tar.gz decompression") | |
289 | ||
290 | def basename(self): | |
9
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
291 | pieces = os.path.basename(self.filename).split('_') |
2
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
292 | if len(pieces) == 1: |
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
293 | return pieces[0] |
9
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
294 | last_piece = pieces.pop() |
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
295 | if (len(last_piece) > 10) or (not last_piece.endswith('.deb')): |
2
1570351bf863
[svn] Fix a small bug that would crash the program if an archive was empty.
brett
parents:
1
diff
changeset
|
296 | return BaseExtractor.basename(self) |
9
920417b8acc9
[svn] Fix issues with basename methods. First, string's rsplit method only
brett
parents:
8
diff
changeset
|
297 | return '_'.join(pieces) |
1 | 298 | |
299 | def check_contents(self): | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
300 | self.check_included_archives(os.listdir('.')) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
301 | self.content_type = BOMB |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
302 | |
1 | 303 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
304 | class CompressionExtractor(BaseExtractor): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
305 | name_checker = FilenameChecker |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
306 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
307 | def basename(self): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
308 | pieces = os.path.basename(self.filename).split('.') |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
309 | extension = '.' + pieces[-1] |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
310 | if mimetypes.encodings_map.has_key(extension): |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
311 | pieces.pop() |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
312 | return '.'.join(pieces) |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
313 | |
15
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
314 | def get_filenames(self): |
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
315 | yield self.basename() |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
316 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
317 | def extract(self): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
318 | self.content_type = ONE_ENTRY_KNOWN |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
319 | self.content_name = self.basename() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
320 | output_fd, self.target = tempfile.mkstemp(prefix='.dtrx-', dir='.') |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
321 | self.run_pipes(output_fd) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
322 | os.close(output_fd) |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
323 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
324 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
325 | class BaseHandler(object): |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
326 | def __init__(self, extractor, options): |
19 | 327 | self.logger = logging.getLogger('dtrx-log') |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
328 | self.extractor = extractor |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
329 | self.options = options |
17
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
330 | self.target = None |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
331 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
332 | def handle(self): |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
333 | command = 'find' |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
334 | status = subprocess.call(['find', self.extractor.target, '-type', 'd', |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
335 | '-exec', 'chmod', 'u+rwx', '{}', ';']) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
336 | if status == 0: |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
337 | command = 'chmod' |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
338 | status = subprocess.call(['chmod', '-R', 'u+rwX', |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
339 | self.extractor.target]) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
340 | if status != 0: |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
341 | return "%s returned with exit status %s" % (command, status) |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
342 | return self.organize() |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
343 | |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
344 | |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
345 | # The "where to extract" table, with options and archive types. |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
346 | # This dictates the contents of each can_handle method. |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
347 | # |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
348 | # Flat Overwrite None |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
349 | # File basename basename FilenameChecked |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
350 | # Match . . tempdir + checked |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
351 | # Bomb . basename DirectoryChecked |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
352 | |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
353 | class FlatHandler(BaseHandler): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
354 | def can_handle(contents, options): |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
355 | return ((options.flat and (contents != ONE_ENTRY_KNOWN)) or |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
356 | (options.overwrite and (contents == MATCHING_DIRECTORY))) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
357 | can_handle = staticmethod(can_handle) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
358 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
359 | def organize(self): |
17
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
360 | self.target = '.' |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
361 | for curdir, dirs, filenames in os.walk(self.extractor.target, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
362 | topdown=False): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
363 | path_parts = curdir.split(os.sep) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
364 | if path_parts[0] == '.': |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
365 | path_parts.pop(1) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
366 | else: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
367 | path_parts.pop(0) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
368 | newdir = os.path.join(*path_parts) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
369 | if not os.path.isdir(newdir): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
370 | os.makedirs(newdir) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
371 | for filename in filenames: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
372 | os.rename(os.path.join(curdir, filename), |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
373 | os.path.join(newdir, filename)) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
374 | os.rmdir(curdir) |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
375 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
376 | |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
377 | class OverwriteHandler(BaseHandler): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
378 | def can_handle(contents, options): |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
379 | return ((options.flat and (contents == ONE_ENTRY_KNOWN)) or |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
380 | (options.overwrite and (contents != MATCHING_DIRECTORY))) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
381 | can_handle = staticmethod(can_handle) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
382 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
383 | def organize(self): |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
384 | self.target = self.extractor.basename() |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
385 | result = run_command(['rm', '-rf', self.target], |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
386 | "removing %s to overwrite" % (self.target,)) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
387 | if result is None: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
388 | os.rename(self.extractor.target, self.target) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
389 | return result |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
390 | |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
391 | |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
392 | class MatchHandler(BaseHandler): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
393 | def can_handle(contents, options): |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
394 | return ((contents == MATCHING_DIRECTORY) or |
22
b240777ae53e
[svn] Improve the way we check archive contents. If all the entries look like
brett
parents:
20
diff
changeset
|
395 | ((contents == ONE_ENTRY) and |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
396 | options.one_entry_policy.ok_for_match())) |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
397 | can_handle = staticmethod(can_handle) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
398 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
399 | def organize(self): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
400 | if self.options.one_entry_policy == EXTRACT_HERE: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
401 | destination = self.extractor.content_name.rstrip('/') |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
402 | else: |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
403 | destination = self.extractor.basename() |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
404 | self.target = self.extractor.name_checker(destination).check() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
405 | if os.path.isdir(self.extractor.target): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
406 | os.rename(os.path.join(self.extractor.target, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
407 | os.listdir(self.extractor.target)[0]), |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
408 | self.target) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
409 | os.rmdir(self.extractor.target) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
410 | else: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
411 | os.rename(self.extractor.target, self.target) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
412 | |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
413 | |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
414 | class EmptyHandler(object): |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
415 | def can_handle(contents, options): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
416 | return contents == EMPTY |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
417 | can_handle = staticmethod(can_handle) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
418 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
419 | def __init__(self, extractor, options): pass |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
420 | def handle(self): pass |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
421 | |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
422 | |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
423 | class BombHandler(BaseHandler): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
424 | def can_handle(contents, options): |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
425 | return True |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
426 | can_handle = staticmethod(can_handle) |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
427 | |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
428 | def organize(self): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
429 | basename = self.extractor.basename() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
430 | self.target = self.extractor.name_checker(basename).check() |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
431 | os.rename(self.extractor.target, self.target) |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
432 | |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
433 | |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
434 | class BasePolicy(object): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
435 | def __init__(self, options): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
436 | self.current_policy = None |
26 | 437 | if options.batch: |
438 | self.permanent_policy = self.answers[''] | |
439 | else: | |
440 | self.permanent_policy = None | |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
441 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
442 | def ask_question(self, question): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
443 | question = textwrap.wrap(question) + self.choices |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
444 | while True: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
445 | print "\n".join(question) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
446 | try: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
447 | answer = raw_input(self.prompt) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
448 | except EOFError: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
449 | return self.answers[''] |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
450 | try: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
451 | return self.answers[answer.lower()] |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
452 | except KeyError: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
453 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
454 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
455 | def __cmp__(self, other): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
456 | return cmp(self.current_policy, other) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
457 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
458 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
459 | class OneEntryPolicy(BasePolicy): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
460 | answers = {'h': EXTRACT_HERE, 'i': EXTRACT_WRAP, 'r': EXTRACT_RENAME, |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
461 | '': EXTRACT_WRAP} |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
462 | choices = ["You can:", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
463 | " * extract it Inside another directory", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
464 | " * extract it and Rename the directory", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
465 | " * extract it Here"] |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
466 | prompt = "What do you want to do? (I/r/h) " |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
467 | |
26 | 468 | def prep(self, archive_filename, entry_name): |
469 | question = ("%s contains one entry: %s." % | |
470 | (archive_filename, entry_name)) | |
471 | self.current_policy = (self.permanent_policy or | |
472 | self.ask_question(question)) | |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
473 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
474 | def ok_for_match(self): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
475 | return self.current_policy in (EXTRACT_RENAME, EXTRACT_HERE) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
476 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
477 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
478 | class RecursionPolicy(BasePolicy): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
479 | answers = {'o': RECURSE_ONCE, 'a': RECURSE_ALWAYS, 'n': RECURSE_NOT_NOW, |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
480 | 'v': RECURSE_NEVER, '': RECURSE_NOT_NOW} |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
481 | choices = ["You can:", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
482 | " * Always extract included archives", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
483 | " * extract included archives this Once", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
484 | " * choose Not to extract included archives", |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
485 | " * neVer extract included archives"] |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
486 | prompt = "What do you want to do? (a/o/N/v) " |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
487 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
488 | def __init__(self, options): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
489 | BasePolicy.__init__(self, options) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
490 | if options.recursive: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
491 | self.permanent_policy = RECURSE_ALWAYS |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
492 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
493 | def prep(self, current_filename, included_archives): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
494 | archive_count = len(included_archives) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
495 | if (self.permanent_policy is not None) or (archive_count == 0): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
496 | self.current_policy = self.permanent_policy or RECURSE_NOT_NOW |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
497 | return |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
498 | elif archive_count > 1: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
499 | question = ("%s contains %s other archive files." % |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
500 | (current_filename, archive_count)) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
501 | else: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
502 | question = ("%s contains another archive: %s." % |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
503 | (current_filename, included_archives[0])) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
504 | self.current_policy = self.ask_question(question) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
505 | if self.current_policy in (RECURSE_ALWAYS, RECURSE_NEVER): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
506 | self.permanent_policy = self.current_policy |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
507 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
508 | def ok_to_recurse(self): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
509 | return self.current_policy in (RECURSE_ALWAYS, RECURSE_ONCE) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
510 | |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
511 | |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
512 | extractor_map = {'application/x-tar': TarExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
513 | 'application/zip': ZipExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
514 | 'application/x-msdos-program': ZipExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
515 | 'application/x-debian-package': DebExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
516 | 'application/x-redhat-package-manager': RPMExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
517 | 'application/x-rpm': RPMExtractor, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
518 | 'application/x-cpio': CpioExtractor} |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
519 | |
16
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
520 | handlers = [FlatHandler, OverwriteHandler, MatchHandler, EmptyHandler, |
29794d4d41aa
[svn] There's now an entirely new object hierarchy for handlers, because the
brett
parents:
15
diff
changeset
|
521 | BombHandler] |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
522 | |
5 | 523 | class ExtractorApplication(object): |
524 | def __init__(self, arguments): | |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
525 | self.parse_options(arguments) |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
526 | self.setup_logger() |
5 | 527 | self.successes = [] |
528 | self.failures = [] | |
529 | ||
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
530 | def parse_options(self, arguments): |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
531 | parser = optparse.OptionParser( |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
532 | usage="%prog [options] archive [archive2 ...]", |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
533 | description="Intelligent archive extractor", |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
534 | version=VERSION_BANNER |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
535 | ) |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
536 | parser.add_option('-r', '--recursive', dest='recursive', |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
537 | action='store_true', default=False, |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
538 | help='extract archives contained in the ones listed') |
13
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
539 | parser.add_option('-q', '--quiet', dest='quiet', |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
540 | action='count', default=3, |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
541 | help='suppress warning/error messages') |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
542 | parser.add_option('-v', '--verbose', dest='verbose', |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
543 | action='count', default=0, |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
544 | help='be verbose/print debugging information') |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
545 | parser.add_option('-o', '--overwrite', dest='overwrite', |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
546 | action='store_true', default=False, |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
547 | help='overwrite any existing target directory') |
15
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
548 | parser.add_option('-f', '--flat', '--no-directory', dest='flat', |
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
549 | action='store_true', default=False, |
28dbd52a8bb8
[svn] Add a -f/--flat option, which will extract the archive contents into the
brett
parents:
14
diff
changeset
|
550 | help="don't put contents in their own directory") |
19 | 551 | parser.add_option('-l', '-t', '--list', '--table', dest='show_list', |
552 | action='store_true', default=False, | |
553 | help="list contents of archives on standard output") | |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
554 | parser.add_option('-n', '--noninteractive', dest='batch', |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
555 | action='store_true', default=False, |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
556 | help="don't ask how to handle special cases") |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
557 | self.options, filenames = parser.parse_args(arguments) |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
558 | if not filenames: |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
559 | parser.error("you did not list any archives") |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
560 | self.options.one_entry_policy = OneEntryPolicy(self.options) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
561 | self.options.recursion_policy = RecursionPolicy(self.options) |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
562 | self.archives = {os.path.realpath(os.curdir): filenames} |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
563 | |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
564 | def setup_logger(self): |
19 | 565 | self.logger = logging.getLogger('dtrx-log') |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
566 | handler = logging.StreamHandler() |
13
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
567 | # WARNING is the default. |
0a3ef1b9f6d4
[svn] Add options to tweak the logging level to taste.
brett
parents:
12
diff
changeset
|
568 | handler.setLevel(10 * (self.options.quiet - self.options.verbose)) |
19 | 569 | formatter = logging.Formatter("dtrx: %(levelname)s: %(message)s") |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
570 | handler.setFormatter(formatter) |
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
571 | self.logger.addHandler(handler) |
1 | 572 | |
5 | 573 | def get_extractor(self): |
574 | mimetype, encoding = mimetypes.guess_type(self.current_filename) | |
1 | 575 | try: |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
576 | extractor = extractor_map[mimetype] |
1 | 577 | except KeyError: |
14
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
578 | if encoding: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
579 | extractor = CompressionExtractor |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
580 | else: |
6f9e1bb59719
[svn] Add support for just decompressing files that are compressed. So, if you
brett
parents:
13
diff
changeset
|
581 | return "not a known archive type" |
5 | 582 | try: |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
583 | self.current_extractor = extractor(self.current_filename, encoding) |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
584 | except ExtractorError, error: |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
585 | return str(error) |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
586 | |
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
587 | def get_handler(self): |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
588 | for var_name in ('type', 'name'): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
589 | exec('content_%s = self.current_extractor.content_%s' % |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
590 | (var_name, var_name)) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
591 | if content_type == ONE_ENTRY: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
592 | self.options.one_entry_policy.prep(self.current_filename, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
593 | content_name) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
594 | for handler in handlers: |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
595 | if handler.can_handle(content_type, self.options): |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
596 | self.current_handler = handler(self.current_extractor, |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
597 | self.options) |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
598 | break |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
599 | |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
600 | def recurse(self): |
25
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
601 | archives = self.current_extractor.included_archives |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
602 | self.options.recursion_policy.prep(self.current_filename, archives) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
603 | if self.options.recursion_policy.ok_to_recurse(): |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
604 | for filename in archives: |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
605 | tail_path, basename = os.path.split(filename) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
606 | directory = os.path.join(self.current_directory, |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
607 | self.current_handler.target, tail_path) |
ef62f2f55eb8
[svn] Move policy-handling code into a dedicated set of classes. This makes
brett
parents:
23
diff
changeset
|
608 | self.archives.setdefault(directory, []).append(basename) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
609 | |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
610 | def report(self, function, *args): |
17
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
611 | try: |
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
612 | error = function(*args) |
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
613 | except (ExtractorError, IOError, OSError), exception: |
481a2b4be471
[svn] Lots of tests for various boundary cases, and slightly better handling for
brett
parents:
16
diff
changeset
|
614 | error = str(exception) |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
615 | self.logger.debug(traceback.format_exception(*sys.exc_info())) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
616 | if error: |
12
5d202467c589
[svn] Introduce a real logging system. Right now all this really gets us is the
brett
parents:
11
diff
changeset
|
617 | self.logger.error("%s: %s", self.current_filename, error) |
5 | 618 | return False |
619 | return True | |
620 | ||
19 | 621 | def record_status(self, success): |
622 | if success: | |
623 | self.successes.append(self.current_filename) | |
624 | else: | |
625 | self.failures.append(self.current_filename) | |
626 | ||
627 | def extract(self): | |
6
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
628 | while self.archives: |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
629 | self.current_directory, filenames = self.archives.popitem() |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
630 | for filename in filenames: |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
631 | os.chdir(self.current_directory) |
77043f4e6a9f
[svn] The big thing here is recursive extraction. Find archive files in the
brett
parents:
5
diff
changeset
|
632 | self.current_filename = filename |
20
69c93c3e6972
[svn] If the archive contains one directory with the "wrong" name, ask the user
brett
parents:
19
diff
changeset
|
633 | success = (self.report(self.get_extractor) and |
28
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
634 | self.report(self.current_extractor.extract) and |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
635 | self.report(self.get_handler) and |
4d88f2231d33
[svn] Change all the license notices from GPLv2 to GPLv3.
brett
parents:
27
diff
changeset
|
636 | self.report(self.current_handler.handle)) |
8
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
637 | if success: |
97388f5ff770
[svn] Make ExtractorApplication suck less. Now the strategies for handling
brett
parents:
7
diff
changeset
|
638 | self.recurse() |
19 | 639 | self.record_status(success) |
27 | 640 | self.options.one_entry_policy.permanent_policy = EXTRACT_WRAP |
19 | 641 | |
642 | def show_contents(self): | |
643 | for filename in self.current_extractor.get_filenames(): | |
644 | print filename | |
645 | ||
646 | def show_list(self): | |
647 | filenames = self.archives.values()[0] | |
648 | if len(filenames) > 1: | |
649 | header = "%s:\n" | |
650 | else: | |
651 | header = None | |
652 | for filename in filenames: | |
653 | if header: | |
654 | print header % (filename,), | |
655 | header = "\n%s:\n" | |
656 | self.current_filename = filename | |
657 | success = (self.report(self.get_extractor) and | |
658 | self.report(self.show_contents)) | |
659 | self.record_status(success) | |
660 | ||
661 | def run(self): | |
662 | if self.options.show_list: | |
663 | self.show_list() | |
664 | else: | |
665 | self.extract() | |
5 | 666 | if self.failures: |
667 | return 1 | |
668 | return 0 | |
669 | ||
1 | 670 | |
671 | if __name__ == '__main__': | |
5 | 672 | app = ExtractorApplication(sys.argv[1:]) |
673 | sys.exit(app.run()) |