From 838576fbe7a26199ba3f60c8ee7a2dfd567d330c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Eray=20=C3=96zkural?= Date: Thu, 20 Apr 2006 19:16:01 +0000 Subject: [PATCH] * improve the code so that it uses the factory pattern for both read and write * fix: set file size correctly for LZMA * add some comments, and also write the code for (fast) buffered decompression/compress but disable for now * support bzip2 that python guys forgot * the ID is now called ZIP_LZMA_BOGUS because 255 is a number that I made up, we'll get the real ID and add it soon. Until then, we are using this one and going to alpha with LZMA compression * build/install test successful with this commit (as was with the previous one?) * we are really close to 1.1 release because there are no important bugs with high priority right now. --- pisi/archive.py | 2 +- pisi/zipfileext.py | 328 ++++++++++++++++++++++----------------------- 2 files changed, 164 insertions(+), 166 deletions(-) diff --git a/pisi/archive.py b/pisi/archive.py index 6cd21765..4916faef 100644 --- a/pisi/archive.py +++ b/pisi/archive.py @@ -130,7 +130,7 @@ class ArchiveZip(ArchiveBase): attr.external_attr = self.symmagic self.zip_obj.writestr(attr, dest) else: - self.zip_obj.write(file_name, file_name, zipfileext.ZIP_LZMA) + self.zip_obj.write(file_name, file_name, zipfileext.ZIP_LZMA_BOGUS) zinfo = self.zip_obj.getinfo(file_name) zinfo.create_system = 3 diff --git a/pisi/zipfileext.py b/pisi/zipfileext.py index 57171a1b..a35feb52 100644 --- a/pisi/zipfileext.py +++ b/pisi/zipfileext.py @@ -9,10 +9,11 @@ # # Please read the COPYING file. # -# Authors: Eray Ozkural -# Faik Uygur +# Authors: Eray Ozkural +# Faik Uygur # -# Extends zipfile module with lzma and bzip2 support + +"""Extends zipfile module with lzma and bzip2 support""" # python standard library modules import os @@ -20,92 +21,174 @@ import struct import time import binascii +# we are really extending the zipfile module, not rewriting it. from zipfile import * try: import zlib except ImportError: zlib = None - try: - import bzip2 + import bz2 except ImportError: - bzip2 = None - + bz2 = None try: import pylzma except ImportError: pylzma = None ZIP_BZIP2 = 12 -ZIP_LZMA = 255 +ZIP_LZMA_BOGUS = 255 # FIXME: we are going to add the official ID when PKWARE gives it to us, and keep this one for a while -compression_methods = [ZIP_STORED, ZIP_DEFLATED, ZIP_BZIP2, ZIP_LZMA] +compression_methods = [ZIP_STORED, ZIP_DEFLATED, ZIP_BZIP2, ZIP_LZMA_BOGUS] -class FileEntry: - """File-like object used to access entries in a ZipFile""" + +class ZipFileEntry: + """File-like object used to access entries in a ZipFile, we're using a factory + design pattern that's a little better than switch blocks thrown about""" - def __init__(self, fp, length): + def __init__(self, fp): self.fp = fp self.readBytes = 0 - self.returnedBytes = 0 - self.length = length + self.returnedBytes = 0 #FIXME: we don't use studlyCaps in python self.finished = 0 - -class LzmaFileEntry(FileEntry): - """File-like object used to read a LZMA entry in a ZipFile""" - - def __init__(self, fp, length): - FileEntry.__init__(self, fp, length) - self.decomp = pylzma.decompressobj() self.buffer = "" - + def tell(self): return self.returnedBytes - - def read(self, n=None): + + def read(self, compress_size, n=None): + """read the whole file, or n bytes and return the decompressed stuff + does it in a buffered fashion""" + self.length = compress_size if self.finished: return "" if n is None: - result = [self.buffer,] - result.append(self.decomp.decompress(self.fp.read(self.length - self.readBytes))) - result.append(self.decomp.flush()) - self.buffer = "" + compr_data = self.fp.read(self.length) + data = self.decompress(compr_data) self.finished = 1 - result = "".join(result) - self.returnedBytes += len(result) - return result + self.returnedBytes += len(data) + return data else: - while len(self.buffer) < n: - data = self.fp.read(min(n, 1024, self.length - self.readBytes)) - self.readBytes += len(data) + # FIXME: must always decompress in streaming mode, not just when n is given + while len(self.buffer) < n: + compr_data = self.fp.read(min(n, 1024 * 8, self.length - self.readBytes)) + self.readBytes += len(compr_data) if not data: - result = self.buffer + self.decomp.decompress() + self.decomp.flush() + result = self.buffer # + self.flush_decompressor() self.finished = 1 self.buffer = "" self.returnedBytes += len(result) return result else: - self.buffer += self.decomp.decompress(data) + self.buffer += self.decompress(data) result = self.buffer[:n] self.buffer = self.buffer[n:] self.returnedBytes += len(result) return result - + + #TODO: use this instead of bulk compression + def write(self, infile): + self.CRC = 0 + self.file_size = 0 + self.compress_size = 0 + while 1: + buf = infile.read(1024 * 8) + if not buf: + break + self.file_size = self.file_size + len(buf) + self.CRC = binascii.crc32(buf, self.CRC) + if compressor: + buf = self.compress(buf) + self.compress_size = self.compress_size + len(buf) + self.fp.write(buf) + return self.compress_size +# if cmpr: + # buf = cmpr.flush() + # compress_size = compress_size + len(buf) + # self.fp.write(buf) + def close(self): self.finished = 1 del self.fp -class ZipFileEntry(FileEntry): + +#TODO: test deflate and bzip2 support thoroughly +class DeflatedZipFileEntry(ZipFileEntry): + """File-like object used to read a deflated entry in a ZipFile""" + + def __init__(self, fp): + ZipFileEntry.__init__(self, fp) + self.decomp = zlib.decompressobj(-15) + self.compressor = zlib.compressobj(zlib.Z_DEFAULT_COMPRESSION, + zlib.DEFLATED, -15) + + def decompress(self, compr_data): + return self.decomp.decompress(compr_data) + self.decomp.decompress("Z") + self.decomp.flush() + + def compress(self, data): + # big deal with flush.... + return self.compressor.compress(data) + compressor.flush() + + def write(self, infile): + data = infile.read() + compressed = self.compress(data) + self.fp.write(compressed) + return len(compressed) + + +class Bzip2ZipFileEntry(ZipFileEntry): + """File-like object used to read a BZIP2 entry in a ZipFile""" + + def __init__(self, fp): + ZipFileEntry.__init__(self, fp) + self.decomp = bz2.Decompressor() + + def decompress(self, compr_data): + return self.decomp.decompress(compr_data) + + def write(self, infile): + data = infile.read() + self.file_size = len(data) + self.CRC = binascii.crc32(data) + compressed = bz2.compress(data) + self.fp.write(compressed) + self.compress_size = len(compressed) + return self.compress_size + + +class LzmaZipFileEntry(ZipFileEntry): + """File-like object used to read a LZMA entry in a ZipFile""" + + def __init__(self, fp): + ZipFileEntry.__init__(self, fp) + self.decompressor = pylzma.decompressobj() + + def decompress(self, compr_data): + return self.decompressor.decompress(compr_data) + self.decompressor.flush() + + def write(self, infile): + #TODO: use the buffered write in superclass + data = infile.read() + self.file_size = len(data) + self.CRC = binascii.crc32(data) + compressed = pylzma.compress(data, eos=1) + self.fp.write(compressed) + self.compress_size = len(compressed) + return len(compressed) + + +class StoredZipFileEntry(ZipFileEntry): """File-like object used to read an uncompressed entry in a ZipFile""" - def __init__(self, fp, length): - FileEntry.__init__(self, fp, length) + def __init__(self, fp): + ZipFileEntry.__init__(self, fp) def tell(self): return self.readBytes - def read(self, n=None): + def read(self, length, n=None): + self.length = length if n is None: n = self.length - self.readBytes if n == 0 or self.finished: @@ -122,53 +205,32 @@ class ZipFileEntry(FileEntry): del self.fp -class DeflatedZipFileEntry(FileEntry): - """File-like object used to read a deflated entry in a ZipFile""" - - def __init__(self, fp, length): - FileEntry.__init__(self, fp, length) - self.decomp = zlib.decompressobj(-15) - self.buffer = "" - - def tell(self): - return self.returnedBytes - - def read(self, n=None): - if self.finished: - return "" - if n is None: - result = [self.buffer,] - result.append(self.decomp.decompress(self.fp.read(self.length - self.readBytes))) - result.append(self.decomp.decompress("Z")) - result.append(self.decomp.flush()) - self.buffer = "" - self.finished = 1 - result = "".join(result) - self.returnedBytes += len(result) - return result - else: - while len(self.buffer) < n: - data = self.fp.read(min(n, 1024, self.length - self.readBytes)) - self.readBytes += len(data) - if not data: - result = self.buffer + self.decomp.decompress("Z") + self.decomp.flush() - self.finished = 1 - self.buffer = "" - self.returnedBytes += len(result) - return result - else: - self.buffer += self.decomp.decompress(data) - result = self.buffer[:n] - self.buffer = self.buffer[n:] - self.returnedBytes += len(result) - return result - - def close(self): - self.finished = 1 - del self.fp - class ZipFileExt(ZipFile): + def build_file_entry(self, compress_type): + "a small factory method" + if compress_type == ZIP_STORED: + return StoredZipFileEntry(self.fp) + elif compress_type == ZIP_DEFLATED: + if not zlib: + raise RuntimeError, \ + "Compression requires the missing %s module" % "zlib" + return DeflatedZipFileEntry(self.fp) + elif compress_type == ZIP_BZIP2: + if not bz2: + raise RuntimeError, \ + "Compression method requires the missing %s module" % "bz2" + return Bzip2ZipFileEntry(self.fp) + elif compress_type == ZIP_LZMA_BOGUS: + if not pylzma: + raise RuntimeError, \ + "Compression method requires the missing %s module" % "pylzma" + return LzmaZipFileEntry(self.fp) + else: + raise BadZipfile, \ + "Unsupported compression method %d for file %s" % \ + (compress_type, name) + def _writecheck(self, zinfo): """Check for errors before writing a file to the archive.""" if zinfo.filename in self.NameToInfo: @@ -179,20 +241,8 @@ class ZipFileExt(ZipFile): if not self.fp: raise RuntimeError, \ "Attempt to write ZIP archive that was already closed" - if zinfo.compress_type not in compression_methods: - raise RuntimeError, \ - "Compression method %s is not supported" % zinfo.compress_type - if zinfo.compress_type == ZIP_DEFLATED and not zlib: - raise RuntimeError, \ - "Compression requires the (missing) zlib module" - if zinfo.compress_type == ZIP_BZIP2 and not bzip2: - raise RuntimeError, \ - "Compression requires the (missing) bzip2 module" - if zinfo.compress_type == ZIP_LZMA and not pylzma: - raise RuntimeError, \ - "Compression requires the (missing) pylzma module" - def write(self, filename, arcname=None, compress_type=None): + def write(self, filename, arcname=None, compress_type=ZIP_DEFLATED): """Put the bytes from filename into the archive under the name arcname.""" st = os.stat(filename) @@ -219,90 +269,38 @@ class ZipFileExt(ZipFile): self.fp.write(zinfo.FileHeader()) zinfo.file_offset = self.fp.tell() # Start of file bytes - # really compress - if zinfo.compress_type == ZIP_DEFLATED or zinfo.compress_type == ZIP_STORED: - if zinfo.compress_type == ZIP_DEFLATED: - cmpr = zlib.compressobj(zlib.Z_DEFAULT_COMPRESSION, - zlib.DEFLATED, -15) - else: - cmpr = None + # build a zipfileentry object from factory, and write compressed data + fileentry = self.build_file_entry(zinfo.compress_type) + fileentry.write(fp) + # update zinfo + zinfo.CRC = fileentry.CRC + zinfo.file_size = fileentry.file_size + zinfo.compress_size = fileentry.compress_size - while 1: - buf = fp.read(1024 * 8) - if not buf: - break - file_size = file_size + len(buf) - CRC = binascii.crc32(buf, CRC) - if cmpr: - buf = cmpr.compress(buf) - compress_size = compress_size + len(buf) - self.fp.write(buf) - fp.close() - if cmpr: - buf = cmpr.flush() - compress_size = compress_size + len(buf) - self.fp.write(buf) - zinfo.compress_size = compress_size - else: - zinfo.compress_size = file_size - - elif zinfo.compress_type == ZIP_LZMA: - # unfortunately pylzma.compressobj is not implemented yet. - # So in order to calculate the CRC, we are going to read - # all the file at once until it is implemented. - buf = fp.read() - CRC = binascii.crc32(buf, CRC) - compressed = pylzma.compress(buf, eos=1) - self.fp.write(compressed) - zinfo.compress_size = len(compressed) - - zinfo.CRC = CRC - zinfo.file_size = file_size # Seek backwards and write CRC and file sizes position = self.fp.tell() # Preserve current position in file self.fp.seek(zinfo.header_offset + 14, 0) self.fp.write(struct.pack("