Use urlgrabber library to fetch stuff.. Spiffy!

This commit is contained in:
A. Murat Eren
2007-11-21 18:43:27 +00:00
parent 77f7399b5a
commit 62c5377255
+103 -289
View File
@@ -16,18 +16,11 @@ all kinds of things: source tarballs, index files, packages, and God
knows what.""" knows what."""
# python standard library modules # python standard library modules
import urllib2
import urllib
import ftplib
import os import os
import socket
import sys import sys
import mimetypes import time
import mimetools
import base64 import base64
import shutil import shutil
import time
import httplib
import gettext import gettext
__trans = gettext.translation('pisi', fallback=True) __trans = gettext.translation('pisi', fallback=True)
@@ -42,330 +35,151 @@ import pisi.uri
class FetchError(pisi.Error): class FetchError(pisi.Error):
pass pass
class RangeError(pisi.Error):
pass
# helper functions class UIHandler:
def fetch_url(url, destdir, progress=None, resume=True): def __init__(self, progress):
fetch = Fetcher(url, destdir) self.filename = None
if not resume: self.url = None
fetch.resume = False self.basename = None
fetch.progress = progress self.downloaded_size = 0
fetch.fetch() self.percent = None
self.rate = 0.0
self.eta = '--:--:--'
self.symbol = None
self.last_updated = 0
self.exist_size = 0
def start(self, archive, url, basename, total_size, text):
if os.path.exists(archive):
self.exist_size = os.path.getsize(archive)
self.filename = basename
self.url = url
self.basename = basename
self.total_size = total_size
self.text = text
self.now = lambda: time.time()
self.Tdiff = lambda: self.now() - self.s_time
self.s_time = self.now()
def update(self, size):
self.size = size
self.percent = (size * 100.0) / self.total_size
if int(self.now()) != int(self.last_updated) and size > 0:
self.rate, self.symbol = util.human_readable_rate((size - self.exist_size) / (self.now() - self.s_time))
self.eta = '%02d:%02d:%02d' %\
tuple([i for i in time.gmtime((self.Tdiff() * (100 - self.percent)) / self.percent)[3:6]])
self._update_ui()
def end(self, read):
pass
def _update_ui(self):
ctx.ui.display_progress(operation = "fetching",
percent = self.percent,
filename = self.filename,
total_size = self.total_size,
downloaded_size = self.size,
rate = self.rate,
eta = self.eta,
symbol = self.symbol)
self.last_updated = self.now()
class Fetcher: class Fetcher:
"""Fetcher can fetch a file from various sources using various """Fetcher can fetch a file from various sources using various
protocols.""" protocols."""
def __init__(self, url, destdir, resume = True): def __init__(self, url, destdir):
if not isinstance(url, pisi.uri.URI): if not isinstance(url, pisi.uri.URI):
url = pisi.uri.URI(url) url = pisi.uri.URI(url)
if ctx.config.get_option("authinfo"): if ctx.config.get_option("authinfo"):
url.set_auth_info(ctx.config.get_option("authinfo")) url.set_auth_info(ctx.config.get_option("authinfo"))
self.resume = resume self.url = url
self.scheme = url.scheme() self.destdir = destdir
self.url = url
self.destdir = destdir
util.check_dir(self.destdir)
self.eta = '??:??:??'
self.percent = 0
self.rate = 0.0
self.progress = None self.progress = None
self.exist_size = 0
util.check_dir(self.destdir)
def fetch (self): def fetch (self):
"""Return value: Fetched file's full path..""" """Return value: Fetched file's full path.."""
# import urlgrabber module
try:
import urlgrabber
except ImportError:
raise FetchError(_('Urlgrabber needs to be installed to run this command'))
if not self.url.filename(): if not self.url.filename():
self.err(_('Filename error')) FetchError(_('Filename error'))
if not os.access(self.destdir, os.W_OK): if not os.access(self.destdir, os.W_OK):
self.err(_('Access denied to write to destination directory: "%s"') % (self.destdir)) FetchError(_('Access denied to write to destination directory: "%s"') % (self.destdir))
archive_file = os.path.join(self.destdir, self.url.filename()) archive_file = os.path.join(self.destdir, self.url.filename())
if os.path.exists(archive_file) and not os.access(archive_file, os.W_OK): if os.path.exists(archive_file) and not os.access(archive_file, os.W_OK):
self.err(_('Access denied to destination file: "%s"') % (archive_file)) FetchError(_('Access denied to destination file: "%s"') % (archive_file))
partial_file = archive_file + '.part' partial_file = archive_file + '.part'
if self.url.is_local_file(): urlgrabber.urlgrab(self.url.get_uri(),
self.fetchLocalFile(partial_file) partial_file,
else: progress_obj = UIHandler(self.progress),
self.fetchRemoteFile(partial_file) http_headers = self._get_http_headers(),
ftp_headers = self._get_ftp_headers(),
proxies = self._get_proxies(),
user_agent = 'PiSi Fetcher/' + pisi.__version__,
reget = 'check_timestamp')
if os.stat(partial_file).st_size == 0: if os.stat(partial_file).st_size == 0:
os.remove(partial_file) os.remove(partial_file)
self.err(_('A problem occured. Please check the archive address and/or permissions again.')) FetchError(_('A problem occurred. Please check the archive address and/or permissions again.'))
shutil.move(partial_file, archive_file) shutil.move(partial_file, archive_file)
return archive_file return archive_file
def _do_grab(self, fileURI, dest, total_size): def _get_http_headers(self):
bs, tt, = 1024, int(time.time()) headers = []
s_time = time.time() if self.url.auth_info() and (self.url.scheme() == "http" or self.url.scheme() == "https"):
Tdiff = lambda: time.time() - s_time
downloaded_size = exist_size = self.exist_size
symbol = 'B/s'
st = time.time()
chunk = fileURI.read(bs)
downloaded_size += len(chunk)
if self.progress:
p = self.progress(total_size, exist_size)
self.percent = p.update(downloaded_size)
self.complete = False
while chunk:
dest.write(chunk)
chunk = fileURI.read(bs)
downloaded_size += len(chunk)
ct = time.time()
if int(tt) != int(ct):
self.rate = (downloaded_size - exist_size) / (ct - st)
if self.percent:
self.eta = '%02d:%02d:%02d' %\
tuple([i for i in time.gmtime((Tdiff() * (100 - self.percent)) / self.percent)[3:6]])
self.rate, symbol = util.human_readable_rate(self.rate)
tt = time.time()
if self.progress:
if p.update(downloaded_size):
self.percent = p.percent
if not self.complete:
ctx.ui.display_progress(operation = "fetching",
percent = self.percent,
filename = self.url.filename(),
total_size = total_size,
downloaded_size = downloaded_size,
rate = self.rate,
eta = self.eta,
symbol = symbol)
if self.percent == 100: #FIXME: will be superseded by a
self.complete = True # working progress interface
dest.close()
def fetchLocalFile (self, archive_file):
url = self.url
if not os.access(url.path(), os.F_OK):
self.err(_('No such file or no permission to read'))
dest = open(archive_file, 'w')
total_size = os.path.getsize(url.path())
fileObj = open(url.path())
self._do_grab(fileObj, dest, total_size)
def fetchRemoteFile (self, archive_file):
if os.path.exists(archive_file) and self.resume:
if self.scheme == 'http' or self.scheme == 'https' or self.scheme == 'ftp':
self.exist_size = os.path.getsize(archive_file)
dest = open(archive_file, 'ab')
else:
dest = open(archive_file, 'wb')
uri = self.url.get_uri()
flag = 1
try:
try:
try:
fileObj = urllib2.urlopen(self.formatRequest(urllib2.Request(uri)))
except RangeError:
ctx.ui.info(_('Requested range not satisfiable, starting again.'))
dest = open(archive_file, 'wb')
self.exist_size = 0
fileObj = urllib2.urlopen(self.formatRequest(urllib2.Request(uri)))
headers = fileObj.info()
flag = 0
except ValueError, e:
self.err(_('Cannot fetch %s; value error: %s') % (uri, e))
except urllib2.HTTPError, e:
self.err(_('Cannot fetch %s; %s') % (uri, e))
except urllib2.URLError, e:
self.err(_('Please check your network connections and try again. (%s)') % e[-1][-1])
except OSError, e:
self.err(_('Cannot fetch %s; %s') % (uri, e))
except httplib.HTTPException, e:
self.err(_('Cannot fetch %s; (%s): %s') % (uri, e.__class__.__name__, e))
finally:
if flag:
if os.stat(archive_file).st_size == 0:
os.remove(archive_file)
try:
total_size = int(headers['Content-Length']) + self.exist_size
except KeyboardInterrupt:
raise
except Exception, e: #FIXME: what exception could we catch here, replace with that.
total_size = 0
self._do_grab(fileObj, dest, total_size)
def formatRequest(self, request):
if self.url.auth_info():
enc = base64.encodestring('%s:%s' % self.url.auth_info()) enc = base64.encodestring('%s:%s' % self.url.auth_info())
request.add_header('Authorization', 'Basic %s' % enc) headers.append(('Authorization', 'Basic %s' % enc),)
return tuple(headers)
range_handlers = { def _get_ftp_headers(self):
'http' : HTTPRangeHandler, headers = []
'https': HTTPRangeHandler, if self.url.auth_info() and self.url.scheme() == "ftp":
'ftp' : FTPRangeHandler enc = base64.encodestring('%s:%s' % self.url.auth_info())
} headers.append(('Authorization', 'Basic %s' % enc),)
return tuple(headers)
if self.exist_size and range_handlers.has_key(self.scheme): def _get_proxies(self):
opener = urllib2.build_opener(range_handlers.get(self.scheme)()) proxies = {}
urllib2.install_opener(opener)
request.add_header('Range', 'bytes=%d-' % self.exist_size)
proxy_handler = None
if ctx.config.values.general.http_proxy and self.url.scheme() == "http": if ctx.config.values.general.http_proxy and self.url.scheme() == "http":
http_proxy = ctx.config.values.general.http_proxy proxies[pisi.uri.URI(http_proxy).scheme()] = ctx.config.values.general.http_proxy
proxy_handler = urllib2.ProxyHandler({pisi.uri.URI(http_proxy).scheme(): http_proxy})
elif ctx.config.values.general.https_proxy and self.url.scheme() == "https": if ctx.config.values.general.https_proxy and self.url.scheme() == "https":
https_proxy = ctx.config.values.general.https_proxy proxies[pisi.uri.URI(https_proxy).scheme()] = ctx.config.values.general.https_proxy
proxy_handler = urllib2.ProxyHandler({pisi.uri.URI(https_proxy): https_proxy})
elif ctx.config.values.general.ftp_proxy and self.url.scheme() == "ftp": if ctx.config.values.general.ftp_proxy and self.url.scheme() == "ftp":
ftp_proxy = ctx.config.values.general.ftp_proxy proxies[pisi.uri.URI(ftp_proxy).scheme()] = ctx.config.values.general.ftp_proxy
proxy_handler = urllib2.ProxyHandler({pisi.uri.URI(http_proxy): ftp_proxy})
if proxy_handler: if self.url.scheme() in proxies:
ctx.ui.info(_("Proxy configuration has been found for '%s' protocol") % self.url.scheme()) ctx.ui.info(_("Proxy configuration has been found for '%s' protocol") % self.url.scheme())
opener = urllib2.build_opener(proxy_handler)
urllib2.install_opener(opener)
return request return proxies
def err (self, error):
raise FetchError(error)
class HTTPRangeHandler(urllib2.BaseHandler):
"""
to override the urllib2 error: 'Error 206: Partial Content'
this reponse from the HTTP server is already what we expected to get.
Don't give up, resume downloading..
"""
def http_error_206(self, request, fp, errcode, msg, headers):
return urllib.addinfourl(fp, headers, request.get_full_url())
def http_error_416(self, request, fp, errcode, msg, headers):
# HTTP 1.1's 'Range Not Satisfiable' error..
raise RangeError
class FTPRangeHandler(urllib2.FTPHandler): # helper function
""" def fetch_url(url, destdir, progress=None):
FTP Range support.. fetch = Fetcher(url, destdir)
""" fetch.progress = progress
def ftp_open(self, req): fetch.fetch()
host = req.get_host()
host, port = urllib.splitport(host)
if port is None:
port = ftplib.FTP_PORT
try:
host = socket.gethostbyname(host)
except socket.error, msg:
raise FetchError(msg)
path, attrs = urllib.splitattr(req.get_selector())
dirs = path.split('/')
dirs = map(urllib.unquote, dirs)
dirs, f = dirs[:-1], dirs[-1]
if dirs and not dirs[0]:
dirs = dirs[1:]
try:
fw = self.connect_ftp('', '', host, port, dirs)
t = f and 'I' or 'D'
for attr in attrs:
attr, value = urllib.splitattr(attr)
if attr.lower() == 'type' and \
value in ('a', 'A', 'i', 'I', 'd', 'D'):
t = value.upper()
rawr = req.headers.get('Range', None)
if rawr:
rest = int(rawr.split("=")[1].rstrip("-"))
else:
rest = 0
fp, retrlen = fw.retrfile(f, t, rest)
fb, lb = rest, retrlen
if retrlen is None or retrlen == 0:
raise RangeError
retrlen = lb - fb
if retrlen < 0:
# beginning of range is larger than file
raise RangeError
headers = ''
mtype = mimetypes.guess_type(req.get_full_url())[0]
if mtype:
headers += 'Content-Type: %s\n' % mtype
if retrlen is not None and retrlen >= 0:
headers += 'Content-Length: %d\n' % retrlen
try:
import cStringIO as StringIO
except ImportError, msg:
import StringIO
return urllib.addinfourl(fp, mimetools.Message(StringIO.StringIO(headers)), req.get_full_url())
except ftplib.all_errors, msg:
raise IOError, (_('ftp error'), msg), sys.exc_info()[2]
def connect_ftp(self, user, passwd, host, port, dirs):
fw = ftpwrapper(user, passwd, host, port, dirs)
return fw
class ftpwrapper(urllib.ftpwrapper):
def retrfile(self, file, type, rest=None):
self.endtransfer()
if type in ('d', 'D'): cmd = 'TYPE A'; isdir = 1
else: cmd = 'TYPE ' + type; isdir = 0
try:
self.ftp.voidcmd(cmd)
except ftplib.all_errors:
self.init()
self.ftp.voidcmd(cmd)
conn = None
if file and not isdir:
try:
self.ftp.nlst(file)
except ftplib.error_perm, reason:
raise IOError, (_('ftp error'), reason), sys.exc_info()[2]
# Restore the transfer mode!
self.ftp.voidcmd(cmd)
try:
cmd = 'RETR ' + file
conn = self.ftp.ntransfercmd(cmd, rest)
except ftplib.error_perm, reason:
if str(reason)[:3] == '501':
# workaround for REST not suported error
fp, retrlen = self.retrfile(file, type)
# WTF? No global (RangeableFileObject) found. RangeableFileObject only defined in urlgrabber / caglar
fp = RangeableFileObject(fp, (rest,''))
return (fp, retrlen)
elif str(reason)[:3] != '550':
raise IOError, (_('ftp error'), reason), sys.exc_info()[2]
if not conn:
self.ftp.voidcmd('TYPE A')
if file: cmd = 'LIST ' + file
else: cmd = 'LIST'
conn = self.ftp.ntransfercmd(cmd)
self.busy = 1
return (urllib.addclosehook(conn[0].makefile('rb'),
self.endtransfer), conn[1])