diff --git a/bin/resync b/bin/resync old mode 100644 new mode 100755 index e69de29..10b0902 --- a/bin/resync +++ b/bin/resync @@ -0,0 +1,282 @@ +#!/usr/bin/python +""" +resync: The ResourceSync command line client + +Created by Simeon Warner on 2012-04... +""" + +import logging +import logging.config +import optparse +import sys + +from resync.client import Client, ClientFatalError +from resync.utils import UTCFormatter + +DEFAULT_CLIENT_LOGFILE = 'resync-client.log' + +def init_logging(to_file=False, logfile=None, human=True, + verbose=False, eval_mode=False): + """Initialize logging + + Use of log levels: + DEBUG - very verbose, for evaluation of output (-e) + INFO - verbose, only seen by users if they ask for it (-v) + WARNING - messages output messages to console + """ + + fmt = '%(asctime)s | %(name)s | %(levelname)s | %(message)s' + formatter = UTCFormatter(fmt) + + if human: + # Create a special handler designed just for human readable output + hh = logging.StreamHandler() + hh.setLevel( logging.INFO if (verbose) else logging.WARNING ) + hh.setFormatter(logging.Formatter(fmt='%(message)s')) + if to_file: + if (logfile is None): + logfile = DEFAULT_CLIENT_LOGFILE + fh = logging.FileHandler(filename=logfile, mode='a') + fh.setFormatter(formatter) + fh.setLevel( logging.DEBUG if (eval_mode) else logging.INFO ) + + loggers = ['client','inventory_builder','sitemap'] + for logger in loggers: + log = logging.getLogger(logger) + log.setLevel(logging.DEBUG) #control at handler instead + if human: + log.addHandler(hh) + if to_file: + log.addHandler(fh) + + log=logging.getLogger('client') + if (to_file): + log.info("Writing detailed log to %s" % (logfile)) + + +def count_true_args(*args): + """Count number of list of arguments that evaluate True""" + count=0 + for arg in args: + if (arg): + count+=1 + return(count) + +def parse_links(args_link): + links={} + if (args_link is not None): + for link_str in args_link: + try: + (href, atts) = parse_link(link_str) + links[href]=atts + except ValueError as e: + raise ClientFatalError("Bad --link option '%s' (%s)"%(link_str,str(e))) + return(links) + +def parse_link(link_str): + """Parse --link option to add to capabilities + + Input string of the form: href,att1=val1,att2=val2 + """ + atts={} + try: + segs = link_str.split(',') + # First segment is relation + href = segs.pop(0) + # Remaining segments are attributes + for term in segs: + (k,v)=term.split('=') + atts[k]=v + except ValueError as e: + raise ClientFatalError("Bad component of --link option '%s' (%s)"%(link_str,str(e))) + return(href,atts) + +def main(): + + if (sys.version_info < (2,6)): + sys.exit("This program requires python version 2.6 or later") + + # Options and arguments + p = optparse.OptionParser(description='ResourceSync sync script', + usage='usage: %prog [options] uri_path local_path (-h for help)', + add_help_option=False) + + # Modes + # a. modes using remote sitemap/resources + rem = p.add_option_group('REMOTE MODES', + 'These modes use a remote source that is specified in a set of uri=path mappings ' + 'and potentially also using an explicit --sitemap location. The default mode is ' + '--baseline') + rem.add_option('--baseline', '-b', action='store_true', + help='baseline sync of resources from remote source (src) to local filesystem (dst)') + rem.add_option('--inc', '-i', action='store_true', + help='incremental sync of resources from remote source (src) to local filesystem (dst)') + rem.add_option('--audit', '-a', action='store_true', + help="audit sync state of destination wrt source") + rem.add_option('--parse', '-p', action='store_true', + help="just parse the remote sitemap (from mapping or explicit --sitemap)") + rem.add_option('--explore', action='store_true', + help="just explore links between sitemap and changesets starting from " + "mapping or explicit --sitemap)") + # b. modes based solely on files on local disk + loc = p.add_option_group('LOCAL MODES', + 'These modes act on files on the local disk') + loc.add_option('--write', '-w', action='store_true', + help="write a sitemap based on files on disk using uri=path mappings " + "in reverse to calculate URIs from the local paths. To STDOUT by " + "default, override with --outfile") + loc.add_option('--changeset', '-c', action='store_true', + help="write a changeset sitemap by comparison of a reference sitemap " + "(specify file with --reference) and either files on disk (using " + "the map provided) or a second sitemap (specify file with " + "--newreference). Otherwise follows --write options. Also accepts " + "the --empty option (with no mapping) to write and empy changeset.") + + # Specification of map between remote URI and local file paths, and remote sitemap + nam = p.add_option_group('FILE/URI NAMING OPTIONS') + nam.add_option('--outfile', type=str, action='store', + help='write sitemap to specified file rather than STDOUT') + nam.add_option('--sitemap', type=str, action='store', + help='explicitly set sitemap name, overriding default sitemap.xml ' + 'appended to first source URI specified in the mappings') + nam.add_option('--reference', type=str, action='store', + help='reference sitemap name for --changeset calculation') + nam.add_option('--newreference', type=str, action='store', + help='updated reference sitemap name for --changeset calculation') + nam.add_option('--dump', metavar="DUMPFILE", type=str, action='store', + help='write dump to specified file for --write or --changeset') + nam.add_option('--changeset-uri', type=str, action='store', + help='explicitly set the changeset URI that will be use in --inc mode, ' + 'overrides process of getting this from the sitemap') + + # Options that apply to multiple modes + opt = p.add_option_group('OPTIONS') + opt.add_option('--checksum', action='store_true', + help="use checksum (md5) in addition to last modification time and size") + opt.add_option('--delete', action='store_true', + help="allow files on destination to be deleted") + opt.add_option('--exclude', type=str, action='append', + help="exclude resources with URI or filename matching pattern " + "(repeat option for multiple excludes)") + opt.add_option('--empty', action='store_true', + help="combine with --changeset to write and empty changeset, perhaps with links") + opt.add_option('--link', type=str, action='append', + help="add discovery links to the output sitemap, " + "format: href,att1=val1,att2=val2 " + "(repeat option for multiple links)") + opt.add_option('--prev', type=str, action='store', + help="add prev discovery link to this URI to the output sitemap") + opt.add_option('--next', type=str, action='store', + help="add next discovery link to this URI to the output sitemap") + opt.add_option('--current', type=str, action='store', + help="add current discovery link to this URI to the output sitemap") + opt.add_option('--multifile', '-m', action='store_true', + help="disable reading and output of sitemapindex for multifile sitemap") + opt.add_option('--noauth', action='store_true', + help="disable checking of URL paths to ensure that the sitemaps refer " + "only to resources on the same server/sub-path etc. Use with care.") + opt.add_option('--warc', action='store_true', + help="write dumps in WARC format (instead of ZIP+Sitemap default)") + opt.add_option('--dryrun', '-n', action='store_true', + help="don't update local resources, say what would be done") + opt.add_option('--ignore-failures', action='store_true', + help="continue past download failures") + # These likely only useful for experimentation + opt.add_option('--max-sitemap-entries', type=int, action='store', + help="override default size limits") + # Want these to show at the end + opt.add_option('--verbose', '-v', action='store_true', + help="verbose") + opt.add_option('--logger', '-l', action='store_true', + help="create detailed log of client actions (will write " + "to %s unless specified with --logfile" % + DEFAULT_CLIENT_LOGFILE) + opt.add_option('--logfile', type='str', action='store', + help="create detailed log of client actions") + opt.add_option('--eval', '-e', action='store_true', + help="output evaluation of source/client synchronization performance... " + "be warned, this is very verbose") + opt.add_option('--help', '-h', action='help', + help="this help") + + (args, map) = p.parse_args() + + # Implement exclusive arguments and default --baseline (support for exclusive + # groups in argparse is incomplete is python2.6) + if (not args.baseline and not args.inc and not args.audit and + not args.parse and not args.explore and not args.write and + not args.changeset): + args.baseline=True + elif (count_true_args(args.baseline,args.inc,args.audit,args.parse, + args.explore,args.write,args.changeset)>1): + p.error("Only one of --baseline, --inc, --audit, --parse, --explore, --write, --changeset modes allowed") + + # Configure logging module and create logger instance + init_logging( to_file=args.logger, logfile=args.logfile, + verbose=args.verbose, eval_mode=args.eval ) + + c = Client( checksum=args.checksum, + verbose=args.verbose, + dryrun=args.dryrun ) + + try: + if (map): + # Mappings apply to (almost) everything + c.set_mappings(map) + if (args.sitemap): + c.sitemap_name=args.sitemap + if (args.warc): + c.dump_format='warc' + if (args.exclude): + c.exclude_patterns=args.exclude + if (args.multifile): + c.allow_multifile=not args.multifile + if (args.noauth): + c.noauth=args.noauth + if (args.max_sitemap_entries): + c.max_sitemap_entries=args.max_sitemap_entries + if (args.ignore_failures): + c.ignore_failures=args.ignore_failures + + # Links apply to anything that writes sitemaps + links = parse_links(args.link) + if (args.prev): + links[args.prev] = { 'rel': 'prev http://www.openarchives.org/rs/changeset' } + if (args.next): + links[args.next] = { 'rel': 'next http://www.openarchives.org/rs/changeset' } + if (args.current): + links[args.current] = { 'rel': 'current http://www.openarchives.org/rs/changeset' } + + # Finally, do something... + if (args.baseline or args.audit): + c.baseline_or_audit(allow_deletion=args.delete, + audit_only=args.audit) + elif (args.inc): + c.incremental(allow_deletion=args.delete, + changeset_uri=args.changeset_uri) + elif (args.parse): + c.parse_sitemap() + elif (args.explore): + c.explore_links() + elif (args.write): + c.write_sitemap(outfile=args.outfile, + capabilities=links, + dump=args.dump) + elif (args.changeset): + if (not args.reference and not args.empty): + p.error("Must supply --reference sitemap for --changeset, or --empty") + c.changeset_sitemap(ref_sitemap=args.reference, + newref_sitemap=( args.newreference if (args.newreference) else None ), + empty=args.empty, + outfile=args.outfile, + capabilities=links, + dump=args.dump) + else: + p.error("Unknown mode requested") + # Any problem we expect will come as a ClientFatalError, anything else + # is... an exception ;-) + except ClientFatalError as e: + sys.stderr.write("\nFatalError: " + str(e) + "\n") + +if __name__ == '__main__': + main() diff --git a/resync/changelist.py b/resync/changelist.py index e69de29..c3d72e9 100644 --- a/resync/changelist.py +++ b/resync/changelist.py @@ -0,0 +1,55 @@ +"""ResourceSync ChangeList object + +A ChangeList is a list of resource descriptions which includes +both metadata associated with the resource at some point in +time, and also metadata about a change that may have occurred +to bring the resource to that states. These descriptions +are Resource objects. + +Different from an inventory, a changelist may include multiple +descriptions for the same resource. The changelist is ordered +from first entry to last entry. + +Different from an inventory, dereference by a URI yields a +ChangeList containing descriptions pertaining to that +particular resource. +""" + +import collections + +from resource_container import ResourceContainer +from resource import Resource + +class ChangeList(ResourceContainer): + """Class representing an Change List""" + + def __init__(self, resources=None, capabilities=None): + if (resources is None): + resources = list() + super(ChangeList, self).__init__(resources, capabilities) + + def __len__(self): + """Number of entries in this changelist""" + return(len(self.resources)) + + def add(self, resource): + """Add a resource_change or an iterable collection to this ChangeList + + Allows multiple resourec_change objects for the same resource (ie. URI) and + preserves the order of addition. + """ + if isinstance(resource, collections.Iterable): + for r in resource: + self.resources.append(r) + else: + self.resources.append(resource) + + def add_changed_resources(self, resources, changeid=None, changetype=None): + """Add items from a ResourceContainer resources to this ChangeList + + If changeid or changetype is specified then these attributes + are set in the Resource objects created. + """ + for resource in resources: + rc = Resource( resource=resource, changeid=changeid, changetype=changetype ) + self.add(rc) diff --git a/resync/client.py b/resync/client.py new file mode 100644 index 0000000..3902453 --- /dev/null +++ b/resync/client.py @@ -0,0 +1,522 @@ +"""ResourceSync client implementation""" + +import sys +import urllib +import os.path +import datetime +import distutils.dir_util +import re +import time +import logging +import ConfigParser + +from resync.inventory_builder import InventoryBuilder +from resync.inventory import Inventory +from resync.changelist import ChangeList +from resync.mapper import Mapper +from resync.sitemap import Sitemap +from resync.dump import Dump +from resync.resource import Resource +from resync.url_authority import UrlAuthority + +class ClientFatalError(Exception): + """Non-recoverable error in client, should include message to user""" + pass + +class Client(object): + """Implementation of a ResourceSync client + + Logging is used for both console output and for detailed logs for + automated analysis. Levels used: + warning - usually shown to user + info - verbose output + debug - very verbose for automated analysis + """ + + def __init__(self, checksum=False, verbose=False, dryrun=False): + super(Client, self).__init__() + self.checksum = checksum + self.verbose = verbose + self.dryrun = dryrun + self.logger = logging.getLogger('client') + self.mapper = None + self.sitemap_name = 'sitemap.xml' + self.dump_format = None + self.exclude_patterns = [] + self.allow_multifile = True + self.noauth = False + self.max_sitemap_entries = None + self.ignore_failures = False + self.status_file = '.resync-client-status.cfg' + + @property + def mappings(self): + """Provide access to mappings list within Mapper object""" + if (self.mapper is None): + raise ClientFatalError("No mappings specified") + return(self.mapper.mappings) + + def set_mappings(self,mappings): + """Build and set Mapper object based on input mappings""" + self.mapper = Mapper(mappings) + + def sitemap_changelist_uri(self,basename): + """Get full URI (filepath) for sitemap/changelist based on basename""" + if (re.match(r"\w+:",basename)): + # looks like URI + return(basename) + elif (re.match(r"/",basename)): + # looks like full path + return(basename) + else: + # build from mapping with name appended + return(self.mappings[0].src_uri + '/' + basename) + + @property + def sitemap(self): + """Return the sitemap URI based on maps or explicit settings""" + return(self.sitemap_changelist_uri(self.sitemap_name)) + + @property + def inventory(self): + """Return inventory on disk based on current mappings + + Return inventory. Uses existing self.mapper settings. + """ + ### 0. Sanity checks + if (len(self.mappings)<1): + raise ClientFatalError("No source to destination mapping specified") + ### 1. Build from disk + ib = InventoryBuilder(do_md5=self.checksum,mapper=self.mapper) + ib.add_exclude_files(self.exclude_patterns) + return( ib.from_disk() ) + + def log_event(self, change): + """Log a Resource object as an event for automated analysis""" + self.logger.debug( "Event: "+repr(change) ) + + def baseline_or_audit(self, allow_deletion=False, audit_only=False): + """Baseline synchonization or audit + + Both functions implemented in this routine because audit is a prerequisite + for a baseline sync. + """ + action = ( 'audit' if (audit_only) else 'baseline sync' ) + self.logger.debug("Starting "+action) + ### 0. Sanity checks + if (len(self.mappings)<1): + raise ClientFatalError("No source to destination mapping specified") + ### 1. Get inventories from both src and dst + # 1.a source inventory + ib = InventoryBuilder(mapper=self.mapper) + try: + self.logger.info("Reading sitemap %s" % (self.sitemap)) + src_sitemap = Sitemap(allow_multifile=self.allow_multifile, mapper=self.mapper) + src_inventory = src_sitemap.read(uri=self.sitemap) + self.logger.debug("Finished reading sitemap") + except Exception as e: + raise ClientFatalError("Can't read source inventory from %s (%s)" % (self.sitemap,str(e))) + self.logger.info("Read source inventory, %d resources listed" % (len(src_inventory))) + if (len(src_inventory)==0): + raise ClientFatalError("Aborting as there are no resources to sync") + if (self.checksum and not src_inventory.has_md5()): + self.checksum=False + self.logger.info("Not calculating checksums on destination as not present in source inventory") + # 1.b destination inventory mapped back to source URIs + ib.do_md5=self.checksum + dst_inventory = ib.from_disk() + ### 2. Compare these inventorys respecting any comparison options + (same,updated,deleted,created)=dst_inventory.compare(src_inventory) + ### 3. Report status and planned actions + status = " IN SYNC " + if (len(updated)>0 or len(deleted)>0 or len(created)>0): + status = "NOT IN SYNC" + self.logger.warning("Status: %s (same=%d, updated=%d, deleted=%d, created=%d)" %\ + (status,len(same),len(updated),len(deleted),len(created))) + if (audit_only): + self.logger.debug("Completed "+action) + return + ### 4. Check that sitemap has authority over URIs listed + uauth = UrlAuthority(self.sitemap) + for resource in src_inventory: + if (not uauth.has_authority_over(resource.uri)): + if (self.noauth): + #self.logger.info("Sitemap (%s) mentions resource at a location it does not have authority over (%s)" % (self.sitemap,resource.uri)) + pass + else: + raise ClientFatalError("Aborting as sitemap (%s) mentions resource at a location it does not have authority over (%s), override with --noauth" % (self.sitemap,resource.uri)) + ### 5. Grab files to do sync + for resource in updated: + uri = resource.uri + file = self.mapper.src_to_dst(uri) + self.logger.info("updated: %s -> %s" % (uri,file)) + self.update_resource(resource,file,'UPDATED') + for resource in created: + uri = resource.uri + file = self.mapper.src_to_dst(uri) + self.logger.info("created: %s -> %s" % (uri,file)) + self.update_resource(resource,file,'CREATED') + for resource in deleted: + uri = resource.uri + file = self.mapper.src_to_dst(uri) + self.delete_resource(resource,file,allow_deletion) + ### 6. For sync reset any incremental status for site + if (not audit_only): + links = self.extract_links(src_inventory) + if ('next' in links): + self.write_incremental_status(self.sitemap,links['next']) + self.logger.info("Written config with next incremental at %s" % (links['next'])) + else: + self.write_incremental_status(self.sitemap) + self.logger.debug("Completed "+action) + + def incremental(self, allow_deletion=False, changelist_uri=None): + """Incremental synchronization""" + self.logger.debug("Starting incremental sync") + ### 0. Sanity checks + if (len(self.mappings)<1): + raise ClientFatalError("No source to destination mapping specified") + # Get current config + inc_config_next=self.read_incremental_status(self.sitemap) + ### 1. Get URI of changelist, from sitemap or explicit + if (inc_config_next is not None): + # We have config from last run for this site + changelist = inc_config_next + self.logger.info("ChangeList location from last incremental run %s" % (changelist)) + elif (changelist_uri): + # Translate as necessary using maps + changelist = self.sitemap_changelist_uri(changelist_uri) + else: + # Get sitemap + try: + self.logger.info("Reading sitemap %s" % (self.sitemap)) + src_sitemap = Sitemap(allow_multifile=self.allow_multifile, mapper=self.mapper) + src_inventory = src_sitemap.read(uri=self.sitemap, index_only=True) + self.logger.debug("Finished reading sitemap/sitemapindex") + except Exception as e: + raise ClientFatalError("Can't read source sitemap from %s (%s)" % (self.sitemap,str(e))) + # Extract changelist location + # FIXME - need to completely rework the way we handle/store capabilities + links = self.extract_links(src_inventory) + if ('current' not in links): + raise ClientFatalError("Failed to extract changelist location from sitemap %s" % (self.sitemap)) + changelist = links['current'] + ### 2. Read changelist from source + ib = InventoryBuilder(mapper=self.mapper) + try: + self.logger.info("Reading changelist %s" % (changelist)) + src_sitemap = Sitemap(allow_multifile=self.allow_multifile, mapper=self.mapper) + src_changelist = src_sitemap.read(uri=changelist, changelist=True) + self.logger.debug("Finished reading changelist") + except Exception as e: + raise ClientFatalError("Can't read source changelist from %s (%s)" % (changelist,str(e))) + self.logger.info("Read source changelist, %d resources listed" % (len(src_changelist))) + #if (len(src_changelist)==0): + # raise ClientFatalError("Aborting as there are no resources to sync") + if (self.checksum and not src_changelist.has_md5()): + self.checksum=False + self.logger.info("Not calculating checksums on destination as not present in source inventory") + ### 3. Check that sitemap has authority over URIs listed + # FIXME - What does authority mean for changelist? Here use both the + # changelist URI and, if we used it, the sitemap URI + uauth_cs = UrlAuthority(changelist) + if (not changelist_uri): + uauth_sm = UrlAuthority(self.sitemap) + for resource in src_changelist: + if (not uauth_cs.has_authority_over(resource.uri) and + (changelist_uri or not uauth_sm.has_authority_over(resource.uri))): + if (self.noauth): + #self.logger.info("ChangeList (%s) mentions resource at a location it does not have authority over (%s)" % (changelist,resource.uri)) + pass + else: + raise ClientFatalError("Aborting as changelist (%s) mentions resource at a location it does not have authority over (%s), override with --noauth" % (changelist,resource.uri)) + ### 3. Apply changes + num_updated = 0 + num_deleted = 0 + num_created = 0 + for resource in src_changelist: + uri = resource.uri + file = self.mapper.src_to_dst(uri) + if (resource.changetype == 'UPDATED'): + self.logger.info("updated: %s -> %s" % (uri,file)) + self.update_resource(resource,file,'UPDATED') + num_updated+=1 + elif (resource.changetype == 'CREATED'): + self.logger.info("created: %s -> %s" % (uri,file)) + self.update_resource(resource,file,'CREATED') + num_created+=1 + elif (resource.changetype == 'DELETED'): + self.delete_resource(resource,file,allow_deletion) + num_deleted+=1 + else: + raise ClientError("Unknown change type %s" % (resource.changetype) ) + # 4. Report status and planned actions + status = "NO CHANGES" + if ((num_updated+num_deleted+num_created)>0): + status = " CHANGES " + self.logger.warning("Status: %s (updated=%d, deleted=%d, created=%d)" %\ + (status,num_updated,num_deleted,num_created)) + # 5. Store next link if available + if ((num_updated+num_deleted+num_created)>0): + links = self.extract_links(src_changelist) + if ('next' in links): + self.write_incremental_status(self.sitemap,links['next']) + self.logger.info("Written config with next incremental at %s" % (links['next'])) + else: + self.logger.warning("Failed to extract next changelist location from changelist %s" % (changelist)) + # 6. Done + self.logger.debug("Completed incremental sync") + + def update_resource(self, resource, file, changetype=None): + """Update resource from uri to file on local system + + Update means two things: + 1. GET resources + 2. set mtime in local time to be equal to timestamp in UTC (should perhaps + or at least warn if different from LastModified from the GET response instead + but maybe warn if different (or just earlier than) the lastmod we expected + from the inventory + """ + path = os.path.dirname(file) + distutils.dir_util.mkpath(path) + if (self.dryrun): + self.logger.info("dryrun: would GET %s --> %s" % (resource.uri,file)) + else: + try: + urllib.urlretrieve(resource.uri,file) + except IOError as e: + msg = "Failed to GET %s -- %s" % (resource.uri,str(e)) + if (self.ignore_failures): + self.logger.warning(msg) + return + else: + raise ClientFatalError(msg) + # sanity check + size = os.stat(file).st_size + if (resource.size != size): + self.logger.info("Downloaded size for %s of %d bytes does not match expected %d bytes" % (resource.uri,size,resource.size)) + # set timestamp if we have one + if (resource.timestamp is not None): + unixtime = int(resource.timestamp) #no fractional + os.utime(file,(unixtime,unixtime)) + self.log_event(Resource(resource=resource, changetype=changetype)) + + def delete_resource(self, resource, file, allow_deletion=False): + """Delete copy of resource in file on local system + """ + uri = resource.uri + if (allow_deletion): + if (self.dryrun): + self.logger.info("dryrun: would delete %s -> %s" % (uri,file)) + else: + try: + os.unlink(file) + except OSError as e: + msg = "Failed to DELETE %s -> %s : %s" % (uri,file,str(e)) + if (self.ignore_failures): + self.logger.warning(msg) + return + else: + raise ClientFatalError(msg) + self.logger.info("deleted: %s -> %s" % (uri,file)) + self.log_event(Resource(resource=resource, changetype="DELETED")) + else: + self.logger.info("nodelete: would delete %s (--delete to enable)" % uri) + + def parse_sitemap(self): + s=Sitemap(allow_multifile=self.allow_multifile) + self.logger.info("Reading sitemap(s) from %s ..." % (self.sitemap)) + i = s.read(self.sitemap) + num_entries = len(i) + self.logger.warning("Read sitemap with %d entries in %d sitemaps" % (num_entries,s.sitemaps_created)) + if (self.verbose): + to_show = 100 + override_str = ' (override with --max-sitemap-entries)' + if (self.max_sitemap_entries): + to_show = self.max_sitemap_entries + override_str = '' + if (num_entries>to_show): + print "Showing first %d entries sorted by URI%s..." % (to_show,override_str) + n=0 + for r in i: + print r + n+=1 + if ( n >= to_show ): + break + + def explore_links(self): + """Explore links from sitemap and between changelists""" + seen = dict() + is_changelist,links = self.explore_links_get(self.sitemap, seen=seen) + starting_changelist = self.sitemap + if (not is_changelist): + if ('current' in links): + starting_changelist = links['current'] + is_changelist,links = self.explore_links_get(links['current'], seen=seen) + # Can we go backward? + if ('prev' in links and not links['prev'] in seen): + self.logger.warning("Will follow links backwards...") + while ('prev' in links and not links['prev'] in seen): + self.logger.warning("Following \"prev\" link") + is_changelist,links = self.explore_links_get(links['prev'], seen=seen) + else: + self.logger.warning("No links backwards") + # Can we go forward? + links = seen[starting_changelist] + if ('next' in links and not links['next'] in seen): + self.logger.warning("Will follow links forwards...") + while ('next' in links and not links['next'] in seen): + self.logger.warning("Following \"next\" link") + is_changelist,links = self.explore_links_get(links['next'], seen=seen) + else: + self.logger.warning("No links forwards") + + def explore_links_get(self, uri, seen=[]): + # Check we haven't been here before + if (uri in seen): + self.logger.warning("Already see %s, skipping" % (uri)) + s=Sitemap(allow_multifile=self.allow_multifile) + self.logger.info("Reading sitemap from %s ..." % (uri)) + i = s.read(uri, index_only=True) + self.logger.warning("Read %s from %s" % (s.read_type,uri)) + links = self.extract_links(i, verbose=True) + if ('next' in links and links['next']==uri): + self.logger.warning("- self reference \"next\" link") + seen[uri]=links + return(s.changelist_read,links) + + def write_sitemap(self,outfile=None,capabilities=None,dump=None): + # Set up base_path->base_uri mappings, get inventory from disk + i = self.inventory + i.capabilities = capabilities + s=Sitemap(pretty_xml=True, allow_multifile=self.allow_multifile, mapper=self.mapper) + if (self.max_sitemap_entries is not None): + s.max_sitemap_entries = self.max_sitemap_entries + if (outfile is None): + print s.resources_as_xml(i,capabilities=i.capabilities) + else: + s.write(i,basename=outfile) + self.write_dump_if_requested(i,dump) + + def changelist_sitemap(self,outfile=None,ref_sitemap=None,newref_sitemap=None, + empty=None,capabilities=None,dump=None): + changelist = ChangeList() + changelist.capabilities = capabilities + if (not empty): + # 1. Get and parse reference sitemap + old_inv = self.read_reference_sitemap(ref_sitemap) + # 2. Depending on whether a newref_sitemap was specified, either read that + # or build inventory from files on disk + if (newref_sitemap is None): + # Get inventory from disk + new_inv = self.inventory + else: + new_inv = self.read_reference_sitemap(newref_sitemap,name='new reference') + # 3. Calculate changelist + (same,updated,deleted,created)=old_inv.compare(new_inv) + changelist.add_changed_resources( updated, changetype='UPDATED' ) + changelist.add_changed_resources( deleted, changetype='DELETED' ) + changelist.add_changed_resources( created, changetype='CREATED' ) + # 4. Write out changelist + s = Sitemap(pretty_xml=True, allow_multifile=self.allow_multifile, mapper=self.mapper) + if (self.max_sitemap_entries is not None): + s.max_sitemap_entries = self.max_sitemap_entries + if (outfile is None): + print s.resources_as_xml(changelist,changelist=True) + else: + s.write(changelist,basename=outfile,changelist=True) + self.write_dump_if_requested(changelist,dump) + + def write_dump_if_requested(self,inventory,dump): + if (dump is None): + return + self.logger.info("Writing dump to %s..." % (dump)) + d = Dump(format=self.dump_format) + d.write(inventory=inventory,dumpfile=dump) + + def read_reference_sitemap(self,ref_sitemap,name='reference'): + """Read reference sitemap and return the inventory + + name parameter just uses in output messages to say what type + of sitemap is being read. + """ + sitemap = Sitemap(allow_multifile=self.allow_multifile, mapper=self.mapper) + self.logger.info("Reading %s sitemap(s) from %s ..." % (name,ref_sitemap)) + i = sitemap.read(ref_sitemap) + num_entries = len(i) + self.logger.warning("Read %s sitemap with %d entries in %d sitemaps" % (name,num_entries,sitemap.sitemaps_created)) + if (self.verbose): + to_show = 100 + override_str = ' (override with --max-sitemap-entries)' + if (self.max_sitemap_entries): + to_show = self.max_sitemap_entries + override_str = '' + if (num_entries>to_show): + print "Showing first %d entries sorted by URI%s..." % (to_show,override_str) + n=0 + for r in i: + print r + n+=1 + if ( n >= to_show ): + break + return(i) + + def extract_links(self, rc, verbose=False): + """Extract links from capabilities inventory or changelist + + FIXME - when we finalize the form of links this should probably + go along with other capabilities functions somewhere general. + """ + links = dict() + for href in rc.capabilities.keys(): + atts = rc.capabilities[href].get('attributes') + self.logger.debug("Capability: %s" % (str(rc.capabilities[href]))) + if (atts is not None): + # split on spaces, check is changelist rel and diraction + if ('http://www.openarchives.org/rs/changelist' in atts): + for linktype in ['next','prev','current']: + if (linktype in atts): + if (linktype in links): + raise ClientFatalError("Duplicate link type %s, links to %s and %s" % (linktype,links[linktype],href)) + links[linktype] = href; + if (verbose): + self.logger.warning("- got \"%s\" link to %s" % (linktype,href)) + return(links) + + def write_incremental_status(self,site,next=None): + """Write status dict to client status file + + FIXME - should have some file lock to avoid race + """ + parser = ConfigParser.SafeConfigParser() + parser.read(self.status_file) + status_section = 'incremental' + if (not parser.has_section(status_section)): + parser.add_section(status_section) + if (next is None): + parser.remove_option(status_section, self.config_site_to_name(site)) + else: + parser.set(status_section, self.config_site_to_name(site), next) + with open(self.status_file, 'wb') as configfile: + parser.write(configfile) + configfile.close() + + def read_incremental_status(self,site): + """Read client status file and return dict""" + parser = ConfigParser.SafeConfigParser() + status_section = 'incremental' + parser.read(self.status_file) + next = None + try: + next = parser.get(status_section,self.config_site_to_name(site)) + except ConfigParser.NoSectionError as e: + pass + except ConfigParser.NoOptionError as e: + pass + return(next) + + def config_site_to_name(self, name): + return( re.sub(r"[^\w]",'_',name) ) + +if __name__ == '__main__': + main() diff --git a/resync/dump.py b/resync/dump.py index e69de29..9df9c57 100644 --- a/resync/dump.py +++ b/resync/dump.py @@ -0,0 +1,85 @@ +"""Dump handler for ResourceSync""" + +import os.path +from zipfile import ZipFile, ZIP_STORED, ZIP_DEFLATED +from sitemap import Sitemap + +class DumpError(Exception): + pass + +class Dump(object): + """Dump of resource content associated with an inventory or change set + + The inventory must be comprised of Resource objects + which have the path attributes set to indicate the local + location of the copies of the resources. + """ + + def __init__(self, format=None, compress=True): + self.format = ('zip' if (format is None) else format) + self.compress = compress + self.max_size = 100*1024*1024 #100MB + self.max_files = 50000 + + def write(self, inventory=None, dumpfile=None): + """Write a dump file""" + self.check_files(inventory) + if (self.format == 'zip'): + self.write_zip(inventory,dumpfile) + elif (self.format == 'warc'): + self.write_warc(inventory,dumpfile) + else: + raise DumpError("Unknown dump format '%s'" % (self.format)) + + def write_zip(self, inventory=None, dumpfile=None): + """Write a ZIP dump file""" + compression = ( ZIP_DEFLATED if self.compress else ZIP_STORED ) + zf = ZipFile(dumpfile, mode="w", compression=compression, allowZip64=True) + # Write inventory first + s = Sitemap(pretty_xml=True, allow_multifile=False) + zf.writestr('manifest.xml',s.resources_as_xml(inventory)) + # Add all files in the inventory + for resource in inventory: + zf.write(resource.uri) + zf.close() + zipsize = os.path.getsize(dumpfile) + print "Wrote ZIP file dump %s with size %d bytes" % (dumpfile,zipsize) + + def write_warc(self, inventory=None, dumpfile=None): + """Write a WARC dump file""" + # Load library late as we want to be able to run rest of code + # without this installed + try: + from warc import WARCFile,WARCHeader,WARCRecord + except: + raise DumpError("Failed to load WARC library") + wf = WARCFile(dumpfile, mode="w", compress=self.compress) + # Add all files in the inventory + for resource in inventory: + wh = WARCHeader({}) + wh.url = resource.uri + wh.ip_address = None + wh.date = resource.lastmod + wh.content_type = 'text/plain' + wh.result_code = 200 + wh.checksum = 'aabbcc' + wh.location = 'loc' + wf.write_record( WARCRecord( header=wh, payload=resource.path ) ) + wf.close() + warcsize = os.path.getsize(dumpfile) + print "Wrote WARC file dump %s with size %d bytes" % (dumpfile,warcsize) + + def check_files(self,inventory): + """Go though and check all files in inventory, add up size""" + if (len(inventory) > self.max_files): + raise DumpError("Number of files to dump (%d) exceeds maximum (%d)" % (len(inventory),self.max_files)) + total_size = 0 #total size of all files in bytes + for resource in inventory: + if (resource.path is None): + #explicit test because exception raised by getsize otherwise confusing + raise DumpError("No file path defined for resource %s" % resource.uri) + total_size += os.path.getsize(resource.path) + self.total_size = total_size + print "Total size of files to include in dump %d bytes" % (total_size) + if (total_size > self.max_size): + raise DumpError("Size of files to dump (%d) exceeds maximum (%d)" % (total_size,self.max_size)) diff --git a/resync/inventory.py b/resync/inventory.py index e69de29..35f0138 100644 --- a/resync/inventory.py +++ b/resync/inventory.py @@ -0,0 +1,143 @@ +"""ResourceSync inventory object + +An inventory is a set of resources with some metadata for each +resource. Comparison of inventories from a source and a +destination allows understanding of whether the two are in +sync or whether some resources need to be updated at the +destination. + +The inventory object may also contain metadata regarding +capabilities and discovery information. +""" + +import collections +import os +from datetime import datetime +import re +import sys +import StringIO + +from resource_container import ResourceContainer + +class InventoryDict(dict): + """Default implementation of class to store resources in Inventory + + Key properties of this class are: + - has add(resource) method + - is iterable and results given in alphanumeric order by resource.uri + """ + + def __iter__(self): + """Iterator over all the resources in this inventory""" + self._iter_next_list = sorted(self.keys()) + self._iter_next_list.reverse() + return(iter(self._iter_next, None)) + + def _iter_next(self): + if (len(self._iter_next_list)>0): + return(self[self._iter_next_list.pop()]) + else: + return(None) + + def add(self, resource, replace=False): + """Add just a single resource""" + uri = resource.uri + if (uri in self and not replace): + raise InventoryDupeError("Attempt to add resource already in inventory") + self[uri]=resource + +class InventoryDupeError(Exception): + pass + +class Inventory(ResourceContainer): + """Class representing an inventory of resources + + This same class is used for both the source and the destination + and is the central point of comparison the decide whether they + are in sync or what needs to be copied to bring the destinaton + into sync. + + An inventory will admit only one resource with any given URI. + + Storage is unordered but the iterator imposes a canonical order + which is currently alphabetical by URI. + """ + + def __init__(self, resources=None, capabilities=None): + self.resources=(resources if (resources is not None) else InventoryDict()) + self.capabilities=(capabilities if (capabilities is not None) else {}) + + def __iter__(self): + """Iterator over all the resources in this inventory""" + return(iter(self.resources)) + + def __len__(self): + """Return number of resources in this inventory""" + return(len(self.resources)) + + def add(self, resource, replace=False): + """Add a resource or an iterable collection of resources + + Will throw a ValueError if the resource (ie. same uri) already + exists in the inventory, unless replace=True. + """ + if isinstance(resource, collections.Iterable): + for r in resource: + self.resources.add(r,replace) + else: + self.resources.add(resource,replace) + + def compare(self,src): + """Compare the current inventory object with the specified inventory + + The parameter src must also be an inventory object, it is assumed + to be the source, and the current object is the destination. This + written to work for any objects in self and sc, provided that the + == operator can be used to compare them. + + The functioning of this method depends on the iterators for self and + src providing access to the resource objects in URI order. + """ + dst_iter = iter(self.resources) + src_iter = iter(src.resources) + same=Inventory() + updated=Inventory() + deleted=Inventory() + created=Inventory() + dst_cur=next(dst_iter,None) + src_cur=next(src_iter,None) + while ((dst_cur is not None) and (src_cur is not None)): + #print 'dst='+dst_cur+' src='+src_cur + if (dst_cur.uri == src_cur.uri): + if (dst_cur==src_cur): + same.add(dst_cur) + else: + updated.add(src_cur) + dst_cur=next(dst_iter,None) + src_cur=next(src_iter,None) + elif (not src_cur or dst_cur.uri < src_cur.uri): + deleted.add(dst_cur) + dst_cur=next(dst_iter,None) + elif (not dst_cur or dst_cur.uri > src_cur.uri): + created.add(src_cur) + src_cur=next(src_iter,None) + else: + raise InternalError("this should not be possible") + # what do we have leftover in src or dst lists? + while (dst_cur is not None): + deleted.add(dst_cur) + dst_cur=next(dst_iter,None) + while (src_cur is not None): + created.add(src_cur) + src_cur=next(src_iter,None) + # have now gone through both lists + return(same,updated,deleted,created) + + def has_md5(self): + """Return true if at least one contained resource-like object has md5 data""" + if (self.resources is None): + return(False) + for resource in self: + if (resource.md5 is not None): + return(True) + return(False) diff --git a/resync/inventory_builder.py b/resync/inventory_builder.py new file mode 100644 index 0000000..df788aa --- /dev/null +++ b/resync/inventory_builder.py @@ -0,0 +1,127 @@ +"""InventoryBuilder to create Inventory objects from various sources + +Attributes: +- do_md5 set true to calculate MD5 sums for all files +- do_size set true to include file size in inventory +- exclude_dirs is a list of directory names to exclude + (defaults to ['CVS','.git')) +""" + +import os +import os.path +import re +import time +import logging +from urllib import URLopener +from xml.etree.ElementTree import parse + +from resource import Resource +from inventory import Inventory +from sitemap import Sitemap +from utils import compute_md5_for_file + +class InventoryBuilder(): + + def __init__(self, do_md5=False, do_size=True, mapper=None): + """Create InventoryBuilder object, optionally set options + + Optionaly sets the following attributes: + - do_md5 - True to add md5 digests for each resource + - do_size - False to not add size for each resources + """ + self.do_md5 = do_md5 + self.do_size = do_size + self.mapper = mapper + self.exclude_files = ['sitemap\d{0,5}.xml'] + self.exclude_dirs = ['CVS','.git'] + self.include_symlinks = False + # Used internally only: + self.logger = logging.getLogger('inventory_builder') + self.compiled_exclude_files = [] + + def add_exclude_files(self, exclude_patterns): + """Add more patterns of files to exclude while building inventory""" + for pattern in exclude_patterns: + self.exclude_files.append(pattern) + + def compile_excludes(self): + self.compiled_exclude_files = [] + for pattern in self.exclude_files: + self.compiled_exclude_files.append(re.compile(pattern)) + + def exclude_file(self, file): + """True if file should be exclude based on name pattern""" + for pattern in self.compiled_exclude_files: + if (pattern.match(file)): + return(True) + return(False) + + def from_disk(self,inventory=None): + """Create or extend inventory with resources from disk scan + + Assumes very simple disk path to URL mapping: chop path and + replace with url_path. Returns the new or extended Inventory + object. + + If a inventory is specified then items are added to that rather + than creating a new one. + + mapper=Mapper('http://example.org/path','/path/to/files') + mb = InventoryBuilder(mapper=mapper) + m = inventory_from_disk() + """ + num=0 + # Either use inventory passed in or make a new one + if (inventory is None): + inventory = Inventory() + # Compile exclude pattern matches + self.compile_excludes() + # Run for each map in the mappings + for map in self.mapper.mappings: + self.logger.info("Scanning disk for %s" % (str(map))) + self.from_disk_add_map(inventory=inventory, map=map) + return(inventory) + + def from_disk_add_map(self, inventory=None, map=None): + # sanity + if (inventory is None or map is None): + raise ValueError("Must specify inventory and map") + path=map.dst_path + #print "walking: %s" % (path) + # for each file: create Resource object, add, increment counter + num_files=0 + for dirpath, dirs, files in os.walk(path,topdown=True): + for file_in_dirpath in files: + num_files+=1 + if (num_files%50000 == 0): + self.logger.info("InventoryBuilder.from_disk_add_map: %d files..." % (num_files)) + try: + if self.exclude_file(file_in_dirpath): + self.logger.debug("Excluding file %s" % (file_in_dirpath)) + continue + # get abs filename and also URL + file = os.path.join(dirpath,file_in_dirpath) + if (not os.path.isfile(file) or not (self.include_symlinks or not os.path.islink(file))): + continue + uri = map.dst_to_src(file) + if (uri is None): + raise Exception("Internal error, mapping failed") + file_stat=os.stat(file) + except OSError as e: + sys.stderr.write("Ignoring file %s (error: %s)" % (file,str(e))) + continue + timestamp = file_stat.st_mtime #UTC + r = Resource(uri=uri,timestamp=timestamp,path=file) + if (self.do_md5): + # add md5 + r.md5=compute_md5_for_file(file) + if (self.do_size): + # add size + r.size=file_stat.st_size + inventory.add(r) + # prune list of dirs based on self.exclude_dirs + for exclude in self.exclude_dirs: + if exclude in dirs: + self.logger.debug("Excluding dir %s" % (exclude)) + dirs.remove(exclude) + return(inventory) diff --git a/resync/mapper.py b/resync/mapper.py new file mode 100644 index 0000000..45f5c4e --- /dev/null +++ b/resync/mapper.py @@ -0,0 +1,116 @@ +"""Map between source URIs and destination paths""" +import os +import os.path +import re + +class MapperError(Exception): + pass + +class Mapper(): + + def __init__(self,mappings=None): + self.mappings=[] + if (mappings): + self.parse(mappings) + + def __len__(self): + """Length is number of mappings""" + return(len(self.mappings)) + + def parse(self, mappings): + """Parse a list of map strings (mappings) + + Accepts two distinct formats: + 1. If there are exactly two entries then these may be the source base URI + and the destination base path. Neither may contain an equals (=) sign. + 2. For any number of mapping stings interpret each as a mapping URI=path. + These are in the order they will be tried. + """ + if (len(mappings)==2 and + re.search(r"=",mappings[0])==None and + re.search(r"=",mappings[1])==None): + self.mappings.append(Map(mappings[0],mappings[1])) + else: + for mapping in mappings: + l=mapping.split('=') + if (len(l)!=2): + raise MapperError("Bad mapping argument (%s), got %s"%(mapping,str(l))) + (src_uri,dst_path)=l + # Check for dupes + for map in self.mappings: + if (src_uri == map.src_uri): + raise MapperError("Attempt to set duplicate mapping for source URI %s (with destination path %s)" % (src_uri,dst_path)) + if (dst_path == map.dst_path): + raise MapperError("Attempt to set duplicate mapping for destination path %s (with source URI %s)" % (dst_path,src_uri)) + self.mappings.append(Map(src_uri, dst_path)) + + def dst_to_src(self,dst_file): + for map in self.mappings: + src_uri = map.dst_to_src(dst_file) + if (src_uri is not None): + return(src_uri) + # Must have failed if loop exited + raise MapperError("Unable to translate destination path (%s) into a source URI." % (dst_file)) + + def src_to_dst(self,src_uri): + for map in self.mappings: + dst_path = map.src_to_dst(src_uri) + if (dst_path is not None): + return(dst_path) + # Must have failed if loop exited + raise MapperError("Unable to translate source URI (%s) into a destination path." % (src_uri)) + + def __repr__(self): + s = 'Mapper: with %d maps:\n' % (len(self.mappings)) + for map in self.mappings: + s += str(map) + '\n' + return(s) + + +class Map: + """A single map from source URI to destination path + + Both URI and destination paths are assumed to use / as the path + separator. No account is take for other path separators used + for paths on non-unix systems. This translation must be done + elsewhere by consideration of os.sep. + """ + + def __init__(self,src_uri=None,dst_path=None): + self.src_uri = self.strip_trailing_slashes(src_uri) + self.dst_path = self.strip_trailing_slashes(dst_path) + + def strip_trailing_slashes(self,path): + """Return input path minus any trailing slashes""" + m=re.match(r"(.*)/+$",path) + if (m is None): + return(path) + return(m.group(1)) + + def dst_to_src(self,dst_file): + """Return the src URI from the dst filepath + + This does not rely on the destination filepath actually existing on the local + filesystem, just on pattern matching. Return source URI on success, None on + failure. + """ + m=re.match(self.dst_path+"/(.*)$",dst_file) + if (m is None): + return(None) + rel_path=m.group(1) + return(self.src_uri+'/'+rel_path) + + def src_to_dst(self,src_uri): + """Return the dst filepath from the src URI + + FIXME -- look at whether urlparse can be used here? + Returns None on failure, destination path on success. + """ + m=re.match(self.src_uri+"/(.*)$",src_uri) + if (m is None): + return(None) + rel_path=m.group(1) + return(self.dst_path+'/'+rel_path) + + def __repr__(self): + return("Map( %s -> %s )" % (self.src_uri, self.dst_path)) diff --git a/resync/resource.py b/resync/resource.py index e69de29..f5f7eae 100644 --- a/resync/resource.py +++ b/resync/resource.py @@ -0,0 +1,140 @@ +"""Information about a web resource + +Each web resource is identified by a URI and may optionally have +other metadata such as timestamp, size, md5. The lastmod property +provides ISO8601 format string access to the timestamp. + +The timestamp is assumed to be stored in UTC. +""" + +import re +from urlparse import urlparse +from posixpath import basename +from w3c_datetime import str_to_datetime, datetime_to_str + +class Resource(object): + __slots__=('uri', 'timestamp', 'size', 'md5', 'sha1', + 'changetype', 'changeid', 'path') + + def __init__(self, uri = None, timestamp = None, size = None, + md5 = None, sha1 = None, lastmod = None, + changetype = None, changeid = None, path = None, + resource = None ): + """ Initialize object either from parameters specified or + from an existing Resource object. If explicit parameters + are specified then they will override values copied from + a resource object supplied. + """ + # Create from a Resource? + self.uri = None + self.timestamp = None + self.size = None + self.md5 = None + self.sha1 = None + self.changetype = None + self.changeid = None + self.path = None + if (resource is not None): + self.uri = resource.uri + self.timestamp = resource.timestamp + self.size = resource.size + self.md5 = resource.md5 + self.sha1 = resource.sha1 + self.changetype = resource.changetype + self.changeid = resource.changeid + self.path = resource.path + if (uri is not None): + self.uri = uri + if (timestamp is not None): + self.timestamp = timestamp + if (size is not None): + self.size = size + if (md5 is not None): + self.md5 = md5 + if (sha1 is not None): + self.sha1 = sha1 + if (changetype is not None): + self.changetype = changetype + if (changeid is not None): + self.changeid = changeid + if (path is not None): + self.path = path + if (lastmod is not None): + self.lastmod=lastmod + # Sanity check + if (self.uri is None): + raise ValueError("Cannot create resoure without a URI") + + @property + def lastmod(self): + """The Last-Modified data in W3C Datetime syntax, Z notation""" + if (self.timestamp is None): + return None + return datetime_to_str(self.timestamp) + + @lastmod.setter + def lastmod(self, lastmod): + """Set timestamp from an W3C Datetime Last-Modified value""" + if (lastmod is None): + self.timestamp = None + return + if (lastmod == ''): + raise ValueError('Attempt to set empty lastmod') + self.timestamp = str_to_datetime(lastmod) + + @property + def basename(self): + """The resource basename (http://example.com/resource/1 -> 1)""" + parse_object = urlparse(self.uri) + return basename(parse_object.path) + + def __eq__(self,other): + """Equality test for resources allowing <1s difference in timestamp""" + return( self.equal(other,delta=1.0) ) + + def equal(self,other,delta=0.0): + """Equality or near equality test for resources + + Equality means: + 1. same uri, AND + 2. same timestamp WITHIN delta if specified for either, AND + 3. same md5 if specified for both, AND + 4. same size if specified for both + """ + if (other is None): return False + + if (self.uri != other.uri): + return(False) + if ( self.timestamp is not None or other.timestamp is not None ): + # not equal if only one timestamp specified + if ( self.timestamp is None or + other.timestamp is None or + abs(self.timestamp-other.timestamp)>=delta ): + return(False) + if ( ( self.md5 is not None and other.md5 is not None ) and + self.md5 != other.md5 ): + return(False) + if ( ( self.size is not None and other.size is not None ) and + self.size != other.size ): + return(False) + return(True) + + def __str__(self): + """Return a human readable string for this resource""" + s = [ str(self.uri), str(self.lastmod), str(self.size), + str(self.md5 if self.md5 else self.sha1) ] + if (self.changetype is not None): + s.add(str(self.changetype)) + #s.add(str(self.changeid)) + if (self.path is not None): + s.add(str(self.path)) + return "[ " + " | ".join(s) + " ]" + + def __repr__(self): + """Return an unambigous representation""" + dict_repr = dict((name, getattr(self, name)) + for name in dir(self) if not (name.startswith('__') + or name == 'equal' + or name == 'basename' + or name == 'lastmod')) + return str(dict_repr) diff --git a/resync/resource_container.py b/resync/resource_container.py index e69de29..f7606cb 100644 --- a/resync/resource_container.py +++ b/resync/resource_container.py @@ -0,0 +1,48 @@ +"""ResourceSync Resource Container object + +Both Inventory and Change Set objects are collections of Resource +objects with additional metadata regarding capabilities and +discovery information. + +This is a superclass for the Inventory and ChangeSet classes which +contains common functionality. +""" + +class ResourceContainer(object): + """Class containing resource-like objects + + Core functionality:: + - resources property that is the set/list of resources + -- add() to add a resource-like object to self.resources + -- iter() to get iterator over self.resource in appropriate order + - capabilities property that is a dict of capabilities + + Derived classes may add extra functionality such as len() etc.. + However, any code designed to work with any ResourceContainer + should use only the core functionality. + """ + + def __init__(self, resources=None, capabilities=None): + self.resources=resources + self.capabilities=(capabilities if (capabilities is not None) else {}) + + def __iter__(self): + """Iterator over all the resources in this inventory + + Baseline implementation use iterator given by resources property + """ + return(iter(self.resources)) + + def add(self, resource): + """Add a resource or an iterable collection of resources to this container + + Must be implemented in derived class + """ + raise NotImplemented("add() not implemented") + + def __str__(self): + """Return string of all resources in order given by interator""" + s = '' + for resource in self: + s += str(resource) + "\n" + return(s) diff --git a/resync/sitemap.py b/resync/sitemap.py new file mode 100644 index 0000000..b428e7a --- /dev/null +++ b/resync/sitemap.py @@ -0,0 +1,629 @@ +"""Read and write ResourceSync inventories and changelist as sitemaps""" + +import re +import os +import sys +import logging +from urllib import URLopener +from xml.etree.ElementTree import ElementTree, Element, parse, tostring +from datetime import datetime +import StringIO + +from resource import Resource +from inventory import Inventory, InventoryDupeError +from changelist import ChangeList +from mapper import Mapper, MapperError +from url_authority import UrlAuthority + +SITEMAP_NS = 'http://www.sitemaps.org/schemas/sitemap/0.9' +RS_NS = 'http://www.openarchives.org/rs/terms/' +#XHTML_NS = 'http://www.w3.org/1999/xhtml' +XHTML_NS = 'http://www.w3.org/1999/xhtml_DEFANGED' + +class SitemapIndexError(Exception): + """Exception on attempt to read a sitemapindex instead of sitemap""" + + def __init__(self, message=None, etree=None): + self.message = message + self.etree = etree + + def __repr__(self): + return(self.message) + +class SitemapIndex(Inventory): + """Reuse an inventory to hold the set of sitemaps""" + pass + +class SitemapError(Exception): + pass + +class Sitemap(object): + """Read and write sitemaps + + Implemented as a separate class that uses ResourceContainer (Inventory or + ChangeList) and Resource classes as data objects. Reads and write sitemaps, + including multiple file sitemaps. + """ + + def __init__(self, pretty_xml=False, allow_multifile=True, mapper=None): + self.logger = logging.getLogger('sitemap') + self.pretty_xml=pretty_xml + self.allow_multifile=allow_multifile + self.mapper=mapper + self.max_sitemap_entries=50000 + # Classes used when parsing + self.inventory_class=Inventory + self.resource_class=Resource + self.changelist_class=ChangeList + self.resourcechange_class=Resource + # Information recorded for logging + self.resources_created=None # Set during parsing sitemap + self.sitemaps_created=None # Set during parsing sitemapindex + self.content_length=None # Size of last sitemap read + self.bytes_read=0 # Aggregate of content_length values + self.changelist_read=None # Set true if changelist read + self.read_type=None # Either sitemap/sitemapindex/changelist/changelistindex + + ##### General sitemap methods that also handle sitemapindexes ##### + + def write(self, resources=None, basename='/tmp/sitemap.xml', changelist=False): + """Write one or a set of sitemap files to disk + + resources is a ResourceContainer that may be an Inventory or + a ChangeList. This may be a generator so data is read as needed + and length is determined at the end. + + basename is used as the name of the single sitemap file or the + sitemapindex for a set of sitemap files. + + if changelist is set true then type information is added to indicate + that this sitemap file is a changelist and not an inventory. + + Uses self.max_sitemap_entries to determine whether the inventory can + be written as one sitemap. If there are more entries and + self.allow_multifile is set true then a set of sitemap files, + with an sitemapindex, will be written. + """ + # Access resources trough iterator only + resources_iter = iter(resources) + ( chunk, next ) = self.get_resources_chunk(resources_iter) + if (next is not None): + # Have more than self.max_sitemap_entries => sitemapindex + if (not self.allow_multifile): + raise Exception("Too many entries for a single sitemap but multifile disabled") + # Work out how to name the sitemaps, attempt to add %05d before ".xml$", else append + sitemap_prefix = basename + sitemap_suffix = '.xml' + if (basename[-4:] == '.xml'): + sitemap_prefix = basename[:-4] + # Use iterator over all resources and count off sets of + # max_sitemap_entries to go into each sitemap, store the + # names of the sitemaps as we go + sitemaps={} + while (len(chunk)>0): + file = sitemap_prefix + ( "%05d" % (len(sitemaps)) ) + sitemap_suffix + self.logger.info("Writing sitemap %s..." % (file)) + f = open(file, 'w') + f.write(self.resources_as_xml(chunk,changelist=changelist)) + f.close() + # Record timestamp + sitemaps[file] = os.stat(file).st_mtime + # Get next chunk + ( chunk, next ) = self.get_resources_chunk(resources_iter,next) + self.logger.info("Wrote %d sitemaps" % (len(sitemaps))) + f = open(basename, 'w') + self.logger.info("Writing sitemapindex %s..." % (basename)) + f.write(self.sitemapindex_as_xml(sitemaps=sitemaps,inventory=resources,capabilities=resources.capabilities,changelist=changelist)) + f.close() + self.logger.info("Wrote sitemapindex %s" % (basename)) + else: + f = open(basename, 'w') + self.logger.info("Writing sitemap %s..." % (basename)) + f.write(self.resources_as_xml(chunk,capabilities=resources.capabilities,changelist=changelist)) + f.close() + self.logger.info("Wrote sitemap %s" % (basename)) + + def get_resources_chunk(self, resource_iter, first=None): + """Return next chunk of resources from resource_iter, and next item + + If first parameter is specified then this will be prepended to + the list. + + The chunk will contain self.max_sitemap_entries if the iterator + returns that many. next will have the value of the next value from + the iterator, providing indication of whether more is available. + Use this as first when asking for the following chunk. + """ + chunk = [] + next = None + if (first is not None): + chunk.append(first) + for r in resource_iter: + chunk.append(r) + if (len(chunk)>self.max_sitemap_entries): + break + if (len(chunk)>self.max_sitemap_entries): + next = chunk.pop() + return(chunk,next) + + def read(self, uri=None, resources=None, changelist=None, index_only=False): + """Read sitemap from a URI including handling sitemapindexes + + Returns the inventory or changelist. If changelist is not specified (None) + then it is assumed that an Inventory is to be read, unless the XML + indicates a Changelist. + + If changelist is True then a Changelist if expected; if changelist if False + then an Inventory is expected. + + If index_only is True then individual sitemaps references in a sitemapindex + will not be read. This will result in no resources being returned and is + useful only to read the capabilities and metadata listed in the sitemapindex. + + Will set self.read_type to a string value sitemap/sitemapindex/changelist/changelistindex + depleding on the type of the file expected/read. + + Includes the subtlety that if the input URI is a local file and is a + sitemapindex which contains URIs for the individual sitemaps, then these + are mapped to the filesystem also. + """ + try: + fh = URLopener().open(uri) + except IOError as e: + raise Exception("Failed to load sitemap/sitemapindex from %s (%s)" % (uri,str(e))) + # Get the Content-Length if we can (works fine for local files) + try: + self.content_length = int(fh.info()['Content-Length']) + self.bytes_read += self.content_length + self.logger.debug( "Read %d bytes from %s" % (self.content_length,uri) ) + except KeyError: + # If we don't get a length then c'est la vie + self.logger.debug( "Read ????? bytes from %s" % (uri) ) + pass + self.logger.info( "Read sitemap/sitemapindex from %s" % (uri) ) + etree = parse(fh) + # check root element: urlset (for sitemap), sitemapindex or bad + self.sitemaps_created=0 + root = etree.getroot() + # assume inventory but look to see whether this is a changelist + # as indicated with rs:type="changelist" on the root + resources_class = self.inventory_class + sitemap_xml_parser = self.inventory_parse_xml + self.changelist_read = False + self.read_type = 'sitemap' + root_type = root.attrib.get('{'+RS_NS+'}type',None) + if (root_type is not None): + if (root_type == 'changelist'): + self.changelist_read = True + else: + self.logger.info("Bad value of rs:type on root element (%s), ignoring" % (root_type)) + elif (changelist is True): + self.changelist_read = True + if (self.changelist_read): + self.read_type = 'changelist' + resources_class = self.changelist_class + sitemap_xml_parser = self.changelist_parse_xml + # now have make sure we have a place to put the data we read + if (resources is None): + resources=resources_class() + # sitemap or sitemapindex? + if (root.tag == '{'+SITEMAP_NS+"}urlset"): + self.logger.info( "Parsing as sitemap" ) + sitemap_xml_parser(etree=etree, resources=resources) + self.sitemaps_created+=1 + elif (root.tag == '{'+SITEMAP_NS+"}sitemapindex"): + self.read_type += 'index' + if (not self.allow_multifile): + raise Exception("Got sitemapindex from %s but support for sitemapindex disabled" % (uri)) + self.logger.info( "Parsing as sitemapindex" ) + sitemaps=self.sitemapindex_parse_xml(etree=etree) + sitemapindex_is_file = self.is_file_uri(uri) + if (index_only): + return(resources) + # now loop over all entries to read each sitemap and add to resources + self.logger.info( "Now reading %d sitemaps" % len(sitemaps) ) + for sitemap_uri in sorted(sitemaps.resources.keys()): + if (sitemapindex_is_file): + if (not self.is_file_uri(sitemap_uri)): + # Attempt to map URI to local file + remote_uri = sitemap_uri + sitemap_uri = self.mapper.src_to_dst(remote_uri) + else: + # The individual sitemaps should be at a URL (scheme/server/path) + # that the sitemapindex URL can speak authoritatively about + if (not UrlAuthority(uri).has_authority_over(sitemap_uri)): + raise Exception("The sitemapindex (%s) refers to sitemap at a location it does not have authority over (%s)" % (uri,sitemap_uri)) + try: + fh = URLopener().open(sitemap_uri) + except IOError as e: + raise Exception("Failed to load sitemap from %s listed in sitemap index %s (%s)" % (sitemap_uri,uri,str(e))) + # Get the Content-Length if we can (works fine for local files) + try: + self.content_length = int(fh.info()['Content-Length']) + self.bytes_read += self.content_length + except KeyError: + # If we don't get a length then c'est la vie + pass + self.logger.info( "Read sitemap from %s (%d)" % (sitemap_uri,self.content_length) ) + sitemap_xml_parser( fh=fh, resources=resources ) + self.sitemaps_created+=1 + else: + raise ValueError("XML read from %s is not a sitemap or sitemapindex" % (uri)) + return(resources) + + ##### Resource methods ##### + + def resource_etree_element(self, resource, element_name='url'): + """Return xml.etree.ElementTree.Element representing the resource + + Returns and element for the specified resource, of the form + with enclosed properties that are based on the sitemap with extensions + for ResourceSync. + """ + e = Element(element_name) + sub = Element('loc') + sub.text=resource.uri + e.append(sub) + if (resource.timestamp is not None): + lastmod_name = 'lastmod' + lastmod_attrib = {} + if (hasattr(resource,'changetype') and + resource.changetype is not None): + # Not a plain old , use with + # rs:type attribute or + if (resource.changetype == 'CREATED'): + lastmod_attrib = {'rs:type': 'created'} + elif (resource.changetype == 'UPDATED'): + lastmod_attrib = {'rs:type': 'updated'} + elif (resource.changetype == 'DELETED'): + lastmod_name = 'expires' + else: + raise Exception("Unknown change type '%s' for resource %s" % (resource.changetype,resource.uri)) + # Create appriate element for timestamp + sub = Element(lastmod_name,lastmod_attrib) + sub.text = str(resource.lastmod) #W3C Datetime in UTC + e.append(sub) + if (resource.size is not None): + sub = Element('rs:size') + sub.text = str(resource.size) + e.append(sub) + if (resource.md5 is not None): + sub = Element('rs:fixity') + sub.attrib = {'type':'md5'} + sub.text = str(resource.md5) + e.append(sub) + if (self.pretty_xml): + e.tail="\n" + return(e) + + def resource_as_xml(self,resource,indent=' '): + """Return string for the the resource as part of an XML sitemap + + """ + e = self.resource_etree_element(resource) + if (sys.version_info < (2,7)): + #must not specify method='xml' in python2.6 + return(tostring(e, encoding='UTF-8')) + else: + return(tostring(e, encoding='UTF-8', method='xml')) + + def resource_from_etree(self, etree, resource_class): + """Construct a Resource from an etree + + Parameters: + etree - the etree to parse + resource_class - class of Resource object to create + + The parsing is properly namespace aware but we search just for + the elements wanted and leave everything else alone. Provided + there is a element then we'll go ahead and extract as much + as possible. + """ + loc = etree.findtext('{'+SITEMAP_NS+"}loc") + if (loc is None): + raise SitemapError("Missing element while parsing in sitemap") + # We at least have a URI, make this object + resource=resource_class(uri=loc) + # and then proceed to look for other resource attributes + changetype = None + lastmod_element = etree.find('{'+SITEMAP_NS+"}lastmod") + if (lastmod_element is not None): + lastmod = lastmod_element.text + if (lastmod is not None): + resource.lastmod=lastmod + type = lastmod_element.attrib.get('{'+RS_NS+'}type',None) + if (type is not None): + if (type == 'created'): + changetype='CREATED' + elif (type == 'updated'): + changetype='UPDATED' + else: + self.logger.warning("Bad rs:type for for %s" % (loc)) + expires = etree.findtext('{'+SITEMAP_NS+"}expires") + if (expires is not None): + resource.lastmod=expires + changetype='DELETED' + if (lastmod_element is not None): + self.logger.warning("Got and for %s" % (loc)) + # If we have a changetype, see whether we can set it + if (changetype is not None): + try: + resource.changetype = changetype + except AttributeError as e: + self.logger.warning("Cannot record changetype %s for %s" % (changetype,loc)) + # size in bytes + size = etree.findtext('{'+RS_NS+"}size") + if (size is not None): + try: + resource.size=int(size) + except ValueError as e: + raise Exception("Invalid for %s" % (loc)) + # The ResourceSync v0.1 spec lists md5, sha-1 and sha-256 fixity + # digest types. Currently support only md5, warn if anything else + # ignored + fixity_element = etree.find('{'+RS_NS+'}fixity') + if (fixity_element is not None): + #type = fixity_element.get('{'+RS_NS+'}type',None) + type = fixity_element.get('type',None) + if (type is not None): + if (type == 'md5'): + resource.md5=fixity_element.text #FIXME - should check valid + elif (type == 'sha-1' or type == 'sha-256'): + self.logger.warning("Unsupported type (%s) in for %s" % (type,loc)) + return(resource) + + ##### ResourceContainer (Inventory or Changelist) methods ##### + + def resources_as_xml(self, resources, num_resources=None, capabilities=None, changelist=False): + """Return XML for a set of resources in sitemap format + + resources is either an iterable or iterator of Resource objects. + + If num_resources is not None then only that number will be written + before exiting. + """ + # will include capabilities if allowed and if there are some + namespaces = { 'xmlns': SITEMAP_NS, 'xmlns:rs': RS_NS } + if ( capabilities is not None and len(capabilities)>0 ): + namespaces['xmlns:xhtml'] = XHTML_NS + root = Element('urlset', namespaces) + if (changelist): + root.set('rs:type','changelist') + if (self.pretty_xml): + root.text="\n" + if ( capabilities is not None and len(capabilities)>0 ): + self.add_capabilities_to_etree(root,capabilities) + # now add the entries from either an iterable or an iterator + for r in resources: + e=self.resource_etree_element(r) + root.append(e) + if (num_resources is not None): + num_resources-=1 + if (num_resources==0): + break + # have tree, now serialize + tree = ElementTree(root); + xml_buf=StringIO.StringIO() + if (sys.version_info < (2,7)): + tree.write(xml_buf,encoding='UTF-8') + else: + tree.write(xml_buf,encoding='UTF-8',xml_declaration=True,method='xml') + return(xml_buf.getvalue()) + + def inventory_parse_xml(self, fh=None, etree=None, resources=None): + """Parse XML Sitemap from fh or etree and add resources to an Inventory object + + Returns the inventory. + + Also sets self.resources_created to be the number of resources created. + We adopt a very lax approach here. The parsing is properly namespace + aware but we search just for the elements wanted and leave everything + else alone. + + The one exception is detection of Sitemap indexes. If the root element + indicates a sitemapindex then an SitemapIndexError() is thrown + and the etree passed along with it. + """ + inventory = resources #use inventory locally but want common argument name + if (inventory is None): + inventory=self.inventory_class() + if (fh is not None): + etree=parse(fh) + elif (etree is None): + raise ValueError("Neither fh or etree set") + # check root element: urlset (for sitemap), sitemapindex or bad + if (etree.getroot().tag == '{'+SITEMAP_NS+"}urlset"): + self.resources_created=0 + for url_element in etree.findall('{'+SITEMAP_NS+"}url"): + r = self.resource_from_etree(url_element, self.resource_class) + try: + inventory.add( r ) + except InventoryDupeError: + self.logger.warning("dupe: %s (%s =? %s)" % + (r.uri,r.lastmod,inventory.resources[r.uri].lastmod)) + self.resources_created+=1 + inventory.capabilities = self.capabilities_from_etree(etree) + return(inventory) + elif (etree.getroot().tag == '{'+SITEMAP_NS+"}sitemapindex"): + raise SitemapIndexError("Got sitemapindex when expecting sitemap",etree) + else: + raise ValueError("XML is not sitemap or sitemapindex") + + def changelist_parse_xml(self, fh=None, etree=None, resources=None): + """Parse XML Sitemap from fh or etree and add resources to an Changelist object + + Returns the Changelist. + + Also sets self.resources_created to be the number of resources created. + We adopt a very lax approach here. The parsing is properly namespace + aware but we search just for the elements wanted and leave everything + else alone. + + The one exception is detection of Sitemap indexes. If the root element + indicates a sitemapindex then an SitemapIndexError() is thrown + and the etree passed along with it. + """ + changelist = resources #use inventory locally but want common argument name + if (changelist is None): + changelist=self.changelist_class() + if (fh is not None): + etree=parse(fh) + elif (etree is None): + raise ValueError("Neither fh or etree set") + # check root element: urlset (for sitemap), sitemapindex or bad + if (etree.getroot().tag == '{'+SITEMAP_NS+"}urlset"): + self.resources_created=0 + for url_element in etree.findall('{'+SITEMAP_NS+"}url"): + r = self.resource_from_etree(url_element, self.resourcechange_class) + changelist.add( r ) + self.resources_created+=1 + changelist.capabilities = self.capabilities_from_etree(etree) + return(changelist) + elif (etree.getroot().tag == '{'+SITEMAP_NS+"}sitemapindex"): + raise SitemapIndexError("Got sitemapindex when expecting sitemap",etree) + else: + raise ValueError("XML is not sitemap or sitemapindex") + + ##### Sitemap Index ##### + + def sitemapindex_as_xml(self, file=None, sitemaps={}, inventory=None, capabilities=None, changelist=False ): + """Return a sitemapindex as an XML string + + Format: + + + http://www.example.com/sitemap1.xml.gz + 2004-10-01T18:23:17+00:00 + + ...more... + + """ + include_capabilities = capabilities and (len(capabilities)>0) + namespaces = { 'xmlns': SITEMAP_NS } + if (include_capabilities): + namespaces['xmlns:xhtml'] = XHTML_NS + root = Element('sitemapindex', namespaces) + if (changelist): + root.set('rs:type','changelist') + if (self.pretty_xml): + root.text="\n" + if (include_capabilities): + self.add_capabilities_to_etree(root,capabilities) + for file in sitemaps.keys(): + try: + uri = self.mapper.dst_to_src(file) + except MapperError: + uri = 'file://'+file + self.logger.error("sitemapindex: can't map %s into URI space, writing %s" % (file,uri)) + # Make a Resource for the Sitemap and serialize + smr = Resource( uri=uri, timestamp=sitemaps[file] ) + root.append( self.resource_etree_element(smr, element_name='sitemap') ) + tree = ElementTree(root); + xml_buf=StringIO.StringIO() + if (sys.version_info < (2,7)): + tree.write(xml_buf,encoding='UTF-8') + else: + tree.write(xml_buf,encoding='UTF-8',xml_declaration=True,method='xml') + return(xml_buf.getvalue()) + + def sitemapindex_parse_xml(self, fh=None, etree=None, sitemapindex=None): + """Parse XML SitemapIndex from fh and return sitemap info + + Returns the SitemapIndex object. + + Also sets self.sitemaps_created to be the number of resources created. + We adopt a very lax approach here. The parsing is properly namespace + aware but we search just for the elements wanted and leave everything + else alone. + + The one exception is detection of a Sitemap when an index is expected. + If the root element indicates a sitemap then a SitemapIndexError() is + thrown and the etree passed along with it. + """ + if (sitemapindex is None): + sitemapindex=SitemapIndex() + if (fh is not None): + etree=parse(fh) + elif (etree is None): + raise ValueError("Neither fh or etree set") + # check root element: urlset (for sitemap), sitemapindex or bad + if (etree.getroot().tag == '{'+SITEMAP_NS+"}sitemapindex"): + self.sitemaps_created=0 + for sitemap_element in etree.findall('{'+SITEMAP_NS+"}sitemap"): + # We can parse the inside just like a element indicating a resource + sitemapindex.add( self.resource_from_etree(sitemap_element,self.resource_class) ) + self.sitemaps_created+=1 + return(sitemapindex) + sitemapindex.capabilities = self.capabilities_from_etree(etree) + elif (etree.getroot().tag == '{'+SITEMAP_NS+"}urlset"): + raise SitemapIndexError("Got sitemap when expecting sitemapindex",etree) + else: + raise ValueError("XML is not sitemap or sitemapindex") + + + ##### Capabilities ##### + + def add_capabilities_to_etree(self, etree, capabilities): + """ Add capabilities to the etree supplied + + Each capability is written out as on xhtml:link element where the + attributes are represented as a dictionary. + """ + for c in sorted(capabilities.keys()): + # make attributes by space concatenating any capability dict values + # that are arrays + atts = { 'href': c } + for a in capabilities[c]: + value=capabilities[c][a] + if (a == 'attributes'): + a='rel' + if (isinstance(value, str)): + atts[a]=value + else: + atts[a]=' '.join(value) + e = Element('xhtml:link', atts) + if (self.pretty_xml): + e.tail="\n" + etree.append(e) + + def capabilities_from_etree(self, etree): + """Read capabilities from sitemap or sitemapindex etree + """ + capabilities = {} + for link in etree.findall('{'+XHTML_NS+"}link"): + c = link.get('href') + if (c is None): + raise Exception("xhtml:link without href") + capabilities[c]={} + rel = link.get('rel') + #if (rel is None): + # raise Exception('xhtml:link href="%s" without rel attribute' % (c)) + if (rel is not None): + attributes = [] + for r in rel.split(' '): + attributes.append(r) + if (len(attributes)==1): + attributes = attributes[0] + capabilities[c]['attributes']=attributes + type = link.get('type') #fudge, take either + #if (type is None): + # raise Exception('xhtml:link href="%s" without type attribute' % (c)) + if (type is not None): + types = [] + for t in type.split(' '): + types.append(t) + if (len(types)==1): + types = types[0] + capabilities[c]['type']=types + # print capabilities[c] + #for meta in etree.findall('{'+XHTML_NS+"}meta"): + # print meta + return(capabilities) + + ##### Utility ##### + + def is_file_uri(self, uri): + """Return true is uri looks like a local file URI, false otherwise""" + return(re.match('file:',uri) or re.match('/',uri)) diff --git a/resync/test/test_changelist.py b/resync/test/test_changelist.py index e69de29..0c1b800 100644 --- a/resync/test/test_changelist.py +++ b/resync/test/test_changelist.py @@ -0,0 +1,87 @@ +import unittest +from resync.resource import Resource +from resync.changelist import ChangeList +from resync.inventory import Inventory + +class TestChangeList(unittest.TestCase): + + def test1_set_with_repeats(self): + src = ChangeList() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=1) ) + src.add( Resource('c',timestamp=1) ) + src.add( Resource('a',timestamp=2) ) + src.add( Resource('b',timestamp=2) ) + self.assertEqual(len(src), 5, "5 changes in changelist") + + def test2_with_repeats_again(self): + r1 = Resource(uri='a',size=1) + r2 = Resource(uri='b',size=2) + i = ChangeList() + i.add(r1) + i.add(r2) + self.assertEqual( len(i), 2 ) + # Can add another Resource with same URI + r1d = Resource(uri='a',size=10) + i.add(r1d) + self.assertEqual( len(i), 3 ) + + def test3_changelist(self): + src = ChangeList() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=2) ) + src.add( Resource('c',timestamp=3) ) + src.add( Resource('d',timestamp=4)) + src.add( Resource('e',timestamp=5) ) + self.assertEqual(len(src), 5, "5 things in src") + + def test4_iter(self): + i = ChangeList() + i.add( Resource('a',timestamp=1) ) + i.add( Resource('b',timestamp=2) ) + i.add( Resource('c',timestamp=3) ) + i.add( Resource('d',timestamp=4) ) + resources=[] + for r in i: + resources.append(r) + self.assertEqual(len(resources), 4) + self.assertEqual( resources[0].uri, 'a') + self.assertEqual( resources[3].uri, 'd') + + def test5_add_changed_resources(self): + added = Inventory() + added.add( Resource('a',timestamp=1) ) + added.add( Resource('d',timestamp=4)) + self.assertEqual(len(added), 2, "2 things in added inventory") + changes = ChangeList() + changes.add_changed_resources( added, changetype='created' ) + self.assertEqual(len(changes), 2, "2 things added") + i = iter(changes) + first = i.next() + self.assertEqual(first.uri, 'a', "changes[0].uri=a") + self.assertEqual(first.timestamp, 1, "changes[0].timestamp=1") + self.assertEqual(first.changetype, 'created') #, "changes[0].changetype=created") + second = i.next() + self.assertEqual(second.timestamp, 4, "changes[1].timestamp=4") + self.assertEqual(second.changetype, 'created', "changes[1].changetype=created") + # Now add some with updated (one same, one diff) + updated = Inventory() + updated.add( Resource('a',timestamp=5) ) + updated.add( Resource('b',timestamp=6)) + self.assertEqual(len(updated), 2, "2 things in updated inventory") + changes.add_changed_resources( updated, changetype='updated' ) + self.assertEqual(len(changes), 4, "4 = 2 old + 2 things updated") + # Make new inventory from the changes which should not have dupes + dst = Inventory() + dst.add( changes, replace=True ) + self.assertEqual(len(dst), 3, "3 unique resources") + self.assertEqual(dst.resources['a'].timestamp, 5 ) # 5 was later in last the 1 + self.assertEqual(dst.resources['a'].changetype, 'updated') + self.assertEqual(dst.resources['b'].timestamp, 6) + self.assertEqual(dst.resources['b'].changetype, 'updated') + self.assertEqual(dst.resources['d'].timestamp, 4) + self.assertEqual(dst.resources['d'].changetype, 'created') + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestChangeList) + unittest.TextTestRunner().run(suite) diff --git a/resync/test/test_client.py b/resync/test/test_client.py new file mode 100644 index 0000000..54ab897 --- /dev/null +++ b/resync/test/test_client.py @@ -0,0 +1,24 @@ +import unittest +import logging +from resync.client import Client, ClientFatalError + +class TestResource(unittest.TestCase): + + def test1_make_inventory_empty(self): + c = Client() + # No mapping is error + # + def wrap_inventory_property_call(c): + # do this because assertRaises( ClientFatalError, c.inventory ) doesn't work + return(c.inventory) + self.assertRaises( ClientFatalError, wrap_inventory_property_call, c ) + + def test2_bad_source_uri(self): + c = Client() + self.assertRaises( ClientFatalError, c.baseline_or_audit ) + c.set_mappings( ['http://example.org/bbb','/tmp/this_does_not_exist'] ) + self.assertRaises( ClientFatalError, c.baseline_or_audit ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestClientResource) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_dump.py b/resync/test/test_dump.py index e69de29..1485da9 100644 --- a/resync/test/test_dump.py +++ b/resync/test/test_dump.py @@ -0,0 +1,20 @@ +import unittest +from resync.dump import Dump, DumpError +from resync.inventory import Inventory +from resync.resource import Resource + +class TestDump(unittest.TestCase): + + def test00_dump_creation(self): + i=Inventory() + i.add( Resource('http://ex.org/a', size=1, path='resync/test/testdata/a') ) + i.add( Resource('http://ex.org/b', size=2, path='resync/test/testdata/b') ) + d=Dump() + d.check_files(inventory=i) + self.assertEqual(d.total_size, 28) + + #FIXME -- need some code to actually write and read dump + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestDump) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_inventory.py b/resync/test/test_inventory.py index e69de29..a483ea8 100644 --- a/resync/test/test_inventory.py +++ b/resync/test/test_inventory.py @@ -0,0 +1,129 @@ +import unittest +from resync.resource import Resource +from resync.inventory import Inventory, InventoryDupeError + +class TestInventory(unittest.TestCase): + + def test1_same(self): + src = Inventory() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=2) ) + dst = Inventory() + dst.add( Resource('a',timestamp=1) ) + dst.add( Resource('b',timestamp=2) ) + ( same, changed, deleted, added ) = dst.compare(src) + self.assertEqual( len(same), 2, "2 things unchanged" ) + i = iter(same) + self.assertEqual( i.next().uri, 'a', "first was a" ) + self.assertEqual( i.next().uri, 'b', "second was b" ) + self.assertEqual( len(changed), 0, "nothing changed" ) + self.assertEqual( len(deleted), 0, "nothing deleted" ) + self.assertEqual( len(added), 0, "nothing added" ) + + def test2_changed(self): + src = Inventory() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=2) ) + dst = Inventory() + dst.add( Resource('a',timestamp=3) ) + dst.add( Resource('b',timestamp=4) ) + ( same, changed, deleted, added ) = dst.compare(src) + self.assertEqual( len(same), 0, "0 things unchanged" ) + self.assertEqual( len(changed), 2, "2 things changed" ) + i = iter(changed) + self.assertEqual( i.next().uri, 'a', "first was a" ) + self.assertEqual( i.next().uri, 'b', "second was b" ) + self.assertEqual( len(deleted), 0, "nothing deleted" ) + self.assertEqual( len(added), 0, "nothing added" ) + + def test3_deleted(self): + src = Inventory() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=2) ) + dst = Inventory() + dst.add( Resource('a',timestamp=1) ) + dst.add( Resource('b',timestamp=2) ) + dst.add( Resource('c',timestamp=3) ) + dst.add( Resource('d',timestamp=4) ) + ( same, changed, deleted, added ) = dst.compare(src) + self.assertEqual( len(same), 2, "2 things unchanged" ) + self.assertEqual( len(changed), 0, "nothing changed" ) + self.assertEqual( len(deleted), 2, "c and d deleted" ) + i = iter(deleted) + self.assertEqual( i.next().uri, 'c', "first was c" ) + self.assertEqual( i.next().uri, 'd', "second was d" ) + self.assertEqual( len(added), 0, "nothing added" ) + + def test4_added(self): + src = Inventory() + src.add( Resource('a',timestamp=1) ) + src.add( Resource('b',timestamp=2) ) + src.add( Resource('c',timestamp=3) ) + src.add( Resource('d',timestamp=4) ) + dst = Inventory() + dst.add( Resource('a',timestamp=1) ) + dst.add( Resource('c',timestamp=3) ) + ( same, changed, deleted, added ) = dst.compare(src) + self.assertEqual( len(same), 2, "2 things unchanged" ) + self.assertEqual( len(changed), 0, "nothing changed" ) + self.assertEqual( len(deleted), 0, "nothing deleted" ) + self.assertEqual( len(added), 2, "b and d added" ) + i = iter(added) + self.assertEqual( i.next().uri, 'b', "first was b" ) + self.assertEqual( i.next().uri, 'd', "second was d" ) + + def test5_add(self): + r1 = Resource(uri='a',size=1) + r2 = Resource(uri='b',size=2) + i = Inventory() + i.add(r1) + self.assertRaises( InventoryDupeError, i.add, r1) + i.add(r2) + self.assertRaises( InventoryDupeError, i.add, r2) + # allow dupes + r1d = Resource(uri='a',size=10) + i.add(r1d,replace=True) + self.assertEqual( len(i), 2 ) + self.assertEqual( i.resources['a'].size, 10 ) + + def test5_add_iterable(self): + r1 = Resource(uri='a',size=1) + r2 = Resource(uri='b',size=2) + i = Inventory() + i.add( [r1,r2] ) + self.assertRaises( InventoryDupeError, i.add, r1) + self.assertRaises( InventoryDupeError, i.add, r2) + # allow dupes + r1d = Resource(uri='a',size=10) + i.add( [r1d] ,replace=True) + self.assertEqual( len(i), 2 ) + self.assertEqual( i.resources['a'].size, 10 ) + + def test6_has_md5(self): + r1 = Resource(uri='a') + r2 = Resource(uri='b') + i = Inventory() + self.assertFalse( i.has_md5() ) + i.add(r1) + i.add(r2) + self.assertFalse( i.has_md5() ) + r1.md5="aabbcc" + self.assertTrue( i.has_md5() ) + + def test7_iter(self): + i = Inventory() + i.add( Resource('a',timestamp=1) ) + i.add( Resource('b',timestamp=2) ) + i.add( Resource('c',timestamp=3) ) + i.add( Resource('d',timestamp=4) ) + resources=[] + for r in i: + resources.append(r) + self.assertEqual(len(resources), 4) + self.assertEqual( resources[0].uri, 'a') + self.assertEqual( resources[3].uri, 'd') + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestInventory) +# unittest.TextTestRunner(verbosity=1).run(suite) + unittest.TextTestRunner().run(suite) diff --git a/resync/test/test_inventory_builder.py b/resync/test/test_inventory_builder.py new file mode 100644 index 0000000..4ec1daa --- /dev/null +++ b/resync/test/test_inventory_builder.py @@ -0,0 +1,56 @@ +import unittest +import re +import os +import time +from resync.inventory_builder import InventoryBuilder +from resync.sitemap import Sitemap +from resync.mapper import Mapper + +class TestInventoryBuilder(unittest.TestCase): + + def setUp(self): + # Set timestamps (mtime) for test data. Timestamps on disk are + # in UTC so no conversion issues. + # Test case file_a: 1343236426 = 2012-07-25T17:13:46Z + # Test case file_b: 1000000000 = 2001-09-09T01:46:40Z + os.utime( "resync/test/testdata/dir1/file_a", (0, 1343236426 ) ) + os.utime( "resync/test/testdata/dir1/file_b", (0, 1000000000 ) ) + + def test1_simple_output(self): + ib = InventoryBuilder() + ib.mapper = Mapper(['http://example.org/t','resync/test/testdata/dir1']) + i = ib.from_disk() + self.assertEqual(Sitemap().resources_as_xml(i),'\nhttp://example.org/t/file_a2012-07-25T17:13:46Z20http://example.org/t/file_b2001-09-09T01:46:40Z45' ) + + def test2_pretty_output(self): + ib = InventoryBuilder() + ib.mapper = Mapper(['http://example.org/t','resync/test/testdata/dir1']) + i = ib.from_disk() + s = Sitemap() + s.pretty_xml=True + self.assertEqual(s.resources_as_xml(i),'\n\nhttp://example.org/t/file_a2012-07-25T17:13:46Z20\nhttp://example.org/t/file_b2001-09-09T01:46:40Z45\n' ) + + def test3_with_md5(self): + ib = InventoryBuilder(do_md5=True) + ib.mapper = Mapper(['http://example.org/t','resync/test/testdata/dir1']) + i = ib.from_disk() + s = Sitemap() + xml = s.resources_as_xml(i) + self.assertNotEqual( None, re.search('http://example.org/t/file_a[\w\:\-]+Z20a/Jv1mYBtSjS4LR\+qoft/Q==',xml) ) #must escape + in md5 + self.assertNotEqual( None, re.search('http://example.org/t/file_b[\w\:\-]+Z45RS5Uva4WJqxdbnvoGzneIQ==',xml) ) + + def test4_data(self): + ib = InventoryBuilder(do_md5=True) + ib.mapper = Mapper(['http://example.org/t','resync/test/testdata/dir1']) + i = ib.from_disk() + self.assertEqual( len(i), 2) + r1 = i.resources.get('http://example.org/t/file_a') + self.assertTrue( r1 is not None ) + self.assertEqual( r1.uri, 'http://example.org/t/file_a' ) + self.assertEqual( r1.lastmod, '2012-07-25T17:13:46Z' ) + self.assertEqual( r1.md5, 'a/Jv1mYBtSjS4LR+qoft/Q==' ) + self.assertEqual( r1.path, 'resync/test/testdata/dir1/file_a' ) + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestInventoryBuilder) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_mapper.py b/resync/test/test_mapper.py new file mode 100644 index 0000000..73bf78a --- /dev/null +++ b/resync/test/test_mapper.py @@ -0,0 +1,57 @@ +import unittest +from resync.mapper import Mapper, MapperError + +class TestMapper(unittest.TestCase): + + def test00_mapper_creation(self): + m1=Mapper( ['http://e.org/p/','/tmp/q/'] ) + self.assertEqual( len(m1), 1 ) + m2=Mapper( mappings=['http://e.org/p','/tmp/q'] ) + self.assertEqual( len(m2), 1 ) + self.assertEqual( str(m1), str(m2) ) + m3=Mapper( ['http://e.org/p/=/tmp/q/'] ) + self.assertEqual( len(m3), 1 ) + self.assertEqual( str(m1), str(m3) ) + m4=Mapper( ['http://e.org/p/=/tmp/q/','http://e.org/r/=/tmp/s/'] ) + m5=Mapper( ['http://e.org/r/=/tmp/s/','http://e.org/p/=/tmp/q/'] ) + self.assertEqual( len(m4), 2 ) + self.assertEqual( len(m5), 2 ) + self.assertNotEqual( str(m4), str(m5) ) + + + def test01_mapper_src_to_dst(self): + m=Mapper( ['http://e.org/p/','/tmp/q/'] ) + self.assertEqual( m.src_to_dst('http://e.org/p/'), '/tmp/q/') + self.assertEqual( m.src_to_dst('http://e.org/p/aa'), '/tmp/q/aa') + self.assertEqual( m.src_to_dst('http://e.org/p/aa/bb'), '/tmp/q/aa/bb') + self.assertEqual( m.src_to_dst('http://e.org/p/aa/bb/'), '/tmp/q/aa/bb/') + self.assertRaises( MapperError, m.src_to_dst, 'http://e.org/p' ) + self.assertRaises( MapperError, m.src_to_dst, 'http://e.org/pa' ) + self.assertRaises( MapperError, m.src_to_dst, 'nomatch' ) + + def test02_mapper_dst_to_src(self): + m=Mapper( ['http://e.org/p/','/tmp/q/'] ) + self.assertEqual( m.dst_to_src('/tmp/q/'), 'http://e.org/p/') + self.assertEqual( m.dst_to_src('/tmp/q/bb'), 'http://e.org/p/bb') + self.assertEqual( m.dst_to_src('/tmp/q/bb/cc'), 'http://e.org/p/bb/cc') + self.assertRaises( MapperError, m.dst_to_src, '/tmp/q' ) + self.assertRaises( MapperError, m.dst_to_src, '/tmp/qa') + self.assertRaises( MapperError, m.dst_to_src, 'nomatch' ) + + def test03_mapper2_src_to_dst(self): + m=Mapper( ['http://e.org/p=/tmp/q','http://e.org/r=/tmp/s'] ) + self.assertEqual( m.src_to_dst('http://e.org/p/'), '/tmp/q/') + self.assertEqual( m.src_to_dst('http://e.org/p/aa'), '/tmp/q/aa') + self.assertEqual( m.src_to_dst('http://e.org/r/'), '/tmp/s/') + self.assertEqual( m.src_to_dst('http://e.org/r/aa'), '/tmp/s/aa') + + def test04_mapper2_dst_to_src(self): + m=Mapper( ['http://e.org/p=/tmp/q','http://e.org/r=/tmp/s'] ) + self.assertEqual( m.dst_to_src('/tmp/q/'), 'http://e.org/p/') + self.assertEqual( m.dst_to_src('/tmp/q/bb'), 'http://e.org/p/bb') + self.assertEqual( m.dst_to_src('/tmp/s/'), 'http://e.org/r/') + self.assertEqual( m.dst_to_src('/tmp/s/bb'), 'http://e.org/r/bb') + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestMapper) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_resource.py b/resync/test/test_resource.py index e69de29..dd01eb5 100644 --- a/resync/test/test_resource.py +++ b/resync/test/test_resource.py @@ -0,0 +1,99 @@ +import unittest +import re +from resync.resource import Resource + +class TestResource(unittest.TestCase): + + def test1a_same(self): + r1 = Resource('a') + r2 = Resource('a') + self.assertEqual( r1, r1 ) + self.assertEqual( r1, r2 ) + + def test1b_same(self): + r1 = Resource(uri='a',timestamp=1234.0) + r2 = Resource(uri='a',timestamp=1234.0) + self.assertEqual( r1, r1 ) + self.assertEqual( r1, r2 ) + + def test1c_same(self): + """Same with lastmod instead of direct timestamp""" + r1 = Resource('a') + r1lm = '2012-01-01T00:00:00Z' + r1.lastmod = r1lm + r2 = Resource('a') + for r2lm in ('2012', + '2012-01', + '2012-01-01', + '2012-01-01T00:00Z', + '2012-01-01T00:00:00Z', + '2012-01-01T00:00:00.000000Z', + '2012-01-01T00:00:00.000000000000Z', + '2012-01-01T00:00:00.000000000001Z', #below resolution + '2012-01-01T00:00:00.00+00:00', + '2012-01-01T00:00:00.00-00:00', + '2012-01-01T02:00:00.00-02:00', + '2011-12-31T23:00:00.00+01:00' + ): + r2.lastmod = r2lm + self.assertEqual( r1.timestamp, r2.timestamp, ('%s (%f) == %s (%f)' % (r1lm,r1.timestamp,r2lm,r2.timestamp)) ) + self.assertEqual( r1, r2 ) + + def test1d_same(self): + """Same with slight timestamp diff""" + r1 = Resource('a') + r1.lastmod='2012-01-02T01:02:03Z' + r2 = Resource('a') + r2.lastmod='2012-01-02T01:02:03.99Z' + self.assertNotEqual( r1.timestamp, r2.timestamp ) + self.assertEqual( r1, r2 ) + + def test2a_diff(self): + r1 = Resource('a') + r2 = Resource('b') + self.assertNotEqual(r1,r2) + + def test2b_diff(self): + r1 = Resource('a',lastmod='2012-01-11') + r2 = Resource('a',lastmod='2012-01-22') + #print 'r1 == r2 : '+str(r1==r2) + self.assertNotEqual( r1, r2 ) + + def test4_bad_lastmod(self): + def setlastmod(r,v): + r.lastmod=v + r = Resource('4') + # Bad formats + self.assertRaises( ValueError, setlastmod, r, "bad_lastmod" ) + self.assertRaises( ValueError, setlastmod, r, "" ) + self.assertRaises( ValueError, setlastmod, r, "2012-13-01" ) + self.assertRaises( ValueError, setlastmod, r, "2012-12-32" ) + self.assertRaises( ValueError, setlastmod, r, "2012-11-01T10:10:60" ) + self.assertRaises( ValueError, setlastmod, r, "2012-11-01T10:10:59.9x" ) + # Valid ISO8601 but not alloed in W3C Datetime + self.assertRaises( ValueError, setlastmod, r, "2012-11-01T01:01:01" ) + self.assertRaises( ValueError, setlastmod, r, "2012-11-01 01:01:01Z" ) + self.assertRaises( ValueError, setlastmod, r, "2012-11-01T01:01:01+0000" ) + self.assertRaises( ValueError, setlastmod, r, "2012-11-01T01:01:01-1000" ) + + def test5_lastmod_roundtrips(self): + r = Resource('a') + r.lastmod='2012-03-14' + self.assertEqual( r.lastmod, '2012-03-14T00:00:00Z' ) + r.lastmod='2012-03-14T00:00:00+00:00' + print r.timestamp + self.assertEqual( r.lastmod, '2012-03-14T00:00:00Z' ) + r.lastmod='2012-03-14T00:00:00-00:00' + print r.timestamp + self.assertEqual( r.lastmod, '2012-03-14T00:00:00Z' ) + r.lastmod='2012-03-14T18:37:36Z' + print r.timestamp + self.assertEqual( r.lastmod, '2012-03-14T18:37:36Z' ) + + def test6_str(self): + r1 = Resource('abc',lastmod='2012-01-01') + self.assertTrue( re.match( r"\[ abc \| 2012-01-01T", str(r1) ) ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestResource) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_resource_container.py b/resync/test/test_resource_container.py index e69de29..421754c 100644 --- a/resync/test/test_resource_container.py +++ b/resync/test/test_resource_container.py @@ -0,0 +1,29 @@ +import unittest +from resync.resource import Resource +from resync.resource_container import ResourceContainer + +class TestResourceContainer(unittest.TestCase): + + def test1_create_and_add(self): + rc = ResourceContainer( resources=[] ) + self.assertEqual( len(rc.resources), 0, "empty" ) + rc.resources.append( Resource('a',timestamp=1) ) + rc.resources.append( Resource('b',timestamp=2) ) + self.assertEqual( len(rc.resources), 2, "two resources" ) + + def test2_iter(self): + rc = ResourceContainer( resources=[] ) + rc.resources.append( Resource('a',timestamp=1) ) + rc.resources.append( Resource('b',timestamp=2) ) + rc.resources.append( Resource('c',timestamp=3) ) + rc.resources.append( Resource('d',timestamp=4) ) + resources=[] + for r in rc: + resources.append(r) + self.assertEqual(len(resources), 4) + self.assertEqual( resources[0].uri, 'a') + self.assertEqual( resources[3].uri, 'd') + +if __name__ == '__main__': + suite = unittest.defaultTestLoader.loadTestsFromTestCase(TestResourceContainer) + unittest.TextTestRunner().run(suite) diff --git a/resync/test/test_sitemap.py b/resync/test/test_sitemap.py new file mode 100644 index 0000000..230d371 --- /dev/null +++ b/resync/test/test_sitemap.py @@ -0,0 +1,164 @@ +import sys +import unittest +import StringIO +from resync.resource import Resource +from resync.inventory import Inventory +from resync.sitemap import Sitemap, SitemapIndexError + +# etree gives ParseError in 2.7, ExpatError in 2.6 +etree_error_class = None +if (sys.version_info < (2,7)): + from xml.parsers.expat import ExpatError + etree_error_class = ExpatError +else: + from xml.etree.ElementTree import ParseError + etree_error_class = ParseError + +class TestSitemap(unittest.TestCase): + + def test_01_resource_str(self): + r1 = Resource('a3') + r1.lastmod='2012-01-11T01:02:03Z' + self.assertEqual( Sitemap().resource_as_xml(r1), "\na32012-01-11T01:02:03Z" ) + + def test_02_resource_str(self): + r1 = Resource('3b',1234.1,9999,'ab54de') + self.assertEqual( Sitemap().resource_as_xml(r1), "\n3b1970-01-01T00:20:34.100000Z9999ab54de" ) + + def test_08_print(self): + r1 = Resource(uri='a',lastmod='2001-01-01',size=1234) + r2 = Resource(uri='b',lastmod='2002-02-02',size=56789) + r3 = Resource(uri='c',lastmod='2003-03-03',size=0) + m = Inventory() + m.add(r1) + m.add(r2) + m.add(r3) + #print m + self.assertEqual( Sitemap().resources_as_xml(m), "\na2001-01-01T00:00:00Z1234b2002-02-02T00:00:00Z56789c2003-03-03T00:00:00Z0") + + def test_09_print_subset(self): + r1 = Resource(uri='a',lastmod='2001-01-01',size=1234) + r2 = Resource(uri='b',lastmod='2002-02-02',size=56789) + r3 = Resource(uri='d',lastmod='2003-03-04',size=444) + m = Inventory() + m.add(r1) + m.add(r2) + m.add(r3) + self.assertEqual( Sitemap().resources_as_xml(m, num_resources=2), "\na2001-01-01T00:00:00Z1234b2002-02-02T00:00:00Z56789") + + def test_09s_print_from_iter(self): + r1 = Resource(uri='a',lastmod='2001-01-01',size=1234) + r2 = Resource(uri='b',lastmod='2002-02-02',size=56789) + r3 = Resource(uri='c',lastmod='2003-03-03',size=0) + r4 = Resource(uri='d',lastmod='2004-04-04',size=444) + m = Inventory() + m.add(r1) + m.add(r2) + m.add(r3) + m.add(r4) + i = iter(m) + self.assertEqual( Sitemap().resources_as_xml(i, num_resources=2), "\na2001-01-01T00:00:00Z1234b2002-02-02T00:00:00Z56789") + self.assertEqual( Sitemap().resources_as_xml(i, num_resources=1), "\nc2003-03-03T00:00:00Z0") + self.assertEqual( Sitemap().resources_as_xml(i), "\nd2004-04-04T00:00:00Z444") + + def test_10_sitemap(self): + xml='\n\ +\ +http://e.com/a2012-03-14T18:37:36Z12Q2hlY2sgSW50ZWdyaXR5IQ==\ +' + s=Sitemap() + i=s.inventory_parse_xml(fh=StringIO.StringIO(xml)) + self.assertEqual( s.resources_created, 1, 'got 1 resources') + r=i.resources['http://e.com/a'] + self.assertTrue( r is not None, 'got the uri expected') + self.assertEqual( r.uri, 'http://e.com/a' ) + self.assertEqual( r.lastmod, '2012-03-14T18:37:36Z' ) + self.assertEqual( r.size, 12 ) + self.assertEqual( r.md5, 'Q2hlY2sgSW50ZWdyaXR5IQ==' ) + + def test_11_parse_2(self): + xml='\n\ +\ +/tmp/rs_test/src/file_a2012-03-14T18:37:36Z12\ +/tmp/rs_test/src/file_b2012-03-14T18:37:36Z32\ +' + s=Sitemap() + i=s.inventory_parse_xml(fh=StringIO.StringIO(xml)) + self.assertEqual( s.resources_created, 2, 'got 2 resources') + + def test_13_parse_illformed(self): + s=Sitemap() + # ExpatError in python2.6, ParserError in 2.7 + self.assertRaises( etree_error_class, s.inventory_parse_xml, StringIO.StringIO('not xml') ) + self.assertRaises( etree_error_class, s.inventory_parse_xml, StringIO.StringIO('something') ) + + def test_13_parse_valid_xml_but_other(self): + s=Sitemap() + self.assertRaises( ValueError, s.inventory_parse_xml, StringIO.StringIO(' ') ) + self.assertRaises( ValueError, s.inventory_parse_xml, StringIO.StringIO(' ') ) + + def test_14_parse_sitemapindex_as_sitemap(self): + s=Sitemap() + self.assertRaises( SitemapIndexError, s.inventory_parse_xml, StringIO.StringIO(' ') ) + + def test_20_parse_sitemapindex_empty(self): + s=Sitemap() + si = s.sitemapindex_parse_xml( fh=StringIO.StringIO(' ') ) + self.assertEqual( s.sitemaps_created, 0, '0 sitemaps in sitemapindex') + self.assertEqual( len(si.resources), 0, '0 sitemaps') + + def test_21_parse_sitemapindex(self): + s=Sitemap() + si = s.sitemapindex_parse_xml( fh=StringIO.StringIO('aaabbb') ) + self.assertEqual( s.sitemaps_created, 2, '2 sitemaps in sitemapindex') + self.assertEqual( len(si.resources), 2, '2 sitemaps') + sms = sorted(si.resources.keys()) + self.assertEqual( sms, ['aaa','bbb'] ) + # add a couple more + s.sitemapindex_parse_xml( fh=StringIO.StringIO('ccdd'), sitemapindex=si ) + self.assertEqual( s.sitemaps_created, 2, '2 sitemaps created to sitemapindex') + self.assertEqual( len(si.resources), 4, '4 sitemaps total') + sms = sorted(si.resources.keys()) + self.assertEqual( sms, ['aaa','bbb', 'cc', 'dd'] ) + + def test_22_parse_sitemapindex_file(self): + s=Sitemap() + fh=open('resync/test/testdata/sitemapindex1/sitemap.xml') + si = s.sitemapindex_parse_xml( fh=fh ) + self.assertEqual( s.sitemaps_created, 3, '3 sitemaps in sitemapindex') + self.assertEqual( len(si.resources), 3, '3 sitemaps') + sms = sorted(si.resources.keys()) + self.assertEqual( sms, ['http://localhost:8888/sitemap00000.xml','http://localhost:8888/sitemap00001.xml','http://localhost:8888/sitemap00002.xml'] ) + self.assertEqual( si.resources['http://localhost:8888/sitemap00000.xml'].lastmod, '2012-06-13T18:09:13Z' ) + + def test_21_parse_multi_sitemapindex(self): + i = Sitemap().read( uri='resync/test/testdata/sitemapindex2/sitemap.xml' ) + self.assertEqual( len(i.resources), 17, '17 resources from 3 sitemaps') + sr = sorted(i.resources.keys()) + self.assertEqual( sr[0], 'http://localhost:8888/resources/1' ) + self.assertEqual( sr[1], 'http://localhost:8888/resources/10' ) + self.assertEqual( sr[2], 'http://localhost:8888/resources/100' ) + self.assertEqual( sr[3], 'http://localhost:8888/resources/1000' ) + self.assertEqual( sr[16], 'http://localhost:8888/resources/826' ) + + def test_30_parse_changelist(self): + xml='\n\ +\ +/tmp/rs_test/src/file_a2012-03-14T18:37:36Z12\ +/tmp/rs_test/src/file_b2012-03-14T18:37:36Z32\ +' + s=Sitemap() + s.resource_class=Resource + c=s.changelist_parse_xml(fh=StringIO.StringIO(xml)) + self.assertEqual( s.resources_created, 2, 'got 2 resources') + i = iter(c) + r1 = i.next() + self.assertEqual( r1.uri, '/tmp/rs_test/src/file_a' ) + self.assertEqual( r1.changetype, 'UPDATED' ) + r2 = i.next() + self.assertEqual( r2.uri, '/tmp/rs_test/src/file_b' ) + self.assertEqual( r2.changetype, None ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestSitemap) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_url_authority.py b/resync/test/test_url_authority.py new file mode 100644 index 0000000..66914aa --- /dev/null +++ b/resync/test/test_url_authority.py @@ -0,0 +1,35 @@ +import unittest + +from resync.url_authority import UrlAuthority + +class TestUrlAuthority(unittest.TestCase): + + def test1(self): + uauth = UrlAuthority( 'http://example.org/sitemap.xml' ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/sitemap.xml' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/sitemap.xml?anything' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/sitemap.xml#frag' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/same_level' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/one/deeper' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/one/two/deeper' ) ) + self.assertTrue( uauth.has_authority_over( 'http://example.org/' ) ) + self.assertTrue( uauth.has_authority_over( 'http://sub.example.org/subdomain' ) ) + self.assertTrue( uauth.has_authority_over( 'http://sub.sub.example.org/subsubdomain' ) ) + + def test2_no_authority(self): + uauth = UrlAuthority( 'http://example.org/dir/sitemap.xml' ) + self.assertFalse( uauth.has_authority_over( 'http://example.org/sitemap.xml' ) ) + self.assertFalse( uauth.has_authority_over( 'http://sub.example.org/sitemap.xml' ) ) + self.assertFalse( uauth.has_authority_over( 'https://example.org/dir/sitemap.xml' ) ) + self.assertFalse( uauth.has_authority_over( 'unknown://example.org/dir/sitemap.xml' ) ) + + def test3_domains(self): + uauth = UrlAuthority( 'http://a.example.org/sitemap.xml' ) + self.assertTrue( uauth.has_authority_over( 'http://a.example.org/sitemap.xml' ) ) + self.assertTrue( uauth.has_authority_over( 'http://sub.a.example.org/sitemap.xml' ) ) + self.assertFalse( uauth.has_authority_over( 'http://b.example.org/sitemap.xml' ) ) + self.assertFalse( uauth.has_authority_over( 'http://sub.b.example.org/sitemap.xml' ) ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestUrlAuthority) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/test_util.py b/resync/test/test_util.py deleted file mode 100644 index e69de29..0000000 diff --git a/resync/test/test_utils.py b/resync/test/test_utils.py new file mode 100644 index 0000000..ee7520e --- /dev/null +++ b/resync/test/test_utils.py @@ -0,0 +1,16 @@ +import unittest +import resync.utils + +class TestUtill(unittest.TestCase): + + def test1_string(self): + self.assertEqual( resync.utils.compute_md5_for_string('A file\n'), + 'j912liHgA/48DCHpkptJHg==') + + def test2_file(self): + # Should be same as the string above + self.assertEqual( resync.utils.compute_md5_for_file('resync/test/testdata/a'), + 'j912liHgA/48DCHpkptJHg==') + +if __name__ == '__main__': + unittest.main() diff --git a/resync/test/test_w3c_datetime.py b/resync/test/test_w3c_datetime.py new file mode 100644 index 0000000..37f3b85 --- /dev/null +++ b/resync/test/test_w3c_datetime.py @@ -0,0 +1,83 @@ +import unittest +import re +from resync.w3c_datetime import str_to_datetime, datetime_to_str + +def rt(dts): + """ Do simple round-trip """ + return( datetime_to_str(str_to_datetime( dts )) ) + +class TestW3cDatetime(unittest.TestCase): + + def test1_datetime_to_str(self): + """Writing...""" + self.assertEqual( datetime_to_str(0), "1970-01-01T00:00:00Z" ) + self.assertEqual( datetime_to_str(0.000001), "1970-01-01T00:00:00.000001Z" ) + self.assertEqual( datetime_to_str(0.1), "1970-01-01T00:00:00.100000Z" ) + self.assertEqual( datetime_to_str(1), "1970-01-01T00:00:01Z" ) + self.assertEqual( datetime_to_str(60), "1970-01-01T00:01:00Z" ) + self.assertEqual( datetime_to_str(60*60), "1970-01-01T01:00:00Z" ) + self.assertEqual( datetime_to_str(60*60*24), "1970-01-02T00:00:00Z" ) + self.assertEqual( datetime_to_str(60*60*24*31), "1970-02-01T00:00:00Z" ) + self.assertEqual( datetime_to_str(60*60*24*365), "1971-01-01T00:00:00Z" ) + # + self.assertEqual( datetime_to_str(1234567890), "2009-02-13T23:31:30Z" ) + + def test2_str_to_datetime(self): + """Reading...""" + self.assertEqual( str_to_datetime("1970-01-01T00:00:00Z"), 0 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00.000Z"), 0 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00+00:00"), 0 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00-00:00"), 0 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00.000001Z"), 0.000001 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00.1Z"), 0.1 ) + self.assertEqual( str_to_datetime("1970-01-01T00:00:00.100000Z"), 0.1 ) + # + self.assertEqual( str_to_datetime("2009-02-13T23:31:30Z"), 1234567890 ) + + def test2_same(self): + """Datetime values that are the same""" + astr = '2012-01-01T00:00:00Z' + a = str_to_datetime( astr ) + for bstr in ('2012', + '2012-01', + '2012-01-01', + '2012-01-01T00:00Z', + '2012-01-01T00:00:00Z', + '2012-01-01T00:00:00.000000Z', + '2012-01-01T00:00:00.000000000000Z', + '2012-01-01T00:00:00.000000000001Z', #below resolution + '2012-01-01T00:00:00.00+00:00', + '2012-01-01T00:00:00.00-00:00', + '2012-01-01T02:00:00.00-02:00', + '2011-12-31T23:00:00.00+01:00' + ): + b = str_to_datetime( bstr ) + self.assertEqual( a, b, ('%s (%f) == %s (%f)' % (astr,a,bstr,b)) ) + + def test4_bad_str(self): + # Bad formats + self.assertRaises( ValueError, str_to_datetime, "bad_lastmod" ) + self.assertRaises( ValueError, str_to_datetime, "" ) + self.assertRaises( ValueError, str_to_datetime, "2012-13-01" ) + self.assertRaises( ValueError, str_to_datetime, "2012-12-32" ) + self.assertRaises( ValueError, str_to_datetime, "2012-11-01T10:10:60" ) + self.assertRaises( ValueError, str_to_datetime, "2012-11-01T10:10:59.9x" ) + # Valid ISO8601 but not allowed in W3C Datetime + self.assertRaises( ValueError, str_to_datetime, "2012-11-01T01:01:01" ) + self.assertRaises( ValueError, str_to_datetime, "2012-11-01 01:01:01Z" ) + self.assertRaises( ValueError, str_to_datetime, "2012-11-01T01:01:01+0000" ) + self.assertRaises( ValueError, str_to_datetime, "2012-11-01T01:01:01-1000" ) + + def test5_roundtrips(self): + self.assertEqual( rt('2012-03-14T00:00:00+00:00'), + '2012-03-14T00:00:00Z') + self.assertEqual( rt('2012-03-14T00:00:00-00:00'), + '2012-03-14T00:00:00Z') + self.assertEqual( rt('2012-03-14T11:00:00-11:00' ), + '2012-03-14T00:00:00Z') + self.assertEqual( rt('2012-03-14T18:37:36Z' ), + '2012-03-14T18:37:36Z' ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestW3cDatetime) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/testdata/a b/resync/test/testdata/a new file mode 100644 index 0000000..51f466f --- /dev/null +++ b/resync/test/testdata/a @@ -0,0 +1 @@ +A file diff --git a/resync/test/testdata/b b/resync/test/testdata/b new file mode 100644 index 0000000..b804125 --- /dev/null +++ b/resync/test/testdata/b @@ -0,0 +1 @@ +nother file called b diff --git a/resync/test/testdata/c b/resync/test/testdata/c new file mode 100644 index 0000000..a271ed7 --- /dev/null +++ b/resync/test/testdata/c @@ -0,0 +1 @@ +A third file, called c ;-) \ No newline at end of file diff --git a/resync/test/testdata/dir1/file_a b/resync/test/testdata/dir1/file_a new file mode 100644 index 0000000..7a446cf --- /dev/null +++ b/resync/test/testdata/dir1/file_a @@ -0,0 +1 @@ +I am file a in dir1 diff --git a/resync/test/testdata/dir1/file_b b/resync/test/testdata/dir1/file_b new file mode 100644 index 0000000..06a57a5 --- /dev/null +++ b/resync/test/testdata/dir1/file_b @@ -0,0 +1 @@ +I am file b in dir1, I am bigger than file_a diff --git a/resync/test/testdata/dir2/file_x b/resync/test/testdata/dir2/file_x new file mode 100644 index 0000000..4d65398 --- /dev/null +++ b/resync/test/testdata/dir2/file_x @@ -0,0 +1 @@ +I am the mysterious file_x in testdata/dir2. Wow. Pazow. Zap!!! \ No newline at end of file diff --git a/resync/test/testdata/examples_from_spec/ex2_1.xml b/resync/test/testdata/examples_from_spec/ex2_1.xml new file mode 100644 index 0000000..1a968f7 --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex2_1.xml @@ -0,0 +1,9 @@ + + + + http://example.com/res1 + + + http://example.com/res2 + + diff --git a/resync/test/testdata/examples_from_spec/ex2_3.xml b/resync/test/testdata/examples_from_spec/ex2_3.xml new file mode 100644 index 0000000..6f89960 --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex2_3.xml @@ -0,0 +1,15 @@ + + + + + http://example.com/res1 + 2012-08-08T08:15:00Z + + + + http://example.com/res2 + 2012-08-08T13:22:00Z + + + diff --git a/resync/test/testdata/examples_from_spec/ex2_4.xml b/resync/test/testdata/examples_from_spec/ex2_4.xml new file mode 100644 index 0000000..21e147a --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex2_4.xml @@ -0,0 +1,17 @@ + + + + + http://example.com/res1 + 2012-08-08T08:15:00Z + + + + http://example.com/res2 + 2012-08-08T13:22:00Z + + + diff --git a/resync/test/testdata/examples_from_spec/ex3_4.xml b/resync/test/testdata/examples_from_spec/ex3_4.xml new file mode 100644 index 0000000..ca4daa2 --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex3_4.xml @@ -0,0 +1,14 @@ + + + + + http://example.com/res1 + 2012-08-08T08:15:00Z + + + http://example.com/res2 + 2012-08-08T13:22:00Z + + diff --git a/resync/test/testdata/examples_from_spec/ex3_5.xml b/resync/test/testdata/examples_from_spec/ex3_5.xml new file mode 100644 index 0000000..22320a8 --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex3_5.xml @@ -0,0 +1,16 @@ + + + + + http://example.com/res1 + 2012-08-08T08:15:00Z + Q2hlY2sgSW50ZWdyaXR5IQ== + + + http://example.com/res2 + 2012-08-08T13:22:00Z + A7kjY2sgSW50ZWdyaX6sgt== + + diff --git a/resync/test/testdata/examples_from_spec/ex3_6.xml b/resync/test/testdata/examples_from_spec/ex3_6.xml new file mode 100644 index 0000000..0afc1a7 --- /dev/null +++ b/resync/test/testdata/examples_from_spec/ex3_6.xml @@ -0,0 +1,18 @@ + + + + + http://example.com/res1 + 2012-08-08T08:15:00Z + Q2hlY2sgSW50ZWdyaXR5IQ== + 15672 + + + http://example.com/res2 + 2012-08-08T13:22:00Z + A7kjY2sgSW50ZWdyaX6sgt== + 93660664 + + diff --git a/resync/test/testdata/sitemapindex1/sitemap.xml b/resync/test/testdata/sitemapindex1/sitemap.xml new file mode 100644 index 0000000..cffada3 --- /dev/null +++ b/resync/test/testdata/sitemapindex1/sitemap.xml @@ -0,0 +1,6 @@ + + +http://localhost:8888/sitemap00000.xml2012-06-13T18:09:13Z +http://localhost:8888/sitemap00001.xml2012-06-13T18:09:13Z +http://localhost:8888/sitemap00002.xml2012-06-13T18:09:13Z + diff --git a/resync/test/testdata/sitemapindex1/sitemap00000.xml b/resync/test/testdata/sitemapindex1/sitemap00000.xml new file mode 100644 index 0000000..24c05d7 --- /dev/null +++ b/resync/test/testdata/sitemapindex1/sitemap00000.xml @@ -0,0 +1,9 @@ + + +http://localhost:8888/resources/12012-06-13T17:13:47+00:00794 +http://localhost:8888/resources/102012-06-13T17:13:47+00:00491 +http://localhost:8888/resources/1002012-06-13T17:13:47+00:00631 +http://localhost:8888/resources/10002012-06-13T17:13:47+00:00167 +http://localhost:8888/resources/10012012-06-13T17:13:51+00:00983 +http://localhost:8888/resources/1012012-06-13T17:13:47+00:00533 + diff --git a/resync/test/testdata/sitemapindex1/sitemap00001.xml b/resync/test/testdata/sitemapindex1/sitemap00001.xml new file mode 100644 index 0000000..81c4d3f --- /dev/null +++ b/resync/test/testdata/sitemapindex1/sitemap00001.xml @@ -0,0 +1,8 @@ + + +http://localhost:8888/resources/4592012-06-13T17:13:47-00:00287 +http://localhost:8888/resources/462012-06-13T17:13:47-00:00727 +http://localhost:8888/resources/4602012-06-13T17:13:47-00:00671 +http://localhost:8888/resources/4612012-06-13T17:13:47-00:00886 +http://localhost:8888/resources/4622012-06-13T17:13:47-00:00388 + diff --git a/resync/test/testdata/sitemapindex1/sitemap00002.xml b/resync/test/testdata/sitemapindex1/sitemap00002.xml new file mode 100644 index 0000000..d0a345a --- /dev/null +++ b/resync/test/testdata/sitemapindex1/sitemap00002.xml @@ -0,0 +1,9 @@ + + +http://localhost:8888/resources/8212012-06-13T17:13:47549 +http://localhost:8888/resources/8222012-06-13T17:13:47515 +http://localhost:8888/resources/8232012-06-13T17:13:47189 +http://localhost:8888/resources/8242012-06-13T17:13:47813 +http://localhost:8888/resources/8252012-06-13T17:13:47959 +http://localhost:8888/resources/8262012-06-13T17:13:47324 + diff --git a/resync/test/testdata/sitemapindex2/sitemap.xml b/resync/test/testdata/sitemapindex2/sitemap.xml new file mode 100644 index 0000000..d3eeec6 --- /dev/null +++ b/resync/test/testdata/sitemapindex2/sitemap.xml @@ -0,0 +1,6 @@ + + +resync/test/testdata/sitemapindex2/sitemap00000.xml2012-06-13T18:09:13Z +resync/test/testdata/sitemapindex2/sitemap00001.xml2012-06-13T18:09:13Z +resync/test/testdata/sitemapindex2/sitemap00002.xml2012-06-13T18:09:13Z + diff --git a/resync/test/testdata/sitemapindex2/sitemap00000.xml b/resync/test/testdata/sitemapindex2/sitemap00000.xml new file mode 100644 index 0000000..96ed43c --- /dev/null +++ b/resync/test/testdata/sitemapindex2/sitemap00000.xml @@ -0,0 +1,9 @@ + + +http://localhost:8888/resources/12012-06-13T17:13:47Z794 +http://localhost:8888/resources/102012-06-13T17:13:47Z491 +http://localhost:8888/resources/1002012-06-13T17:13:47Z631 +http://localhost:8888/resources/10002012-06-13T17:13:47Z167 +http://localhost:8888/resources/10012012-06-13T17:13:51Z983 +http://localhost:8888/resources/1012012-06-13T17:13:47Z533 + diff --git a/resync/test/testdata/sitemapindex2/sitemap00001.xml b/resync/test/testdata/sitemapindex2/sitemap00001.xml new file mode 100644 index 0000000..8b00f47 --- /dev/null +++ b/resync/test/testdata/sitemapindex2/sitemap00001.xml @@ -0,0 +1,8 @@ + + +http://localhost:8888/resources/4592012-06-13T17:13:47Z287 +http://localhost:8888/resources/462012-06-13T17:13:47Z727 +http://localhost:8888/resources/4602012-06-13T17:13:47Z671 +http://localhost:8888/resources/4612012-06-13T17:13:47Z886 +http://localhost:8888/resources/4622012-06-13T17:13:47Z388 + diff --git a/resync/test/testdata/sitemapindex2/sitemap00002.xml b/resync/test/testdata/sitemapindex2/sitemap00002.xml new file mode 100644 index 0000000..3aa51b2 --- /dev/null +++ b/resync/test/testdata/sitemapindex2/sitemap00002.xml @@ -0,0 +1,9 @@ + + +http://localhost:8888/resources/8212012-06-13T17:13:47Z549 +http://localhost:8888/resources/8222012-06-13T17:13:47Z515 +http://localhost:8888/resources/8232012-06-13T17:13:47Z189 +http://localhost:8888/resources/8242012-06-13T17:13:47Z813 +http://localhost:8888/resources/8252012-06-13T17:13:47Z959 +http://localhost:8888/resources/8262012-06-13T17:13:47Z324 + diff --git a/resync/url_authority.py b/resync/url_authority.py new file mode 100644 index 0000000..856f0b3 --- /dev/null +++ b/resync/url_authority.py @@ -0,0 +1,39 @@ +"""Determine whether one resource can speak authoritatively about another""" + +import urlparse +import os.path + +class UrlAuthority(object): + + def __init__(self, url=None): + self.url = url + if (self.url is not None): + self.set_master(self.url) + else: + self.master_scheme='none' + self.master_netloc='none.none.none' + self.master_path='/not/very/likely' + + def set_master(self, url): + """Set the master url that this object works with""" + m = urlparse.urlparse(url) + self.master_scheme=m.scheme + self.master_netloc=m.netloc + self.master_path=os.path.dirname(m.path) + + def has_authority_over(self, url): + """Returns True of the current master has authority over url""" + s = urlparse.urlparse(url) + if (s.scheme != self.master_scheme): + return(False) + if (s.netloc != self.master_netloc): + if (not s.netloc.endswith('.'+self.master_netloc)): + return(False) + #Maybe should allow parallel for 3+ components, eg. a.example.org, b.example.org + path = os.path.dirname(s.path) + if (path != self.master_path and + not path.startswith(self.master_path)): + return(False) + return(True) + + diff --git a/resync/utils.py b/resync/utils.py index e69de29..e19eb2f 100644 --- a/resync/utils.py +++ b/resync/utils.py @@ -0,0 +1,64 @@ +""" +util.py: A collection of common util functions used in source and/or client. + +""" + +from logging import Formatter +from datetime import datetime + +class UTCFormatter(Formatter): + # based on http://bit.ly/T2n3Xk + def formatTime(self, record, datefmt=None): + timestamp = record.created + return datetime.utcfromtimestamp(timestamp).isoformat() + 'Z' + +"""Compute digests for ResourceSync + +These are all base64 encoded according to the rules of +http://www.ietf.org/rfc/rfc4648.txt + +MD5 + +ResourceSync defined to be the same as for Content-MD5 in HTTP, +http://www.ietf.org/rfc/rfc2616.txt which, in turn, defined the +digest string as the "base64 of 128 bit MD5 digest as per RFC 1864" +http://www.ietf.org/rfc/rfc1864.txt + +Unfortunately, RFC1864 is rather vague and contains only and example +which doesn't use encoding characters for 62 or 63. It points to +RFC1521 to describe base64 which is explicit that the encoding alphabet +is [A-Za-z0-9+/] with = to pad. + +The above corresponds with the alphabet of "3. Base 64 Encoding" in RFC3548 +http://www.ietf.org/rfc/rfc3548.txt +and not the url safe version, "Base 64 Encoding with URL and Filename Safe +Alphabet" which replaces + and / with - and _ respectively. + +This is the same as the alphabet of "4. Base 64 Encoding" in RFC4648 +http://www.ietf.org/rfc/rfc4648.txt. + +This algorithm is implemented by base64.standard_b64encode() or +base64.b64encode() with no altchars specified. Available in python2.4 and +up [http://docs.python.org/library/base64.html] +""" +import base64 +import hashlib + +def compute_md5_for_string(string): + """Compute MD5 digest over some string payload""" + return base64.b64encode(hashlib.md5(string).digest()) + +def compute_md5_for_file(file, block_size=2**14): + """Compute MD5 digest for a file + + Optional block_size parameter controls memory used to do MD5 calculation. + This should be a multiple of 128 bytes. + """ + f = open(file, 'r') + md5 = hashlib.md5() + while True: + data = f.read(block_size) + if not data: + break + md5.update(data) + return base64.b64encode(md5.digest()) diff --git a/resync/w3c_datetime.py b/resync/w3c_datetime.py new file mode 100644 index 0000000..e74384f --- /dev/null +++ b/resync/w3c_datetime.py @@ -0,0 +1,94 @@ +"""Write and parse W3C datetime + +Each web resource is identified by a URI and may optionally have +other metadata such as timestamp, size, md5. The lastmod property +provides ISO8601 format string access to the timestamp. + +The timestamp is assumed to be stored in UTC. +""" + +import time +from calendar import timegm +from datetime import datetime +from dateutil import parser as dateutil_parser +import re + +def datetime_to_str(dt): + """The Last-Modified data in ISO8601 syntax, Z notation + + The lastmod is stored as unix timestamp which is already + in UTC. At preesent this code will return 6 decimal digits + if any fraction of a second is given. It would perhaps be + better to return only the number of decimal digits necessary, + up to a resultion of 1 microsecond.""" + if (dt is None): + return None + return datetime.utcfromtimestamp(dt).isoformat() + 'Z' + +def str_to_datetime(s): + """Set timestamp from an W3C Datetime Last-Modified value + + The sitemaps.org specification says that values + must comply with the W3C Datetime format + (http://www.w3.org/TR/NOTE-datetime). This is a restricted + subset of ISO8601. In particular, all forms that include a + time must include a timezone indication so there is no + notion of local time (which would be tricky on the web). The + forms allowed are: + + Year: + YYYY (eg 1997) + Year and month: + YYYY-MM (eg 1997-07) + Complete date: + YYYY-MM-DD (eg 1997-07-16) + Complete date plus hours and minutes: + YYYY-MM-DDThh:mmTZD (eg 1997-07-16T19:20+01:00) + Complete date plus hours, minutes and seconds: + YYYY-MM-DDThh:mm:ssTZD (eg 1997-07-16T19:20:30+01:00) + Complete date plus hours, minutes, seconds and a decimal fraction + of a second + YYYY-MM-DDThh:mm:ss.sTZD (eg 1997-07-16T19:20:30.45+01:00) + where: + TZD = time zone designator (Z or +hh:mm or -hh:mm) + + We do not anticipate the YYYY and YYYY-MM forms being used but + interpret them as YYYY-01-01 and YYYY-MM-01 respectively. All + dates are interpreted as having time 00:00:00.0 UTC. + + Datetimes not specified to the level of seconds are intepreted + as 00.0 seconds. + """ + t = None + if (s is None): + return(t) + if (s == ''): + raise ValueError('Attempt to set empty datetime') + # Make a date into a full datetime + m = re.match(r"\d\d\d\d(\-\d\d(\-\d\d)?)?$",s) + if (m is not None): + if (m.group(1) is None): + s += '-01-01' + elif (m.group(2) is None): + s += '-01' + s += 'T00:00:00Z' + # Now have datetime with timezone info + m = re.match(r"(.*\d{2}:\d{2}:\d{2})(\.\d+)([^\d].*)?$",s) + # Chop out fractional seconds + fractional_seconds = 0 + if (m is not None): + s = m.group(1) + if (m.group(3) is not None): + s += m.group(3) + fractional_seconds = float(m.group(2)) + # Now check that only allowed formats supplied (the parse + # function is rather lax) + m = re.match(r"\d\d\d\d\-\d\d\-\d\dT\d\d:\d\d(:\d\d)?(Z|[+-]\d\d:\d\d)$",s) + if (m is None): + raise ValueError("Bad datetime format (%s)" % s) + dt = dateutil_parser.parse(s) + # timetuple ignores timezone information + #offset_seconds = dt.tzinfo.utcoffset(0).total_seconds() #only >=2.7 + offset = dt.tzinfo.utcoffset(0) + offset_seconds = (offset.seconds + offset.days * 24 * 3600) + return( timegm(dt.timetuple()) + offset_seconds + fractional_seconds )