diff --git a/bin/resync-explorer b/bin/resync-explorer index eba4684..2ca0801 100755 --- a/bin/resync-explorer +++ b/bin/resync-explorer @@ -90,6 +90,7 @@ def main(): init_logging( to_file=args.logger, logfile=args.logfile, default_logfile=DEFAULT_LOGFILE, verbose=args.verbose ) + print "----- ResourceSync Explorer -----" c = Explorer( checksum=args.checksum, verbose=args.verbose ) @@ -99,8 +100,6 @@ def main(): c.set_mappings(map) if (args.sitemap): c.sitemap_name=args.sitemap - if (args.warc): - c.dump_format='warc' if (args.exclude): c.exclude_patterns=args.exclude if (args.multifile): diff --git a/resync/explorer.py b/resync/explorer.py index 2d60fdc..0b62496 100644 --- a/resync/explorer.py +++ b/resync/explorer.py @@ -1,4 +1,9 @@ -"""ResourceSync explorer""" +"""ResourceSync explorer + +This is the guts of a client designed to 'explore' the ResourceSync +facilities offered by a source. Will use standard practices to +look for and interpret capabilities. +""" import sys import urllib @@ -11,89 +16,157 @@ import time import logging import requests -from resync.resource_list_builder import ResourceListBuilder -from resync.resource_list import ResourceList -from resync.change_list import ChangeList -from resync.capability_list import CapabilityList -from resync.capability_list_index import CapabilityListIndex from resync.mapper import Mapper from resync.sitemap import Sitemap -from resync.dump import Dump -from resync.resource import Resource -from resync.url_authority import UrlAuthority -from resync.utils import compute_md5_for_file from resync.client import Client,ClientFatalError from resync.client_state import ClientState from w3c_datetime import str_to_datetime,datetime_to_str +class XResource(object): + """Information about a resource for the explorer + + Must have a uri but may also store: + + acceptable_capabilities - None for any acceptable, 'resource' if a + resource rather than a capability document is expected, else + a list of capability names + checks - a set of information to check when then XResource is inspected + base_uri - on creation interpret any relative URI specified in the + context of base_uri, store the resulting full URI + """ + def __init__(self, uri, acceptable_capabilities=None, checks=None, context=None): + self.uri=urlparse.urljoin(context,uri) + self.acceptable_capabilities=acceptable_capabilities + self.checks=checks + +class HeadResponse(object): + """Object to mock up requests.head(...) response""" + def __init__(self): + self.status_code=None + self.headers={} + +class ExplorerQuit(Exception): + """Exception raised when user quits normally, no error""" + pass + class Explorer(Client): """Extension of the client code to explore a ResourceSync source + + Designed to support a text-based command-line client that starts from + a given URI, server, or local file, and then allows interactive + exploration of ResourceSync capabilities offered by the source. """ + def explore(self): - """Explore capabilities of a server interactvely + """INTERACTIVE exploration source capabilities Will use sitemap URI taken either from explicit self.sitemap_name or derived from the mappings supplied. """ - uri = None + # Where do we start? Build options in starts which has entries + # that are a pair comprised of the uri and a list of acceptable + # capabilities + starts = [] if (self.sitemap_name is not None): - uri = self.sitemap - print "Taking location from --sitemap option" - acceptable_capabilities = None #ie. any + print "Starting from explicit --sitemap %s" % (uri) + starts.append( XResource(uri) ) elif (len(self.mapper)>0): - pu = urlparse.urlparse(self.mapper.default_src_uri()) - uri = urlparse.urlunparse( [ pu[0], pu[1], '/.well-known/resourcesync', '', '', '' ] ) - print "Will look for discovery information based on mappings" - acceptable_capabilities = [ 'capabilitylist', 'capabilitylistindex' ] + uri = self.mapper.default_src_uri() + (scheme, netloc, path, params, query, fragment) = urlparse.urlparse(uri) + if (not scheme and not netloc): + if (os.path.isdir(path)): + # have a dir, look for 'likely' file names + print "Looking for capability documents in local directory %s" % (path) + for name in ['resourcesync','capabilities.xml', + 'resourcelist.xml','changelist.xml']: + file = os.path.join(path,name) + if (os.path.isfile(file)): + starts.append( XResource(file) ) + if (len(starts)==0): + raise ClientFatalError( "No likely capability files found in local directory %s" % + (path) ) + else: + # local file, might be anything (or not exist) + print "Starting from local file %s" % (path) + starts.append( XResource(path) ) + else: + # remote, can't tell whether we have a sitemap or a server name or something + # else, build list of options depending on whether there is a path and whether + # there is an extension/name + well_known = urlparse.urlunparse( [ scheme,netloc,'/.well-known/resourcesync','','','' ] ) + if (not path): + # root, just look for .well-known + starts.append( XResource(well_known, ['capabilitylist','capabilitylistindex']) ) + else: + starts.append( XResource(uri) ) + starts.append( XResource(well_known, ['capabilitylist','capabilitylistindex']) ) + print "Looking for discovery information based on mappings" else: - raise ClientFatalError("Neither explicit sitemap nor mapping specified") - history = [] - inp = None - checks = None - while (inp!='q'): - print - if (inp=='b'): - if (len(history)<2): - break #can't do this, exit - history.pop() #throw away current - uri=history.pop() - acceptable_capabilities=None - history.append(uri) - (uri,checks,acceptable_capabilities,inp) = self.explore_uri(uri,checks,acceptable_capabilities,len(history)>1) - print "resync-explorer done, bye..." + raise ClientFatalError("No source information (server base uri or capability uri) specified, use -h for help") + # + # Have list of one or more possible starting point, try them in turn + try: + for start in starts: + # For each starting point we create a fresh history + history = [ start ] + input = None + while (len(history)>0): + print + xr = history.pop() + new_xr = self.explore_uri(xr,len(history)>0) + if (new_xr): + # Add current and new to history + history.append( xr ) + history.append( new_xr ) + except ExplorerQuit: + pass # expected way to exit + print "\nresync-explorer done, bye...\n" - def explore_uri(self, uri, checks, caps, show_back=True): - """Interactive exploration of document at uri + def explore_uri(self, explorer_resource, show_back=True): + """INTERACTIVE exploration of capabilities document(s) starting at a given URI Will flag warnings if the document is not of type listed in caps """ - s=Sitemap() + uri = explorer_resource.uri + caps = explorer_resource.acceptable_capabilities + checks = explorer_resource.checks print "Reading %s" % (uri) options={} capability=None try: if (caps=='resource'): + # Not expecting a capability document self.explore_show_head(uri,check_headers=checks) else: + s=Sitemap() list = s.parse_xml(urllib.urlopen(uri)) - (options,capability)=self.explore_show_summary(list,s.parsed_index,caps) + (options,capability)=self.explore_show_summary(list,s.parsed_index,caps,context=uri) except IOError as e: print "Cannot read %s (%s)\nGoing back" % (uri,str(e)) - return('','','','b') + return(None) except Exception as e: print "Cannot parse %s (%s)\nGoing back" % (uri,str(e)) - return('','','','b') + return(None) + # + # Loop until we have some valide input + # while (True): # don't offer number option for no resources/capabilities num_prompt = '' if (len(options)==0) else 'number, ' up_prompt = 'b(ack), ' if (show_back) else '' - inp = raw_input( "Follow [%s%sq(uit)]?" % (num_prompt,up_prompt) ) - if (inp in options.keys()): + input = raw_input( "Follow [%s%sq(uit)]?" % (num_prompt,up_prompt) ) + if (input in options.keys()): break - if (inp == 'q' or inp == 'b'): - return('','','',inp) + if (input == 'q'): + raise ExplorerQuit() + if (input == 'b'): + return(None) + # + # Got input that is one of the options + # checks = {} - if ( options[inp].capability is None ): + r = options[input] + if ( r.capability is None ): if (capability == 'capabilitylistindex'): # all links should be to capabilitylist documents caps = ['capabilitylist'] @@ -101,30 +174,39 @@ class Explorer(Client): 'resourcedump','changedump']): caps = 'resource' else: - r = options[inp] caps = [r.capability] - if (r.length is not None): - checks['content-length']=r.length - if (r.lastmod is not None): - checks['last-modified']=r.lastmod - # FIXME - could do sanity check here and issue warnings if odd - return( options[inp].uri, checks, caps, inp ) + # Record anything we know about the resource to check + if (r.length is not None): + checks['content-length']=r.length + if (r.lastmod is not None): + checks['last-modified']=r.lastmod + # FIXME - could do sanity check here and issue warnings if odd + return( XResource(options[input].uri, caps, checks) ) - def explore_show_summary(self,list,parsed_index,caps): + def explore_show_summary(self, list, index=False, expected=None, context=None): """Show summary of one capability document - Used as part of --explore. + Given a capability document or index (in list, index True if it is an + index), write out a simply textual summary of the document with all + related documents shown as numbered options (of the form + [#] ...description...) which will then form a menu for the next + exploration. + + If expected is not None then it should be a list of expected document + types. If this is set then a warning will be printed if list is not + one of these document types. + FIXME - should look for link and show that """ num_entries = len(list.resources) capability = '(unknown capability)' if ('capability' in list.md): capability = list.md['capability'] - if (parsed_index): + if (index): capability += 'index' print "Parsed %s document with %d entries:" % (capability,num_entries) - if (caps is not None and capability not in caps): - print "WARNING - expected a %s document" % (','.join(caps)) + if (expected is not None and capability not in expected): + print "WARNING - expected a %s document" % (','.join(expected)) to_show = num_entries if (num_entries>21): to_show = 20 @@ -139,9 +221,17 @@ class Explorer(Client): entry_caps = ['changelist'] options = {} n=0 + # Look for elements in this document + if ('resourcesync' in list.ln): + options['rs']=list.ln['resourcesync'] + print "[%s] %s" % ('rs',list.ln['resourcesync'].uri) if ('up' in list.ln): options['up']=list.ln['up'] print "[%s] %s" % ('up',list.ln['up'].uri) + if ('db' in list.ln): + options['db']=list.ln['describedby'] + print "[%s] %s" % ('db',list.ln['describedby'].uri) + # Show listed resources as numbered options for r in list.resources: if (n>=to_show): print "(not showing remaining %d entries)" % (num_entries-n) @@ -149,6 +239,12 @@ class Explorer(Client): n+=1 options[str(n)]=r print "[%d] %s" % (n,r.uri) + if (self.verbose): + print " " + str(r) + full_uri = urlparse.urljoin(context,r.uri) + if (full_uri != r.uri): + print " WARNING - expanded relative URI to %s" % (full_uri) + r.uri = full_uri if (r.capability is not None): warning = '' if (r.capability not in entry_caps): @@ -166,19 +262,37 @@ class Explorer(Client): check_headers. """ print "HEAD %s" % (uri) - response = requests.head(uri) + if (re.match(r'^\w+:', uri)): + # Looks like a URI + response = requests.head(uri) + else: + # Mock up response if we have a local file + response = self.head_on_file(uri) print " status: %s" % (response.status_code) - # generate normalized lastmod + if (response.status_code=='200'): + # generate normalized lastmod # if ('last-modified' in response.headers): # response.headers.add['lastmod'] = datetime_to_str(str_to_datetime(response.headers['last-modified'])) # print some of the headers - for header in ['content-length','last-modified','lastmod','content-type','etag']: - if header in response.headers: - check_str='' - if (check_headers is not None and - header in check_headers): - if (response.headers[header] == check_headers[header]): - check_str=' MATCHES EXPECTED VALUE' - else: - check_STR=' EXPECTED %s' % (check_headers[header]) - print " %s: %s%s" % (header, response.headers[header], check_str) + for header in ['content-length','last-modified','lastmod','content-type','etag']: + if header in response.headers: + check_str='' + if (check_headers is not None and + header in check_headers): + if (response.headers[header] == check_headers[header]): + check_str=' MATCHES EXPECTED VALUE' + else: + check_str=' EXPECTED %s' % (check_headers[header]) + print " %s: %s%s" % (header, response.headers[header], check_str) + + def head_on_file(self,file): + """Mock up requests.head(..) response on local file + """ + response = HeadResponse() + if (not os.path.isfile(file)): + response.status_code='404' + else: + response.status_code='200' + response.headers['last-modified']=datetime_to_str(os.path.getmtime(file)) + response.headers['content-length']=os.path.getsize(file) + return(response) diff --git a/resync/test/test_explorer.py b/resync/test/test_explorer.py new file mode 100644 index 0000000..dc9d9ad --- /dev/null +++ b/resync/test/test_explorer.py @@ -0,0 +1,51 @@ +import unittest +import re +import logging +import sys, StringIO, contextlib + +from resync.client import Client, ClientFatalError +from resync.capability_list import CapabilityList +from resync.explorer import Explorer +from resync.resource import Resource + +# From http://stackoverflow.com/questions/2654834/capturing-stdout-within-the-same-process-in-python +class Data(object): + pass + +@contextlib.contextmanager +def capture_stdout(): + old = sys.stdout + capturer = StringIO.StringIO() + sys.stdout = capturer + data = Data() + yield data + sys.stdout = old + data.result = capturer.getvalue() + + +class TestExplorer(unittest.TestCase): + + @classmethod + def setUpClass(cls): + logging.basicConfig(level=logging.INFO) + + def test01_create(self): + # dumb test that we can create Exporer object + e = Explorer() + self.assertTrue( e ) + + def test02_explore_show_summary(self): + # Create dummy capabilities object and display + cl = CapabilityList() + cl.add( Resource('uri:resourcelist') ) + cl.add( Resource('uri:changelist') ) + e = Explorer() + with capture_stdout() as capturer: + e.explore_show_summary(cl,False,[]) + self.assertTrue( re.search(r'Parsed \(unknown capability\) document with 2 entries:',capturer.result) ) + self.assertTrue( re.search(r'\[1\] uri:changelist',capturer.result) ) + self.assertTrue( re.search(r'\[2\] uri:resourcelist',capturer.result) ) + +if __name__ == '__main__': + suite = unittest.TestLoader().loadTestsFromTestCase(TestExplorer) + unittest.TextTestRunner(verbosity=2).run(suite) diff --git a/resync/test/testdata/explore1/caps1.xml b/resync/test/testdata/explore1/caps1.xml index 14701af..45e1c87 100644 --- a/resync/test/testdata/explore1/caps1.xml +++ b/resync/test/testdata/explore1/caps1.xml @@ -2,23 +2,23 @@ + href="dataset1/info_about_source.html"/> - http://example.com/dataset1/resourcelist.xml + dataset1/resourcelist.xml - http://example.com/dataset1/resourcedump.xml + dataset1/resourcedump.xml - http://example.com/dataset1/changelist.xml + dataset1/changelist.xml - http://example.com/dataset1/changedump.xml + dataset1/changedump.xml - \ No newline at end of file + diff --git a/resync/test/testdata/explore1/caps2.xml b/resync/test/testdata/explore1/caps2.xml index 14701af..57617ac 100644 --- a/resync/test/testdata/explore1/caps2.xml +++ b/resync/test/testdata/explore1/caps2.xml @@ -2,23 +2,23 @@ + href="http://example.com/dataset2/info_about_source.xml"/> - http://example.com/dataset1/resourcelist.xml + http://example.com/dataset2/resourcelist.xml - http://example.com/dataset1/resourcedump.xml + http://example.com/dataset2/resourcedump.xml - http://example.com/dataset1/changelist.xml + http://example.com/dataset2/changelist.xml - http://example.com/dataset1/changedump.xml + http://example.com/dataset2/changedump.xml - \ No newline at end of file + diff --git a/resync/test/testdata/explore1/caps3.xml b/resync/test/testdata/explore1/caps3.xml index 14701af..901fbe9 100644 --- a/resync/test/testdata/explore1/caps3.xml +++ b/resync/test/testdata/explore1/caps3.xml @@ -2,23 +2,23 @@ + href="dataset3/info_about_source.xml"/> - http://example.com/dataset1/resourcelist.xml + dataset3/resourcelist.xml - http://example.com/dataset1/resourcedump.xml + dataset3/resourcedump.xml - http://example.com/dataset1/changelist.xml + dataset3/changelist.xml - http://example.com/dataset1/changedump.xml + dataset3/changedump.xml - \ No newline at end of file + diff --git a/resync/test/testdata/explore1/dataset1/info_about_source.html b/resync/test/testdata/explore1/dataset1/info_about_source.html new file mode 100644 index 0000000..ccbcac6 --- /dev/null +++ b/resync/test/testdata/explore1/dataset1/info_about_source.html @@ -0,0 +1,8 @@ + + + Info for dataset 1 + + +

This is dataset 1

+ + diff --git a/resync/test/testdata/explore1/dataset1/resource1 b/resync/test/testdata/explore1/dataset1/resource1 new file mode 100644 index 0000000..d0c0eae --- /dev/null +++ b/resync/test/testdata/explore1/dataset1/resource1 @@ -0,0 +1 @@ +I am resource 1 diff --git a/resync/test/testdata/explore1/dataset1/resource2 b/resync/test/testdata/explore1/dataset1/resource2 new file mode 100644 index 0000000..ad9a323 --- /dev/null +++ b/resync/test/testdata/explore1/dataset1/resource2 @@ -0,0 +1 @@ +And I am resource 2 diff --git a/resync/test/testdata/explore1/dataset1/resourcelist.xml b/resync/test/testdata/explore1/dataset1/resourcelist.xml new file mode 100644 index 0000000..a8e0b83 --- /dev/null +++ b/resync/test/testdata/explore1/dataset1/resourcelist.xml @@ -0,0 +1,19 @@ + + + + + + resource1 + 2013-05-28T17:23:34Z + + + + resource2 + + + resource3 + + diff --git a/resync/test/testdata/explore1/capsindex.xml b/resync/test/testdata/explore1/resourcesync similarity index 74% rename from resync/test/testdata/explore1/capsindex.xml rename to resync/test/testdata/explore1/resourcesync index 440e123..61422c2 100644 --- a/resync/test/testdata/explore1/capsindex.xml +++ b/resync/test/testdata/explore1/resourcesync @@ -6,14 +6,14 @@ - resync/test/testdata/explore1/caps1.xml + caps1.xml - resync/test/testdata/explore1/caps2.xml + caps2.xml - resync/test/testdata/explore1/caps3.xml + caps3.xml