diff --git a/downloader.py b/downloader.py index d440392..06afd13 100644 --- a/downloader.py +++ b/downloader.py @@ -94,6 +94,9 @@ def main(argv, parser.add_option("-l", "--list", action="store_true", dest="list", help="Get list of valid story URLs from page given.",) + parser.add_option("-n", "--normalize-list", + action="store_true", dest="normalize",default=False, + help="Get list of valid story URLs from page given, but normalized to standard forms.",) parser.add_option("-d", "--debug", action="store_true", dest="debug", help="Show debug output while downloading.",) @@ -169,8 +172,8 @@ def main(argv, (var,val) = opt.split('=') configuration.set("overrides",var,val) - if options.list: - retlist = get_urls_from_page(args[0], configuration) + if options.list or options.normalize: + retlist = get_urls_from_page(args[0], configuration, normalize=options.normalize) print "\n".join(retlist) return diff --git a/fanficdownloader/geturls.py b/fanficdownloader/geturls.py index 60cd132..25b1c54 100644 --- a/fanficdownloader/geturls.py +++ b/fanficdownloader/geturls.py @@ -25,7 +25,7 @@ from gziphttp import GZipProcessor import adapters from configurable import Configuration -def get_urls_from_page(url,configuration=None): +def get_urls_from_page(url,configuration=None,normalize=False): if not configuration: configuration = Configuration("test1.com","EPUB") @@ -54,11 +54,11 @@ def get_urls_from_page(url,configuration=None): opener = u2.build_opener(u2.HTTPCookieProcessor(),GZipProcessor()) data = opener.open(url).read() - return get_urls_from_html(data,url) + return get_urls_from_html(data,url,configuration,normalize) -def get_urls_from_html(data,url=None,configuration=None): +def get_urls_from_html(data,url=None,configuration=None,normalize=False): - normalized = set() # normalized url + normalized = [] # normalized url retlist = [] # orig urls. if not configuration: @@ -82,16 +82,19 @@ def get_urls_from_html(data,url=None,configuration=None): href = href.replace('&index=1','') adapter = adapters.getAdapter(configuration,href) if adapter.story.getMetadata('storyUrl') not in normalized: - normalized.add(adapter.story.getMetadata('storyUrl')) + normalized.append(adapter.story.getMetadata('storyUrl')) retlist.append(href) except: pass - return retlist + if normalize: + return normalized + else: + return retlist -def get_urls_from_text(data,configuration=None): +def get_urls_from_text(data,configuration=None,normalize=False): - normalized = set() # normalized url + normalized = [] # normalized url retlist = [] # orig urls. if not configuration: @@ -109,12 +112,15 @@ def get_urls_from_text(data,configuration=None): href = href.replace('&index=1','') adapter = adapters.getAdapter(configuration,href) if adapter.story.getMetadata('storyUrl') not in normalized: - normalized.add(adapter.story.getMetadata('storyUrl')) + normalized.append(adapter.story.getMetadata('storyUrl')) retlist.append(href) except: pass - return retlist + if normalize: + return normalized + else: + return retlist def form_url(parenturl,url): url = url.strip() # ran across an image with a space in the