diff --git a/app.yaml b/app.yaml index 852b064..a21f6c6 100644 --- a/app.yaml +++ b/app.yaml @@ -1,6 +1,6 @@ # ffd-retief-hrd fanfictiondownloader application: fanfictiondownloader -version: 4-1-1 +version: 4-2-0 runtime: python27 api_version: 1 threadsafe: true diff --git a/calibre-plugin/__init__.py b/calibre-plugin/__init__.py new file mode 100644 index 0000000..a25fee9 --- /dev/null +++ b/calibre-plugin/__init__.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python +# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai +# -*- coding: utf-8 -*- +from __future__ import (unicode_literals, division, absolute_import, + print_function) + +__license__ = 'GPL v3' +__copyright__ = '2011, Jim Miller' +__docformat__ = 'restructuredtext en' + +# The class that all Interface Action plugin wrappers must inherit from +from calibre.customize import InterfaceActionBase + +class InterfacePluginDemo(InterfaceActionBase): + ''' + This class is a simple wrapper that provides information about the + actual plugin class. The actual interface plugin class is called + InterfacePlugin and is defined in the ffdl_plugin.py file, as + specified in the actual_plugin field below. + + The reason for having two classes is that it allows the command line + calibre utilities to run without needing to load the GUI libraries. + ''' + name = 'FanFictionDownLoader' + description = 'UI plugin to download FanFiction stories from various sites.' + supported_platforms = ['windows', 'osx', 'linux'] + author = 'Jim Miller' + version = (1, 0, 4) + minimum_calibre_version = (0, 8, 30) + +# action_menu_clone_qaction = True + + #: This field defines the GUI plugin class that contains all the code + #: that actually does something. Its format is module_path:class_name + #: The specified class must be defined in the specified module. + actual_plugin = 'calibre_plugins.fanfictiondownloader_plugin.ffdl_plugin:FanFictionDownLoaderPlugin' + + def is_customizable(self): + ''' + This method must return True to enable customization via + Preferences->Plugins + ''' + return True + + def config_widget(self): + ''' + Implement this method and :meth:`save_settings` in your plugin to + use a custom configuration dialog. + + This method, if implemented, must return a QWidget. The widget can have + an optional method validate() that takes no arguments and is called + immediately after the user clicks OK. Changes are applied if and only + if the method returns True. + + If for some reason you cannot perform the configuration at this time, + return a tuple of two strings (message, details), these will be + displayed as a warning dialog to the user and the process will be + aborted. + + The base class implementation of this method raises NotImplementedError + so by default no user configuration is possible. + ''' + # It is important to put this import statement here rather than at the + # top of the module as importing the config class will also cause the + # GUI libraries to be loaded, which we do not want when using calibre + # from the command line + from calibre_plugins.fanfictiondownloader_plugin.config import ConfigWidget + return ConfigWidget() + + def save_settings(self, config_widget): + ''' + Save the settings specified by the user with config_widget. + + :param config_widget: The widget returned by :meth:`config_widget`. + ''' + config_widget.save_settings() + + # Apply the changes + ac = self.actual_plugin_ + if ac is not None: + ac.apply_settings() + +# For testing, run from command line with this: +# calibre-debug -e __init__.py +# +if __name__ == '__main__': + from PyQt4.Qt import QApplication + from calibre.gui2.preferences import test_widget + app = QApplication([]) + test_widget('Advanced', 'Plugins') diff --git a/calibre-plugin/about.txt b/calibre-plugin/about.txt new file mode 100644 index 0000000..1d11808 --- /dev/null +++ b/calibre-plugin/about.txt @@ -0,0 +1,11 @@ +
FanFictionDownLoader Plugin
+http://code.google.com/p/fanficdownloader/
+ +Created by Jim Miller, borrowing heavily from +Kovid Goyal's 'The InterfacePlugin Demo' and +Grant Drake's 'Count Pages' plugins.
+ +Requires calibre >= 0.8.30
+ diff --git a/calibre-plugin/config.py b/calibre-plugin/config.py new file mode 100644 index 0000000..8fdb9bd --- /dev/null +++ b/calibre-plugin/config.py @@ -0,0 +1,172 @@ +#!/usr/bin/env python +# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai +from __future__ import (unicode_literals, division, absolute_import, + print_function) + +__license__ = 'GPL v3' +__copyright__ = '2011, Jim Miller' +__docformat__ = 'restructuredtext en' + +from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QLabel, QLineEdit, + QTextEdit, QComboBox, QCheckBox, QPushButton) + +from calibre.utils.config import JSONConfig + +from calibre_plugins.fanfictiondownloader_plugin.dialogs import (OVERWRITE, ADDNEW, SKIP,CALIBREONLY,UPDATE) + +# This is where all preferences for this plugin will be stored +# Remember that this name (i.e. plugins/fanfictiondownloader_plugin) is also +# in a global namespace, so make it as unique as possible. +# You should always prefix your config file name with plugins/, +# so as to ensure you dont accidentally clobber a calibre config file +prefs = JSONConfig('plugins/fanfictiondownloader_plugin') + +# urlsfrompriority values +CLIP = 'Clipboard' +SELECTED = 'Selected Stories' + +# Set defaults +prefs.defaults['personal.ini'] = get_resources('example.ini') +prefs.defaults['updatemeta'] = True +prefs.defaults['onlyoverwriteifnewer'] = False +prefs.defaults['urlsfromclip'] = True +prefs.defaults['urlsfromselected'] = True +prefs.defaults['urlsfrompriority'] = SELECTED +prefs.defaults['fileform'] = 'epub' +prefs.defaults['collision'] = OVERWRITE +prefs.defaults['deleteotherforms'] = False + +class ConfigWidget(QWidget): + + def __init__(self): + QWidget.__init__(self) + self.l = QVBoxLayout() + self.setLayout(self.l) + + horz = QHBoxLayout() + label = QLabel('Default Output &Format:') + horz.addWidget(label) + self.fileform = QComboBox(self) + self.fileform.addItem('epub') + self.fileform.addItem('mobi') + self.fileform.addItem('html') + self.fileform.addItem('txt') + self.fileform.setCurrentIndex(self.fileform.findText(prefs['fileform'])) + self.fileform.setToolTip('Choose output format to create. May set default from plugin configuration.') + label.setBuddy(self.fileform) + horz.addWidget(self.fileform) + self.l.addLayout(horz) + + horz = QHBoxLayout() + label = QLabel('Default If Story Already Exists?') + label.setToolTip("What to do if there's already an existing story with the same title and author.") + horz.addWidget(label) + self.collision = QComboBox(self) + self.collision.addItem(OVERWRITE) + self.collision.addItem(UPDATE) + self.collision.addItem(ADDNEW) + self.collision.addItem(SKIP) + self.collision.addItem(CALIBREONLY) + self.collision.setCurrentIndex(self.collision.findText(prefs['collision'])) + self.collision.setToolTip('Overwrite will replace the existing story. Add New will create a new story with the same title and author.') + label.setBuddy(self.collision) + horz.addWidget(self.collision) + self.l.addLayout(horz) + + self.updatemeta = QCheckBox('Default Update Calibre &Metadata?',self) + self.updatemeta.setToolTip('Update metadata for story in Calibre from web site?') + self.updatemeta.setChecked(prefs['updatemeta']) + self.l.addWidget(self.updatemeta) + + self.onlyoverwriteifnewer = QCheckBox('Default Only Overwrite Story if Newer',self) + self.onlyoverwriteifnewer.setToolTip("Don't overwrite existing book unless the story on the web site is newer or from the same day.") + self.onlyoverwriteifnewer.setChecked(prefs['onlyoverwriteifnewer']) + self.l.addWidget(self.onlyoverwriteifnewer) + + self.urlsfromclip = QCheckBox('Take URLs from Clipboard?',self) + self.urlsfromclip.setToolTip('Prefill URLs from valid URLs in Clipboard when starting plugin?') + self.urlsfromclip.setChecked(prefs['urlsfromclip']) + self.l.addWidget(self.urlsfromclip) + + self.urlsfromselected = QCheckBox('Take URLs from Selected Stories?',self) + self.urlsfromselected.setToolTip('Prefill URLs from valid URLs in selected stories when starting plugin?') + self.urlsfromselected.setChecked(prefs['urlsfromselected']) + self.l.addWidget(self.urlsfromselected) + + horz = QHBoxLayout() + label = QLabel('Take URLs from which first:') + label.setToolTip("If both clipbaord and selected enabled and populated, which is used?") + horz.addWidget(label) + self.urlsfrompriority = QComboBox(self) + self.urlsfrompriority.addItem(SELECTED) + self.urlsfrompriority.addItem(CLIP) + self.urlsfrompriority.setCurrentIndex(self.urlsfrompriority.findText(prefs['urlsfrompriority'])) + self.urlsfrompriority.setToolTip('If both clipbaord and selected enabled and populated, which is used?') + label.setBuddy(self.urlsfrompriority) + horz.addWidget(self.urlsfrompriority) + self.l.addLayout(horz) + + self.deleteotherforms = QCheckBox('Delete other existing formats?',self) + self.deleteotherforms.setToolTip('Check this to automatically delete all other ebook formats when updating an existing book.\nHandy if you have both a Nook(epub) and Kindle(mobi), for example.') + self.deleteotherforms.setChecked(prefs['deleteotherforms']) + self.l.addWidget(self.deleteotherforms) + + self.label = QLabel('personal.ini:') + self.l.addWidget(self.label) + + self.ini = QTextEdit(self) + self.ini.setLineWrapMode(QTextEdit.NoWrap) + self.ini.setText(prefs['personal.ini']) + self.l.addWidget(self.ini) + + self.defaults = QPushButton('View Defaults', self) + self.defaults.setToolTip("View all of the plugin's configurable settings\nand their default settings.") + self.defaults.clicked.connect(self.show_defaults) + self.l.addWidget(self.defaults) + + def save_settings(self): + prefs['fileform'] = unicode(self.fileform.currentText()) + prefs['collision'] = unicode(self.collision.currentText()) + prefs['updatemeta'] = self.updatemeta.isChecked() + prefs['urlsfrompriority'] = unicode(self.urlsfrompriority.currentText()) + prefs['urlsfromclip'] = self.urlsfromclip.isChecked() + prefs['urlsfromselected'] = self.urlsfromselected.isChecked() + prefs['onlyoverwriteifnewer'] = self.onlyoverwriteifnewer.isChecked() + prefs['deleteotherforms'] = self.deleteotherforms.isChecked() + + ini = unicode(self.ini.toPlainText()) + if ini: + prefs['personal.ini'] = ini + else: + # if they've removed everything, clear it so they get the + # default next time. + del prefs['personal.ini'] + + def show_defaults(self): + text = get_resources('defaults.ini') + ShowDefaultsIniDialog(self.windowIcon(),text,self).exec_() + +class ShowDefaultsIniDialog(QDialog): + + def __init__(self, icon, text, parent=None): + QDialog.__init__(self, parent) + self.resize(600, 500) + self.l = QVBoxLayout() + self.setLayout(self.l) + self.label = QLabel("Plugin Defaults (Read-Only)") + self.label.setToolTip("These all of the plugin's configurable settings\nand their default settings.") + self.setWindowTitle(_('Plugin Defaults')) + self.setWindowIcon(icon) + self.l.addWidget(self.label) + + self.ini = QTextEdit(self) + self.ini.setToolTip("These all of the plugin's configurable settings\nand their default settings.") + self.ini.setLineWrapMode(QTextEdit.NoWrap) + self.ini.setText(text) + self.ini.setReadOnly(True) + self.l.addWidget(self.ini) + + self.ok_button = QPushButton('OK', self) + self.ok_button.clicked.connect(self.hide) + self.l.addWidget(self.ok_button) + diff --git a/calibre-plugin/dialogs.py b/calibre-plugin/dialogs.py new file mode 100644 index 0000000..534ef4c --- /dev/null +++ b/calibre-plugin/dialogs.py @@ -0,0 +1,286 @@ +#!/usr/bin/env python +# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai +from __future__ import (unicode_literals, division, + print_function) + +__license__ = 'GPL v3' +__copyright__ = '2011, Jim Miller' +__docformat__ = 'restructuredtext en' + +import traceback + +from PyQt4.Qt import (QDialog, QMessageBox, QVBoxLayout, QHBoxLayout, QGridLayout, + QPushButton, QProgressDialog, QString, QLabel, QCheckBox, + QTextEdit, QLineEdit, QInputDialog, QComboBox, QClipboard, + QProgressDialog, QTimer, QDialogButtonBox, QPixmap, Qt ) + +from calibre.gui2 import error_dialog, warning_dialog, question_dialog, info_dialog + +from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters,writers,exceptions + +OVERWRITE='Overwrite' +UPDATE='Update EPUB' +ADDNEW='Add New' +SKIP='Skip' +CALIBREONLY='Update Calibre Metadata Only' + +class DownloadDialog(QDialog): + + def __init__(self, gui, prefs, icon, url_list_text, do_user_config, start_downloads): + QDialog.__init__(self, gui) + self.gui = gui + self.do_user_config = do_user_config + self.start_downloads = start_downloads + + self.setMinimumWidth(300) + self.l = QVBoxLayout() + self.setLayout(self.l) + + self.setWindowTitle('FanFictionDownLoader') + self.setWindowIcon(icon) + + self.l.addWidget(QLabel('Story URL(s), one per line:')) + self.url = QTextEdit(self) + self.url.setToolTip('URLs for stories, one per line.\nWill take URLs from selected books or clipboard, but only valid URLs.') + self.url.setLineWrapMode(QTextEdit.NoWrap) + self.url.setText(url_list_text) + self.l.addWidget(self.url) + + self.ffdl_button = QPushButton( + 'Download Stories', self) + self.ffdl_button.setToolTip('Start download(s).') + self.ffdl_button.clicked.connect(self.ffdl) + # if there's already URL(s), focus 'go' button + if url_list_text: + self.ffdl_button.setFocus() + self.l.addWidget(self.ffdl_button) + + horz = QHBoxLayout() + label = QLabel('Output &Format:') + horz.addWidget(label) + self.fileform = QComboBox(self) + self.fileform.addItem('epub') + self.fileform.addItem('mobi') + self.fileform.addItem('html') + self.fileform.addItem('txt') + self.fileform.setCurrentIndex(self.fileform.findText(prefs['fileform'])) + self.fileform.setToolTip('Choose output format to create. May set default from plugin configuration.') + label.setBuddy(self.fileform) + horz.addWidget(self.fileform) + self.l.addLayout(horz) + + horz = QHBoxLayout() + label = QLabel('If Story Already Exists?') + label.setToolTip("What to do if there's already an existing story with the same title and author.") + horz.addWidget(label) + self.collision = QComboBox(self) + self.collision.addItem(OVERWRITE) + self.collision.addItem(UPDATE) + self.collision.addItem(ADDNEW) + self.collision.addItem(SKIP) + self.collision.addItem(CALIBREONLY) + self.collision.setCurrentIndex(self.collision.findText(prefs['collision'])) + self.collision.setToolTip(OVERWRITE+' will replace the existing story.\n'+ + UPDATE+' will download new chapters only and add to existing EPUB.\n'+ + ADDNEW+' will create a new story with the same title and author.\n'+ + SKIP+' will not download existing stories.\n'+ + CALIBREONLY+' will not download stories, but will update Calibre metadata.') + label.setBuddy(self.collision) + horz.addWidget(self.collision) + self.l.addLayout(horz) + + self.updatemeta = QCheckBox('Update Calibre &Metadata?',self) + self.updatemeta.setToolTip('Update metadata for story in Calibre from web site?') + self.updatemeta.setChecked(prefs['updatemeta']) + self.l.addWidget(self.updatemeta) + + self.onlyoverwriteifnewer = QCheckBox('Only Overwrite Story if Newer',self) + self.onlyoverwriteifnewer.setToolTip("Don't overwrite existing book unless the story on the web site is newer.\n"+ + "From the same day counts as 'newer' because the sites don't give update time.") + self.onlyoverwriteifnewer.setChecked(prefs['onlyoverwriteifnewer']) + self.l.addWidget(self.onlyoverwriteifnewer) + + horz = QHBoxLayout() + self.about_button = QPushButton('About', self) + self.about_button.clicked.connect(self.about) + horz.addWidget(self.about_button) + self.conf_button = QPushButton( + 'Configure this plugin', self) + self.conf_button.clicked.connect(self.config) + horz.addWidget(self.conf_button) + self.l.addLayout(horz) + + self.resize(self.sizeHint()) + + def about(self): + # Get the about text from a file inside the plugin zip file + # The get_resources function is a builtin function defined for all your + # plugin code. It loads files from the plugin zip file. It returns + # the bytes from the specified file. + # + # Note that if you are loading more than one file, for performance, you + # should pass a list of names to get_resources. In this case, + # get_resources will return a dictionary mapping names to bytes. Names that + # are not found in the zip file will not be in the returned dictionary. + text = get_resources('about.txt') +# QMessageBox.about(self, 'About the FanFictionDownLoader Plugin', + # text.decode('utf-8')) + AboutDialog(self.windowIcon(),text,self).exec_() + + def ffdl(self): + self.start_downloads(unicode(self.url.toPlainText()), + unicode(self.fileform.currentText()), + unicode(self.collision.currentText()), + self.updatemeta.isChecked(), + self.onlyoverwriteifnewer.isChecked()) + self.hide() + + def config(self): + self.do_user_config(parent=self) + + +class UserPassDialog(QDialog): + ''' + Need to collect User/Pass for some sites. + ''' + def __init__(self, gui, site): + QDialog.__init__(self, gui) + self.gui = gui + self.status=False + self.setWindowTitle('User/Password') + + self.l = QGridLayout() + self.setLayout(self.l) + + self.l.addWidget(QLabel("%s requires you to login to download this story."%site),0,0,1,2) + + self.l.addWidget(QLabel("User:"),1,0) + self.user = QLineEdit(self) + self.l.addWidget(self.user,1,1) + + self.l.addWidget(QLabel("Password:"),2,0) + self.passwd = QLineEdit(self) + self.l.addWidget(self.passwd,2,1) + + self.ok_button = QPushButton('OK', self) + self.ok_button.clicked.connect(self.ok) + self.l.addWidget(self.ok_button,3,0) + + self.cancel_button = QPushButton('Cancel', self) + self.cancel_button.clicked.connect(self.cancel) + self.l.addWidget(self.cancel_button,3,1) + + self.resize(self.sizeHint()) + + def ok(self): + self.status=True + self.hide() + + def cancel(self): + self.status=False + self.hide() + +class MetadataProgressDialog(QProgressDialog): + ''' + ProgressDialog displayed while fetching metadata for each story. + ''' + def __init__(self, gui, loop_list, fileform, getadapter_function, download_list_function, db): + QProgressDialog.__init__(self, + "Fetching metadata for stories...", + QString(), 0, len(loop_list), gui) + self.setWindowTitle("Downloading metadata for stories") + self.setMinimumWidth(500) + self.gui = gui + self.db = db + self.loop_list = loop_list + self.fileform = fileform + self.getadapter_function = getadapter_function + self.download_list_function = download_list_function + self.i, self.loop_bad, self.loop_good = 0, [], [] + + ## self.do_loop does QTimer.singleShot on self.do_loop also. + ## A weird way to do a loop, but that was the example I had. + QTimer.singleShot(0, self.do_loop) + self.exec_() + + def updateStatus(self): + self.setLabelText("Fetched metadata for %d of %d"%(self.i+1,len(self.loop_list))) + self.setValue(self.i+1) + print(self.labelText()) + + def do_loop(self): + print("self.i:%d"%self.i) + + if self.i == 0: + self.setValue(0) + + if self.i >= len(self.loop_list) or self.wasCanceled(): + return self.do_when_finished() + + else: + current = self.loop_list[self.i] + try: + ## collision spec passed into getadapter by partial from ffdl_plugin + ## no retval only if it exists, but collision is SKIP + retval = self.getadapter_function(current,self.fileform) + self.loop_good.append((current,retval)) + except Exception as e: + print("%s:%s"%(current,e)) + self.loop_bad.append((current,e)) + traceback.print_exc() + + self.updateStatus() + self.i += 1 + QTimer.singleShot(0, self.do_loop) + + def do_when_finished(self): + self.hide() + + # Queues a job to process these ePub/Mobi books in the background. + self.download_list_function(self.loop_good,self.fileform) + + if self.loop_bad != []: + res = [] + for j in self.loop_bad: + res.append('%s : %s'%j) + msg = '%s' % '\n'.join(res) + warning_dialog(self.gui, _('Not going to download some stories'), + _('Not going to download %d of %d stories.') % + (len(self.loop_bad), len(self.loop_list)), + msg).exec_() + # else: + # info_dialog(self.gui, "Starting Downloads", + # "Got metadata and started download for %d stories."%len(self.loop_good), + # show_copy_button=False).exec_() + self.gui = None + +class AboutDialog(QDialog): + + def __init__(self, icon, text, parent=None): + QDialog.__init__(self, parent) + self.resize(400, 250) + self.l = QGridLayout() + self.setLayout(self.l) + self.logo = QLabel() + self.logo.setMaximumWidth(110) + self.logo.setPixmap(QPixmap(icon.pixmap(100,100))) + self.label = QLabel(text) + self.label.setOpenExternalLinks(True) + self.label.setWordWrap(True) + self.setWindowTitle(_('Update available!')) + self.setWindowIcon(icon) + self.l.addWidget(self.logo, 0, 0) + self.l.addWidget(self.label, 0, 1) + self.bb = QDialogButtonBox(self) + b = self.bb.addButton(_('OK'), self.bb.AcceptRole) + b.setDefault(True) + self.l.addWidget(self.bb, 2, 0, 1, -1) + self.bb.accepted.connect(self.accept) + + +class NotGoingToDownload(Exception): + def __init__(self,error): + self.error=error + + def __str__(self): + return self.error diff --git a/calibre-plugin/ffdl_plugin.py b/calibre-plugin/ffdl_plugin.py new file mode 100644 index 0000000..4adbea7 --- /dev/null +++ b/calibre-plugin/ffdl_plugin.py @@ -0,0 +1,464 @@ +#!/usr/bin/env python +# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai +from __future__ import (unicode_literals, division, absolute_import, + print_function) + +__license__ = 'GPL v3' +__copyright__ = '2011, Jim Miller' +__docformat__ = 'restructuredtext en' + +import ConfigParser, os +from StringIO import StringIO +from functools import partial +from datetime import datetime + +from PyQt4.Qt import (QApplication) + +# The class that all interface action plugins must inherit from +from calibre.ptempfile import PersistentTemporaryFile +from calibre.ebooks.metadata import MetaInformation +from calibre.ebooks.metadata.meta import get_metadata +from calibre.gui2 import error_dialog, warning_dialog, question_dialog, info_dialog +from calibre.gui2.actions import InterfaceAction +from calibre.gui2.threaded_jobs import ThreadedJob + +from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions +from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge + +from calibre_plugins.fanfictiondownloader_plugin.config import (prefs, CLIP, SELECTED) +from calibre_plugins.fanfictiondownloader_plugin.dialogs import ( + DownloadDialog, MetadataProgressDialog, UserPassDialog, + OVERWRITE, UPDATE, ADDNEW, SKIP, CALIBREONLY, NotGoingToDownload ) + +# because calibre immediately transforms html into zip and don't want +# to have an 'if html'. db.has_format is cool with the case mismatch, +# but if I'm doing it anyway... +formmapping = { + 'epub':'EPUB', + 'mobi':'MOBI', + 'html':'ZIP', + 'txt':'TXT' + } + +class FanFictionDownLoaderPlugin(InterfaceAction): + + name = 'FanFictionDownLoader' + + # Declare the main action associated with this plugin + # The keyboard shortcut can be None if you dont want to use a keyboard + # shortcut. Remember that currently calibre has no central management for + # keyboard shortcuts, so try to use an unusual/unused shortcut. + # (text, icon_path, tooltip, keyboard shortcut) + # icon_path isn't in the zip--icon loaded below. + action_spec = ('FanFictionDownLoader', None, + 'Download FanFiction stories from various web sites', None) + + action_type = 'global' + + def genesis(self): + # This method is called once per plugin, do initial setup here + + # Set the icon for this interface action + # The get_icons function is a builtin function defined for all your + # plugin code. It loads icons from the plugin zip file. It returns + # QIcon objects, if you want the actual data, use the analogous + # get_resources builtin function. + # + # Note that if you are loading more than one icon, for performance, you + # should pass a list of names to get_icons. In this case, get_icons + # will return a dictionary mapping names to QIcons. Names that + # are not found in the zip file will result in null QIcons. + icon = get_icons('images/icon.png') + + # The qaction is automatically created from the action_spec defined + # above + self.qaction.setIcon(icon) + # Call function when plugin triggered. + self.qaction.triggered.connect(self.show_dialog) + + def show_dialog(self): + # The base plugin object defined in __init__.py + base_plugin_object = self.interface_action_base_plugin + # Show the config dialog + # The config dialog can also be shown from within + # Preferences->Plugins, which is why the do_user_config + # method is defined on the base plugin class + do_user_config = base_plugin_object.do_user_config + + # The current database shown in the GUI + # db is an instance of the class LibraryDatabase2 from database.py + # This class has many, many methods that allow you to do a lot of + # things. + self.db = self.gui.current_db + + # pre-pop urls from selected stories or clipboard. but with + # configurable priority and on/off option on each + if prefs['urlsfrompriority'] == SELECTED: + from_first_func = self.get_urls_select + from_second_func = self.get_urls_clip + elif prefs['urlsfrompriority'] == CLIP: + from_first_func = self.get_urls_clip + from_second_func = self.get_urls_select + + url_list_text = from_first_func() + if not url_list_text: + url_list_text = from_second_func() + + # self.gui is the main calibre GUI. It acts as the gateway to access + # all the elements of the calibre user interface, it should also be the + # parent of the dialog + # DownloadDialog just collects URLs, format and presents buttons. + d = DownloadDialog(self.gui, + prefs, + self.qaction.icon(), + url_list_text, + do_user_config, # method for config button + self.start_downloads, # method to start downloads + ) + d.show() + + ## if there's rows selected, try to find a source URL from + ## either identifier in the metadata, or from the epub + ## metadata. + def get_urls_select(self): + url_list = [] + rows = self.gui.library_view.selectionModel().selectedRows() + if rows and prefs['urlsfromselected']: + book_ids = self.gui.library_view.get_selected_ids() + #print("book_ids: %s"%book_ids) + for book_id in book_ids: + #print("book_on_device:%s"%self.db.book_on_device(book_id)) + identifiers = self.db.get_identifiers(book_id,index_is_id=True) + if 'url' in identifiers: + # identifiers have :->| in url. + #print("url from book:"+identifiers['url'].replace('|',':')) + url_list.append(identifiers['url'].replace('|',':')) + else: + ## only epub has that in it. + if self.db.has_format(book_id,'EPUB',index_is_id=True): + existingepub = self.db.format(book_id,'EPUB',index_is_id=True, as_file=True) + mi = get_metadata(existingepub,'EPUB') + #print("mi:%s"%mi) + identifiers = mi.get_identifiers() + if 'url' in identifiers: + #print("url from epub:"+identifiers['url'].replace('|',':')) + url_list.append(identifiers['url'].replace('|',':')) + return self.get_valid_urls(url_list) + + def get_urls_clip(self): + url_list = [] + if prefs['urlsfromclip']: + # no rows selected, check for valid URLs in the clipboard. + cliptext = unicode(QApplication.instance().clipboard().text()) + url_list.extend(cliptext.split()) + return self.get_valid_urls(url_list) + + def get_valid_urls(self,url_list): + url_list_text = "" + + # Check and make sure the URLs are valid ffdl URLs. + if url_list: + # this is the accepted way to 'check for existance'? really? + try: + self.dummyconfig + except AttributeError: + self.dummyconfig = ConfigParser.SafeConfigParser() + + alreadyin=[] + for url in url_list: + if url in alreadyin: + continue + alreadyin.append(url) + # pulling up an adapter is pretty low over-head. If + # it fails, it's a bad url. + try: + adapters.getAdapter(self.dummyconfig,url) + except: + pass + else: + if url_list_text: + url_list_text += "\n" + url_list_text += url + + return url_list_text + + def apply_settings(self): + # No need to do anything with perfs here, but we could. + prefs + + def start_downloads(self,urls,fileform, + collision,updatemeta,onlyoverwriteifnewer): + + url_list = get_url_list(urls) + + self.fetchmeta_qpd = \ + MetadataProgressDialog(self.gui, + url_list, + fileform, + partial(self.get_adapter_for_story, collision=collision,onlyoverwriteifnewer=onlyoverwriteifnewer), + partial(self.download_list,collision=collision,updatemeta=updatemeta,onlyoverwriteifnewer=onlyoverwriteifnewer), + self.db) + + def get_adapter_for_story(self,url,fileform, + collision=SKIP, + onlyoverwriteifnewer=False): + ''' + Returns adapter object for story at URL. To be called from + MetadataProgressDialog 'loop' to build up list of adapters. Also + pops dialogs for is adult, user/pass, duplicate + ''' + + print("URL:"+url) + ## was self.ffdlconfig, but we need to be able to change it + ## when doing epub update. + ffdlconfig = ConfigParser.SafeConfigParser() + ffdlconfig.readfp(StringIO(get_resources("defaults.ini"))) + ffdlconfig.readfp(StringIO(prefs['personal.ini'])) + adapter = adapters.getAdapter(ffdlconfig,url) + + try: + adapter.getStoryMetadataOnly() + except exceptions.FailedToLogin: + print("Login Failed, Need Username/Password.") + userpass = UserPassDialog(self.gui,url) + userpass.exec_() # exec_ will make it act modal + if userpass.status: + adapter.username = userpass.user.text() + adapter.password = userpass.passwd.text() + # else: + # del adapter + # return + except exceptions.AdultCheckRequired: + if question_dialog(self.gui, 'Are You Adult?', ''+ + "%s requires that you be an adult. Please confirm you are an adult in your locale:"%url, + show_copy_button=False): + adapter.is_adult=True + # else: + # del adapter + # return + + # let exceptions percolate up. + story = adapter.getStoryMetadataOnly() + + if collision != ADDNEW: + mi = MetaInformation(story.getMetadata("title"), + (story.getMetadata("author"),)) # author is a list. + + identicalbooks = self.db.find_identical_books(mi) + print(identicalbooks) + ## more than one match will need to be handled differently. + if identicalbooks: + book_id = identicalbooks.pop() + if collision == SKIP: + raise NotGoingToDownload("Skipping duplicate story.") + + # print("Attempting to add to list.") + # rl_plugin = self.gui.iactions['Reading List'] + # rl_plugin.remove_books_from_list('Send',[book_id],refresh_screen=False) + # rl_plugin.add_books_to_list('Send',[book_id],refresh_screen=False) + #rl_plugin.view_list('Send') + + if collision == OVERWRITE and len(identicalbooks) > 1: + raise NotGoingToDownload("More than one identical books--can't tell which to overwrite.") + + if collision == OVERWRITE and \ + onlyoverwriteifnewer and \ + self.db.has_format(book_id,fileform,index_is_id=True): + # check make sure incoming is newer. + lastupdated=story.getMetadataRaw('dateUpdated').date() + fileupdated=datetime.fromtimestamp(os.stat(self.db.format_abspath(book_id, fileform, index_is_id=True))[8]).date() + if fileupdated > lastupdated: + raise NotGoingToDownload("Not Overwriting, story is not newer.") + + if collision == UPDATE: + if fileform != 'epub': + raise NotGoingToDownload("Not updating non-epub format.") + # 'book' can exist without epub. If there's no existing epub, + # let it go and it will download it. + if self.db.has_format(book_id,fileform,index_is_id=True): + toupdateio = StringIO() + (epuburl,chaptercount) = doMerge(toupdateio, + [StringIO(self.db.format(book_id,'EPUB', + index_is_id=True))], + titlenavpoints=False, + striptitletoc=True, + forceunique=False) + + urlchaptercount = int(story.getMetadata('numChapters')) + if chaptercount == urlchaptercount: # and not onlyoverwriteifnewer: + raise NotGoingToDownload("%s already contains %d chapters." % (url,chaptercount)) + elif chaptercount > urlchaptercount: + raise NotGoingToDownload("%s contains %d chapters, more than epub." % (url,chaptercount)) + else: + print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount)) + + else: # not identicalbooks + if collision == CALIBREONLY: + raise NotGoingToDownload("Not updating Calibre Metadata, no existing book to update.") + + return adapter + + def download_list(self,adaptertuple_list,fileform, + collision=ADDNEW, + updatemeta=True, + onlyoverwriteifnewer=True): + ''' + Called by MetadataProgressDialog to start story downloads BG processing. + adapter_list is a list of tuples of (url,adapter) + ''' + print("download_list") + + job = ThreadedJob('FanFictionDownload', + 'Downloading FanFiction Stories', + func=self.do_story_downloads, + args=(adaptertuple_list, fileform, self.db), + kwargs={'collision':collision,'updatemeta':updatemeta, + 'onlyoverwriteifnewer':onlyoverwriteifnewer}, + callback=self._get_stories_completed) + + + self.gui.job_manager.run_threaded_job(job) + + self.gui.status_bar.show_message('Downloading %d stories'%len(adaptertuple_list)) + + def _get_stories_completed(self, job): + print("_get_stories_completed") + + def do_story_downloads(self, adaptertuple_list, fileform, db, + **kwargs): + ''' + Master job, loop to download this list of stories + ''' + print("do_story_downloads") + abort = kwargs['abort'] + notifications=kwargs['notifications'] + log = kwargs['log'] + notifications.put((0.01, 'Start Downloading Stories')) + count = 0.01 + total = len(adaptertuple_list) + # Queue all the jobs + for (url,adapter) in adaptertuple_list: + if abort.is_set(): + notifications.put(1.0,'Aborting...') + return + notifications.put((float(count)/total, + 'Downloading %s'%adapter.getStoryMetadataOnly().getMetadata("title"))) + log.prints(log.INFO,'Downloading %s'%adapter.getStoryMetadataOnly().getMetadata("title")) + try: + self.do_story_download(adapter,fileform,db, + kwargs['collision'],kwargs['updatemeta'],kwargs['onlyoverwriteifnewer']) + except Exception as e: + log.prints(log.ERROR,'Failed Downloading %s: %s'% + (adapter.getStoryMetadataOnly().getMetadata("title"),e)) + + count = count + 1 + return + + def do_story_download(self,adapter,fileform,db,collision, + updatemeta,onlyoverwriteifnewer): + print("do_story_download") + + story = adapter.getStoryMetadataOnly() + + mi = MetaInformation(story.getMetadata("title"), + (story.getMetadata("author"),)) # author is a list. + + writer = writers.getWriter(fileform,adapter.config,adapter) + tmp = PersistentTemporaryFile("."+fileform) + titleauth = "%s by %s"%(story.getMetadata("title"), story.getMetadata("author")) + url = story.getMetadata("storyUrl") + print(titleauth) + print("tmp: "+tmp.name) + + mi.set_identifiers({'url':story.getMetadata("storyUrl")}) + mi.publisher = story.getMetadata("site") + mi.tags = writer.getTags() + mi.languages = ['en'] + mi.pubdate = story.getMetadataRaw('datePublished') + mi.timestamp = story.getMetadataRaw('dateCreated') + mi.comments = story.getMetadata("description") + + identicalbooks = self.db.find_identical_books(mi) + #print(identicalbooks) + addedcount=0 + if identicalbooks and collision != ADDNEW: + ## more than one match? add to first off the list. + ## Shouldn't happen--we checked above. + book_id = identicalbooks.pop() + + if collision == UPDATE: + if self.db.has_format(book_id,fileform,index_is_id=True): + urlchaptercount = int(story.getMetadata('numChapters')) + ## First, get existing epub with titlepage and tocpage stripped. + updateio = StringIO() + (epuburl,chaptercount) = doMerge(updateio, + [StringIO(self.db.format(book_id,'EPUB', + index_is_id=True))], + titlenavpoints=False, + striptitletoc=True, + forceunique=False) + print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount)) + + ## Get updated title page/metadata by itself in an epub. + ## Even if the title page isn't included, this carries the metadata. + titleio = StringIO() + writer.writeStory(outstream=titleio,metaonly=True) + + newchaptersio = None + if urlchaptercount > chaptercount : + ## Go get the new chapters only in another epub. + newchaptersio = StringIO() + adapter.setChaptersRange(chaptercount+1,urlchaptercount) + + adapter.config.set("overrides",'include_tocpage','false') + adapter.config.set("overrides",'include_titlepage','false') + writer.writeStory(outstream=newchaptersio) + + ## Merge the three epubs together. + doMerge(tmp, + [titleio,updateio,newchaptersio], + fromfirst=True, + titlenavpoints=False, + striptitletoc=False, + forceunique=False) + + + else: # update, but there's no epub extant, so do overwrite. + collision = OVERWRITE + + if collision == OVERWRITE: + writer.writeStory(tmp) + + db.add_format_with_hooks(book_id, fileform, tmp, index_is_id=True) + + # get all formats. + if prefs['deleteotherforms'] and collision in (OVERWRITE, UPDATE): + fmts = set([x.lower() for x in db.formats(book_id, index_is_id=True).split(',')]) + for fmt in fmts: + if fmt != fileform: + print("remove f:"+fmt) + db.remove_format(book_id, fmt,index_is_id=True)#, notify=False + + if updatemeta or collision == CALIBREONLY: + db.set_metadata(book_id,mi) + + else: # no matching, adding new. + writer.writeStory(tmp) + (notadded,addedcount)=db.add_books([tmp],[fileform],[mi], add_duplicates=True) + + + # Otherwise list of books doesn't update right away. + if addedcount: + self.gui.library_view.model().books_added(addedcount) + + self.gui.library_view.model().refresh() + + del adapter + del writer + +def f(x): + if x.strip(): return True + else: return False + +def get_url_list(urls): + return filter(f,urls.strip().splitlines()) diff --git a/calibre-plugin/images/icon.png b/calibre-plugin/images/icon.png new file mode 100644 index 0000000..75b2806 Binary files /dev/null and b/calibre-plugin/images/icon.png differ diff --git a/calibre-plugin/images/icon.xcf b/calibre-plugin/images/icon.xcf new file mode 100644 index 0000000..36c63a5 Binary files /dev/null and b/calibre-plugin/images/icon.xcf differ diff --git a/calibre-plugin/plugin-import-name-fanfictiondownloader_plugin.txt b/calibre-plugin/plugin-import-name-fanfictiondownloader_plugin.txt new file mode 100644 index 0000000..e69de29 diff --git a/defaults.ini b/defaults.ini index 6da575a..f791b51 100644 --- a/defaults.ini +++ b/defaults.ini @@ -105,6 +105,10 @@ zip_filename: ${title}-${siteabbrev}_${storyId}${formatext}.zip ## zip_filename. allow_unsafe_filename: false +## entries to make epub subjects and calibre tags +## lastupdate creates two tags: "Last Update Year/Month: %Y/%m" and "Last Update: %Y/%m/%d" +include_subject_tags: extratags, genre, category, characters, lastupdate, status + ## extra tags (comma separated) to include, primarily for epub. extratags: FanFiction @@ -137,10 +141,6 @@ windows_eol: true ## epub is already a zip file. zip_output: false -## entries to make epub subject tags -## lastupdate creates two tags: "Last Update Year/Month: %Y/%m" and "Last Update: %Y/%m/%d" -include_subject_tags: extratags, genre, category, characters, lastupdate, status - ## epub carries the TOC in metadata. ## mobi generated from epub will have a TOC at the end. include_tocpage: false diff --git a/fanficdownloader/adapters/__init__.py b/fanficdownloader/adapters/__init__.py index 0800ca4..89ef2c9 100644 --- a/fanficdownloader/adapters/__init__.py +++ b/fanficdownloader/adapters/__init__.py @@ -15,21 +15,52 @@ # limitations under the License. # -import os, re, sys, glob +import os, re, sys, glob, types from os.path import dirname, basename, normpath import logging import urlparse as up -import fanficdownloader.exceptions as exceptions +from .. import exceptions as exceptions + +## must import each adapter here. + +import adapter_test1 +import adapter_fanfictionnet +import adapter_fanficcastletvnet +import adapter_fanfictionnet +import adapter_fictionalleyorg +import adapter_fictionpresscom +import adapter_ficwadcom +import adapter_fimfictionnet +import adapter_harrypotterfanfictioncom +import adapter_mediaminerorg +import adapter_potionsandsnitchesnet +import adapter_tenhawkpresentscom +import adapter_adastrafanficcom +import adapter_thewriterscoffeeshopcom +import adapter_tthfanficorg +import adapter_twilightednet +import adapter_twiwritenet +import adapter_whoficcom +import adapter_siyecouk ## This bit of complexity allows adapters to be added by just adding -## the source file. It eliminates the long if/else clauses we used to -## need to pick out the adapter. +## importing. It eliminates the long if/else clauses we used to need +## to pick out the adapter. ## List of registered site adapters. - __class_list = [] +def imports(): + for name, val in globals().items(): + if isinstance(val, types.ModuleType): + yield val.__name__ + +for x in imports(): + if "fanficdownloader.adapters.adapter_" in x: + #print x + __class_list.append(sys.modules[x].getClass()) + def getAdapter(config,url): ## fix up leading protocol. fixedurl = re.sub(r"(?i)^[htp]+[:/]+","http://",url.strip()) @@ -63,16 +94,3 @@ def getClassFor(domain): for cls in __class_list: if cls.matchesSite(domain): return cls - -## Automatically import each adapter_*.py file. -## Each implement getClass() to their class - -filelist = glob.glob(dirname(__file__)+'/adapter_*.py') -sys.path.insert(0,normpath(dirname(__file__))) - -for file in filelist: - #print "file: "+basename(file)[:-3] - module = __import__(basename(file)[:-3]) - __class_list.append(module.getClass()) - -del sys.path[0] diff --git a/fanficdownloader/adapters/adapter_adastrafanficcom.py b/fanficdownloader/adapters/adapter_adastrafanficcom.py index be600c6..bc908a2 100644 --- a/fanficdownloader/adapters/adapter_adastrafanficcom.py +++ b/fanficdownloader/adapters/adapter_adastrafanficcom.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_fanficcastletvnet.py b/fanficdownloader/adapters/adapter_fanficcastletvnet.py index 96d5d1b..ee31b47 100644 --- a/fanficdownloader/adapters/adapter_fanficcastletvnet.py +++ b/fanficdownloader/adapters/adapter_fanficcastletvnet.py @@ -20,9 +20,9 @@ import logging import re import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_fanfictionnet.py b/fanficdownloader/adapters/adapter_fanfictionnet.py index 42e2b88..4afa74e 100644 --- a/fanficdownloader/adapters/adapter_fanfictionnet.py +++ b/fanficdownloader/adapters/adapter_fanfictionnet.py @@ -21,8 +21,8 @@ import re import urllib2 import time -import fanficdownloader.BeautifulSoup as bs -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_fictionalleyorg.py b/fanficdownloader/adapters/adapter_fictionalleyorg.py index 14a03ec..7eb9e37 100644 --- a/fanficdownloader/adapters/adapter_fictionalleyorg.py +++ b/fanficdownloader/adapters/adapter_fictionalleyorg.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_ficwadcom.py b/fanficdownloader/adapters/adapter_ficwadcom.py index 2782199..8d63aba 100644 --- a/fanficdownloader/adapters/adapter_ficwadcom.py +++ b/fanficdownloader/adapters/adapter_ficwadcom.py @@ -22,9 +22,9 @@ import urllib2 import time import httplib, urllib -import fanficdownloader.BeautifulSoup as bs -import fanficdownloader.exceptions as exceptions -from fanficdownloader.htmlcleanup import stripHTML +from .. import BeautifulSoup as bs +from .. import exceptions as exceptions +from ..htmlcleanup import stripHTML from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_fimfictionnet.py b/fanficdownloader/adapters/adapter_fimfictionnet.py index 8332203..cddcb8f 100644 --- a/fanficdownloader/adapters/adapter_fimfictionnet.py +++ b/fanficdownloader/adapters/adapter_fimfictionnet.py @@ -21,9 +21,9 @@ import re import urllib2 import cookielib as cl -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate @@ -88,8 +88,7 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter): soup = bs.BeautifulSoup(data).find("div", {"class":"content_box post_content_box"}) - title = soup.find("h2").find("a").text # first a link in first h2 is title. - author = soup.find("h2").find("span",{'class':'author'}).find("a").text + title, author = [link.text for link in soup.find("h2").findAll("a")] self.story.setMetadata("title", title) self.story.setMetadata("author", author) self.story.setMetadata("authorId", author) # The author's name will be unique @@ -159,4 +158,4 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter): if soup == None: raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) return utf8FromSoup(soup) - + \ No newline at end of file diff --git a/fanficdownloader/adapters/adapter_harrypotterfanfictioncom.py b/fanficdownloader/adapters/adapter_harrypotterfanfictioncom.py index 155ebf7..2d44f9b 100644 --- a/fanficdownloader/adapters/adapter_harrypotterfanfictioncom.py +++ b/fanficdownloader/adapters/adapter_harrypotterfanfictioncom.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_mediaminerorg.py b/fanficdownloader/adapters/adapter_mediaminerorg.py index 62aa50c..23c72e3 100644 --- a/fanficdownloader/adapters/adapter_mediaminerorg.py +++ b/fanficdownloader/adapters/adapter_mediaminerorg.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_potionsandsnitchesnet.py b/fanficdownloader/adapters/adapter_potionsandsnitchesnet.py index 6cb4b41..de9a722 100644 --- a/fanficdownloader/adapters/adapter_potionsandsnitchesnet.py +++ b/fanficdownloader/adapters/adapter_potionsandsnitchesnet.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_siyecouk.py b/fanficdownloader/adapters/adapter_siyecouk.py new file mode 100644 index 0000000..8222b15 --- /dev/null +++ b/fanficdownloader/adapters/adapter_siyecouk.py @@ -0,0 +1,275 @@ +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return SiyeCoUkAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class SiyeCoUkAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8",]# 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + # self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + # self.password = "" + # self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + logging.debug("storyId: (%s)"%self.story.getMetadata('storyId')) + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/siye/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','siye') # XXX + + # If all stories from the site fall into the same category, + # the site itself isn't likely to label them as such, so we + # do. + self.story.addToList("category","Harry Potter") # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y.%m.%d" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.siye.co.uk' # XXX + + @classmethod + def getAcceptDomains(cls): + return ['www.siye.co.uk','siye.co.uk'] + + def getSiteExampleURLs(self): + return "http://"+self.getSiteDomain()+"/siye/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+r"(www\.)?"+re.escape("siye.co.uk/siye/viewstory.php?sid=")+r"\d+$" + + # ## Login seems to be reasonably standard across eFiction sites. + # def needToLoginCheck(self, data): + # if 'Registered Users Only' in data \ + # or 'There is no such account on our website' in data \ + # or "That password doesn't match the one in our database" in data: + # return True + # else: + # return False + + # def performLogin(self, url): + # params = {} + + # if self.password: + # params['penname'] = self.username + # params['password'] = self.password + # else: + # params['penname'] = self.getConfig("username") + # params['password'] = self.getConfig("password") + # params['cookiecheck'] = '1' + # params['submit'] = 'Submit' + + # loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + # logging.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + # params['penname'])) + + # d = self._fetchUrl(loginUrl, params) + + # if "Member Account" not in d : #Member Account + # logging.info("Failed to login to URL %s as %s" % (loginUrl, + # params['penname'])) + # raise exceptions.FailedToLogin(url,params['penname']) + # return False + # else: + # return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # if self.is_adult or self.getConfig("is_adult"): + # # Weirdly, different sites use different warning numbers. + # # If the title search below fails, there's a good chance + # # you need a different number. print data at that point + # # and see what the 'click here to continue' url says. + # addurl = "&ageconsent=ok&warning=4" # XXX + # else: + # addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + # Except it doesn't this time. :-/ + url = self.url #+'&index=1'+addurl + logging.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # if self.needToLoginCheck(data): + # # need to log in for this one. + # self.performLogin(url) + # data = self._fetchUrl(url) + + # # The actual text that is used to announce you need to be an + # # adult varies from site to site. Again, print data before + # # the title search to troubleshoot. + # if "Age Consent Required" in data: # XXX + # raise exceptions.AdultCheckRequired(self.url) + + # if "Access denied. This story has not been validated by the adminstrators of this site." in data: + # raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/siye/'+a['href']) + self.story.setMetadata('author',a.string) + + # need(or easier) to pull other metadata from the author's list page. + authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + ## Title + titlea = authsoup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',titlea.string) + + # Find the chapters (from soup, not authsoup): + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/siye/'+chapter['href'])) + + if self.chapterUrls: + self.story.setMetadata('numChapters',len(self.chapterUrls)) + else: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + self.story.setMetadata('numChapters',1) + + # The stuff we can get from the chapter list/one-shot page are + # in the first table with 95% width. + metatable = soup.find('table',{'width':'95%'}) + + # Categories + cat_as = metatable.findAll('a', href=re.compile(r'categories.php')) + for cat_a in cat_as: + self.story.addToList('category',stripHTML(cat_a)) + + moremetaparts = stripHTML(metatable).split('\n') + for part in moremetaparts: + part = part.strip() + if part.startswith("Characters:"): + part = part[part.find(':')+1:] + for item in part.split(','): + if item.strip() != "None": + self.story.addToList('characters',item) + + if part.startswith("Genres:"): + part = part[part.find(':')+1:] + for item in part.split(','): + if item.strip() != "None": + self.story.addToList('genre',item) + + if part.startswith("Warnings:"): + part = part[part.find(':')+1:] + for item in part.split(','): + if item.strip() != "None": + self.story.addToList('warnings',item) + + if part.startswith("Rating:"): + part = part[part.find(':')+1:] + self.story.setMetadata('rating',part) + + if part.startswith("Summary:"): + part = part[part.find(':')+1:] + self.story.setMetadata('description',part) + + + + # want to get the next tr of the table. + #print("%s"%titlea.parent.parent.findNextSibling('tr')) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + moremeta = stripHTML(titlea.parent.parent.findNextSibling('tr')) + for part in moremeta.replace(' - ','\n').split('\n'): + #print("part:%s"%part) + try: + (name,value) = part.split(': ') + except: + # not going to worry about fancier processing for the bits + # that don't match. + continue + name=name.strip() + value=value.strip() + if name == 'Published': + self.story.setMetadata('datePublished', makeDate(value, self.dateformat)) + if name == 'Updated': + self.story.setMetadata('dateUpdated', makeDate(value, self.dateformat)) + if name == 'Completed': + if value == 'Yes': + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + if name == 'Words': + self.story.setMetadata('numWords', value) + + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logging.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + # not the most unique thing in the work, bit it appears to be + # the best we can do there. + story = soup.find('span', {'style' : 'font-size: 100%;'}) + + if None == story: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return utf8FromSoup(story) diff --git a/fanficdownloader/adapters/adapter_tenhawkpresentscom.py b/fanficdownloader/adapters/adapter_tenhawkpresentscom.py index 6ec8a24..a4922d2 100644 --- a/fanficdownloader/adapters/adapter_tenhawkpresentscom.py +++ b/fanficdownloader/adapters/adapter_tenhawkpresentscom.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_test1.py b/fanficdownloader/adapters/adapter_test1.py index 1debdfc..f00b963 100644 --- a/fanficdownloader/adapters/adapter_test1.py +++ b/fanficdownloader/adapters/adapter_test1.py @@ -19,8 +19,8 @@ import datetime import time import logging -import fanficdownloader.BeautifulSoup as bs -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from .. import exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate @@ -54,6 +54,12 @@ class TestSiteAdapter(BaseSiteAdapter): if self.story.getMetadata('storyId') == '666': raise exceptions.StoryDoesNotExist(self.url) + if self.story.getMetadata('storyId').startswith('670'): + time.sleep(1.0) + + if self.story.getMetadata('storyId').startswith('671'): + time.sleep(1.0) + if self.getConfig("username"): self.username = self.getConfig("username") @@ -63,18 +69,18 @@ class TestSiteAdapter(BaseSiteAdapter): if self.story.getMetadata('storyId') == '664': self.story.setMetadata(u'title',"Test Story Title "+self.crazystring) else: - self.story.setMetadata(u'title',"Test Story Title") + self.story.setMetadata(u'title',"Test Story Title "+self.story.getMetadata('storyId')) self.story.setMetadata('storyUrl',self.url) self.story.setMetadata('description',u'Description '+self.crazystring+u''' Done Some more longer description. "I suck at summaries!" "Better than it sounds!" "My first fic" ''') - self.story.setMetadata('datePublished',makeDate("1972-01-31","%Y-%m-%d")) + self.story.setMetadata('datePublished',makeDate("1975-03-15","%Y-%m-%d")) self.story.setMetadata('dateCreated',datetime.datetime.now()) if self.story.getMetadata('storyId') == '669': self.story.setMetadata('dateUpdated',datetime.datetime.now()) else: - self.story.setMetadata('dateUpdated',makeDate("1975-01-31","%Y-%m-%d")) + self.story.setMetadata('dateUpdated',makeDate("1975-04-15","%Y-%m-%d")) self.story.setMetadata('numWords','123456') self.story.setMetadata('status','In-Completed') self.story.setMetadata('rating','Tweenie') @@ -100,7 +106,7 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!" ('Chapter 3, Over Cinnabar',self.url+"&chapter=4"), ('Chapter 4',self.url+"&chapter=5"), ('Chapter 5',self.url+"&chapter=6"), - # ('Chapter 6',self.url+"&chapter=6"), + ('Chapter 6',self.url+"&chapter=6"), # ('Chapter 7',self.url+"&chapter=6"), # ('Chapter 8',self.url+"&chapter=6"), # ('Chapter 9',self.url+"&chapter=6"), @@ -128,8 +134,9 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!" if self.story.getMetadata('storyId') == '667': raise exceptions.FailedToDownload("Error downloading Chapter: %s!" % url) - if self.story.getMetadata('storyId') == '670': - time.sleep(2.0) + if self.story.getMetadata('storyId').startswith('670') or \ + self.story.getMetadata('storyId').startswith('672'): + time.sleep(1.0) if "chapter=1" in url : text=u''' @@ -143,6 +150,8 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
http://test1.com?sid=668 - raises FailedToLogin unless username='Me'
http://test1.com?sid=669 - Succeeds with Updated Date=now
http://test1.com?sid=670 - Succeeds, but sleeps 2sec on each chapter
+http://test1.com?sid=671 - Succeeds, but sleeps 2sec metadata only
+http://test1.com?sid=672 - Succeeds, quick meta, sleeps 2sec chapters only
And other storyId will succeed with the same output.
''' diff --git a/fanficdownloader/adapters/adapter_thewriterscoffeeshopcom.py b/fanficdownloader/adapters/adapter_thewriterscoffeeshopcom.py index 5a399a6..fecc2bd 100644 --- a/fanficdownloader/adapters/adapter_thewriterscoffeeshopcom.py +++ b/fanficdownloader/adapters/adapter_thewriterscoffeeshopcom.py @@ -21,9 +21,9 @@ import re import urllib import urllib2 -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate diff --git a/fanficdownloader/adapters/adapter_tthfanficorg.py b/fanficdownloader/adapters/adapter_tthfanficorg.py index 1906aaa..3964172 100644 --- a/fanficdownloader/adapters/adapter_tthfanficorg.py +++ b/fanficdownloader/adapters/adapter_tthfanficorg.py @@ -21,9 +21,9 @@ import re import urllib2 import time -import fanficdownloader.BeautifulSoup as bs -from fanficdownloader.htmlcleanup import stripHTML -import fanficdownloader.exceptions as exceptions +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate @@ -34,6 +34,8 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter): self.story.setMetadata('siteabbrev','tth') self.dateformat = "%d %b %y" self.is_adult=False + self.username = None + self.password = None # get storyId from url--url validation guarantees query correct m = re.match(self.getSiteURLPattern(),url) if m: @@ -60,6 +62,48 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter): def getSiteURLPattern(self): return r"http://www.tthfanfic.org/(T-\d+/)?Story-(?P