Merge calibre-plugin branch back to default trunk.

This commit is contained in:
Jim Miller committed 2012-01-01 09:26:17 -06:00
commit c944d837fb
38 files changed
+1589 -119

No files matched your search

+1 -1
View File
@@ -1,6 +1,6 @@
# ffd-retief-hrd fanfictiondownloader
application: fanfictiondownloader
version: 4-1-1
version: 4-2-0
runtime: python27
api_version: 1
threadsafe: true
+90
View File
@@ -0,0 +1,90 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
# -*- coding: utf-8 -*-
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2011, Jim Miller'
__docformat__ = 'restructuredtext en'
# The class that all Interface Action plugin wrappers must inherit from
from calibre.customize import InterfaceActionBase
class InterfacePluginDemo(InterfaceActionBase):
'''
This class is a simple wrapper that provides information about the
actual plugin class. The actual interface plugin class is called
InterfacePlugin and is defined in the ffdl_plugin.py file, as
specified in the actual_plugin field below.
The reason for having two classes is that it allows the command line
calibre utilities to run without needing to load the GUI libraries.
'''
name = 'FanFictionDownLoader'
description = 'UI plugin to download FanFiction stories from various sites.'
supported_platforms = ['windows', 'osx', 'linux']
author = 'Jim Miller'
version = (1, 0, 4)
minimum_calibre_version = (0, 8, 30)
# action_menu_clone_qaction = True
#: This field defines the GUI plugin class that contains all the code
#: that actually does something. Its format is module_path:class_name
#: The specified class must be defined in the specified module.
actual_plugin = 'calibre_plugins.fanfictiondownloader_plugin.ffdl_plugin:FanFictionDownLoaderPlugin'
def is_customizable(self):
'''
This method must return True to enable customization via
Preferences->Plugins
'''
return True
def config_widget(self):
'''
Implement this method and :meth:`save_settings` in your plugin to
use a custom configuration dialog.
This method, if implemented, must return a QWidget. The widget can have
an optional method validate() that takes no arguments and is called
immediately after the user clicks OK. Changes are applied if and only
if the method returns True.
If for some reason you cannot perform the configuration at this time,
return a tuple of two strings (message, details), these will be
displayed as a warning dialog to the user and the process will be
aborted.
The base class implementation of this method raises NotImplementedError
so by default no user configuration is possible.
'''
# It is important to put this import statement here rather than at the
# top of the module as importing the config class will also cause the
# GUI libraries to be loaded, which we do not want when using calibre
# from the command line
from calibre_plugins.fanfictiondownloader_plugin.config import ConfigWidget
return ConfigWidget()
def save_settings(self, config_widget):
'''
Save the settings specified by the user with config_widget.
:param config_widget: The widget returned by :meth:`config_widget`.
'''
config_widget.save_settings()
# Apply the changes
ac = self.actual_plugin_
if ac is not None:
ac.apply_settings()
# For testing, run from command line with this:
# calibre-debug -e __init__.py
#
if __name__ == '__main__':
from PyQt4.Qt import QApplication
from calibre.gui2.preferences import test_widget
app = QApplication([])
test_widget('Advanced', 'Plugins')
+11
View File
@@ -0,0 +1,11 @@
<p>FanFictionDownLoader Plugin</p>
<hr />
<p><a href="http://code.google.com/p/fanficdownloader/">http://code.google.com/p/fanficdownloader/</a></p>
<p>Created by Jim Miller, borrowing heavily from
Kovid Goyal's 'The InterfacePlugin Demo' and
Grant Drake's 'Count Pages' plugins.</p>
<p>Requires calibre >= 0.8.30</p>
+172
View File
@@ -0,0 +1,172 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2011, Jim Miller'
__docformat__ = 'restructuredtext en'
from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QLabel, QLineEdit,
QTextEdit, QComboBox, QCheckBox, QPushButton)
from calibre.utils.config import JSONConfig
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (OVERWRITE, ADDNEW, SKIP,CALIBREONLY,UPDATE)
# This is where all preferences for this plugin will be stored
# Remember that this name (i.e. plugins/fanfictiondownloader_plugin) is also
# in a global namespace, so make it as unique as possible.
# You should always prefix your config file name with plugins/,
# so as to ensure you dont accidentally clobber a calibre config file
prefs = JSONConfig('plugins/fanfictiondownloader_plugin')
# urlsfrompriority values
CLIP = 'Clipboard'
SELECTED = 'Selected Stories'
# Set defaults
prefs.defaults['personal.ini'] = get_resources('example.ini')
prefs.defaults['updatemeta'] = True
prefs.defaults['onlyoverwriteifnewer'] = False
prefs.defaults['urlsfromclip'] = True
prefs.defaults['urlsfromselected'] = True
prefs.defaults['urlsfrompriority'] = SELECTED
prefs.defaults['fileform'] = 'epub'
prefs.defaults['collision'] = OVERWRITE
prefs.defaults['deleteotherforms'] = False
class ConfigWidget(QWidget):
def __init__(self):
QWidget.__init__(self)
self.l = QVBoxLayout()
self.setLayout(self.l)
horz = QHBoxLayout()
label = QLabel('Default Output &Format:')
horz.addWidget(label)
self.fileform = QComboBox(self)
self.fileform.addItem('epub')
self.fileform.addItem('mobi')
self.fileform.addItem('html')
self.fileform.addItem('txt')
self.fileform.setCurrentIndex(self.fileform.findText(prefs['fileform']))
self.fileform.setToolTip('Choose output format to create. May set default from plugin configuration.')
label.setBuddy(self.fileform)
horz.addWidget(self.fileform)
self.l.addLayout(horz)
horz = QHBoxLayout()
label = QLabel('Default If Story Already Exists?')
label.setToolTip("What to do if there's already an existing story with the same title and author.")
horz.addWidget(label)
self.collision = QComboBox(self)
self.collision.addItem(OVERWRITE)
self.collision.addItem(UPDATE)
self.collision.addItem(ADDNEW)
self.collision.addItem(SKIP)
self.collision.addItem(CALIBREONLY)
self.collision.setCurrentIndex(self.collision.findText(prefs['collision']))
self.collision.setToolTip('Overwrite will replace the existing story. Add New will create a new story with the same title and author.')
label.setBuddy(self.collision)
horz.addWidget(self.collision)
self.l.addLayout(horz)
self.updatemeta = QCheckBox('Default Update Calibre &Metadata?',self)
self.updatemeta.setToolTip('Update metadata for story in Calibre from web site?')
self.updatemeta.setChecked(prefs['updatemeta'])
self.l.addWidget(self.updatemeta)
self.onlyoverwriteifnewer = QCheckBox('Default Only Overwrite Story if Newer',self)
self.onlyoverwriteifnewer.setToolTip("Don't overwrite existing book unless the story on the web site is newer or from the same day.")
self.onlyoverwriteifnewer.setChecked(prefs['onlyoverwriteifnewer'])
self.l.addWidget(self.onlyoverwriteifnewer)
self.urlsfromclip = QCheckBox('Take URLs from Clipboard?',self)
self.urlsfromclip.setToolTip('Prefill URLs from valid URLs in Clipboard when starting plugin?')
self.urlsfromclip.setChecked(prefs['urlsfromclip'])
self.l.addWidget(self.urlsfromclip)
self.urlsfromselected = QCheckBox('Take URLs from Selected Stories?',self)
self.urlsfromselected.setToolTip('Prefill URLs from valid URLs in selected stories when starting plugin?')
self.urlsfromselected.setChecked(prefs['urlsfromselected'])
self.l.addWidget(self.urlsfromselected)
horz = QHBoxLayout()
label = QLabel('Take URLs from which first:')
label.setToolTip("If both clipbaord and selected enabled and populated, which is used?")
horz.addWidget(label)
self.urlsfrompriority = QComboBox(self)
self.urlsfrompriority.addItem(SELECTED)
self.urlsfrompriority.addItem(CLIP)
self.urlsfrompriority.setCurrentIndex(self.urlsfrompriority.findText(prefs['urlsfrompriority']))
self.urlsfrompriority.setToolTip('If both clipbaord and selected enabled and populated, which is used?')
label.setBuddy(self.urlsfrompriority)
horz.addWidget(self.urlsfrompriority)
self.l.addLayout(horz)
self.deleteotherforms = QCheckBox('Delete other existing formats?',self)
self.deleteotherforms.setToolTip('Check this to automatically delete all other ebook formats when updating an existing book.\nHandy if you have both a Nook(epub) and Kindle(mobi), for example.')
self.deleteotherforms.setChecked(prefs['deleteotherforms'])
self.l.addWidget(self.deleteotherforms)
self.label = QLabel('personal.ini:')
self.l.addWidget(self.label)
self.ini = QTextEdit(self)
self.ini.setLineWrapMode(QTextEdit.NoWrap)
self.ini.setText(prefs['personal.ini'])
self.l.addWidget(self.ini)
self.defaults = QPushButton('View Defaults', self)
self.defaults.setToolTip("View all of the plugin's configurable settings\nand their default settings.")
self.defaults.clicked.connect(self.show_defaults)
self.l.addWidget(self.defaults)
def save_settings(self):
prefs['fileform'] = unicode(self.fileform.currentText())
prefs['collision'] = unicode(self.collision.currentText())
prefs['updatemeta'] = self.updatemeta.isChecked()
prefs['urlsfrompriority'] = unicode(self.urlsfrompriority.currentText())
prefs['urlsfromclip'] = self.urlsfromclip.isChecked()
prefs['urlsfromselected'] = self.urlsfromselected.isChecked()
prefs['onlyoverwriteifnewer'] = self.onlyoverwriteifnewer.isChecked()
prefs['deleteotherforms'] = self.deleteotherforms.isChecked()
ini = unicode(self.ini.toPlainText())
if ini:
prefs['personal.ini'] = ini
else:
# if they've removed everything, clear it so they get the
# default next time.
del prefs['personal.ini']
def show_defaults(self):
text = get_resources('defaults.ini')
ShowDefaultsIniDialog(self.windowIcon(),text,self).exec_()
class ShowDefaultsIniDialog(QDialog):
def __init__(self, icon, text, parent=None):
QDialog.__init__(self, parent)
self.resize(600, 500)
self.l = QVBoxLayout()
self.setLayout(self.l)
self.label = QLabel("Plugin Defaults (Read-Only)")
self.label.setToolTip("These all of the plugin's configurable settings\nand their default settings.")
self.setWindowTitle(_('Plugin Defaults'))
self.setWindowIcon(icon)
self.l.addWidget(self.label)
self.ini = QTextEdit(self)
self.ini.setToolTip("These all of the plugin's configurable settings\nand their default settings.")
self.ini.setLineWrapMode(QTextEdit.NoWrap)
self.ini.setText(text)
self.ini.setReadOnly(True)
self.l.addWidget(self.ini)
self.ok_button = QPushButton('OK', self)
self.ok_button.clicked.connect(self.hide)
self.l.addWidget(self.ok_button)
+286
View File
@@ -0,0 +1,286 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2011, Jim Miller'
__docformat__ = 'restructuredtext en'
import traceback
from PyQt4.Qt import (QDialog, QMessageBox, QVBoxLayout, QHBoxLayout, QGridLayout,
QPushButton, QProgressDialog, QString, QLabel, QCheckBox,
QTextEdit, QLineEdit, QInputDialog, QComboBox, QClipboard,
QProgressDialog, QTimer, QDialogButtonBox, QPixmap, Qt )
from calibre.gui2 import error_dialog, warning_dialog, question_dialog, info_dialog
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters,writers,exceptions
OVERWRITE='Overwrite'
UPDATE='Update EPUB'
ADDNEW='Add New'
SKIP='Skip'
CALIBREONLY='Update Calibre Metadata Only'
class DownloadDialog(QDialog):
def __init__(self, gui, prefs, icon, url_list_text, do_user_config, start_downloads):
QDialog.__init__(self, gui)
self.gui = gui
self.do_user_config = do_user_config
self.start_downloads = start_downloads
self.setMinimumWidth(300)
self.l = QVBoxLayout()
self.setLayout(self.l)
self.setWindowTitle('FanFictionDownLoader')
self.setWindowIcon(icon)
self.l.addWidget(QLabel('Story URL(s), one per line:'))
self.url = QTextEdit(self)
self.url.setToolTip('URLs for stories, one per line.\nWill take URLs from selected books or clipboard, but only valid URLs.')
self.url.setLineWrapMode(QTextEdit.NoWrap)
self.url.setText(url_list_text)
self.l.addWidget(self.url)
self.ffdl_button = QPushButton(
'Download Stories', self)
self.ffdl_button.setToolTip('Start download(s).')
self.ffdl_button.clicked.connect(self.ffdl)
# if there's already URL(s), focus 'go' button
if url_list_text:
self.ffdl_button.setFocus()
self.l.addWidget(self.ffdl_button)
horz = QHBoxLayout()
label = QLabel('Output &Format:')
horz.addWidget(label)
self.fileform = QComboBox(self)
self.fileform.addItem('epub')
self.fileform.addItem('mobi')
self.fileform.addItem('html')
self.fileform.addItem('txt')
self.fileform.setCurrentIndex(self.fileform.findText(prefs['fileform']))
self.fileform.setToolTip('Choose output format to create. May set default from plugin configuration.')
label.setBuddy(self.fileform)
horz.addWidget(self.fileform)
self.l.addLayout(horz)
horz = QHBoxLayout()
label = QLabel('If Story Already Exists?')
label.setToolTip("What to do if there's already an existing story with the same title and author.")
horz.addWidget(label)
self.collision = QComboBox(self)
self.collision.addItem(OVERWRITE)
self.collision.addItem(UPDATE)
self.collision.addItem(ADDNEW)
self.collision.addItem(SKIP)
self.collision.addItem(CALIBREONLY)
self.collision.setCurrentIndex(self.collision.findText(prefs['collision']))
self.collision.setToolTip(OVERWRITE+' will replace the existing story.\n'+
UPDATE+' will download new chapters only and add to existing EPUB.\n'+
ADDNEW+' will create a new story with the same title and author.\n'+
SKIP+' will not download existing stories.\n'+
CALIBREONLY+' will not download stories, but will update Calibre metadata.')
label.setBuddy(self.collision)
horz.addWidget(self.collision)
self.l.addLayout(horz)
self.updatemeta = QCheckBox('Update Calibre &Metadata?',self)
self.updatemeta.setToolTip('Update metadata for story in Calibre from web site?')
self.updatemeta.setChecked(prefs['updatemeta'])
self.l.addWidget(self.updatemeta)
self.onlyoverwriteifnewer = QCheckBox('Only Overwrite Story if Newer',self)
self.onlyoverwriteifnewer.setToolTip("Don't overwrite existing book unless the story on the web site is newer.\n"+
"From the same day counts as 'newer' because the sites don't give update time.")
self.onlyoverwriteifnewer.setChecked(prefs['onlyoverwriteifnewer'])
self.l.addWidget(self.onlyoverwriteifnewer)
horz = QHBoxLayout()
self.about_button = QPushButton('About', self)
self.about_button.clicked.connect(self.about)
horz.addWidget(self.about_button)
self.conf_button = QPushButton(
'Configure this plugin', self)
self.conf_button.clicked.connect(self.config)
horz.addWidget(self.conf_button)
self.l.addLayout(horz)
self.resize(self.sizeHint())
def about(self):
# Get the about text from a file inside the plugin zip file
# The get_resources function is a builtin function defined for all your
# plugin code. It loads files from the plugin zip file. It returns
# the bytes from the specified file.
#
# Note that if you are loading more than one file, for performance, you
# should pass a list of names to get_resources. In this case,
# get_resources will return a dictionary mapping names to bytes. Names that
# are not found in the zip file will not be in the returned dictionary.
text = get_resources('about.txt')
# QMessageBox.about(self, 'About the FanFictionDownLoader Plugin',
# text.decode('utf-8'))
AboutDialog(self.windowIcon(),text,self).exec_()
def ffdl(self):
self.start_downloads(unicode(self.url.toPlainText()),
unicode(self.fileform.currentText()),
unicode(self.collision.currentText()),
self.updatemeta.isChecked(),
self.onlyoverwriteifnewer.isChecked())
self.hide()
def config(self):
self.do_user_config(parent=self)
class UserPassDialog(QDialog):
'''
Need to collect User/Pass for some sites.
'''
def __init__(self, gui, site):
QDialog.__init__(self, gui)
self.gui = gui
self.status=False
self.setWindowTitle('User/Password')
self.l = QGridLayout()
self.setLayout(self.l)
self.l.addWidget(QLabel("%s requires you to login to download this story."%site),0,0,1,2)
self.l.addWidget(QLabel("User:"),1,0)
self.user = QLineEdit(self)
self.l.addWidget(self.user,1,1)
self.l.addWidget(QLabel("Password:"),2,0)
self.passwd = QLineEdit(self)
self.l.addWidget(self.passwd,2,1)
self.ok_button = QPushButton('OK', self)
self.ok_button.clicked.connect(self.ok)
self.l.addWidget(self.ok_button,3,0)
self.cancel_button = QPushButton('Cancel', self)
self.cancel_button.clicked.connect(self.cancel)
self.l.addWidget(self.cancel_button,3,1)
self.resize(self.sizeHint())
def ok(self):
self.status=True
self.hide()
def cancel(self):
self.status=False
self.hide()
class MetadataProgressDialog(QProgressDialog):
'''
ProgressDialog displayed while fetching metadata for each story.
'''
def __init__(self, gui, loop_list, fileform, getadapter_function, download_list_function, db):
QProgressDialog.__init__(self,
"Fetching metadata for stories...",
QString(), 0, len(loop_list), gui)
self.setWindowTitle("Downloading metadata for stories")
self.setMinimumWidth(500)
self.gui = gui
self.db = db
self.loop_list = loop_list
self.fileform = fileform
self.getadapter_function = getadapter_function
self.download_list_function = download_list_function
self.i, self.loop_bad, self.loop_good = 0, [], []
## self.do_loop does QTimer.singleShot on self.do_loop also.
## A weird way to do a loop, but that was the example I had.
QTimer.singleShot(0, self.do_loop)
self.exec_()
def updateStatus(self):
self.setLabelText("Fetched metadata for %d of %d"%(self.i+1,len(self.loop_list)))
self.setValue(self.i+1)
print(self.labelText())
def do_loop(self):
print("self.i:%d"%self.i)
if self.i == 0:
self.setValue(0)
if self.i >= len(self.loop_list) or self.wasCanceled():
return self.do_when_finished()
else:
current = self.loop_list[self.i]
try:
## collision spec passed into getadapter by partial from ffdl_plugin
## no retval only if it exists, but collision is SKIP
retval = self.getadapter_function(current,self.fileform)
self.loop_good.append((current,retval))
except Exception as e:
print("%s:%s"%(current,e))
self.loop_bad.append((current,e))
traceback.print_exc()
self.updateStatus()
self.i += 1
QTimer.singleShot(0, self.do_loop)
def do_when_finished(self):
self.hide()
# Queues a job to process these ePub/Mobi books in the background.
self.download_list_function(self.loop_good,self.fileform)
if self.loop_bad != []:
res = []
for j in self.loop_bad:
res.append('%s : %s'%j)
msg = '%s' % '\n'.join(res)
warning_dialog(self.gui, _('Not going to download some stories'),
_('Not going to download %d of %d stories.') %
(len(self.loop_bad), len(self.loop_list)),
msg).exec_()
# else:
# info_dialog(self.gui, "Starting Downloads",
# "Got metadata and started download for %d stories."%len(self.loop_good),
# show_copy_button=False).exec_()
self.gui = None
class AboutDialog(QDialog):
def __init__(self, icon, text, parent=None):
QDialog.__init__(self, parent)
self.resize(400, 250)
self.l = QGridLayout()
self.setLayout(self.l)
self.logo = QLabel()
self.logo.setMaximumWidth(110)
self.logo.setPixmap(QPixmap(icon.pixmap(100,100)))
self.label = QLabel(text)
self.label.setOpenExternalLinks(True)
self.label.setWordWrap(True)
self.setWindowTitle(_('Update available!'))
self.setWindowIcon(icon)
self.l.addWidget(self.logo, 0, 0)
self.l.addWidget(self.label, 0, 1)
self.bb = QDialogButtonBox(self)
b = self.bb.addButton(_('OK'), self.bb.AcceptRole)
b.setDefault(True)
self.l.addWidget(self.bb, 2, 0, 1, -1)
self.bb.accepted.connect(self.accept)
class NotGoingToDownload(Exception):
def __init__(self,error):
self.error=error
def __str__(self):
return self.error
+464
View File
@@ -0,0 +1,464 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2011, Jim Miller'
__docformat__ = 'restructuredtext en'
import ConfigParser, os
from StringIO import StringIO
from functools import partial
from datetime import datetime
from PyQt4.Qt import (QApplication)
# The class that all interface action plugins must inherit from
from calibre.ptempfile import PersistentTemporaryFile
from calibre.ebooks.metadata import MetaInformation
from calibre.ebooks.metadata.meta import get_metadata
from calibre.gui2 import error_dialog, warning_dialog, question_dialog, info_dialog
from calibre.gui2.actions import InterfaceAction
from calibre.gui2.threaded_jobs import ThreadedJob
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
from calibre_plugins.fanfictiondownloader_plugin.config import (prefs, CLIP, SELECTED)
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (
DownloadDialog, MetadataProgressDialog, UserPassDialog,
OVERWRITE, UPDATE, ADDNEW, SKIP, CALIBREONLY, NotGoingToDownload )
# because calibre immediately transforms html into zip and don't want
# to have an 'if html'. db.has_format is cool with the case mismatch,
# but if I'm doing it anyway...
formmapping = {
'epub':'EPUB',
'mobi':'MOBI',
'html':'ZIP',
'txt':'TXT'
}
class FanFictionDownLoaderPlugin(InterfaceAction):
name = 'FanFictionDownLoader'
# Declare the main action associated with this plugin
# The keyboard shortcut can be None if you dont want to use a keyboard
# shortcut. Remember that currently calibre has no central management for
# keyboard shortcuts, so try to use an unusual/unused shortcut.
# (text, icon_path, tooltip, keyboard shortcut)
# icon_path isn't in the zip--icon loaded below.
action_spec = ('FanFictionDownLoader', None,
'Download FanFiction stories from various web sites', None)
action_type = 'global'
def genesis(self):
# This method is called once per plugin, do initial setup here
# Set the icon for this interface action
# The get_icons function is a builtin function defined for all your
# plugin code. It loads icons from the plugin zip file. It returns
# QIcon objects, if you want the actual data, use the analogous
# get_resources builtin function.
#
# Note that if you are loading more than one icon, for performance, you
# should pass a list of names to get_icons. In this case, get_icons
# will return a dictionary mapping names to QIcons. Names that
# are not found in the zip file will result in null QIcons.
icon = get_icons('images/icon.png')
# The qaction is automatically created from the action_spec defined
# above
self.qaction.setIcon(icon)
# Call function when plugin triggered.
self.qaction.triggered.connect(self.show_dialog)
def show_dialog(self):
# The base plugin object defined in __init__.py
base_plugin_object = self.interface_action_base_plugin
# Show the config dialog
# The config dialog can also be shown from within
# Preferences->Plugins, which is why the do_user_config
# method is defined on the base plugin class
do_user_config = base_plugin_object.do_user_config
# The current database shown in the GUI
# db is an instance of the class LibraryDatabase2 from database.py
# This class has many, many methods that allow you to do a lot of
# things.
self.db = self.gui.current_db
# pre-pop urls from selected stories or clipboard. but with
# configurable priority and on/off option on each
if prefs['urlsfrompriority'] == SELECTED:
from_first_func = self.get_urls_select
from_second_func = self.get_urls_clip
elif prefs['urlsfrompriority'] == CLIP:
from_first_func = self.get_urls_clip
from_second_func = self.get_urls_select
url_list_text = from_first_func()
if not url_list_text:
url_list_text = from_second_func()
# self.gui is the main calibre GUI. It acts as the gateway to access
# all the elements of the calibre user interface, it should also be the
# parent of the dialog
# DownloadDialog just collects URLs, format and presents buttons.
d = DownloadDialog(self.gui,
prefs,
self.qaction.icon(),
url_list_text,
do_user_config, # method for config button
self.start_downloads, # method to start downloads
)
d.show()
## if there's rows selected, try to find a source URL from
## either identifier in the metadata, or from the epub
## metadata.
def get_urls_select(self):
url_list = []
rows = self.gui.library_view.selectionModel().selectedRows()
if rows and prefs['urlsfromselected']:
book_ids = self.gui.library_view.get_selected_ids()
#print("book_ids: %s"%book_ids)
for book_id in book_ids:
#print("book_on_device:%s"%self.db.book_on_device(book_id))
identifiers = self.db.get_identifiers(book_id,index_is_id=True)
if 'url' in identifiers:
# identifiers have :->| in url.
#print("url from book:"+identifiers['url'].replace('|',':'))
url_list.append(identifiers['url'].replace('|',':'))
else:
## only epub has that in it.
if self.db.has_format(book_id,'EPUB',index_is_id=True):
existingepub = self.db.format(book_id,'EPUB',index_is_id=True, as_file=True)
mi = get_metadata(existingepub,'EPUB')
#print("mi:%s"%mi)
identifiers = mi.get_identifiers()
if 'url' in identifiers:
#print("url from epub:"+identifiers['url'].replace('|',':'))
url_list.append(identifiers['url'].replace('|',':'))
return self.get_valid_urls(url_list)
def get_urls_clip(self):
url_list = []
if prefs['urlsfromclip']:
# no rows selected, check for valid URLs in the clipboard.
cliptext = unicode(QApplication.instance().clipboard().text())
url_list.extend(cliptext.split())
return self.get_valid_urls(url_list)
def get_valid_urls(self,url_list):
url_list_text = ""
# Check and make sure the URLs are valid ffdl URLs.
if url_list:
# this is the accepted way to 'check for existance'? really?
try:
self.dummyconfig
except AttributeError:
self.dummyconfig = ConfigParser.SafeConfigParser()
alreadyin=[]
for url in url_list:
if url in alreadyin:
continue
alreadyin.append(url)
# pulling up an adapter is pretty low over-head. If
# it fails, it's a bad url.
try:
adapters.getAdapter(self.dummyconfig,url)
except:
pass
else:
if url_list_text:
url_list_text += "\n"
url_list_text += url
return url_list_text
def apply_settings(self):
# No need to do anything with perfs here, but we could.
prefs
def start_downloads(self,urls,fileform,
collision,updatemeta,onlyoverwriteifnewer):
url_list = get_url_list(urls)
self.fetchmeta_qpd = \
MetadataProgressDialog(self.gui,
url_list,
fileform,
partial(self.get_adapter_for_story, collision=collision,onlyoverwriteifnewer=onlyoverwriteifnewer),
partial(self.download_list,collision=collision,updatemeta=updatemeta,onlyoverwriteifnewer=onlyoverwriteifnewer),
self.db)
def get_adapter_for_story(self,url,fileform,
collision=SKIP,
onlyoverwriteifnewer=False):
'''
Returns adapter object for story at URL. To be called from
MetadataProgressDialog 'loop' to build up list of adapters. Also
pops dialogs for is adult, user/pass, duplicate
'''
print("URL:"+url)
## was self.ffdlconfig, but we need to be able to change it
## when doing epub update.
ffdlconfig = ConfigParser.SafeConfigParser()
ffdlconfig.readfp(StringIO(get_resources("defaults.ini")))
ffdlconfig.readfp(StringIO(prefs['personal.ini']))
adapter = adapters.getAdapter(ffdlconfig,url)
try:
adapter.getStoryMetadataOnly()
except exceptions.FailedToLogin:
print("Login Failed, Need Username/Password.")
userpass = UserPassDialog(self.gui,url)
userpass.exec_() # exec_ will make it act modal
if userpass.status:
adapter.username = userpass.user.text()
adapter.password = userpass.passwd.text()
# else:
# del adapter
# return
except exceptions.AdultCheckRequired:
if question_dialog(self.gui, 'Are You Adult?', '<p>'+
"%s requires that you be an adult. Please confirm you are an adult in your locale:"%url,
show_copy_button=False):
adapter.is_adult=True
# else:
# del adapter
# return
# let exceptions percolate up.
story = adapter.getStoryMetadataOnly()
if collision != ADDNEW:
mi = MetaInformation(story.getMetadata("title"),
(story.getMetadata("author"),)) # author is a list.
identicalbooks = self.db.find_identical_books(mi)
print(identicalbooks)
## more than one match will need to be handled differently.
if identicalbooks:
book_id = identicalbooks.pop()
if collision == SKIP:
raise NotGoingToDownload("Skipping duplicate story.")
# print("Attempting to add to list.")
# rl_plugin = self.gui.iactions['Reading List']
# rl_plugin.remove_books_from_list('Send',[book_id],refresh_screen=False)
# rl_plugin.add_books_to_list('Send',[book_id],refresh_screen=False)
#rl_plugin.view_list('Send')
if collision == OVERWRITE and len(identicalbooks) > 1:
raise NotGoingToDownload("More than one identical books--can't tell which to overwrite.")
if collision == OVERWRITE and \
onlyoverwriteifnewer and \
self.db.has_format(book_id,fileform,index_is_id=True):
# check make sure incoming is newer.
lastupdated=story.getMetadataRaw('dateUpdated').date()
fileupdated=datetime.fromtimestamp(os.stat(self.db.format_abspath(book_id, fileform, index_is_id=True))[8]).date()
if fileupdated > lastupdated:
raise NotGoingToDownload("Not Overwriting, story is not newer.")
if collision == UPDATE:
if fileform != 'epub':
raise NotGoingToDownload("Not updating non-epub format.")
# 'book' can exist without epub. If there's no existing epub,
# let it go and it will download it.
if self.db.has_format(book_id,fileform,index_is_id=True):
toupdateio = StringIO()
(epuburl,chaptercount) = doMerge(toupdateio,
[StringIO(self.db.format(book_id,'EPUB',
index_is_id=True))],
titlenavpoints=False,
striptitletoc=True,
forceunique=False)
urlchaptercount = int(story.getMetadata('numChapters'))
if chaptercount == urlchaptercount: # and not onlyoverwriteifnewer:
raise NotGoingToDownload("%s already contains %d chapters." % (url,chaptercount))
elif chaptercount > urlchaptercount:
raise NotGoingToDownload("%s contains %d chapters, more than epub." % (url,chaptercount))
else:
print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount))
else: # not identicalbooks
if collision == CALIBREONLY:
raise NotGoingToDownload("Not updating Calibre Metadata, no existing book to update.")
return adapter
def download_list(self,adaptertuple_list,fileform,
collision=ADDNEW,
updatemeta=True,
onlyoverwriteifnewer=True):
'''
Called by MetadataProgressDialog to start story downloads BG processing.
adapter_list is a list of tuples of (url,adapter)
'''
print("download_list")
job = ThreadedJob('FanFictionDownload',
'Downloading FanFiction Stories',
func=self.do_story_downloads,
args=(adaptertuple_list, fileform, self.db),
kwargs={'collision':collision,'updatemeta':updatemeta,
'onlyoverwriteifnewer':onlyoverwriteifnewer},
callback=self._get_stories_completed)
self.gui.job_manager.run_threaded_job(job)
self.gui.status_bar.show_message('Downloading %d stories'%len(adaptertuple_list))
def _get_stories_completed(self, job):
print("_get_stories_completed")
def do_story_downloads(self, adaptertuple_list, fileform, db,
**kwargs):
'''
Master job, loop to download this list of stories
'''
print("do_story_downloads")
abort = kwargs['abort']
notifications=kwargs['notifications']
log = kwargs['log']
notifications.put((0.01, 'Start Downloading Stories'))
count = 0.01
total = len(adaptertuple_list)
# Queue all the jobs
for (url,adapter) in adaptertuple_list:
if abort.is_set():
notifications.put(1.0,'Aborting...')
return
notifications.put((float(count)/total,
'Downloading %s'%adapter.getStoryMetadataOnly().getMetadata("title")))
log.prints(log.INFO,'Downloading %s'%adapter.getStoryMetadataOnly().getMetadata("title"))
try:
self.do_story_download(adapter,fileform,db,
kwargs['collision'],kwargs['updatemeta'],kwargs['onlyoverwriteifnewer'])
except Exception as e:
log.prints(log.ERROR,'Failed Downloading %s: %s'%
(adapter.getStoryMetadataOnly().getMetadata("title"),e))
count = count + 1
return
def do_story_download(self,adapter,fileform,db,collision,
updatemeta,onlyoverwriteifnewer):
print("do_story_download")
story = adapter.getStoryMetadataOnly()
mi = MetaInformation(story.getMetadata("title"),
(story.getMetadata("author"),)) # author is a list.
writer = writers.getWriter(fileform,adapter.config,adapter)
tmp = PersistentTemporaryFile("."+fileform)
titleauth = "%s by %s"%(story.getMetadata("title"), story.getMetadata("author"))
url = story.getMetadata("storyUrl")
print(titleauth)
print("tmp: "+tmp.name)
mi.set_identifiers({'url':story.getMetadata("storyUrl")})
mi.publisher = story.getMetadata("site")
mi.tags = writer.getTags()
mi.languages = ['en']
mi.pubdate = story.getMetadataRaw('datePublished')
mi.timestamp = story.getMetadataRaw('dateCreated')
mi.comments = story.getMetadata("description")
identicalbooks = self.db.find_identical_books(mi)
#print(identicalbooks)
addedcount=0
if identicalbooks and collision != ADDNEW:
## more than one match? add to first off the list.
## Shouldn't happen--we checked above.
book_id = identicalbooks.pop()
if collision == UPDATE:
if self.db.has_format(book_id,fileform,index_is_id=True):
urlchaptercount = int(story.getMetadata('numChapters'))
## First, get existing epub with titlepage and tocpage stripped.
updateio = StringIO()
(epuburl,chaptercount) = doMerge(updateio,
[StringIO(self.db.format(book_id,'EPUB',
index_is_id=True))],
titlenavpoints=False,
striptitletoc=True,
forceunique=False)
print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount))
## Get updated title page/metadata by itself in an epub.
## Even if the title page isn't included, this carries the metadata.
titleio = StringIO()
writer.writeStory(outstream=titleio,metaonly=True)
newchaptersio = None
if urlchaptercount > chaptercount :
## Go get the new chapters only in another epub.
newchaptersio = StringIO()
adapter.setChaptersRange(chaptercount+1,urlchaptercount)
adapter.config.set("overrides",'include_tocpage','false')
adapter.config.set("overrides",'include_titlepage','false')
writer.writeStory(outstream=newchaptersio)
## Merge the three epubs together.
doMerge(tmp,
[titleio,updateio,newchaptersio],
fromfirst=True,
titlenavpoints=False,
striptitletoc=False,
forceunique=False)
else: # update, but there's no epub extant, so do overwrite.
collision = OVERWRITE
if collision == OVERWRITE:
writer.writeStory(tmp)
db.add_format_with_hooks(book_id, fileform, tmp, index_is_id=True)
# get all formats.
if prefs['deleteotherforms'] and collision in (OVERWRITE, UPDATE):
fmts = set([x.lower() for x in db.formats(book_id, index_is_id=True).split(',')])
for fmt in fmts:
if fmt != fileform:
print("remove f:"+fmt)
db.remove_format(book_id, fmt,index_is_id=True)#, notify=False
if updatemeta or collision == CALIBREONLY:
db.set_metadata(book_id,mi)
else: # no matching, adding new.
writer.writeStory(tmp)
(notadded,addedcount)=db.add_books([tmp],[fileform],[mi], add_duplicates=True)
# Otherwise list of books doesn't update right away.
if addedcount:
self.gui.library_view.model().books_added(addedcount)
self.gui.library_view.model().refresh()
del adapter
del writer
def f(x):
if x.strip(): return True
else: return False
def get_url_list(urls):
return filter(f,urls.strip().splitlines())
Binary file not shown.

After

Width:  |  Height:  |  Size: 23 KiB

Binary file not shown.
Whitespace-only changes.
+4 -4
View File
@@ -105,6 +105,10 @@ zip_filename: ${title}-${siteabbrev}_${storyId}${formatext}.zip
## zip_filename.
allow_unsafe_filename: false
## entries to make epub subjects and calibre tags
## lastupdate creates two tags: "Last Update Year/Month: %Y/%m" and "Last Update: %Y/%m/%d"
include_subject_tags: extratags, genre, category, characters, lastupdate, status
## extra tags (comma separated) to include, primarily for epub.
extratags: FanFiction
@@ -137,10 +141,6 @@ windows_eol: true
## epub is already a zip file.
zip_output: false
## entries to make epub subject tags
## lastupdate creates two tags: "Last Update Year/Month: %Y/%m" and "Last Update: %Y/%m/%d"
include_subject_tags: extratags, genre, category, characters, lastupdate, status
## epub carries the TOC in metadata.
## mobi generated from epub will have a TOC at the end.
include_tocpage: false
+36 -18
View File
@@ -15,21 +15,52 @@
# limitations under the License.
#
import os, re, sys, glob
import os, re, sys, glob, types
from os.path import dirname, basename, normpath
import logging
import urlparse as up
import fanficdownloader.exceptions as exceptions
from .. import exceptions as exceptions
## must import each adapter here.
import adapter_test1
import adapter_fanfictionnet
import adapter_fanficcastletvnet
import adapter_fanfictionnet
import adapter_fictionalleyorg
import adapter_fictionpresscom
import adapter_ficwadcom
import adapter_fimfictionnet
import adapter_harrypotterfanfictioncom
import adapter_mediaminerorg
import adapter_potionsandsnitchesnet
import adapter_tenhawkpresentscom
import adapter_adastrafanficcom
import adapter_thewriterscoffeeshopcom
import adapter_tthfanficorg
import adapter_twilightednet
import adapter_twiwritenet
import adapter_whoficcom
import adapter_siyecouk
## This bit of complexity allows adapters to be added by just adding
## the source file. It eliminates the long if/else clauses we used to
## need to pick out the adapter.
## importing. It eliminates the long if/else clauses we used to need
## to pick out the adapter.
## List of registered site adapters.
__class_list = []
def imports():
for name, val in globals().items():
if isinstance(val, types.ModuleType):
yield val.__name__
for x in imports():
if "fanficdownloader.adapters.adapter_" in x:
#print x
__class_list.append(sys.modules[x].getClass())
def getAdapter(config,url):
## fix up leading protocol.
fixedurl = re.sub(r"(?i)^[htp]+[:/]+","http://",url.strip())
@@ -63,16 +94,3 @@ def getClassFor(domain):
for cls in __class_list:
if cls.matchesSite(domain):
return cls
## Automatically import each adapter_*.py file.
## Each implement getClass() to their class
filelist = glob.glob(dirname(__file__)+'/adapter_*.py')
sys.path.insert(0,normpath(dirname(__file__)))
for file in filelist:
#print "file: "+basename(file)[:-3]
module = __import__(basename(file)[:-3])
__class_list.append(module.getClass())
del sys.path[0]
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -20,9 +20,9 @@ import logging
import re
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,8 +21,8 @@ import re
import urllib2
import time
import fanficdownloader.BeautifulSoup as bs
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -22,9 +22,9 @@ import urllib2
import time
import httplib, urllib
import fanficdownloader.BeautifulSoup as bs
import fanficdownloader.exceptions as exceptions
from fanficdownloader.htmlcleanup import stripHTML
from .. import BeautifulSoup as bs
from .. import exceptions as exceptions
from ..htmlcleanup import stripHTML
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib2
import cookielib as cl
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -88,8 +88,7 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
soup = bs.BeautifulSoup(data).find("div", {"class":"content_box post_content_box"})
title = soup.find("h2").find("a").text # first a link in first h2 is title.
author = soup.find("h2").find("span",{'class':'author'}).find("a").text
title, author = [link.text for link in soup.find("h2").findAll("a")]
self.story.setMetadata("title", title)
self.story.setMetadata("author", author)
self.story.setMetadata("authorId", author) # The author's name will be unique
@@ -159,4 +158,4 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
if soup == None:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return utf8FromSoup(soup)
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -0,0 +1,275 @@
# -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
import re
import urllib2
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
# This function is called by the downloader in all adapter_*.py files
# in this dir to register the adapter class. So it needs to be
# updated to reflect the class below it. That, plus getSiteDomain()
# take care of 'Registering'.
def getClass():
return SiyeCoUkAdapter # XXX
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class SiyeCoUkAdapter(BaseSiteAdapter): # XXX
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["Windows-1252",
"utf8",]# 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
# self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
# self.password = ""
# self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
logging.debug("storyId: (%s)"%self.story.getMetadata('storyId'))
# normalized story URL.
self._setURL('http://' + self.getSiteDomain() + '/siye/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','siye') # XXX
# If all stories from the site fall into the same category,
# the site itself isn't likely to label them as such, so we
# do.
self.story.addToList("category","Harry Potter") # XXX
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%Y.%m.%d" # XXX
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'www.siye.co.uk' # XXX
@classmethod
def getAcceptDomains(cls):
return ['www.siye.co.uk','siye.co.uk']
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/siye/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+r"(www\.)?"+re.escape("siye.co.uk/siye/viewstory.php?sid=")+r"\d+$"
# ## Login seems to be reasonably standard across eFiction sites.
# def needToLoginCheck(self, data):
# if 'Registered Users Only' in data \
# or 'There is no such account on our website' in data \
# or "That password doesn't match the one in our database" in data:
# return True
# else:
# return False
# def performLogin(self, url):
# params = {}
# if self.password:
# params['penname'] = self.username
# params['password'] = self.password
# else:
# params['penname'] = self.getConfig("username")
# params['password'] = self.getConfig("password")
# params['cookiecheck'] = '1'
# params['submit'] = 'Submit'
# loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login'
# logging.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
# params['penname']))
# d = self._fetchUrl(loginUrl, params)
# if "Member Account" not in d : #Member Account
# logging.info("Failed to login to URL %s as %s" % (loginUrl,
# params['penname']))
# raise exceptions.FailedToLogin(url,params['penname'])
# return False
# else:
# return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
# if self.is_adult or self.getConfig("is_adult"):
# # Weirdly, different sites use different warning numbers.
# # If the title search below fails, there's a good chance
# # you need a different number. print data at that point
# # and see what the 'click here to continue' url says.
# addurl = "&ageconsent=ok&warning=4" # XXX
# else:
# addurl=""
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
# Except it doesn't this time. :-/
url = self.url #+'&index=1'+addurl
logging.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
# if self.needToLoginCheck(data):
# # need to log in for this one.
# self.performLogin(url)
# data = self._fetchUrl(url)
# # The actual text that is used to announce you need to be an
# # adult varies from site to site. Again, print data before
# # the title search to troubleshoot.
# if "Age Consent Required" in data: # XXX
# raise exceptions.AdultCheckRequired(self.url)
# if "Access denied. This story has not been validated by the adminstrators of this site." in data:
# raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = bs.BeautifulSoup(data)
# print data
# Now go hunting for all the meta data and the chapter list.
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+"))
self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/siye/'+a['href'])
self.story.setMetadata('author',a.string)
# need(or easier) to pull other metadata from the author's list page.
authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl')))
## Title
titlea = authsoup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',titlea.string)
# Find the chapters (from soup, not authsoup):
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/siye/'+chapter['href']))
if self.chapterUrls:
self.story.setMetadata('numChapters',len(self.chapterUrls))
else:
self.chapterUrls.append((self.story.getMetadata('title'),url))
self.story.setMetadata('numChapters',1)
# The stuff we can get from the chapter list/one-shot page are
# in the first table with 95% width.
metatable = soup.find('table',{'width':'95%'})
# Categories
cat_as = metatable.findAll('a', href=re.compile(r'categories.php'))
for cat_a in cat_as:
self.story.addToList('category',stripHTML(cat_a))
moremetaparts = stripHTML(metatable).split('\n')
for part in moremetaparts:
part = part.strip()
if part.startswith("Characters:"):
part = part[part.find(':')+1:]
for item in part.split(','):
if item.strip() != "None":
self.story.addToList('characters',item)
if part.startswith("Genres:"):
part = part[part.find(':')+1:]
for item in part.split(','):
if item.strip() != "None":
self.story.addToList('genre',item)
if part.startswith("Warnings:"):
part = part[part.find(':')+1:]
for item in part.split(','):
if item.strip() != "None":
self.story.addToList('warnings',item)
if part.startswith("Rating:"):
part = part[part.find(':')+1:]
self.story.setMetadata('rating',part)
if part.startswith("Summary:"):
part = part[part.find(':')+1:]
self.story.setMetadata('description',part)
# want to get the next tr of the table.
#print("%s"%titlea.parent.parent.findNextSibling('tr'))
# eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly.
moremeta = stripHTML(titlea.parent.parent.findNextSibling('tr'))
for part in moremeta.replace(' - ','\n').split('\n'):
#print("part:%s"%part)
try:
(name,value) = part.split(': ')
except:
# not going to worry about fancier processing for the bits
# that don't match.
continue
name=name.strip()
value=value.strip()
if name == 'Published':
self.story.setMetadata('datePublished', makeDate(value, self.dateformat))
if name == 'Updated':
self.story.setMetadata('dateUpdated', makeDate(value, self.dateformat))
if name == 'Completed':
if value == 'Yes':
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
if name == 'Words':
self.story.setMetadata('numWords', value)
# grab the text for an individual chapter.
def getChapterText(self, url):
logging.debug('Getting chapter text from: %s' % url)
soup = bs.BeautifulSoup(self._fetchUrl(url))
# not the most unique thing in the work, bit it appears to be
# the best we can do there.
story = soup.find('span', {'style' : 'font-size: 100%;'})
if None == story:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return utf8FromSoup(story)
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
+17 -8
View File
@@ -19,8 +19,8 @@ import datetime
import time
import logging
import fanficdownloader.BeautifulSoup as bs
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from .. import exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -54,6 +54,12 @@ class TestSiteAdapter(BaseSiteAdapter):
if self.story.getMetadata('storyId') == '666':
raise exceptions.StoryDoesNotExist(self.url)
if self.story.getMetadata('storyId').startswith('670'):
time.sleep(1.0)
if self.story.getMetadata('storyId').startswith('671'):
time.sleep(1.0)
if self.getConfig("username"):
self.username = self.getConfig("username")
@@ -63,18 +69,18 @@ class TestSiteAdapter(BaseSiteAdapter):
if self.story.getMetadata('storyId') == '664':
self.story.setMetadata(u'title',"Test Story Title "+self.crazystring)
else:
self.story.setMetadata(u'title',"Test Story Title")
self.story.setMetadata(u'title',"Test Story Title "+self.story.getMetadata('storyId'))
self.story.setMetadata('storyUrl',self.url)
self.story.setMetadata('description',u'Description '+self.crazystring+u''' Done
Some more longer description. "I suck at summaries!" "Better than it sounds!" "My first fic"
''')
self.story.setMetadata('datePublished',makeDate("1972-01-31","%Y-%m-%d"))
self.story.setMetadata('datePublished',makeDate("1975-03-15","%Y-%m-%d"))
self.story.setMetadata('dateCreated',datetime.datetime.now())
if self.story.getMetadata('storyId') == '669':
self.story.setMetadata('dateUpdated',datetime.datetime.now())
else:
self.story.setMetadata('dateUpdated',makeDate("1975-01-31","%Y-%m-%d"))
self.story.setMetadata('dateUpdated',makeDate("1975-04-15","%Y-%m-%d"))
self.story.setMetadata('numWords','123456')
self.story.setMetadata('status','In-Completed')
self.story.setMetadata('rating','Tweenie')
@@ -100,7 +106,7 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
('Chapter 3, Over Cinnabar',self.url+"&chapter=4"),
('Chapter 4',self.url+"&chapter=5"),
('Chapter 5',self.url+"&chapter=6"),
# ('Chapter 6',self.url+"&chapter=6"),
('Chapter 6',self.url+"&chapter=6"),
# ('Chapter 7',self.url+"&chapter=6"),
# ('Chapter 8',self.url+"&chapter=6"),
# ('Chapter 9',self.url+"&chapter=6"),
@@ -128,8 +134,9 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
if self.story.getMetadata('storyId') == '667':
raise exceptions.FailedToDownload("Error downloading Chapter: %s!" % url)
if self.story.getMetadata('storyId') == '670':
time.sleep(2.0)
if self.story.getMetadata('storyId').startswith('670') or \
self.story.getMetadata('storyId').startswith('672'):
time.sleep(1.0)
if "chapter=1" in url :
text=u'''
@@ -143,6 +150,8 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
<p>http://test1.com?sid=668 - raises FailedToLogin unless username='Me'</p>
<p>http://test1.com?sid=669 - Succeeds with Updated Date=now</p>
<p>http://test1.com?sid=670 - Succeeds, but sleeps 2sec on each chapter</p>
<p>http://test1.com?sid=671 - Succeeds, but sleeps 2sec metadata only</p>
<p>http://test1.com?sid=672 - Succeeds, quick meta, sleeps 2sec chapters only</p>
<p>And other storyId will succeed with the same output.</p>
</div>
'''
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib2
import time
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -34,6 +34,8 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
self.story.setMetadata('siteabbrev','tth')
self.dateformat = "%d %b %y"
self.is_adult=False
self.username = None
self.password = None
# get storyId from url--url validation guarantees query correct
m = re.match(self.getSiteURLPattern(),url)
if m:
@@ -60,6 +62,48 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
def getSiteURLPattern(self):
return r"http://www.tthfanfic.org/(T-\d+/)?Story-(?P<id>\d+)(-\d+)?(/.*)?$"
# tth won't send you future updates if you aren't 'caught up'
# on the story. Login isn't required for F21, but logging in will
# mark stories you've downloaded as 'read' on tth.
def performLogin(self):
params = {}
if self.password:
params['urealname'] = self.username
params['password'] = self.password
else:
params['urealname'] = self.getConfig("username")
params['password'] = self.getConfig("password")
params['loginsubmit'] = 'Login'
if not params['password']:
return
loginUrl = 'http://' + self.getSiteDomain() + '/login.php'
logging.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['urealname']))
## need to pull empty login page first to get ctkn and
## password name, which are BUSs
# <form method='post' action='/login.php' accept-charset="utf-8">
# <input type='hidden' name='ctkn' value='4bdf761f5bea06bf4477072afcbd0f8d721d1a4f989c09945a9e87afb7a66de1'/>
# <input type='text' id='urealname' name='urealname' value=''/>
# <input type='password' id='password' name='6bb3fcd148d148629223690bf19733b8'/>
# <input type='submit' value='Login' name='loginsubmit'/>
soup = bs.BeautifulSoup(self._fetchUrl(loginUrl))
params['ctkn']=soup.find('input', {'name':'ctkn'})['value']
params[soup.find('input', {'id':'password'})['name']] = params['password']
d = self._fetchUrl(loginUrl, params)
if "Stories Published" not in d : #Member Account
logging.info("Failed to login to URL %s as %s" % (loginUrl,
params['penname']))
raise exceptions.FailedToLogin(url,params['penname'])
return False
else:
return True
def extractChapterUrlsAndMetadata(self):
# fetch the chapter. From that we will get almost all the
# metadata and chapter list
@@ -67,6 +111,11 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
url=self.url
logging.debug("URL: "+url)
# tth won't send you future updates if you aren't 'caught up'
# on the story. Login isn't required for F21, but logging in will
# mark stories you've downloaded as 'read' on tth.
self.performLogin()
# use BeautifulSoup HTML parser to make everything easier to find.
try:
data = self._fetchUrl(url)
@@ -125,10 +174,10 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
BtVS = True
for cat in verticaltable.findAll('a', href=re.compile(r"^/Category-")):
if cat.string not in ['General', 'Non-BtVS/AtS Stories', 'BtVS/AtS Non-Crossover']:
if cat.string not in ['General', 'Non-BtVS/AtS Stories', 'BtVS/AtS Non-Crossover', 'Non-BtVS Crossovers']:
self.story.addToList('category',cat.string)
else:
if cat.string == 'Non-BtVS/AtS Stories':
if 'Non-BtVS' in cat.string:
BtVS = False
if BtVS:
self.story.addToList('category','Buffy: The Vampire Slayer')
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -21,9 +21,9 @@ import re
import urllib
import urllib2
import fanficdownloader.BeautifulSoup as bs
from fanficdownloader.htmlcleanup import stripHTML
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
@@ -20,8 +20,8 @@ import logging
import re
import urllib2
import fanficdownloader.BeautifulSoup as bs
import fanficdownloader.exceptions as exceptions
from .. import BeautifulSoup as bs
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, utf8FromSoup, makeDate
+6 -5
View File
@@ -39,13 +39,13 @@ except:
pass
#logging.info("Hook to make default deadline 10.0 NOT installed--not using appengine")
from fanficdownloader.story import Story
from fanficdownloader.configurable import Configurable
from fanficdownloader.htmlcleanup import removeEntities, removeAllEntities, stripHTML
from fanficdownloader.exceptions import InvalidStoryURL
from ..story import Story
from ..configurable import Configurable
from ..htmlcleanup import removeEntities, removeAllEntities, stripHTML
from ..exceptions import InvalidStoryURL
try:
import fanficdownloader.chardet as chardet
from .. import chardet as chardet
except ImportError:
chardet = None
@@ -63,6 +63,7 @@ class BaseSiteAdapter(Configurable):
return re.match(self.getSiteURLPattern(), self.url)
def __init__(self, config, url):
self.config = config
Configurable.__init__(self, config)
self.addConfigSection(self.getSiteDomain())
self.addConfigSection("overrides")
+2 -1
View File
@@ -53,11 +53,12 @@ class Story:
def addToList(self,listname,value):
if value==None:
return
value = conditionalRemoveEntities(value)
if not self.listables.has_key(listname):
self.listables[listname]=[]
# prevent duplicates.
if not value in self.listables[listname]:
self.listables[listname].append(conditionalRemoveEntities(value))
self.listables[listname].append(value)
def getList(self,listname):
if not self.listables.has_key(listname):
+1 -1
View File
@@ -18,7 +18,7 @@
## This could (should?) use a dynamic loader like adapters, but for
## now, it's static, since there's so few of them.
from fanficdownloader.exceptions import FailedToDownload
from ..exceptions import FailedToDownload
from writer_html import HTMLWriter
from writer_txt import TextWriter
+24 -2
View File
@@ -24,8 +24,8 @@ import zipfile
from zipfile import ZipFile, ZIP_DEFLATED
import logging
from fanficdownloader.configurable import Configurable
from fanficdownloader.htmlcleanup import removeEntities, removeAllEntities, stripHTML
from ..configurable import Configurable
from ..htmlcleanup import removeEntities, removeAllEntities, stripHTML
class BaseStoryWriter(Configurable):
@@ -102,6 +102,9 @@ class BaseStoryWriter(Configurable):
self.story.setMetadata('formatname',self.getFormatName())
self.story.setMetadata('formatext',self.getFormatExt())
def getMetadata(self,key):
return stripHTML(self.story.getMetadata(key))
def getOutputFileName(self):
if self.getConfig('zip_output'):
return self.getZipFileName()
@@ -242,6 +245,25 @@ class BaseStoryWriter(Configurable):
if close:
outstream.close()
def getTags(self):
# set to avoid duplicates subject tags.
subjectset = set()
for entry in self.validEntries:
if entry in self.getConfigList("include_subject_tags") and \
entry not in self.story.getLists() and \
self.story.getMetadata(entry):
subjectset.add(self.getMetadata(entry))
# listables all go into dc:subject tags, but only if they are configured.
for (name,lst) in self.story.getLists().iteritems():
if name in self.getConfigList("include_subject_tags"):
for tag in lst:
subjectset.add(tag)
for tag in self.getConfigList("extratags"):
subjectset.add(tag)
return subjectset
def writeStoryImpl(self, out):
"Must be overriden by sub classes."
pass
+2 -17
View File
@@ -26,7 +26,7 @@ from zipfile import ZipFile, ZIP_STORED, ZIP_DEFLATED
from xml.dom.minidom import parse, parseString, getDOMImplementation
from base_writer import *
from fanficdownloader.htmlcleanup import stripHTML
from ..htmlcleanup import stripHTML
class EpubWriter(BaseStoryWriter):
@@ -156,9 +156,6 @@ h6 { text-align: center; }
</html>
''')
def getMetadata(self,key):
return stripHTML(self.story.getMetadata(key))
def writeStoryImpl(self, out):
## Python 2.5 ZipFile is rather more primative than later
@@ -260,19 +257,7 @@ h6 { text-align: center; }
metadata.appendChild(newTag(contentdom,"dc:description",text=
self.getMetadata('description')))
# set to avoid duplicates subject tags.
subjectset = set()
for entry in self.validEntries:
if entry in self.getConfigList("include_subject_tags") and \
entry not in self.story.getLists() and \
self.story.getMetadata(entry):
subjectset.add(self.getMetadata(entry))
# listables all go into dc:subject tags, but only if they are configured.
for (name,lst) in self.story.getLists().iteritems():
if name in self.getConfigList("include_subject_tags"):
for tag in lst:
subjectset.add(tag)
for subject in subjectset:
for subject in self.getTags():
metadata.appendChild(newTag(contentdom,"dc:subject",text=subject))
+2 -5
View File
@@ -20,8 +20,8 @@ import string
import StringIO
from base_writer import *
from fanficdownloader.htmlcleanup import stripHTML
from fanficdownloader.mobi import Converter
from ..htmlcleanup import stripHTML
from ..mobi import Converter
class MobiWriter(BaseStoryWriter):
@@ -124,9 +124,6 @@ class MobiWriter(BaseStoryWriter):
</html>
''')
def getMetadata(self,key):
return stripHTML(self.story.getMetadata(key))
def writeStoryImpl(self, out):
files = []
+1 -1
View File
@@ -21,7 +21,7 @@ from textwrap import wrap
from base_writer import *
from fanficdownloader.html2text import html2text, BODY_WIDTH
from ..html2text import html2text, BODY_WIDTH
## In BaseStoryWriter, we define _write to encode <unicode> objects
## back into <string> for true output. But txt needs to write the
+7 -8
View File
@@ -61,14 +61,8 @@
considers Python 2.7 Experimental still, so there may be issues.
</p>
<p>
<b>Good news!</b><br />
The issue that was causing problems with downloading large stories
has been fixed.
</p>
<p>
<b>New Feature</b><br /> You can now set a custom
parameter for background_color that will be used with html
and epub output. (Note: many epub readers ignore the bg color.)
<b>New Site</b><br />
Now supporting www.siye.co.uk.
</p>
<p>
If you have any problems with this application, please
@@ -215,6 +209,11 @@
<br /><a href="http://www.tthfanfic.org/Story-5583/Greywizard+Marked+By+Kane.htm">http://www.tthfanfic.org/Story-5583/Greywizard+Marked+By+Kane.htm</a>.
<br /><a href="http://www.tthfanfic.org/T-99999999/Story-26448-15/batzulger+Willow+Rosenberg+and+the+Mind+Riders.htm">http://www.tthfanfic.org/T-99999999/Story-26448-15/batzulger+Willow+Rosenberg+and+the+Mind+Riders.htm</a>.
</dd>
<dt>www.siye.co.uk</dt>
<dd>
Use the URL of the story's chapter list, such as
<br /><a href="http://www.siye.co.uk/siye/viewstory.php?sid=123">http://www.siye.co.uk/siye/viewstory.php?sid=123</a>.
</dd>
</dl>
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/python
# -*- coding: utf-8 -*-
# epubmerge.py 1.0
# Copyright 2011, Jim Miller
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
# http://www.apache.org/licenses/LICENSE-2.0
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os
from glob import glob
from makezip import createZipFile
if __name__=="__main__":
filename="FanFictionDownLoaderPlugin.zip"
exclude=['*.pyc','*~','*.xcf']
# from top dir. 'w' for overwrite
createZipFile(filename,"w",
['defaults.ini','example.ini','epubmerge.py','fanficdownloader'],
exclude=exclude)
#from calibre-plugin dir. 'a' for append
os.chdir('calibre-plugin')
files=['about.txt','images',]
files.extend(glob('*.py'))
files.extend(glob('plugin-import-name-*.txt'))
createZipFile("../"+filename,"a",
files,exclude=exclude)
+54
View File
@@ -0,0 +1,54 @@
#!/usr/bin/python
# -*- coding: utf-8 -*-
# epubmerge.py 1.0
# Copyright 2011, Jim Miller
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
# http://www.apache.org/licenses/LICENSE-2.0
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import os, zipfile, sys
from glob import glob
def addFolderToZip(myZipFile,folder,exclude=[]):
folder = folder.encode('ascii') #convert path to ascii for ZipFile Method
excludelist=[]
for ex in exclude:
excludelist.extend(glob(folder+"/"+ex))
for file in glob(folder+"/*"):
if file in excludelist:
continue
if os.path.isfile(file):
#print file
myZipFile.write(file, file, zipfile.ZIP_DEFLATED)
elif os.path.isdir(file):
addFolderToZip(myZipFile,file,exclude=exclude)
def createZipFile(filename,mode,files,exclude=[]):
myZipFile = zipfile.ZipFile( filename, mode ) # Open the zip file for writing
excludelist=[]
for ex in exclude:
excludelist.extend(glob(ex))
for file in files:
if file in excludelist:
continue
file = file.encode('ascii') #convert path to ascii for ZipFile Method
if os.path.isfile(file):
(filepath, filename) = os.path.split(file)
#print file
myZipFile.write( file, filename, zipfile.ZIP_DEFLATED )
if os.path.isdir(file):
addFolderToZip(myZipFile,file,exclude=exclude)
myZipFile.close()
return (1,filename)