Compare commits

...
126 Commits
Author SHA1 Message Date
M Clark a0b4332da8 better description selector 2016-10-28 15:07:25 +08:00
M Clark a1bd9c8379 avoid stripping html from description 2016-10-27 12:18:53 +08:00
Jim Miller e02c969371 Bump test version. 2016-10-26 11:02:35 -05:00
Jim Miller b18b2177b0 Merge pull request #146 from wassname/master
fix list index out of range error
2016-10-26 11:01:40 -05:00
gitea 5ca33837d4 fix list index out of range error
- fixes "Failed List Index Out of Range" on url
"http://royalroadl.com/fiction/1612"
2016-10-26 14:22:10 +08:00
Jim Miller eade66f513 Bump test version. 2016-10-25 19:38:08 -05:00
Jim Miller 28c4557d22 Allow '_u#.xhtml' file names in updates. For Calibre Convert on Anthologies, then manually split. 2016-10-25 19:37:27 -05:00
Jim Miller d6eda82767 Change royalroadl rating to stars to match usual rating definition. 2016-10-24 12:17:58 -05:00
Jim Miller 28eff8ac12 Update inis and bump test version for royalroadl.com 2016-10-24 12:07:55 -05:00
Jim Miller 7ae40e539d Merge pull request #144 from wassname/master
Add royalroad adapter,  Thanks @wassname
2016-10-24 12:03:53 -05:00
gitea ab6436ca0b add royalroad adapter 2016-10-24 12:35:22 +08:00
Jim Miller b5a04b0f97 Remove reviews count from wraithbait.com, no longer displayed. 2016-10-22 12:44:08 -05:00
Jim Miller c7bbb765b2 Change default for post_process_apply_filename_safepattern to false. 2016-10-20 14:28:12 -05:00
Jim Miller 8a84043d29 Allow --infile missing last newline. 2016-10-20 14:13:09 -05:00
Jim Miller f115d68e52 Add post_process_apply_filename_safepattern setting (default:true). 2016-10-20 14:03:04 -05:00
Jim Miller e2707f5459 Bump versions to 2.5.0. 2016-10-18 10:47:00 -05:00
Jim Miller aa85efd6d9 Update translations. 2016-10-18 10:45:32 -05:00
Jim Miller 9e2c0a3563 Correct a typo in an ini comment. 2016-10-18 10:44:17 -05:00
Jim Miller d4f3fee053 Fix ini comments for internalize_text_links 2016-10-06 23:50:09 -05:00
Jim Miller e87c7b7009 Bump micro version for test version. 2016-10-06 23:23:48 -05:00
Jim Miller 5b6228166c Update translations. 2016-10-06 23:23:32 -05:00
Jim Miller f91092d9d8 Add internalize_text_links option for epub and html formats. Also fix for growing whitespace in epub updates and missing always_reload_first_chapter in highlight list. 2016-10-06 23:21:01 -05:00
Jim Miller a40383bada Adding normalize_chapterurl() for xenforoforum and normalize_text_links option. 2016-10-06 20:56:37 -05:00
Jim Miller c9205dd6bc Tiny fix to html output--close TOCTOP <a> tag. 2016-10-06 20:22:44 -05:00
Jim Miller a0acbb8893 Fix for CLI not working correctly with -u epubfile 2016-10-05 17:46:56 -05:00
Jim Miller 7d66d93b70 Adding words_added metadata for epub logpage only. 2016-10-05 17:33:26 -05:00
Jim Miller 277b1ef92d Fix for unusual (changed?) chapter URLs on storiesonline.net. 2016-10-02 12:57:02 -05:00
Jim Miller 72217423c7 Fix CLI post_process_cmd to use processed metadata instead of raw. 2016-10-01 12:28:53 -05:00
Jim Miller 99e4ea9a59 Bump CLI only to 2.4.3. 2016-09-30 13:17:31 -05:00
Jim Miller f1bb729b33 Fix CLI --infile to work properly. 2016-09-30 13:16:16 -05:00
Jim Miller 26fe5f42ef Bump CLI only to 2.4.2 due cli.py also having version. 2016-09-30 12:09:15 -05:00
Jim Miller 5f8e75d4c1 Bump CLI only to 2.4.1 due to pypi won't allow re-upload of 2.4.0 properly. 2016-09-30 12:08:01 -05:00
Jim Miller 6dcfad0832 Fix fictionhunt.com example URLs. 2016-09-30 11:31:22 -05:00
Jim Miller b11e06e697 Bump versions--start bumping minor per version instead of micro. 2016-09-30 11:17:20 -05:00
Jim Miller 63515d1366 New version updating tool. 2016-09-30 11:16:54 -05:00
Jim Miller cfcba87c32 Update translations. 2016-09-30 11:15:04 -05:00
Jim Miller 709da65ea2 New site: fictionhunt.com 2016-09-21 19:59:08 -05:00
Jim Miller a21b555255 New site: fictionhunt.com 2016-09-21 19:07:50 -05:00
Jim Miller f25b46d268 Clean up file a little. 2016-09-21 19:04:22 -05:00
Jim Miller 6e2635a110 Need to provide 'more info' on failure with individual URL in CLI. 2016-09-21 17:50:17 -05:00
Jim Miller ab44185cfa Fix for webservice over-long error msgs causing problems, specific errmsg for unknown site. 2016-09-21 16:55:38 -05:00
Jim Miller c21ad04a9c Fix for calibre series w/injectseries always being applied when doing BG metadata. 2016-09-20 11:02:47 -05:00
Jim Miller 314baf038f Fix for poor HTML in some nfacommunity.com descs. 2016-09-20 09:31:08 -05:00
Jim Miller 860da1208e Improvements to CLI/defaults.ini suggested by MrTyton. 2016-09-19 17:41:08 -05:00
Jim Miller fc477034e5 Back out single-source version number changes. Not worth the head aches. 2016-09-18 21:16:38 -05:00
Jim Miller 51542d4e31 Add 'Reject w/o Confirm' option for Series Anthology Check. 2016-09-18 19:33:01 -05:00
Jim Miller fbabf92861 Remove prints. 2016-09-18 17:34:41 -05:00
Jim Miller a00becc3f8 Changes to CLI options for --imap and download from imap and list. 2016-09-18 14:52:15 -05:00
Jim Miller d503d4914b Adding CLI options --imap-list and --download-list 2016-09-15 16:27:58 -05:00
Jim Miller b6bd88c4b2 Adding CLI options --imap-list and --download-list 2016-09-15 12:11:20 -05:00
Jim Miller c4153ac59b Add support for Pillow (vs PIL) image lib in CLI. 2016-09-10 12:29:41 -05:00
Jim Miller dce0b5a335 Adding Update Existing Only option to Email URL Fetch. 2016-09-08 10:50:33 -05:00
Jim Miller d4291f2cd6 Adding Update Existing Only option to Email URL Fetch. 2016-09-08 10:50:01 -05:00
Jim Miller 43637fa67b Single sourcing version number in fanficfare/__init__.py 2016-08-30 21:50:35 -05:00
Jim Miller 173fca8773 Bump versions. 2016-08-24 13:17:35 -05:00
Jim Miller 2ce514a38e Fix for change to AO3 author links. 2016-08-24 09:57:11 -05:00
Jim Miller d0e4999712 Force overall cover for Anthologies if any sub-books have covers to avoid Calibre Polish issues. 2016-08-20 14:05:36 -05:00
Jim Miller 5c8940d333 Fix ffnet category fall back. 2016-08-20 12:08:19 -05:00
Jim Miller 664684b04f Add extra decode code for calling get_urls_from_imap() from outside calibre. 2016-08-20 00:10:18 -05:00
Jim Miller 75064b2267 Update translations. 2016-08-19 23:40:01 -05:00
Jim Miller 06d3fd9080 Fail more gracefully when anthology cover doesn't exist. Polish can remove them. 2016-08-18 19:33:22 -05:00
Jim Miller c6cd9c57e2 Adding logpage_at_end INI option. 2016-08-17 12:14:14 -05:00
Jim Miller 1887cf2cb9 Fix for Publisher and Language metadata for Anthologies. 2016-08-14 11:22:32 -05:00
Jim Miller 2e6a1ef65d Merge pull request #128 from PlushBeaver/masseffect2in-textual-dates
Fix date parsing for masseffect2.in.
2016-08-13 21:06:32 -05:00
Dmitry Kozliuk a0b276beb4 Fix date parsing for masseffect2.in.
The site displays `Вчера' for yesterday and `Сегодня' for today now.
2016-08-14 01:39:14 +03:00
Jim Miller 1ac9e5d36c Tweaks to adultfanfictionorg 2016-08-13 10:38:01 -05:00
Jim Miller 93712de4b7 Add options for controlling Anthology book comments on the Standard Columns config tab. (plugin only) 2016-08-12 21:40:55 -05:00
Jim Miller e7941298b6 Add always_reload_first_chapter setting for base_xenforoforum stories with index chapter. 2016-08-11 12:18:30 -05:00
Jim Miller 65f286be99 Add always_reload_first_chapter setting for base_xenforoforum stories with index chapter. 2016-08-11 12:12:34 -05:00
Jim Miller 1030cf44af Progress Bar - add extra timer loop for Win10 display. 2016-08-11 11:52:55 -05:00
Jim Miller e9190b9a12 New sites from GComyn, including adult-fanfiction.org. 2016-08-10 14:09:02 -05:00
Jim Miller dc63773fad Update translations. 2016-08-09 13:27:30 -05:00
Jim Miller 710800f976 Make base_efiction code accept https story urls. 2016-08-09 09:44:27 -05:00
Jim Miller 3eca202567 Tweaks for tthfanfic.org, including a slow down. 2016-08-09 09:41:38 -05:00
Jim Miller 6d423e586f For lierotica.com, throw a StoryDoesNotExist exception for 'This submission is awaiting moderator's approval' 2016-08-01 22:02:42 -05:00
Jim Miller bda62aab9f Tweaks for quotev.com, including a slow down. 2016-08-01 12:44:33 -05:00
Jim Miller f6d65c334f Bump versions. 2016-07-22 12:02:48 -05:00
Jim Miller 2c5371d95e Update translations. 2016-07-22 11:59:42 -05:00
Jim Miller 6f4660763f Save addheaders when setting cookiejar. For ffn referer. 2016-07-17 17:08:33 -05:00
Jim Miller 75f8f72266 Include 'prefix' tags in forumtags in base_xenforoforum. 2016-07-17 16:38:53 -05:00
Jim Miller b1d689ba3e Add n_anthaver and r_anthaver modes to custom_columns_settings for averaging metadata for anthologies before setting in integer and float calibre custom columns. 2016-07-12 14:57:10 -05:00
Jim Miller 39bb6e37f6 Add n_anthaver and r_anthaver modes to custom_columns_settings for averaging metadata for anthologies before setting in integer and float calibre custom columns. 2016-07-12 14:55:51 -05:00
Jim Miller 288f12afed Add n_anthaver and r_anthaver modes to custom_columns_settings for averaging metadata for anthologies before setting in integer and float calibre custom columns. 2016-07-12 14:53:18 -05:00
Jim Miller 2a25aef7ac Update translations. 2016-07-12 14:49:03 -05:00
Jim Miller 085fb47b08 Remove Django from app.yaml--old version going away. 2016-07-09 09:27:46 -05:00
Jim Miller 5fdcbab46a Allow old goto/post chapter URLs in base_xenforoforum. 2016-07-05 23:22:25 -05:00
Jim Miller 2d83fa8f5d Fix for SIYE when author puts story URL in bio. 2016-06-30 10:38:13 -05:00
Jim Miller 4a752e05e1 Change ffnet metadata colletion to allow for chars with (' - ') in them. 2016-06-26 11:34:36 -05:00
Jim Miller f1d9760aa9 Bump versions. 2016-06-23 17:08:47 -05:00
Jim Miller b11bdd82db Update translations 2016-06-23 17:07:12 -05:00
Jim Miller 6e5de8060e Update translations. 2016-06-22 23:52:19 -05:00
Jim Miller 2c1b9456bf Fix for previously failedtoload img tags causing lookup sleeps. 2016-06-21 11:23:57 -05:00
Jim Miller 60439cc658 Fix for older harrypotterfanfictioncom stories lacking reviewjs.js. 2016-06-21 11:23:20 -05:00
Jim Miller fb95e1e168 Change adapter_fictionmaniatv to set status Completed instead of Complete (no d). 2016-06-16 17:25:46 -05:00
Jim Miller 0d490d2e50 Update translations. 2016-06-15 23:29:41 -05:00
Jim Miller 838510c011 Update translations. 2016-06-15 23:28:50 -05:00
Jim Miller 954ad00ca6 Update StoriesOnlineNet/FineStoriesCom login URL. 2016-06-15 23:25:58 -05:00
Jim Miller ee462d3742 Different detect StoryDoesNotExist string for ficwad.com. 2016-06-10 09:54:45 -05:00
Jim Miller 56401e6dfa Allow old showpost.php chapter URLs in base_xenforoforum. 2016-06-10 09:52:02 -05:00
Jim Miller 6070accbf5 Update (& Fix) Translations. 2016-05-31 19:36:07 -05:00
Jim Miller eebb1ee8d0 Bump versions. 2016-05-23 13:47:11 -05:00
Jim Miller bda307ed3d Update translations. 2016-05-23 13:44:21 -05:00
Jim Miller 6b0c519f99 Make 'Go back to fix errors?' dialog scroll error list for smaller dialog size. 2016-05-22 15:55:00 -05:00
Jim Miller 6533ee0187 Remove defaults.ini sections for removed sites. 2016-05-21 16:45:16 -05:00
Jim Miller 3cc3b32012 Merge pull request #121 from cryzed/master
Fix for #120 
(ProceedQuestion got an unexpected keyword argument 'log_viewer_unique_name')
2016-05-21 10:11:58 -05:00
cryzed 315536a75b Fix for #120 2016-05-21 16:45:05 +02:00
Jim Miller 1c4250c0c1 Add merged tags to anthology epubs. 2016-05-19 09:27:19 -05:00
Jim Miller 6f7e700cd2 Fix for adapter_ashwindersycophanthexcom login. 2016-05-19 09:13:18 -05:00
Jim Miller a593bd201c Bump versions. 2016-05-17 12:34:08 -05:00
Jim Miller 37885a9039 Update translations. 2016-05-13 13:50:01 -05:00
Jim Miller 97437aa7da Fixes for ficbook.net changes. 2016-05-13 13:39:24 -05:00
Jim Miller ccfcd801a3 Fixes for ficbook.net changes. 2016-05-13 13:31:39 -05:00
Jim Miller 4df321c18e Remove support for dead sites. 2016-05-13 12:48:30 -05:00
Jim Miller 664fb639bd Unique name for ViewLog size. 2016-05-09 12:28:18 -05:00
Jim Miller 0cd71263b1 Change adapter_ncisfictionnet to eFiction Base due to site change. 2016-05-01 21:54:55 -05:00
Jim Miller 2fc1a14ac4 Change log status of NotGoingToDownload books to 'Skipped'. 2016-04-28 16:07:12 -05:00
Jim Miller 8772b50451 Update included html2text for web/plugin, require as dependency for CLI. 2016-04-28 14:22:33 -05:00
Jim Miller 7390424f0f Fix for Mac OS X freezes - set log level to CRITICAL unless calibre in debug or in BG job. 2016-04-24 23:38:24 -05:00
Jim Miller 3a16c49c2e Fix for Mac OS X crashes - changes to progress bar dialogs implementation. 2016-04-24 15:56:08 -05:00
Jim Miller d8066d4bcf Update translations. 2016-04-24 14:25:45 -05:00
Jim Miller 61fff980b5 Fix for Yet Another Site Change on quotevcom. 2016-04-22 21:41:53 -05:00
Jim Miller bf704fcdc7 Don't need seconds for published/updated on xenforo. 2016-04-22 21:30:53 -05:00
Jim Miller c9f86bd784 Fix for %p (am/pm) date parsing on non-en_US locales. 2016-04-22 21:24:00 -05:00
Jim Miller 76faea3b4e Comment out a bunch of adapter_storiesonlinenet debug output to avoid encoding issues. 2016-04-20 16:35:45 -05:00
Jim Miller 6b3a19bb45 Remove some more prints in the hopes it will help Mac. 2016-04-20 16:33:55 -05:00
Jim Miller df7bf64845 Update some webservice links/references. 2016-04-19 10:35:37 -05:00
74 changed files with 11471 additions and 7917 deletions
+18 -11
View File
@@ -7,15 +7,21 @@ __license__ = 'GPL v3'
__copyright__ = '2016, Jim Miller' __copyright__ = '2016, Jim Miller'
__docformat__ = 'restructuredtext en' __docformat__ = 'restructuredtext en'
import sys import sys, os
if sys.version_info >= (2, 7): if sys.version_info >= (2, 7):
import logging import logging
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
loghandler=logging.StreamHandler() loghandler=logging.StreamHandler()
loghandler.setFormatter(logging.Formatter("FFF: %(levelname)s: %(asctime)s: %(filename)s(%(lineno)d): %(message)s")) loghandler.setFormatter(logging.Formatter("FFF: %(levelname)s: %(asctime)s: %(filename)s(%(lineno)d): %(message)s"))
logger.addHandler(loghandler) logger.addHandler(loghandler)
loghandler.setLevel(logging.DEBUG)
logger.setLevel(logging.DEBUG) from calibre.constants import DEBUG
if os.environ.get('CALIBRE_WORKER', None) is not None or DEBUG:
loghandler.setLevel(logging.DEBUG)
logger.setLevel(logging.DEBUG)
else:
loghandler.setLevel(logging.CRITICAL)
logger.setLevel(logging.CRITICAL)
# pulls in translation files for _() strings # pulls in translation files for _() strings
try: try:
@@ -42,7 +48,7 @@ class FanFicFareBase(InterfaceActionBase):
description = _('UI plugin to download FanFiction stories from various sites.') description = _('UI plugin to download FanFiction stories from various sites.')
supported_platforms = ['windows', 'osx', 'linux'] supported_platforms = ['windows', 'osx', 'linux']
author = 'Jim Miller' author = 'Jim Miller'
version = (2, 3, 1) version = (2, 5, 3)
minimum_calibre_version = (1, 48, 0) minimum_calibre_version = (1, 48, 0)
#: This field defines the GUI plugin class that contains all the code #: This field defines the GUI plugin class that contains all the code
@@ -110,22 +116,23 @@ class FanFicFareBase(InterfaceActionBase):
from calibre_plugins.fanficfare_plugin.fanficfare.cli import main as fff_main from calibre_plugins.fanficfare_plugin.fanficfare.cli import main as fff_main
from calibre_plugins.fanficfare_plugin.prefs import PrefsFacade from calibre_plugins.fanficfare_plugin.prefs import PrefsFacade
from calibre.utils.config import prefs as calibre_prefs from calibre.utils.config import prefs as calibre_prefs
from optparse import OptionParser from optparse import OptionParser
parser = OptionParser('%prog --run-plugin '+self.name+' -- [options] <storyurl>') parser = OptionParser('%prog --run-plugin '+self.name+' -- [options] <storyurl>')
parser.add_option('--library-path', '--with-library', default=None, help=_('Path to the calibre library. Default is to use the path stored in the settings.')) parser.add_option('--library-path', '--with-library', default=None, help=_('Path to the calibre library. Default is to use the path stored in the settings.'))
# parser.add_option('--dont-notify-gui', default=False, action='store_true', # parser.add_option('--dont-notify-gui', default=False, action='store_true',
# help=_('Do not notify the running calibre GUI (if any) that the database has' # help=_('Do not notify the running calibre GUI (if any) that the database has'
# ' changed. Use with care, as it can lead to database corruption!')) # ' changed. Use with care, as it can lead to database corruption!'))
pargs = [x for x in argv if x.startswith('--with-library') or x.startswith('--library-path') pargs = [x for x in argv if x.startswith('--with-library') or x.startswith('--library-path')
or not x.startswith('-')] or not x.startswith('-')]
opts, args = parser.parse_args(pargs) opts, args = parser.parse_args(pargs)
fff_prefs = PrefsFacade(db(path=opts.library_path, fff_prefs = PrefsFacade(db(path=opts.library_path,
read_only=True)) read_only=True))
fff_main(argv[1:], fff_main(argv[1:],
parser=parser, parser=parser,
passed_defaultsini=StringIO(get_resources("fanficfare/defaults.ini")), passed_defaultsini=StringIO(get_resources("fanficfare/defaults.ini")),
passed_personalini=StringIO(fff_prefs["personal.ini"])) passed_personalini=StringIO(fff_prefs["personal.ini"]),
)
+150 -119
View File
@@ -14,11 +14,13 @@ import traceback, copy, threading, re
from collections import OrderedDict from collections import OrderedDict
try: try:
from PyQt5 import QtWidgets as QtGui
from PyQt5.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QGridLayout, from PyQt5.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QGridLayout,
QLabel, QLineEdit, QFont, QWidget, QTextEdit, QComboBox, QLabel, QLineEdit, QFont, QWidget, QTextEdit, QComboBox,
QCheckBox, QPushButton, QTabWidget, QScrollArea, QCheckBox, QPushButton, QTabWidget, QScrollArea,
QDialogButtonBox, QGroupBox, QButtonGroup, QRadioButton, Qt) QDialogButtonBox, QGroupBox, QButtonGroup, QRadioButton, Qt)
except ImportError as e: except ImportError as e:
from PyQt4 import QtGui
from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QGridLayout, from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QGridLayout,
QLabel, QLineEdit, QFont, QWidget, QTextEdit, QComboBox, QLabel, QLineEdit, QFont, QWidget, QTextEdit, QComboBox,
QCheckBox, QPushButton, QTabWidget, QScrollArea, QCheckBox, QPushButton, QTabWidget, QScrollArea,
@@ -87,7 +89,7 @@ from calibre_plugins.fanficfare_plugin.prefs \
from calibre_plugins.fanficfare_plugin.dialogs \ from calibre_plugins.fanficfare_plugin.dialogs \
import (UPDATE, UPDATEALWAYS, collision_order, save_collisions, RejectListDialog, import (UPDATE, UPDATEALWAYS, collision_order, save_collisions, RejectListDialog,
EditTextDialog, IniTextDialog, RejectUrlEntry) EditTextDialog, IniTextDialog, RejectUrlEntry)
from calibre_plugins.fanficfare_plugin.fanficfare.adapters \ from calibre_plugins.fanficfare_plugin.fanficfare.adapters \
import getSiteSections import getSiteSections
@@ -112,7 +114,7 @@ class RejectURLList:
#print("rue.url:%s"%rue.url) #print("rue.url:%s"%rue.url)
if rue.valid: if rue.valid:
cache[rue.url] = rue cache[rue.url] = rue
return cache return cache
def _get_listcache(self): def _get_listcache(self):
if self.listcache == None: if self.listcache == None:
@@ -124,7 +126,7 @@ class RejectURLList:
self.prefs['rejecturls'] = '\n'.join([x.to_line() for x in listcache.values()]) self.prefs['rejecturls'] = '\n'.join([x.to_line() for x in listcache.values()])
self.prefs.save_to_db() self.prefs.save_to_db()
self.listcache = None self.listcache = None
def clear_cache(self): def clear_cache(self):
self.listcache = None self.listcache = None
@@ -133,7 +135,7 @@ class RejectURLList:
with self.sync_lock: with self.sync_lock:
listcache = self._get_listcache() listcache = self._get_listcache()
return url in listcache return url in listcache
def get_note(self,url): def get_note(self,url):
with self.sync_lock: with self.sync_lock:
listcache = self._get_listcache() listcache = self._get_listcache()
@@ -159,7 +161,7 @@ class RejectURLList:
def add_text(self,rejecttext,addreasontext): def add_text(self,rejecttext,addreasontext):
self.add(self._read_list_from_text(rejecttext,addreasontext).values()) self.add(self._read_list_from_text(rejecttext,addreasontext).values())
def add(self,rejectlist,clear=False): def add(self,rejectlist,clear=False):
with self.sync_lock: with self.sync_lock:
if clear: if clear:
@@ -172,7 +174,7 @@ class RejectURLList:
def get_list(self): def get_list(self):
return self._get_listcache().values() return self._get_listcache().values()
def get_reject_reasons(self): def get_reject_reasons(self):
return self.prefs['rejectreasons'].splitlines() return self.prefs['rejectreasons'].splitlines()
@@ -183,7 +185,7 @@ class ConfigWidget(QWidget):
def __init__(self, plugin_action): def __init__(self, plugin_action):
QWidget.__init__(self) QWidget.__init__(self)
self.plugin_action = plugin_action self.plugin_action = plugin_action
self.l = QVBoxLayout() self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -192,7 +194,7 @@ class ConfigWidget(QWidget):
+_('List of Supported Sites')+'</a> -- <a href="'\ +_('List of Supported Sites')+'</a> -- <a href="'\
+'https://github.com/JimmXinu/FanFicFare/wiki/FAQs">'\ +'https://github.com/JimmXinu/FanFicFare/wiki/FAQs">'\
+_('FAQs')+'</a>') +_('FAQs')+'</a>')
label.setOpenExternalLinks(True) label.setOpenExternalLinks(True)
self.l.addWidget(label) self.l.addWidget(label)
@@ -204,13 +206,13 @@ class ConfigWidget(QWidget):
tab_widget = QTabWidget(self) tab_widget = QTabWidget(self)
self.scroll_area.setWidget(tab_widget) self.scroll_area.setWidget(tab_widget)
self.basic_tab = BasicTab(self, plugin_action) self.basic_tab = BasicTab(self, plugin_action)
tab_widget.addTab(self.basic_tab, _('Basic')) tab_widget.addTab(self.basic_tab, _('Basic'))
self.personalini_tab = PersonalIniTab(self, plugin_action) self.personalini_tab = PersonalIniTab(self, plugin_action)
tab_widget.addTab(self.personalini_tab, 'personal.ini') tab_widget.addTab(self.personalini_tab, 'personal.ini')
self.readinglist_tab = ReadingListTab(self, plugin_action) self.readinglist_tab = ReadingListTab(self, plugin_action)
tab_widget.addTab(self.readinglist_tab, 'Reading Lists') tab_widget.addTab(self.readinglist_tab, 'Reading Lists')
if 'Reading List' not in plugin_action.gui.iactions: if 'Reading List' not in plugin_action.gui.iactions:
@@ -258,6 +260,7 @@ class ConfigWidget(QWidget):
prefs['adddialogstaysontop'] = self.basic_tab.adddialogstaysontop.isChecked() prefs['adddialogstaysontop'] = self.basic_tab.adddialogstaysontop.isChecked()
prefs['lookforurlinhtml'] = self.basic_tab.lookforurlinhtml.isChecked() prefs['lookforurlinhtml'] = self.basic_tab.lookforurlinhtml.isChecked()
prefs['checkforseriesurlid'] = self.basic_tab.checkforseriesurlid.isChecked() prefs['checkforseriesurlid'] = self.basic_tab.checkforseriesurlid.isChecked()
prefs['auto_reject_seriesurlid'] = self.basic_tab.auto_reject_seriesurlid.isChecked()
prefs['checkforurlchange'] = self.basic_tab.checkforurlchange.isChecked() prefs['checkforurlchange'] = self.basic_tab.checkforurlchange.isChecked()
prefs['injectseries'] = self.basic_tab.injectseries.isChecked() prefs['injectseries'] = self.basic_tab.injectseries.isChecked()
prefs['matchtitleauth'] = self.basic_tab.matchtitleauth.isChecked() prefs['matchtitleauth'] = self.basic_tab.matchtitleauth.isChecked()
@@ -285,7 +288,7 @@ class ConfigWidget(QWidget):
prefs['personal.ini'] = get_resources('plugin-example.ini') prefs['personal.ini'] = get_resources('plugin-example.ini')
prefs['cal_cols_pass_in'] = self.personalini_tab.cal_cols_pass_in.isChecked() prefs['cal_cols_pass_in'] = self.personalini_tab.cal_cols_pass_in.isChecked()
# Covers tab # Covers tab
prefs['updatecalcover'] = prefs_save_options[unicode(self.calibrecover_tab.updatecalcover.currentText())] prefs['updatecalcover'] = prefs_save_options[unicode(self.calibrecover_tab.updatecalcover.currentText())]
# for backward compatibility: # for backward compatibility:
@@ -306,7 +309,7 @@ class ConfigWidget(QWidget):
# Count Pages tab # Count Pages tab
countpagesstats = [] countpagesstats = []
if self.countpages_tab.pagecount.isChecked(): if self.countpages_tab.pagecount.isChecked():
countpagesstats.append('PageCount') countpagesstats.append('PageCount')
if self.countpages_tab.wordcount.isChecked(): if self.countpages_tab.wordcount.isChecked():
@@ -317,10 +320,10 @@ class ConfigWidget(QWidget):
countpagesstats.append('FleschGrade') countpagesstats.append('FleschGrade')
if self.countpages_tab.gunningfog.isChecked(): if self.countpages_tab.gunningfog.isChecked():
countpagesstats.append('GunningFog') countpagesstats.append('GunningFog')
prefs['countpagesstats'] = countpagesstats prefs['countpagesstats'] = countpagesstats
prefs['wordcountmissing'] = self.countpages_tab.wordcount.isChecked() and self.countpages_tab.wordcountmissing.isChecked() prefs['wordcountmissing'] = self.countpages_tab.wordcount.isChecked() and self.countpages_tab.wordcountmissing.isChecked()
# Standard Columns tab # Standard Columns tab
colsnewonly = {} colsnewonly = {}
for (col,checkbox) in self.std_columns_tab.stdcol_newonlycheck.iteritems(): for (col,checkbox) in self.std_columns_tab.stdcol_newonlycheck.iteritems():
@@ -328,6 +331,8 @@ class ConfigWidget(QWidget):
prefs['std_cols_newonly'] = colsnewonly prefs['std_cols_newonly'] = colsnewonly
prefs['set_author_url'] =self.std_columns_tab.set_author_url.isChecked() prefs['set_author_url'] =self.std_columns_tab.set_author_url.isChecked()
prefs['includecomments'] =self.std_columns_tab.includecomments.isChecked()
prefs['anth_comments_newonly'] =self.std_columns_tab.anth_comments_newonly.isChecked()
# Custom Columns tab # Custom Columns tab
# error column # error column
@@ -350,7 +355,7 @@ class ConfigWidget(QWidget):
for (col,checkbox) in self.cust_columns_tab.custcol_newonlycheck.iteritems(): for (col,checkbox) in self.cust_columns_tab.custcol_newonlycheck.iteritems():
colsnewonly[col] = checkbox.isChecked() colsnewonly[col] = checkbox.isChecked()
prefs['custom_cols_newonly'] = colsnewonly prefs['custom_cols_newonly'] = colsnewonly
prefs['allow_custcol_from_ini'] = self.cust_columns_tab.allow_custcol_from_ini.isChecked() prefs['allow_custcol_from_ini'] = self.cust_columns_tab.allow_custcol_from_ini.isChecked()
prefs['imapserver'] = unicode(self.imap_tab.imapserver.text()).strip() prefs['imapserver'] = unicode(self.imap_tab.imapserver.text()).strip()
@@ -360,8 +365,9 @@ class ConfigWidget(QWidget):
prefs['imapmarkread'] = self.imap_tab.imapmarkread.isChecked() prefs['imapmarkread'] = self.imap_tab.imapmarkread.isChecked()
prefs['imapsessionpass'] = self.imap_tab.imapsessionpass.isChecked() prefs['imapsessionpass'] = self.imap_tab.imapsessionpass.isChecked()
prefs['auto_reject_from_email'] = self.imap_tab.auto_reject_from_email.isChecked() prefs['auto_reject_from_email'] = self.imap_tab.auto_reject_from_email.isChecked()
prefs['update_existing_only_from_email'] = self.imap_tab.update_existing_only_from_email.isChecked()
prefs['download_from_email_immediately'] = self.imap_tab.download_from_email_immediately.isChecked() prefs['download_from_email_immediately'] = self.imap_tab.download_from_email_immediately.isChecked()
prefs.save_to_db() prefs.save_to_db()
def edit_shortcuts(self): def edit_shortcuts(self):
@@ -378,7 +384,7 @@ class BasicTab(QWidget):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
topl = QVBoxLayout() topl = QVBoxLayout()
self.setLayout(topl) self.setLayout(topl)
@@ -433,7 +439,7 @@ class BasicTab(QWidget):
self.updateepubcover.setToolTip(_("On each download, FanFicFare offers an option to update the book cover image <i>inside</i> the EPUB from the web site when the EPUB is updated.<br />This sets whether that will default to on or off.")) self.updateepubcover.setToolTip(_("On each download, FanFicFare offers an option to update the book cover image <i>inside</i> the EPUB from the web site when the EPUB is updated.<br />This sets whether that will default to on or off."))
self.updateepubcover.setChecked(prefs['updateepubcover']) self.updateepubcover.setChecked(prefs['updateepubcover'])
horz.addWidget(self.updateepubcover) horz.addWidget(self.updateepubcover)
self.bgmeta = QCheckBox(_('Default Background Metadata?'),self) self.bgmeta = QCheckBox(_('Default Background Metadata?'),self)
self.bgmeta.setToolTip(_("On each download, FanFicFare offers an option to Collect Metadata from sites in a Background process.<br />This returns control to you quicker while updating, but you won't be asked for username/passwords or if you are an adult--stories that need those will just fail.<br />Only available for Update/Overwrite of existing books in case URL given isn't canonical or matches to existing book by Title/Author.")) self.bgmeta.setToolTip(_("On each download, FanFicFare offers an option to Collect Metadata from sites in a Background process.<br />This returns control to you quicker while updating, but you won't be asked for username/passwords or if you are an adult--stories that need those will just fail.<br />Only available for Update/Overwrite of existing books in case URL given isn't canonical or matches to existing book by Title/Author."))
self.bgmeta.setChecked(prefs['bgmeta']) self.bgmeta.setChecked(prefs['bgmeta'])
@@ -449,7 +455,7 @@ class BasicTab(QWidget):
self.deleteotherforms.setToolTip(_('Check this to automatically delete all other ebook formats when updating an existing book.\nHandy if you have both a Nook(epub) and Kindle(mobi), for example.')) self.deleteotherforms.setToolTip(_('Check this to automatically delete all other ebook formats when updating an existing book.\nHandy if you have both a Nook(epub) and Kindle(mobi), for example.'))
self.deleteotherforms.setChecked(prefs['deleteotherforms']) self.deleteotherforms.setChecked(prefs['deleteotherforms'])
self.l.addWidget(self.deleteotherforms) self.l.addWidget(self.deleteotherforms)
self.keeptags = QCheckBox(_('Keep Existing Tags when Updating Metadata?'),self) self.keeptags = QCheckBox(_('Keep Existing Tags when Updating Metadata?'),self)
self.keeptags.setToolTip(_("Existing tags will be kept and any new tags added.\n%(cmplt)s and %(inprog)s tags will be still be updated, if known.\n%(lul)s tags will be updated if %(lus)s in %(is)s.\n(If Tags is set to 'New Only' in the Standard Columns tab, this has no effect.)")%no_trans) self.keeptags.setToolTip(_("Existing tags will be kept and any new tags added.\n%(cmplt)s and %(inprog)s tags will be still be updated, if known.\n%(lul)s tags will be updated if %(lus)s in %(is)s.\n(If Tags is set to 'New Only' in the Standard Columns tab, this has no effect.)")%no_trans)
self.keeptags.setChecked(prefs['keeptags']) self.keeptags.setChecked(prefs['keeptags'])
@@ -466,10 +472,18 @@ class BasicTab(QWidget):
self.l.addWidget(self.suppresstitlesort) self.l.addWidget(self.suppresstitlesort)
self.checkforseriesurlid = QCheckBox(_("Check for existing Series Anthology books?"),self) self.checkforseriesurlid = QCheckBox(_("Check for existing Series Anthology books?"),self)
self.checkforseriesurlid.setToolTip(_("Check for existings Series Anthology books using each new story's series URL before downloading.\nOffer to skip downloading if a Series Anthology is found.\nDoesn't work when Collect Metadata in Background is selected.")) self.checkforseriesurlid.setToolTip(_("Check for existing Series Anthology books using each new story's series URL before downloading.\nOffer to skip downloading if a Series Anthology is found.\nDoesn't work when Collect Metadata in Background is selected."))
self.checkforseriesurlid.setChecked(prefs['checkforseriesurlid']) self.checkforseriesurlid.setChecked(prefs['checkforseriesurlid'])
self.l.addWidget(self.checkforseriesurlid) self.l.addWidget(self.checkforseriesurlid)
self.auto_reject_seriesurlid = QCheckBox(_("Reject Without Confirmation?"),self)
self.auto_reject_seriesurlid.setToolTip(_("Automatically reject storys with existing Series Anthology books.\nOnly works if 'Check for existing Series Anthology books' is on.\nDoesn't work when Collect Metadata in Background is selected."))
self.auto_reject_seriesurlid.setChecked(prefs['auto_reject_seriesurlid'])
horz = QHBoxLayout()
horz.addItem(QtGui.QSpacerItem(20, 1))
horz.addWidget(self.auto_reject_seriesurlid)
self.l.addLayout(horz)
self.checkforurlchange = QCheckBox(_("Check for changed Story URL?"),self) self.checkforurlchange = QCheckBox(_("Check for changed Story URL?"),self)
self.checkforurlchange.setToolTip(_("Warn you if an update will change the URL of an existing book.\nfanfiction.net URLs will change from http to https silently.")) self.checkforurlchange.setToolTip(_("Warn you if an update will change the URL of an existing book.\nfanfiction.net URLs will change from http to https silently."))
self.checkforurlchange.setChecked(prefs['checkforurlchange']) self.checkforurlchange.setChecked(prefs['checkforurlchange'])
@@ -499,7 +513,7 @@ class BasicTab(QWidget):
self.smarten_punctuation.setChecked(prefs['smarten_punctuation']) self.smarten_punctuation.setChecked(prefs['smarten_punctuation'])
self.l.addWidget(self.smarten_punctuation) self.l.addWidget(self.smarten_punctuation)
tooltip = _("Calculate Word Counts using Calibre internal methods.\n" tooltip = _("Calculate Word Counts using Calibre internal methods.\n"
"Many sites include Word Count, but many do not.\n" "Many sites include Word Count, but many do not.\n"
"This will count the words in each book and include it as if it came from the site.") "This will count the words in each book and include it as if it came from the site.")
@@ -516,7 +530,7 @@ class BasicTab(QWidget):
horz.addWidget(self.do_wordcount) horz.addWidget(self.do_wordcount)
self.l.addLayout(horz) self.l.addLayout(horz)
self.autoconvert = QCheckBox(_("Automatically Convert new/update books?"),self) self.autoconvert = QCheckBox(_("Automatically Convert new/update books?"),self)
self.autoconvert.setToolTip(_("Automatically call calibre's Convert for new/update books.\nConverts to the current output format as chosen in calibre's\nPreferences->Behavior settings.")) self.autoconvert.setToolTip(_("Automatically call calibre's Convert for new/update books.\nConverts to the current output format as chosen in calibre's\nPreferences->Behavior settings."))
self.autoconvert.setChecked(prefs['autoconvert']) self.autoconvert.setChecked(prefs['autoconvert'])
@@ -568,12 +582,12 @@ class BasicTab(QWidget):
self.rejectlist.setToolTip(_("Edit list of URLs FanFicFare will automatically Reject.")) self.rejectlist.setToolTip(_("Edit list of URLs FanFicFare will automatically Reject."))
self.rejectlist.clicked.connect(self.show_rejectlist) self.rejectlist.clicked.connect(self.show_rejectlist)
self.l.addWidget(self.rejectlist) self.l.addWidget(self.rejectlist)
self.reject_urls = QPushButton(_('Add Reject URLs'), self) self.reject_urls = QPushButton(_('Add Reject URLs'), self)
self.reject_urls.setToolTip(_("Add additional URLs to Reject as text.")) self.reject_urls.setToolTip(_("Add additional URLs to Reject as text."))
self.reject_urls.clicked.connect(self.add_reject_urls) self.reject_urls.clicked.connect(self.add_reject_urls)
self.l.addWidget(self.reject_urls) self.l.addWidget(self.reject_urls)
self.reject_reasons = QPushButton(_('Edit Reject Reasons List'), self) self.reject_reasons = QPushButton(_('Edit Reject Reasons List'), self)
self.reject_reasons.setToolTip(_("Customize the Reasons presented when Rejecting URLs")) self.reject_reasons.setToolTip(_("Customize the Reasons presented when Rejecting URLs"))
self.reject_reasons.clicked.connect(self.show_reject_reasons) self.reject_reasons.clicked.connect(self.show_reject_reasons)
@@ -596,13 +610,13 @@ class BasicTab(QWidget):
vertright.addWidget(gui_gb) vertright.addWidget(gui_gb)
vertright.addWidget(misc_gb) vertright.addWidget(misc_gb)
vertright.addWidget(rej_gb) vertright.addWidget(rej_gb)
horz.addLayout(vertleft) horz.addLayout(vertleft)
horz.addLayout(vertright) horz.addLayout(vertright)
topl.addLayout(horz) topl.addLayout(horz)
topl.insertStretch(-1) topl.insertStretch(-1)
def set_collisions(self): def set_collisions(self):
prev=self.collision.currentText() prev=self.collision.currentText()
self.collision.clear() self.collision.clear()
@@ -612,7 +626,7 @@ class BasicTab(QWidget):
i = self.collision.findText(prev) i = self.collision.findText(prev)
if i > -1: if i > -1:
self.collision.setCurrentIndex(i) self.collision.setCurrentIndex(i)
def show_rejectlist(self): def show_rejectlist(self):
d = RejectListDialog(self, d = RejectListDialog(self,
rejecturllist.get_list(), rejecturllist.get_list(),
@@ -621,12 +635,12 @@ class BasicTab(QWidget):
show_delete=False, show_delete=False,
show_all_reasons=False) show_all_reasons=False)
d.exec_() d.exec_()
if d.result() != d.Accepted: if d.result() != d.Accepted:
return return
rejecturllist.add(d.get_reject_list(),clear=True) rejecturllist.add(d.get_reject_list(),clear=True)
def show_reject_reasons(self): def show_reject_reasons(self):
d = EditTextDialog(self, d = EditTextDialog(self,
prefs['rejectreasons'], prefs['rejectreasons'],
@@ -638,7 +652,7 @@ class BasicTab(QWidget):
d.exec_() d.exec_()
if d.result() == d.Accepted: if d.result() == d.Accepted:
prefs['rejectreasons'] = d.get_plain_text() prefs['rejectreasons'] = d.get_plain_text()
def add_reject_urls(self): def add_reject_urls(self):
d = EditTextDialog(self, d = EditTextDialog(self,
"http://example.com/story.php?sid=5,"+_("Reason why I rejected it")+"\nhttp://example.com/story.php?sid=6,"+_("Title by Author")+" - "+_("Reason why I rejected it"), "http://example.com/story.php?sid=5,"+_("Reason why I rejected it")+"\nhttp://example.com/story.php?sid=6,"+_("Title by Author")+" - "+_("Reason why I rejected it"),
@@ -652,14 +666,14 @@ class BasicTab(QWidget):
d.exec_() d.exec_()
if d.result() == d.Accepted: if d.result() == d.Accepted:
rejecturllist.add_text(d.get_plain_text(),d.get_reason_text()) rejecturllist.add_text(d.get_plain_text(),d.get_reason_text())
class PersonalIniTab(QWidget): class PersonalIniTab(QWidget):
def __init__(self, parent_dialog, plugin_action): def __init__(self, parent_dialog, plugin_action):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.l = QVBoxLayout() self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -676,16 +690,16 @@ class PersonalIniTab(QWidget):
self.l.addWidget(groupbox) self.l.addWidget(groupbox)
horz = QHBoxLayout() horz = QHBoxLayout()
vert.addLayout(horz) vert.addLayout(horz)
self.ini_button = QPushButton(_('Edit personal.ini'), self) self.ini_button = QPushButton(_('Edit personal.ini'), self)
#self.ini_button.setToolTip(_("Edit personal.ini file.")) #self.ini_button.setToolTip(_("Edit personal.ini file."))
self.ini_button.clicked.connect(self.add_ini_button) self.ini_button.clicked.connect(self.add_ini_button)
horz.addWidget(self.ini_button) horz.addWidget(self.ini_button)
label = QLabel(_("FanFicFare now includes find, color coding, and error checking for personal.ini editing. Red generally indicates errors.")) label = QLabel(_("FanFicFare now includes find, color coding, and error checking for personal.ini editing. Red generally indicates errors."))
label.setWordWrap(True) label.setWordWrap(True)
horz.addWidget(label) horz.addWidget(label)
vert.addSpacing(5) vert.addSpacing(5)
horz = QHBoxLayout() horz = QHBoxLayout()
@@ -694,18 +708,18 @@ class PersonalIniTab(QWidget):
#self.ini_button.setToolTip(_("Edit personal.ini file.")) #self.ini_button.setToolTip(_("Edit personal.ini file."))
self.ini_button.clicked.connect(self.safe_ini_button) self.ini_button.clicked.connect(self.safe_ini_button)
horz.addWidget(self.ini_button) horz.addWidget(self.ini_button)
label = QLabel(_("View your personal.ini with usernames and passwords removed. For safely sharing your personal.ini settings with others.")) label = QLabel(_("View your personal.ini with usernames and passwords removed. For safely sharing your personal.ini settings with others."))
label.setWordWrap(True) label.setWordWrap(True)
horz.addWidget(label) horz.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
groupbox = QGroupBox(_("defaults.ini")) groupbox = QGroupBox(_("defaults.ini"))
horz = QHBoxLayout() horz = QHBoxLayout()
groupbox.setLayout(horz) groupbox.setLayout(horz)
self.l.addWidget(groupbox) self.l.addWidget(groupbox)
view_label = _("View all of the plugin's configurable settings\nand their default settings.") view_label = _("View all of the plugin's configurable settings\nand their default settings.")
self.defaults = QPushButton(_('View Defaults')+' (plugin-defaults.ini)', self) self.defaults = QPushButton(_('View Defaults')+' (plugin-defaults.ini)', self)
self.defaults.setToolTip(view_label) self.defaults.setToolTip(view_label)
@@ -715,7 +729,7 @@ class PersonalIniTab(QWidget):
label = QLabel(view_label) label = QLabel(view_label)
label.setWordWrap(True) label.setWordWrap(True)
horz.addWidget(label) horz.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
groupbox = QGroupBox(_("Calibre Columns")) groupbox = QGroupBox(_("Calibre Columns"))
@@ -730,13 +744,13 @@ class PersonalIniTab(QWidget):
self.cal_cols_pass_in.setToolTip(pass_label) self.cal_cols_pass_in.setToolTip(pass_label)
self.cal_cols_pass_in.setChecked(prefs['cal_cols_pass_in']) self.cal_cols_pass_in.setChecked(prefs['cal_cols_pass_in'])
horz.addWidget(self.cal_cols_pass_in) horz.addWidget(self.cal_cols_pass_in)
label = QLabel(pass_label) label = QLabel(pass_label)
label.setWordWrap(True) label.setWordWrap(True)
horz.addWidget(label) horz.addWidget(label)
vert.addSpacing(5) vert.addSpacing(5)
horz = QHBoxLayout() horz = QHBoxLayout()
vert.addLayout(horz) vert.addLayout(horz)
col_label = _("FanFicFare can pass the Calibre Columns into the download/update process.<br>This will show you the columns available by name.") col_label = _("FanFicFare can pass the Calibre Columns into the download/update process.<br>This will show you the columns available by name.")
@@ -754,7 +768,7 @@ class PersonalIniTab(QWidget):
self.l.addWidget(label) self.l.addWidget(label)
self.l.insertStretch(-1) self.l.insertStretch(-1)
def show_defaults(self): def show_defaults(self):
IniTextDialog(self, IniTextDialog(self,
get_resources('plugin-defaults.ini'), get_resources('plugin-defaults.ini'),
@@ -764,10 +778,10 @@ class PersonalIniTab(QWidget):
use_find=True, use_find=True,
read_only=True, read_only=True,
save_size_name='fff:defaults.ini').exec_() save_size_name='fff:defaults.ini').exec_()
def safe_ini_button(self): def safe_ini_button(self):
personalini = re.sub(r'((username|password) *[=:]).*$',r'\1XXXXXXXX',self.personalini,flags=re.MULTILINE) personalini = re.sub(r'((username|password) *[=:]).*$',r'\1XXXXXXXX',self.personalini,flags=re.MULTILINE)
d = EditTextDialog(self, d = EditTextDialog(self,
personalini, personalini,
icon=self.windowIcon(), icon=self.windowIcon(),
@@ -776,7 +790,7 @@ class PersonalIniTab(QWidget):
save_size_name='fff:safe personal.ini', save_size_name='fff:safe personal.ini',
read_only=True) read_only=True)
d.exec_() d.exec_()
def add_ini_button(self): def add_ini_button(self):
d = IniTextDialog(self, d = IniTextDialog(self,
self.personalini, self.personalini,
@@ -788,13 +802,13 @@ class PersonalIniTab(QWidget):
d.exec_() d.exec_()
if d.result() == d.Accepted: if d.result() == d.Accepted:
self.personalini = d.get_plain_text() self.personalini = d.get_plain_text()
def show_showcalcols(self): def show_showcalcols(self):
lines=[]#[('calibre_std_user_categories',_('User Categories'))] lines=[]#[('calibre_std_user_categories',_('User Categories'))]
for k,f in field_metadata.iteritems(): for k,f in field_metadata.iteritems():
if f['name'] and k not in STD_COLS_SKIP: # only if it has a human readable name. if f['name'] and k not in STD_COLS_SKIP: # only if it has a human readable name.
lines.append(('calibre_std_'+k,f['name'])) lines.append(('calibre_std_'+k,f['name']))
for k, column in self.plugin_action.gui.library_view.model().custom_columns.iteritems(): for k, column in self.plugin_action.gui.library_view.model().custom_columns.iteritems():
if k != prefs['savemetacol']: if k != prefs['savemetacol']:
# custom always have name. # custom always have name.
@@ -809,14 +823,14 @@ class PersonalIniTab(QWidget):
label=_('Label (entry_name)'), label=_('Label (entry_name)'),
read_only=True, read_only=True,
save_size_name='fff:showcalcols').exec_() save_size_name='fff:showcalcols').exec_()
class ReadingListTab(QWidget): class ReadingListTab(QWidget):
def __init__(self, parent_dialog, plugin_action): def __init__(self, parent_dialog, plugin_action):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.l = QVBoxLayout() self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -825,21 +839,21 @@ class ReadingListTab(QWidget):
reading_lists = rl_plugin.get_list_names() reading_lists = rl_plugin.get_list_names()
except KeyError: except KeyError:
reading_lists= [] reading_lists= []
label = QLabel(_('These settings provide integration with the %(rl)s Plugin. %(rl)s can automatically send to devices and change custom columns. You have to create and configure the lists in %(rl)s to be useful.')%no_trans) label = QLabel(_('These settings provide integration with the %(rl)s Plugin. %(rl)s can automatically send to devices and change custom columns. You have to create and configure the lists in %(rl)s to be useful.')%no_trans)
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
self.addtolists = QCheckBox(_('Add new/updated stories to "Send to Device" Reading List(s).'),self) self.addtolists = QCheckBox(_('Add new/updated stories to "Send to Device" Reading List(s).'),self)
self.addtolists.setToolTip(_('Automatically add new/updated stories to these lists in the %(rl)s plugin.')%no_trans) self.addtolists.setToolTip(_('Automatically add new/updated stories to these lists in the %(rl)s plugin.')%no_trans)
self.addtolists.setChecked(prefs['addtolists']) self.addtolists.setChecked(prefs['addtolists'])
self.l.addWidget(self.addtolists) self.l.addWidget(self.addtolists)
horz = QHBoxLayout() horz = QHBoxLayout()
label = QLabel(_('"Send to Device" Reading Lists')) label = QLabel(_('"Send to Device" Reading Lists'))
label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists.")) label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
horz.addWidget(label) horz.addWidget(label)
self.send_lists_box = EditWithComplete(self) self.send_lists_box = EditWithComplete(self)
self.send_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists.")) self.send_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
self.send_lists_box.update_items_cache(reading_lists) self.send_lists_box.update_items_cache(reading_lists)
@@ -847,16 +861,16 @@ class ReadingListTab(QWidget):
horz.addWidget(self.send_lists_box) horz.addWidget(self.send_lists_box)
self.send_lists_box.setCursorPosition(0) self.send_lists_box.setCursorPosition(0)
self.l.addLayout(horz) self.l.addLayout(horz)
self.addtoreadlists = QCheckBox(_('Add new/updated stories to "To Read" Reading List(s).'),self) self.addtoreadlists = QCheckBox(_('Add new/updated stories to "To Read" Reading List(s).'),self)
self.addtoreadlists.setToolTip(_('Automatically add new/updated stories to these lists in the %(rl)s plugin.\nAlso offers menu option to remove stories from the "To Read" lists.')%no_trans) self.addtoreadlists.setToolTip(_('Automatically add new/updated stories to these lists in the %(rl)s plugin.\nAlso offers menu option to remove stories from the "To Read" lists.')%no_trans)
self.addtoreadlists.setChecked(prefs['addtoreadlists']) self.addtoreadlists.setChecked(prefs['addtoreadlists'])
self.l.addWidget(self.addtoreadlists) self.l.addWidget(self.addtoreadlists)
horz = QHBoxLayout() horz = QHBoxLayout()
label = QLabel(_('"To Read" Reading Lists')) label = QLabel(_('"To Read" Reading Lists'))
label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists.")) label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
horz.addWidget(label) horz.addWidget(label)
self.read_lists_box = EditWithComplete(self) self.read_lists_box = EditWithComplete(self)
self.read_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists.")) self.read_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
self.read_lists_box.update_items_cache(reading_lists) self.read_lists_box.update_items_cache(reading_lists)
@@ -864,31 +878,31 @@ class ReadingListTab(QWidget):
horz.addWidget(self.read_lists_box) horz.addWidget(self.read_lists_box)
self.read_lists_box.setCursorPosition(0) self.read_lists_box.setCursorPosition(0)
self.l.addLayout(horz) self.l.addLayout(horz)
self.addtolistsonread = QCheckBox(_('Add stories back to "Send to Device" Reading List(s) when marked "Read".'),self) self.addtolistsonread = QCheckBox(_('Add stories back to "Send to Device" Reading List(s) when marked "Read".'),self)
self.addtolistsonread.setToolTip(_('Menu option to remove from "To Read" lists will also add stories back to "Send to Device" Reading List(s)')) self.addtolistsonread.setToolTip(_('Menu option to remove from "To Read" lists will also add stories back to "Send to Device" Reading List(s)'))
self.addtolistsonread.setChecked(prefs['addtolistsonread']) self.addtolistsonread.setChecked(prefs['addtolistsonread'])
self.l.addWidget(self.addtolistsonread) self.l.addWidget(self.addtolistsonread)
self.autounnew = QCheckBox(_('Automatically run Remove "New" Chapter Marks when marking books "Read".'),self) self.autounnew = QCheckBox(_('Automatically run Remove "New" Chapter Marks when marking books "Read".'),self)
self.autounnew.setToolTip(_('Menu option to remove from "To Read" lists will also remove "(new)" chapter marks created by personal.ini <i>mark_new_chapters</i> setting.')) self.autounnew.setToolTip(_('Menu option to remove from "To Read" lists will also remove "(new)" chapter marks created by personal.ini <i>mark_new_chapters</i> setting.'))
self.autounnew.setChecked(prefs['autounnew']) self.autounnew.setChecked(prefs['autounnew'])
self.l.addWidget(self.autounnew) self.l.addWidget(self.autounnew)
self.l.insertStretch(-1) self.l.insertStretch(-1)
class CalibreCoverTab(QWidget): class CalibreCoverTab(QWidget):
def __init__(self, parent_dialog, plugin_action): def __init__(self, parent_dialog, plugin_action):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.gencov_elements=[] ## used to disable/enable when gen self.gencov_elements=[] ## used to disable/enable when gen
## cover is off/on. This is more ## cover is off/on. This is more
## about being a visual que than real ## about being a visual que than real
## necessary function. ## necessary function.
topl = self.l = QVBoxLayout() topl = self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -898,7 +912,7 @@ class CalibreCoverTab(QWidget):
except KeyError: except KeyError:
gc_settings= [] gc_settings= []
label = QLabel(_("The Calibre cover image for a downloaded book can come" label = QLabel(_("The Calibre cover image for a downloaded book can come"
" from the story site(if EPUB and images are enabled), or" " from the story site(if EPUB and images are enabled), or"
" from either Calibre's built-in random cover generator or" " from either Calibre's built-in random cover generator or"
@@ -906,7 +920,7 @@ class CalibreCoverTab(QWidget):
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
tooltip = _("Update Calibre book cover image from EPUB when Calibre metadata is updated.\n" tooltip = _("Update Calibre book cover image from EPUB when Calibre metadata is updated.\n"
"Doesn't go looking for new images on 'Update Calibre Metadata Only'.\n" "Doesn't go looking for new images on 'Update Calibre Metadata Only'.\n"
"Cover in EPUB could be from site or previously injected into the EPUB.\n" "Cover in EPUB could be from site or previously injected into the EPUB.\n"
@@ -947,7 +961,7 @@ class CalibreCoverTab(QWidget):
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_YES])) # self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_YES]))
# else: # doesn't have own value, old value not set, NO. # else: # doesn't have own value, old value not set, NO.
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_NO])) # self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_NO]))
self.gencalcover.setToolTip(tooltip) self.gencalcover.setToolTip(tooltip)
label.setBuddy(self.gencalcover) label.setBuddy(self.gencalcover)
horz.addWidget(self.gencalcover) horz.addWidget(self.gencalcover)
@@ -960,7 +974,7 @@ class CalibreCoverTab(QWidget):
self.gencov_gb = QGroupBox() self.gencov_gb = QGroupBox()
horz = QHBoxLayout() horz = QHBoxLayout()
self.gencov_gb.setLayout(horz) self.gencov_gb.setLayout(horz)
self.plugin_gen_cover = QRadioButton(_('Plugin %(gc)s')%no_trans,self) self.plugin_gen_cover = QRadioButton(_('Plugin %(gc)s')%no_trans,self)
self.plugin_gen_cover.setToolTip(_("Use plugin to create covers. Additional settings are below.")) self.plugin_gen_cover.setToolTip(_("Use plugin to create covers. Additional settings are below."))
self.gencov_rdgrp.addButton(self.plugin_gen_cover) self.gencov_rdgrp.addButton(self.plugin_gen_cover)
@@ -1005,7 +1019,7 @@ class CalibreCoverTab(QWidget):
self.gencov_elements.append(self.gcp_gb) self.gencov_elements.append(self.gcp_gb)
self.gencov_rdgrp.buttonClicked.connect(self.endisable_elements) self.gencov_rdgrp.buttonClicked.connect(self.endisable_elements)
label = QLabel(_('The %(gc)s plugin can create cover images for books using various metadata (including existing cover image). If you have %(gc)s installed, FanFicFare can run %(gc)s on new downloads and metadata updates. Pick a %(gc)s setting by site and/or one to use by Default.')%no_trans) label = QLabel(_('The %(gc)s plugin can create cover images for books using various metadata (including existing cover image). If you have %(gc)s installed, FanFicFare can run %(gc)s on new downloads and metadata updates. Pick a %(gc)s setting by site and/or one to use by Default.')%no_trans)
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
@@ -1019,7 +1033,7 @@ class CalibreCoverTab(QWidget):
self.sl = QVBoxLayout() self.sl = QVBoxLayout()
scrollcontent.setLayout(self.sl) scrollcontent.setLayout(self.sl)
self.gc_dropdowns = {} self.gc_dropdowns = {}
sitelist = getSiteSections() sitelist = getSiteSections()
@@ -1052,9 +1066,9 @@ class CalibreCoverTab(QWidget):
horz.addWidget(dropdown) horz.addWidget(dropdown)
self.sl.addLayout(horz) self.sl.addLayout(horz)
self.sl.insertStretch(-1) self.sl.insertStretch(-1)
self.allow_gc_from_ini = QCheckBox(_('Allow %(gcset)s from %(pini)s to override')%no_trans,self) self.allow_gc_from_ini = QCheckBox(_('Allow %(gcset)s from %(pini)s to override')%no_trans,self)
self.allow_gc_from_ini.setToolTip(_("The %(pini)s parameter %(gcset)s allows you to choose a %(gc)s setting based on metadata" self.allow_gc_from_ini.setToolTip(_("The %(pini)s parameter %(gcset)s allows you to choose a %(gc)s setting based on metadata"
" rather than site, but it's much more complex.<br \>%(gcset)s is ignored when this is off.")%no_trans) " rather than site, but it's much more complex.<br \>%(gcset)s is ignored when this is off.")%no_trans)
@@ -1068,7 +1082,7 @@ class CalibreCoverTab(QWidget):
"Clearing house function for setting elements of Calibre" "Clearing house function for setting elements of Calibre"
"Cover tab enabled/disabled depending on all factors." "Cover tab enabled/disabled depending on all factors."
## First, cover gen on/off ## First, cover gen on/off
for e in self.gencov_elements: for e in self.gencov_elements:
e.setEnabled(prefs_save_options[unicode(self.gencalcover.currentText())] != SAVE_NO) e.setEnabled(prefs_save_options[unicode(self.gencalcover.currentText())] != SAVE_NO)
@@ -1082,15 +1096,15 @@ class CalibreCoverTab(QWidget):
if not 'Generate Cover' in self.plugin_action.gui.iactions: if not 'Generate Cover' in self.plugin_action.gui.iactions:
self.plugin_gen_cover.setEnabled(False) self.plugin_gen_cover.setEnabled(False)
self.gcp_gb.setEnabled(False) self.gcp_gb.setEnabled(False)
class CountPagesTab(QWidget): class CountPagesTab(QWidget):
def __init__(self, parent_dialog, plugin_action): def __init__(self, parent_dialog, plugin_action):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.l = QVBoxLayout() self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -1113,7 +1127,7 @@ class CountPagesTab(QWidget):
self.l.addWidget(self.pagecount) self.l.addWidget(self.pagecount)
horz = QHBoxLayout() horz = QHBoxLayout()
self.wordcount = QCheckBox('Word Count',self) self.wordcount = QCheckBox('Word Count',self)
self.wordcount.setToolTip(tooltip+"\n"+_('Will overwrite word count from FanFicFare metadata if set to update the same custom column.')) self.wordcount.setToolTip(tooltip+"\n"+_('Will overwrite word count from FanFicFare metadata if set to update the same custom column.'))
self.wordcount.setChecked('WordCount' in prefs['countpagesstats']) self.wordcount.setChecked('WordCount' in prefs['countpagesstats'])
@@ -1126,33 +1140,33 @@ class CountPagesTab(QWidget):
horz.addWidget(self.wordcountmissing) horz.addWidget(self.wordcountmissing)
self.wordcount.stateChanged.connect(lambda x : self.wordcountmissing.setEnabled(self.wordcount.isChecked())) self.wordcount.stateChanged.connect(lambda x : self.wordcountmissing.setEnabled(self.wordcount.isChecked()))
self.l.addLayout(horz) self.l.addLayout(horz)
self.fleschreading = QCheckBox('Flesch Reading Ease',self) self.fleschreading = QCheckBox('Flesch Reading Ease',self)
self.fleschreading.setToolTip(tooltip) self.fleschreading.setToolTip(tooltip)
self.fleschreading.setChecked('FleschReading' in prefs['countpagesstats']) self.fleschreading.setChecked('FleschReading' in prefs['countpagesstats'])
self.l.addWidget(self.fleschreading) self.l.addWidget(self.fleschreading)
self.fleschgrade = QCheckBox('Flesch-Kincaid Grade Level',self) self.fleschgrade = QCheckBox('Flesch-Kincaid Grade Level',self)
self.fleschgrade.setToolTip(tooltip) self.fleschgrade.setToolTip(tooltip)
self.fleschgrade.setChecked('FleschGrade' in prefs['countpagesstats']) self.fleschgrade.setChecked('FleschGrade' in prefs['countpagesstats'])
self.l.addWidget(self.fleschgrade) self.l.addWidget(self.fleschgrade)
self.gunningfog = QCheckBox('Gunning Fog Index',self) self.gunningfog = QCheckBox('Gunning Fog Index',self)
self.gunningfog.setToolTip(tooltip) self.gunningfog.setToolTip(tooltip)
self.gunningfog.setChecked('GunningFog' in prefs['countpagesstats']) self.gunningfog.setChecked('GunningFog' in prefs['countpagesstats'])
self.l.addWidget(self.gunningfog) self.l.addWidget(self.gunningfog)
self.l.insertStretch(-1) self.l.insertStretch(-1)
class OtherTab(QWidget): class OtherTab(QWidget):
def __init__(self, parent_dialog, plugin_action): def __init__(self, parent_dialog, plugin_action):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.l = QVBoxLayout() self.l = QVBoxLayout()
self.setLayout(self.l) self.setLayout(self.l)
@@ -1160,7 +1174,7 @@ class OtherTab(QWidget):
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
keyboard_shortcuts_button = QPushButton(_('Keyboard shortcuts...'), self) keyboard_shortcuts_button = QPushButton(_('Keyboard shortcuts...'), self)
keyboard_shortcuts_button.setToolTip(_('Edit the keyboard shortcuts associated with this plugin')) keyboard_shortcuts_button.setToolTip(_('Edit the keyboard shortcuts associated with this plugin'))
keyboard_shortcuts_button.clicked.connect(parent_dialog.edit_shortcuts) keyboard_shortcuts_button.clicked.connect(parent_dialog.edit_shortcuts)
@@ -1170,14 +1184,14 @@ class OtherTab(QWidget):
reset_confirmation_button.setToolTip(_('Reset all show me again dialogs for the FanFicFare plugin')) reset_confirmation_button.setToolTip(_('Reset all show me again dialogs for the FanFicFare plugin'))
reset_confirmation_button.clicked.connect(self.reset_dialogs) reset_confirmation_button.clicked.connect(self.reset_dialogs)
self.l.addWidget(reset_confirmation_button) self.l.addWidget(reset_confirmation_button)
view_prefs_button = QPushButton(_('&View library preferences...'), self) view_prefs_button = QPushButton(_('&View library preferences...'), self)
view_prefs_button.setToolTip(_('View data stored in the library database for this plugin')) view_prefs_button.setToolTip(_('View data stored in the library database for this plugin'))
view_prefs_button.clicked.connect(self.view_prefs) view_prefs_button.clicked.connect(self.view_prefs)
self.l.addWidget(view_prefs_button) self.l.addWidget(view_prefs_button)
self.l.insertStretch(-1) self.l.insertStretch(-1)
def reset_dialogs(self): def reset_dialogs(self):
for key in dynamic.keys(): for key in dynamic.keys():
if key.startswith('fanfictiondownloader_') and key.endswith('_again') \ if key.startswith('fanfictiondownloader_') and key.endswith('_again') \
@@ -1187,7 +1201,7 @@ class OtherTab(QWidget):
_('Confirmation dialogs have all been reset'), _('Confirmation dialogs have all been reset'),
show=True, show=True,
show_copy_button=False) show_copy_button=False)
def view_prefs(self): def view_prefs(self):
d = PrefsViewerDialog(self.plugin_action.gui, PREFS_NAMESPACE) d = PrefsViewerDialog(self.plugin_action.gui, PREFS_NAMESPACE)
d.exec_() d.exec_()
@@ -1269,7 +1283,7 @@ class CustomColumnsTab(QWidget):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
custom_columns = self.plugin_action.gui.library_view.model().custom_columns custom_columns = self.plugin_action.gui.library_view.model().custom_columns
self.l = QVBoxLayout() self.l = QVBoxLayout()
@@ -1279,7 +1293,7 @@ class CustomColumnsTab(QWidget):
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
self.custcol_dropdowns = {} self.custcol_dropdowns = {}
self.custcol_newonlycheck = {} self.custcol_newonlycheck = {}
@@ -1291,7 +1305,7 @@ class CustomColumnsTab(QWidget):
self.sl = QVBoxLayout() self.sl = QVBoxLayout()
scrollcontent.setLayout(self.sl) scrollcontent.setLayout(self.sl)
for key, column in custom_columns.iteritems(): for key, column in custom_columns.iteritems():
if column['datatype'] in permitted_values: if column['datatype'] in permitted_values:
@@ -1321,9 +1335,9 @@ class CustomColumnsTab(QWidget):
if key in prefs['custom_cols_newonly']: if key in prefs['custom_cols_newonly']:
newonlycheck.setChecked(prefs['custom_cols_newonly'][key]) newonlycheck.setChecked(prefs['custom_cols_newonly'][key])
horz.addWidget(newonlycheck) horz.addWidget(newonlycheck)
self.sl.addLayout(horz) self.sl.addLayout(horz)
self.sl.insertStretch(-1) self.sl.insertStretch(-1)
self.l.addSpacing(5) self.l.addSpacing(5)
@@ -1331,11 +1345,11 @@ class CustomColumnsTab(QWidget):
self.allow_custcol_from_ini.setToolTip(_("The %(pini)s parameter %(ccset)s allows you to set custom columns to site specific values that aren't common to all sites.<br />%(ccset)s is ignored when this is off.")%no_trans) self.allow_custcol_from_ini.setToolTip(_("The %(pini)s parameter %(ccset)s allows you to set custom columns to site specific values that aren't common to all sites.<br />%(ccset)s is ignored when this is off.")%no_trans)
self.allow_custcol_from_ini.setChecked(prefs['allow_custcol_from_ini']) self.allow_custcol_from_ini.setChecked(prefs['allow_custcol_from_ini'])
self.l.addWidget(self.allow_custcol_from_ini) self.l.addWidget(self.allow_custcol_from_ini)
label = QLabel(_("Special column:")) label = QLabel(_("Special column:"))
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
horz = QHBoxLayout() horz = QHBoxLayout()
label = QLabel(_("Update/Overwrite Error Column:")) label = QLabel(_("Update/Overwrite Error Column:"))
tooltip=_("When an update or overwrite of an existing story fails, record the reason in this column.\n(Text and Long Text columns only.)") tooltip=_("When an update or overwrite of an existing story fails, record the reason in this column.\n(Text and Long Text columns only.)")
@@ -1350,7 +1364,7 @@ class CustomColumnsTab(QWidget):
self.errorcol.addItem(column['name'],key) self.errorcol.addItem(column['name'],key)
self.errorcol.setCurrentIndex(self.errorcol.findData(prefs['errorcol'])) self.errorcol.setCurrentIndex(self.errorcol.findData(prefs['errorcol']))
horz.addWidget(self.errorcol) horz.addWidget(self.errorcol)
self.save_all_errors = QCheckBox(_('Save All Errors'),self) self.save_all_errors = QCheckBox(_('Save All Errors'),self)
self.save_all_errors.setToolTip(_('If unchecked, these errors will not be saved:%s')%( self.save_all_errors.setToolTip(_('If unchecked, these errors will not be saved:%s')%(
'\n'+ '\n'+
@@ -1358,9 +1372,9 @@ class CustomColumnsTab(QWidget):
_("Already contains %d chapters.").replace('%d','X'))))) _("Already contains %d chapters.").replace('%d','X')))))
self.save_all_errors.setChecked(prefs['save_all_errors']) self.save_all_errors.setChecked(prefs['save_all_errors'])
horz.addWidget(self.save_all_errors) horz.addWidget(self.save_all_errors)
self.l.addLayout(horz) self.l.addLayout(horz)
horz = QHBoxLayout() horz = QHBoxLayout()
label = QLabel(_("Saved Metadata Column:")) label = QLabel(_("Saved Metadata Column:"))
tooltip=_("If set, FanFicFare will save a copy of all its metadata in this column when the book is downloaded or updated.<br/>The metadata from this column can later be used to update custom columns without having to request the metadata from the server again.<br/>(Long Text columns only.)") tooltip=_("If set, FanFicFare will save a copy of all its metadata in this column when the book is downloaded or updated.<br/>The metadata from this column can later be used to update custom columns without having to request the metadata from the server again.<br/>(Long Text columns only.)")
@@ -1377,7 +1391,7 @@ class CustomColumnsTab(QWidget):
label = QLabel('') label = QLabel('')
horz.addWidget(label) # empty spacer for alignment with error column line. horz.addWidget(label) # empty spacer for alignment with error column line.
self.l.addLayout(horz) self.l.addLayout(horz)
#print("prefs['custom_cols'] %s"%prefs['custom_cols']) #print("prefs['custom_cols'] %s"%prefs['custom_cols'])
@@ -1391,7 +1405,7 @@ class StandardColumnsTab(QWidget):
QWidget.__init__(self) QWidget.__init__(self)
columns=OrderedDict() columns=OrderedDict()
columns["title"]=_("Title") columns["title"]=_("Title")
columns["authors"]=_("Author(s)") columns["authors"]=_("Author(s)")
columns["publisher"]=_("Publisher") columns["publisher"]=_("Publisher")
@@ -1410,7 +1424,7 @@ class StandardColumnsTab(QWidget):
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label) self.l.addWidget(label)
self.l.addSpacing(5) self.l.addSpacing(5)
self.stdcol_newonlycheck = {} self.stdcol_newonlycheck = {}
for key, column in columns.iteritems(): for key, column in columns.iteritems():
@@ -1425,9 +1439,9 @@ class StandardColumnsTab(QWidget):
if key in prefs['std_cols_newonly']: if key in prefs['std_cols_newonly']:
newonlycheck.setChecked(prefs['std_cols_newonly'][key]) newonlycheck.setChecked(prefs['std_cols_newonly'][key])
horz.addWidget(newonlycheck) horz.addWidget(newonlycheck)
self.l.addLayout(horz) self.l.addLayout(horz)
self.l.addSpacing(5) self.l.addSpacing(5)
label = QLabel(_("Other Standard Column Options")) label = QLabel(_("Other Standard Column Options"))
label.setWordWrap(True) label.setWordWrap(True)
@@ -1438,7 +1452,18 @@ class StandardColumnsTab(QWidget):
self.set_author_url.setToolTip(_("Set Calibre Author URL to Author's URL on story site.")) self.set_author_url.setToolTip(_("Set Calibre Author URL to Author's URL on story site."))
self.set_author_url.setChecked(prefs['set_author_url']) self.set_author_url.setChecked(prefs['set_author_url'])
self.l.addWidget(self.set_author_url) self.l.addWidget(self.set_author_url)
self.includecomments = QCheckBox(_("Include Books' Comments in Anthology Comments?"),self)
self.includecomments.setToolTip(_('''Include all the merged books' comments in the new book's comments.
Default is a list of included titles only.'''))
self.includecomments.setChecked(prefs['includecomments'])
self.l.addWidget(self.includecomments)
self.anth_comments_newonly = QCheckBox(_("Set Anthology Comments only for new books"),self)
self.anth_comments_newonly.setToolTip(_("Comments will only be set for New Anthologies, not updates.\nThat way comments you set manually are retained."))
self.anth_comments_newonly.setChecked(prefs['anth_comments_newonly'])
self.l.addWidget(self.anth_comments_newonly)
self.l.insertStretch(-1) self.l.insertStretch(-1)
class ImapTab(QWidget): class ImapTab(QWidget):
@@ -1447,11 +1472,11 @@ class ImapTab(QWidget):
self.parent_dialog = parent_dialog self.parent_dialog = parent_dialog
self.plugin_action = plugin_action self.plugin_action = plugin_action
QWidget.__init__(self) QWidget.__init__(self)
self.l = QGridLayout() self.l = QGridLayout()
self.setLayout(self.l) self.setLayout(self.l)
row=0 row=0
label = QLabel(_('These settings will allow FanFicFare to fetch story URLs from your email account. It will only look for story URLs in unread emails in the folder specified below.')) label = QLabel(_('These settings will allow FanFicFare to fetch story URLs from your email account. It will only look for story URLs in unread emails in the folder specified below.'))
label.setWordWrap(True) label.setWordWrap(True)
self.l.addWidget(label,row,0,1,-1) self.l.addWidget(label,row,0,1,-1)
@@ -1466,21 +1491,21 @@ class ImapTab(QWidget):
self.imapserver.setText(prefs['imapserver']) self.imapserver.setText(prefs['imapserver'])
self.l.addWidget(self.imapserver,row,1) self.l.addWidget(self.imapserver,row,1)
row+=1 row+=1
label = QLabel(_('IMAP User Name')) label = QLabel(_('IMAP User Name'))
tooltip = _("Name of IMAP user. Eg: yourname@gmail.com\nNote that Gmail accounts need to have IMAP enabled in Gmail Settings first.") tooltip = _("Name of IMAP user. Eg: yourname@gmail.com\nNote that Gmail accounts need to have IMAP enabled in Gmail Settings first.")
label.setToolTip(tooltip) label.setToolTip(tooltip)
self.l.addWidget(label,row,0) self.l.addWidget(label,row,0)
self.imapuser = QLineEdit(self) self.imapuser = QLineEdit(self)
self.imapuser.setToolTip(tooltip) self.imapuser.setToolTip(tooltip)
self.imapuser.setText(prefs['imapuser']) self.imapuser.setText(prefs['imapuser'])
self.l.addWidget(self.imapuser,row,1) self.l.addWidget(self.imapuser,row,1)
row+=1 row+=1
label = QLabel(_('IMAP User Password')) label = QLabel(_('IMAP User Password'))
tooltip = _("IMAP password. If left empty, FanFicFare will ask you for your password when you use the feature.") tooltip = _("IMAP password. If left empty, FanFicFare will ask you for your password when you use the feature.")
label.setToolTip(tooltip) label.setToolTip(tooltip)
self.l.addWidget(label,row,0) self.l.addWidget(label,row,0)
self.imappass = QLineEdit(self) self.imappass = QLineEdit(self)
self.imappass.setToolTip(tooltip) self.imappass.setToolTip(tooltip)
self.imappass.setEchoMode(QLineEdit.Password) self.imappass.setEchoMode(QLineEdit.Password)
@@ -1493,35 +1518,41 @@ class ImapTab(QWidget):
self.imapsessionpass.setChecked(prefs['imapsessionpass']) self.imapsessionpass.setChecked(prefs['imapsessionpass'])
self.l.addWidget(self.imapsessionpass,row,0,1,-1) self.l.addWidget(self.imapsessionpass,row,0,1,-1)
row+=1 row+=1
label = QLabel(_('IMAP Folder Name')) label = QLabel(_('IMAP Folder Name'))
tooltip = _("Name of IMAP folder to search for new emails. The folder (or label) has to already exist. Use INBOX for your default inbox.") tooltip = _("Name of IMAP folder to search for new emails. The folder (or label) has to already exist. Use INBOX for your default inbox.")
label.setToolTip(tooltip) label.setToolTip(tooltip)
self.l.addWidget(label,row,0) self.l.addWidget(label,row,0)
self.imapfolder = QLineEdit(self) self.imapfolder = QLineEdit(self)
self.imapfolder.setToolTip(tooltip) self.imapfolder.setToolTip(tooltip)
self.imapfolder.setText(prefs['imapfolder']) self.imapfolder.setText(prefs['imapfolder'])
self.l.addWidget(self.imapfolder,row,1) self.l.addWidget(self.imapfolder,row,1)
row+=1 row+=1
self.imapmarkread = QCheckBox(_('Mark Emails Read'),self) self.imapmarkread = QCheckBox(_('Mark Emails Read'),self)
self.imapmarkread.setToolTip(_('If checked, emails will be marked as having been read if they contain any story URLs.')) self.imapmarkread.setToolTip(_('If checked, emails will be marked as having been read if they contain any story URLs.'))
self.imapmarkread.setChecked(prefs['imapmarkread']) self.imapmarkread.setChecked(prefs['imapmarkread'])
self.l.addWidget(self.imapmarkread,row,0,1,-1) self.l.addWidget(self.imapmarkread,row,0,1,-1)
row+=1 row+=1
self.auto_reject_from_email = QCheckBox(_('Discard URLs on Reject List'),self) self.auto_reject_from_email = QCheckBox(_('Discard URLs on Reject List'),self)
self.auto_reject_from_email.setToolTip(_('If checked, FanFicFare will silently discard story URLs from emails that are on your Reject URL List.<br>Otherwise they will appear and you will see the normal Reject URL dialog.<br>The Emails will still be marked Read if configured to.')) self.auto_reject_from_email.setToolTip(_('If checked, FanFicFare will silently discard story URLs from emails that are on your Reject URL List.<br>Otherwise they will appear and you will see the normal Reject URL dialog.<br>The Emails will still be marked Read if configured to.'))
self.auto_reject_from_email.setChecked(prefs['auto_reject_from_email']) self.auto_reject_from_email.setChecked(prefs['auto_reject_from_email'])
self.l.addWidget(self.auto_reject_from_email,row,0,1,-1) self.l.addWidget(self.auto_reject_from_email,row,0,1,-1)
row+=1 row+=1
self.update_existing_only_from_email = QCheckBox(_('Update Existing Books Only'),self)
self.update_existing_only_from_email.setToolTip(_('If checked, FanFicFare will silently discard story URLs from emails that are not already in your library.<br>Otherwise all story URLs, new and existing, will be used.<br>The Emails will still be marked Read if configured to.'))
self.update_existing_only_from_email.setChecked(prefs['update_existing_only_from_email'])
self.l.addWidget(self.update_existing_only_from_email,row,0,1,-1)
row+=1
self.download_from_email_immediately = QCheckBox(_('Download from Email Immediately'),self) self.download_from_email_immediately = QCheckBox(_('Download from Email Immediately'),self)
self.download_from_email_immediately.setToolTip(_('If checked, FanFicFare will start downloading story URLs from emails immediately.<br>Otherwise the usual Download from URLs dialog will appear.')) self.download_from_email_immediately.setToolTip(_('If checked, FanFicFare will start downloading story URLs from emails immediately.<br>Otherwise the usual Download from URLs dialog will appear.'))
self.download_from_email_immediately.setChecked(prefs['download_from_email_immediately']) self.download_from_email_immediately.setChecked(prefs['download_from_email_immediately'])
self.l.addWidget(self.download_from_email_immediately,row,0,1,-1) self.l.addWidget(self.download_from_email_immediately,row,0,1,-1)
row+=1 row+=1
label = QLabel(_("<b>It's safest if you create a separate email account that you use only " label = QLabel(_("<b>It's safest if you create a separate email account that you use only "
"for your story update notices. FanFicFare and calibre cannot guarantee that " "for your story update notices. FanFicFare and calibre cannot guarantee that "
"malicious code cannot get your email password once you've entered it. " "malicious code cannot get your email password once you've entered it. "
@@ -1530,5 +1561,5 @@ class ImapTab(QWidget):
self.l.addWidget(label,row,0,1,-1,Qt.AlignTop) self.l.addWidget(label,row,0,1,-1,Qt.AlignTop)
self.l.setRowStretch(row,1) self.l.setRowStretch(row,1)
row+=1 row+=1
+54 -21
View File
@@ -21,19 +21,19 @@ from datetime import datetime
try: try:
from PyQt5 import QtWidgets as QtGui from PyQt5 import QtWidgets as QtGui
from PyQt5 import QtCore from PyQt5 import QtCore
from PyQt5.Qt import (QDialog, QTableWidget, QVBoxLayout, QHBoxLayout, QGridLayout, from PyQt5.Qt import (QDialog, QWidget, QTableWidget, QVBoxLayout, QHBoxLayout,
QPushButton, QFont, QLabel, QCheckBox, QIcon, QLineEdit, QGridLayout, QPushButton, QFont, QLabel, QCheckBox, QIcon,
QComboBox, QProgressDialog, QTimer, QDialogButtonBox, QLineEdit, QComboBox, QProgressDialog, QTimer, QDialogButtonBox,
QPixmap, Qt, QAbstractItemView, QTextEdit, pyqtSignal, QScrollArea, QPixmap, Qt, QAbstractItemView, QTextEdit,
QGroupBox, QFrame, QTextBrowser, QSize, QAction) pyqtSignal, QGroupBox, QFrame, QTextBrowser, QSize, QAction)
except ImportError as e: except ImportError as e:
from PyQt4 import QtGui from PyQt4 import QtGui
from PyQt4 import QtCore from PyQt4 import QtCore
from PyQt4.Qt import (QDialog, QTableWidget, QVBoxLayout, QHBoxLayout, QGridLayout, from PyQt4.Qt import (QDialog, QWidget, QTableWidget, QVBoxLayout, QHBoxLayout,
QPushButton, QFont, QLabel, QCheckBox, QIcon, QLineEdit, QGridLayout, QPushButton, QFont, QLabel, QCheckBox, QIcon,
QComboBox, QProgressDialog, QTimer, QDialogButtonBox, QLineEdit, QComboBox, QProgressDialog, QTimer, QDialogButtonBox,
QPixmap, Qt, QAbstractItemView, QTextEdit, pyqtSignal, QScrollArea, QPixmap, Qt, QAbstractItemView, QTextEdit,
QGroupBox, QFrame, QTextBrowser, QSize, QAction) pyqtSignal, QGroupBox, QFrame, QTextBrowser, QSize, QAction)
try: try:
from calibre.gui2 import QVariant from calibre.gui2 import QVariant
@@ -581,14 +581,34 @@ class UserPassDialog(QDialog):
self.status=False self.status=False
self.hide() self.hide()
class LoopProgressDialog(QProgressDialog): def LoopProgressDialog(gui,
book_list,
foreach_function,
finish_function,
init_label=_("Fetching metadata for stories..."),
win_title=_("Downloading metadata for stories"),
status_prefix=_("Fetched metadata for")):
ld = _LoopProgressDialog(gui,
book_list,
foreach_function,
init_label,
win_title,
status_prefix)
# Mac OS X gets upset if the finish_function is called from inside
# the real _LoopProgressDialog class.
# reflect old behavior.
if not ld.wasCanceled():
finish_function(book_list)
class _LoopProgressDialog(QProgressDialog):
''' '''
ProgressDialog displayed while fetching metadata for each story. ProgressDialog displayed while fetching metadata for each story.
''' '''
def __init__(self, gui, def __init__(self, gui,
book_list, book_list,
foreach_function, foreach_function,
finish_function,
init_label=_("Fetching metadata for stories..."), init_label=_("Fetching metadata for stories..."),
win_title=_("Downloading metadata for stories"), win_title=_("Downloading metadata for stories"),
status_prefix=_("Fetched metadata for")): status_prefix=_("Fetched metadata for")):
@@ -599,10 +619,10 @@ class LoopProgressDialog(QProgressDialog):
self.setMinimumWidth(500) self.setMinimumWidth(500)
self.book_list = book_list self.book_list = book_list
self.foreach_function = foreach_function self.foreach_function = foreach_function
self.finish_function = finish_function
self.status_prefix = status_prefix self.status_prefix = status_prefix
self.i = 0 self.i = 0
self.start_time = datetime.now() self.start_time = datetime.now()
self.first = True
# can't import at file load. # can't import at file load.
from calibre_plugins.fanficfare_plugin.prefs import prefs from calibre_plugins.fanficfare_plugin.prefs import prefs
@@ -629,8 +649,14 @@ class LoopProgressDialog(QProgressDialog):
def do_loop(self): def do_loop(self):
if self.i == 0: if self.first:
self.setValue(0) ## Windows 10 doesn't want to show the prog dialog content
## until after the timer's been called again. Something to
## do with cooperative multi threading maybe?
## So this just trips the timer loop an extra time at the start.
self.first = False
QTimer.singleShot(0, self.do_loop)
return
book = self.book_list[self.i] book = self.book_list[self.i]
try: try:
@@ -639,6 +665,7 @@ class LoopProgressDialog(QProgressDialog):
self.foreach_function(book) self.foreach_function(book)
except NotGoingToDownload as d: except NotGoingToDownload as d:
book['status']=_('Skipped')
book['good']=False book['good']=False
book['showerror']=d.showerror book['showerror']=d.showerror
book['comment']=unicode(d) book['comment']=unicode(d)
@@ -647,8 +674,7 @@ class LoopProgressDialog(QProgressDialog):
except Exception as e: except Exception as e:
book['good']=False book['good']=False
book['comment']=unicode(e) book['comment']=unicode(e)
logger.error("Exception: %s:%s"%(book,unicode(e))) logger.error("Exception: %s:%s"%(book,unicode(e)),exc_info=True)
traceback.print_exc()
self.updateStatus() self.updateStatus()
self.i += 1 self.i += 1
@@ -660,8 +686,6 @@ class LoopProgressDialog(QProgressDialog):
def do_when_finished(self): def do_when_finished(self):
self.hide() self.hide()
# Queues a job to process these books in the background.
self.finish_function(self.book_list)
def time_duration_format(seconds): def time_duration_format(seconds):
""" """
@@ -1433,6 +1457,15 @@ class ViewLog(SizePersistedDialog):
self.lineno = None self.lineno = None
scrollable = QScrollArea()
scrollcontent = QWidget()
scrollable.setWidget(scrollcontent)
scrollable.setWidgetResizable(True)
self.l.addWidget(scrollable)
self.sl = QVBoxLayout()
scrollcontent.setLayout(self.sl)
## error = (lineno, msg) ## error = (lineno, msg)
for (lineno, error_msg) in errors: for (lineno, error_msg) in errors:
# print('adding label for error:%s: %s'%(lineno, error_msg)) # print('adding label for error:%s: %s'%(lineno, error_msg))
@@ -1443,7 +1476,7 @@ class ViewLog(SizePersistedDialog):
label.setStyleSheet("QLabel { margin-left: 2em; color : blue; } QLabel:hover { color: red; }"); label.setStyleSheet("QLabel { margin-left: 2em; color : blue; } QLabel:hover { color: red; }");
label.setToolTip(_('Click to go to line %s')%lineno) label.setToolTip(_('Click to go to line %s')%lineno)
label.mouseReleaseEvent = partial(self.label_clicked, lineno=lineno) label.mouseReleaseEvent = partial(self.label_clicked, lineno=lineno)
self.l.addWidget(label) self.sl.addWidget(label)
# html='<p>'+'</p><p>'.join([ '(lineno: %s) %s'%e for e in errors ])+'</p>' # html='<p>'+'</p><p>'.join([ '(lineno: %s) %s'%e for e in errors ])+'</p>'
@@ -1453,7 +1486,7 @@ class ViewLog(SizePersistedDialog):
# self.tb.setHtml(html) # self.tb.setHtml(html)
# l.addWidget(self.tb) # l.addWidget(self.tb)
self.l.insertStretch(-1) self.sl.insertStretch(-1)
horz = QHBoxLayout() horz = QHBoxLayout()
+181 -74
View File
@@ -73,7 +73,7 @@ except:
from calibre.library.field_metadata import FieldMetadata from calibre.library.field_metadata import FieldMetadata
field_metadata = FieldMetadata() field_metadata = FieldMetadata()
from calibre_plugins.fanficfare_plugin.common_utils import ( from calibre_plugins.fanficfare_plugin.common_utils import (
set_plugin_icon_resources, get_icon, create_menu_action_unique, set_plugin_icon_resources, get_icon, create_menu_action_unique,
get_library_uuid) get_library_uuid)
@@ -83,7 +83,7 @@ from calibre_plugins.fanficfare_plugin.fanficfare import (
from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import ( from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import (
get_dcsource, get_dcsource_chaptercount, get_story_url_from_html, get_dcsource, get_dcsource_chaptercount, get_story_url_from_html,
reset_orig_chapters_epub) reset_orig_chapters_epub, get_cover_data)
from calibre_plugins.fanficfare_plugin.fanficfare.geturls import ( from calibre_plugins.fanficfare_plugin.fanficfare.geturls import (
get_urls_from_page, get_urls_from_html,get_urls_from_text, get_urls_from_page, get_urls_from_html,get_urls_from_text,
@@ -116,6 +116,13 @@ formmapping = {
'txt':'TXT' 'txt':'TXT'
} }
imagetypes = {
'image/jpeg':'jpg',
'image/png':'png',
'image/gif':'gif',
'image/svg+xml':'svg',
}
PLUGIN_ICONS = ['images/icon.png'] PLUGIN_ICONS = ['images/icon.png']
class FanFicFarePlugin(InterfaceAction): class FanFicFarePlugin(InterfaceAction):
@@ -477,9 +484,16 @@ class FanFicFarePlugin(InterfaceAction):
reject_list=set() reject_list=set()
if prefs['auto_reject_from_email']: if prefs['auto_reject_from_email']:
# need to normalize for reject list. # need to normalize for reject list.
reject_list = set([x for x in url_list if rejecturllist.check(adapters.getNormalStoryURLSite(x)[0])]) reject_list = set([x for x in url_list if rejecturllist.check(adapters.getNormalStoryURL(x))])
url_list = url_list - reject_list url_list = url_list - reject_list
## feature for update-only - check url_list with
## self.do_id_search(url)
notupdate_list = set()
if prefs['update_existing_only_from_email']:
notupdate_list = set([x for x in url_list if not self.do_id_search(adapters.getNormalStoryURL(x))])
url_list = url_list - notupdate_list
self.gui.status_bar.show_message(_('No Valid Story URLs Found in Unread Emails.'),3000) self.gui.status_bar.show_message(_('No Valid Story URLs Found in Unread Emails.'),3000)
self.restore_cursor() self.restore_cursor()
@@ -497,19 +511,21 @@ class FanFicFarePlugin(InterfaceAction):
},"\n".join(url_list)) },"\n".join(url_list))
else: else:
self.gui.status_bar.show_message(_('Finished Fetching Story URLs from Email.'),3000) self.gui.status_bar.show_message(_('Finished Fetching Story URLs from Email.'),3000)
else: else:
if url_list: if url_list:
self.add_dialog("\n".join(url_list),merge=False) self.add_dialog("\n".join(url_list),merge=False)
else: else:
msg = _('No Valid Story URLs Found in Unread Emails.') msg = _('No Valid Story URLs Found in Unread Emails.')
if reject_list: if reject_list:
msg = msg + '<p>'+(_('(%d Story URLs Skipped, on Rejected URL List)')%len(reject_list))+'</p>' msg = msg + '<p>'+(_('(%d Story URLs Skipped, on Rejected URL List)')%len(reject_list))+'</p>'
if notupdate_list:
msg = msg + '<p>'+(_("(%d Story URLs Skipped, no Existing Book in Library)")%len(notupdate_list))+'</p>'
info_dialog(self.gui, _('Get Story URLs from Email'), info_dialog(self.gui, _('Get Story URLs from Email'),
msg, msg,
show=True, show=True,
show_copy_button=False) show_copy_button=False)
def get_urls_from_page_menu(self,anthology=False): def get_urls_from_page_menu(self,anthology=False):
urltxt = "" urltxt = ""
@@ -534,7 +550,7 @@ class FanFicFarePlugin(InterfaceAction):
self.gui.status_bar.show_message(_('Finished Fetching Story URLs from Page.'),3000) self.gui.status_bar.show_message(_('Finished Fetching Story URLs from Page.'),3000)
self.restore_cursor() self.restore_cursor()
if url_list: if url_list:
self.add_dialog("\n".join(url_list),merge=d.anthology,anthology_url=url) self.add_dialog("\n".join(url_list),merge=d.anthology,anthology_url=url)
else: else:
@@ -606,7 +622,7 @@ class FanFicFarePlugin(InterfaceAction):
self.gui.status_bar.show_message(_('Can only UnNew books in library'), self.gui.status_bar.show_message(_('Can only UnNew books in library'),
3000) 3000)
return return
if not self.gui.current_view().selectionModel().selectedRows() : if not self.gui.current_view().selectionModel().selectedRows() :
self.gui.status_bar.show_message(_('No Selected Books to Get URLs From'), self.gui.status_bar.show_message(_('No Selected Books to Get URLs From'),
3000) 3000)
@@ -632,7 +648,7 @@ class FanFicFarePlugin(InterfaceAction):
suffix='.epub', suffix='.epub',
dir=tdir) dir=tdir)
db.copy_format_to(book['calibre_id'],'EPUB',tmp,index_is_id=True) db.copy_format_to(book['calibre_id'],'EPUB',tmp,index_is_id=True)
unnewtmp = PersistentTemporaryFile(prefix='unnew-%s-'%book['calibre_id'], unnewtmp = PersistentTemporaryFile(prefix='unnew-%s-'%book['calibre_id'],
suffix='.epub', suffix='.epub',
dir=tdir) dir=tdir)
@@ -656,7 +672,7 @@ class FanFicFarePlugin(InterfaceAction):
if fmt.lower() != 'epub' and db.has_format(book['calibre_id'],fmt,index_is_id=True): if fmt.lower() != 'epub' and db.has_format(book['calibre_id'],fmt,index_is_id=True):
logger.debug("autoconvert remove f:"+fmt) logger.debug("autoconvert remove f:"+fmt)
db.remove_format(book['calibre_id'], fmt, index_is_id=True)#, notify=False db.remove_format(book['calibre_id'], fmt, index_is_id=True)#, notify=False
def get_unnew_books_finish(self, book_list, tdir=None): def get_unnew_books_finish(self, book_list, tdir=None):
remove_dir(tdir) remove_dir(tdir)
if prefs['autoconvert']: if prefs['autoconvert']:
@@ -787,7 +803,7 @@ class FanFicFarePlugin(InterfaceAction):
self.busy_cursor() self.busy_cursor()
self.gui.status_bar.show_message(_('Fetching Story URLs for Series...')) self.gui.status_bar.show_message(_('Fetching Story URLs for Series...'))
# get list from identifiers:url/uri if present, but only if # get list from identifiers:url/uri if present, but only if
# it's *not* a valid story URL. # it's *not* a valid story URL.
mergeurl = self.get_story_url(db,book_id) mergeurl = self.get_story_url(db,book_id)
@@ -798,7 +814,7 @@ class FanFicFarePlugin(InterfaceAction):
self.gui.status_bar.show_message(_('Finished Fetching Story URLs for Series.'),3000) self.gui.status_bar.show_message(_('Finished Fetching Story URLs for Series.'),3000)
self.restore_cursor() self.restore_cursor()
#print("urlmapfile:%s"%urlmapfile) #print("urlmapfile:%s"%urlmapfile)
# AddNewDialog collects URLs, format and presents buttons. # AddNewDialog collects URLs, format and presents buttons.
@@ -958,7 +974,7 @@ class FanFicFarePlugin(InterfaceAction):
init_label=_("Fetching metadata for stories...") init_label=_("Fetching metadata for stories...")
win_title=_("Downloading metadata for stories") win_title=_("Downloading metadata for stories")
status_prefix=_("Fetched metadata for") status_prefix=_("Fetched metadata for")
self.gui.status_bar.show_message(status_bar, 3000) self.gui.status_bar.show_message(status_bar, 3000)
LoopProgressDialog(self.gui, LoopProgressDialog(self.gui,
books, books,
@@ -1072,7 +1088,7 @@ class FanFicFarePlugin(InterfaceAction):
## Getting metadata from configured column. ## Getting metadata from configured column.
custom_columns = self.gui.library_view.model().custom_columns custom_columns = self.gui.library_view.model().custom_columns
if ( collision in (CALIBREONLYSAVECOL) and if ( collision in (CALIBREONLYSAVECOL) and
prefs['savemetacol'] != '' and prefs['savemetacol'] != '' and
prefs['savemetacol'] in custom_columns ): prefs['savemetacol'] in custom_columns ):
savedmeta_book_id = book['calibre_id'] savedmeta_book_id = book['calibre_id']
@@ -1081,13 +1097,13 @@ class FanFicFarePlugin(InterfaceAction):
identicalbooks = self.do_id_search(url) identicalbooks = self.do_id_search(url)
if len(identicalbooks) == 1: if len(identicalbooks) == 1:
savedmeta_book_id = identicalbooks.pop() savedmeta_book_id = identicalbooks.pop()
if savedmeta_book_id: if savedmeta_book_id:
label = custom_columns[prefs['savemetacol']]['label'] label = custom_columns[prefs['savemetacol']]['label']
savedmetadata = db.get_custom(savedmeta_book_id, label=label, index_is_id=True) savedmetadata = db.get_custom(savedmeta_book_id, label=label, index_is_id=True)
else: else:
savedmetadata = None savedmetadata = None
if savedmetadata: if savedmetadata:
# sets flag inside story so getStoryMetadataOnly won't hit server. # sets flag inside story so getStoryMetadataOnly won't hit server.
adapter.setStoryMetadata(savedmetadata) adapter.setStoryMetadata(savedmetadata)
@@ -1123,19 +1139,19 @@ class FanFicFarePlugin(InterfaceAction):
if userpass.status: if userpass.status:
adapter.username = userpass.user.text() adapter.username = userpass.user.text()
adapter.password = userpass.passwd.text() adapter.password = userpass.passwd.text()
except exceptions.AdultCheckRequired: except exceptions.AdultCheckRequired:
if question_dialog(self.gui, _('Are You an Adult?'), '<p>'+ if question_dialog(self.gui, _('Are You an Adult?'), '<p>'+
_("%s requires that you be an adult. Please confirm you are an adult in your locale:")%url, _("%s requires that you be an adult. Please confirm you are an adult in your locale:")%url,
show_copy_button=False): show_copy_button=False):
adapter.is_adult=True adapter.is_adult=True
# let other exceptions percolate up. # let other exceptions percolate up.
story = adapter.getStoryMetadataOnly(get_cover=False) story = adapter.getStoryMetadataOnly(get_cover=False)
book['title'] = story.getMetadata('title') book['title'] = story.getMetadata('title')
book['author'] = [story.getMetadata('author')] book['author'] = [story.getMetadata('author')]
book['url'] = story.getMetadata('storyUrl') book['url'] = story.getMetadata('storyUrl')
## Check reject list. Redundant with below for when story ## Check reject list. Redundant with below for when story
## URL changes, but also kept here to avoid network hit in ## URL changes, but also kept here to avoid network hit in
## most common case where given url is story url. ## most common case where given url is story url.
@@ -1148,7 +1164,8 @@ class FanFicFarePlugin(InterfaceAction):
# try to find *series anthology* by *seriesUrl* identifier url or uri first. # try to find *series anthology* by *seriesUrl* identifier url or uri first.
identicalbooks = self.do_id_search(story.getMetadata('seriesUrl')) identicalbooks = self.do_id_search(story.getMetadata('seriesUrl'))
# print("identicalbooks:%s"%identicalbooks) # print("identicalbooks:%s"%identicalbooks)
if len(identicalbooks) > 0 and question_dialog(self.gui, _('Skip Story?'),''' if len(identicalbooks) > 0 and (prefs['auto_reject_seriesurlid'] or \
question_dialog(self.gui, _('Skip Story?'),'''
<h3>%s</h3> <h3>%s</h3>
<p>%s</p> <p>%s</p>
<p>%s</p> <p>%s</p>
@@ -1158,7 +1175,7 @@ class FanFicFarePlugin(InterfaceAction):
_('"<b>%s</b>" is in series "<b><a href="%s">%s</a></b>" that you have an anthology book for.')%(story.getMetadata('title'),story.getMetadata('seriesUrl'),series[:series.index(' [')]), _('"<b>%s</b>" is in series "<b><a href="%s">%s</a></b>" that you have an anthology book for.')%(story.getMetadata('title'),story.getMetadata('seriesUrl'),series[:series.index(' [')]),
_("Click '<b>Yes</b>' to Skip."), _("Click '<b>Yes</b>' to Skip."),
_("Click '<b>No</b>' to download anyway.")), _("Click '<b>No</b>' to download anyway.")),
show_copy_button=False): show_copy_button=False)):
book['comment'] = _("Story in Series Anthology(%s).")%series book['comment'] = _("Story in Series Anthology(%s).")%series
book['title'] = story.getMetadata('title') book['title'] = story.getMetadata('title')
book['author'] = [story.getMetadata('author')] book['author'] = [story.getMetadata('author')]
@@ -1167,7 +1184,7 @@ class FanFicFarePlugin(InterfaceAction):
book['icon']='rotate-right.png' book['icon']='rotate-right.png'
book['status'] = _('Skipped') book['status'] = _('Skipped')
return return
################################################################################################################################################33 ################################################################################################################################################33
book['is_adult'] = adapter.is_adult book['is_adult'] = adapter.is_adult
@@ -1176,7 +1193,7 @@ class FanFicFarePlugin(InterfaceAction):
book['icon'] = 'plus.png' book['icon'] = 'plus.png'
book['status'] = _('Add') book['status'] = _('Add')
if not bgmeta: if not bgmeta:
# set PI version instead of default. # set PI version instead of default.
if 'version' in options: if 'version' in options:
@@ -1187,7 +1204,7 @@ class FanFicFarePlugin(InterfaceAction):
if prefs['savemetacol'] != '': if prefs['savemetacol'] != '':
# get metadata to save in configured column. # get metadata to save in configured column.
book['savemetacol'] = story.dump_html_metadata() book['savemetacol'] = story.dump_html_metadata()
book['title'] = story.getMetadata("title", removeallentities=True) book['title'] = story.getMetadata("title", removeallentities=True)
book['author_sort'] = book['author'] = story.getList("author", removeallentities=True) book['author_sort'] = book['author'] = story.getList("author", removeallentities=True)
book['publisher'] = story.getMetadata("site") book['publisher'] = story.getMetadata("site")
@@ -1198,7 +1215,7 @@ class FanFicFarePlugin(InterfaceAction):
else: else:
book['comments']='' book['comments']=''
book['series'] = story.getMetadata("series", removeallentities=True) book['series'] = story.getMetadata("series", removeallentities=True)
if story.getMetadataRaw('datePublished'): if story.getMetadataRaw('datePublished'):
book['pubdate'] = story.getMetadataRaw('datePublished').replace(tzinfo=local_tz) book['pubdate'] = story.getMetadataRaw('datePublished').replace(tzinfo=local_tz)
if story.getMetadataRaw('dateUpdated'): if story.getMetadataRaw('dateUpdated'):
@@ -1207,7 +1224,7 @@ class FanFicFarePlugin(InterfaceAction):
book['timestamp'] = story.getMetadataRaw('dateCreated').replace(tzinfo=local_tz) book['timestamp'] = story.getMetadataRaw('dateCreated').replace(tzinfo=local_tz)
else: else:
book['timestamp'] = None # need *something* there for calibre. book['timestamp'] = None # need *something* there for calibre.
if not merge:# skip all the collision code when d/ling for merging. if not merge:# skip all the collision code when d/ling for merging.
if collision in (CALIBREONLY, CALIBREONLYSAVECOL): if collision in (CALIBREONLY, CALIBREONLYSAVECOL):
book['icon'] = 'metadata.png' book['icon'] = 'metadata.png'
@@ -1334,7 +1351,7 @@ class FanFicFarePlugin(InterfaceAction):
# check make sure incoming is newer. # check make sure incoming is newer.
lastupdated=story.getMetadataRaw('dateUpdated') lastupdated=story.getMetadataRaw('dateUpdated')
logger.debug("OVERWRITE site updated: %s"%lastupdated) logger.debug("OVERWRITE site updated: %s"%lastupdated)
# updated doesn't have time (or is midnight), use dates only. # updated doesn't have time (or is midnight), use dates only.
# updated does have time, use full timestamps. # updated does have time, use full timestamps.
if (lastupdated.time() == time.min and fileupdated.date() > lastupdated.date()) or \ if (lastupdated.time() == time.min and fileupdated.date() > lastupdated.date()) or \
@@ -1377,19 +1394,19 @@ class FanFicFarePlugin(InterfaceAction):
if not label and k in field_metadata: if not label and k in field_metadata:
label=field_metadata[k]['name'] label=field_metadata[k]['name']
key='calibre_std_'+k key='calibre_std_'+k
# if k == 'user_categories': # if k == 'user_categories':
# value=u', '.join(mi.get(k)) # value=u', '.join(mi.get(k))
# label=_('User Categories') # label=_('User Categories')
if label: # only if it has a human readable name. if label: # only if it has a human readable name.
if value is None or not book['calibre_id']: if value is None or not book['calibre_id']:
## if existing book, populate existing calibre column ## if existing book, populate existing calibre column
## values in metadata, else '' to hide. ## values in metadata, else '' to hide.
value='' value=''
book['calibre_columns'][key]={'val':value,'label':label} book['calibre_columns'][key]={'val':value,'label':label}
#logger.debug("%s(%s): %s"%(label,key,value)) #logger.debug("%s(%s): %s"%(label,key,value))
# custom columns # custom columns
for k, column in self.gui.library_view.model().custom_columns.iteritems(): for k, column in self.gui.library_view.model().custom_columns.iteritems():
if k != prefs['savemetacol']: if k != prefs['savemetacol']:
@@ -1405,7 +1422,7 @@ class FanFicFarePlugin(InterfaceAction):
value='' value=''
book['calibre_columns'][key]={'val':value,'label':label} book['calibre_columns'][key]={'val':value,'label':label}
# logger.debug("%s(%s): %s"%(label,key,value)) # logger.debug("%s(%s): %s"%(label,key,value))
# if still 'good', make a temp file to write the output to. # if still 'good', make a temp file to write the output to.
# For HTML format users, make the filename inside the zip something reasonable. # For HTML format users, make the filename inside the zip something reasonable.
# For crazy long titles/authors, limit it to 200chars. # For crazy long titles/authors, limit it to 200chars.
@@ -1472,10 +1489,20 @@ class FanFicFarePlugin(InterfaceAction):
htmllog = htmllog + '</table></body></html>' htmllog = htmllog + '</table></body></html>'
payload = ([], book_list, options) payload = ([], book_list, options)
self.gui.proceed_question(self.update_error_column,
payload, htmllog, # log_viewer_unique_name implemented here: https://github.com/kovidgoyal/calibre/compare/v2.56.0...v2.57.0
_('FanFicFare log'), _('FanFicFare download ended'), msg, if calibre_version >= (2, 57, 0):
show_copy_button=False) self.gui.proceed_question(self.update_error_column,
payload, htmllog,
_('FanFicFare log'), _('FanFicFare download ended'), msg,
show_copy_button=False,
log_viewer_unique_name="FanFicFare log viewer")
else:
self.gui.proceed_question(self.update_error_column,
payload, htmllog,
_('FanFicFare log'), _('FanFicFare download ended'), msg,
show_copy_button=False)
return return
cookiejarfile = PersistentTemporaryFile(suffix='.cookiejar', cookiejarfile = PersistentTemporaryFile(suffix='.cookiejar',
@@ -1709,12 +1736,21 @@ class FanFicFarePlugin(InterfaceAction):
do_update_func = self.do_download_list_update do_update_func = self.do_download_list_update
self.gui.proceed_question(do_update_func, # log_viewer_unique_name implemented here: https://github.com/kovidgoyal/calibre/compare/v2.56.0...v2.57.0
payload, htmllog, if calibre_version >= (2, 57, 0):
_('FanFicFare log'), _('FanFicFare download complete'), msg, self.gui.proceed_question(do_update_func,
show_copy_button=False) payload, htmllog,
_('FanFicFare log'), _('FanFicFare download complete'), msg,
show_copy_button=False,
log_viewer_unique_name="FanFicFare log viewer")
else:
self.gui.proceed_question(do_update_func,
payload, htmllog,
_('FanFicFare log'), _('FanFicFare download complete'), msg,
show_copy_button=False)
def do_download_merge_update(self, payload): def do_download_merge_update(self, payload):
db = self.gui.current_db
(good_list,bad_list,options) = payload (good_list,bad_list,options) = payload
total_good = len(good_list) total_good = len(good_list)
@@ -1740,19 +1776,62 @@ class FanFicFarePlugin(InterfaceAction):
#print("mergebook:\n%s"%mergebook) #print("mergebook:\n%s"%mergebook)
if mergebook['good']: # there shouldn't be any !'good' books at this point. # make a temp file to write the output to.
# if still 'good', make a temp file to write the output to. tmp = PersistentTemporaryFile(suffix='.'+options['fileform'],
tmp = PersistentTemporaryFile(suffix='.'+options['fileform'], dir=options['tdir'])
dir=options['tdir']) # logger.debug("title:"+mergebook['title'])
logger.debug("title:"+mergebook['title']) logger.debug("outfile:"+tmp.name)
logger.debug("outfile:"+tmp.name) mergebook['outfile'] = tmp.name
mergebook['outfile'] = tmp.name
## Calibre's Polish heuristics for covers can cause problems
## if a merged anthology book has sub-book covers, but not a
## proper main cover. So, now if there are any covers, we
## will force a main cover.
## start with None. If no subbook covers, don't force one
## here. User can configure FFF to always create/polish a
## cover if they want. This is about when we force it.
coverpath = None
coverimgtype = None
## first, look for covers inside the subbooks. Stop at the
## first one, which will be used if there isn't a pre-existing
## calibre cover.
if not coverpath:
for book in good_list:
coverdata = get_cover_data(book['outfile'])
if coverdata: # found a cover.
(coverimgtype,coverimgdata) = coverdata[4:6]
logger.debug('coverimgtype:%s [%s]'%(coverimgtype,imagetypes[coverimgtype]))
tmpcover = PersistentTemporaryFile(suffix='.'+imagetypes[coverimgtype],
dir=options['tdir'])
tmpcover.write(coverimgdata)
tmpcover.flush()
tmpcover.close()
coverpath = tmpcover.name
break
# logger.debug('coverpath:%s'%coverpath)
## if updating an existing book and there is at least one
## subbook cover:
if coverpath and mergebook['calibre_id']:
# Couldn't find a better way to get the cover path.
calcoverpath = os.path.join(db.library_path,
db.path(mergebook['calibre_id'], index_is_id=True),
'cover.jpg')
## if there's an existing cover, use it. Calibre will set
## it for us during lots of different actions anyway.
if os.path.exists(calcoverpath):
coverpath = calcoverpath
# logger.debug('coverpath:%s'%coverpath)
self.get_epubmerge_plugin().do_merge(tmp.name, self.get_epubmerge_plugin().do_merge(tmp.name,
[ x['outfile'] for x in good_list ], [ x['outfile'] for x in good_list ],
tags=mergebook['tags'],
titleopt=mergebook['title'], titleopt=mergebook['title'],
keepmetadatafiles=True, keepmetadatafiles=True,
source=mergebook['url']) source=mergebook['url'],
coverjpgpath=coverpath)
options['collision']=OVERWRITEALWAYS options['collision']=OVERWRITEALWAYS
self.update_books_loop(mergebook,self.gui.current_db,options) self.update_books_loop(mergebook,self.gui.current_db,options)
@@ -1872,7 +1951,8 @@ class FanFicFarePlugin(InterfaceAction):
#print("mi.tags:%s"%mi.tags) #print("mi.tags:%s"%mi.tags)
if book['all_metadata']['langcode']: if book['all_metadata']['langcode']:
mi.languages=[book['all_metadata']['langcode']] # split due to anthologies. Gives list of one for non-anth.
mi.languages=book['all_metadata']['langcode'].split(', ')
else: else:
# Set language english, but only if not already set. # Set language english, but only if not already set.
if not oldmi.languages: if not oldmi.languages:
@@ -1939,14 +2019,17 @@ class FanFicFarePlugin(InterfaceAction):
configuration = None configuration = None
if prefs['allow_custcol_from_ini']: if prefs['allow_custcol_from_ini']:
configuration = get_fff_config(book['url'],options['fileform']) configuration = get_fff_config(book['url'],options['fileform'])
# meta => custcol[,a|n|r] # meta => custcol[,a|n|r|n_anthaver,r_anthaver]
# cliches=>\#acolumn,r # cliches=>\#acolumn,r
for line in configuration.getConfig('custom_columns_settings').splitlines(): for line in configuration.getConfig('custom_columns_settings').splitlines():
if "=>" in line: if "=>" in line:
(meta,custcol) = map( lambda x: x.strip(), line.split("=>") ) (meta,custcol) = map( lambda x: x.strip(), line.split("=>") )
flag='r' flag='r'
anthaver=False
if "," in custcol: if "," in custcol:
(custcol,flag) = map( lambda x: x.strip(), custcol.split(",") ) (custcol,flag) = map( lambda x: x.strip(), custcol.split(",") )
anthaver = 'anthaver' in flag
flag=flag[0] # first char only.
if meta not in book['all_metadata']: if meta not in book['all_metadata']:
# if double quoted, use as a literal value. # if double quoted, use as a literal value.
@@ -1968,8 +2051,15 @@ class FanFicFarePlugin(InterfaceAction):
if flag == 'r' or (flag == 'n' and book['added']): if flag == 'r' or (flag == 'n' and book['added']):
if coldef['datatype'] in ('int','float'): # for favs, etc--site specific metadata. if coldef['datatype'] in ('int','float'): # for favs, etc--site specific metadata.
if 'anthology_meta_list' in book and meta in book['anthology_meta_list']: if 'anthology_meta_list' in book and meta in book['anthology_meta_list']:
# re-split list, strip commas, convert to floats, sum up. # re-split list, strip commas, convert to floats
val = sum([ float(x.replace(",","")) for x in val.split(", ") ]) items = [ float(x.replace(",","")) for x in val.split(", ") ]
if anthaver:
if items:
val = sum(items) / float(len(items))
else:
val = 0
else:
val = sum(items)
else: else:
val = unicode(val).replace(",","") val = unicode(val).replace(",","")
else: else:
@@ -2042,7 +2132,7 @@ class FanFicFarePlugin(InterfaceAction):
prefs['gencalcover'] == SAVE_YES ## yes, always prefs['gencalcover'] == SAVE_YES ## yes, always
or (prefs['gencalcover'] == SAVE_YES_UNLESS_IMG ## yes, unless image. or (prefs['gencalcover'] == SAVE_YES_UNLESS_IMG ## yes, unless image.
and book['all_metadata']['cover_image'] not in ('specific','first','default')) ): and book['all_metadata']['cover_image'] not in ('specific','first','default')) ):
cover_generated = False # flag for polish below. cover_generated = False # flag for polish below.
# Yes, should do gencov. Which? # Yes, should do gencov. Which?
if prefs['calibre_gen_cover'] and HAS_CALGC: if prefs['calibre_gen_cover'] and HAS_CALGC:
@@ -2055,19 +2145,19 @@ class FanFicFarePlugin(InterfaceAction):
cover_generated = True cover_generated = True
elif prefs['plugin_gen_cover'] and 'Generate Cover' in self.gui.iactions: elif prefs['plugin_gen_cover'] and 'Generate Cover' in self.gui.iactions:
# plugin, if available. # plugin, if available.
#logger.debug("Do Generate Cover added:%s gcnewonly:%s"%(book['added'],prefs['gcnewonly'])) #logger.debug("Do Generate Cover added:%s gcnewonly:%s"%(book['added'],prefs['gcnewonly']))
# force a refresh if generating cover so complex composite # force a refresh if generating cover so complex composite
# custom columns are current and correct # custom columns are current and correct
db.refresh_ids([book_id]) db.refresh_ids([book_id])
gc_plugin = self.gui.iactions['Generate Cover'] gc_plugin = self.gui.iactions['Generate Cover']
setting_name = None setting_name = None
if prefs['allow_gc_from_ini']: if prefs['allow_gc_from_ini']:
if not configuration: # might already have it from allow_custcol_from_ini if not configuration: # might already have it from allow_custcol_from_ini
configuration = get_fff_config(book['url'],options['fileform']) configuration = get_fff_config(book['url'],options['fileform'])
# template => regexp to match => GC Setting to use. # template => regexp to match => GC Setting to use.
# generate_cover_settings: # generate_cover_settings:
# ${category} => Buffy:? the Vampire Slayer => Buffy # ${category} => Buffy:? the Vampire Slayer => Buffy
@@ -2076,25 +2166,25 @@ class FanFicFarePlugin(InterfaceAction):
# (template,regexp,setting) = map( lambda x: x.strip(), line.split("=>") ) # (template,regexp,setting) = map( lambda x: x.strip(), line.split("=>") )
for (template,regexp,setting) in configuration.get_generate_cover_settings(): for (template,regexp,setting) in configuration.get_generate_cover_settings():
value = Template(template).safe_substitute(book['all_metadata']).encode('utf8') value = Template(template).safe_substitute(book['all_metadata']).encode('utf8')
print("%s(%s) => %s => %s"%(template,value,regexp,setting)) # print("%s(%s) => %s => %s"%(template,value,regexp,setting))
if re.search(regexp,value): if re.search(regexp,value):
setting_name = setting setting_name = setting
break break
if setting_name: if setting_name:
logger.debug("Generate Cover Setting from generate_cover_settings(%s)"%setting_name) logger.debug("Generate Cover Setting from generate_cover_settings(%s)"%setting_name)
if setting_name not in gc_plugin.get_saved_setting_names(): if setting_name not in gc_plugin.get_saved_setting_names():
logger.info("GC Name %s not found, discarding! (check personal.ini for typos)"%setting_name) logger.info("GC Name %s not found, discarding! (check personal.ini for typos)"%setting_name)
setting_name = None setting_name = None
if not setting_name and book['all_metadata']['site'] in prefs['gc_site_settings']: if not setting_name and book['all_metadata']['site'] in prefs['gc_site_settings']:
setting_name = prefs['gc_site_settings'][book['all_metadata']['site']] setting_name = prefs['gc_site_settings'][book['all_metadata']['site']]
logger.debug("Generate Cover Setting from site(%s)"%setting_name) logger.debug("Generate Cover Setting from site(%s)"%setting_name)
if not setting_name and 'Default' in prefs['gc_site_settings']: if not setting_name and 'Default' in prefs['gc_site_settings']:
setting_name = prefs['gc_site_settings']['Default'] setting_name = prefs['gc_site_settings']['Default']
logger.debug("Generate Cover Setting from Default(%s)"%setting_name) logger.debug("Generate Cover Setting from Default(%s)"%setting_name)
if setting_name: if setting_name:
logger.debug("Running Generate Cover with settings %s."%setting_name) logger.debug("Running Generate Cover with settings %s."%setting_name)
## fetch updated mi object from ## fetch updated mi object from
@@ -2103,14 +2193,14 @@ class FanFicFarePlugin(InterfaceAction):
realmi = db.get_metadata(book_id, index_is_id=True) realmi = db.get_metadata(book_id, index_is_id=True)
gc_plugin.generate_cover_for_book(realmi,saved_setting_name=setting_name) gc_plugin.generate_cover_for_book(realmi,saved_setting_name=setting_name)
cover_generated = True cover_generated = True
if cover_generated and prefs['gc_polish_cover'] and \ if cover_generated and prefs['gc_polish_cover'] and \
options['fileform'] == "epub": options['fileform'] == "epub":
# set cover inside epub from calibre's polish feature # set cover inside epub from calibre's polish feature
from calibre.ebooks.oeb.polish.main import polish, ALL_OPTS from calibre.ebooks.oeb.polish.main import polish, ALL_OPTS
from calibre.utils.logging import Log from calibre.utils.logging import Log
from collections import namedtuple from collections import namedtuple
# Couldn't find a better way to get the cover path. # Couldn't find a better way to get the cover path.
cover_path = os.path.join(db.library_path, cover_path = os.path.join(db.library_path,
db.path(book_id, index_is_id=True), db.path(book_id, index_is_id=True),
@@ -2121,7 +2211,7 @@ class FanFicFarePlugin(InterfaceAction):
opts.update(data) opts.update(data)
O = namedtuple('Options', ' '.join(ALL_OPTS.iterkeys())) O = namedtuple('Options', ' '.join(ALL_OPTS.iterkeys()))
opts = O(**opts) opts = O(**opts)
log = Log(level=Log.DEBUG) log = Log(level=Log.DEBUG)
outfile = db.format_abspath(book_id, outfile = db.format_abspath(book_id,
formmapping[options['fileform']], formmapping[options['fileform']],
@@ -2381,7 +2471,9 @@ class FanFicFarePlugin(InterfaceAction):
#print("book series:%s"%serieslist[-1]) #print("book series:%s"%serieslist[-1])
if b['publisher']: if b['publisher']:
if 'publisher' not in book: if not book['publisher']:
## not set in all_metadata because it's not one of
## the permitted metadata--use site instead.
book['publisher']=b['publisher'] book['publisher']=b['publisher']
elif book['publisher']!=b['publisher']: elif book['publisher']!=b['publisher']:
book['publisher']=None # if any are different, don't use. book['publisher']=None # if any are different, don't use.
@@ -2439,18 +2531,33 @@ class FanFicFarePlugin(InterfaceAction):
book['anthology_meta_list'][k]=True book['anthology_meta_list'][k]=True
logger.debug("book['url']:%s"%book['url']) logger.debug("book['url']:%s"%book['url'])
book['comments'] = _("Anthology containing:")+"\n\n"
if len(book['author']) > 1:
mkbooktitle = lambda x : _("%s by %s") % (x['title'],' & '.join(x['author']))
else:
mkbooktitle = lambda x : x['title']
if prefs['includecomments']:
def mkbookcomments(x):
if x['comments']:
return '<b>%s</b>\n\n%s'%(mkbooktitle(x),x['comments'])
else:
return '<b>%s</b>\n'%mkbooktitle(x)
book['comments'] += ('<div class="mergedbook">' +
'<hr></div><div class="mergedbook">'.join([ mkbookcomments(x) for x in book_list]) +
'</div>')
else:
book['comments'] += '\n'.join( [ mkbooktitle(x) for x in book_list ] )
configuration = get_fff_config(book['url'],fileform) configuration = get_fff_config(book['url'],fileform)
if existingbook: if existingbook:
book['title'] = deftitle = existingbook['title'] book['title'] = deftitle = existingbook['title']
book['comments'] = existingbook['comments'] if prefs['anth_comments_newonly']:
book['comments'] = existingbook['comments']
else: else:
book['title'] = deftitle = book_list[0]['title'] book['title'] = deftitle = book_list[0]['title']
if len(book['author']) > 1:
book['comments'] = _("Anthology containing:")+"\n" + \
"\n".join([ _("%s by %s")%(b['title'],', '.join(b['author'])) for b in book_list ])
else:
book['comments'] = _("Anthology containing:")+"\n" + \
"\n".join([ b['title'] for b in book_list ])
# book['all_metadata']['description'] # book['all_metadata']['description']
# if all same series, use series for name. But only if all and not previous named # if all same series, use series for name. But only if all and not previous named
@@ -2487,7 +2594,7 @@ class FanFicFarePlugin(InterfaceAction):
def restore_cursor(self): def restore_cursor(self):
QApplication.restoreOverrideCursor() QApplication.restoreOverrideCursor()
def split_text_to_urls(urls): def split_text_to_urls(urls):
# remove dups while preserving order. # remove dups while preserving order.
dups=set() dups=set()
+2 -3
View File
@@ -161,7 +161,7 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
adapter.set_pagecache(options['pagecache']) adapter.set_pagecache(options['pagecache'])
story = adapter.getStoryMetadataOnly() story = adapter.getStoryMetadataOnly()
if 'calibre_series' in book: if not story.getMetadata("series") and 'calibre_series' in book:
adapter.setSeries(book['calibre_series'][0],book['calibre_series'][1]) adapter.setSeries(book['calibre_series'][0],book['calibre_series'][1])
# set PI version instead of default. # set PI version instead of default.
@@ -319,8 +319,7 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
book['comment']=unicode(e) book['comment']=unicode(e)
book['icon']='dialog_error.png' book['icon']='dialog_error.png'
book['status'] = _('Error') book['status'] = _('Error')
logger.info("Exception: %s:%s"%(book,unicode(e))) logger.info("Exception: %s:%s"%(book,unicode(e)),exc_info=True)
traceback.print_exc()
#time.sleep(10) #time.sleep(10)
return book return book
+162 -119
View File
@@ -25,7 +25,7 @@
## titlepage_entries: category,genre, status,dateUpdated,rating ## titlepage_entries: category,genre, status,dateUpdated,rating
## [epub] ## [epub]
## # overrides defaults & site section ## # overrides defaults & site section
## titlepage_entries: category,genre, status,datePublished,dateUpdated,dateCreated ## titlepage_entries: category,genre,status,datePublished,dateUpdated,dateCreated
## [www.whofic.com:epub] ## [www.whofic.com:epub]
## # overrides defaults, site section & format section ## # overrides defaults, site section & format section
## titlepage_entries: category,genre, status,datePublished ## titlepage_entries: category,genre, status,datePublished
@@ -150,7 +150,7 @@ extratags: FanFiction
## Can also be used for other metadata values ## Can also be used for other metadata values
#default_value_category:FanFiction #default_value_category:FanFiction
## number of seconds to sleep between calls to the story site. May by ## number of seconds to sleep between calls to the story site. May be
## useful if pulling large numbers of stories or if the site is slow. ## useful if pulling large numbers of stories or if the site is slow.
#slow_down_sleep_time:0.5 #slow_down_sleep_time:0.5
@@ -474,8 +474,8 @@ add_to_include_subject_tags:,tagsfromtitle.SPLIT,forumtags
## base_xenforoforum reads Published and Updated datetimes from ## base_xenforoforum reads Published and Updated datetimes from
## Threadmarks if used, or from the posted & updated times of the ## Threadmarks if used, or from the posted & updated times of the
## 'first' post if no threadmarks. ## 'first' post if no threadmarks.
datePublished_format:%%Y-%%m-%%d %%H:%%M:%%S datePublished_format:%%Y-%%m-%%d %%H:%%M
dateUpdated_format:%%Y-%%m-%%d %%H:%%M:%%S dateUpdated_format:%%Y-%%m-%%d %%H:%%M
## Only take the first X characters of the 'first' post to use as ## Only take the first X characters of the 'first' post to use as
## the description. ## the description.
@@ -510,6 +510,20 @@ first_post_title:First Post
## if threadmarks are used. Can result in a duplicated chapter. ## if threadmarks are used. Can result in a duplicated chapter.
always_include_first_post:false always_include_first_post:false
## In normal operation, when updating an existing epub, old chapters
## will be reused as-is. Normally, that works fine, but forum stories
## sometimes have an index post as the first 'chapter', and the
## version in the first chapter gets out of sync.
##
## If always_reload_first_chapter:true, then the first chapter will
## always be downloaded again (ie, reloaded). It will NOT be maked
## '(new)' (see mark_new_chapters). Because it is reloaded, manual
## edits made to the first chapter will be lost.
##
## While intended for base_xenforoforum sites, this setting can be
## applied to other sites.
always_reload_first_chapter:false
## In normal operation, forumtags will only be populated when ## In normal operation, forumtags will only be populated when
## threadmarks are used for chapters (see minimum_threadmarks above). ## threadmarks are used for chapters (see minimum_threadmarks above).
## When always_use_forumtags:true, always populate forumtags. ## When always_use_forumtags:true, always populate forumtags.
@@ -585,6 +599,11 @@ include_logpage: false
## end up with Completed stories that have just one logpage entry. ## end up with Completed stories that have just one logpage entry.
#include_logpage: smart #include_logpage: smart
## By default, logpage is placed before the story chapters. This
## setting, if true, will place the logpage after the chapters
## instead.
logpage_at_end: false
## items to include in the log page Empty metadata entries, or those ## items to include in the log page Empty metadata entries, or those
## that haven't changed since the last update, will *not* appear, even ## that haven't changed since the last update, will *not* appear, even
## if in the list. You can include extra text or HTML that will be ## if in the list. You can include extra text or HTML that will be
@@ -701,6 +720,18 @@ remove_transparency: true
## true--replace_br_with_p also fixes the problem. ## true--replace_br_with_p also fixes the problem.
nook_img_fix:true nook_img_fix:true
## Apply adapter's normalize_chapterurl() to all links in chapter
## texts, if they match chapter URLs. Currently only implemented by
## base_xenforoforum adapters.
#normalize_text_links:false
## Search all links in chapter texts and, if they match any included
## chapter URLs, replace them with links to the chapter in the
## download. Only works with epub and html output formats.
## base_xenforoforum adapters should also use normalize_text_links
## with this.
#internalize_text_links:false
[mobi] [mobi]
## mobi TOC cannot be turned off right now. ## mobi TOC cannot be turned off right now.
#include_tocpage: true #include_tocpage: true
@@ -748,6 +779,19 @@ extratags: FanFiction,Testing,Text
[test1.com:html] [test1.com:html]
extratags: FanFiction,Testing,HTML extratags: FanFiction,Testing,HTML
[adult-fanfiction.org]
extra_valid_entries:eroticatags,disclaimer
eroticatags_label:Erotica Tags
disclaimer_label:Disclaimer
extra_titlepage_entries:eroticatags,disclaimer
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[archive.skyehawke.com] [archive.skyehawke.com]
[archiveofourown.org] [archiveofourown.org]
@@ -847,18 +891,6 @@ extracategories:The Sentinel
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[bdsm-geschichten.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## This site offers no index page so we can either guess the chapter URLs
## by dec/incrementing numbers ('guess') or walk all the chapters in the metadata
## parsing state ('parse'). Since guessing can lead to errors for non-standard
## story URLs, the default is to parse
#find_chapters:guess
[bloodshedverse.com] [bloodshedverse.com]
## website encoding(s) In theory, each website reports the character ## website encoding(s) In theory, each website reports the character
## encoding they use for each page. In practice, some sites report it ## encoding they use for each page. In practice, some sites report it
@@ -869,6 +901,12 @@ extracategories:The Sentinel
## it has +90% confidence. 'auto' is not reliable. ## it has +90% confidence. 'auto' is not reliable.
website_encodings:Windows-1252,ISO-8859-1,auto website_encodings:Windows-1252,ISO-8859-1,auto
## dateUpdate doesn't usually have time, but it does on
## bloodshedverse.com. See
## http://docs.python.org/library/datetime.html#strftime-strptime-behavior
## Note that ini format requires % to be escaped as %%.
dateUpdated_format:%%Y-%%m-%%d %%H:%%M
## Extra metadata that this adapter knows about. See [dramione.org] ## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them. ## for examples of how to use them.
extra_valid_entries:warnings,reviews extra_valid_entries:warnings,reviews
@@ -898,15 +936,6 @@ strip_text_links:true
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Blood Ties extracategories:Blood Ties
[buffynfaith.net]
## Site dedicated to these categories/characters/ships
extracategories:Buffy: The Vampire Slayer
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
[fanfic.castletv.net] [fanfic.castletv.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1023,15 +1052,24 @@ cliches_label:Character Cliches
## 'mode'. 'r' to Replace any existing values, 'a' to Add to existing ## 'mode'. 'r' to Replace any existing values, 'a' to Add to existing
## value (use with tag-like columns), and 'n' for setting on New books ## value (use with tag-like columns), and 'n' for setting on New books
## only. (Default is 'r'.) ## only. (Default is 'r'.)
## Literal strings can be set into custom columns using double quotes. ## Literal strings can be set into custom columns using double quotes.
## Each metadata=>column mapping must be on a separate line and each ## Each metadata=>column mapping must be on a separate line and each
## needs to have one space at the start of each line. ## needs to have one space at the start of each line.
## 'r_anthaver' and 'n_anthaver' can be used to indicate the same as
## 'r' and 'n' for normal downloads, but to average the metadata for
## the differents story in an anthology before setting in integer and
## float type custom columns. This can be useful for a averrating
## column, for example. Default is to sum the values of all stories,
## and numChapters and numWords are always summed.
#custom_columns_settings: #custom_columns_settings:
# cliches=>#acolumn # cliches=>#acolumn
# themes=>#bcolumn,a # themes=>#bcolumn,a
# timeline=>#ccolumn,n # timeline=>#ccolumn,n
# "FanFiction"=>#collection # "FanFiction"=>#collection
# averrating=>#averrating,r_anthaver
[efiction.esteliel.de] [efiction.esteliel.de]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
@@ -1151,13 +1189,20 @@ romance_label: Romance
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[ficwad.com] [fictionhunt.com]
## Some sites require login (or login for some rated stories) The ## Archive only site for ffnet HP stories.
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not ## Site dedicated to these categories/characters/ships
## defaults.ini. extracategories:Harry Potter
#username:YourName
#password:yourpassword extra_valid_entries: origin,originUrl,originHTML,reviews
originHTML_label:Original Story URL
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:origin
add_to_extra_titlepage_entries:originHTML
[fictionmania.tv] [fictionmania.tv]
## website encoding(s) In theory, each website reports the character ## website encoding(s) In theory, each website reports the character
@@ -1215,6 +1260,14 @@ views_label:Views
likes_label:Likes likes_label:Likes
dislikes_label:Dislikes dislikes_label:Dislikes
[ficwad.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[finestories.com] [finestories.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1256,13 +1309,9 @@ universe_as_series: true
## cover image. This lets you exclude them. ## cover image. This lets you exclude them.
cover_exclusion_regexp:/css/bir.png cover_exclusion_regexp:/css/bir.png
[forums.spacebattles.com] [forum.questionablequesting.com]
## see [base_xenforoforum] ## see [base_xenforoforum]
[forums.sufficientvelocity.com]
## see [base_xenforoforum]
[grangerenchanted.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not ## commandline version, this should go in your personal.ini, not
@@ -1270,18 +1319,24 @@ cover_exclusion_regexp:/css/bir.png
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
## Some sites also require the user to confirm they are adult for [forums.spacebattles.com]
## adult content. In commandline version, this should go in your ## see [base_xenforoforum]
## personal.ini, not defaults.ini.
[forums.sufficientvelocity.com]
## see [base_xenforoforum]
[harem.lucifael.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
## Site dedicated to these categories/characters/ships ## Some sites require login (or login for some rated stories) The
extracategories:Harry Potter ## program can prompt you, or you can save it in config. In
extracharacters:Hermione Granger ## commandline version, this should go in your personal.ini, not
## defaults.ini.
## Extra metadata that this adapter knows about. See [dramione.org] #username:YourName
## for examples of how to use them. #password:yourpassword
extra_valid_entries:read,reviews
[hlfiction.net] [hlfiction.net]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
@@ -1395,6 +1450,19 @@ extra_titlepage_entries: eroticatags
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Merlin extracategories:Merlin
[mujaji.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
[national-library.net] [national-library.net]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:West Wing extracategories:West Wing
@@ -1496,15 +1564,21 @@ extracategories:My Little Pony: Friendship is Magic
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:The Pretender extracategories:The Pretender
[forum.questionablequesting.com] [quotev.com]
## see [base_xenforoforum] extra_valid_entries:pages,readers,reads,favorites,searchtags,comments
pages_label:Pages
readers_label:Readers
reads_label:Reads
favorites_label:Favorites
searchtags_label:Search Tags
comments_label:Comments
## Some sites require login (or login for some rated stories) The include_in_category:category,searchtags
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not [royalroadl.com]
## defaults.ini. extra_valid_entries:stars
#username:YourName
#password:yourpassword #add_to_extra_titlepage_entries:,stars
[samandjack.net] [samandjack.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
@@ -1530,22 +1604,6 @@ extracategories:Supernatural
extracharacters:Sam,Dean extracharacters:Sam,Dean
extraships:Sam/Dean extraships:Sam/Dean
[scarhead.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
[sheppardweir.com] [sheppardweir.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1687,15 +1745,6 @@ readings_label: Readings
## personal.ini, not defaults.ini. ## personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[tokra.fandomnet.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1
[tolkienfanfiction.com] [tolkienfanfiction.com]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Lord of the Rings extracategories:Lord of the Rings
@@ -1763,6 +1812,12 @@ extraships:Draco Malfoy/Ginny Weasley
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1 extracategories:Stargate: SG-1
[www.deepinmysoul.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[www.destinysgateway.com] [www.destinysgateway.com]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
@@ -1857,6 +1912,10 @@ check_next_chapter:false
#password:yourpassword #password:yourpassword
[www.ficbook.net] [www.ficbook.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[www.fictionalley.org] [www.fictionalley.org]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
@@ -2059,11 +2118,6 @@ cover_exclusion_regexp:/stories/999/images/.*?_trophy.png
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:NCIS extracategories:NCIS
[www.nickngreg.nl]
## Site dedicated to these categories/characters/ships
extracategories:CSI
extraships:Nick Stokes/Greg Sanders
[www.phoenixsong.net] [www.phoenixsong.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -2118,6 +2172,8 @@ extracategories:Psych
extracategories:Queer as Folk extracategories:Queer as Folk
[quotev.com] [quotev.com]
user_agent:
slow_down_sleep_time:2
extra_valid_entries:pages,readers,reads,favorites,searchtags,comments extra_valid_entries:pages,readers,reads,favorites,searchtags,comments
pages_label:Pages pages_label:Pages
readers_label:Readers readers_label:Readers
@@ -2194,23 +2250,9 @@ extracategories:Lord of the Rings
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
[www.twcslibrary.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## twcslibrary.net (ab)uses series as personal reading lists.
collect_series: false
[www.tthfanfic.org] [www.tthfanfic.org]
user_agent:
slow_down_sleep_time:2
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
@@ -2245,6 +2287,22 @@ pairingcat_to_characters_ships:true
## instead be added to characters Buffy, Spike and ships Buffy/Spike ## instead be added to characters Buffy, Spike and ships Buffy/Spike
romancecat_to_characters_ships:true romancecat_to_characters_ships:true
[www.twcslibrary.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## twcslibrary.net (ab)uses series as personal reading lists.
collect_series: false
[www.twilightarchives.com] [www.twilightarchives.com]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Twilight extracategories:Twilight
@@ -2293,8 +2351,9 @@ extracharacters:Wolverine,Rogue
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis extracategories:Stargate: Atlantis
extra_valid_entries:reviews ##site stopped showing reviews ~ Oct 2016
reviews_label:Reviews #extra_valid_entries:reviews
#reviews_label:Reviews
[buffygiles.velocitygrass.com] [buffygiles.velocitygrass.com]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
@@ -2357,20 +2416,6 @@ extracategories:Andromeda
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Artemis Fowl extracategories:Artemis Fowl
[www.therabidreader.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[www.naiceanilme.net] [www.naiceanilme.net]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
@@ -2384,8 +2429,6 @@ extracategories:Artemis Fowl
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
[overrides] [overrides]
## It may sometimes be useful to override all of the specific format, ## It may sometimes be useful to override all of the specific format,
## site and site:format sections in your private configuration. For ## site and site:format sections in your private configuration. For
+4
View File
@@ -130,6 +130,7 @@ default_prefs['deleteotherforms'] = False
default_prefs['adddialogstaysontop'] = False default_prefs['adddialogstaysontop'] = False
default_prefs['lookforurlinhtml'] = False default_prefs['lookforurlinhtml'] = False
default_prefs['checkforseriesurlid'] = True default_prefs['checkforseriesurlid'] = True
default_prefs['auto_reject_seriesurlid'] = False
default_prefs['checkforurlchange'] = True default_prefs['checkforurlchange'] = True
default_prefs['injectseries'] = False default_prefs['injectseries'] = False
default_prefs['matchtitleauth'] = True default_prefs['matchtitleauth'] = True
@@ -166,6 +167,8 @@ default_prefs['allow_custcol_from_ini'] = True
default_prefs['std_cols_newonly'] = {} default_prefs['std_cols_newonly'] = {}
default_prefs['set_author_url'] = True default_prefs['set_author_url'] = True
default_prefs['includecomments'] = False
default_prefs['anth_comments_newonly'] = True
default_prefs['imapserver'] = '' default_prefs['imapserver'] = ''
default_prefs['imapuser'] = '' default_prefs['imapuser'] = ''
@@ -174,6 +177,7 @@ default_prefs['imapsessionpass'] = False
default_prefs['imapfolder'] = 'INBOX' default_prefs['imapfolder'] = 'INBOX'
default_prefs['imapmarkread'] = True default_prefs['imapmarkread'] = True
default_prefs['auto_reject_from_email'] = False default_prefs['auto_reject_from_email'] = False
default_prefs['update_existing_only_from_email'] = False
default_prefs['download_from_email_immediately'] = False default_prefs['download_from_email_immediately'] = False
def set_library_config(library_config,db): def set_library_config(library_config,db):
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+2 -3
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2015 Fanficdownloader team, 2015 FanFicFare team # Copyright 2015 Fanficdownloader team, 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the 'License'); # Licensed under the Apache License, Version 2.0 (the 'License');
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -17,7 +17,7 @@
try: try:
# just a way to switch between web service and CLI/PI # just a way to switch between web service and CLI/PI
import google.appengine.api import google.appengine.api
except: except:
try: # just a way to switch between CLI and PI try: # just a way to switch between CLI and PI
import calibre.constants import calibre.constants
@@ -31,4 +31,3 @@ except:
logger.addHandler(loghandler) logger.addHandler(loghandler)
loghandler.setLevel(logging.DEBUG) loghandler.setLevel(logging.DEBUG)
logger.setLevel(logging.DEBUG) logger.setLevel(logging.DEBUG)
+8 -10
View File
@@ -85,7 +85,6 @@ import adapter_hpfanficarchivecom
import adapter_twilightarchivescom import adapter_twilightarchivescom
import adapter_nhamagicalworldsus import adapter_nhamagicalworldsus
import adapter_hlfictionnet import adapter_hlfictionnet
import adapter_grangerenchantedcom
import adapter_dracoandginnycom import adapter_dracoandginnycom
import adapter_scarvesandcoffeenet import adapter_scarvesandcoffeenet
import adapter_thepetulantpoetesscom import adapter_thepetulantpoetesscom
@@ -102,13 +101,9 @@ import adapter_efictionestelielde
import adapter_pommedesangcom import adapter_pommedesangcom
import adapter_restrictedsectionorg import adapter_restrictedsectionorg
import adapter_imagineeficcom import adapter_imagineeficcom
import adapter_buffynfaithnet
import adapter_psychficcom import adapter_psychficcom
import adapter_tokrafandomnetcom
import adapter_asr3slashzoneorg import adapter_asr3slashzoneorg
import adapter_nickandgregnet
import adapter_potterheadsanonymouscom import adapter_potterheadsanonymouscom
import adapter_scarheadnet
import adapter_fictionpadcom import adapter_fictionpadcom
import adapter_storiesonlinenet import adapter_storiesonlinenet
import adapter_trekiverseorg import adapter_trekiverseorg
@@ -120,7 +115,6 @@ import adapter_nocturnallightnet
import adapter_fanfichu import adapter_fanfichu
import adapter_fanfictioncsodaidokhu import adapter_fanfictioncsodaidokhu
import adapter_fictionmaniatv import adapter_fictionmaniatv
import adapter_bdsmgeschichten
import adapter_tolkienfanfiction import adapter_tolkienfanfiction
import adapter_themaplebookshelf import adapter_themaplebookshelf
import adapter_fannation import adapter_fannation
@@ -139,14 +133,18 @@ import adapter_ninelivesarchivecom
import adapter_masseffect2in import adapter_masseffect2in
import adapter_quotevcom import adapter_quotevcom
import adapter_mcstoriescom import adapter_mcstoriescom
import adapter_lucifaelff import adapter_lucifaelff
import adapter_buffygilescom import adapter_buffygilescom
#import adapter_rubyquillcom import adapter_andromedawebcom
import adapter_andromedawebcom # Not all lables are captured
import adapter_artemisfowlcom import adapter_artemisfowlcom
import adapter_rabidreadercom
import adapter_naiceanilmenet import adapter_naiceanilmenet
import adapter_deepinmysoulnet
import adapter_haremlucifaelcom
import adapter_kiarepositorymujajinet
import adapter_fanfictionlucifaelcom
import adapter_adultfanfictionorg
import adapter_fictionhuntcom
import adapter_royalroadl
## This bit of complexity allows adapters to be added by just adding ## This bit of complexity allows adapters to be added by just adding
## importing. It eliminates the long if/else clauses we used to need ## importing. It eliminates the long if/else clauses we used to need
@@ -0,0 +1,375 @@
# -*- coding: utf-8 -*-
# Copyright 2013 Fanficdownloader team, 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
################################################################################
### Written by GComyn
################################################################################
import time
import logging
logger = logging.getLogger(__name__)
import re
import sys
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
################################################################################
def getClass():
return AdultFanFictionOrgAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class AdultFanFictionOrgAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
logger.debug("AdultFanFictionOrgAdapter.__init__ - url='{0}'".format(url))
self.decode = ["utf8",
"Windows-1252"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
#Setting the 'Zone' for each "Site"
self.zone = self.parsedUrl.netloc.split('.')[0]
# normalized story URL. (checking self.zone against list
# removed--it was redundant w/getAcceptDomains and
# getSiteURLPattern both)
self._setURL('http://' + self.zone + '.' + self.getBaseDomain() + '/story.php?no='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
#self.story.setMetadata('siteabbrev',self.getSiteAbbrev())
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev',self.zone+'aff')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%Y-%m-%d"
##This method will be moved to the sub-adapters
# @classmethod
# def getSiteAbbrev(self):
# return self.zone+'aff'
## Added because adult-fanfiction.org does send you to
## www.adult-fanfiction.org when you go to it and it also moves
## the site & examples down the web service front page so the
## first screen isn't dominated by 'adult' links.
def getBaseDomain(self):
return 'adult-fanfiction.org'
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'www.adult-fanfiction.org'
@classmethod
def getAcceptDomains(cls):
# mobile.fimifction.com isn't actually a valid domain, but we can still get the story id from URLs anyway
return ['anime.adult-fanfiction.org',
'anime2.adult-fanfiction.org',
'bleach.adult-fanfiction.org',
'books.adult-fanfiction.org',
'buffy.adult-fanfiction.org',
'cartoon.adult-fanfiction.org',
'celeb.adult-fanfiction.org',
'comics.adult-fanfiction.org',
'ff.adult-fanfiction.org',
'games.adult-fanfiction.org',
'hp.adult-fanfiction.org',
'inu.adult-fanfiction.org',
'lotr.adult-fanfiction.org',
'manga.adult-fanfiction.org',
'movies.adult-fanfiction.org',
'naruto.adult-fanfiction.org',
'ne.adult-fanfiction.org',
'original.adult-fanfiction.org',
'tv.adult-fanfiction.org',
'xmen.adult-fanfiction.org',
'ygo.adult-fanfiction.org',
'yuyu.adult-fanfiction.org']
@classmethod
def getSiteExampleURLs(self):
return ("http://anime.adult-fanfiction.org/story.php?no=123456789 "
+ "http://anime2.adult-fanfiction.org/story.php?no=123456789 "
+ "http://bleach.adult-fanfiction.org/story.php?no=123456789 "
+ "http://books.adult-fanfiction.org/story.php?no=123456789 "
+ "http://buffy.adult-fanfiction.org/story.php?no=123456789 "
+ "http://cartoon.adult-fanfiction.org/story.php?no=123456789 "
+ "http://celeb.adult-fanfiction.org/story.php?no=123456789 "
+ "http://comics.adult-fanfiction.org/story.php?no=123456789 "
+ "http://ff.adult-fanfiction.org/story.php?no=123456789 "
+ "http://games.adult-fanfiction.org/story.php?no=123456789 "
+ "http://hp.adult-fanfiction.org/story.php?no=123456789 "
+ "http://inu.adult-fanfiction.org/story.php?no=123456789 "
+ "http://lotr.adult-fanfiction.org/story.php?no=123456789 "
+ "http://manga.adult-fanfiction.org/story.php?no=123456789 "
+ "http://movies.adult-fanfiction.org/story.php?no=123456789 "
+ "http://naruto.adult-fanfiction.org/story.php?no=123456789 "
+ "http://ne.adult-fanfiction.org/story.php?no=123456789 "
+ "http://original.adult-fanfiction.org/story.php?no=123456789 "
+ "http://tv.adult-fanfiction.org/story.php?no=123456789 "
+ "http://xmen.adult-fanfiction.org/story.php?no=123456789 "
+ "http://ygo.adult-fanfiction.org/story.php?no=123456789 "
+ "http://yuyu.adult-fanfiction.org/story.php?no=123456789")
def getSiteURLPattern(self):
return r'http?://(anime|anime2|bleach|books|buffy|cartoon|celeb|comics|ff|games|hp|inu|lotr|manga|movies|naruto|ne|original|tv|xmen|ygo|yuyu)\.adult-fanfiction\.org/story\.php\?no=\d+$'
##This is not working right now, so I'm commenting it out, but leaving it for future testing
## Login seems to be reasonably standard across eFiction sites.
#def needToLoginCheck(self, data):
##This adapter will always require a login
# return True
# <form name="login" method="post" action="">
# <div class="top">E-mail: <span id="sprytextfield1">
# <input name="email" type="text" id="email" size="20" maxlength="255" />
# <span class="textfieldRequiredMsg">Email is required.</span><span class="textfieldInvalidFormatMsg">Invalid E-mail.</span></span></div>
# <div class="top">Password: <span id="sprytextfield2">
# <input name="pass1" type="password" id="pass1" size="20" maxlength="32" />
# <span class="textfieldRequiredMsg">password is required.</span><span class="textfieldMinCharsMsg">Minimum 8 characters8.</span><span class="textfieldMaxCharsMsg">Exceeded 32 characters.</span></span></div>
# <div class="top"><br /> <input name="loginsubmittop" type="hidden" id="loginsubmit" value="TRUE" />
# <input type="submit" value="Login" />
# </div>
# </form>
##This is not working right now, so I'm commenting it out, but leaving it for future testing
#def performLogin(self, url, soup):
# params = {}
# if self.password:
# params['email'] = self.username
# params['pass1'] = self.password
# else:
# params['email'] = self.getConfig("username")
# params['pass1'] = self.getConfig("password")
# params['submit'] = 'Login'
# # copy all hidden input tags to pick up appropriate tokens.
# for tag in soup.findAll('input',{'type':'hidden'}):
# params[tag['name']] = tag['value']
# logger.debug("Will now login to URL {0} as {1} with password: {2}".format(url, params['email'],params['pass1']))
# d = self._postUrl(url, params, usecache=False)
# d = self._fetchUrl(url, params, usecache=False)
# soup = self.make_soup(d)
#if not (soup.find('form', {'name' : 'login'}) == None):
# logger.info("Failed to login to URL %s as %s" % (url, params['email']))
# raise exceptions.FailedToLogin(url,params['email'])
# return False
#else:
# return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def doExtractChapterUrlsAndMetadata(self, get_cover=True):
## You need to have your is_adult set to true to get this story
if not (self.is_adult or self.getConfig("is_adult")):
raise exceptions.AdultCheckRequired(self.url)
url = self.url
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code in 404:
raise exceptions.StoryDoesNotExist("Code: 404. %s"%self.url)
elif e.code == 410:
raise exceptions.StoryDoesNotExist("Code: 410. %s"%self.url)
elif e.code == 401:
self.needToLogin = True
data = ''
else:
raise e
if "The dragons running the back end of the site can not seem to find the story you are looking for." in data:
raise exceptions.StoryDoesNotExist(self.zone+'.'+self.getBaseDomain()
+" says: The dragons running the back end of the site can not seem to find the story you are looking for.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
##This is not working right now, so I'm commenting it out, but leaving it for future testing
#self.performLogin(url, soup)
# Now go hunting for all the meta data and the chapter list.
## Title
## Some of the titles have a backslash on the story page, but not on the Author's page
## So I am removing it from the title, so it can be found on the Author's page further in the code.
## Also, some titles may have extra spaces ' ', and the search on the Author's page removes them,
## so I have to here as well. I used multiple replaces to make sure, since I did the same below.
a = soup.find('a', href=re.compile(r'story.php\?no='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a).replace('\\','').replace(' ',' ').replace(' ',' ').replace(' ',' ').strip())
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"profile.php\?no=\d+"))
self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl',a['href'])
self.story.setMetadata('author',stripHTML(a))
# Find the chapters:
chapters = soup.find('div',{'id':'snav'})
for i, chapter in enumerate(chapters.findAll('a')):
self.chapterUrls.append((stripHTML(chapter),self.url+'&chapter='+str(i+1)))
self.story.setMetadata('numChapters', len(self.chapterUrls))
##The story page does not give much Metadata, so we go to the Author's page
##Get the first Author page to see if there are multiple pages.
##AFF doesn't care if the page number is larger than the actual pages,
##it will continue to show the last page even if the variable is larger than the actual page
author_Url = self.story.getMetadata('authorUrl')+'&view=story&zone='+self.zone+'&page=1'
##I'm resetting the author page to the zone for this story
self.story.setMetadata('authorUrl',author_Url)
logger.debug('Getting the author page: {0}'.format(author_Url))
try:
adata = self._fetchUrl(author_Url)
except urllib2.HTTPError, e:
if e.code in 404:
raise exceptions.StoryDoesNotExist("Author Page: Code: 404. %s"%author_Url)
elif e.code == 410:
raise exceptions.StoryDoesNotExist("Author Page: Code: 410. %s"%author_Url)
else:
raise e
if "The member you are looking for does not exist." in adata:
raise exceptions.StoryDoesNotExist(self.zone+'.'+self.getBaseDomain() +" says: The member you are looking for does not exist.")
asoup = self.make_soup(adata)
##Getting the number of pages
pages=asoup.find('div',{'class' : 'pagination'}).findAll('li')[-1].find('a')
if not pages == None:
pages = pages['href'].split('=')[-1]
else:
pages = 0
logger.info(pages)
##If there is only 1 page of stories, check it to get the Metadata,
if pages == 0:
a = asoup.findAll('li')
for lc2 in a:
if lc2.find('a', href=re.compile(r'story.php\?no='+self.story.getMetadata('storyId')+"$")):
break
## otherwise go through the pages
else:
page=1
i=0
while i == 0:
##We already have the first page, so if this is the first time through, skip getting the page
if page != 1:
author_Url = self.story.getMetadata('authorUrl')+'&view=story&zone='+self.zone+'&page='+str(page)
logger.debug('Getting the author page: {0}'.format(author_Url))
try:
adata = self._fetchUrl(author_Url)
except urllib2.HTTPError, e:
if e.code in 404:
raise exceptions.StoryDoesNotExist("Author Page: Code: 404. %s"%author_Url)
elif e.code == 410:
raise exceptions.StoryDoesNotExist("Author Page: Code: 410. %s"%author_Url)
else:
raise e
##This will probably never be needed, since AFF doesn't seem to care what number you put as
## the page number, it will default to the last page, even if you use 1000, for an author
## that only hase 5 pages of stories, but I'm keeping it in to appease Saint Justin Case (just in case).
if "The member you are looking for does not exist." in adata:
raise exceptions.StoryDoesNotExist(self.zone+'.'+self.getBaseDomain() +" says: The member you are looking for does not exist.")
asoup = self.make_soup(adata)
a = asoup.findAll('li')
for lc2 in a:
if lc2.find('a', href=re.compile(r'story.php\?no='+self.story.getMetadata('storyId')+"$")):
i=1
break
page = page + 1
if page > pages:
break
##Split the Metadata up into a list
##We have to change the soup type to a string, then remove the newlines, and double spaces,
##then changes the <br/> to '-:-', which seperates the different elemeents.
##Then we strip the HTML elements from the string.
##There is also a double <br/>, so we have to fix that, then remove the leading and trailing '-:-'.
##They are always in the same order.
liMetadata = stripHTML(str(lc2).replace('\n','').replace('\r','').replace('\t',' ').replace(' ',' ').replace(' ',' ').replace(' ',' ').replace(r'<br/>','-:-'))
liMetadata = liMetadata.replace(r'-:--:-','-:-').strip('-:-').strip('-:-')
for i, value in enumerate(liMetadata.split('-:-')):
##The item 6 is the reviews... We are disregarding them.
##The item 7 is the 'Dragon Prints'... not sure what they are, so disregarding them.
##The 0 item is the title
if i == 0:
if value <> self.story.getMetadata('title'):
raise exceptions.StoryDoesNotExist('Did not find story in author story list: {0}'.format(author_Url))
elif i == 1:
##Get the description
self.story.setMetadata('description',stripHTML(value.strip()))
elif i == 2:
##The Get the Category
self.story.setMetadata('category',value.replace(r'&gt;',r'>').replace(r'Located :',r'').strip())
elif i == 3:
##Get the Erotic Tags
value = stripHTML(value.replace(r'Content Tags :',r'')).strip()
for code in re.split(r'\s',value):
self.story.addToList('eroticatags',code)
elif i == 4:
##Get the Posted Date
value = value.replace(r'Posted :',r'').strip()
self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat))
elif i == 5:
##Get the 'Updated' Edited date
##AFF has the time for the Updated date, and we only want the date,
##so we take the first 10 characters only
value = value.replace(r'Edited :',r'').strip()[0:10]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat))
# grab the text for an individual chapter.
def getChapterText(self, url):
#Since each chapter is on 1 page, we don't need to do anything special, just get the content of the page.
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
chaptertag = soup.find('div',{'class' : 'pagination'}).parent.findNext('td')
if None == chaptertag:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,chaptertag)
@@ -187,7 +187,7 @@ class ArchiveOfOurOwnOrgAdapter(BaseSiteAdapter):
else: else:
for a in alist: for a in alist:
self.story.addToList('authorId',a['href'].split('/')[-1]) self.story.addToList('authorId',a['href'].split('/')[-1])
self.story.addToList('authorUrl',a['href']) self.story.addToList('authorUrl','http://'+self.host+a['href'])
self.story.addToList('author',a.text) self.story.addToList('author',a.text)
byline = metasoup.find('h3',{'class':'byline'}) byline = metasoup.find('h3',{'class':'byline'})
@@ -77,7 +77,8 @@ class AshwinderSycophantHexComAdapter(BaseSiteAdapter):
## Login seems to be reasonably standard across eFiction sites. ## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data): def needToLoginCheck(self, data):
if 'This story contains adult content and/or themes.' in data \ if 'This story contains adult content and/or themes.' in data \
or "That password doesn't match the one in our database" in data: or "That password doesn't match the one in our database" in data \
or "Member Login" in data:
return True return True
else: else:
return False return False
@@ -1,347 +0,0 @@
# -*- coding: utf-8 -*-
# Copyright 2014 Fanficdownloader team, 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
import urlparse
import time
from bs4.element import Tag, Comment
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
def _translate_date_german_english(date):
fullmon = {"Januar":"01",
"Februar":"02",
u"März":"03",
"April":"04",
"Mai":"05",
"Juni":"06",
"Juli":"07",
"August":"08",
"September":"09",
"Oktober":"10",
"November":"11",
"Dezember":"12"}
for (name,num) in fullmon.items():
date = date.replace(name,num)
return date
_REGEX_TRAILING_DIGIT = re.compile("(\d+)$")
_REGEX_DASH_TO_END = re.compile("-[^-]+$")
_REGEX_CHAPTER_TITLE = re.compile(ur"""
\s*
[\u2013-]?
\s*
([\dIVX-]+)?
\.?
\s*
[\[\(]?
\s*
(Teil|Kapitel|Tag)?
\s*
([\dIVX-]+)?
\s*
[\]\)]?
\s*
$
""", re.VERBOSE)
_INITIAL_STEP = 5
class BdsmGeschichtenAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["utf8", "Windows-1252"]
self.story.setMetadata('siteabbrev','bdsmgesch')
# Replace possible chapter numbering
chapterMatch = _REGEX_TRAILING_DIGIT.search(url)
if chapterMatch is None:
self.maxChapter = 1
else:
self.maxChapter = int(chapterMatch.group(1))
# url = re.sub(_REGEX_TRAILING_DIGIT, "1", url)
# set storyId
self.story.setMetadata('storyId', re.compile(self.getSiteURLPattern()).match(url).group('storyId'))
# normalize URL
self._setURL('http://%s/%s' % (self.getSiteDomain(), self.story.getMetadata('storyId')))
self.dateformat = '%d. %m %Y - %H:%M'
@staticmethod
def getSiteDomain():
return 'bdsm-geschichten.net'
@classmethod
def getAcceptDomains(cls):
return ['www.bdsm-geschichten.net', 'www.bdsm-geschichten.net']
@classmethod
def getSiteExampleURLs(cls):
return "http://www.bdsm-geschichten.net/title-of-story-1 http://bdsm-geschichten.net/title-of-story-1"
def getSiteURLPattern(self):
return r"http://(www\.)?bdsm-geschichten.net/(?P<storyId>[a-zA-Z0-9_-]+)"
def extractChapterUrlsAndMetadata(self):
if not (self.is_adult or self.getConfig("is_adult")):
raise exceptions.AdultCheckRequired(self.url)
try:
data1 = self._fetchUrl(self.url)
soup = self.make_soup(data1)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
#strip comments from soup
[comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, Comment))]
# Cache the soups so we won't have to redownload in getChapterText later
self.soupsCache = {}
self.soupsCache[self.url] = soup
# author
authorDiv = soup.find("div", "author-pane-line author-name")
authorId = authorDiv.string.strip()
self.story.setMetadata('authorId', authorId)
self.story.setMetadata('author', authorId)
# TODO not really true need to be loggedin for this to work or fetch userid
self.story.setMetadata('authorUrl','http://'+self.host+'/'+authorId)
# TODO better metadata
date = soup.find("div", {"class": "submitted"}).string.strip()
# 11. April 2015 - 17:08
date = re.sub(r"(\d+\. \D+ \d+ - \d+:\d+).*", r"\1", date)
date = _translate_date_german_english(date)
self.story.setMetadata('datePublished', makeDate(date, self.dateformat))
title1 = soup.find("h1", {'class': 'title'}).string
for tagLink in soup.find("ul", "taxonomy").findAll("a"):
self.story.addToList('category', tagLink.string)
## Retrieve chapter soups
if self.getConfig('find_chapters') == 'guess':
self.chapterUrls = []
self._find_chapters_by_guessing(title1)
else:
self._find_chapters_by_parsing(soup)
firstChapterUrl = self.chapterUrls[0][1]
if firstChapterUrl in self.soupsCache:
firstChapterSoup = self.soupsCache[firstChapterUrl]
h1 = firstChapterSoup.find("h1").text
else:
h1 = soup.find("h1").text
h1 = re.sub(_REGEX_CHAPTER_TITLE, "", h1)
self.story.setMetadata('title', h1)
self.story.setMetadata('numChapters', len(self.chapterUrls))
return
def _find_chapters_by_parsing(self, soup):
# store original soup
origSoup = soup
#
# find first chapter
#
firstLink = None
firstLinkDiv = soup.find("div", "field-field-erster-teil")
if firstLinkDiv is not None:
firstLink = "http://%s%s" % (self.getSiteDomain(), firstLinkDiv.findNext("a")['href'])
logger.debug("Found first chapter right away <%s>" % firstLink)
try:
soup = self.make_soup(self._fetchUrl(firstLink))
self.soupsCache[firstLink] = soup
self.chapterUrls.insert(0, (soup.find("h1").text, firstLink))
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise exceptions.StoryDoesNotExist(firstLink)
else:
logger.debug("DIDN'T find first chapter right away")
# parse previous Link until first
while True:
prevLink = None
prevLinkDiv = soup.find("div", "field-field-vorheriger-teil")
if prevLinkDiv is not None:
prevLink = prevLinkDiv.find("a")
if prevLink is None:
prevLink = soup.find("a", text=re.compile("&lt;&lt;&lt;")) # <<<
if prevLink is None:
logger.debug("Couldn't find prev part")
break
else:
logger.debug("Previous Chapter <%s>" % prevLink)
if type(prevLink) != Tag or prevLink.name != "a":
prevLink = prevLink.findParent("a")
if prevLink is None or '#' in prevLink['href']:
logger.debug("Couldn't find prev part (false positive) <%s>" % prevLink)
break
prevLink = prevLink['href']
try:
soup = self.make_soup(self._fetchUrl(prevLink))
self.soupsCache[prevLink] = soup
prevTtitle = soup.find("h1", {'class': 'title'}).string
self.chapterUrls.insert(0, (prevTtitle, prevLink))
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(nextLink)
else:
raise e
firstLink = prevLink
# if first chapter couldn't be determined, assume the URL originally
# passed is the first chapter
if firstLink is None:
logger.debug("Couldn't set first chapter")
firstLink = self.url
self.chapterUrls.insert(0, (soup.find("h1").text, firstLink))
# set first URL
logger.debug("Set first link: %s" % firstLink)
self._setURL(firstLink)
self.story.setMetadata('storyId', re.compile(self.getSiteURLPattern()).match(firstLink).group('storyId'))
#
# Parse next chapters
#
while True:
nextLink = None
nextLinkDiv = soup.find("div", "field-field-naechster-teil")
if nextLinkDiv is not None:
nextLink = nextLinkDiv.find("a")
if nextLink is None:
nextLink = soup.find("a", text=re.compile("&gt;&gt;&gt;"))
if nextLink is None:
nextLink = soup.find("a", text=re.compile("Fortsetzung"))
if nextLink is None:
logger.debug("Couldn't find next part")
break
else:
if type(nextLink) != Tag or nextLink.name != "a":
nextLink = nextLink.findParent("a")
if nextLink is None or '#' in nextLink['href']:
logger.debug("Couldn't find next part (false positive) <%s>" % nextLink)
break
nextLink = nextLink['href']
if not nextLink.startswith('http:'):
nextLink = 'http://' + self.getSiteDomain() + nextLink
for loadedChapter in self.chapterUrls:
if loadedChapter[0] == nextLink:
logger.debug("ERROR: Repeating chapter <%s> Try to fix it" % nextLink)
nextLinkMatch = _REGEX_TRAILING_DIGIT.match(nextLink)
if nextLinkMatch is not None:
curChap = nextLinkMatch.group(1)
nextLink = re.sub(_REGEX_TRAILING_DIGIT, unicode(int(curChap) + 1), nextLink)
else:
break
try:
data = self._fetchUrl(nextLink)
soup = self.make_soup(data)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(nextLink)
else:
raise e
title2 = soup.find("h1", {'class': 'title'}).string
self.chapterUrls.append((title2, nextLink))
logger.debug("Grabbing next chapter URL " + nextLink)
self.soupsCache[nextLink] = soup
# [comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, Comment))]
logger.debug("Chapters: %s" % self.chapterUrls)
def _find_chapters_by_guessing(self, title1):
step = _INITIAL_STEP
curMax = self.maxChapter + step
lastHit = True
while True:
nextChapterUrl = re.sub(_REGEX_TRAILING_DIGIT, unicode(curMax), self.url)
if nextChapterUrl == self.url:
logger.debug("Unable to guess next chapter because URL doesn't end in numbers")
break;
try:
logger.debug("Trying chapter URL " + nextChapterUrl)
data = self._fetchUrl(nextChapterUrl)
hit = True
except urllib2.HTTPError, e:
if e.code == 404:
hit = False
else:
raise e
if hit:
logger.debug("Found chapter URL " + nextChapterUrl)
self.maxChapter = curMax
self.soupsCache[nextChapterUrl] = self.make_soup(data)
if not lastHit:
break
lastHit = curMax
curMax += step
else:
lastHit = False
curMax -= 1
logger.debug(curMax)
for i in xrange(1, self.maxChapter):
nextChapterUrl = re.sub(_REGEX_TRAILING_DIGIT, unicode(i), self.url)
nextChapterTitle = re.sub("1", unicode(i), title1)
self.chapterUrls.append((nextChapterTitle, nextChapterUrl))
def getChapterText(self, url):
if url in self.soupsCache:
logger.debug('Getting chapter <%s> from cache' % url)
soup = self.soupsCache[url]
else:
logger.debug('Downloading chapter <%s>' % url)
data1 = self._fetchUrl(url)
soup = self.make_soup(data1)
#strip comments from soup
[comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, Comment))]
# get story text
storyDiv1 = soup.new_tag("div")
for para in soup.find("div", "full-node").find('div', 'content').findAll("p"):
storyDiv1.append(para)
storyDiv1.append(soup.new_tag("br"))
storytext = self.utf8FromSoup(url,storyDiv1)
return storytext
def getClass():
return BdsmGeschichtenAdapter
@@ -28,7 +28,7 @@ class BloodshedverseComAdapter(BaseSiteAdapter):
READ_URL_TEMPLATE = BASE_URL + 'stories.php?go=read&no=%s' READ_URL_TEMPLATE = BASE_URL + 'stories.php?go=read&no=%s'
STARTED_DATETIME_FORMAT = '%m/%d/%Y' STARTED_DATETIME_FORMAT = '%m/%d/%Y'
UPDATED_DATETIME_FORMAT = '%m/%d/%Y %I:%M' UPDATED_DATETIME_FORMAT = '%m/%d/%Y %I:%M %p'
def __init__(self, config, url): def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url) BaseSiteAdapter.__init__(self, config, url)
@@ -168,14 +168,8 @@ class BloodshedverseComAdapter(BaseSiteAdapter):
self.story.setMetadata('datePublished', makeDate(value, self.STARTED_DATETIME_FORMAT)) self.story.setMetadata('datePublished', makeDate(value, self.STARTED_DATETIME_FORMAT))
elif key == 'Updated': elif key == 'Updated':
date_string, period = value.rsplit(' ', 1) date = makeDate(value, self.UPDATED_DATETIME_FORMAT)
date = makeDate(date_string, self.UPDATED_DATETIME_FORMAT) # ugly %p(am/pm) hack moved into makeDate so other sites can use it.
# Rather ugly hack to work around Calibre's changing of
# Python's locale setting, causing am/pm to not be properly
# parsed by strptime() when using a non-english locale
if period == 'pm':
date += timedelta(hours=12)
self.story.setMetadata('dateUpdated', date) self.story.setMetadata('dateUpdated', date)
if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')): if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')):
@@ -1,292 +0,0 @@
# -*- coding: utf-8 -*-
# Copyright 2013 Fanficdownloader team, 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
import cookielib as cl
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
# This function is called by the downloader in all adapter_*.py files
# in this dir to register the adapter class. So it needs to be
# updated to reflect the class below it. That, plus getSiteDomain()
# take care of 'Registering'.
def getClass():
return BuffyNFaithNetAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class BuffyNFaithNetAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.setHeader()
self.decode = ["Windows-1252",
"utf8"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query correct
m = re.match(self.getSiteURLPattern(),url)
if m:
self.story.setMetadata('storyId',m.group('id'))
# normalized story URL. gets rid of chapter if there, left with ch 1 URL on this site
nurl = "http://"+self.getSiteDomain()+"/fanfictions/index.php?act=vie&id="+self.story.getMetadata('storyId')
self._setURL(nurl)
else:
raise exceptions.InvalidStoryURL(url,
self.getSiteDomain(),
self.getSiteExampleURLs())
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','bnfnet')
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'buffynfaith.net'
@classmethod
def stripURLParameters(cls,url):
"Only needs to be overriden if URL contains more than one parameter"
## This adapter needs at least two parameters left on the URL, act and id
return re.sub(r"(\?act=(vie|ovr)&id=\d+)&.*$",r"\1",url)
def setHeader(self):
"buffynfaith.net wants a Referer for images. Used both above and below(after cookieproc added)"
self.opener.addheaders.append(('Referer', 'http://'+self.getSiteDomain()+'/'))
@classmethod
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=ovr&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234&ch=2"
def getSiteURLPattern(self):
#http://buffynfaith.net/fanfictions/index.php?act=vie&id=963
#http://buffynfaith.net/fanfictions/index.php?act=vie&id=949
#http://buffynfaith.net/fanfictions/index.php?act=vie&id=949&ch=2
p = re.escape("http://"+self.getSiteDomain()+"/fanfictions/index.php?act=")+\
r"(vie|ovr)&id=(?P<id>\d+)(&ch=(?P<ch>\d+))?$"
return p
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def extractChapterUrlsAndMetadata(self):
dateformat = "%d %B %Y"
url = self.url
logger.debug("URL: "+url)
#set a cookie to get past adult check
if self.is_adult or self.getConfig("is_adult"):
cookie = cl.Cookie(version=0, name='my_age', value='yes',
port=None, port_specified=False,
domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False,
path='/', path_specified=True,
secure=False,
expires=time.time()+10000,
discard=False,
comment=None,
comment_url=None,
rest={'HttpOnly': None},
rfc2109=False)
self.get_cookiejar().set_cookie(cookie)
self.setHeader()
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
#print data
if "ADULT CONTENT WARNING" in data:
raise exceptions.AdultCheckRequired(self.url)
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
# Now go hunting for all the meta data and the chapter list.
#stuff in <head>: description
svalue = soup.head.find('meta',attrs={'name':'description'})['content']
#self.story.setMetadata('description',svalue)
self.setDescription(url,svalue)
#useful stuff in rest of doc, all contained in this:
doc = soup.body.find('div', id='my_wrapper')
#first the site category (more of a genre to me, meh) and title, in this element:
mt = doc.find('div',attrs={'class':'maintitle'})
self.story.addToList('genre',mt.findAll('a')[1].string)
self.story.setMetadata('title',stripHTML(mt).split(u'»')[-1].strip())
del mt
#the actual category, for me, is 'Buffy: The Vampire Slayer'
#self.story.addToList('category','Buffy: The Vampire Slayer')
#No need to do it here, it is better to set it in in plugin-defaults.ini and defaults.ini
#then a block that sits in a table cell like so:
#(contains a lot of metadata)
mblock = doc.find('td', align='left', width = '70%').contents
while len(mblock) > 0:
i = mblock.pop(0)
if 'Author:' in i.string:
#drop empty space
mblock.pop(0)
#get author link
a = mblock.pop(0)
authre = re.escape('./index.php?act=bio&id=')+'(?P<authid>\d+)'
m = re.match(authre,a['href'])
self.story.setMetadata('author',a.string)
self.story.setMetadata('authorId',m.group('authid'))
authurl = u'http://%s/fanfictions/index.php?act=bio&id=%s' % ( self.getSiteDomain(),
self.story.getMetadata('authorId'))
self.story.setMetadata('authorUrl',authurl,condremoveentities=False)
#drop empty space
mblock.pop(0)
if 'Rating:' in i.string:
self.story.setMetadata('rating',mblock.pop(0).strip())
if 'Published:' in i.string:
date = mblock.pop(0).strip()
#get rid of 'st', 'nd', 'rd', 'th' after day number
date = date[0:2]+date[4:]
self.story.setMetadata('datePublished',makeDate(date, dateformat))
if 'Last Updated:' in i.string:
date = mblock.pop(0).strip()
#get rid of 'st', 'nd', 'rd', 'th' after day number
date = date[0:2]+date[4:]
self.story.setMetadata('dateUpdated',makeDate(date, dateformat))
if 'Genre:' in i.string:
genres = mblock.pop(0).strip()
genres = genres.split('/')
for genre in genres: self.story.addToList('genre',genre)
#end ifs
#end while
# Find the chapter selector
select = soup.find('select', { 'name' : 'ch' } )
if select is None:
# no selector found, so it's a one-chapter story.
#self.chapterUrls.append((self.story.getMetadata('title'),url))
self.chapterUrls.append((self.story.getMetadata('title'),url))
else:
allOptions = select.findAll('option')
for o in allOptions:
url = u'http://%s/fanfictions/index.php?act=vie&id=%s&ch=%s' % ( self.getSiteDomain(),
self.story.getMetadata('storyId'),
o['value'])
title = u"%s" % o
title = stripHTML(title)
ts = title.split(' ',1)
title = ts[0]+'. '+ts[1]
self.chapterUrls.append((title,url))
self.story.setMetadata('numChapters',len(self.chapterUrls))
## Go scrape the rest of the metadata from the author's page.
data = self._fetchUrl(self.story.getMetadata('authorUrl'))
soup = self.make_soup(data)
#find the story link and its parent div
storya = soup.find('a',{'href':self.story.getMetadata('storyUrl')})
storydiv = storya.parent
#warnings come under a <spawn> tag. Never seen that before...
#appears to just be a line of freeform text, not necessarily a list
#optional
spawn = storydiv.find('spawn',{'id':'warnings'})
if spawn is not None:
warns = spawn.nextSibling.strip()
self.story.addToList('warnings',warns)
#some meta in spans - this should get all, even the ones jammed in a table
spans = storydiv.findAll('span')
for s in spans:
if s.string == 'Ship:':
list = s.nextSibling.strip().split()
self.story.extendList('ships',list)
if s.string == 'Characters:':
list = s.nextSibling.strip().split(',')
self.story.extendList('characters',list)
if s.string == 'Status:':
st = s.nextSibling.strip()
self.story.setMetadata('status',st)
if s.string == 'Words:':
st = s.nextSibling.strip()
self.story.setMetadata('numWords',st)
#reviews - is this worth having?
#ffnet adapter gathers it, don't know if anything else does
#or if it's ever going to be used!
a = storydiv.find('a',{'id':'bold-blue'})
if a:
revs = a.nextSibling.strip()[1:-1]
self.story.setMetadata('reviews',st)
else:
revs = '0'
self.story.setMetadata('reviews',st)
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
div = soup.find('div', {'id' : 'fanfiction'})
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
#remove all the unnecessary bookmark tags
[s.extract() for s in div('div',{'class':"tiny_box2"})]
#is there a review link?
r = div.find('a',href=re.compile(re.escape("./index.php?act=irv")+".*$"))
if r is not None:
#remove the review link and its parent div
r.parent.extract()
#There might also be a link to the sequel on the last chapter
#I'm inclined to keep it in, but the URL needs to be changed from relative to absolute
#Shame there isn't proper series metadata available
#(I couldn't find it anyway)
s = div.find('a',href=re.compile(re.escape("./index.php?act=ovr")+".*$"))
if s is not None:
s['href'] = 'http://'+self.getSiteDomain()+'/fanfictions'+s['href'][1:]
return self.utf8FromSoup(url,div)
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2012 Fanficdownloader team, 2015 FanFicFare team # Copyright 2013 Fanficdownloader team, 2015 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -29,11 +29,11 @@ from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate from base_adapter import BaseSiteAdapter, makeDate
def getClass(): def getClass():
return GrangerEnchantedCom return DeepInMySoulNetAdapter ## XXX
# Class name has to be unique. Our convention is camel case the # Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped. # sitename with Adapter at the end. www is skipped.
class GrangerEnchantedCom(BaseSiteAdapter): class DeepInMySoulNetAdapter(BaseSiteAdapter): # XXX
def __init__(self, config, url): def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url) BaseSiteAdapter.__init__(self, config, url)
@@ -50,37 +50,29 @@ class GrangerEnchantedCom(BaseSiteAdapter):
# get storyId from url--url validation guarantees query is only sid=1234 # get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
self.section=self.parsedUrl.path.split('/',)[1]
# normalized story URL. # normalized story URL.
if "malfoymanor" in self.parsedUrl.netloc: # XXX Most sites don't have the /fiction part. Replace all to remove it usually.
self._setURL('http://malfoymanor.' + self.getSiteDomain() + '/themanor/viewstory.php?sid='+self.story.getMetadata('storyId')) self._setURL('http://' + self.getSiteDomain() + '/fiction/viewstory.php?sid='+self.story.getMetadata('storyId'))
self.story.addToList("category","The Manor")
else:
self._setURL('http://' + self.getSiteDomain() + '/enchant/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation. # Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','gech') self.story.setMetadata('siteabbrev','dimsn') ## XXX
# The date format will vary from site to site. # The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior # http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%d/%b/%Y" self.dateformat = "%B %d, %Y"
@staticmethod # must be @staticmethod, don't remove it. @staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain(): def getSiteDomain():
# The site domain. Does have www here, if it uses it. # The site domain. Does have www here, if it uses it.
return 'grangerenchanted.com' return 'www.deepinmysoul.net' # XXX
@classmethod
def getAcceptDomains(cls):
return ['grangerenchanted.com','malfoymanor.grangerenchanted.com']
@classmethod @classmethod
def getSiteExampleURLs(cls): def getSiteExampleURLs(cls):
return "http://grangerenchanted.com/enchant/viewstory.php?sid=1234 http://malfoymanor.grangerenchanted.com/themanor/viewstory.php?sid=1234" return "http://"+cls.getSiteDomain()+"/fiction/viewstory.php?sid=1234"
def getSiteURLPattern(self): def getSiteURLPattern(self):
return r"http://(malfoymanor.)?grangerenchanted.com/(enchant|themanor)?/viewstory.php\?sid=\d+$" return re.escape("http://"+self.getSiteDomain()+"/fiction/viewstory.php?sid=")+r"\d+$"
## Login seems to be reasonably standard across eFiction sites. ## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data): def needToLoginCheck(self, data):
@@ -103,10 +95,7 @@ class GrangerEnchantedCom(BaseSiteAdapter):
params['cookiecheck'] = '1' params['cookiecheck'] = '1'
params['submit'] = 'Submit' params['submit'] = 'Submit'
if "enchant" in self.section: loginUrl = 'http://' + self.getSiteDomain() + '/fiction/user.php?action=login'
loginUrl = 'http://grangerenchanted.com/enchant/user.php?action=login'
else:
loginUrl = 'http://malfoymanor.grangerenchanted.com/themanor/user.php?action=login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['penname'])) params['penname']))
@@ -128,13 +117,13 @@ class GrangerEnchantedCom(BaseSiteAdapter):
# If the title search below fails, there's a good chance # If the title search below fails, there's a good chance
# you need a different number. print data at that point # you need a different number. print data at that point
# and see what the 'click here to continue' url says. # and see what the 'click here to continue' url says.
addurl = "&ageconsent=ok&warning=1" addurl = "&warning=4"
else: else:
addurl="" addurl=""
# index=1 makes sure we see the story chapter index. Some # index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories. # sites skip that for one-chapter stories.
url = self.url+addurl url = self.url+'&index=1'+addurl
logger.debug("URL: "+url) logger.debug("URL: "+url)
try: try:
@@ -150,7 +139,16 @@ class GrangerEnchantedCom(BaseSiteAdapter):
self.performLogin(url) self.performLogin(url)
data = self._fetchUrl(url) data = self._fetchUrl(url)
m = re.search(r"'viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data) # Since the warning text can change by warning level, let's
# look for the warning pass url. ksarchive uses
# &amp;warning= -- actually, so do other sites. Must be an
# eFiction book.
# fiction/viewstory.php?sid=1882&amp;warning=4
# fiction/viewstory.php?sid=1654&amp;ageconsent=ok&amp;warning=5
#print data
m = re.search(r"'fiction/viewstory.php\?sid=29(&amp;warning=4)'",data)
m = re.search(r"'fiction/viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data)
if m != None: if m != None:
if self.is_adult or self.getConfig("is_adult"): if self.is_adult or self.getConfig("is_adult"):
# We tried the default and still got a warning, so # We tried the default and still got a warning, so
@@ -173,20 +171,22 @@ class GrangerEnchantedCom(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url) raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data: if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find. # use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data) soup = self.make_soup(data)
# print data
# Now go hunting for all the meta data and the chapter list. # Now go hunting for all the meta data and the chapter list.
pagetitle = soup.find('div',{'id':'pagecontent'})
## Title ## Title
a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a)) self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url. # Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+"))
self.story.setMetadata('authorId',a['href'].split('=')[1]) self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href'])
self.story.setMetadata('author',a.string) self.story.setMetadata('author',a.string)
@@ -194,9 +194,9 @@ class GrangerEnchantedCom(BaseSiteAdapter):
# Find the chapters: # Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles. # just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/'+chapter['href']+addurl)) self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fiction/'+chapter['href']+addurl))
self.story.setMetadata('numChapters',len(self.chapterUrls)) self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data # eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly. # formating, so it's a little ugly.
@@ -217,10 +217,11 @@ class GrangerEnchantedCom(BaseSiteAdapter):
if 'Summary' in label: if 'Summary' in label:
## Everything until the next span class='label' ## Everything until the next span class='label'
svalue = "" svalue = ""
while value and 'label' not in defaultGetattr(value,'class') and '<span class="label">' not in unicode(value): while 'label' not in defaultGetattr(value,'class'):
svalue += unicode(value) svalue += unicode(value)
value = value.nextSibling value = value.nextSibling
self.setDescription(url,svalue) self.setDescription(url,svalue)
#self.story.setMetadata('description',stripHTML(svalue))
if 'Rated' in label: if 'Rated' in label:
self.story.setMetadata('rating', value) self.story.setMetadata('rating', value)
@@ -228,9 +229,6 @@ class GrangerEnchantedCom(BaseSiteAdapter):
if 'Word count' in label: if 'Word count' in label:
self.story.setMetadata('numWords', value) self.story.setMetadata('numWords', value)
if 'Read' in label:
self.story.setMetadata('read', value)
if 'Categories' in label: if 'Categories' in label:
cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))
for cat in cats: for cat in cats:
@@ -242,12 +240,12 @@ class GrangerEnchantedCom(BaseSiteAdapter):
self.story.addToList('characters',char.string) self.story.addToList('characters',char.string)
if 'Genre' in label: if 'Genre' in label:
genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1'))
for genre in genres: for genre in genres:
self.story.addToList('genre',genre.string) self.story.addToList('genre',genre.string)
if 'Warnings' in label: if 'Warnings' in label:
warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3'))
for warning in warnings: for warning in warnings:
self.story.addToList('warnings',warning.string) self.story.addToList('warnings',warning.string)
@@ -261,39 +259,31 @@ class GrangerEnchantedCom(BaseSiteAdapter):
self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat))
if 'Updated' in label: if 'Updated' in label:
# there's a stray [ at the end.
#value = value[0:-1]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat))
try: try:
# Find Series name from series URL. # Find Series name from series URL.
a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) a = soup.find('a', href=re.compile(r"fiction/viewseries.php\?seriesid=\d+"))
series_name = a.string series_name = a.string
series_url = 'http://'+self.host+'/'+self.section+'/'+a['href'] series_url = 'http://'+self.host+'/'+a['href']
# use BeautifulSoup HTML parser to make everything easier to find. # use BeautifulSoup HTML parser to make everything easier to find.
seriessoup = self.make_soup(self._fetchUrl(series_url)) seriessoup = self.make_soup(self._fetchUrl(series_url))
# can't use ^viewstory...$ in case of higher rated stories with javascript href. storyas = seriessoup.findAll('a', href=re.compile(r'^fiction/viewstory.php\?sid=\d+$'))
storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+'))
i=1 i=1
for a in storyas: for a in storyas:
# skip 'report this' and 'TOC' links if a['href'] == ('fiction/viewstory.php?sid='+self.story.getMetadata('storyId')):
if 'contact.php' not in a['href'] and 'index' not in a['href']: self.setSeries(series_name, i)
if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): self.story.setMetadata('seriesUrl',series_url)
self.setSeries(series_name, i) break
self.story.setMetadata('seriesUrl',series_url) i+=1
break
i+=1
except: except:
# I find it hard to care if the series parsing fails # I find it hard to care if the series parsing fails
pass pass
try:
self.story.setMetadata('reviews',
stripHTML(soup.find('div',{'id':'sort'}).
findAll('a', href=re.compile(r'^reviews.php'))[1]))
except:
# I find it hard to care if the series parsing fails
pass
# grab the text for an individual chapter. # grab the text for an individual chapter.
def getChapterText(self, url): def getChapterText(self, url):
@@ -301,9 +291,10 @@ class GrangerEnchantedCom(BaseSiteAdapter):
soup = self.make_soup(self._fetchUrl(url)) soup = self.make_soup(self._fetchUrl(url))
div = soup.find('div', {'id' : 'story1'}) div = soup.find('div', {'id' : 'story'})
if None == div: if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div) return self.utf8FromSoup(url,div)
@@ -0,0 +1,36 @@
# -*- coding: utf-8 -*-
# Copyright 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
from base_efiction_adapter import BaseEfictionAdapter
class FanfictionLucifaelComAdapter(BaseEfictionAdapter):
@staticmethod
def getSiteDomain():
return 'fanfiction.lucifael.com'
@classmethod
def getSiteAbbrev(self):
return 'luci'
@classmethod
def getDateFormat(self):
return "%d/%m/%Y"
def getClass():
return FanfictionLucifaelComAdapter
+48 -55
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team, 2015 FanFicFare team # Copyright 2011 Fanficdownloader team, 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -15,14 +15,12 @@
# limitations under the License. # limitations under the License.
# #
import time
from datetime import datetime from datetime import datetime
import logging import logging
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
import re import re
import urllib2 import urllib2
from urllib import unquote_plus from urllib import unquote_plus
import time
from .. import exceptions as exceptions from .. import exceptions as exceptions
from ..htmlcleanup import stripHTML from ..htmlcleanup import stripHTML
@@ -170,9 +168,11 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
elif 'Crossover' in categories[0]['href']: elif 'Crossover' in categories[0]['href']:
caturl = "https://%s%s"%(self.getSiteDomain(),categories[0]['href']) caturl = "https://%s%s"%(self.getSiteDomain(),categories[0]['href'])
catsoup = self.make_soup(self._fetchUrl(caturl)) catsoup = self.make_soup(self._fetchUrl(caturl))
found = False
for a in catsoup.findAll('a',href=re.compile(r"^/crossovers/.+?/\d+/")): for a in catsoup.findAll('a',href=re.compile(r"^/crossovers/.+?/\d+/")):
self.story.addToList('category',stripHTML(a)) self.story.addToList('category',stripHTML(a))
else: found = True
if not found:
# Fall back. I ran across a story with a Crossver # Fall back. I ran across a story with a Crossver
# category link to a broken page once. # category link to a broken page once.
# http://www.fanfiction.net/s/2622060/1/ # http://www.fanfiction.net/s/2622060/1/
@@ -181,8 +181,6 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
for c in stripHTML(categories[0]).replace(" Crossover","").split(' + '): for c in stripHTML(categories[0]).replace(" Crossover","").split(' + '):
self.story.addToList('category',c) self.story.addToList('category',c)
a = soup.find('a', href=re.compile(r'https?://www\.fictionratings\.com/')) a = soup.find('a', href=re.compile(r'https?://www\.fictionratings\.com/'))
rating = a.string rating = a.string
if 'Fiction' in rating: # if rating has 'Fiction ', strip that out for consistency with past. if 'Fiction' in rating: # if rating has 'Fiction ', strip that out for consistency with past.
@@ -206,6 +204,12 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
# b.extract() # b.extract()
metatext = stripHTML(grayspan).replace('Hurt/Comfort','Hurt-Comfort') metatext = stripHTML(grayspan).replace('Hurt/Comfort','Hurt-Comfort')
#logger.debug("metatext:(%s)"%metatext) #logger.debug("metatext:(%s)"%metatext)
if 'Status: Complete' in metatext:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
metalist = metatext.split(" - ") metalist = metatext.split(" - ")
#logger.debug("metalist:(%s)"%metalist) #logger.debug("metalist:(%s)"%metalist)
@@ -240,35 +244,43 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
self.story.setMetadata('dateUpdated',datetime.fromtimestamp(float(dates[0]['data-xutime']))) self.story.setMetadata('dateUpdated',datetime.fromtimestamp(float(dates[0]['data-xutime'])))
self.story.setMetadata('datePublished',datetime.fromtimestamp(float(dates[-1]['data-xutime']))) self.story.setMetadata('datePublished',datetime.fromtimestamp(float(dates[-1]['data-xutime'])))
donechars = False # Meta key titles and the metadata they go into, if any.
metakeys = {
# These are already handled separately.
'Chapters':False,
'Status':False,
'id':False,
'Updated':False,
'Published':False,
'Reviews':'reviews',
'Favs':'favs',
'Follows':'follows',
'Words':'numWords',
}
chars_ships_list=[]
while len(metalist) > 0: while len(metalist) > 0:
if metalist[0].startswith('Chapters') or metalist[0].startswith('Status') or metalist[0].startswith('id:') or metalist[0].startswith('Updated:') or metalist[0].startswith('Published:'): m = metalist.pop(0)
pass if ':' in m:
elif metalist[0].startswith('Reviews'): key = m.split(':')[0].strip()
self.story.setMetadata('reviews',metalist[0].split(':')[1].strip()) if key in metakeys:
elif metalist[0].startswith('Favs:'): if metakeys[key]:
self.story.setMetadata('favs',metalist[0].split(':')[1].strip()) self.story.setMetadata(metakeys[key],m.split(':')[1].strip())
elif metalist[0].startswith('Follows:'): continue
self.story.setMetadata('follows',metalist[0].split(':')[1].strip()) # no ':' or not found in metakeys
elif metalist[0].startswith('Words'): chars_ships_list.append(m)
self.story.setMetadata('numWords',metalist[0].split(':')[1].strip())
elif not donechars:
# with 'pairing' support, pairings are bracketed w/o comma after
# [Caspian X, Lucy Pevensie] Edmund Pevensie, Peter Pevensie
self.story.extendList('characters',metalist[0].replace('[','').replace(']',',').split(','))
l = metalist[0] # all because sometimes chars can have ' - ' in them.
while '[' in l: chars_ships_text = (' - ').join(chars_ships_list)
self.story.addToList('ships',l[l.index('[')+1:l.index(']')].replace(', ','/')) # print("chars_ships_text:%s"%chars_ships_text)
l = l[l.index(']')+1:] # with 'pairing' support, pairings are bracketed w/o comma after
# [Caspian X, Lucy Pevensie] Edmund Pevensie, Peter Pevensie
self.story.extendList('characters',chars_ships_text.replace('[','').replace(']',',').split(','))
donechars = True l = chars_ships_text
metalist=metalist[1:] while '[' in l:
self.story.addToList('ships',l[l.index('[')+1:l.index(']')].replace(', ','/'))
if 'Status: Complete' in metatext: l = l[l.index(']')+1:]
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
if get_cover: if get_cover:
# Try the larger image first. # Try the larger image first.
@@ -317,7 +329,7 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
# Find the chapter selector # Find the chapter selector
select = soup.find('select', { 'name' : 'chapter' } ) select = soup.find('select', { 'name' : 'chapter' } )
if select is None: if select is None:
# no selector found, so it's a one-chapter story. # no selector found, so it's a one-chapter story.
self.chapterUrls.append((self.story.getMetadata('title'),url)) self.chapterUrls.append((self.story.getMetadata('title'),url))
@@ -337,36 +349,17 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
return return
def getChapterText(self, url): def getChapterText(self, url):
# time.sleep(4.0) ## ffnet(and, I assume, fpcom) tends to fail
# ## more if hit too fast. This is in
# ## additional to what ever the
# ## slow_down_sleep_time setting is.
logger.debug('Getting chapter text from: %s' % url) logger.debug('Getting chapter text from: %s' % url)
## ffnet(and, I assume, fpcom) tends to fail more if hit too
## fast. This is in additional to what ever the
## slow_down_sleep_time setting is.
data = self._fetchUrl(url,extrasleep=4.0) data = self._fetchUrl(url,extrasleep=4.0)
if "Please email this error message in full to <a href='mailto:support@fanfiction.com'>support@fanfiction.com</a>" in data: if "Please email this error message in full to <a href='mailto:support@fanfiction.com'>support@fanfiction.com</a>" in data:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! FanFiction.net Site Error!" % url) raise exceptions.FailedToDownload("Error downloading Chapter: %s! FanFiction.net Site Error!" % url)
# some ancient stories have body tags inside them that cause
# soup parsing to discard the content. For story text we
# don't care about anything before "<div role='main'" and
# this kills any body tags.
# XXX needed with new BS? -- No, doesn't look like it
# divstr = "<div role='main'"
# if divstr not in data:
# raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
# else:
# data = data[data.index(divstr):]
# data = data.replace("<body","<notbody").replace("<BODY","<NOTBODY")
soup = self.make_soup(data) soup = self.make_soup(data)
## Remove the 'share' button.
## No longer appears in the story text.
# sharediv = soup.find('div', {'class' : 'a2a_kit a2a_default_style'})
# if sharediv:
# sharediv.extract()
div = soup.find('div', {'id' : 'storytextp'}) div = soup.find('div', {'id' : 'storytextp'})
if None == div: if None == div:
+63 -42
View File
@@ -55,7 +55,7 @@ class FicBookNetAdapter(BaseSiteAdapter):
# normalized story URL. # normalized story URL.
self._setURL('http://' + self.getSiteDomain() + '/readfic/'+self.story.getMetadata('storyId')) self._setURL('https://' + self.getSiteDomain() + '/readfic/'+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation. # Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','fbn') self.story.setMetadata('siteabbrev','fbn')
@@ -71,10 +71,10 @@ class FicBookNetAdapter(BaseSiteAdapter):
@classmethod @classmethod
def getSiteExampleURLs(cls): def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/readfic/12345 http://"+cls.getSiteDomain()+"/readfic/93626/246417#part_content" return "https://"+cls.getSiteDomain()+"/readfic/12345 https://"+cls.getSiteDomain()+"/readfic/93626/246417#part_content"
def getSiteURLPattern(self): def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/readfic/")+r"\d+" return r"https?://"+re.escape(self.getSiteDomain()+"/readfic/")+r"\d+"
## Getting the chapter list and the meta data, plus 'is adult' checking. ## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self): def extractChapterUrlsAndMetadata(self):
@@ -92,39 +92,51 @@ class FicBookNetAdapter(BaseSiteAdapter):
# use BeautifulSoup HTML parser to make everything easier to find. # use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data) soup = self.make_soup(data)
adult_div = soup.find('div',id='adultCoverWarning')
if adult_div:
if self.is_adult or self.getConfig("is_adult"):
adult_div.extract()
else:
raise exceptions.AdultCheckRequired(self.url)
# Now go hunting for all the meta data and the chapter list. # Now go hunting for all the meta data and the chapter list.
table = soup.find('td',{'width':'50%'})
## Title ## Title
a = soup.find('h1') a = soup.find('section',{'class':'chapter-info'}).find('h1')
# kill '+' marks if present.
sup = a.find('sup')
if sup:
sup.extract()
self.story.setMetadata('title',stripHTML(a)) self.story.setMetadata('title',stripHTML(a))
logger.debug("Title: (%s)"%self.story.getMetadata('title')) logger.debug("Title: (%s)"%self.story.getMetadata('title'))
# Find authorid and URL from... author url. # Find authorid and URL from... author url.
a = table.find('a') # assume first avatar-nickname -- there can be a second marked 'beta'.
a = soup.find('a',{'class':'avatar-nickname'})
self.story.setMetadata('authorId',a.text) # Author's name is unique self.story.setMetadata('authorId',a.text) # Author's name is unique
self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) self.story.setMetadata('authorUrl','https://'+self.host+'/'+a['href'])
self.story.setMetadata('author',a.text) self.story.setMetadata('author',a.text)
logger.debug("Author: (%s)"%self.story.getMetadata('author')) logger.debug("Author: (%s)"%self.story.getMetadata('author'))
# Find the chapters: # Find the chapters:
chapters = soup.find('div', {'class' : 'part_list'}) chapters = soup.find('ul', {'class' : 'table-of-contents'})
if chapters != None: if chapters != None:
chapters=chapters.findAll('a', href=re.compile(r'/readfic/'+self.story.getMetadata('storyId')+"/\d+#part_content$")) chapters=chapters.findAll('a', href=re.compile(r'/readfic/'+self.story.getMetadata('storyId')+"/\d+#part_content$"))
self.story.setMetadata('numChapters',len(chapters)) self.story.setMetadata('numChapters',len(chapters))
for x in range(0,len(chapters)): for x in range(0,len(chapters)):
chapter=chapters[x] chapter=chapters[x]
churl='http://'+self.host+chapter['href'] churl='https://'+self.host+chapter['href']
self.chapterUrls.append((stripHTML(chapter),churl)) self.chapterUrls.append((stripHTML(chapter),churl))
if x == 0: if x == 0:
pubdate = translit.translit(stripHTML(self.make_soup(self._fetchUrl(churl)).find('div', {'class' : 'part_added'}).find('span'))) pubdate = translit.translit(stripHTML(chapter.parent.find('span')))
# pubdate = translit.translit(stripHTML(self.make_soup(self._fetchUrl(churl)).find('div', {'class' : 'part_added'}).find('span')))
if x == len(chapters)-1: if x == len(chapters)-1:
update = translit.translit(stripHTML(self.make_soup(self._fetchUrl(churl)).find('div', {'class' : 'part_added'}).find('span'))) update = translit.translit(stripHTML(chapter.parent.find('span')))
# update = translit.translit(stripHTML(self.make_soup(self._fetchUrl(churl)).find('div', {'class' : 'part_added'}).find('span')))
else: else:
self.chapterUrls.append((self.story.getMetadata('title'),url)) self.chapterUrls.append((self.story.getMetadata('title'),url))
self.story.setMetadata('numChapters',1) self.story.setMetadata('numChapters',1)
pubdate=translit.translit(stripHTML(soup.find('div', {'class' : 'part_added'}).find('span'))) pubdate=translit.translit(stripHTML(soup.find('div',{'class':'title-area'}).find('span')))
update=pubdate update=pubdate
logger.debug("numChapters: (%s)"%self.story.getMetadata('numChapters')) logger.debug("numChapters: (%s)"%self.story.getMetadata('numChapters'))
@@ -158,54 +170,63 @@ class FicBookNetAdapter(BaseSiteAdapter):
self.story.setMetadata('dateUpdated', makeDate(update, self.dateformat)) self.story.setMetadata('dateUpdated', makeDate(update, self.dateformat))
self.story.setMetadata('datePublished', makeDate(pubdate, self.dateformat)) self.story.setMetadata('datePublished', makeDate(pubdate, self.dateformat))
self.story.setMetadata('language','Russian') self.story.setMetadata('language','Russian')
pr=soup.find('a', href=re.compile(r'/printfic/\w+')) ## after site change, I don't see word count anywhere.
pr='http://'+self.host+pr['href'] # pr=soup.find('a', href=re.compile(r'/printfic/\w+'))
pr = self.make_soup(self._fetchUrl(pr)) # pr='https://'+self.host+pr['href']
pr=pr.findAll('div', {'class' : 'part_text'}) # pr = self.make_soup(self._fetchUrl(pr))
# pr=pr.findAll('div', {'class' : 'part_text'})
# i=0
# for part in pr:
# i=i+len(stripHTML(part).split(' '))
# self.story.setMetadata('numWords', unicode(i))
dlinfo = soup.find('dl',{'class':'info'})
i=0 i=0
for part in pr: fandoms = dlinfo.find('dd').findAll('a', href=re.compile(r'/fanfiction/\w+'))
i=i+len(stripHTML(part).split(' '))
self.story.setMetadata('numWords', unicode(i))
i=0
fandoms = table.findAll('a', href=re.compile(r'/fanfiction/\w+'))
for fandom in fandoms: for fandom in fandoms:
self.story.addToList('category',fandom.string) self.story.addToList('category',fandom.string)
i=i+1 i=i+1
if i > 1: if i > 1:
self.story.addToList('genre', u'Кроссовер') self.story.addToList('genre', u'Кроссовер')
meta=table.findAll('a', href=re.compile(r'/ratings/'))
i=0
for m in meta:
if i == 0:
self.story.setMetadata('rating', stripHTML(m))
i=1
elif i == 1:
if not "," in m.nextSibling:
i=2
self.story.addToList('genre', m.find('b').text)
elif i == 2:
self.story.addToList('warnings', m.find('b').text)
if table.find('span', {'style' : 'color: green'}): for genre in dlinfo.findAll('a',href=re.compile(r'/genres/')):
self.story.addToList('genre',stripHTML(genre))
ratingdt = dlinfo.find('dt',text='Рейтинг:')
self.story.setMetadata('rating', stripHTML(ratingdt.next_sibling))
# meta=table.findAll('a', href=re.compile(r'/ratings/'))
# i=0
# for m in meta:
# if i == 0:
# self.story.setMetadata('rating', stripHTML(m))
# i=1
# elif i == 1:
# if not "," in m.nextSibling:
# i=2
# self.story.addToList('genre', m.find('b').text)
# elif i == 2:
# self.story.addToList('warnings', m.find('b').text)
if dlinfo.find('span', {'style' : 'color: green'}):
self.story.setMetadata('status', 'Completed') self.story.setMetadata('status', 'Completed')
else: else:
self.story.setMetadata('status', 'In-Progress') self.story.setMetadata('status', 'In-Progress')
tags = table.findAll('b') tags = dlinfo.findAll('dt')
for tag in tags: for tag in tags:
label = translit.translit(tag.text) label = translit.translit(tag.text)
if 'Piersonazhi:' in label or u'Персонажи:' in label: if 'Piersonazhi:' in label or u'Персонажи:' in label:
chars=tag.nextSibling.string.split(', ') chars=stripHTML(tag.next_sibling).split(', ')
for char in chars: for char in chars:
self.story.addToList('characters',char) self.story.addToList('characters',char)
break break
summary=soup.find('span', {'class' : 'urlize'}) summary=soup.find('div', {'class' : 'urlize'})
self.setDescription(url,summary) self.setDescription(url,summary)
#self.story.setMetadata('description', summary.text) #self.story.setMetadata('description', summary.text)
@@ -0,0 +1,147 @@
# -*- coding: utf-8 -*-
# Copyright 2016 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
from .. import exceptions as exceptions
from ..htmlcleanup import stripHTML
from base_adapter import BaseSiteAdapter, makeDate
class FictionHuntComSiteAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.story.setMetadata('siteabbrev','fichunt')
# get storyId from url--url validation guarantees second part is storyId
self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2])
# normalized story URL.
self._setURL("http://"+self.getSiteDomain()\
+"/read/"+self.story.getMetadata('storyId')+"/1")
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%d-%m-%Y"
@staticmethod
def getSiteDomain():
return 'fictionhunt.com'
@classmethod
def getSiteExampleURLs(cls):
return "http://fictionhunt.com/read/1234/1"
def getSiteURLPattern(self):
return r"http://(www.)?fictionhunt.com/read/\d+(/\d+)?(/|/[^/]+)?/?$"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def doExtractChapterUrlsAndMetadata(self,get_cover=True):
# fetch the chapter. From that we will get almost all the
# metadata and chapter list
url = self.url
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.meta)
else:
raise e
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
self.story.setMetadata('title',stripHTML(soup.find('div',{'class':'title'})).strip())
self.setDescription(url,'<i>(Story descriptions not available on fictionhunt.com)</i>')
# Find authorid and URL from... author url.
# fictionhunt doesn't have author pages, use ffnet original author link.
a = soup.find('a', href=re.compile(r"fanfiction.net/u/\d+"))
self.story.setMetadata('authorId',a['href'].split('/')[-1])
self.story.setMetadata('authorUrl','https://www.fanfiction.net/u/'+self.story.getMetadata('authorId'))
self.story.setMetadata('author',a.string)
# Find original ffnet URL
a = soup.find('a', href=re.compile(r"fanfiction.net/s/\d+"))
self.story.setMetadata('origin',stripHTML(a))
self.story.setMetadata('originUrl',a['href'])
# Fleur D. & Harry P. & Hermione G. & Susan B. - Words: 42,848 - Rated: M - English - None - Chapters: 9 - Reviews: 248 - Updated: 21-09-2016 - Published: 16-05-2015 - by Elven Sorcerer (FFN)
# None - Words: 13,087 - Rated: M - English - Romance & Supernatural - Chapters: 3 - Reviews: 5 - Updated: 21-09-2016 - Published: 20-09-2016
# Harry P. & OC - Words: 10,910 - Rated: M - English - None - Chapters: 5 - Reviews: 6 - Updated: 21-09-2016 - Published: 11-09-2016
# Dudley D. & Harry P. & Nagini & Vernon D. - Words: 4,328 - Rated: K+ - English - None - Chapters: 2 - Updated: 21-09-2016 - Published: 20-09-2016 -
details = soup.find('div',{'class':'details'})
detail_re = \
r'(?P<characters>.+) - Words: (?P<numWords>[0-9,]+) - Rated: (?P<rating>[a-zA-Z\\+]+) - (?P<language>.+) - (?P<genre>.+)'+ \
r' - Chapters: (?P<numChapters>[0-9,]+)( - Reviews: (?P<reviews>[0-9,]+))? - Updated: (?P<dateUpdated>[0-9-]+)'+ \
r' - Published: (?P<datePublished>[0-9-]+)(?P<completed> - Complete)?'
details_dict = re.match(detail_re,stripHTML(details)).groupdict()
# lists
for meta in ('characters','genre'):
if details_dict[meta] != 'None':
self.story.extendList(meta,details_dict[meta].split(' & '))
# scalars
for meta in ('numWords','numChapters','rating','language','reviews'):
self.story.setMetadata(meta,details_dict[meta])
# dates
for meta in ('datePublished','dateUpdated'):
self.story.setMetadata(meta, makeDate(details_dict[meta], self.dateformat))
# status
if details_dict['completed']:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
# It's assumed that the number of chapters is correct.
# There's no complete list of chapters, so the only
# alternative is to get the num of chaps from the last
# indiated chapter list instead.
for i in range(1,1+int(self.story.getMetadata('numChapters'))):
self.chapterUrls.append(("Chapter "+unicode(i),"http://"+self.getSiteDomain()\
+"/read/"+self.story.getMetadata('storyId')+"/%s"%i))
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
data = self._fetchUrl(url)
soup = self.make_soup(data)
div = soup.find('div', {'class' : 'text'})
return self.utf8FromSoup(url,div)
def getClass():
return FictionHuntComSiteAdapter
@@ -117,7 +117,7 @@ class FictionManiaTVAdapter(BaseSiteAdapter):
self.story.setMetadata('rating', value) self.story.setMetadata('rating', value)
elif key == 'Complete': elif key == 'Complete':
self.story.setMetadata('status', 'Complete' if value == 'Complete' else 'In-Progress') self.story.setMetadata('status', 'Completed' if value == 'Complete' else 'In-Progress')
elif key == 'Categories': elif key == 'Categories':
for element in cells[1]('a'): for element in cells[1]('a'):
+1 -1
View File
@@ -93,7 +93,7 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
try: try:
data = self._fetchUrl(url) data = self._fetchUrl(url)
# non-existent/removed story urls get thrown to the front page. # non-existent/removed story urls get thrown to the front page.
if "<h2>Welcome to FicWad</h2>" in data: if "<h4>Featured Story</h4>" in data:
raise exceptions.StoryDoesNotExist(self.url) raise exceptions.StoryDoesNotExist(self.url)
soup = self.make_soup(data) soup = self.make_soup(data)
except urllib2.HTTPError, e: except urllib2.HTTPError, e:
@@ -14,7 +14,6 @@
# See the License for the specific language governing permissions and # See the License for the specific language governing permissions and
# limitations under the License. # limitations under the License.
# #
import re
from base_xenforoforum_adapter import BaseXenForoForumAdapter from base_xenforoforum_adapter import BaseXenForoForumAdapter
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2016 FanFicFare team # Copyright 2015 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -18,19 +18,19 @@
# Software: eFiction # Software: eFiction
from base_efiction_adapter import BaseEfictionAdapter from base_efiction_adapter import BaseEfictionAdapter
class RabidReaderComAdapter(BaseEfictionAdapter): class HaremLucifaelComAdapter(BaseEfictionAdapter):
@staticmethod @staticmethod
def getSiteDomain(): def getSiteDomain():
return 'www.therabidreader.com' return 'harem.lucifael.com'
@classmethod @classmethod
def getSiteAbbrev(self): def getSiteAbbrev(self):
return 'rrcom' return 'seraglio'
@classmethod @classmethod
def getDateFormat(self): def getDateFormat(self):
return "%d %b %Y" return "%d/%m/%Y"
def getClass(): def getClass():
return RabidReaderComAdapter return HaremLucifaelComAdapter
@@ -188,9 +188,13 @@ class HarryPotterFanFictionComSiteAdapter(BaseSiteAdapter):
data = self._fetchUrl(url) data = self._fetchUrl(url)
# remove everything after here--the site's chapters break the try:
# BS4 parser. # remove everything after here--the site's chapters break
data = data[:data.index('<script type="text/javascript" src="reviewjs.js">')] # the BS4 parser.
data = data[:data.index('<script type="text/javascript" src="reviewjs.js">')]
except:
# some older stories don't have the code at the end that breaks things.
pass
soup = self.make_soup(data) soup = self.make_soup(data)
@@ -29,11 +29,11 @@ from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate from base_adapter import BaseSiteAdapter, makeDate
def getClass(): def getClass():
return ScarHeadNetAdapter return KiaRepositoryMujajiNetAdapter ## XXX
# Class name has to be unique. Our convention is camel case the # Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped. # sitename with Adapter at the end. www is skipped.
class ScarHeadNetAdapter(BaseSiteAdapter): class KiaRepositoryMujajiNetAdapter(BaseSiteAdapter): # XXX
def __init__(self, config, url): def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url) BaseSiteAdapter.__init__(self, config, url)
@@ -52,26 +52,27 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
# normalized story URL. # normalized story URL.
self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) # XXX Most sites don't have the /fiction part. Replace all to remove it usually.
self._setURL('http://' + self.getSiteDomain() + '/repository/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation. # Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','shn') self.story.setMetadata('siteabbrev','kia') ## XXX
# The date format will vary from site to site. # The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior # http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%d/%m/%y" self.dateformat = "%d %b %Y" ## XXX
@staticmethod # must be @staticmethod, don't remove it. @staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain(): def getSiteDomain():
# The site domain. Does have www here, if it uses it. # The site domain. Does have www here, if it uses it.
return 'scarhead.net' return 'mujaji.net' # XXX
@classmethod @classmethod
def getSiteExampleURLs(cls): def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" return "http://"+cls.getSiteDomain()+"/repository/viewstory.php?sid=1234"
def getSiteURLPattern(self): def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" return re.escape("http://"+self.getSiteDomain()+"/repository/viewstory.php?sid=")+r"\d+$"
## Login seems to be reasonably standard across eFiction sites. ## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data): def needToLoginCheck(self, data):
@@ -94,7 +95,7 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
params['cookiecheck'] = '1' params['cookiecheck'] = '1'
params['submit'] = 'Submit' params['submit'] = 'Submit'
loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' loginUrl = 'http://' + self.getSiteDomain() + '/repository/user.php?action=login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['penname'])) params['penname']))
@@ -116,7 +117,7 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
# If the title search below fails, there's a good chance # If the title search below fails, there's a good chance
# you need a different number. print data at that point # you need a different number. print data at that point
# and see what the 'click here to continue' url says. # and see what the 'click here to continue' url says.
addurl = "&ageconsent=ok&warning=5" addurl = "&warning=4"
else: else:
addurl="" addurl=""
@@ -143,11 +144,11 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
# &amp;warning= -- actually, so do other sites. Must be an # &amp;warning= -- actually, so do other sites. Must be an
# eFiction book. # eFiction book.
# viewstory.php?sid=1882&amp;warning=4 # fiction/viewstory.php?sid=1882&amp;warning=4
# viewstory.php?sid=1654&amp;ageconsent=ok&amp;warning=5 # fiction/viewstory.php?sid=1654&amp;ageconsent=ok&amp;warning=5
#print data #print data
#m = re.search(r"'viewstory.php\?sid=1882(&amp;warning=4)'",data) m = re.search(r"'repository/viewstory.php\?sid=29(&amp;warning=4)'",data)
m = re.search(r"'viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data) m = re.search(r"'repository/viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data)
if m != None: if m != None:
if self.is_adult or self.getConfig("is_adult"): if self.is_adult or self.getConfig("is_adult"):
# We tried the default and still got a warning, so # We tried the default and still got a warning, so
@@ -170,18 +171,18 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url) raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data: if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find. # use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data) soup = self.make_soup(data)
# print data
# Now go hunting for all the meta data and the chapter list. # Now go hunting for all the meta data and the chapter list.
pagetitle = soup.find('tr',{'valign':'top'}) pagetitle = soup.find('div',{'id':'pagetitle'})
## Title ## Title
a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a)) self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url. # Find authorid and URL from... author url.
@@ -193,88 +194,87 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
# Find the chapters: # Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles. # just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/repository/'+chapter['href']+addurl))
self.story.setMetadata('numChapters',len(self.chapterUrls)) self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data # eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly. # formating, so it's a little ugly.
cats = soup.findAll('a',href=re.compile(r'browse.php\?type=categories')) # utility method
for cat in cats: def defaultGetattr(d,k):
if '/' == cat.string[0]:
self.story.addToList('ships','Harry Potter'+cat.string.split('(')[0])
elif 'Harry' in cat.string:
self.story.addToList('ships',cat.string.split('(')[0])
else:
self.story.addToList('category',cat.string)
if '(' in cat.string:
self.story.addToList('category',cat.string.split('(')[1].split(')')[0])
chars = soup.findAll('a',href=re.compile(r'browse.php\?type=characters'))
for char in chars:
self.story.addToList('characters',char.string)
genres = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2'))
for genre in genres:
self.story.addToList('genre',genre.string)
warnings = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1'))
for warning in warnings:
self.story.addToList('warnings',warning.string)
textsoup = stripHTML(soup)
a = textsoup.split('Published: ')[1].split(' ')[0]
self.story.setMetadata('datePublished', makeDate(stripHTML(a), self.dateformat))
a = textsoup.split('Updated: ')[1].split(' ')[0]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(a), self.dateformat))
a = textsoup.split('Rating: ')[1].split(' ')[0]
self.story.setMetadata('rating', a)
a = textsoup.split('Length: ')[1].split('(')[1].split(' ')[0]
self.story.setMetadata('numWords', a)
a = textsoup.split('Completed: ')[1].split(' ')[0]
if 'Yes' in a:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
#a = textsoup.split('Summary: ')[1].split('Add Story to Favorites')[0]
#self.setDescription(url,a)
a=soup.find(text=re.compile("Summary: "))
i=0
svalue = ""
while i == 0:
try: try:
b = unicode(a) return d[k]
svalue += b.split('Summary: ')[1]
except: except:
svalue += unicode(a) return ""
if a.nextSibling != None:
a = a.nextSibling # <span class="label">Rated:</span> NC-17<br /> etc
else: labels = soup.findAll('span',{'class':'label'})
a = a.parent.nextSibling for labelspan in labels:
if 'Disclaimer: ' in stripHTML(a): value = labelspan.nextSibling
i=1 label = labelspan.string
self.setDescription(url,svalue)
if 'Summary' in label:
## Everything until the next span class='label'
svalue = ""
while 'label' not in defaultGetattr(value,'class'):
svalue += unicode(value)
value = value.nextSibling
self.setDescription(url,svalue)
#self.story.setMetadata('description',stripHTML(svalue))
if 'Rated' in label:
self.story.setMetadata('rating', value)
if 'Word count' in label:
self.story.setMetadata('numWords', value)
if 'Categories' in label:
cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))
for cat in cats:
self.story.addToList('category',cat.string)
if 'Characters' in label:
chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters'))
for char in chars:
self.story.addToList('characters',char.string)
if 'Genre' in label:
genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1'))
for genre in genres:
self.story.addToList('genre',genre.string)
if 'Warnings' in label:
warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3'))
for warning in warnings:
self.story.addToList('warnings',warning.string)
if 'Completed' in label:
if 'Yes' in value:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
if 'Published' in label:
self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat))
if 'Updated' in label:
# there's a stray [ at the end.
#value = value[0:-1]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat))
try: try:
# Find Series name from series URL. # Find Series name from series URL.
a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) a = soup.find('a', href=re.compile(r"repository/viewseries.php\?seriesid=\d+"))
series_name = a.string series_name = a.string
series_url = 'http://'+self.host+'/'+a['href'] series_url = 'http://'+self.host+'/'+a['href']
# use BeautifulSoup HTML parser to make everything easier to find. # use BeautifulSoup HTML parser to make everything easier to find.
seriessoup = self.make_soup(self._fetchUrl(series_url)) seriessoup = self.make_soup(self._fetchUrl(series_url))
storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) storyas = seriessoup.findAll('a', href=re.compile(r'^repository/viewstory.php\?sid=\d+$'))
i=1 i=1
for a in storyas: for a in storyas:
if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): if a['href'] == ('repository/viewstory.php?sid='+self.story.getMetadata('storyId')):
self.setSeries(series_name, i) self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl',series_url) self.story.setMetadata('seriesUrl',series_url)
break break
@@ -283,7 +283,7 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
except: except:
# I find it hard to care if the series parsing fails # I find it hard to care if the series parsing fails
pass pass
# grab the text for an individual chapter. # grab the text for an individual chapter.
def getChapterText(self, url): def getChapterText(self, url):
@@ -297,3 +297,4 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div) return self.utf8FromSoup(url,div)
@@ -144,6 +144,9 @@ class LiteroticaSiteAdapter(BaseSiteAdapter):
else: else:
raise e raise e
if "This submission is awaiting moderator's approval" in data1:
raise exceptions.StoryDoesNotExist("This submission is awaiting moderator's approval. %s"%self.url)
# author # author
a = soup1.find("span", "b-story-user-y") a = soup1.find("span", "b-story-user-y")
self.story.setMetadata('authorId', urlparse.parse_qs(a.a['href'].split('?')[1])['uid'][0]) self.story.setMetadata('authorId', urlparse.parse_qs(a.a['href'].split('?')[1])['uid'][0])
+16 -1
View File
@@ -459,7 +459,22 @@ class Chapter(object):
.strip(u'| \n') .strip(u'| \n')
except AttributeError: except AttributeError:
raise ParsingError(u'Failed to locate date.') raise ParsingError(u'Failed to locate date.')
date = makeDate(dateText, '%d.%m.%Y')
# The site uses Europe/Moscow (MSK, UTC+0300) server time.
def todayInMoscow():
now = datetime.datetime.now() + datetime.timedelta(hours=3)
today = datetime.datetime(now.year, now.month, now.day)
return today
def parseDateText(text):
if text == u'Вчера':
return todayInMoscow() - datetime.timedelta(days=1)
elif text == u'Сегодня':
return todayInMoscow()
else:
return makeDate(text, '%d.%m.%Y')
date = parseDateText(dateText)
return date return date
def _getInfoBarElement(self): def _getInfoBarElement(self):
+14 -187
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2012 Fanficdownloader team, 2015 FanFicFare team # Copyright 2012 Fanficdownloader team, 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -15,195 +15,22 @@
# limitations under the License. # limitations under the License.
# #
import time # Software: eFiction
import logging from base_efiction_adapter import BaseEfictionAdapter
logger = logging.getLogger(__name__)
import re
import urllib2
class NCISFictionNetAdapter(BaseEfictionAdapter):
from ..htmlcleanup import stripHTML @staticmethod
from .. import exceptions as exceptions def getSiteDomain():
return 'ncisfiction.net'
from base_adapter import BaseSiteAdapter, makeDate @classmethod
def getSiteAbbrev(self):
return 'ncisfn'
@classmethod
def getDateFormat(self):
return "%m/%d/%Y"
def getClass(): def getClass():
return NCISFictionNetAdapter return NCISFictionNetAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class NCISFictionNetAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["iso-8859-1",
"Windows-1252"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
# normalized story URL.
self._setURL("http://"+self.getSiteDomain()\
+"/chapters.php?stid="+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','ncisfn')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%d/%m/%Y"
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'www.ncisfiction.net'
## Changed from www.ncisfiction.com to www.ncisfiction.net Oct
## 2012 due to the ncisfiction.com domain expiring. Still accept
## .com domains for existing updates, etc.
@classmethod
def getAcceptDomains(cls):
return ['www.ncisfiction.net','www.ncisfiction.com']
@classmethod
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/story.php?stid=01234 http://"+cls.getSiteDomain()+"/chapters.php?stid=1234"
def getSiteURLPattern(self):
return r'http://www\.ncisfiction\.(net|com)/(chapters|story)?.php\?stid=\d+'
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
url = self.url
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
# print data
# Now go hunting for all the meta data and the chapter list.
## Title and author
a = soup.find('div', {'class' : 'main_title'})
aut = a.find('a')
self.story.setMetadata('authorId',aut['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/'+aut['href'])
self.story.setMetadata('author',aut.string)
aut.extract()
self.story.setMetadata('title',stripHTML(a)[:len(stripHTML(a))-2])
# Find the chapters:
i=0
chapters=soup.findAll('table', {'class' : 'story_table'})
for chapter in chapters:
ch=chapter.find('a')
# just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(ch),'http://'+self.host+'/'+ch['href']))
if i == 0:
self.story.setMetadata('datePublished', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat))
if i == len(chapters)-1:
self.story.setMetadata('dateUpdated', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat))
i=i+1
self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly.
info = soup.find('table', {'class' : 'story_info'})
# no convenient way to calculate word count as it is logged differently for stories with and without series
labels = info.findAll('tr')
for tr in labels:
value = tr.find('td')
label = tr.find('th').string
if 'Summary' in label:
self.setDescription(url,value)
if 'Rating' in label:
self.story.setMetadata('rating', value.string)
if 'Category' in label:
cats = value.findAll('a')
for cat in cats:
self.story.addToList('category',cat.string)
if 'Characters' in label:
chars = value.findAll('a')
for char in chars:
self.story.addToList('characters',char.string)
if 'Pairing' in label:
ships = value.findAll('a')
for ship in ships:
self.story.addToList('ships',ship.string)
if 'Genre' in label:
genres = value.findAll('a')
for genre in genres:
self.story.addToList('genre',genre.string)
if 'Warnings' in label:
warnings = value.findAll('a')
for warning in warnings:
self.story.addToList('warnings',warning.string)
if 'Status' in label:
if 'not completed' in value.text:
self.story.setMetadata('status', 'In-Progress')
else:
self.story.setMetadata('status', 'Completed')
try:
# Find Series name from series URL.
a = soup.find('div',{'class' : 'sub_header'})
series_name = a.find('a').string
i = a.text.split('#')[1]
self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl','http://'+self.host+'/'+a.find('a')['href'])
except:
# I find it hard to care if the series parsing fails
pass
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
div = soup.find('div', {'class' : 'story_text'})
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div)
@@ -184,7 +184,6 @@ class NfaCommunityComAdapter(BaseSiteAdapter): # XXX
except: except:
return "" return ""
# <span class="label">Rated:</span> NC-17<br /> etc
labels = soup.findAll('span',{'class':'label'}) labels = soup.findAll('span',{'class':'label'})
for labelspan in labels: for labelspan in labels:
value = labelspan.nextSibling value = labelspan.nextSibling
@@ -195,7 +194,13 @@ class NfaCommunityComAdapter(BaseSiteAdapter): # XXX
svalue = "" svalue = ""
while 'label' not in defaultGetattr(value,'class'): while 'label' not in defaultGetattr(value,'class'):
svalue += unicode(value) svalue += unicode(value)
value = value.nextSibling # poor HTML(unclosed <p> for one) can cause run on
# over the next label.
if '<span class="label">' in svalue:
svalue = svalue[0:svalue.find('<span class="label">')]
break
else:
value = value.nextSibling
self.setDescription(url,svalue) self.setDescription(url,svalue)
#self.story.setMetadata('description',stripHTML(svalue)) #self.story.setMetadata('description',stripHTML(svalue))
@@ -1,179 +0,0 @@
# -*- coding: utf-8 -*-
# Copyright 2013 Fanficdownloader team, 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
def getClass():
return NickAndGregNetAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class NickAndGregNetAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["Windows-1252",
"utf8"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
# normalized story URL.
# XXX Most sites don't have the /fanfic part. Replace all to remove it usually.
self._setURL('http://' + self.getSiteDomain() + '/desert_archive/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','nag')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%Y/%m/%d"
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'www.nickngreg.nl'
@classmethod
def getAcceptDomains(cls):
return ['www.nickngreg.nl','www.nickandgreg.net']
@classmethod
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/desert_archive/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return "http://("+self.getSiteDomain()+"|www.nickandgreg.net)"+re.escape("/desert_archive/viewstory.php?sid=")+r"\d+$"
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
url = self.url+'&i=1'
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
# print data
# Now go hunting for all the meta data and the chapter list.
## Title
a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+"))
self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/desert_archive/'+a['href'])
self.story.setMetadata('author',a.string)
# Find the chapters:
chapters = soup.find('select')
for chapter in chapters.findAll('option'):
if chapter.text != 'Story Index' and chapter.text != 'Chapters':
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/desert_archive/'+chapter['value']))
self.story.setMetadata('numChapters',len(self.chapterUrls))
asoup = self.make_soup(self._fetchUrl(self.story.getMetadata('authorUrl')))
for div in asoup.findAll('td', {'class' : 'tblborder6'}):
a = div.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
if a != None:
break
self.setDescription(url,div.find('br').nextSibling)
a=div.text.split('Rating:')
if len(a) == 2: self.story.setMetadata('rating', a[1].split(' -')[0])
a=div.text.split('Characters:')
if len(a) == 2:
for char in a[1].split(' -')[0].split(', '):
self.story.addToList('characters',char)
a=div.text.split('Genres:')
if len(a) == 2:
for genre in a[1].split(' -')[0].split(', '):
self.story.addToList('genre',genre)
a=div.text.split('Warnings:')
if len(a) == 2:
for warn in a[1].split(' -')[0].split(', '):
if 'none' not in warn:
self.story.addToList('warnings',warn)
a=div.text.split('Completed:')
if len(a) ==2:
if 'Yes' in a[1]:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
a=div.text.split('Published:')
if len(a) == 2: self.story.setMetadata('datePublished', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat))
a=div.text.split('Updated:')
if len(a) == 2: self.story.setMetadata('dateUpdated', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat))
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
# wrap a div around it.
divsoup = self.make_soup('<div class="story"></div>')
div = divsoup.find('div')
div.append(soup.find('table', {'class' : 'tblborder6'}))
return self.utf8FromSoup(url,div)
+6 -6
View File
@@ -63,17 +63,17 @@ class QuotevComAdapter(BaseSiteAdapter):
title = element.find('h1') title = element.find('h1')
self.story.setMetadata('title', title.get_text()) self.story.setMetadata('title', title.get_text())
authdiv = title.next_sibling # soup.find('div', {'style':"font-size:0.7em;color:#aaa;margin-top:-10px;text-align:center;margin-left:45px;cursor:pointer"}) authdiv = soup.find('div', {'class':"quizAuthorList"})
if authdiv: if authdiv:
#print("div:%s"%authdiv.find_all('a')) print("div:%s"%authdiv)
for a in authdiv.find_all('a'): for a in authdiv.find_all('a'):
self.story.addToList('author', a.get_text()) self.story.addToList('author', a.get_text())
self.story.addToList('authorId', a['href'].split('/')[-1]) self.story.addToList('authorId', a['href'].split('/')[-1])
self.story.addToList('authorUrl', urlparse.urljoin(self.url, a['href'])) self.story.addToList('authorUrl', urlparse.urljoin(self.url, a['href']))
else: if not self.story.getList('author'):
self.story.setMetadata('author','Anonymous') self.story.addToList('author','Anonymous')
self.story.setMetadata('authorUrl','http://www.quotev.com') self.story.addToList('authorUrl','http://www.quotev.com')
self.story.setMetadata('authorId','0') self.story.addToList('authorId','0')
self.setDescription(self.url, soup.find('div', id='qdesct')) self.setDescription(self.url, soup.find('div', id='qdesct'))
imgmeta = soup.find('meta',{'property':"og:image" }) imgmeta = soup.find('meta',{'property':"og:image" })
+181
View File
@@ -0,0 +1,181 @@
# -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team, 2016 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
import cookielib as cl
from datetime import datetime
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
def getClass():
return RoyalRoadAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class RoyalRoadAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["utf8",
"Windows-1252"
] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only fiction/1234
self.story.setMetadata('storyId',re.match('/fiction/(\d+)(:/.+)?$',self.parsedUrl.path).groups()[0])
# normalized story URL.
self._setURL('http://' + self.getSiteDomain() + '/fiction/'+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','rylrdl')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = '%d/%m/%Y %H:%M:%S %p'
def make_date(self, parenttag):
# locale dates differ but the timestamp is easily converted
ts = parenttag.find('time')['unixtime']
return datetime.fromtimestamp(float(ts))
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'royalroadl.com'
@classmethod
def getAcceptDomains(cls):
return ['royalroadl.com','www.royalroadl.com']
@classmethod
def getSiteExampleURLs(cls):
return "https://royalroadl.com/fiction/3056"
def getSiteURLPattern(self):
return "https?"+re.escape("://")+r"(www\.|)royalroadl\.com/fiction/\d+$"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
url = self.url
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
# print data
## Title
title=soup.h2.text
self.story.setMetadata('title',title)
# Find authorid and URL from... author url.
author = soup.find('',{'class':'mt-card-social'})
author_link = author.findAll('li')[-1]
if author_link:
authorId = author_link.a['href'].split('=')[-1]
self.story.setMetadata('authorId', authorId)
self.story.setMetadata('authorUrl','http://'+self.host+'/member.php?action=profile&uid='+authorId)
self.story.setMetadata('author',soup.find(attrs=dict(property="books:author"))['content'])
chapters = soup.find('table',{'id':'chapters'}).find('tbody')
tds = [tr.findAll('td')[0] for tr in chapters.findAll('tr')]
for td in tds:
chapterUrl = 'http://' + self.getSiteDomain() + td.a['href']
self.chapterUrls.append((stripHTML(td.text), chapterUrl))
self.story.setMetadata('numChapters',len(self.chapterUrls))
# this is forum based so it's a bit ugly
description = soup.find('div', {'property': 'description', 'class': 'hidden-content'})
self.setDescription(url,description)
dates = [tr.findAll('td')[1] for tr in chapters.findAll('tr')]
self.story.setMetadata('dateUpdated', self.make_date(dates[-1]))
self.story.setMetadata('datePublished', self.make_date(dates[0]))
genre=[tag.text for tag in soup.find('input',{'property':'genre'}).parent.findChildren('span')]
if not "Unspecified" in genre:
for tag in genre:
self.story.addToList('genre',tag)
# 'rating' in FFF speak means G, PG, Teen, Restricted, etc.
# 'stars' is used instead for RR's 1-5 stars rating.
stars=soup.find(attrs=dict(property="books:rating:value"))['content']
self.story.setMetadata('stars',stars)
logger.debug(self.story.getMetadata('stars'))
warning = soup.find('strong',text='Warning')
if warning != None:
warnings=[c.text for c in warning.parent.children if getattr(c,'text',None)][1:]
for warntag in warnings:
self.story.addToList('warnings',warntag)
# get cover
img = soup.find('',{'class':'row fic-header'}).find('img')
if img:
cover_url = img['src']
self.setCoverImage(url,cover_url)
# some content is show as tables, this will preserve them
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
div = soup.find('div',{'class':"chapter-inner chapter-content"})
# TODO: these stories often have tables in, but these wont render correctly
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div)
+5
View File
@@ -112,6 +112,11 @@ class SiyeCoUkAdapter(BaseSiteAdapter): # XXX
# need(or easier) to pull other metadata from the author's list page. # need(or easier) to pull other metadata from the author's list page.
authsoup = self.make_soup(self._fetchUrl(self.story.getMetadata('authorUrl'))) authsoup = self.make_soup(self._fetchUrl(self.story.getMetadata('authorUrl')))
# remove author profile incase they've put the story URL in their bio.
profile = authsoup.find('div',{'id':'profile'})
if profile: # in case it changes.
profile.extract()
## Title ## Title
titlea = authsoup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) titlea = authsoup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(titlea)) self.story.setMetadata('title',stripHTML(titlea))
+20 -20
View File
@@ -97,7 +97,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
params['page'] = 'http://'+self.getSiteDomain()+'/' params['page'] = 'http://'+self.getSiteDomain()+'/'
params['submit'] = 'Login' params['submit'] = 'Login'
loginUrl = 'https://' + self.getSiteDomain() + '/login.php' loginUrl = 'https://' + self.getSiteDomain() + '/sol-secure/login.php'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['theusername'])) params['theusername']))
@@ -171,7 +171,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
self.story.addToList('author',stripHTML(a).replace("'s Page","")) self.story.addToList('author',stripHTML(a).replace("'s Page",""))
# Find the chapters: # Find the chapters:
chapters = soup.findAll('a', href=re.compile(r'^/s/'+self.story.getMetadata('storyId')+":\d+$")) chapters = soup.findAll('a', href=re.compile(r'^/s/'+self.story.getMetadata('storyId')+":\d+(/.*)?$"))
if len(chapters) != 0: if len(chapters) != 0:
for chapter in chapters: for chapter in chapters:
# just in case there's tags, like <i> in chapter titles. # just in case there's tags, like <i> in chapter titles.
@@ -211,7 +211,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
try: try:
a = lc4.find('a', href=re.compile(r"/series/\d+/.*")) a = lc4.find('a', href=re.compile(r"/series/\d+/.*"))
logger.debug("Looking for series - a='{0}'".format(a)) # logger.debug("Looking for series - a='{0}'".format(a))
if a: if a:
# if there's a number after the series name, series_contents is a two element list: # if there's a number after the series name, series_contents is a two element list:
# [<a href="...">Title</a>, u' (2)'] # [<a href="...">Title</a>, u' (2)']
@@ -220,34 +220,34 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
seriesUrl = 'http://'+self.host+a['href'] seriesUrl = 'http://'+self.host+a['href']
self.story.setMetadata('seriesUrl',seriesUrl) self.story.setMetadata('seriesUrl',seriesUrl)
series_name = stripHTML(a) series_name = stripHTML(a)
logger.debug("Series name= %s" % series_name) # logger.debug("Series name= %s" % series_name)
series_soup = self.make_soup(self._fetchUrl(seriesUrl)) series_soup = self.make_soup(self._fetchUrl(seriesUrl))
if series_soup: if series_soup:
logger.debug("Retrieving Series - looking for name") # logger.debug("Retrieving Series - looking for name")
series_name = stripHTML(series_soup.find('span', {'id' : 'ptitle'})) series_name = stripHTML(series_soup.find('span', {'id' : 'ptitle'}))
series_name = re.sub(r' . a series by.*$','',series_name) series_name = re.sub(r' . a series by.*$','',series_name)
logger.debug("Series name: '{0}'".format(series_name)) # logger.debug("Series name: '%s'" % series_name)
self.setSeries(series_name, i) self.setSeries(series_name, i)
desc = lc4.contents[2] desc = lc4.contents[2]
# Check if series is in a universe # Check if series is in a universe
universe_url = self.story.getList('authorUrl')[0] + "&type=uni" universe_url = self.story.getList('authorUrl')[0] + "&type=uni"
universes_soup = self.make_soup(self._fetchUrl(universe_url) ) universes_soup = self.make_soup(self._fetchUrl(universe_url) )
logger.debug("Universe url='{0}'".format(universe_url)) # logger.debug("Universe url='{0}'".format(universe_url))
if universes_soup: if universes_soup:
universes = universes_soup.findAll('div', {'class' : 'ser-box'}) universes = universes_soup.findAll('div', {'class' : 'ser-box'})
logger.debug("Number of Universes: %d" % len(universes)) # logger.debug("Number of Universes: %d" % len(universes))
for universe in universes: for universe in universes:
logger.debug("universe.find('a')={0}".format(universe.find('a'))) # logger.debug("universe.find('a')={0}".format(universe.find('a')))
# The universe id is in an "a" tag that has an id but nothing else. It is the first tag. # The universe id is in an "a" tag that has an id but nothing else. It is the first tag.
# The id is prefixed with the letter "u". # The id is prefixed with the letter "u".
universe_id = universe.find('a')['id'][1:] universe_id = universe.find('a')['id'][1:]
logger.debug("universe_id='%s'" % universe_id) # logger.debug("universe_id='%s'" % universe_id)
universe_name = stripHTML(universe.find('div', {'class' : 'ser-name'})).partition(' ')[2] universe_name = stripHTML(universe.find('div', {'class' : 'ser-name'})).partition(' ')[2]
logger.debug("universe_name='%s'" % universe_name) # logger.debug("universe_name='%s'" % universe_name)
# If there is link to the story, we have the right universe # If there is link to the story, we have the right universe
story_a = universe.find('a', href=re.compile('/s/'+self.story.getMetadata('storyId'))) story_a = universe.find('a', href=re.compile('/s/'+self.story.getMetadata('storyId')))
if story_a: if story_a:
logger.debug("Story is in a series that is in a universe! The universe is '%s'" % universe_name) # logger.debug("Story is in a series that is in a universe! The universe is '%s'" % universe_name)
self.story.setMetadata("universe", universe_name) self.story.setMetadata("universe", universe_name)
self.story.setMetadata('universeUrl','http://'+self.host+ '/library/universe.php?id=' + universe_id) self.story.setMetadata('universeUrl','http://'+self.host+ '/library/universe.php?id=' + universe_id)
break break
@@ -258,24 +258,24 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
pass pass
try: try:
a = lc4.find('a', href=re.compile(r"/universe/\d+/.*")) a = lc4.find('a', href=re.compile(r"/universe/\d+/.*"))
logger.debug("Looking for universe - a='{0}'".format(a)) # logger.debug("Looking for universe - a='{0}'".format(a))
if a: if a:
self.story.setMetadata("universe",stripHTML(a)) self.story.setMetadata("universe",stripHTML(a))
desc = lc4.contents[2] desc = lc4.contents[2]
# Assumed only one universe, but it does have a URL--use universeHTML # Assumed only one universe, but it does have a URL--use universeHTML
universe_name = stripHTML(a) universe_name = stripHTML(a)
universeUrl = 'http://'+self.host+a['href'] universeUrl = 'http://'+self.host+a['href']
logger.debug("Retrieving Universe - about to get page - universeUrl='{0}".format(universeUrl)) # logger.debug("Retrieving Universe - about to get page - universeUrl='{0}".format(universeUrl))
universe_soup = self.make_soup(self._fetchUrl(universeUrl)) universe_soup = self.make_soup(self._fetchUrl(universeUrl))
logger.debug("Retrieving Universe - have page") logger.debug("Retrieving Universe - have page")
if universe_soup: if universe_soup:
logger.debug("Retrieving Universe - looking for name") logger.debug("Retrieving Universe - looking for name")
universe_name = stripHTML(universe_soup.find('h1', {'id' : 'ptitle'})) universe_name = stripHTML(universe_soup.find('h1', {'id' : 'ptitle'}))
universe_name = re.sub(r' . A Universe from the Mind.*$','',universe_name) universe_name = re.sub(r' . A Universe from the Mind.*$','',universe_name)
logger.debug("Universes name: '{0}'".format(universe_name)) # logger.debug("Universes name: '{0}'".format(universe_name))
self.story.setMetadata('universeUrl',universeUrl) self.story.setMetadata('universeUrl',universeUrl)
logger.debug("Setting universe name: '{0}'".format(universe_name)) # logger.debug("Setting universe name: '{0}'".format(universe_name))
self.story.setMetadata('universe',universe_name) self.story.setMetadata('universe',universe_name)
if self.getConfig("universe_as_series"): if self.getConfig("universe_as_series"):
self.setSeries(universe_name, 0) self.setSeries(universe_name, 0)
@@ -320,12 +320,12 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
# http://storiesonline.net/s/11999 # http://storiesonline.net/s/11999
# http://storiesonline.net/s/10823 # http://storiesonline.net/s/10823
if get_cover: if get_cover:
logger.debug("Looking for the cover image...") # logger.debug("Looking for the cover image...")
cover_url = "" cover_url = ""
img = soup.find('img') img = soup.find('img')
if img: if img:
cover_url=img['src'] cover_url=img['src']
logger.debug("cover_url: %s"%cover_url) # logger.debug("cover_url: %s"%cover_url)
if cover_url: if cover_url:
self.setCoverImage(url,cover_url) self.setCoverImage(url,cover_url)
@@ -363,7 +363,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
urls=pager.findAll('a') urls=pager.findAll('a')
urls=urls[:len(urls)-1] urls=urls[:len(urls)-1]
logger.debug("pager urls:%s"%urls) # logger.debug("pager urls:%s"%urls)
pager.extract() pager.extract()
chaptertag.contents = chaptertag.contents[2:] chaptertag.contents = chaptertag.contents[2:]
@@ -372,7 +372,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
pagetag = soup.find('div', {'id' : 'story'}) pagetag = soup.find('div', {'id' : 'story'})
if not pagetag: if not pagetag:
logger.debug("div id=story not found, try article") # logger.debug("div id=story not found, try article")
pagetag = soup.find('article', {'id' : 'story'}) pagetag = soup.find('article', {'id' : 'story'})
self.cleanPage(pagetag) self.cleanPage(pagetag)
@@ -1,238 +0,0 @@
# -*- coding: utf-8 -*-
# Copyright 2013 Fanficdownloader team, 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
def getClass():
return TokraFandomnetComAdapter
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class TokraFandomnetComAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["Windows-1252",
"utf8"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
# normalized story URL.
self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','tokra')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%m/%d/%Y"
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it. But it
# doesn't matter too much anymore.
return 'tokra.fandomnet.com'
@classmethod
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
if self.is_adult or self.getConfig("is_adult"):
# Weirdly, different sites use different warning numbers.
# If the title search below fails, there's a good chance
# you need a different number. print data at that point
# and see what the 'click here to continue' url says.
addurl = "&ageconsent=ok&warning=3"
else:
addurl=""
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
url = self.url+'&index=1'+addurl
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
m = re.search(r"'viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data)
if m != None:
if self.is_adult or self.getConfig("is_adult"):
# We tried the default and still got a warning, so
# let's pull the warning number from the 'continue'
# link and reload data.
addurl = m.group(1)
# correct stupid &amp; error in url.
addurl = addurl.replace("&amp;","&")
url = self.url+'&index=1'+addurl
logger.debug("URL 2nd try: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
else:
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
#print data
# Now go hunting for all the meta data and the chapter list.
## Title
a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+"))
self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href'])
self.story.setMetadata('author',a.string)
# Rating
rate = stripHTML(soup.find('div',{'id':'pagetitle'}))
rate = rate[rate.rindex('[')+1:rate.rindex(']')]
self.story.setMetadata('rating', rate)
# Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl))
self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly.
metadiv = soup.find('div',{'class':'content'})
smalldiv = metadiv.find('div',{'class':'small'})
# tokra categories -> genre
# categories will be filled from ini.
genres = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))
for genre in genres:
self.story.addToList('genre',genre.string)
chars = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=characters'))
for char in chars:
self.story.addToList('characters',char.string)
metatext = stripHTML(smalldiv)
if 'Completed: Yes' in metatext:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
wordstart=metatext.rindex('Word count:')+12
words = metatext[wordstart:metatext.index(' ',wordstart)]
self.story.setMetadata('numWords', words)
datesdiv = soup.find('div',{'class':'bottom'})
dates = stripHTML(datesdiv).split()
# Published: 04/26/2011 Updated: 03/06/2013
self.story.setMetadata('datePublished', makeDate(dates[1], self.dateformat))
self.story.setMetadata('dateUpdated', makeDate(dates[3], self.dateformat))
try:
# Find Series name from series URL.
a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+"))
series_name = a.string
series_url = 'http://'+self.host+'/'+a['href']
# use BeautifulSoup HTML parser to make everything easier to find.
seriessoup = self.make_soup(self._fetchUrl(series_url))
# can't use ^viewstory...$ in case of higher rated stories with javascript href.
storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+'))
i=1
for a in storyas:
# skip 'report this' and 'TOC' links
if 'contact.php' not in a['href'] and 'index' not in a['href']:
if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')):
self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl',series_url)
break
i+=1
except:
# I find it hard to care if the series parsing fails
pass
# remove 'small' leaving only summary.
smalldiv.extract()
self.setDescription(url,metadiv)
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = self.make_soup(self._fetchUrl(url))
div = soup.find('div', {'class' : 'content'})
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
# remove some decorations while keeping notes.
remove = div.find('div', {'id' : 'pagetitle'})
remove.extract()
for remove in div.findAll('div', {'class' : 'right'}):
remove.extract()
for remove in div.findAll('div', {'class' : 'left'}):
remove.extract()
return self.utf8FromSoup(url,div)
+4 -3
View File
@@ -125,9 +125,10 @@ class WraithBaitComAdapter(BaseSiteAdapter):
rating=pt.text.split('[')[1].split(']')[0] rating=pt.text.split('[')[1].split(']')[0]
self.story.setMetadata('rating', rating) self.story.setMetadata('rating', rating)
st = soup.find('div', {'class' : 'storytitle'}) # site stopped showing reviews ~ Oct 2016
a = st.findAll('a', href=re.compile(r'reviews.php\?type=ST&item='+self.story.getMetadata('storyId')+"$"))[1] # second one. # st = soup.find('div', {'class' : 'storytitle'})
self.story.setMetadata('reviews',stripHTML(a)) # a = st.findAll('a', href=re.compile(r'reviews.php\?type=ST&item='+self.story.getMetadata('storyId')+"$"))[1] # second one.
# self.story.setMetadata('reviews',stripHTML(a))
# Find the chapters: # Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
+112 -64
View File
@@ -16,7 +16,8 @@
# #
import re import re
import datetime from datetime import datetime, timedelta
import time import time
import logging import logging
import urllib import urllib
@@ -83,7 +84,7 @@ class BaseSiteAdapter(Configurable):
def __init__(self, configuration, url): def __init__(self, configuration, url):
Configurable.__init__(self, configuration) Configurable.__init__(self, configuration)
self.username = "NoneGiven" # if left empty, site doesn't return any message at all. self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = "" self.password = ""
self.is_adult=False self.is_adult=False
@@ -99,7 +100,7 @@ class BaseSiteAdapter(Configurable):
self.metadataDone = False self.metadataDone = False
self.story = Story(configuration) self.story = Story(configuration)
self.story.setMetadata('site',self.getConfigSection()) self.story.setMetadata('site',self.getConfigSection())
self.story.setMetadata('dateCreated',datetime.datetime.now()) self.story.setMetadata('dateCreated',datetime.now())
self.chapterUrls = [] # tuples of (chapter title,chapter url) self.chapterUrls = [] # tuples of (chapter title,chapter url)
self.chapterFirst = None self.chapterFirst = None
self.chapterLast = None self.chapterLast = None
@@ -112,7 +113,7 @@ class BaseSiteAdapter(Configurable):
self.logfile = None self.logfile = None
self.pagecache = self.get_empty_pagecache() self.pagecache = self.get_empty_pagecache()
## order of preference for decoding. ## order of preference for decoding.
self.decode = ["utf8", self.decode = ["utf8",
"Windows-1252"] # 1252 is a superset of "Windows-1252"] # 1252 is a superset of
@@ -131,20 +132,20 @@ class BaseSiteAdapter(Configurable):
def set_cookiejar(self,cj): def set_cookiejar(self,cj):
self.cookiejar = cj self.cookiejar = cj
saveheaders = self.opener.addheaders
self.opener = u2.build_opener(u2.HTTPCookieProcessor(self.cookiejar),GZipProcessor()) self.opener = u2.build_opener(u2.HTTPCookieProcessor(self.cookiejar),GZipProcessor())
self.opener.addheaders = [('User-Agent', self.getConfig('user_agent')), self.opener.addheaders = saveheaders
('X-Clacks-Overhead','GNU Terry Pratchett')]
def load_cookiejar(self,filename): def load_cookiejar(self,filename):
''' '''
Needs to be called after adapter create, but before any fetchs Needs to be called after adapter create, but before any fetchs
are done. Takes file *name*. are done. Takes file *name*.
''' '''
self.get_cookiejar().load(filename, ignore_discard=True, ignore_expires=True) self.get_cookiejar().load(filename, ignore_discard=True, ignore_expires=True)
def get_pagecache(self): def get_pagecache(self):
return self.pagecache return self.pagecache
def set_pagecache(self,d): def set_pagecache(self,d):
self.pagecache=d self.pagecache=d
@@ -158,7 +159,7 @@ class BaseSiteAdapter(Configurable):
def _has_cachekey(self,cachekey): def _has_cachekey(self,cachekey):
return self.use_pagecache() and cachekey in self.get_pagecache() return self.use_pagecache() and cachekey in self.get_pagecache()
def _get_from_pagecache(self,cachekey): def _get_from_pagecache(self,cachekey):
if self.use_pagecache(): if self.use_pagecache():
return self.get_pagecache().get(cachekey) return self.get_pagecache().get(cachekey)
@@ -175,18 +176,18 @@ class BaseSiteAdapter(Configurable):
this and change it to True. this and change it to True.
''' '''
return False return False
# def story_load(self,filename): # def story_load(self,filename):
# d = pickle.load(self.story.metadata,filename) # d = pickle.load(self.story.metadata,filename)
# self.story.metadata = d['metadata'] # self.story.metadata = d['metadata']
# self.chapterUrls = d['chapterlist'] # self.chapterUrls = d['chapterlist']
# self.story.metadataDone = True # self.story.metadataDone = True
def _setURL(self,url): def _setURL(self,url):
self.url = url self.url = url
self.parsedUrl = up.urlparse(url) self.parsedUrl = up.urlparse(url)
self.host = self.parsedUrl.netloc self.host = self.parsedUrl.netloc
self.path = self.parsedUrl.path self.path = self.parsedUrl.path
self.story.setMetadata('storyUrl',self.url,condremoveentities=False) self.story.setMetadata('storyUrl',self.url,condremoveentities=False)
## website encoding(s)--in theory, each website reports the character ## website encoding(s)--in theory, each website reports the character
@@ -200,7 +201,7 @@ class BaseSiteAdapter(Configurable):
decode = self.getConfigList('website_encodings') decode = self.getConfigList('website_encodings')
else: else:
decode = self.decode decode = self.decode
for code in decode: for code in decode:
try: try:
#print code #print code
@@ -229,7 +230,7 @@ class BaseSiteAdapter(Configurable):
usecache=True): usecache=True):
''' '''
When should cache be cleared or not used? logins... When should cache be cleared or not used? logins...
extrasleep is primarily for ffnet adapter which has extra extrasleep is primarily for ffnet adapter which has extra
sleeps. Passed into fetchs so it can be bypassed when sleeps. Passed into fetchs so it can be bypassed when
cache hits. cache hits.
@@ -239,7 +240,7 @@ class BaseSiteAdapter(Configurable):
logger.debug("#####################################\npagecache HIT: %s"%safe_url(cachekey)) logger.debug("#####################################\npagecache HIT: %s"%safe_url(cachekey))
data,redirecturl = self._get_from_pagecache(cachekey) data,redirecturl = self._get_from_pagecache(cachekey)
return data return data
logger.debug("#####################################\npagecache MISS: %s"%safe_url(cachekey)) logger.debug("#####################################\npagecache MISS: %s"%safe_url(cachekey))
self.do_sleep(extrasleep) self.do_sleep(extrasleep)
@@ -260,19 +261,19 @@ class BaseSiteAdapter(Configurable):
parameters=None, parameters=None,
extrasleep=None, extrasleep=None,
usecache=True): usecache=True):
return self._fetchUrlRawOpened(url, return self._fetchUrlRawOpened(url,
parameters, parameters,
extrasleep, extrasleep,
usecache)[0] usecache)[0]
def _fetchUrlRawOpened(self, url, def _fetchUrlRawOpened(self, url,
parameters=None, parameters=None,
extrasleep=None, extrasleep=None,
usecache=True): usecache=True):
''' '''
When should cache be cleared or not used? logins... When should cache be cleared or not used? logins...
extrasleep is primarily for ffnet adapter which has extra extrasleep is primarily for ffnet adapter which has extra
sleeps. Passed into fetchs so it can be bypassed when sleeps. Passed into fetchs so it can be bypassed when
cache hits. cache hits.
@@ -288,7 +289,7 @@ class BaseSiteAdapter(Configurable):
def geturl(self): return self.url def geturl(self): return self.url
def read(self): return self.data def read(self): return self.data
return (data,FakeOpened(data,redirecturl)) return (data,FakeOpened(data,redirecturl))
logger.debug("#####################################\npagecache MISS: %s"%safe_url(cachekey)) logger.debug("#####################################\npagecache MISS: %s"%safe_url(cachekey))
self.do_sleep(extrasleep) self.do_sleep(extrasleep)
if parameters != None: if parameters != None:
@@ -297,13 +298,13 @@ class BaseSiteAdapter(Configurable):
opened = self.opener.open(url.replace(' ','%20'),None,float(self.getConfig('connect_timeout',30.0))) opened = self.opener.open(url.replace(' ','%20'),None,float(self.getConfig('connect_timeout',30.0)))
data = opened.read() data = opened.read()
self._set_to_pagecache(cachekey,data,opened.url) self._set_to_pagecache(cachekey,data,opened.url)
return (data,opened) return (data,opened)
def set_sleep(self,val): def set_sleep(self,val):
logger.debug("\n===========\n set sleep time %s\n==========="%val) logger.debug("\n===========\n set sleep time %s\n==========="%val)
self.override_sleep = val self.override_sleep = val
def do_sleep(self,extrasleep=None): def do_sleep(self,extrasleep=None):
if extrasleep: if extrasleep:
time.sleep(float(extrasleep)) time.sleep(float(extrasleep))
@@ -311,7 +312,7 @@ class BaseSiteAdapter(Configurable):
time.sleep(float(self.override_sleep)) time.sleep(float(self.override_sleep))
elif self.getConfig('slow_down_sleep_time'): elif self.getConfig('slow_down_sleep_time'):
time.sleep(float(self.getConfig('slow_down_sleep_time'))) time.sleep(float(self.getConfig('slow_down_sleep_time')))
def _fetchUrl(self, url, def _fetchUrl(self, url,
parameters=None, parameters=None,
usecache=True, usecache=True,
@@ -329,7 +330,7 @@ class BaseSiteAdapter(Configurable):
excpt=None excpt=None
for sleeptime in [0, 0.5, 4, 9]: for sleeptime in [0, 0.5, 4, 9]:
time.sleep(sleeptime) time.sleep(sleeptime)
try: try:
(data,opened)=self._fetchUrlRawOpened(url, (data,opened)=self._fetchUrlRawOpened(url,
parameters=parameters, parameters=parameters,
@@ -344,7 +345,7 @@ class BaseSiteAdapter(Configurable):
except Exception, e: except Exception, e:
excpt=e excpt=e
logger.warn("Caught an exception reading URL: %s sleeptime(%s) Exception %s."%(unicode(safe_url(url)),sleeptime,unicode(e))) logger.warn("Caught an exception reading URL: %s sleeptime(%s) Exception %s."%(unicode(safe_url(url)),sleeptime,unicode(e)))
logger.error("Giving up on %s" %safe_url(url)) logger.error("Giving up on %s" %safe_url(url))
logger.debug(excpt, exc_info=True) logger.debug(excpt, exc_info=True)
raise(excpt) raise(excpt)
@@ -356,12 +357,16 @@ class BaseSiteAdapter(Configurable):
if last: if last:
self.chapterLast=int(last)-1 self.chapterLast=int(last)-1
self.story.set_chapters_range(first,last) self.story.set_chapters_range(first,last)
# Does the download the first time it's called. # Does the download the first time it's called.
def getStory(self): def getStory(self):
if not self.storyDone: if not self.storyDone:
self.getStoryMetadataOnly(get_cover=True) self.getStoryMetadataOnly(get_cover=True)
## one-off step to normalize old chapter URLs if present.
if self.oldchaptersmap:
self.oldchaptersmap = dict((self.normalize_chapterurl(key), value) for (key, value) in self.oldchaptersmap.items())
for index, (title,url) in enumerate(self.chapterUrls): for index, (title,url) in enumerate(self.chapterUrls):
newchap = False newchap = False
if (self.chapterFirst!=None and index < self.chapterFirst) or \ if (self.chapterFirst!=None and index < self.chapterFirst) or \
@@ -387,18 +392,25 @@ class BaseSiteAdapter(Configurable):
url in self.oldchaptersdata and ( url in self.oldchaptersdata and (
self.oldchaptersdata[url]['chapterorigtitle'] != self.oldchaptersdata[url]['chapterorigtitle'] !=
self.oldchaptersdata[url]['chaptertitle']) ) self.oldchaptersdata[url]['chaptertitle']) )
if not data: if not data:
data = self.getChapterText(url) data = self.getChapterText(url)
# if had to fetch and has existing chapters # if had to fetch and has existing chapters
newchap = bool(self.oldchapters or self.oldchaptersmap) newchap = bool(self.oldchapters or self.oldchaptersmap)
if index == 0 and self.getConfig('always_reload_first_chapter'):
data = self.getChapterText(url)
# first chapter is rarely marked new
# anyway--only if it's replaced during an
# update.
newchap = False
self.story.addChapter(url, self.story.addChapter(url,
removeEntities(title), removeEntities(title),
removeEntities(data), removeEntities(data),
newchap) newchap)
self.storyDone = True self.storyDone = True
# include image, but no cover from story, add default_cover_image cover. # include image, but no cover from story, add default_cover_image cover.
if self.getConfig('include_images') and \ if self.getConfig('include_images') and \
not self.story.cover and \ not self.story.cover and \
@@ -415,26 +427,30 @@ class BaseSiteAdapter(Configurable):
if not self.story.cover and self.oldcover: if not self.story.cover and self.oldcover:
self.story.oldcover = self.oldcover self.story.oldcover = self.oldcover
self.story.setMetadata('cover_image','old') self.story.setMetadata('cover_image','old')
# cheesy way to carry calibre bookmark file forward across update. # cheesy way to carry calibre bookmark file forward across update.
if self.calibrebookmark: if self.calibrebookmark:
self.story.calibrebookmark = self.calibrebookmark self.story.calibrebookmark = self.calibrebookmark
if self.logfile: if self.logfile:
self.story.logfile = self.logfile self.story.logfile = self.logfile
return self.story return self.story
def getStoryMetadataOnly(self,get_cover=True): def getStoryMetadataOnly(self,get_cover=True):
if not self.metadataDone: if not self.metadataDone:
self.doExtractChapterUrlsAndMetadata(get_cover=get_cover) self.doExtractChapterUrlsAndMetadata(get_cover=get_cover)
if not self.story.getMetadataRaw('dateUpdated'): if not self.story.getMetadataRaw('dateUpdated'):
if self.story.getMetadataRaw('datePublished'): if self.story.getMetadataRaw('datePublished'):
self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('datePublished')) self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('datePublished'))
else: else:
self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('dateCreated')) self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('dateCreated'))
self.metadataDone = True self.metadataDone = True
# normalize chapter urls.
for index, (title,url) in enumerate(self.chapterUrls):
self.chapterUrls[index] = (title,self.normalize_chapterurl(url))
return self.story return self.story
def setStoryMetadata(self,metahtml): def setStoryMetadata(self,metahtml):
@@ -445,36 +461,36 @@ class BaseSiteAdapter(Configurable):
if self.story.getMetadataRaw('datePublished'): if self.story.getMetadataRaw('datePublished'):
self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('datePublished')) self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('datePublished'))
else: else:
self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('dateCreated')) self.story.setMetadata('dateUpdated',self.story.getMetadataRaw('dateCreated'))
def hookForUpdates(self,chaptercount): def hookForUpdates(self,chaptercount):
"Usually not needed." "Usually not needed."
return chaptercount return chaptercount
############################### ###############################
@staticmethod @staticmethod
def getSiteDomain(): def getSiteDomain():
"Needs to be overriden in each adapter class." "Needs to be overriden in each adapter class."
return 'no such domain' return 'no such domain'
@classmethod @classmethod
def getConfigSection(cls): def getConfigSection(cls):
"Only needs to be overriden if != site domain." "Only needs to be overriden if != site domain."
return cls.getSiteDomain() return cls.getSiteDomain()
@classmethod @classmethod
def getConfigSections(cls): def getConfigSections(cls):
"Only needs to be overriden if has additional ini sections." "Only needs to be overriden if has additional ini sections."
return [cls.getConfigSection()] return [cls.getConfigSection()]
@classmethod @classmethod
def stripURLParameters(cls,url): def stripURLParameters(cls,url):
"Only needs to be overriden if URL contains more than one parameter" "Only needs to be overriden if URL contains more than one parameter"
## remove any trailing '&' parameters--?sid=999 will be left. ## remove any trailing '&' parameters--?sid=999 will be left.
## that's all that any of the current adapters need or want. ## that's all that any of the current adapters need or want.
return re.sub(r"&.*$","",url) return re.sub(r"&.*$","",url)
## URL pattern validation is done *after* picking an adaptor based ## URL pattern validation is done *after* picking an adaptor based
## on domain instead of *as* the adaptor selector so we can offer ## on domain instead of *as* the adaptor selector so we can offer
## the user example(s) for that particular site. ## the user example(s) for that particular site.
@@ -482,7 +498,7 @@ class BaseSiteAdapter(Configurable):
def getSiteURLPattern(self): def getSiteURLPattern(self):
"Used to validate URL. Should be override in each adapter class." "Used to validate URL. Should be override in each adapter class."
return '^http://'+re.escape(self.getSiteDomain()) return '^http://'+re.escape(self.getSiteDomain())
@classmethod @classmethod
def getSiteExampleURLs(cls): def getSiteExampleURLs(cls):
""" """
@@ -492,7 +508,7 @@ class BaseSiteAdapter(Configurable):
validateURL method. validateURL method.
""" """
return 'no such example' return 'no such example'
def doExtractChapterUrlsAndMetadata(self,get_cover=True): def doExtractChapterUrlsAndMetadata(self,get_cover=True):
''' '''
There are a handful of adapters that fetch a cover image while There are a handful of adapters that fetch a cover image while
@@ -501,7 +517,7 @@ class BaseSiteAdapter(Configurable):
this instead of extractChapterUrlsAndMetadata() this instead of extractChapterUrlsAndMetadata()
''' '''
return self.extractChapterUrlsAndMetadata() return self.extractChapterUrlsAndMetadata()
def extractChapterUrlsAndMetadata(self): def extractChapterUrlsAndMetadata(self):
"Needs to be overriden in each adapter class. Populates self.story metadata and self.chapterUrls" "Needs to be overriden in each adapter class. Populates self.story metadata and self.chapterUrls"
pass pass
@@ -553,7 +569,7 @@ class BaseSiteAdapter(Configurable):
# bs4 # bs4
return soup.attrs.keys() return soup.attrs.keys()
return [] return []
# This gives us a unicode object, not just a string containing bytes. # This gives us a unicode object, not just a string containing bytes.
# (I gave soup a unicode string, you'd think it could give it back...) # (I gave soup a unicode string, you'd think it could give it back...)
# Now also does a bunch of other common processing for us. # Now also does a bunch of other common processing for us.
@@ -562,12 +578,12 @@ class BaseSiteAdapter(Configurable):
fetch=self._fetchUrlRaw fetch=self._fetchUrlRaw
acceptable_attributes = self.getConfigList('keep_html_attrs',['href','name','class','id']) acceptable_attributes = self.getConfigList('keep_html_attrs',['href','name','class','id'])
if self.getConfig("keep_style_attr"): if self.getConfig("keep_style_attr"):
acceptable_attributes.append('style') acceptable_attributes.append('style')
if self.getConfig("keep_title_attr"): if self.getConfig("keep_title_attr"):
acceptable_attributes.append('title') acceptable_attributes.append('title')
#print("include_images:"+self.getConfig('include_images')) #print("include_images:"+self.getConfig('include_images'))
if self.getConfig('include_images'): if self.getConfig('include_images'):
acceptable_attributes.extend(('src','alt','longdesc')) acceptable_attributes.extend(('src','alt','longdesc'))
@@ -584,6 +600,19 @@ class BaseSiteAdapter(Configurable):
if attr not in acceptable_attributes: if attr not in acceptable_attributes:
del soup[attr] ## strip all tag attributes except href and name del soup[attr] ## strip all tag attributes except href and name
## apply adapter's normalize_chapterurls to all links in
## chapter texts, if they match chapter URLs. While this will
## be occasionally helpful by itself, it's really for the next
## feature: internal text links.
if self.getConfig('normalize_text_links'):
for alink in soup.find_all('a'):
# try:
if alink.has_attr('href'):
# logger.debug("normalize_text_links %s -> %s"%(alink['href'],self.normalize_chapterurl(alink['href'])))
alink['href'] = self.normalize_chapterurl(alink['href'])
# except AttributeError as ae:
# logger.info("Parsing for normalize_text_links failed...")
try: try:
# as a generator, each tag will be returned even if there's a # as a generator, each tag will be returned even if there's a
# mismatch at the end. # mismatch at the end.
@@ -591,8 +620,8 @@ class BaseSiteAdapter(Configurable):
for attr in self.get_attr_keys(t): for attr in self.get_attr_keys(t):
if attr not in acceptable_attributes: if attr not in acceptable_attributes:
del t[attr] ## strip all tag attributes except acceptable_attributes del t[attr] ## strip all tag attributes except acceptable_attributes
# these are not acceptable strict XHTML. But we do already have # these are not acceptable strict XHTML. But we do already have
# CSS classes of the same names defined # CSS classes of the same names defined
if t and hasattr(t,'name') and t.name is not None: if t and hasattr(t,'name') and t.name is not None:
if t.name in self.getConfigList('replace_tags_with_spans',['u']): if t.name in self.getConfigList('replace_tags_with_spans',['u']):
@@ -608,11 +637,11 @@ class BaseSiteAdapter(Configurable):
# remove script tags cross the board. # remove script tags cross the board.
if t.name=='script': if t.name=='script':
t.extract() t.extract()
except AttributeError, ae: except AttributeError, ae:
if "%s"%ae != "'NoneType' object has no attribute 'next_element'": if "%s"%ae != "'NoneType' object has no attribute 'next_element'":
logger.error("Error parsing HTML, probably poor input HTML. %s"%ae) logger.error("Error parsing HTML, probably poor input HTML. %s"%ae)
retval = unicode(soup) retval = unicode(soup)
if self.getConfig('nook_img_fix') and not self.getConfig('replace_br_with_p'): if self.getConfig('nook_img_fix') and not self.getConfig('replace_br_with_p'):
@@ -621,16 +650,16 @@ class BaseSiteAdapter(Configurable):
# that under the text for the rest of the chapter. # that under the text for the rest of the chapter.
retval = re.sub(r"(?!<(div|p)>)\s*(?P<imgtag><img[^>]+>)\s*(?!</(div|p)>)", retval = re.sub(r"(?!<(div|p)>)\s*(?P<imgtag><img[^>]+>)\s*(?!</(div|p)>)",
"<div>\g<imgtag></div>",retval) "<div>\g<imgtag></div>",retval)
# Don't want html, head or body tags in chapter html--writers add them. # Don't want html, head or body tags in chapter html--writers add them.
# This is primarily for epub updates. # This is primarily for epub updates.
retval = re.sub(r"</?(html|head|body)[^>]*>\r?\n?","",retval) retval = re.sub(r"</?(html|head|body)[^>]*>\r?\n?","",retval)
if self.getConfig("replace_br_with_p") and allow_replace_br_with_p: if self.getConfig("replace_br_with_p") and allow_replace_br_with_p:
# Apply heuristic processing to replace <br> paragraph # Apply heuristic processing to replace <br> paragraph
# breaks with <p> tags. # breaks with <p> tags.
retval = replace_br_with_p(retval) retval = replace_br_with_p(retval)
if self.getConfig('replace_hr'): if self.getConfig('replace_hr'):
# replacing a self-closing tag with a container tag in the # replacing a self-closing tag with a container tag in the
# soup is more difficult than it first appears. So cheat. # soup is more difficult than it first appears. So cheat.
@@ -640,31 +669,35 @@ class BaseSiteAdapter(Configurable):
def make_soup(self,data): def make_soup(self,data):
''' '''
Convenience method for getting a bs4 soup. Older and Convenience method for getting a bs4 soup. bs3 has been removed.
non-updated adapters call the included bs3 library themselves.
''' '''
## html5lib handles <noscript> oddly. See: ## html5lib handles <noscript> oddly. See:
## https://bugs.launchpad.net/beautifulsoup/+bug/1277464 ## https://bugs.launchpad.net/beautifulsoup/+bug/1277464
## This should 'hide' and restore <noscript> tags. ## This should 'hide' and restore <noscript> tags.
data = data.replace("noscript>","fff_hide_noscript>") data = data.replace("noscript>","fff_hide_noscript>")
## soup and re-soup because BS4/html5lib is more forgiving of ## soup and re-soup because BS4/html5lib is more forgiving of
## incorrectly nested tags that way. ## incorrectly nested tags that way.
soup = bs4.BeautifulSoup(data,'html5lib') soup = bs4.BeautifulSoup(data,'html5lib')
soup = bs4.BeautifulSoup(unicode(soup),'html5lib') soup = bs4.BeautifulSoup(unicode(soup),'html5lib')
for ns in soup.find_all('fff_hide_noscript'): for ns in soup.find_all('fff_hide_noscript'):
ns.name = 'noscript' ns.name = 'noscript'
return soup return soup
## For adapters, especially base_xenforoforum to override. Make
## sure to return unchanged URL if it's NOT a chapter URL...
def normalize_chapterurl(self,url):
return url
def cachedfetch(realfetch,cache,url): def cachedfetch(realfetch,cache,url):
if url in cache: if url in cache:
return cache[url] return cache[url]
else: else:
return realfetch(url) return realfetch(url)
fullmon = {u"January":u"01", u"February":u"02", u"March":u"03", u"April":u"04", u"May":u"05", fullmon = {u"January":u"01", u"February":u"02", u"March":u"03", u"April":u"04", u"May":u"05",
u"June":u"06","July":u"07", u"August":u"08", u"September":u"09", u"October":u"10", u"June":u"06","July":u"07", u"August":u"08", u"September":u"09", u"October":u"10",
u"November":u"11", u"December":u"12" } u"November":u"11", u"December":u"12" }
@@ -679,7 +712,7 @@ def makeDate(string,dateform):
# lie. It has to do something even more complicated to get # lie. It has to do something even more complicated to get
# Russian month names correct everywhere. # Russian month names correct everywhere.
do_abbrev = "%b" in dateform do_abbrev = "%b" in dateform
if u"%B" in dateform or do_abbrev: if u"%B" in dateform or do_abbrev:
dateform = dateform.replace(u"%B",u"%m").replace(u"%b",u"%m") dateform = dateform.replace(u"%B",u"%m").replace(u"%b",u"%m")
for (name,num) in fullmon.items(): for (name,num) in fullmon.items():
@@ -688,8 +721,23 @@ def makeDate(string,dateform):
if name in string: if name in string:
string = string.replace(name,num) string = string.replace(name,num)
break break
return datetime.datetime.strptime(string.encode('utf-8'),dateform.encode('utf-8')) # Many locales don't define %p for AM/PM. So if %p, remove from
# dateform, look for 'pm' in string, remove am/pm from string and
# add 12 hours if pm found.
add_hours = False
if u"%p" in dateform:
dateform = dateform.replace(u"%p",u"")
if 'pm' in string or 'PM' in string:
add_hours = True
string = string.replace(u"AM",u"").replace(u"PM",u"").replace(u"am",u"").replace(u"pm",u"")
date = datetime.strptime(string.encode('utf-8'),dateform.encode('utf-8'))
if add_hours:
date += timedelta(hours=12)
return date
# .? for AO3's ']' in param names. # .? for AO3's ']' in param names.
safe_url_re = re.compile(r'(?P<attr>(password|name|login).?=)[^&]*(?P<amp>&|$)',flags=re.MULTILINE) safe_url_re = re.compile(r'(?P<attr>(password|name|login).?=)[^&]*(?P<amp>&|$)',flags=re.MULTILINE)
+1 -1
View File
@@ -85,7 +85,7 @@ class BaseEfictionAdapter(BaseSiteAdapter):
@classmethod @classmethod
def getSiteURLPattern(self): def getSiteURLPattern(self):
return r"http://(www\.)?%s%s/%s\?sid=(?P<storyId>\d+)" % (self.getSiteDomain(), self.getPathToArchive(), self.getViewStoryPhpName()) return r"https?://(www\.)?%s%s/%s\?sid=(?P<storyId>\d+)" % (self.getSiteDomain(), self.getPathToArchive(), self.getViewStoryPhpName())
@classmethod @classmethod
def getEncoding(cls): def getEncoding(cls):
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2015 FanFicFare team # Copyright 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -85,6 +85,62 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
def getSiteURLPattern(self): def getSiteURLPattern(self):
return r"https?://"+re.escape(self.getSiteDomain())+r"/(?P<tp>threads|posts)/(.+\.)?(?P<id>\d+)/?[^#]*?(#post-(?P<anchorpost>\d+))?$" return r"https?://"+re.escape(self.getSiteDomain())+r"/(?P<tp>threads|posts)/(.+\.)?(?P<id>\d+)/?[^#]*?(#post-(?P<anchorpost>\d+))?$"
## For adapters, especially base_xenforoforum to override. Make
## sure to return unchanged URL if it's NOT a chapter URL. This
## is most helpful for xenforoforum because threadmarks use
## thread-name URLs--which can change if the thread name changes.
def normalize_chapterurl(self,url):
(is_chapter_url,normalized_url) = self._is_normalize_chapterurl(url)
if is_chapter_url:
return normalized_url
else:
return url
## returns (is_chapter_url,normalized_url)
def _is_normalize_chapterurl(self,url):
is_chapter_url = False
## moved from extract metadata to share with normalize_chapterurl.
if not url.startswith('http'):
url = self.getURLPrefix()+'/'+url
if ( url.startswith(self.getURLPrefix()) or
url.startswith('http://'+self.getSiteDomain()) or
url.startswith('https://'+self.getSiteDomain()) ) and \
( '/posts/' in url or '/threads/' in url or 'showpost.php' in url or 'goto/post' in url):
# brute force way to deal with SB's http->https change when hardcoded http urls.
url = url.replace('http://'+self.getSiteDomain(),self.getURLPrefix())
# http://forums.spacebattles.com/showpost.php?p=4755532&postcount=9
url = re.sub(r'showpost\.php\?p=([0-9]+)(&postcount=[0-9]+)?',r'/posts/\1/',url)
# http://forums.spacebattles.com/goto/post?id=15222406#post-15222406
url = re.sub(r'/goto/post\?id=([0-9]+)(#post-[0-9]+)?',r'/posts/\1/',url)
url = re.sub(r'(^[\'"]+|[\'"]+$)','',url) # strip leading or trailing '" from incorrect quoting.
url = re.sub(r'like$','',url) # strip 'like' if incorrect 'like' link instead of proper post URL.
#### moved from getChapterText()
## there's some history of stories with links to the wrong
## page. This changes page#post URLs to perma-link URLs.
## Which will be redirected back to page#posts, but the
## *correct* ones.
# http://forums.sufficientvelocity.com/threads/harry-potter-and-the-not-fatal-at-all-cultural-exchange-program.330/page-4#post-39915
# https://forums.sufficientvelocity.com/posts/39915/
if '#post-' in url:
url = self.getURLPrefix()+'/posts/'+url.split('#post-')[1]+'/'
## Same as above except for for case where author mistakenly
## used the reply link instead of normal link to post.
# "http://forums.spacebattles.com/threads/manager-worm-story-thread-iv.301602/reply?quote=15962513"
# https://forums.spacebattles.com/posts/
if 'reply?quote=' in url:
url = self.getURLPrefix()+'/posts/'+url.split('reply?quote=')[1]+'/'
is_chapter_url = True
return (is_chapter_url,url)
def use_pagecache(self): def use_pagecache(self):
''' '''
adapters that will work with the page cache need to implement adapters that will work with the page cache need to implement
@@ -119,7 +175,7 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
# params[soup.find('input', {'id':'password'})['name']] = params['password'] # params[soup.find('input', {'id':'password'})['name']] = params['password']
d = self._fetchUrl(loginUrl, params) d = self._fetchUrl(loginUrl, params)
if "Log Out" not in d : if "Log Out" not in d :
logger.info("Failed to login to URL %s as %s" % (loginUrl, logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['login'])) params['login']))
@@ -183,7 +239,7 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
threadmark_chaps = True threadmark_chaps = True
if self.getConfig('always_include_first_post'): if self.getConfig('always_include_first_post'):
self.chapterUrls.append((first_post_title,useurl)) self.chapterUrls.append((first_post_title,useurl))
for (atag,url,name) in [ (x,x['href'],stripHTML(x)) for x in markas ]: for (atag,url,name) in [ (x,x['href'],stripHTML(x)) for x in markas ]:
date = self.make_date(atag.find_next_sibling('div',{'class':'extra'})) date = self.make_date(atag.find_next_sibling('div',{'class':'extra'}))
if not self.story.getMetadataRaw('datePublished') or date < self.story.getMetadataRaw('datePublished'): if not self.story.getMetadataRaw('datePublished') or date < self.story.getMetadataRaw('datePublished'):
@@ -196,13 +252,13 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
soup = soup.find('li',{'class':'message'}) # limit first post for date stuff below. ('#' posts above) soup = soup.find('li',{'class':'message'}) # limit first post for date stuff below. ('#' posts above)
if threadmark_chaps or self.getConfig('always_use_forumtags'): if threadmark_chaps or self.getConfig('always_use_forumtags'):
## only use tags if threadmarks for chapters or ## only use tags if threadmarks for chapters or always_use_forumtags is on.
for tag in topsoup.findAll('a',{'class':'tag'}): for tag in topsoup.findAll('a',{'class':'tag'}) + topsoup.findAll('span',{'class':'prefix'}):
tstr = stripHTML(tag) tstr = stripHTML(tag)
if self.getConfig('capitalize_forumtags'): if self.getConfig('capitalize_forumtags'):
tstr = tstr.title() tstr = tstr.title()
self.story.addToList('forumtags',tstr) self.story.addToList('forumtags',tstr)
# Now go hunting for the 'chapter list'. # Now go hunting for the 'chapter list'.
bq = soup.find('blockquote') # assume first posting contains TOC urls. bq = soup.find('blockquote') # assume first posting contains TOC urls.
@@ -222,21 +278,9 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
if not self.chapterUrls: if not self.chapterUrls:
self.chapterUrls.append((first_post_title,useurl)) self.chapterUrls.append((first_post_title,useurl))
for (url,name) in [ (x['href'],stripHTML(x)) for x in bq.find_all('a') ]: for (url,name) in [ (x['href'],stripHTML(x)) for x in bq.find_all('a') ]:
#logger.debug("found chapurl:%s"%url)
if not url.startswith('http'):
url = self.getURLPrefix()+'/'+url
if ( url.startswith(self.getURLPrefix()) or (is_chapter_url,url) = self._is_normalize_chapterurl(url)
url.startswith('http://'+self.getSiteDomain()) or if is_chapter_url:
url.startswith('https://'+self.getSiteDomain()) ) and ('/posts/' in url or '/threads/' in url):
# brute force way to deal with SB's http->https change when hardcoded http urls.
url = url.replace('http://'+self.getSiteDomain(),self.getURLPrefix())
url = re.sub(r'(^[\'"]+|[\'"]+$)','',url) # strip leading or trailing '" from incorrect quoting.
url = re.sub(r'like$','',url) # strip 'like' if incorrect 'like' link instead of proper post URL.
logger.debug("(ch:%s)used chapurl:%s"%(len(self.chapterUrls)+1,url))
self.chapterUrls.append((name,url)) self.chapterUrls.append((name,url))
if url == useurl and first_post_title == self.chapterUrls[0][0] \ if url == useurl and first_post_title == self.chapterUrls[0][0] \
and not self.getConfig('always_include_first_post',False): and not self.getConfig('always_include_first_post',False):
@@ -272,29 +316,13 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
datestr = re.sub(r' (\d[^\d])',r' 0\1',datestr) # add leading 0 for single digit day & hours. datestr = re.sub(r' (\d[^\d])',r' 0\1',datestr) # add leading 0 for single digit day & hours.
return makeDate(datestr, self.dateformat) return makeDate(datestr, self.dateformat)
except: except:
logger.debug('No date found in %s'%parenttag) logger.debug('No date found in %s'%parenttag,exc_info=True)
return None return None
# grab the text for an individual chapter. # grab the text for an individual chapter.
def getChapterText(self, url): def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url) logger.debug('Getting chapter text from: %s' % url)
## there's some history of stories with links to the wrong
## page. This changes page#post URLs to perma-link URLs.
## Which will be redirected back to page#posts, but the
## *correct* ones.
# http://forums.sufficientvelocity.com/threads/harry-potter-and-the-not-fatal-at-all-cultural-exchange-program.330/page-4#post-39915
# https://forums.sufficientvelocity.com/posts/39915/
if '#post-' in url:
url = self.getURLPrefix()+'/posts/'+url.split('#post-')[1]+'/'
## Same as above except for for case where author mistakenly
## used the reply link instead of normal link to post.
# "http://forums.spacebattles.com/threads/manager-worm-story-thread-iv.301602/reply?quote=15962513"
# https://forums.spacebattles.com/posts/
if 'reply?quote=' in url:
url = self.getURLPrefix()+'/posts/'+url.split('reply?quote=')[1]+'/'
try: try:
origurl = url origurl = url
(data,opened) = self._fetchUrlOpened(url) (data,opened) = self._fetchUrlOpened(url)
@@ -302,20 +330,20 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
if '#' in origurl and '#' not in url: if '#' in origurl and '#' not in url:
url = url + origurl[origurl.index('#'):] url = url + origurl[origurl.index('#'):]
logger.debug("chapter URL redirected to: %s"%url) logger.debug("chapter URL redirected to: %s"%url)
soup = self.make_soup(data) soup = self.make_soup(data)
if '#' in url: if '#' in url:
anchorid = url.split('#')[1] anchorid = url.split('#')[1]
soup = soup.find('li',id=anchorid) soup = soup.find('li',id=anchorid)
bq = soup.find('blockquote') bq = soup.find('blockquote')
bq.name='div' bq.name='div'
for iframe in bq.find_all('iframe'): for iframe in bq.find_all('iframe'):
iframe.extract() # calibre book reader & editor don't like iframes to youtube. iframe.extract() # calibre book reader & editor don't like iframes to youtube.
for qdiv in bq.find_all('div',{'class':'quoteExpand'}): for qdiv in bq.find_all('div',{'class':'quoteExpand'}):
qdiv.extract() # Remove <div class="quoteExpand">click to expand</div> qdiv.extract() # Remove <div class="quoteExpand">click to expand</div>
@@ -323,7 +351,7 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
## include lazy load images. ## include lazy load images.
for img in bq.find_all('img',{'class':'lazyload'}): for img in bq.find_all('img',{'class':'lazyload'}):
img['src'] = img['data-src'] img['src'] = img['data-src']
except Exception as e: except Exception as e:
if self.getConfig('continue_on_chapter_error'): if self.getConfig('continue_on_chapter_error'):
bq = self.make_soup("""<div> bq = self.make_soup("""<div>
+200 -112
View File
@@ -26,6 +26,8 @@ import pprint
import string import string
import sys import sys
version="2.5.1"
if sys.version_info < (2, 5): if sys.version_info < (2, 5):
print 'This program requires Python 2.5 or newer.' print 'This program requires Python 2.5 or newer.'
sys.exit(1) sys.exit(1)
@@ -43,13 +45,13 @@ try:
from calibre_plugins.fanficfare_plugin.fanficfare.configurable import Configuration from calibre_plugins.fanficfare_plugin.fanficfare.configurable import Configuration
from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import ( from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import (
get_dcsource_chaptercount, get_update_data, reset_orig_chapters_epub) get_dcsource_chaptercount, get_update_data, reset_orig_chapters_epub)
from calibre_plugins.fanficfare_plugin.fanficfare.geturls import get_urls_from_page from calibre_plugins.fanficfare_plugin.fanficfare.geturls import get_urls_from_page, get_urls_from_imap
except ImportError: except ImportError:
from fanficfare import adapters, writers, exceptions from fanficfare import adapters, writers, exceptions
from fanficfare.configurable import Configuration from fanficfare.configurable import Configuration
from fanficfare.epubutils import ( from fanficfare.epubutils import (
get_dcsource_chaptercount, get_update_data, reset_orig_chapters_epub) get_dcsource_chaptercount, get_update_data, reset_orig_chapters_epub)
from fanficfare.geturls import get_urls_from_page from fanficfare.geturls import get_urls_from_page, get_urls_from_imap
def write_story(config, adapter, writeformat, metaonly=False, outstream=None): def write_story(config, adapter, writeformat, metaonly=False, outstream=None):
@@ -59,13 +61,15 @@ def write_story(config, adapter, writeformat, metaonly=False, outstream=None):
del writer del writer
return output_filename return output_filename
def main(argv=None,
def main(argv=None, parser=None, passed_defaultsini=None, passed_personalini=None): parser=None,
passed_defaultsini=None,
passed_personalini=None):
if argv is None: if argv is None:
argv = sys.argv[1:] argv = sys.argv[1:]
# read in args, anything starting with -- will be treated as --<varible>=<value> # read in args, anything starting with -- will be treated as --<varible>=<value>
if not parser: if not parser:
parser = OptionParser('usage: %prog [options] storyurl') parser = OptionParser('usage: %prog [options] [STORYURL]...')
parser.add_option('-f', '--format', dest='format', default='epub', parser.add_option('-f', '--format', dest='format', default='epub',
help='write story as FORMAT, epub(default), mobi, text or html', metavar='FORMAT') help='write story as FORMAT, epub(default), mobi, text or html', metavar='FORMAT')
@@ -88,41 +92,68 @@ def main(argv=None, parser=None, passed_defaultsini=None, passed_personalini=Non
help='Retrieve metadata and stop. Or, if --update-epub, update metadata title page only.', ) help='Retrieve metadata and stop. Or, if --update-epub, update metadata title page only.', )
parser.add_option('-u', '--update-epub', parser.add_option('-u', '--update-epub',
action='store_true', dest='update', action='store_true', dest='update',
help='Update an existing epub with new chapters, give epub filename instead of storyurl.', ) help='Update an existing epub(if present) with new chapters. Give either epub filename or story URL.', )
parser.add_option('--unnew',
action='store_true', dest='unnew',
help='Remove (new) chapter marks left by mark_new_chapters setting.', )
parser.add_option('--update-cover', parser.add_option('--update-cover',
action='store_true', dest='updatecover', action='store_true', dest='updatecover',
help='Update cover in an existing epub, otherwise existing cover (if any) is used on update. Only valid with --update-epub.', ) help='Update cover in an existing epub, otherwise existing cover (if any) is used on update. Only valid with --update-epub.', )
parser.add_option('--unnew',
action='store_true', dest='unnew',
help='Remove (new) chapter marks left by mark_new_chapters setting.', )
parser.add_option('--force', parser.add_option('--force',
action='store_true', dest='force', action='store_true', dest='force',
help='Force overwrite of an existing epub, download and overwrite all chapters.', ) help='Force overwrite of an existing epub, download and overwrite all chapters.', )
parser.add_option('-i', '--infile', parser.add_option('-i', '--infile',
help='Give a filename to read for URLs (and/or existing EPUB files with -u for updates).', help='Give a filename to read for URLs (and/or existing EPUB files with --update-epub).',
dest='infile', default=None, dest='infile', default=None,
metavar='INFILE') metavar='INFILE')
parser.add_option('-l', '--list', parser.add_option('-l', '--list',
action='store_true', dest='list', dest='list', default=None, metavar='URL',
help='Get list of valid story URLs from page given.', ) help='Get list of valid story URLs from page given.', )
parser.add_option('-n', '--normalize-list', parser.add_option('-n', '--normalize-list',
action='store_true', dest='normalize', default=False, dest='normalize', default=None, metavar='URL',
help='Get list of valid story URLs from page given, but normalized to standard forms.', ) help='Get list of valid story URLs from page given, but normalized to standard forms.', )
parser.add_option('--download-list',
dest='downloadlist', default=None, metavar='URL',
help='Download story URLs retrieved from page given. Update existing EPUBs if used with --update-epub.', )
parser.add_option('--imap',
action='store_true', dest='imaplist',
help='Get list of valid story URLs from unread email from IMAP account configured in ini.', )
parser.add_option('--download-imap',
action='store_true', dest='downloadimap',
help='Download valid story URLs from unread email from IMAP account configured in ini. Update existing EPUBs if used with --update-epub.', )
parser.add_option('-s', '--sites-list', parser.add_option('-s', '--sites-list',
action='store_true', dest='siteslist', default=False, action='store_true', dest='siteslist', default=False,
help='Get list of valid story URLs examples.', ) help='Get list of valid story URLs examples.', )
parser.add_option('-d', '--debug', parser.add_option('-d', '--debug',
action='store_true', dest='debug', action='store_true', dest='debug',
help='Show debug output while downloading.', ) help='Show debug and notice output.', )
parser.add_option('-v', '--version',
action='store_true', dest='version',
help='Display version and quit.', )
options, args = parser.parse_args(argv) options, args = parser.parse_args(argv)
if options.version:
print("Version: %s" % version)
return
if not options.debug: if not options.debug:
logger = logging.getLogger('fanficfare') logger = logging.getLogger('fanficfare')
logger.setLevel(logging.INFO) logger.setLevel(logging.WARNING)
if not (options.siteslist or options.infile) and len(args) != 1: list_only = any((options.imaplist,
parser.error('incorrect number of arguments') options.siteslist,
options.list,
options.normalize,
))
if list_only and (args or any((options.downloadimap,
options.downloadlist))):
parser.error('Incorrect arguments: Cannot download and list URLs at the same time.')
if options.siteslist: if options.siteslist:
for site, examples in adapters.getSiteExamples(): for site, examples in adapters.getSiteExamples():
@@ -137,43 +168,85 @@ def main(argv=None, parser=None, passed_defaultsini=None, passed_personalini=Non
if options.unnew and options.format != 'epub': if options.unnew and options.format != 'epub':
parser.error('--unnew only works with epub') parser.error('--unnew only works with epub')
urls=args
if not list_only and not (args or any((options.infile,
options.downloadimap,
options.downloadlist))):
parser.print_help();
return
if options.list:
configuration = get_configuration(options.list,
passed_defaultsini,
passed_personalini,options)
retlist = get_urls_from_page(options.list, configuration)
print '\n'.join(retlist)
if options.normalize:
configuration = get_configuration(options.normalize,
passed_defaultsini,
passed_personalini,options)
retlist = get_urls_from_page(options.normalize, configuration,normalize=True)
print '\n'.join(retlist)
if options.downloadlist:
configuration = get_configuration(options.downloadlist,
passed_defaultsini,
passed_personalini,options)
retlist = get_urls_from_page(options.downloadlist, configuration)
urls.extend(retlist)
if options.imaplist or options.downloadimap:
# list doesn't have a supported site.
configuration = get_configuration('test1.com',passed_defaultsini,passed_personalini,options)
markread = configuration.getConfig('imap_mark_read') == 'true' or \
(configuration.getConfig('imap_mark_read') == 'downloadonly' and options.downloadimap)
retlist = get_urls_from_imap(configuration.getConfig('imap_server'),
configuration.getConfig('imap_username'),
configuration.getConfig('imap_password'),
configuration.getConfig('imap_folder'),
markread)
if options.downloadimap:
urls.extend(retlist)
else:
print '\n'.join(retlist)
# for passing in a file list # for passing in a file list
if options.infile: if options.infile:
urls=[]
with open(options.infile,"r") as infile: with open(options.infile,"r") as infile:
#print "File exists and is readable" #print "File exists and is readable"
#fileurls = [line.strip() for line in infile]
for url in infile: for url in infile:
url = url[:url.find('#')].strip() if '#' in url:
url = url[:url.find('#')].strip()
url = url.strip()
if len(url) > 0: if len(url) > 0:
#print "URL: (%s)"%url #print "URL: (%s)"%url
urls.append(url) urls.append(url)
else:
urls = args
if len(urls) > 1: if not list_only:
for url in urls: if len(urls) < 1:
try: print "No valid story URLs found"
do_download(url, else:
options, for url in urls:
passed_defaultsini, try:
passed_personalini) do_download(url,
options,
passed_defaultsini,
passed_personalini)
#print("pagecache:%s"%options.pagecache.keys()) #print("pagecache:%s"%options.pagecache.keys())
except Exception, e: except Exception, e:
print "URL(%s) Failed: Exception (%s). Run URL individually for more detail."%(url,e) if len(urls) == 1:
else: raise
do_download(urls[0], print "URL(%s) Failed: Exception (%s). Run URL individually for more detail."%(url,e)
options,
passed_defaultsini,
passed_personalini)
# make rest a function and loop on it. # make rest a function and loop on it.
def do_download(arg, def do_download(arg,
options, options,
passed_defaultsini, passed_defaultsini,
passed_personalini): passed_personalini):
# Attempt to update an existing epub. # Attempt to update an existing epub.
chaptercount = None chaptercount = None
output_filename = None output_filename = None
@@ -182,7 +255,7 @@ def do_download(arg,
# remove mark_new_chapters marks # remove mark_new_chapters marks
reset_orig_chapters_epub(arg,arg) reset_orig_chapters_epub(arg,arg)
return return
if options.update: if options.update:
try: try:
url, chaptercount = get_dcsource_chaptercount(arg) url, chaptercount = get_dcsource_chaptercount(arg)
@@ -197,71 +270,13 @@ def do_download(arg,
url = arg url = arg
else: else:
url = arg url = arg
try:
configuration = Configuration(adapters.getConfigSectionsFor(url), options.format)
except exceptions.UnknownSite, e:
if options.list or options.normalize:
# list for page doesn't have to be a supported site.
configuration = Configuration('test1.com', options.format)
else:
raise e
conflist = [] configuration = get_configuration(url,
homepath = join(expanduser('~'), '.fanficdownloader') passed_defaultsini,
## also look for .fanficfare now, give higher priority than old dir. passed_personalini,
homepath2 = join(expanduser('~'), '.fanficfare') options,
chaptercount,
if passed_defaultsini: output_filename)
configuration.readfp(passed_defaultsini)
# don't need to check existance for our selves.
conflist.append(join(dirname(__file__), 'defaults.ini'))
conflist.append(join(homepath, 'defaults.ini'))
conflist.append(join(homepath2, 'defaults.ini'))
conflist.append('defaults.ini')
if passed_personalini:
configuration.readfp(passed_personalini)
conflist.append(join(homepath, 'personal.ini'))
conflist.append(join(homepath2, 'personal.ini'))
conflist.append('personal.ini')
if options.configfile:
conflist.extend(options.configfile)
logging.debug('reading %s config file(s), if present' % conflist)
configuration.read(conflist)
try:
configuration.add_section('overrides')
except ConfigParser.DuplicateSectionError:
pass
if options.force:
configuration.set('overrides', 'always_overwrite', 'true')
if options.update and chaptercount:
configuration.set('overrides', 'output_filename', output_filename)
if options.update and not options.updatecover:
configuration.set('overrides', 'never_make_cover', 'true')
# images only for epub, even if the user mistakenly turned it
# on else where.
if options.format not in ('epub', 'html'):
configuration.set('overrides', 'include_images', 'false')
if options.options:
for opt in options.options:
(var, val) = opt.split('=')
configuration.set('overrides', var, val)
if options.list or options.normalize:
retlist = get_urls_from_page(arg, configuration, normalize=options.normalize)
print '\n'.join(retlist)
return
try: try:
adapter = adapters.getAdapter(configuration, url) adapter = adapters.getAdapter(configuration, url)
@@ -269,10 +284,10 @@ def do_download(arg,
if not hasattr(options,'pagecache'): if not hasattr(options,'pagecache'):
options.pagecache = adapter.get_empty_pagecache() options.pagecache = adapter.get_empty_pagecache()
options.cookiejar = adapter.get_empty_cookiejar() options.cookiejar = adapter.get_empty_cookiejar()
adapter.set_pagecache(options.pagecache) adapter.set_pagecache(options.pagecache)
adapter.set_cookiejar(options.cookiejar) adapter.set_cookiejar(options.cookiejar)
adapter.setChaptersRange(options.begin, options.end) adapter.setChaptersRange(options.begin, options.end)
# check for updating from URL (vs from file) # check for updating from URL (vs from file)
@@ -290,17 +305,20 @@ def do_download(arg,
if adapter.getConfig('include_images') and not adapter.getConfig('no_image_processing'): if adapter.getConfig('include_images') and not adapter.getConfig('no_image_processing'):
try: try:
from calibre.utils.magick import Image from calibre.utils.magick import Image
logging.debug('Using calibre.utils.magick') logging.debug('Using calibre.utils.magick')
except ImportError: except ImportError:
try: try:
import Image ## Pillow is a more current fork of PIL library
from PIL import Image
logging.debug('Using PIL') logging.debug('Using Pillow')
except ImportError: except ImportError:
print "You have include_images enabled, but Python Image Library(PIL) isn't found.\nImages will be included full size in original format.\nContinue? (y/n)?" try:
if not sys.stdin.readline().strip().lower().startswith('y'): import Image
return logging.debug('Using PIL')
except ImportError:
print "You have include_images enabled, but Python Image Library(PIL) isn't found.\nImages will be included full size in original format.\nContinue? (y/n)?"
if not sys.stdin.readline().strip().lower().startswith('y'):
return
# three tries, that's enough if both user/pass & is_adult needed, # three tries, that's enough if both user/pass & is_adult needed,
# or a couple tries of one or the other # or a couple tries of one or the other
@@ -360,7 +378,10 @@ def do_download(arg,
output_filename = write_story(configuration, adapter, options.format, options.metaonly) output_filename = write_story(configuration, adapter, options.format, options.metaonly)
if not options.metaonly and adapter.getConfig('post_process_cmd'): if not options.metaonly and adapter.getConfig('post_process_cmd'):
metadata = adapter.story.metadata if adapter.getConfig('post_process_apply_filename_safepattern'):
metadata = adapter.story.get_filename_safe_metadata()
else:
metadata = adapter.story.getAllMetadata()
metadata['output_filename'] = output_filename metadata['output_filename'] = output_filename
call(string.Template(adapter.getConfig('post_process_cmd')).substitute(metadata), shell=True) call(string.Template(adapter.getConfig('post_process_cmd')).substitute(metadata), shell=True)
@@ -375,6 +396,73 @@ def do_download(arg,
except exceptions.AccessDenied as ad: except exceptions.AccessDenied as ad:
print ad print ad
def get_configuration(url,
passed_defaultsini,
passed_personalini,
options,
chaptercount=None,
output_filename=None):
try:
configuration = Configuration(adapters.getConfigSectionsFor(url), options.format)
except exceptions.UnknownSite, e:
if options.list or options.normalize or options.downloadlist:
# list for page doesn't have to be a supported site.
configuration = Configuration('test1.com', options.format)
else:
raise e
conflist = []
homepath = join(expanduser('~'), '.fanficdownloader')
## also look for .fanficfare now, give higher priority than old dir.
homepath2 = join(expanduser('~'), '.fanficfare')
if passed_defaultsini:
configuration.readfp(passed_defaultsini)
# don't need to check existance for our selves.
conflist.append(join(dirname(__file__), 'defaults.ini'))
conflist.append(join(homepath, 'defaults.ini'))
conflist.append(join(homepath2, 'defaults.ini'))
conflist.append('defaults.ini')
if passed_personalini:
configuration.readfp(passed_personalini)
conflist.append(join(homepath, 'personal.ini'))
conflist.append(join(homepath2, 'personal.ini'))
conflist.append('personal.ini')
if options.configfile:
conflist.extend(options.configfile)
logging.debug('reading %s config file(s), if present' % conflist)
configuration.read(conflist)
try:
configuration.add_section('overrides')
except ConfigParser.DuplicateSectionError:
pass
if options.force:
configuration.set('overrides', 'always_overwrite', 'true')
if options.update and chaptercount and output_filename:
configuration.set('overrides', 'output_filename', output_filename)
if options.update and not options.updatecover:
configuration.set('overrides', 'never_make_cover', 'true')
# images only for epub, even if the user mistakenly turned it
# on else where.
if options.format not in ('epub', 'html'):
configuration.set('overrides', 'include_images', 'false')
if options.options:
for opt in options.options:
(var, val) = opt.split('=')
configuration.set('overrides', var, val)
return configuration
if __name__ == '__main__': if __name__ == '__main__':
main() main()
+59 -39
View File
@@ -40,7 +40,7 @@ import adapters
def re_compile(regex,line): def re_compile(regex,line):
try: try:
return re.compile(regex) return re.compile(regex)
except Exception, e: except Exception, e:
raise exceptions.RegularExpresssionFailed(e,regex,line) raise exceptions.RegularExpresssionFailed(e,regex,line)
# fall back labels. # fall back labels.
@@ -59,6 +59,7 @@ titleLabels = {
'warnings':'Warnings', 'warnings':'Warnings',
'numChapters':'Chapters', 'numChapters':'Chapters',
'numWords':'Words', 'numWords':'Words',
'words_added':'Words Added', # logpage only
'site':'Site', 'site':'Site',
'storyId':'Story ID', 'storyId':'Story ID',
'authorId':'Author ID', 'authorId':'Author ID',
@@ -78,7 +79,7 @@ formatsections = ['html','txt','epub','mobi']
othersections = ['defaults','overrides'] othersections = ['defaults','overrides']
def get_valid_sections(): def get_valid_sections():
sites = adapters.getConfigSections() sites = adapters.getConfigSections()
sitesections = list(othersections) sitesections = list(othersections)
for section in sites: for section in sites:
sitesections.append(section) sitesections.append(section)
@@ -90,7 +91,7 @@ def get_valid_sections():
else: else:
# add w/ www if doesn't www # add w/ www if doesn't www
sitesections.append('www.%s'%section) sitesections.append('www.%s'%section)
allowedsections = [] allowedsections = []
allowedsections.extend(formatsections) allowedsections.extend(formatsections)
@@ -99,7 +100,7 @@ def get_valid_sections():
for f in formatsections: for f in formatsections:
allowedsections.append('%s:%s'%(section,f)) allowedsections.append('%s:%s'%(section,f))
return allowedsections return allowedsections
def get_valid_list_entries(): def get_valid_list_entries():
return list(['category', return list(['category',
'genre', 'genre',
@@ -127,6 +128,12 @@ def get_valid_set_options():
This is to further restrict keywords to certain sections and/or This is to further restrict keywords to certain sections and/or
values. get_valid_keywords() below is the list of allowed values. get_valid_keywords() below is the list of allowed
keywords. Any keyword listed here must also be listed there. keywords. Any keyword listed here must also be listed there.
This is what's used by the code when you save personal.ini in
plugin that stops and points out possible errors in keyword
*values*. It doesn't flag 'bad' keywords. Note that it's
separate from color highlighting and most keywords need to be
added to both.
''' '''
valdict = {'collect_series':(None,None,boollist), valdict = {'collect_series':(None,None,boollist),
@@ -144,15 +151,15 @@ def get_valid_set_options():
'strip_chapter_numbers':(None,None,boollist), 'strip_chapter_numbers':(None,None,boollist),
'mark_new_chapters':(None,None,boollist), 'mark_new_chapters':(None,None,boollist),
'titlepage_use_table':(None,None,boollist), 'titlepage_use_table':(None,None,boollist),
'use_ssl_unverified_context':(None,None,boollist), 'use_ssl_unverified_context':(None,None,boollist),
'add_chapter_numbers':(None,None,boollist+['toconly']), 'add_chapter_numbers':(None,None,boollist+['toconly']),
'check_next_chapter':(['fanfiction.net'],None,boollist), 'check_next_chapter':(['fanfiction.net'],None,boollist),
'tweak_fg_sleep':(['fanfiction.net'],None,boollist), 'tweak_fg_sleep':(['fanfiction.net'],None,boollist),
'skip_author_cover':(['fanfiction.net'],None,boollist), 'skip_author_cover':(['fanfiction.net'],None,boollist),
'fix_fimf_blockquotes':(['fimfiction.net'],None,boollist), 'fix_fimf_blockquotes':(['fimfiction.net'],None,boollist),
'fail_on_password':(['fimfiction.net'],None,boollist), 'fail_on_password':(['fimfiction.net'],None,boollist),
'do_update_hook':(['fimfiction.net', 'do_update_hook':(['fimfiction.net',
@@ -174,20 +181,24 @@ def get_valid_set_options():
# kept forgetting to add them, so now it's automatic. # kept forgetting to add them, so now it's automatic.
'bulk_load':(adapters.get_bulk_load_sites(), 'bulk_load':(adapters.get_bulk_load_sites(),
None,boollist), None,boollist),
'include_logpage':(None,['epub'],boollist+['smart']), 'include_logpage':(None,['epub'],boollist+['smart']),
'logpage_at_end':(None,['epub'],boollist),
'windows_eol':(None,['txt'],boollist), 'windows_eol':(None,['txt'],boollist),
'include_images':(None,['epub','html'],boollist), 'include_images':(None,['epub','html'],boollist),
'grayscale_images':(None,['epub','html'],boollist), 'grayscale_images':(None,['epub','html'],boollist),
'no_image_processing':(None,['epub','html'],boollist), 'no_image_processing':(None,['epub','html'],boollist),
'normalize_text_links':(None,['epub','html'],boollist),
'internalize_text_links':(None,['epub','html'],boollist),
'capitalize_forumtags':(base_xenforo_list,None,boollist), 'capitalize_forumtags':(base_xenforo_list,None,boollist),
'continue_on_chapter_error':(base_xenforo_list,None,boollist), 'continue_on_chapter_error':(base_xenforo_list,None,boollist),
'minimum_threadmarks':(base_xenforo_list,None,None), 'minimum_threadmarks':(base_xenforo_list,None,None),
'first_post_title':(base_xenforo_list,None,None), 'first_post_title':(base_xenforo_list,None,None),
'always_include_first_post':(base_xenforo_list,None,boollist), 'always_include_first_post':(base_xenforo_list,None,boollist),
'always_reload_first_chapter':(base_xenforo_list,None,boollist),
} }
return dict(valdict) return dict(valdict)
@@ -203,6 +214,7 @@ def get_valid_scalar_entries():
'rating', 'rating',
'numChapters', 'numChapters',
'numWords', 'numWords',
'words_added', # logpage only.
'site', 'site',
'storyId', 'storyId',
'title', 'title',
@@ -225,6 +237,11 @@ def get_valid_entries():
# *known* keywords -- or rather regexps for them. # *known* keywords -- or rather regexps for them.
def get_valid_keywords(): def get_valid_keywords():
'''
Among other things, this list is used by the color highlighting in
personal.ini editing in plugin. Note that it's separate from
value checking and most keywords need to be added to both.
'''
return list(['(in|ex)clude_metadata_(pre|post)', return list(['(in|ex)clude_metadata_(pre|post)',
'add_chapter_numbers', 'add_chapter_numbers',
'add_genre_when_multi_category', 'add_genre_when_multi_category',
@@ -282,6 +299,7 @@ def get_valid_keywords():
'image_max_size', 'image_max_size',
'include_images', 'include_images',
'include_logpage', 'include_logpage',
'logpage_at_end',
'include_subject_tags', 'include_subject_tags',
'include_titlepage', 'include_titlepage',
'include_tocpage', 'include_tocpage',
@@ -356,7 +374,9 @@ def get_valid_keywords():
'minimum_threadmarks', 'minimum_threadmarks',
'first_post_title', 'first_post_title',
'always_include_first_post', 'always_include_first_post',
'', 'always_reload_first_chapter',
'normalize_text_links',
'internalize_text_links',
]) ])
# *known* entry keywords -- or rather regexps for them. # *known* entry keywords -- or rather regexps for them.
@@ -373,9 +393,9 @@ def make_generate_cover_settings(param):
(template,regexp,setting) = map( lambda x: x.strip(), line.split("=>") ) (template,regexp,setting) = map( lambda x: x.strip(), line.split("=>") )
re_compile(regexp,line) re_compile(regexp,line)
vlist.append((template,regexp,setting)) vlist.append((template,regexp,setting))
except Exception, e: except Exception, e:
raise exceptions.PersonalIniFailed(e,line,param) raise exceptions.PersonalIniFailed(e,line,param)
return vlist return vlist
@@ -386,9 +406,9 @@ class Configuration(ConfigParser.SafeConfigParser):
ConfigParser.SafeConfigParser.__init__(self) ConfigParser.SafeConfigParser.__init__(self)
self.lightweight = lightweight self.lightweight = lightweight
self.linenos=dict() # key by section or section,key -> lineno self.linenos=dict() # key by section or section,key -> lineno
## [injected] section has even less priority than [defaults] ## [injected] section has even less priority than [defaults]
self.sectionslist = ['defaults','injected'] self.sectionslist = ['defaults','injected']
@@ -396,17 +416,17 @@ class Configuration(ConfigParser.SafeConfigParser):
## but before site-specific. ## but before site-specific.
for section in sections[:-1]: for section in sections[:-1]:
self.addConfigSection(section) self.addConfigSection(section)
if site.startswith("www."): if site.startswith("www."):
sitewith = site sitewith = site
sitewithout = site.replace("www.","") sitewithout = site.replace("www.","")
else: else:
sitewith = "www."+site sitewith = "www."+site
sitewithout = site sitewithout = site
self.addConfigSection(sitewith) self.addConfigSection(sitewith)
self.addConfigSection(sitewithout) self.addConfigSection(sitewithout)
if fileform: if fileform:
self.addConfigSection(fileform) self.addConfigSection(fileform)
## add other sections:fileform (not including site DN) ## add other sections:fileform (not including site DN)
@@ -416,9 +436,9 @@ class Configuration(ConfigParser.SafeConfigParser):
self.addConfigSection(sitewith+":"+fileform) self.addConfigSection(sitewith+":"+fileform)
self.addConfigSection(sitewithout+":"+fileform) self.addConfigSection(sitewithout+":"+fileform)
self.addConfigSection("overrides") self.addConfigSection("overrides")
self.listTypeEntries = get_valid_list_entries() self.listTypeEntries = get_valid_list_entries()
self.validEntries = get_valid_entries() self.validEntries = get_valid_entries()
self.url_config_set = False self.url_config_set = False
@@ -443,7 +463,7 @@ class Configuration(ConfigParser.SafeConfigParser):
def isListType(self,key): def isListType(self,key):
return key in self.listTypeEntries or self.hasConfig("include_in_"+key) return key in self.listTypeEntries or self.hasConfig("include_in_"+key)
def isValidMetaEntry(self, key): def isValidMetaEntry(self, key):
return key in self.getValidMetaList() return key in self.getValidMetaList()
@@ -473,7 +493,7 @@ class Configuration(ConfigParser.SafeConfigParser):
# used by adapters & writers, non-convention naming style # used by adapters & writers, non-convention naming style
def getConfig(self, key, default=""): def getConfig(self, key, default=""):
return self.get_config(self.sectionslist,key,default) return self.get_config(self.sectionslist,key,default)
def get_config(self, sections, key, default=""): def get_config(self, sections, key, default=""):
val = default val = default
for section in sections: for section in sections:
@@ -493,7 +513,7 @@ class Configuration(ConfigParser.SafeConfigParser):
#print "getConfig(add_to_%s)=[%s]%s" % (key,section,val) #print "getConfig(add_to_%s)=[%s]%s" % (key,section,val)
except (ConfigParser.NoOptionError, ConfigParser.NoSectionError), e: except (ConfigParser.NoOptionError, ConfigParser.NoSectionError), e:
pass pass
return val return val
# split and strip each. # split and strip each.
@@ -505,7 +525,7 @@ class Configuration(ConfigParser.SafeConfigParser):
return default return default
else: else:
return vlist return vlist
# used by adapters & writers, non-convention naming style # used by adapters & writers, non-convention naming style
def getConfigList(self, key, default=[]): def getConfigList(self, key, default=[]):
return self.get_config_list(self.sectionslist, key, default) return self.get_config_list(self.sectionslist, key, default)
@@ -519,7 +539,7 @@ class Configuration(ConfigParser.SafeConfigParser):
return self.linenos.get(section+','+key,None) return self.linenos.get(section+','+key,None)
else: else:
return self.linenos.get(section,None) return self.linenos.get(section,None)
## Copied from Python 2.7 library so as to make read utf8. ## Copied from Python 2.7 library so as to make read utf8.
def read(self, filenames): def read(self, filenames):
"""Read and parse a filename or a list of filenames. """Read and parse a filename or a list of filenames.
@@ -543,7 +563,7 @@ class Configuration(ConfigParser.SafeConfigParser):
fp.close() fp.close()
read_ok.append(filename) read_ok.append(filename)
return read_ok return read_ok
## Copied from Python 2.7 library so as to make it save linenos too. ## Copied from Python 2.7 library so as to make it save linenos too.
# #
# Regular expressions for parsing section headers and options. # Regular expressions for parsing section headers and options.
@@ -623,7 +643,7 @@ class Configuration(ConfigParser.SafeConfigParser):
optval = '' optval = ''
optname = self.optionxform(optname.rstrip()) optname = self.optionxform(optname.rstrip())
cursect[optname] = optval cursect[optname] = optval
self.linenos[cursect['__name__']+','+optname]=lineno self.linenos[cursect['__name__']+','+optname]=lineno
else: else:
# a non-fatal parsing error occurred. set up the # a non-fatal parsing error occurred. set up the
# exception but keep going. the exception will be # exception but keep going. the exception will be
@@ -651,11 +671,11 @@ class Configuration(ConfigParser.SafeConfigParser):
from story import set_in_ex_clude, make_replacements from story import set_in_ex_clude, make_replacements
custom_columns_settings_re = re.compile(r'(add_to_)?custom_columns_settings') custom_columns_settings_re = re.compile(r'(add_to_)?custom_columns_settings')
generate_cover_settings_re = re.compile(r'(add_to_)?generate_cover_settings') generate_cover_settings_re = re.compile(r'(add_to_)?generate_cover_settings')
valdict = get_valid_set_options() valdict = get_valid_set_options()
for section in self.sections(): for section in self.sections():
allow_all_section = allow_all_sections_re.match(section) allow_all_section = allow_all_sections_re.match(section)
if section not in allowedsections and not allow_all_section: if section not in allowedsections and not allow_all_section:
@@ -671,17 +691,17 @@ class Configuration(ConfigParser.SafeConfigParser):
elif sitename in othersections: elif sitename in othersections:
formatname = None formatname = None
sitename = None sitename = None
## check each keyword in section. Due to precedence ## check each keyword in section. Due to precedence
## order of sections, it's possible for bad lines to ## order of sections, it's possible for bad lines to
## never be used. ## never be used.
for keyword,value in self.items(section): for keyword,value in self.items(section):
try: try:
## check regex bearing keywords first. Each ## check regex bearing keywords first. Each
## will raise exceptions if flawed. ## will raise exceptions if flawed.
if clude_metadata_re.match(keyword): if clude_metadata_re.match(keyword):
set_in_ex_clude(value) set_in_ex_clude(value)
if replace_metadata_re.match(keyword): if replace_metadata_re.match(keyword):
make_replacements(value) make_replacements(value)
@@ -713,7 +733,7 @@ class Configuration(ConfigParser.SafeConfigParser):
## used with CLI/web yet. ## used with CLI/web yet.
except Exception as e: except Exception as e:
errors.append((self.get_lineno(section,keyword),"Error:%s in (%s:%s)"%(e,keyword,value))) errors.append((self.get_lineno(section,keyword),"Error:%s in (%s:%s)"%(e,keyword,value)))
return errors return errors
@@ -728,7 +748,7 @@ class Configurable(object):
def addUrlConfigSection(self,url): def addUrlConfigSection(self,url):
self.configuration.addUrlConfigSection(url) self.configuration.addUrlConfigSection(url)
def isListType(self,key): def isListType(self,key):
return self.configuration.isListType(key) return self.configuration.isListType(key)
@@ -737,10 +757,10 @@ class Configurable(object):
def getValidMetaList(self): def getValidMetaList(self):
return self.configuration.getValidMetaList() return self.configuration.getValidMetaList()
def hasConfig(self, key): def hasConfig(self, key):
return self.configuration.hasConfig(key) return self.configuration.hasConfig(key)
def has_config(self, sections, key): def has_config(self, sections, key):
return self.configuration.has_config(sections, key) return self.configuration.has_config(sections, key)
+196 -122
View File
@@ -25,7 +25,7 @@
## titlepage_entries: category,genre, status,dateUpdated,rating ## titlepage_entries: category,genre, status,dateUpdated,rating
## [epub] ## [epub]
## # overrides defaults & site section ## # overrides defaults & site section
## titlepage_entries: category,genre, status,datePublished,dateUpdated,dateCreated ## titlepage_entries: category,genre,status,datePublished,dateUpdated,dateCreated
## [www.whofic.com:epub] ## [www.whofic.com:epub]
## # overrides defaults, site section & format section ## # overrides defaults, site section & format section
## titlepage_entries: category,genre, status,datePublished ## titlepage_entries: category,genre, status,datePublished
@@ -180,7 +180,7 @@ extratags: FanFiction
## Can also be used for other metadata values ## Can also be used for other metadata values
#default_value_category:FanFiction #default_value_category:FanFiction
## number of seconds to sleep between calls to the story site. May by ## number of seconds to sleep between calls to the story site. May be
## useful if pulling large numbers of stories or if the site is slow. ## useful if pulling large numbers of stories or if the site is slow.
#slow_down_sleep_time:0.5 #slow_down_sleep_time:0.5
@@ -189,11 +189,17 @@ extratags: FanFiction
## prevent excessive wait when your network or the site is down. ## prevent excessive wait when your network or the site is down.
connect_timeout:60.0 connect_timeout:60.0
## For use only with stand-alone CLI version--run a command on the ## For use only with CLI version--run a command on the generated file
## generated file after it's produced. All of the titlepage_entries ## after it's produced. All of the titlepage_entries values are
## values are available, plus output_filename. ## available, plus output_filename.
#post_process_cmd: addbook -f "${output_filename}" -t "${title}" #post_process_cmd: addbook -f "${output_filename}" -t "${title}"
## Some operating systems and command shells have problems with some
## characters. When true, the output_filename_safepattern will be
## applied to each metadata item passed to post_process_cmd before
## it's called.
#post_process_apply_filename_safepattern:false
## Use regular expressions to find and replace (or remove) metadata. ## Use regular expressions to find and replace (or remove) metadata.
## For example, you could change Sci-Fi=>SF, remove *-Centered tags, ## For example, you could change Sci-Fi=>SF, remove *-Centered tags,
## etc. See http://docs.python.org/library/re.html (look for re.sub) ## etc. See http://docs.python.org/library/re.html (look for re.sub)
@@ -404,6 +410,40 @@ user_agent:FFF/2.X
## non-intuitive. ## non-intuitive.
#description_limit:1000 #description_limit:1000
## The FFF CLI can fetch story URLs from unread emails when configured
## to read from your IMAP mail server. The example shows GMail, but
## other services that support IMAP can be used. GMail requires you
## to turn on an option to enable IMAP access. Only the CLI uses these
## options--the Calibre Plugin stores these separately.
##
## It's safest if you create a separate email account that you use
## only for your story update notices. FanFicFare cannot guarantee
## that malicious code cannot get your email password once you've
## saved it. Use this feature at your own risk.
##
#imap_server:imap.gmail.com
#imap_username:youraddress@gmail.com
#imap_password:XXXXXXXX
#imap_folder:INBOX
## Mark mails with story URLs read:
## imap_mark_read can be 'true', 'false'(default) or 'downloadonly'.
##
## If 'true', unread emails will be marked as read when
## either CLI option --imap to list the story URLs or --download-imap
## to download story URLs from email are used.
##
## If 'downloadonly', unread emails will be marked as read
## only when CLI --download-imap to download story URLs from email are
## used.
##
## If 'false', unread emails will not be marked as read.
##
## Only unread emails will be searched for story URLs, and only emails
## containing valid story URLs will ever be marked read.
##
#imap_mark_read:true
[base_efiction] [base_efiction]
## At the time of writing, eFiction Base adapters allow downloading ## At the time of writing, eFiction Base adapters allow downloading
## the whole story in bulk using the 'Print' feature. If 'bulk_load' ## the whole story in bulk using the 'Print' feature. If 'bulk_load'
@@ -473,8 +513,8 @@ add_to_include_subject_tags:,tagsfromtitle.SPLIT,forumtags
## base_xenforoforum reads Published and Updated datetimes from ## base_xenforoforum reads Published and Updated datetimes from
## Threadmarks if used, or from the posted & updated times of the ## Threadmarks if used, or from the posted & updated times of the
## 'first' post if no threadmarks. ## 'first' post if no threadmarks.
datePublished_format:%%Y-%%m-%%d %%H:%%M:%%S datePublished_format:%%Y-%%m-%%d %%H:%%M
dateUpdated_format:%%Y-%%m-%%d %%H:%%M:%%S dateUpdated_format:%%Y-%%m-%%d %%H:%%M
## Only take the first X characters of the 'first' post to use as ## Only take the first X characters of the 'first' post to use as
## the description. ## the description.
@@ -509,6 +549,20 @@ first_post_title:First Post
## if threadmarks are used. Can result in a duplicated chapter. ## if threadmarks are used. Can result in a duplicated chapter.
always_include_first_post:false always_include_first_post:false
## In normal operation, when updating an existing epub, old chapters
## will be reused as-is. Normally, that works fine, but forum stories
## sometimes have an index post as the first 'chapter', and the
## version in the first chapter gets out of sync.
##
## If always_reload_first_chapter:true, then the first chapter will
## always be downloaded again (ie, reloaded). It will NOT be maked
## '(new)' (see mark_new_chapters). Because it is reloaded, manual
## edits made to the first chapter will be lost.
##
## While intended for base_xenforoforum sites, this setting can be
## applied to other sites.
always_reload_first_chapter:false
## In normal operation, forumtags will only be populated when ## In normal operation, forumtags will only be populated when
## threadmarks are used for chapters (see minimum_threadmarks above). ## threadmarks are used for chapters (see minimum_threadmarks above).
## When always_use_forumtags:true, always populate forumtags. ## When always_use_forumtags:true, always populate forumtags.
@@ -590,6 +644,11 @@ include_logpage: false
## end up with Completed stories that have just one logpage entry. ## end up with Completed stories that have just one logpage entry.
#include_logpage: smart #include_logpage: smart
## By default, logpage is placed before the story chapters. This
## setting, if true, will place the logpage after the chapters
## instead.
logpage_at_end: false
## items to include in the log page Empty metadata entries, or those ## items to include in the log page Empty metadata entries, or those
## that haven't changed since the last update, will *not* appear, even ## that haven't changed since the last update, will *not* appear, even
## if in the list. You can include extra text or HTML that will be ## if in the list. You can include extra text or HTML that will be
@@ -707,6 +766,18 @@ remove_transparency: true
## true--replace_br_with_p also fixes the problem. ## true--replace_br_with_p also fixes the problem.
nook_img_fix:true nook_img_fix:true
## Apply adapter's normalize_chapterurl() to all links in chapter
## texts, if they match chapter URLs. Currently only implemented by
## base_xenforoforum adapters.
#normalize_text_links:false
## Search all links in chapter texts and, if they match any included
## chapter URLs, replace them with links to the chapter in the
## download. Only works with epub and html output formats.
## base_xenforoforum adapters should also use normalize_text_links
## with this.
#internalize_text_links:false
[mobi] [mobi]
## mobi TOC cannot be turned off right now. ## mobi TOC cannot be turned off right now.
#include_tocpage: true #include_tocpage: true
@@ -754,6 +825,19 @@ extratags: FanFiction,Testing,Text
[test1.com:html] [test1.com:html]
extratags: FanFiction,Testing,HTML extratags: FanFiction,Testing,HTML
[adult-fanfiction.org]
extra_valid_entries:eroticatags,disclaimer
eroticatags_label:Erotica Tags
disclaimer_label:Disclaimer
extra_titlepage_entries:eroticatags,disclaimer
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[archive.skyehawke.com] [archive.skyehawke.com]
[archiveofourown.org] [archiveofourown.org]
@@ -853,18 +937,6 @@ extracategories:The Sentinel
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[bdsm-geschichten.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## This site offers no index page so we can either guess the chapter URLs
## by dec/incrementing numbers ('guess') or walk all the chapters in the metadata
## parsing state ('parse'). Since guessing can lead to errors for non-standard
## story URLs, the default is to parse
#find_chapters:guess
[bloodshedverse.com] [bloodshedverse.com]
## website encoding(s) In theory, each website reports the character ## website encoding(s) In theory, each website reports the character
## encoding they use for each page. In practice, some sites report it ## encoding they use for each page. In practice, some sites report it
@@ -875,6 +947,12 @@ extracategories:The Sentinel
## it has +90% confidence. 'auto' is not reliable. ## it has +90% confidence. 'auto' is not reliable.
website_encodings:Windows-1252,ISO-8859-1,auto website_encodings:Windows-1252,ISO-8859-1,auto
## dateUpdate doesn't usually have time, but it does on
## bloodshedverse.com. See
## http://docs.python.org/library/datetime.html#strftime-strptime-behavior
## Note that ini format requires % to be escaped as %%.
dateUpdated_format:%%Y-%%m-%%d %%H:%%M
## Extra metadata that this adapter knows about. See [dramione.org] ## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them. ## for examples of how to use them.
extra_valid_entries:warnings,reviews extra_valid_entries:warnings,reviews
@@ -904,15 +982,6 @@ strip_text_links:true
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Blood Ties extracategories:Blood Ties
[buffynfaith.net]
## Site dedicated to these categories/characters/ships
extracategories:Buffy: The Vampire Slayer
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
[fanfic.castletv.net] [fanfic.castletv.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1139,13 +1208,20 @@ romance_label: Romance
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[ficwad.com] [fictionhunt.com]
## Some sites require login (or login for some rated stories) The ## Archive only site for ffnet HP stories.
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not ## Site dedicated to these categories/characters/ships
## defaults.ini. extracategories:Harry Potter
#username:YourName
#password:yourpassword extra_valid_entries: origin,originUrl,originHTML,reviews
originHTML_label:Original Story URL
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:origin
add_to_extra_titlepage_entries:originHTML
[fictionmania.tv] [fictionmania.tv]
## website encoding(s) In theory, each website reports the character ## website encoding(s) In theory, each website reports the character
@@ -1203,6 +1279,14 @@ views_label:Views
likes_label:Likes likes_label:Likes
dislikes_label:Dislikes dislikes_label:Dislikes
[ficwad.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[finestories.com] [finestories.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1244,13 +1328,9 @@ universe_as_series: true
## cover image. This lets you exclude them. ## cover image. This lets you exclude them.
cover_exclusion_regexp:/css/bir.png cover_exclusion_regexp:/css/bir.png
[forums.spacebattles.com] [forum.questionablequesting.com]
## see [base_xenforoforum] ## see [base_xenforoforum]
[forums.sufficientvelocity.com]
## see [base_xenforoforum]
[grangerenchanted.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not ## commandline version, this should go in your personal.ini, not
@@ -1258,18 +1338,24 @@ cover_exclusion_regexp:/css/bir.png
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
## Some sites also require the user to confirm they are adult for [forums.spacebattles.com]
## adult content. In commandline version, this should go in your ## see [base_xenforoforum]
## personal.ini, not defaults.ini.
[forums.sufficientvelocity.com]
## see [base_xenforoforum]
[harem.lucifael.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
## Site dedicated to these categories/characters/ships ## Some sites require login (or login for some rated stories) The
extracategories:Harry Potter ## program can prompt you, or you can save it in config. In
extracharacters:Hermione Granger ## commandline version, this should go in your personal.ini, not
## defaults.ini.
## Extra metadata that this adapter knows about. See [dramione.org] #username:YourName
## for examples of how to use them. #password:yourpassword
extra_valid_entries:read,reviews
[hlfiction.net] [hlfiction.net]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
@@ -1383,6 +1469,19 @@ extra_titlepage_entries: eroticatags
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Merlin extracategories:Merlin
[mujaji.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
[national-library.net] [national-library.net]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:West Wing extracategories:West Wing
@@ -1484,15 +1583,21 @@ extracategories:My Little Pony: Friendship is Magic
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:The Pretender extracategories:The Pretender
[forum.questionablequesting.com] [quotev.com]
## see [base_xenforoforum] extra_valid_entries:pages,readers,reads,favorites,searchtags,comments
pages_label:Pages
readers_label:Readers
reads_label:Reads
favorites_label:Favorites
searchtags_label:Search Tags
comments_label:Comments
## Some sites require login (or login for some rated stories) The include_in_category:category,searchtags
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not [royalroadl.com]
## defaults.ini. extra_valid_entries:stars
#username:YourName
#password:yourpassword #add_to_extra_titlepage_entries:,stars
[samandjack.net] [samandjack.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
@@ -1518,22 +1623,6 @@ extracategories:Supernatural
extracharacters:Sam,Dean extracharacters:Sam,Dean
extraships:Sam/Dean extraships:Sam/Dean
[scarhead.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
[sheppardweir.com] [sheppardweir.com]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -1675,15 +1764,6 @@ readings_label: Readings
## personal.ini, not defaults.ini. ## personal.ini, not defaults.ini.
#is_adult:true #is_adult:true
[tokra.fandomnet.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1
[tolkienfanfiction.com] [tolkienfanfiction.com]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Lord of the Rings extracategories:Lord of the Rings
@@ -1751,6 +1831,12 @@ extraships:Draco Malfoy/Ginny Weasley
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1 extracategories:Stargate: SG-1
[www.deepinmysoul.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[www.destinysgateway.com] [www.destinysgateway.com]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
@@ -1836,6 +1922,10 @@ check_next_chapter:false
#password:yourpassword #password:yourpassword
[www.ficbook.net] [www.ficbook.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[www.fictionalley.org] [www.fictionalley.org]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
@@ -2041,11 +2131,6 @@ cover_exclusion_regexp:/stories/999/images/.*?_trophy.png
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:NCIS extracategories:NCIS
[www.nickngreg.nl]
## Site dedicated to these categories/characters/ships
extracategories:CSI
extraships:Nick Stokes/Greg Sanders
[www.phoenixsong.net] [www.phoenixsong.net]
## Some sites require login (or login for some rated stories) The ## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In ## program can prompt you, or you can save it in config. In
@@ -2100,6 +2185,8 @@ extracategories:Psych
extracategories:Queer as Folk extracategories:Queer as Folk
[quotev.com] [quotev.com]
user_agent:
slow_down_sleep_time:2
extra_valid_entries:pages,readers,reads,favorites,searchtags,comments extra_valid_entries:pages,readers,reads,favorites,searchtags,comments
pages_label:Pages pages_label:Pages
readers_label:Readers readers_label:Readers
@@ -2176,23 +2263,9 @@ extracategories:Lord of the Rings
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
[www.twcslibrary.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## twcslibrary.net (ab)uses series as personal reading lists.
collect_series: false
[www.tthfanfic.org] [www.tthfanfic.org]
user_agent:
slow_down_sleep_time:2
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini. ## this should go in your personal.ini, not defaults.ini.
@@ -2227,6 +2300,22 @@ pairingcat_to_characters_ships:true
## instead be added to characters Buffy, Spike and ships Buffy/Spike ## instead be added to characters Buffy, Spike and ships Buffy/Spike
romancecat_to_characters_ships:true romancecat_to_characters_ships:true
[www.twcslibrary.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## twcslibrary.net (ab)uses series as personal reading lists.
collect_series: false
[www.twilightarchives.com] [www.twilightarchives.com]
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Twilight extracategories:Twilight
@@ -2275,8 +2364,9 @@ extracharacters:Wolverine,Rogue
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis extracategories:Stargate: Atlantis
extra_valid_entries:reviews ##site stopped showing reviews ~ Oct 2016
reviews_label:Reviews #extra_valid_entries:reviews
#reviews_label:Reviews
[buffygiles.velocitygrass.com] [buffygiles.velocitygrass.com]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
@@ -2339,20 +2429,6 @@ extracategories:Andromeda
## Site dedicated to these categories/characters/ships ## Site dedicated to these categories/characters/ships
extracategories:Artemis Fowl extracategories:Artemis Fowl
[www.therabidreader.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
[www.naiceanilme.net] [www.naiceanilme.net]
## Some sites do not require a login, but do require the user to ## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version, ## confirm they are adult for adult content. In commandline version,
@@ -2366,8 +2442,6 @@ extracategories:Artemis Fowl
#username:YourName #username:YourName
#password:yourpassword #password:yourpassword
[overrides] [overrides]
## It may sometimes be useful to override all of the specific format, ## It may sometimes be useful to override all of the specific format,
## site and site:format sections in your private configuration. For ## site and site:format sections in your private configuration. For
+32 -22
View File
@@ -21,6 +21,10 @@ def get_dcsource(inputio):
def get_dcsource_chaptercount(inputio): def get_dcsource_chaptercount(inputio):
return get_update_data(inputio,getfilecount=True,getsoups=False)[:2] # (source,filecount) return get_update_data(inputio,getfilecount=True,getsoups=False)[:2] # (source,filecount)
def get_cover_data(inputio):
# (oldcoverhtmlhref,oldcoverhtmltype,oldcoverhtmldata,oldcoverimghref,oldcoverimgtype,oldcoverimgdata)
return get_update_data(inputio,getfilecount=True,getsoups=False)[4]
def get_update_data(inputio, def get_update_data(inputio,
getfilecount=True, getfilecount=True,
getsoups=True): getsoups=True):
@@ -50,26 +54,32 @@ def get_update_data(inputio,
if item.getAttribute("type") == "cover": if item.getAttribute("type") == "cover":
# there is a cover (x)html file, save the soup for it. # there is a cover (x)html file, save the soup for it.
href=relpath+item.getAttribute("href") href=relpath+item.getAttribute("href")
oldcoverhtmlhref = href
oldcoverhtmldata = epub.read(href)
oldcoverhtmltype = "application/xhtml+xml"
for item in contentdom.getElementsByTagName("item"):
if( relpath+item.getAttribute("href") == oldcoverhtmlhref ):
oldcoverhtmltype = item.getAttribute("media-type")
break
soup = bs.BeautifulSoup(oldcoverhtmldata.decode("utf-8"),"html5lib")
src = None src = None
# first img or image tag. try:
imgs = soup.findAll('img') oldcoverhtmlhref = href
if imgs: oldcoverhtmldata = epub.read(href)
src = get_path_part(href)+imgs[0]['src'] oldcoverhtmltype = "application/xhtml+xml"
else: for item in contentdom.getElementsByTagName("item"):
imgs = soup.findAll('image') if( relpath+item.getAttribute("href") == oldcoverhtmlhref ):
oldcoverhtmltype = item.getAttribute("media-type")
break
soup = bs.BeautifulSoup(oldcoverhtmldata.decode("utf-8"),"html5lib")
# first img or image tag.
imgs = soup.findAll('img')
if imgs: if imgs:
src=get_path_part(href)+imgs[0]['xlink:href'] src = get_path_part(href)+imgs[0]['src']
else:
imgs = soup.findAll('image')
if imgs:
src=get_path_part(href)+imgs[0]['xlink:href']
if not src:
continue
except Exception as e:
## Calibre's Polish Book corrupts sub-book covers.
logger.warn("Cover (x)html file %s not found"%href)
logger.warn("Exception: %s"%(unicode(e)))
if not src:
continue
try: try:
# remove all .. and the path part above it, if present. # remove all .. and the path part above it, if present.
# Mostly for epubs edited by Sigil. # Mostly for epubs edited by Sigil.
@@ -85,7 +95,6 @@ def get_update_data(inputio,
except Exception as e: except Exception as e:
logger.warn("Cover Image %s not found"%src) logger.warn("Cover Image %s not found"%src)
logger.warn("Exception: %s"%(unicode(e))) logger.warn("Exception: %s"%(unicode(e)))
traceback.print_exc()
filecount = 0 filecount = 0
soups = [] # list of xhmtl blocks soups = [] # list of xhmtl blocks
@@ -101,12 +110,14 @@ def get_update_data(inputio,
if( item.getAttribute("media-type") == "application/xhtml+xml" ): if( item.getAttribute("media-type") == "application/xhtml+xml" ):
href=relpath+item.getAttribute("href") href=relpath+item.getAttribute("href")
#print("---- item href:%s path part: %s"%(href,get_path_part(href))) #print("---- item href:%s path part: %s"%(href,get_path_part(href)))
if re.match(r'.*/log_page\.x?html',href): if re.match(r'.*/log_page(_u\d+)?\.x?html',href):
try: try:
logfile = epub.read(href).decode("utf-8") logfile = epub.read(href).decode("utf-8")
except: except:
pass # corner case I bumped into while testing. pass # corner case I bumped into while testing.
if re.match(r'.*/(file|chapter)\d+\.x?html',href): if re.match(r'.*/(file|chapter)\d+(_u\d+)?\.x?html',href):
# (_u\d+)? is from calibre convert naming files
# 3/OEBPS/file0005_u3.xhtml etc.
if getsoups: if getsoups:
soup = bs.BeautifulSoup(epub.read(href).decode("utf-8"),"html5lib") soup = bs.BeautifulSoup(epub.read(href).decode("utf-8"),"html5lib")
for img in soup.findAll('img'): for img in soup.findAll('img'):
@@ -127,8 +138,7 @@ def get_update_data(inputio,
# originally. # originally.
if newsrc != u'OEBPS/failedtoload': if newsrc != u'OEBPS/failedtoload':
logger.warn("Image %s not found!\n(originally:%s)"%(newsrc,longdesc)) logger.warn("Image %s not found!\n(originally:%s)"%(newsrc,longdesc))
logger.warn("Exception: %s"%(unicode(e))) logger.warn("Exception: %s"%(unicode(e)),exc_info=True)
traceback.print_exc()
bodysoup = soup.find('body') bodysoup = soup.find('body')
# ffdl epubs have chapter title h3 # ffdl epubs have chapter title h3
h3 = bodysoup.find('h3') h3 = bodysoup.find('h3')
+5 -2
View File
@@ -125,7 +125,10 @@ def get_urls_from_html(data,url=None,configuration=None,normalize=False,restrict
def get_urls_from_text(data,configuration=None,normalize=False): def get_urls_from_text(data,configuration=None,normalize=False):
urls = collections.OrderedDict() urls = collections.OrderedDict()
data=unicode(data) try:
data = unicode(data)
except UnicodeDecodeError:
data=data.decode('utf8') ## for when called outside calibre.
if not configuration: if not configuration:
configuration = Configuration(["test1.com"],"EPUB",lightweight=True) configuration = Configuration(["test1.com"],"EPUB",lightweight=True)
@@ -229,7 +232,7 @@ def get_urls_from_imap(srv,user,passwd,folder,markread=True):
if part.get_content_type() == 'text/html': if part.get_content_type() == 'text/html':
urllist.extend(get_urls_from_html(part.get_payload(decode=True))) urllist.extend(get_urls_from_html(part.get_payload(decode=True)))
except Exception as e: except Exception as e:
logger.error("Failed to read email content: %s"%e) logger.error("Failed to read email content: %s"%e,exc_info=True)
#logger.debug "urls:%s"%get_urls_from_text(get_first_text_block(email_message)) #logger.debug "urls:%s"%get_urls_from_text(get_first_text_block(email_message))
if urllist and markread: if urllist and markread:
-453
View File
@@ -1,453 +0,0 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""html2text: Turn HTML into equivalent Markdown-structured text."""
__version__ = "2.37"
__author__ = "Aaron Swartz (me@aaronsw.com)"
__copyright__ = "(C) 2004-2008 Aaron Swartz. GNU GPL 3."
__contributors__ = ["Martin 'Joey' Schulze", "Ricardo Reyes", "Kevin Jay North"]
# TODO:
# Support decoded entities with unifiable.
if not hasattr(__builtins__, 'True'): True, False = 1, 0
import re, sys, urllib, htmlentitydefs, codecs, StringIO, types
import sgmllib
import urlparse
sgmllib.charref = re.compile('&#([xX]?[0-9a-fA-F]+)[^0-9a-fA-F]')
try: from textwrap import wrap
except: pass
# Use Unicode characters instead of their ascii psuedo-replacements
UNICODE_SNOB = 0
# Put the links after each paragraph instead of at the end.
LINKS_EACH_PARAGRAPH = 0
# Wrap long lines at position. 0 for no wrapping. (Requires Python 2.3.)
BODY_WIDTH = 78
# Don't show internal links (href="#local-anchor") -- corresponding link targets
# won't be visible in the plain text file anyway.
SKIP_INTERNAL_LINKS = False
### Entity Nonsense ###
def name2cp(k):
if k == 'apos': return ord("'")
if hasattr(htmlentitydefs, "name2codepoint"): # requires Python 2.3
return htmlentitydefs.name2codepoint[k]
else:
k = htmlentitydefs.entitydefs[k]
if k.startswith("&#") and k.endswith(";"): return int(k[2:-1]) # not in latin-1
return ord(codecs.latin_1_decode(k)[0])
unifiable = {'rsquo':"'", 'lsquo':"'", 'rdquo':'"', 'ldquo':'"',
'copy':'(C)', 'mdash':'--', 'nbsp':' ', 'rarr':'->', 'larr':'<-', 'middot':'*',
'ndash':'-', 'oelig':'oe', 'aelig':'ae',
'agrave':'a', 'aacute':'a', 'acirc':'a', 'atilde':'a', 'auml':'a', 'aring':'a',
'egrave':'e', 'eacute':'e', 'ecirc':'e', 'euml':'e',
'igrave':'i', 'iacute':'i', 'icirc':'i', 'iuml':'i',
'ograve':'o', 'oacute':'o', 'ocirc':'o', 'otilde':'o', 'ouml':'o',
'ugrave':'u', 'uacute':'u', 'ucirc':'u', 'uuml':'u'}
unifiable_n = {}
for k in unifiable.keys():
unifiable_n[name2cp(k)] = unifiable[k]
def charref(name):
if name[0] in ['x','X']:
c = int(name[1:], 16)
else:
c = int(name)
if not UNICODE_SNOB and c in unifiable_n.keys():
return unifiable_n[c]
else:
return unichr(c)
def entityref(c):
if not UNICODE_SNOB and c in unifiable.keys():
return unifiable[c]
else:
try: name2cp(c)
except KeyError: return "&" + c
else: return unichr(name2cp(c))
def replaceEntities(s):
s = s.group(1)
if s[0] == "#":
return charref(s[1:])
else: return entityref(s)
r_unescape = re.compile(r"&(#?[xX]?(?:[0-9a-fA-F]+|\w{1,8}));")
def unescape(s):
return r_unescape.sub(replaceEntities, s)
def fixattrs(attrs):
# Fix bug in sgmllib.py
if not attrs: return attrs
newattrs = []
for attr in attrs:
newattrs.append((attr[0], unescape(attr[1])))
return newattrs
### End Entity Nonsense ###
def onlywhite(line):
"""Return true if the line does only consist of whitespace characters."""
for c in line:
if c is not ' ' and c is not ' ':
return c is ' '
return line
def optwrap(text,wrap_width=BODY_WIDTH):
"""Wrap all paragraphs in the provided text."""
if not wrap_width:
return text
assert wrap, "Requires Python 2.3."
result = ''
newlines = 0
for para in text.split("\n"):
if len(para) > 0:
if para[0] is not ' ' and para[0] is not '-' and para[0] is not '*':
for line in wrap(para, wrap_width):
result += line + "\n"
result += "\n"
newlines = 2
else:
if not onlywhite(para):
result += para + "\n"
newlines = 1
else:
if newlines < 2:
result += "\n"
newlines += 1
return result
def hn(tag):
if tag[0] == 'h' and len(tag) == 2:
try:
n = int(tag[1])
if n in range(1, 10): return n
except ValueError: return 0
class _html2text(sgmllib.SGMLParser):
def __init__(self, out=None, baseurl=''):
sgmllib.SGMLParser.__init__(self)
if out is None: self.out = self.outtextf
else: self.out = out
self.outtext = u''
self.quiet = 0
self.p_p = 0
self.outcount = 0
self.start = 1
self.space = 0
self.a = []
self.astack = []
self.acount = 0
self.list = []
self.blockquote = 0
self.pre = 0
self.startpre = 0
self.lastWasNL = 0
self.abbr_title = None # current abbreviation definition
self.abbr_data = None # last inner HTML (for abbr being defined)
self.abbr_list = {} # stack of abbreviations to write later
self.baseurl = baseurl
def outtextf(self, s):
self.outtext += s
def close(self):
sgmllib.SGMLParser.close(self)
self.pbr()
self.o('', 0, 'end')
return self.outtext
def handle_charref(self, c):
self.o(charref(c))
def handle_entityref(self, c):
self.o(entityref(c))
def unknown_starttag(self, tag, attrs):
self.handle_tag(tag, attrs, 1)
def unknown_endtag(self, tag):
self.handle_tag(tag, None, 0)
def previousIndex(self, attrs):
""" returns the index of certain set of attributes (of a link) in the
self.a list
If the set of attributes is not found, returns None
"""
if not attrs.has_attr('href'): return None
i = -1
for a in self.a:
i += 1
match = 0
if a.has_attr('href') and a['href'] == attrs['href']:
if a.has_attr('title') or attrs.has_attr('title'):
if (a.has_attr('title') and attrs.has_attr('title') and
a['title'] == attrs['title']):
match = True
else:
match = True
if match: return i
def handle_tag(self, tag, attrs, start):
attrs = fixattrs(attrs)
if hn(tag):
self.p()
if start: self.o(hn(tag)*"#" + ' ')
if tag in ['p', 'div']: self.p()
if tag == "br" and start: self.o(" \n")
if tag == "hr" and start:
self.p()
self.o("* * *")
self.p()
if tag in ["head", "style", 'script']:
if start: self.quiet += 1
else: self.quiet -= 1
if tag in ["body"]:
self.quiet = 0 # sites like 9rules.com never close <head>
if tag == "blockquote":
if start:
self.p(); self.o('> ', 0, 1); self.start = 1
self.blockquote += 1
else:
self.blockquote -= 1
self.p()
if tag in ['em', 'i', 'u']: self.o("_")
if tag in ['strong', 'b']: self.o("**")
if tag == "code" and not self.pre: self.o('`') #TODO: `` `this` ``
if tag == "abbr":
if start:
attrsD = {}
for (x, y) in attrs: attrsD[x] = y
attrs = attrsD
self.abbr_title = None
self.abbr_data = ''
if attrs.has_attr('title'):
self.abbr_title = attrs['title']
else:
if self.abbr_title != None:
self.abbr_list[self.abbr_data] = self.abbr_title
self.abbr_title = None
self.abbr_data = ''
if tag == "a":
if start:
attrsD = {}
for (x, y) in attrs: attrsD[x] = y
attrs = attrsD
if attrs.has_attr('href') and not (SKIP_INTERNAL_LINKS and attrs['href'].startswith('#')):
self.astack.append(attrs)
self.o("[")
else:
self.astack.append(None)
else:
if self.astack:
a = self.astack.pop()
if a:
i = self.previousIndex(a)
if i is not None:
a = self.a[i]
else:
self.acount += 1
a['count'] = self.acount
a['outcount'] = self.outcount
self.a.append(a)
self.o("][" + `a['count']` + "]")
if tag == "img" and start:
attrsD = {}
for (x, y) in attrs: attrsD[x] = y
attrs = attrsD
if attrs.has_attr('src'):
attrs['href'] = attrs['src']
alt = attrs.get('alt', '')
i = self.previousIndex(attrs)
if i is not None:
attrs = self.a[i]
else:
self.acount += 1
attrs['count'] = self.acount
attrs['outcount'] = self.outcount
self.a.append(attrs)
self.o("![")
self.o(alt)
self.o("]["+`attrs['count']`+"]")
if tag == 'dl' and start: self.p()
if tag == 'dt' and not start: self.pbr()
if tag == 'dd' and start: self.o(' ')
if tag == 'dd' and not start: self.pbr()
if tag in ["ol", "ul"]:
if start:
self.list.append({'name':tag, 'num':0})
else:
if self.list: self.list.pop()
self.p()
if tag == 'li':
if start:
self.pbr()
if self.list: li = self.list[-1]
else: li = {'name':'ul', 'num':0}
self.o(" "*len(self.list)) #TODO: line up <ol><li>s > 9 correctly.
if li['name'] == "ul": self.o("* ")
elif li['name'] == "ol":
li['num'] += 1
self.o(`li['num']`+". ")
self.start = 1
else:
self.pbr()
if tag in ["table", "tr"] and start: self.p()
if tag == 'td': self.pbr()
if tag == "pre":
if start:
self.startpre = 1
self.pre = 1
else:
self.pre = 0
self.p()
def pbr(self):
if self.p_p == 0: self.p_p = 1
def p(self): self.p_p = 2
def o(self, data, puredata=0, force=0):
if self.abbr_data is not None: self.abbr_data += data
if not self.quiet:
if puredata and not self.pre:
data = re.sub('\s+', ' ', data)
if data and data[0] == ' ':
self.space = 1
data = data[1:]
if not data and not force: return
if self.startpre:
#self.out(" :") #TODO: not output when already one there
self.startpre = 0
bq = (">" * self.blockquote)
if not (force and data and data[0] == ">") and self.blockquote: bq += " "
if self.pre:
bq += " "
data = data.replace("\n", "\n"+bq)
if self.start:
self.space = 0
self.p_p = 0
self.start = 0
if force == 'end':
# It's the end.
self.p_p = 0
self.out("\n")
self.space = 0
if self.p_p:
self.out(('\n'+bq)*self.p_p)
self.space = 0
if self.space:
if not self.lastWasNL: self.out(' ')
self.space = 0
if self.a and ((self.p_p == 2 and LINKS_EACH_PARAGRAPH) or force == "end"):
if force == "end": self.out("\n")
newa = []
for link in self.a:
if self.outcount > link['outcount']:
self.out(" ["+`link['count']`+"]: " + urlparse.urljoin(self.baseurl, link['href']))
if link.has_attr('title'): self.out(" ("+link['title']+")")
self.out("\n")
else:
newa.append(link)
if self.a != newa: self.out("\n") # Don't need an extra line when nothing was done.
self.a = newa
if self.abbr_list and force == "end":
for abbr, definition in self.abbr_list.items():
self.out(" *[" + abbr + "]: " + definition + "\n")
self.p_p = 0
self.out(data)
self.lastWasNL = data and data[-1] == '\n'
self.outcount += 1
def handle_data(self, data):
if r'\/script>' in data: self.quiet -= 1
self.o(data, 1)
def unknown_decl(self, data): pass
def wrapwrite(text): sys.stdout.write(text.encode('utf8'))
def html2text_file(html, out=wrapwrite, baseurl=''):
h = _html2text(out, baseurl)
h.feed(html)
h.feed("")
return h.close()
def html2text(html, baseurl='', wrap_width=BODY_WIDTH):
return optwrap(html2text_file(html, None, baseurl),wrap_width)
if __name__ == "__main__":
baseurl = ''
if sys.argv[1:]:
arg = sys.argv[1]
if arg.startswith('http://'):
baseurl = arg
j = urllib.urlopen(baseurl)
try:
from feedparser import _getCharacterEncoding as enc
except ImportError:
enc = lambda x, y: ('utf-8', 1)
text = j.read()
encoding = enc(j.headers, text)[0]
if encoding == 'us-ascii': encoding = 'utf-8'
data = text.decode(encoding)
else:
encoding = 'utf8'
if len(sys.argv) > 2:
encoding = sys.argv[2]
data = open(arg, 'r').read().decode(encoding)
else:
data = sys.stdin.read().decode('utf8')
wrapwrite(html2text(data, baseurl))
+20 -9
View File
@@ -997,20 +997,28 @@ class Story(Configurable):
return retval return retval
def get_filename_safe_metadata(self):
origvalues = self.getAllMetadata()
values={}
pattern = re_compile(self.getConfig("output_filename_safepattern",
r"(^\.|/\.|[^a-zA-Z0-9_\. \[\]\(\)&'-]+)"),
"output_filename_safepattern")
for k in origvalues.keys():
if k == 'formatext': # don't do file extension--we set it anyway.
values[k]=self.getMetadata(k)
else:
values[k]=re.sub(pattern,'_', removeAllEntities(self.getMetadata(k)))
return values
def formatFileName(self,template,allowunsafefilename=True): def formatFileName(self,template,allowunsafefilename=True):
values = origvalues = self.getAllMetadata()
# fall back default: # fall back default:
if not template: if not template:
template="${title}-${siteabbrev}_${storyId}${formatext}" template="${title}-${siteabbrev}_${storyId}${formatext}"
if not allowunsafefilename: if allowunsafefilename:
values={} values = self.getAllMetadata()
pattern = re_compile(self.getConfig("output_filename_safepattern",r"(^\.|/\.|[^a-zA-Z0-9_\. \[\]\(\)&'-]+)"),"output_filename_safepattern") else:
for k in origvalues.keys(): values = self.get_filename_safe_metadata()
if k == 'formatext': # don't do file extension--we set it anyway.
values[k]=self.getMetadata(k)
else:
values[k]=re.sub(pattern,'_', removeAllEntities(self.getMetadata(k)))
return string.Template(template).substitute(values).encode('utf8') return string.Template(template).substitute(values).encode('utf8')
@@ -1069,6 +1077,9 @@ class Story(Configurable):
if imgurl not in self.imgurls: if imgurl not in self.imgurls:
try: try:
if imgurl == 'failedtoload':
raise Exception("Previously failed to load")
parsedUrl = urlparse.urlparse(imgurl) parsedUrl = urlparse.urlparse(imgurl)
if self.getConfig('no_image_processing'): if self.getConfig('no_image_processing'):
(data,ext,mime) = no_convert_image(imgurl, (data,ext,mime) = no_convert_image(imgurl,
+81 -45
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team, 2015 FanFicFare team # Copyright 2011 Fanficdownloader team, 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -27,8 +27,11 @@ import re
## use DOM to generate the XML files. ## use DOM to generate the XML files.
from xml.dom.minidom import parse, parseString, getDOMImplementation from xml.dom.minidom import parse, parseString, getDOMImplementation
import bs4
from base_writer import * from base_writer import *
from ..htmlcleanup import stripHTML,removeEntities from ..htmlcleanup import stripHTML,removeEntities
from ..story import commaGroups
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -46,7 +49,7 @@ class EpubWriter(BaseStoryWriter):
BaseStoryWriter.__init__(self, config, story) BaseStoryWriter.__init__(self, config, story)
self.EPUB_CSS = string.Template('''${output_css}''') self.EPUB_CSS = string.Template('''${output_css}''')
self.EPUB_TITLE_PAGE_START = string.Template('''<?xml version="1.0" encoding="UTF-8"?> self.EPUB_TITLE_PAGE_START = string.Template('''<?xml version="1.0" encoding="UTF-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"> <html xmlns="http://www.w3.org/1999/xhtml">
<head> <head>
@@ -117,7 +120,7 @@ ${value}<br />
self.EPUB_TOC_ENTRY = string.Template(''' self.EPUB_TOC_ENTRY = string.Template('''
<a href="file${index}.xhtml">${chapter}</a><br /> <a href="file${index}.xhtml">${chapter}</a><br />
''') ''')
self.EPUB_TOC_PAGE_END = string.Template(''' self.EPUB_TOC_PAGE_END = string.Template('''
</div> </div>
</body> </body>
@@ -156,15 +159,15 @@ ${value}<br />
self.EPUB_LOG_UPDATE_START = string.Template(''' self.EPUB_LOG_UPDATE_START = string.Template('''
<p class='log_entry'> <p class='log_entry'>
''') ''')
self.EPUB_LOG_ENTRY = string.Template(''' self.EPUB_LOG_ENTRY = string.Template('''
<b>${label}:</b> <span id="${id}">${value}</span> <b>${label}:</b> <span id="${id}">${value}</span>
''') ''')
self.EPUB_LOG_UPDATE_END = string.Template(''' self.EPUB_LOG_UPDATE_END = string.Template('''
</p><hr /> </p><hr />
''') ''')
self.EPUB_LOG_PAGE_END = string.Template(''' self.EPUB_LOG_PAGE_END = string.Template('''
</body> </body>
</html> </html>
@@ -184,7 +187,7 @@ div { margin: 0pt; padding: 0pt; }
<img src="${coverimg}" alt="cover"/> <img src="${coverimg}" alt="cover"/>
</div></body></html> </div></body></html>
''') ''')
def writeLogPage(self, out): def writeLogPage(self, out):
""" """
Write the log page, but only include entries that there's Write the log page, but only include entries that there's
@@ -201,12 +204,12 @@ div { margin: 0pt; padding: 0pt; }
END = string.Template(self.getConfig("logpage_end")) END = string.Template(self.getConfig("logpage_end"))
else: else:
END = self.EPUB_LOG_PAGE_END END = self.EPUB_LOG_PAGE_END
# if there's a self.story.logfile, there's an existing log # if there's a self.story.logfile, there's an existing log
# to add to. # to add to.
if self.story.logfile: if self.story.logfile:
logger.debug("existing logfile found, appending") logger.debug("existing logfile found, appending")
logger.debug("existing data:%s"%self._getLastLogData(self.story.logfile)) # logger.debug("existing data:%s"%self._getLastLogData(self.story.logfile))
replace_string = "</body>" # "</h3>" replace_string = "</body>" # "</h3>"
self._write(out,self.story.logfile.replace(replace_string,self._makeLogEntry(self._getLastLogData(self.story.logfile))+replace_string)) self._write(out,self.story.logfile.replace(replace_string,self._makeLogEntry(self._getLastLogData(self.story.logfile))+replace_string))
else: else:
@@ -234,7 +237,7 @@ div { margin: 0pt; padding: 0pt; }
pass pass
return values return values
def _makeLogEntry(self, oldvalues={}): def _makeLogEntry(self, oldvalues={}):
if self.hasConfig("logpage_update_start"): if self.hasConfig("logpage_update_start"):
START = string.Template(self.getConfig("logpage_update_start")) START = string.Template(self.getConfig("logpage_update_start"))
@@ -253,6 +256,14 @@ div { margin: 0pt; padding: 0pt; }
retval = START.substitute(self.story.getAllMetadata()) retval = START.substitute(self.story.getAllMetadata())
## words_added is only used in logpage because it's the only
## place we know the previous version's word count.
if 'words_added' in (self.getConfigList("logpage_entries") + self.getConfigList("extra_logpage_entries")):
new_words = self.story.getMetadata('numWords')
old_words = oldvalues.get('numWords',None)
if new_words and old_words:
self.story.setMetadata('words_added',commaGroups(unicode(int(new_words.replace(',',''))-int(old_words.replace(',','')))))
for entry in self.getConfigList("logpage_entries") + self.getConfigList("extra_logpage_entries"): for entry in self.getConfigList("logpage_entries") + self.getConfigList("extra_logpage_entries"):
if self.isValidMetaEntry(entry): if self.isValidMetaEntry(entry):
val = self.story.getMetadata(entry) val = self.story.getMetadata(entry)
@@ -275,16 +286,16 @@ div { margin: 0pt; padding: 0pt; }
# mostly it makes it easy to tell when you get the # mostly it makes it easy to tell when you get the
# keyword wrong. # keyword wrong.
retval = retval + entry retval = retval + entry
retval = retval + END.substitute(self.story.getAllMetadata()) retval = retval + END.substitute(self.story.getAllMetadata())
if self.getConfig('replace_hr'): if self.getConfig('replace_hr'):
# replacing a self-closing tag with a container tag in the # replacing a self-closing tag with a container tag in the
# soup is more difficult than it first appears. So cheat. # soup is more difficult than it first appears. So cheat.
retval = re.sub("<hr[^>]*>","<div class='center'>* * *</div>",retval) retval = re.sub("<hr[^>]*>","<div class='center'>* * *</div>",retval)
return retval return retval
def writeStoryImpl(self, out): def writeStoryImpl(self, out):
## Python 2.5 ZipFile is rather more primative than later ## Python 2.5 ZipFile is rather more primative than later
@@ -305,7 +316,7 @@ div { margin: 0pt; padding: 0pt; }
## Re-open file for content. ## Re-open file for content.
outputepub = ZipFile(zipio, 'a', compression=ZIP_DEFLATED) outputepub = ZipFile(zipio, 'a', compression=ZIP_DEFLATED)
outputepub.debug=3 outputepub.debug=3
## Create META-INF/container.xml file. The only thing it does is ## Create META-INF/container.xml file. The only thing it does is
## point to content.opf ## point to content.opf
containerdom = getDOMImplementation().createDocument(None, "container", None) containerdom = getDOMImplementation().createDocument(None, "container", None)
@@ -332,7 +343,7 @@ div { margin: 0pt; padding: 0pt; }
self.getMetadata('site'), self.getMetadata('site'),
self.story.getList('authorId')[0], self.story.getList('authorId')[0],
self.getMetadata('storyId')) self.getMetadata('storyId'))
contentdom = getDOMImplementation().createDocument(None, "package", None) contentdom = getDOMImplementation().createDocument(None, "package", None)
package = contentdom.documentElement package = contentdom.documentElement
package.setAttribute("version","2.0") package.setAttribute("version","2.0")
@@ -374,12 +385,12 @@ div { margin: 0pt; padding: 0pt; }
metadata.appendChild(newTag(contentdom,"dc:date", metadata.appendChild(newTag(contentdom,"dc:date",
attrs={"opf:event":"publication"}, attrs={"opf:event":"publication"},
text=self.story.getMetadataRaw('datePublished').strftime("%Y-%m-%d"))) text=self.story.getMetadataRaw('datePublished').strftime("%Y-%m-%d")))
if self.story.getMetadataRaw('dateCreated'): if self.story.getMetadataRaw('dateCreated'):
metadata.appendChild(newTag(contentdom,"dc:date", metadata.appendChild(newTag(contentdom,"dc:date",
attrs={"opf:event":"creation"}, attrs={"opf:event":"creation"},
text=self.story.getMetadataRaw('dateCreated').strftime("%Y-%m-%d"))) text=self.story.getMetadataRaw('dateCreated').strftime("%Y-%m-%d")))
if self.story.getMetadataRaw('dateUpdated'): if self.story.getMetadataRaw('dateUpdated'):
metadata.appendChild(newTag(contentdom,"dc:date", metadata.appendChild(newTag(contentdom,"dc:date",
attrs={"opf:event":"modification"}, attrs={"opf:event":"modification"},
@@ -387,7 +398,7 @@ div { margin: 0pt; padding: 0pt; }
metadata.appendChild(newTag(contentdom,"meta", metadata.appendChild(newTag(contentdom,"meta",
attrs={"name":"calibre:timestamp", attrs={"name":"calibre:timestamp",
"content":self.story.getMetadataRaw('dateUpdated').strftime("%Y-%m-%dT%H:%M:%S")})) "content":self.story.getMetadataRaw('dateUpdated').strftime("%Y-%m-%dT%H:%M:%S")}))
if self.getMetadata('description'): if self.getMetadata('description'):
metadata.appendChild(newTag(contentdom,"dc:description",text= metadata.appendChild(newTag(contentdom,"dc:description",text=
self.getMetadata('description'))) self.getMetadata('description')))
@@ -395,11 +406,11 @@ div { margin: 0pt; padding: 0pt; }
for subject in self.story.getSubjectTags(): for subject in self.story.getSubjectTags():
metadata.appendChild(newTag(contentdom,"dc:subject",text=subject)) metadata.appendChild(newTag(contentdom,"dc:subject",text=subject))
if self.getMetadata('site'): if self.getMetadata('site'):
metadata.appendChild(newTag(contentdom,"dc:publisher", metadata.appendChild(newTag(contentdom,"dc:publisher",
text=self.getMetadata('site'))) text=self.getMetadata('site')))
if self.getMetadata('storyUrl'): if self.getMetadata('storyUrl'):
metadata.appendChild(newTag(contentdom,"dc:identifier", metadata.appendChild(newTag(contentdom,"dc:identifier",
attrs={"opf:scheme":"URL"}, attrs={"opf:scheme":"URL"},
@@ -415,7 +426,7 @@ div { margin: 0pt; padding: 0pt; }
guide = None guide = None
coverIO = None coverIO = None
coverimgid = "image0000" coverimgid = "image0000"
if not self.story.cover and self.story.oldcover: if not self.story.cover and self.story.oldcover:
logger.debug("writer_epub: no new cover, has old cover, write image.") logger.debug("writer_epub: no new cover, has old cover, write image.")
@@ -427,7 +438,7 @@ div { margin: 0pt; padding: 0pt; }
oldcoverimgdata) = self.story.oldcover oldcoverimgdata) = self.story.oldcover
outputepub.writestr(oldcoverhtmlhref,oldcoverhtmldata) outputepub.writestr(oldcoverhtmlhref,oldcoverhtmldata)
outputepub.writestr(oldcoverimghref,oldcoverimgdata) outputepub.writestr(oldcoverimghref,oldcoverimgdata)
coverimgid = "image0" coverimgid = "image0"
items.append((coverimgid, items.append((coverimgid,
oldcoverimghref, oldcoverimghref,
@@ -441,8 +452,8 @@ div { margin: 0pt; padding: 0pt; }
guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover", guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover",
"title":"Cover", "title":"Cover",
"href":oldcoverhtmlhref})) "href":oldcoverhtmlhref}))
if self.getConfig('include_images'): if self.getConfig('include_images'):
imgcount=0 imgcount=0
@@ -459,7 +470,7 @@ div { margin: 0pt; padding: 0pt; }
# just the first image. # just the first image.
coverimgid = items[-1][0] coverimgid = items[-1][0]
items.append(("style","OEBPS/stylesheet.css","text/css",None)) items.append(("style","OEBPS/stylesheet.css","text/css",None))
if self.story.cover: if self.story.cover:
@@ -467,7 +478,7 @@ div { margin: 0pt; padding: 0pt; }
# for it to work on Nook. # for it to work on Nook.
items.append(("cover","OEBPS/cover.xhtml","application/xhtml+xml",None)) items.append(("cover","OEBPS/cover.xhtml","application/xhtml+xml",None))
itemrefs.append("cover") itemrefs.append("cover")
# #
# <meta name="cover" content="cover.jpg"/> # <meta name="cover" content="cover.jpg"/>
metadata.appendChild(newTag(contentdom,"meta",{"content":coverimgid, metadata.appendChild(newTag(contentdom,"meta",{"content":coverimgid,
"name":"cover"})) "name":"cover"}))
@@ -480,14 +491,14 @@ div { margin: 0pt; padding: 0pt; }
guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover", guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover",
"title":"Cover", "title":"Cover",
"href":"OEBPS/cover.xhtml"})) "href":"OEBPS/cover.xhtml"}))
if self.hasConfig("cover_content"): if self.hasConfig("cover_content"):
COVER = string.Template(self.getConfig("cover_content")) COVER = string.Template(self.getConfig("cover_content"))
else: else:
COVER = self.EPUB_COVER COVER = self.EPUB_COVER
coverIO = StringIO.StringIO() coverIO = StringIO.StringIO()
coverIO.write(COVER.substitute(dict(self.story.getAllMetadata().items()+{'coverimg':self.story.cover}.items()))) coverIO.write(COVER.substitute(dict(self.story.getAllMetadata().items()+{'coverimg':self.story.cover}.items())))
if self.getConfig("include_titlepage"): if self.getConfig("include_titlepage"):
items.append(("title_page","OEBPS/title_page.xhtml","application/xhtml+xml","Title Page")) items.append(("title_page","OEBPS/title_page.xhtml","application/xhtml+xml","Title Page"))
itemrefs.append("title_page") itemrefs.append("title_page")
@@ -495,14 +506,15 @@ div { margin: 0pt; padding: 0pt; }
items.append(("toc_page","OEBPS/toc_page.xhtml","application/xhtml+xml","Table of Contents")) items.append(("toc_page","OEBPS/toc_page.xhtml","application/xhtml+xml","Table of Contents"))
itemrefs.append("toc_page") itemrefs.append("toc_page")
## save where to insert logpage.
logpage_indices = (len(items),len(itemrefs))
dologpage = ( self.getConfig("include_logpage") == "smart" and \ dologpage = ( self.getConfig("include_logpage") == "smart" and \
(self.story.logfile or self.story.getMetadataRaw("status") == "In-Progress") ) \ (self.story.logfile or self.story.getMetadataRaw("status") == "In-Progress") ) \
or self.getConfig("include_logpage") == "true" or self.getConfig("include_logpage") == "true"
if dologpage: ## collect chapter urls and file names for internalize_text_links option.
items.append(("log_page","OEBPS/log_page.xhtml","application/xhtml+xml","Update Log")) chapurlmap = {}
itemrefs.append("log_page")
for index, chap in enumerate(self.story.getChapters(fortoc=True)): for index, chap in enumerate(self.story.getChapters(fortoc=True)):
if chap.html: if chap.html:
i=index+1 i=index+1
@@ -511,6 +523,14 @@ div { margin: 0pt; padding: 0pt; }
"application/xhtml+xml", "application/xhtml+xml",
chap.title)) chap.title))
itemrefs.append("file%04d"%i) itemrefs.append("file%04d"%i)
chapurlmap[chap.url]="file%04d.xhtml"%i # url -> relative epub file name.
if dologpage:
if self.getConfig("logpage_at_end") == "true":
## insert logpage after chapters.
logpage_indices = (len(items),len(itemrefs))
items.insert(logpage_indices[0],("log_page","OEBPS/log_page.xhtml","application/xhtml+xml","Update Log"))
itemrefs.insert(logpage_indices[1],"log_page")
manifest = contentdom.createElement("manifest") manifest = contentdom.createElement("manifest")
package.appendChild(manifest) package.appendChild(manifest)
@@ -520,7 +540,7 @@ div { margin: 0pt; padding: 0pt; }
attrs={'id':id, attrs={'id':id,
'href':href, 'href':href,
'media-type':type})) 'media-type':type}))
spine = newTag(contentdom,"spine",attrs={"toc":"ncx"}) spine = newTag(contentdom,"spine",attrs={"toc":"ncx"})
package.appendChild(spine) package.appendChild(spine)
for itemref in itemrefs: for itemref in itemrefs:
@@ -530,10 +550,10 @@ div { margin: 0pt; padding: 0pt; }
# guide only exists if there's a cover. # guide only exists if there's a cover.
if guide: if guide:
package.appendChild(guide) package.appendChild(guide)
# write content.opf to zip. # write content.opf to zip.
contentxml = contentdom.toxml(encoding='utf-8') contentxml = contentdom.toxml(encoding='utf-8')
# tweak for brain damaged Nook STR. Nook insists on name before content. # tweak for brain damaged Nook STR. Nook insists on name before content.
contentxml = contentxml.replace('<meta content="%s" name="cover"/>'%coverimgid, contentxml = contentxml.replace('<meta content="%s" name="cover"/>'%coverimgid,
'<meta name="cover" content="%s"/>'%coverimgid) '<meta name="cover" content="%s"/>'%coverimgid)
@@ -557,11 +577,11 @@ div { margin: 0pt; padding: 0pt; }
attrs={"name":"dtb:totalPageCount", "content":"0"})) attrs={"name":"dtb:totalPageCount", "content":"0"}))
head.appendChild(newTag(tocncxdom,"meta", head.appendChild(newTag(tocncxdom,"meta",
attrs={"name":"dtb:maxPageNumber", "content":"0"})) attrs={"name":"dtb:maxPageNumber", "content":"0"}))
docTitle = tocncxdom.createElement("docTitle") docTitle = tocncxdom.createElement("docTitle")
docTitle.appendChild(newTag(tocncxdom,"text",text=self.getMetadata('title'))) docTitle.appendChild(newTag(tocncxdom,"text",text=self.getMetadata('title')))
ncx.appendChild(docTitle) ncx.appendChild(docTitle)
tocnavMap = tocncxdom.createElement("navMap") tocnavMap = tocncxdom.createElement("navMap")
ncx.appendChild(tocnavMap) ncx.appendChild(tocnavMap)
@@ -586,14 +606,14 @@ div { margin: 0pt; padding: 0pt; }
navLabel.appendChild(newTag(tocncxdom,"text",text=stripHTML(title))) navLabel.appendChild(newTag(tocncxdom,"text",text=stripHTML(title)))
navPoint.appendChild(newTag(tocncxdom,"content",attrs={"src":href})) navPoint.appendChild(newTag(tocncxdom,"content",attrs={"src":href}))
index=index+1 index=index+1
# write toc.ncx to zip file # write toc.ncx to zip file
outputepub.writestr("toc.ncx",tocncxdom.toxml(encoding='utf-8')) outputepub.writestr("toc.ncx",tocncxdom.toxml(encoding='utf-8'))
tocncxdom.unlink() tocncxdom.unlink()
del tocncxdom del tocncxdom
# write stylesheet.css file. # write stylesheet.css file.
outputepub.writestr("OEBPS/stylesheet.css",self.EPUB_CSS.substitute(self.story.getAllMetadata())) outputepub.writestr("OEBPS/stylesheet.css",self.EPUB_CSS.substitute(self.story.getAllMetadata()))
# write title page. # write title page.
if self.getConfig("titlepage_use_table"): if self.getConfig("titlepage_use_table"):
@@ -612,7 +632,7 @@ div { margin: 0pt; padding: 0pt; }
if coverIO: if coverIO:
outputepub.writestr("OEBPS/cover.xhtml",coverIO.getvalue()) outputepub.writestr("OEBPS/cover.xhtml",coverIO.getvalue())
coverIO.close() coverIO.close()
titlepageIO = StringIO.StringIO() titlepageIO = StringIO.StringIO()
self.writeTitlePage(out=titlepageIO, self.writeTitlePage(out=titlepageIO,
START=TITLE_PAGE_START, START=TITLE_PAGE_START,
@@ -624,7 +644,7 @@ div { margin: 0pt; padding: 0pt; }
outputepub.writestr("OEBPS/title_page.xhtml",titlepageIO.getvalue()) outputepub.writestr("OEBPS/title_page.xhtml",titlepageIO.getvalue())
titlepageIO.close() titlepageIO.close()
# write toc page. # write toc page.
tocpageIO = StringIO.StringIO() tocpageIO = StringIO.StringIO()
self.writeTOCPage(tocpageIO, self.writeTOCPage(tocpageIO,
self.EPUB_TOC_PAGE_START, self.EPUB_TOC_PAGE_START,
@@ -645,14 +665,28 @@ div { margin: 0pt; padding: 0pt; }
CHAPTER_START = string.Template(self.getConfig("chapter_start")) CHAPTER_START = string.Template(self.getConfig("chapter_start"))
else: else:
CHAPTER_START = self.EPUB_CHAPTER_START CHAPTER_START = self.EPUB_CHAPTER_START
if self.hasConfig('chapter_end'): if self.hasConfig('chapter_end'):
CHAPTER_END = string.Template(self.getConfig("chapter_end")) CHAPTER_END = string.Template(self.getConfig("chapter_end"))
else: else:
CHAPTER_END = self.EPUB_CHAPTER_END CHAPTER_END = self.EPUB_CHAPTER_END
for index, chap in enumerate(self.story.getChapters()): # (url,title,html) for index, chap in enumerate(self.story.getChapters()): # (url,title,html)
if chap.html: if chap.html:
chap_data = chap.html
if self.getConfig('internalize_text_links'):
soup = bs4.BeautifulSoup(chap.html,'html5lib')
changed=False
for alink in soup.find_all('a'):
if alink.has_attr('href') and alink['href'] in chapurlmap:
alink['href']=chapurlmap[alink['href']]
changed=True
if changed:
chap_data = unicode(soup)
# Don't want html, head or body tags in
# chapter html--bs4 insists on adding them.
chap_data = re.sub(r"</?(html|head|body)[^>]*>\r?\n?","",chap_data)
#logger.debug('Writing chapter text for: %s' % chap.title) #logger.debug('Writing chapter text for: %s' % chap.title)
vals={'url':removeEntities(chap.url), vals={'url':removeEntities(chap.url),
'chapter':removeEntities(chap.title), 'chapter':removeEntities(chap.title),
@@ -664,7 +698,9 @@ div { margin: 0pt; padding: 0pt; }
for k,v in vals.items(): for k,v in vals.items():
if isinstance(v,basestring): vals[k]=v.replace('"','&quot;') if isinstance(v,basestring): vals[k]=v.replace('"','&quot;')
fullhtml = CHAPTER_START.substitute(vals) + \ fullhtml = CHAPTER_START.substitute(vals) + \
chap.html + CHAPTER_END.substitute(vals) chap_data.strip() + \
CHAPTER_END.substitute(vals)
# strip to avoid ever growning numbers of newlines.
# ffnet(& maybe others) gives the whole chapter text # ffnet(& maybe others) gives the whole chapter text
# as one line. This causes problems for nook(at # as one line. This causes problems for nook(at
# least) when the chapter size starts getting big # least) when the chapter size starts getting big
+38 -11
View File
@@ -1,6 +1,6 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team, 2015 FanFicFare team # Copyright 2011 Fanficdownloader team, 2016 FanFicFare team
# #
# Licensed under the Apache License, Version 2.0 (the "License"); # Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License. # you may not use this file except in compliance with the License.
@@ -18,6 +18,8 @@
import logging import logging
import string import string
import bs4
from base_writer import * from base_writer import *
class HTMLWriter(BaseStoryWriter): class HTMLWriter(BaseStoryWriter):
@@ -32,7 +34,7 @@ class HTMLWriter(BaseStoryWriter):
def __init__(self, config, story): def __init__(self, config, story):
BaseStoryWriter.__init__(self, config, story) BaseStoryWriter.__init__(self, config, story)
self.HTML_FILE_START = string.Template('''<!DOCTYPE html> self.HTML_FILE_START = string.Template('''<!DOCTYPE html>
<html> <html>
<head> <head>
@@ -48,7 +50,7 @@ ${output_css}
self.HTML_COVER = string.Template(''' self.HTML_COVER = string.Template('''
<img src="${coverimg}" alt="cover" /> <img src="${coverimg}" alt="cover" />
''') ''')
self.HTML_TITLE_PAGE_START = string.Template(''' self.HTML_TITLE_PAGE_START = string.Template('''
<table class="full"> <table class="full">
''') ''')
@@ -62,14 +64,14 @@ ${output_css}
''') ''')
self.HTML_TOC_PAGE_START = string.Template(''' self.HTML_TOC_PAGE_START = string.Template('''
<a name="TOCTOP"><h2>Table of Contents</h2> <a name="TOCTOP"><h2>Table of Contents</h2></a>
<p> <p>
''') ''')
self.HTML_TOC_ENTRY = string.Template(''' self.HTML_TOC_ENTRY = string.Template('''
<a href="#section${index}">${chapter}</a><br /> <a href="#section${index}">${chapter}</a><br />
''') ''')
self.HTML_TOC_PAGE_END = string.Template(''' self.HTML_TOC_PAGE_END = string.Template('''
</p> </p>
''') ''')
@@ -100,12 +102,12 @@ ${output_css}
FILE_END = string.Template(self.getConfig("file_end")) FILE_END = string.Template(self.getConfig("file_end"))
else: else:
FILE_END = self.HTML_FILE_END FILE_END = self.HTML_FILE_END
self._write(out,FILE_START.substitute(self.story.getAllMetadata())) self._write(out,FILE_START.substitute(self.story.getAllMetadata()))
if self.getConfig('include_images') and self.story.cover: if self.getConfig('include_images') and self.story.cover:
self._write(out,COVER.substitute(dict(self.story.getAllMetadata().items()+{'coverimg':self.story.cover}.items()))) self._write(out,COVER.substitute(dict(self.story.getAllMetadata().items()+{'coverimg':self.story.cover}.items())))
self.writeTitlePage(out, self.writeTitlePage(out,
self.HTML_TITLE_PAGE_START, self.HTML_TITLE_PAGE_START,
self.HTML_TITLE_ENTRY, self.HTML_TITLE_ENTRY,
@@ -120,18 +122,43 @@ ${output_css}
CHAPTER_START = string.Template(self.getConfig("chapter_start")) CHAPTER_START = string.Template(self.getConfig("chapter_start"))
else: else:
CHAPTER_START = self.HTML_CHAPTER_START CHAPTER_START = self.HTML_CHAPTER_START
if self.hasConfig('chapter_end'): if self.hasConfig('chapter_end'):
CHAPTER_END = string.Template(self.getConfig("chapter_end")) CHAPTER_END = string.Template(self.getConfig("chapter_end"))
else: else:
CHAPTER_END = self.HTML_CHAPTER_END CHAPTER_END = self.HTML_CHAPTER_END
## collect chapter urls and file names for internalize_text_links option.
chapurlmap = {}
for index, chap in enumerate(self.story.getChapters()): for index, chap in enumerate(self.story.getChapters()):
if chap.html: if chap.html:
## HTML_CHAPTER_START needs to have matching <a>
## anchor to work. Which it does by default. This
## could also be made configurable if some user
## changed it.
chapurlmap[chap.url]="#section%04d"%(index+1) # url -> index
for index, chap in enumerate(self.story.getChapters()):
if chap.html:
chap_data = chap.html
if self.getConfig('internalize_text_links'):
soup = bs4.BeautifulSoup(chap.html,'html5lib')
changed=False
for alink in soup.find_all('a'):
if alink.has_attr('href') and alink['href'] in chapurlmap:
alink['href']=chapurlmap[alink['href']]
changed=True
if changed:
chap_data = unicode(soup)
# Don't want html, head or body tags in
# chapter html--bs4 insists on adding them.
chap_data = re.sub(r"</?(html|head|body)[^>]*>\r?\n?","",chap_data)
logging.debug('Writing chapter text for: %s' % chap.title) logging.debug('Writing chapter text for: %s' % chap.title)
vals={'url':chap.url, 'chapter':chap.title, 'index':"%04d"%(index+1), 'number':index+1} vals={'url':chap.url, 'chapter':chap.title, 'index':"%04d"%(index+1), 'number':index+1}
self._write(out,CHAPTER_START.substitute(vals)) self._write(out,CHAPTER_START.substitute(vals))
self._write(out,chap.html) self._write(out,chap_data)
self._write(out,CHAPTER_END.substitute(vals)) self._write(out,CHAPTER_END.substitute(vals))
self._write(out,FILE_END.substitute(self.story.getAllMetadata())) self._write(out,FILE_END.substitute(self.story.getAllMetadata()))
@@ -139,4 +166,4 @@ ${output_css}
if self.getConfig('include_images'): if self.getConfig('include_images'):
for imgmap in self.story.getImgUrls(): for imgmap in self.story.getImgUrls():
self.writeFile(imgmap['newsrc'],imgmap['data']) self.writeFile(imgmap['newsrc'],imgmap['data'])
+3 -3
View File
@@ -21,7 +21,7 @@ from textwrap import wrap
from base_writer import * from base_writer import *
from ..html2text import html2text from html2text import html2text
## In BaseStoryWriter, we define _write to encode <unicode> objects ## In BaseStoryWriter, we define _write to encode <unicode> objects
## back into <string> for true output. But txt needs to write the ## back into <string> for true output. But txt needs to write the
@@ -109,7 +109,7 @@ End file.
self.wrap_width = self.getConfig('wrap_width') self.wrap_width = self.getConfig('wrap_width')
if self.wrap_width == '' or self.wrap_width == '0': if self.wrap_width == '' or self.wrap_width == '0':
self.wrap_width = None self.wrap_width = 0
else: else:
self.wrap_width = int(self.wrap_width) self.wrap_width = int(self.wrap_width)
@@ -159,7 +159,7 @@ End file.
logging.debug('Writing chapter text for: %s' % chap.title) logging.debug('Writing chapter text for: %s' % chap.title)
vals={'url':chap.url, 'chapter':chap.title, 'index':"%04d"%(index+1), 'number':index+1} vals={'url':chap.url, 'chapter':chap.title, 'index':"%04d"%(index+1), 'number':index+1}
self._write(out,self.lineends(self.wraplines(removeAllEntities(CHAPTER_START.substitute(vals))))) self._write(out,self.lineends(self.wraplines(removeAllEntities(CHAPTER_START.substitute(vals)))))
self._write(out,self.lineends(html2text(chap.html,wrap_width=self.wrap_width))) self._write(out,self.lineends(html2text(chap.html,bodywidth=self.wrap_width)))
self._write(out,self.lineends(self.wraplines(removeAllEntities(CHAPTER_END.substitute(vals))))) self._write(out,self.lineends(self.wraplines(removeAllEntities(CHAPTER_END.substitute(vals)))))
self._write(out,self.lineends(self.wraplines(FILE_END.substitute(self.story.getAllMetadata())))) self._write(out,self.lineends(self.wraplines(FILE_END.substitute(self.story.getAllMetadata()))))
+857
View File
@@ -0,0 +1,857 @@
#!/usr/bin/env python
# coding: utf-8
"""html2text: Turn HTML into equivalent Markdown-structured text."""
from __future__ import division
import re
import sys
import cgi
try:
from textwrap import wrap
except ImportError: # pragma: no cover
pass
from html2text.compat import urlparse, HTMLParser
from html2text import config
from html2text.utils import (
name2cp,
unifiable_n,
google_text_emphasis,
google_fixed_width_font,
element_style,
hn,
google_has_height,
escape_md,
google_list_style,
list_numbering_start,
dumb_css_parser,
escape_md_section,
skipwrap
)
__version__ = (2016, 4, 2)
# TODO:
# Support decoded entities with UNIFIABLE.
class HTML2Text(HTMLParser.HTMLParser):
def __init__(self, out=None, baseurl='', bodywidth=config.BODY_WIDTH):
"""
Input parameters:
out: possible custom replacement for self.outtextf (which
appends lines of text).
baseurl: base URL of the document we process
"""
kwargs = {}
if sys.version_info >= (3, 4):
kwargs['convert_charrefs'] = False
HTMLParser.HTMLParser.__init__(self, **kwargs)
# Config options
self.split_next_td = False
self.td_count = 0
self.table_start = False
self.unicode_snob = config.UNICODE_SNOB # covered in cli
self.escape_snob = config.ESCAPE_SNOB # covered in cli
self.links_each_paragraph = config.LINKS_EACH_PARAGRAPH
self.body_width = bodywidth # covered in cli
self.skip_internal_links = config.SKIP_INTERNAL_LINKS # covered in cli
self.inline_links = config.INLINE_LINKS # covered in cli
self.protect_links = config.PROTECT_LINKS # covered in cli
self.google_list_indent = config.GOOGLE_LIST_INDENT # covered in cli
self.ignore_links = config.IGNORE_ANCHORS # covered in cli
self.ignore_images = config.IGNORE_IMAGES # covered in cli
self.images_to_alt = config.IMAGES_TO_ALT # covered in cli
self.images_with_size = config.IMAGES_WITH_SIZE # covered in cli
self.ignore_emphasis = config.IGNORE_EMPHASIS # covered in cli
self.bypass_tables = config.BYPASS_TABLES # covered in cli
self.google_doc = False # covered in cli
self.ul_item_mark = '*' # covered in cli
self.emphasis_mark = '_' # covered in cli
self.strong_mark = '**'
self.single_line_break = config.SINGLE_LINE_BREAK # covered in cli
self.use_automatic_links = config.USE_AUTOMATIC_LINKS # covered in cli
self.hide_strikethrough = False # covered in cli
self.mark_code = config.MARK_CODE
self.wrap_links = config.WRAP_LINKS # covered in cli
self.tag_callback = None
if out is None: # pragma: no cover
self.out = self.outtextf
else: # pragma: no cover
self.out = out
# empty list to store output characters before they are "joined"
self.outtextlist = []
self.quiet = 0
self.p_p = 0 # number of newline character to print before next output
self.outcount = 0
self.start = 1
self.space = 0
self.a = []
self.astack = []
self.maybe_automatic_link = None
self.empty_link = False
self.absolute_url_matcher = re.compile(r'^[a-zA-Z+]+://')
self.acount = 0
self.list = []
self.blockquote = 0
self.pre = 0
self.startpre = 0
self.code = False
self.br_toggle = ''
self.lastWasNL = 0
self.lastWasList = False
self.style = 0
self.style_def = {}
self.tag_stack = []
self.emphasis = 0
self.drop_white_space = 0
self.inheader = False
self.abbr_title = None # current abbreviation definition
self.abbr_data = None # last inner HTML (for abbr being defined)
self.abbr_list = {} # stack of abbreviations to write later
self.baseurl = baseurl
try:
del unifiable_n[name2cp('nbsp')]
except KeyError:
pass
config.UNIFIABLE['nbsp'] = '&nbsp_place_holder;'
def feed(self, data):
data = data.replace("</' + 'script>", "</ignore>")
HTMLParser.HTMLParser.feed(self, data)
def handle(self, data):
self.feed(data)
self.feed("")
return self.optwrap(self.close())
def outtextf(self, s):
self.outtextlist.append(s)
if s:
self.lastWasNL = s[-1] == '\n'
def close(self):
HTMLParser.HTMLParser.close(self)
try:
nochr = unicode('')
except NameError:
nochr = str('')
self.pbr()
self.o('', 0, 'end')
outtext = nochr.join(self.outtextlist)
if self.unicode_snob:
try:
nbsp = unichr(name2cp('nbsp'))
except NameError:
nbsp = chr(name2cp('nbsp'))
else:
try:
nbsp = unichr(32)
except NameError:
nbsp = chr(32)
try:
outtext = outtext.replace(unicode('&nbsp_place_holder;'), nbsp)
except NameError:
outtext = outtext.replace('&nbsp_place_holder;', nbsp)
# Clear self.outtextlist to avoid memory leak of its content to
# the next handling.
self.outtextlist = []
return outtext
def handle_charref(self, c):
charref = self.charref(c)
if not self.code and not self.pre:
charref = cgi.escape(charref)
self.handle_data(charref, True)
def handle_entityref(self, c):
entityref = self.entityref(c)
if (not self.code and not self.pre
and entityref != '&nbsp_place_holder;'):
entityref = cgi.escape(entityref)
self.handle_data(entityref, True)
def handle_starttag(self, tag, attrs):
self.handle_tag(tag, attrs, 1)
def handle_endtag(self, tag):
self.handle_tag(tag, None, 0)
def previousIndex(self, attrs):
"""
:type attrs: dict
:returns: The index of certain set of attributes (of a link) in the
self.a list. If the set of attributes is not found, returns None
:rtype: int
"""
if 'href' not in attrs: # pragma: no cover
return None
i = -1
for a in self.a:
i += 1
match = 0
if ('href' in a) and a['href'] == attrs['href']:
if ('title' in a) or ('title' in attrs):
if (('title' in a) and ('title' in attrs) and
a['title'] == attrs['title']):
match = True
else:
match = True
if match:
return i
def handle_emphasis(self, start, tag_style, parent_style):
"""
Handles various text emphases
"""
tag_emphasis = google_text_emphasis(tag_style)
parent_emphasis = google_text_emphasis(parent_style)
# handle Google's text emphasis
strikethrough = 'line-through' in \
tag_emphasis and self.hide_strikethrough
bold = 'bold' in tag_emphasis and not 'bold' in parent_emphasis
italic = 'italic' in tag_emphasis and not 'italic' in parent_emphasis
fixed = google_fixed_width_font(tag_style) and not \
google_fixed_width_font(parent_style) and not self.pre
if start:
# crossed-out text must be handled before other attributes
# in order not to output qualifiers unnecessarily
if bold or italic or fixed:
self.emphasis += 1
if strikethrough:
self.quiet += 1
if italic:
self.o(self.emphasis_mark)
self.drop_white_space += 1
if bold:
self.o(self.strong_mark)
self.drop_white_space += 1
if fixed:
self.o('`')
self.drop_white_space += 1
self.code = True
else:
if bold or italic or fixed:
# there must not be whitespace before closing emphasis mark
self.emphasis -= 1
self.space = 0
if fixed:
if self.drop_white_space:
# empty emphasis, drop it
self.drop_white_space -= 1
else:
self.o('`')
self.code = False
if bold:
if self.drop_white_space:
# empty emphasis, drop it
self.drop_white_space -= 1
else:
self.o(self.strong_mark)
if italic:
if self.drop_white_space:
# empty emphasis, drop it
self.drop_white_space -= 1
else:
self.o(self.emphasis_mark)
# space is only allowed after *all* emphasis marks
if (bold or italic) and not self.emphasis:
self.o(" ")
if strikethrough:
self.quiet -= 1
def handle_tag(self, tag, attrs, start):
# attrs is None for endtags
if attrs is None:
attrs = {}
else:
attrs = dict(attrs)
if self.tag_callback is not None:
if self.tag_callback(self, tag, attrs, start) is True:
return
# first thing inside the anchor tag is another tag that produces some output
if (start and not self.maybe_automatic_link is None
and tag not in ['p', 'div', 'style', 'dl', 'dt']
and (tag != "img" or self.ignore_images)):
self.o("[")
self.maybe_automatic_link = None
self.empty_link = False
if self.google_doc:
# the attrs parameter is empty for a closing tag. in addition, we
# need the attributes of the parent nodes in order to get a
# complete style description for the current element. we assume
# that google docs export well formed html.
parent_style = {}
if start:
if self.tag_stack:
parent_style = self.tag_stack[-1][2]
tag_style = element_style(attrs, self.style_def, parent_style)
self.tag_stack.append((tag, attrs, tag_style))
else:
dummy, attrs, tag_style = self.tag_stack.pop() if self.tag_stack else (None, {}, {})
if self.tag_stack:
parent_style = self.tag_stack[-1][2]
if hn(tag):
self.p()
if start:
self.inheader = True
self.o(hn(tag) * "#" + ' ')
else:
self.inheader = False
return # prevent redundant emphasis marks on headers
if tag in ['p', 'div']:
if self.google_doc:
if start and google_has_height(tag_style):
self.p()
else:
self.soft_br()
else:
self.p()
if tag == "br" and start:
self.o(" \n")
if tag == "hr" and start:
self.p()
self.o("* * *")
self.p()
if tag in ["head", "style", 'script']:
if start:
self.quiet += 1
else:
self.quiet -= 1
if tag == "style":
if start:
self.style += 1
else:
self.style -= 1
if tag in ["body"]:
self.quiet = 0 # sites like 9rules.com never close <head>
if tag == "blockquote":
if start:
self.p()
self.o('> ', 0, 1)
self.start = 1
self.blockquote += 1
else:
self.blockquote -= 1
self.p()
if tag in ['em', 'i', 'u'] and not self.ignore_emphasis:
self.o(self.emphasis_mark)
if tag in ['strong', 'b'] and not self.ignore_emphasis:
self.o(self.strong_mark)
if tag in ['del', 'strike', 's']:
if start:
self.o('~~')
else:
self.o('~~')
if self.google_doc:
if not self.inheader:
# handle some font attributes, but leave headers clean
self.handle_emphasis(start, tag_style, parent_style)
if tag in ["code", "tt"] and not self.pre:
self.o('`') # TODO: `` `this` ``
self.code = not self.code
if tag == "abbr":
if start:
self.abbr_title = None
self.abbr_data = ''
if ('title' in attrs):
self.abbr_title = attrs['title']
else:
if self.abbr_title is not None:
self.abbr_list[self.abbr_data] = self.abbr_title
self.abbr_title = None
self.abbr_data = ''
if tag == "a" and not self.ignore_links:
if start:
if ('href' in attrs) and \
(attrs['href'] is not None) and \
not (self.skip_internal_links and
attrs['href'].startswith('#')):
self.astack.append(attrs)
self.maybe_automatic_link = attrs['href']
self.empty_link = True
if self.protect_links:
attrs['href'] = '<'+attrs['href']+'>'
else:
self.astack.append(None)
else:
if self.astack:
a = self.astack.pop()
if self.maybe_automatic_link and not self.empty_link:
self.maybe_automatic_link = None
elif a:
if self.empty_link:
self.o("[")
self.empty_link = False
self.maybe_automatic_link = None
if self.inline_links:
try:
title = escape_md(a['title'])
except KeyError:
self.o("](" + escape_md(urlparse.urljoin(self.baseurl, a['href'])) + ")")
else:
self.o("](" + escape_md(urlparse.urljoin(self.baseurl, a['href']))
+ ' "' + title + '" )')
else:
i = self.previousIndex(a)
if i is not None:
a = self.a[i]
else:
self.acount += 1
a['count'] = self.acount
a['outcount'] = self.outcount
self.a.append(a)
self.o("][" + str(a['count']) + "]")
if tag == "img" and start and not self.ignore_images:
if 'src' in attrs:
if not self.images_to_alt:
attrs['href'] = attrs['src']
alt = attrs.get('alt') or ''
# If we have images_with_size, write raw html including width,
# height, and alt attributes
if self.images_with_size and \
("width" in attrs or "height" in attrs):
self.o("<img src='" + attrs["src"] + "' ")
if "width" in attrs:
self.o("width='" + attrs["width"] + "' ")
if "height" in attrs:
self.o("height='" + attrs["height"] + "' ")
if alt:
self.o("alt='" + alt + "' ")
self.o("/>")
return
# If we have a link to create, output the start
if not self.maybe_automatic_link is None:
href = self.maybe_automatic_link
if self.images_to_alt and escape_md(alt) == href and \
self.absolute_url_matcher.match(href):
self.o("<" + escape_md(alt) + ">")
self.empty_link = False
return
else:
self.o("[")
self.maybe_automatic_link = None
self.empty_link = False
# If we have images_to_alt, we discard the image itself,
# considering only the alt text.
if self.images_to_alt:
self.o(escape_md(alt))
else:
self.o("![" + escape_md(alt) + "]")
if self.inline_links:
href = attrs.get('href') or ''
self.o("(" + escape_md(urlparse.urljoin(self.baseurl, href)) + ")")
else:
i = self.previousIndex(attrs)
if i is not None:
attrs = self.a[i]
else:
self.acount += 1
attrs['count'] = self.acount
attrs['outcount'] = self.outcount
self.a.append(attrs)
self.o("[" + str(attrs['count']) + "]")
if tag == 'dl' and start:
self.p()
if tag == 'dt' and not start:
self.pbr()
if tag == 'dd' and start:
self.o(' ')
if tag == 'dd' and not start:
self.pbr()
if tag in ["ol", "ul"]:
# Google Docs create sub lists as top level lists
if (not self.list) and (not self.lastWasList):
self.p()
if start:
if self.google_doc:
list_style = google_list_style(tag_style)
else:
list_style = tag
numbering_start = list_numbering_start(attrs)
self.list.append({
'name': list_style,
'num': numbering_start
})
else:
if self.list:
self.list.pop()
if (not self.google_doc) and (not self.list):
self.o('\n')
self.lastWasList = True
else:
self.lastWasList = False
if tag == 'li':
self.pbr()
if start:
if self.list:
li = self.list[-1]
else:
li = {'name': 'ul', 'num': 0}
if self.google_doc:
nest_count = self.google_nest_count(tag_style)
else:
nest_count = len(self.list)
# TODO: line up <ol><li>s > 9 correctly.
self.o(" " * nest_count)
if li['name'] == "ul":
self.o(self.ul_item_mark + " ")
elif li['name'] == "ol":
li['num'] += 1
self.o(str(li['num']) + ". ")
self.start = 1
if tag in ["table", "tr", "td", "th"]:
if self.bypass_tables:
if start:
self.soft_br()
if tag in ["td", "th"]:
if start:
self.o('<{0}>\n\n'.format(tag))
else:
self.o('\n</{0}>'.format(tag))
else:
if start:
self.o('<{0}>'.format(tag))
else:
self.o('</{0}>'.format(tag))
else:
if tag == "table" and start:
self.table_start = True
if tag in ["td", "th"] and start:
if self.split_next_td:
self.o("| ")
self.split_next_td = True
if tag == "tr" and start:
self.td_count = 0
if tag == "tr" and not start:
self.split_next_td = False
self.soft_br()
if tag == "tr" and not start and self.table_start:
# Underline table header
self.o("|".join(["---"] * self.td_count))
self.soft_br()
self.table_start = False
if tag in ["td", "th"] and start:
self.td_count += 1
if tag == "pre":
if start:
self.startpre = 1
self.pre = 1
else:
self.pre = 0
if self.mark_code:
self.out("\n[/code]")
self.p()
# TODO: Add docstring for these one letter functions
def pbr(self):
"Pretty print has a line break"
if self.p_p == 0:
self.p_p = 1
def p(self):
"Set pretty print to 1 or 2 lines"
self.p_p = 1 if self.single_line_break else 2
def soft_br(self):
"Soft breaks"
self.pbr()
self.br_toggle = ' '
def o(self, data, puredata=0, force=0):
"""
Deal with indentation and whitespace
"""
if self.abbr_data is not None:
self.abbr_data += data
if not self.quiet:
if self.google_doc:
# prevent white space immediately after 'begin emphasis'
# marks ('**' and '_')
lstripped_data = data.lstrip()
if self.drop_white_space and not (self.pre or self.code):
data = lstripped_data
if lstripped_data != '':
self.drop_white_space = 0
if puredata and not self.pre:
# This is a very dangerous call ... it could mess up
# all handling of &nbsp; when not handled properly
# (see entityref)
data = re.sub(r'\s+', r' ', data)
if data and data[0] == ' ':
self.space = 1
data = data[1:]
if not data and not force:
return
if self.startpre:
#self.out(" :") #TODO: not output when already one there
if not data.startswith("\n"): # <pre>stuff...
data = "\n" + data
if self.mark_code:
self.out("\n[code]")
self.p_p = 0
bq = (">" * self.blockquote)
if not (force and data and data[0] == ">") and self.blockquote:
bq += " "
if self.pre:
if not self.list:
bq += " "
#else: list content is already partially indented
for i in range(len(self.list)):
bq += " "
data = data.replace("\n", "\n" + bq)
if self.startpre:
self.startpre = 0
if self.list:
# use existing initial indentation
data = data.lstrip("\n")
if self.start:
self.space = 0
self.p_p = 0
self.start = 0
if force == 'end':
# It's the end.
self.p_p = 0
self.out("\n")
self.space = 0
if self.p_p:
self.out((self.br_toggle + '\n' + bq) * self.p_p)
self.space = 0
self.br_toggle = ''
if self.space:
if not self.lastWasNL:
self.out(' ')
self.space = 0
if self.a and ((self.p_p == 2 and self.links_each_paragraph)
or force == "end"):
if force == "end":
self.out("\n")
newa = []
for link in self.a:
if self.outcount > link['outcount']:
self.out(" [" + str(link['count']) + "]: " +
urlparse.urljoin(self.baseurl, link['href']))
if 'title' in link:
self.out(" (" + link['title'] + ")")
self.out("\n")
else:
newa.append(link)
# Don't need an extra line when nothing was done.
if self.a != newa:
self.out("\n")
self.a = newa
if self.abbr_list and force == "end":
for abbr, definition in self.abbr_list.items():
self.out(" *[" + abbr + "]: " + definition + "\n")
self.p_p = 0
self.out(data)
self.outcount += 1
def handle_data(self, data, entity_char=False):
if r'\/script>' in data:
self.quiet -= 1
if self.style:
self.style_def.update(dumb_css_parser(data))
if not self.maybe_automatic_link is None:
href = self.maybe_automatic_link
if (href == data and self.absolute_url_matcher.match(href)
and self.use_automatic_links):
self.o("<" + data + ">")
self.empty_link = False
return
else:
self.o("[")
self.maybe_automatic_link = None
self.empty_link = False
if not self.code and not self.pre and not entity_char:
data = escape_md_section(data, snob=self.escape_snob)
self.o(data, 1)
def unknown_decl(self, data): # pragma: no cover
# TODO: what is this doing here?
pass
def charref(self, name):
if name[0] in ['x', 'X']:
c = int(name[1:], 16)
else:
c = int(name)
if not self.unicode_snob and c in unifiable_n.keys():
return unifiable_n[c]
else:
try:
try:
return unichr(c)
except NameError: # Python3
return chr(c)
except ValueError: # invalid unicode
return ''
def entityref(self, c):
if not self.unicode_snob and c in config.UNIFIABLE.keys():
return config.UNIFIABLE[c]
else:
try:
name2cp(c)
except KeyError:
return "&" + c + ';'
else:
if c == 'nbsp':
return config.UNIFIABLE[c]
else:
try:
return unichr(name2cp(c))
except NameError: # Python3
return chr(name2cp(c))
def replaceEntities(self, s):
s = s.group(1)
if s[0] == "#":
return self.charref(s[1:])
else:
return self.entityref(s)
def unescape(self, s):
return config.RE_UNESCAPE.sub(self.replaceEntities, s)
def google_nest_count(self, style):
"""
Calculate the nesting count of google doc lists
:type style: dict
:rtype: int
"""
nest_count = 0
if 'margin-left' in style:
nest_count = int(style['margin-left'][:-2]) \
// self.google_list_indent
return nest_count
def optwrap(self, text):
"""
Wrap all paragraphs in the provided text.
:type text: str
:rtype: str
"""
if not self.body_width:
return text
assert wrap, "Requires Python 2.3."
result = ''
newlines = 0
# I cannot think of a better solution for now.
# To avoid the non-wrap behaviour for entire paras
# because of the presence of a link in it
if not self.wrap_links:
self.inline_links = False
for para in text.split("\n"):
if len(para) > 0:
if not skipwrap(para, self.wrap_links):
result += "\n".join(wrap(para, self.body_width))
if para.endswith(' '):
result += " \n"
newlines = 1
else:
result += "\n\n"
newlines = 2
else:
# Warning for the tempted!!!
# Be aware that obvious replacement of this with
# line.isspace()
# DOES NOT work! Explanations are welcome.
if not config.RE_SPACE.match(para):
result += para + "\n"
newlines = 1
else:
if newlines < 2:
result += "\n"
newlines += 1
return result
def html2text(html, baseurl='', bodywidth=None):
if bodywidth is None:
bodywidth = config.BODY_WIDTH
h = HTML2Text(baseurl=baseurl, bodywidth=bodywidth)
return h.handle(html)
def unescape(s, unicode_snob=False):
h = HTML2Text()
h.unicode_snob = unicode_snob
return h.unescape(s)
if __name__ == "__main__":
from html2text.cli import main
main()
+13
View File
@@ -0,0 +1,13 @@
import sys
if sys.version_info[0] == 2:
import htmlentitydefs
import urlparse
import HTMLParser
import urllib
else:
import urllib.parse as urlparse
import html.entities as htmlentitydefs
import html.parser as HTMLParser
import urllib.request as urllib
+123
View File
@@ -0,0 +1,123 @@
import re
# Use Unicode characters instead of their ascii psuedo-replacements
UNICODE_SNOB = 0
# Escape all special characters. Output is less readable, but avoids
# corner case formatting issues.
ESCAPE_SNOB = 0
# Put the links after each paragraph instead of at the end.
LINKS_EACH_PARAGRAPH = 0
# Wrap long lines at position. 0 for no wrapping. (Requires Python 2.3.)
BODY_WIDTH = 78
# Don't show internal links (href="#local-anchor") -- corresponding link
# targets won't be visible in the plain text file anyway.
SKIP_INTERNAL_LINKS = True
# Use inline, rather than reference, formatting for images and links
INLINE_LINKS = True
# Protect links from line breaks surrounding them with angle brackets (in
# addition to their square brackets)
PROTECT_LINKS = False
# WRAP_LINKS = True
WRAP_LINKS = True
# Number of pixels Google indents nested lists
GOOGLE_LIST_INDENT = 36
IGNORE_ANCHORS = False
IGNORE_IMAGES = False
IMAGES_TO_ALT = False
IMAGES_WITH_SIZE = False
IGNORE_EMPHASIS = False
MARK_CODE = False
DECODE_ERRORS = 'strict'
# Convert links with same href and text to <href> format if they are absolute links
USE_AUTOMATIC_LINKS = True
# For checking space-only lines on line 771
RE_SPACE = re.compile(r'\s\+')
RE_UNESCAPE = re.compile(r"&(#?[xX]?(?:[0-9a-fA-F]+|\w{1,8}));")
RE_ORDERED_LIST_MATCHER = re.compile(r'\d+\.\s')
RE_UNORDERED_LIST_MATCHER = re.compile(r'[-\*\+]\s')
RE_MD_CHARS_MATCHER = re.compile(r"([\\\[\]\(\)])")
RE_MD_CHARS_MATCHER_ALL = re.compile(r"([`\*_{}\[\]\(\)#!])")
RE_LINK = re.compile(r"(\[.*?\] ?\(.*?\))|(\[.*?\]:.*?)") # to find links in the text
RE_MD_DOT_MATCHER = re.compile(r"""
^ # start of line
(\s*\d+) # optional whitespace and a number
(\.) # dot
(?=\s) # lookahead assert whitespace
""", re.MULTILINE | re.VERBOSE)
RE_MD_PLUS_MATCHER = re.compile(r"""
^
(\s*)
(\+)
(?=\s)
""", flags=re.MULTILINE | re.VERBOSE)
RE_MD_DASH_MATCHER = re.compile(r"""
^
(\s*)
(-)
(?=\s|\-) # followed by whitespace (bullet list, or spaced out hr)
# or another dash (header or hr)
""", flags=re.MULTILINE | re.VERBOSE)
RE_SLASH_CHARS = r'\`*_{}[]()#+-.!'
RE_MD_BACKSLASH_MATCHER = re.compile(r'''
(\\) # match one slash
(?=[%s]) # followed by a char that requires escaping
''' % re.escape(RE_SLASH_CHARS),
flags=re.VERBOSE)
UNIFIABLE = {
'rsquo': "'",
'lsquo': "'",
'rdquo': '"',
'ldquo': '"',
'copy': '(C)',
'mdash': '--',
'nbsp': ' ',
'rarr': '->',
'larr': '<-',
'middot': '*',
'ndash': '-',
'oelig': 'oe',
'aelig': 'ae',
'agrave': 'a',
'aacute': 'a',
'acirc': 'a',
'atilde': 'a',
'auml': 'a',
'aring': 'a',
'egrave': 'e',
'eacute': 'e',
'ecirc': 'e',
'euml': 'e',
'igrave': 'i',
'iacute': 'i',
'icirc': 'i',
'iuml': 'i',
'ograve': 'o',
'oacute': 'o',
'ocirc': 'o',
'otilde': 'o',
'ouml': 'o',
'ugrave': 'u',
'uacute': 'u',
'ucirc': 'u',
'uuml': 'u',
'lrm': '',
'rlm': ''
}
BYPASS_TABLES = False
# Use a single line break after a block element rather an two line breaks.
# NOTE: Requires body width setting to be 0.
SINGLE_LINE_BREAK = False
+246
View File
@@ -0,0 +1,246 @@
import sys
from html2text import config
from html2text.compat import htmlentitydefs
def name2cp(k):
"""Return sname to codepoint"""
if k == 'apos':
return ord("'")
return htmlentitydefs.name2codepoint[k]
unifiable_n = {}
for k in config.UNIFIABLE.keys():
unifiable_n[name2cp(k)] = config.UNIFIABLE[k]
def hn(tag):
if tag[0] == 'h' and len(tag) == 2:
try:
n = int(tag[1])
if n in range(1, 10): # pragma: no branch
return n
except ValueError:
return 0
def dumb_property_dict(style):
"""
:returns: A hash of css attributes
"""
out = dict([(x.strip(), y.strip()) for x, y in
[z.split(':', 1) for z in
style.split(';') if ':' in z
]
]
)
return out
def dumb_css_parser(data):
"""
:type data: str
:returns: A hash of css selectors, each of which contains a hash of
css attributes.
:rtype: dict
"""
# remove @import sentences
data += ';'
importIndex = data.find('@import')
while importIndex != -1:
data = data[0:importIndex] + data[data.find(';', importIndex) + 1:]
importIndex = data.find('@import')
# parse the css. reverted from dictionary comprehension in order to
# support older pythons
elements = [x.split('{') for x in data.split('}') if '{' in x.strip()]
try:
elements = dict([(a.strip(), dumb_property_dict(b))
for a, b in elements])
except ValueError: # pragma: no cover
elements = {} # not that important
return elements
def element_style(attrs, style_def, parent_style):
"""
:type attrs: dict
:type style_def: dict
:type style_def: dict
:returns: A hash of the 'final' style attributes of the element
:rtype: dict
"""
style = parent_style.copy()
if 'class' in attrs:
for css_class in attrs['class'].split():
css_style = style_def.get('.' + css_class, {})
style.update(css_style)
if 'style' in attrs:
immediate_style = dumb_property_dict(attrs['style'])
style.update(immediate_style)
return style
def google_list_style(style):
"""
Finds out whether this is an ordered or unordered list
:type style: dict
:rtype: str
"""
if 'list-style-type' in style:
list_style = style['list-style-type']
if list_style in ['disc', 'circle', 'square', 'none']:
return 'ul'
return 'ol'
def google_has_height(style):
"""
Check if the style of the element has the 'height' attribute
explicitly defined
:type style: dict
:rtype: bool
"""
if 'height' in style:
return True
return False
def google_text_emphasis(style):
"""
:type style: dict
:returns: A list of all emphasis modifiers of the element
:rtype: list
"""
emphasis = []
if 'text-decoration' in style:
emphasis.append(style['text-decoration'])
if 'font-style' in style:
emphasis.append(style['font-style'])
if 'font-weight' in style:
emphasis.append(style['font-weight'])
return emphasis
def google_fixed_width_font(style):
"""
Check if the css of the current element defines a fixed width font
:type style: dict
:rtype: bool
"""
font_family = ''
if 'font-family' in style:
font_family = style['font-family']
if 'Courier New' == font_family or 'Consolas' == font_family:
return True
return False
def list_numbering_start(attrs):
"""
Extract numbering from list element attributes
:type attrs: dict
:rtype: int or None
"""
if 'start' in attrs:
try:
return int(attrs['start']) - 1
except ValueError:
pass
return 0
def skipwrap(para, wrap_links):
# If it appears to contain a link
# don't wrap
if (len(config.RE_LINK.findall(para)) > 0) and not wrap_links:
return True
# If the text begins with four spaces or one tab, it's a code block;
# don't wrap
if para[0:4] == ' ' or para[0] == '\t':
return True
# If the text begins with only two "--", possibly preceded by
# whitespace, that's an emdash; so wrap.
stripped = para.lstrip()
if stripped[0:2] == "--" and len(stripped) > 2 and stripped[2] != "-":
return False
# I'm not sure what this is for; I thought it was to detect lists,
# but there's a <br>-inside-<span> case in one of the tests that
# also depends upon it.
if stripped[0:1] == '-' or stripped[0:1] == '*':
return True
# If the text begins with a single -, *, or +, followed by a space,
# or an integer, followed by a ., followed by a space (in either
# case optionally proceeded by whitespace), it's a list; don't wrap.
if config.RE_ORDERED_LIST_MATCHER.match(stripped) or \
config.RE_UNORDERED_LIST_MATCHER.match(stripped):
return True
return False
def wrapwrite(text):
text = text.encode('utf-8')
try: # Python3
sys.stdout.buffer.write(text)
except AttributeError:
sys.stdout.write(text)
def wrap_read(): # pragma: no cover
"""
:rtype: str
"""
try:
return sys.stdin.read()
except AttributeError:
return sys.stdin.buffer.read()
def escape_md(text):
"""
Escapes markdown-sensitive characters within other markdown
constructs.
"""
return config.RE_MD_CHARS_MATCHER.sub(r"\\\1", text)
def escape_md_section(text, snob=False):
"""
Escapes markdown-sensitive characters across whole document sections.
"""
text = config.RE_MD_BACKSLASH_MATCHER.sub(r"\\\1", text)
if snob:
text = config.RE_MD_CHARS_MATCHER_ALL.sub(r"\\\1", text)
text = config.RE_MD_DOT_MATCHER.sub(r"\1\\\2", text)
text = config.RE_MD_PLUS_MATCHER.sub(r"\1\\\2", text)
text = config.RE_MD_DASH_MATCHER.sub(r"\1\\\2", text)
return text
+1 -1
View File
@@ -36,7 +36,7 @@ if __name__=="__main__":
os.chdir('../included_dependencies') os.chdir('../included_dependencies')
# 'a' for append # 'a' for append
files=['gif.py','six.py','bs4','html5lib','chardet'] files=['gif.py','six.py','bs4','html5lib','chardet','html2text']
createZipFile("../"+filename,"a", createZipFile("../"+filename,"a",
files, files,
exclude=exclude) exclude=exclude)
+4 -6
View File
@@ -16,14 +16,12 @@ from os import path
# Get the long description from the relevant file # Get the long description from the relevant file
with codecs.open('DESCRIPTION.rst', encoding='utf-8') as f: with codecs.open('DESCRIPTION.rst', encoding='utf-8') as f:
long_description = f.read() long_description = f.read()
setup( setup(
name="FanFicFare", name="FanFicFare",
# Versions should comply with PEP440. For a discussion on single-sourcing # Versions should comply with PEP440.
# the version across setup.py and the project code, see version="2.5.1",
# https://packaging.python.org/en/latest/single_source_version.html
version="2.3.1",
description='A tool for downloading fanfiction to eBook formats', description='A tool for downloading fanfiction to eBook formats',
long_description=long_description, long_description=long_description,
@@ -81,7 +79,7 @@ setup(
# your project is installed. For an analysis of "install_requires" vs pip's # your project is installed. For an analysis of "install_requires" vs pip's
# requirements files see: # requirements files see:
# https://packaging.python.org/en/latest/requirements.html # https://packaging.python.org/en/latest/requirements.html
install_requires=['beautifulsoup4','chardet','html5lib'], # html5lib requires 'six'. install_requires=['beautifulsoup4','chardet','html5lib','html2text'], # html5lib requires 'six'.
# List additional groups of dependencies here (e.g. development # List additional groups of dependencies here (e.g. development
# dependencies). You can install these using the following syntax, # dependencies). You can install these using the following syntax,
+79
View File
@@ -0,0 +1,79 @@
# -*- coding: utf-8 -*-
import codecs, sys, re
from tempfile import mkstemp
from os import rename, close, unlink
#print sys.argv[1:]
## Files that contain version numbers that will need to be updated.
version_files = [
# 'version_test.sml',
# 'version_test.txt',
'setup.py',
'calibre-plugin/__init__.py',
'webservice/app.yaml',
'fanficfare/cli.py',
]
## save version from this file for index.html link.
# save_file='version_test.txt'
save_file='webservice/app.yaml'
saved_version = None
def main(args):
## major.minor.micro
'''
version = (2, 3, 6)
version="2.3.6",
version: 2-3-06a
version="2.3.6"
'''
version_re = \
r'^(?P<prefix>[ ]*)version(?P<infix>[ =:"\\(]+)' \
r'(?P<major>[0-9]+)(?P<dot1>[, \\.-]+)' \
r'(?P<minor>[0-9]+)(?P<dot2>[, \\.-]+)' \
r'(?P<micro>[0-9]+[a-z]?)(?P<suffix>[",\\)]*\r?\n)$'
version_subs = '\g<prefix>version\g<infix>%s\g<dot1>%s\g<dot2>%s\g<suffix>' % tuple(args)
do_loop(version_files, version_re, version_subs)
if saved_version:
# index_files = ['index.html']
index_files = ['webservice/index.html']
index_re = 'http://([0-9-]+[a-z]?)\\.fanficfare\\.appspot\\.com'
index_subs = 'http://%s-%s-%s.fanficfare.appspot.com'%saved_version
do_loop(index_files, index_re, index_subs)
def do_loop(files, pattern, substring):
global saved_version
for source_file_path in files:
print "src:"+source_file_path
fh, target_file_path = mkstemp()
with codecs.open(target_file_path, 'w', 'utf-8') as target_file:
with codecs.open(source_file_path, 'r', 'utf-8') as source_file:
for line in source_file:
repline = re.sub(pattern, substring, line)
if line != repline and source_file_path == save_file:
m = re.match(pattern,line)
saved_version = (m.group('major'),m.group('minor'),m.group('micro'))
print("<-%s->%s"%(line,repline))
target_file.write(repline)
close(fh)
unlink(source_file_path)
rename(target_file_path,source_file_path)
if __name__ == '__main__':
args = sys.argv[1:]
try:
if len(args) != 3:
raise Exception()
[int(x) for x in args]
except:
print "Requires exactly 3 numeric args: major minor micro"
exit()
main(args)
# print saved_version
+1 -11
View File
@@ -1,6 +1,6 @@
# ffd-retief-hrd fanficfare # ffd-retief-hrd fanficfare
application: fanficfare application: fanficfare
version: 2-3-01 version: 2-5-0
runtime: python27 runtime: python27
api_version: 1 api_version: 1
threadsafe: true threadsafe: true
@@ -34,13 +34,3 @@ handlers:
- url: /.* - url: /.*
script: main.app script: main.app
#builtins:
#- datastore_admin: on
libraries:
- name: django
version: "1.2"
- name: PIL
version: "1.1.7"
+2 -2
View File
@@ -39,8 +39,8 @@
<img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif" <img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif"
alt="Powered by Google App Engine" /> alt="Powered by Google App Engine" />
<br/><br/> <br/><br/>
This is a web front-end to <A href="http://code.google.com/p/fanficdownloader/">FanFicFare</a><br/> This is a web front-end to <a href="https://github.com/JimmXinu/FanFicFare/">FanFicFare</a><br/>
Copyright &copy; Fanficdownloader team Copyright &copy; FanFicFare team
</div> </div>
</div> </div>
+7 -7
View File
@@ -24,7 +24,7 @@
<div id='urlbox'> <div id='urlbox'>
<div id='greeting'> <div id='greeting'>
<p>Hi, {{ nickname }}! This is FanFicFare, which makes reading stories from various websites <p>Hi, {{ nickname }}! This is FanFicFare, which makes reading stories from various websites
much easier. </p> much easier by helping you download them to EBook files. </p>
</div> </div>
<p> <p>
@@ -35,7 +35,7 @@
If you have any problems with this application, please If you have any problems with this application, please
report them in report them in
the <a href="http://groups.google.com/group/fanfic-downloader">FanFicFare Google Group</a>. The the <a href="http://groups.google.com/group/fanfic-downloader">FanFicFare Google Group</a>. The
<a href="http://2-3-00.fanficfare.appspot.com">previous version <a href="http://2-4-0.fanficfare.appspot.com">previous version
</a> is also available for you to use if necessary. </a> is also available for you to use if necessary.
</p> </p>
<div id='error'> <div id='error'>
@@ -75,7 +75,7 @@
<div id='urlbox'> <div id='urlbox'>
<div id='greeting'> <div id='greeting'>
<p> <p>
This is a FanFicFare, which makes reading stories from various websites much easier. Before you This is a FanFicFare, which makes reading stories from various websites much easier by helping you download them to EBook files. Before you
can start downloading fanfics, you need to login, so FanFicFare can remember your fanfics and store them. can start downloading fanfics, you need to login, so FanFicFare can remember your fanfics and store them.
</p> </p>
<p><a href="{{ login_url }}">Login using Google account</a></p> <p><a href="{{ login_url }}">Login using Google account</a></p>
@@ -85,17 +85,17 @@
<div id='typebox'> <div id='typebox'>
<p> <p>
<b>FanFicFare calibre Plugin</b> <b>FanFicFare Calibre Plugin</b>
<br /><br /> <br /><br />
There's also a version of this downloader that runs inside There's also a version of this downloader that runs inside
the popular <a href="http://calibre-ebook.com/">calibre</a> the popular <a href="http://calibre-ebook.com/">Calibre</a>
ebook management package as a plugin. ebook management package as a plugin.
<br /><br /> <br /><br />
Once you have calibre installed and running, inside Once you have Calibre installed and running, inside
calibre, you can go to 'Get plugins to enhance calibre' or Calibre, you can go to 'Get plugins to enhance calibre' or
'Get new plugins' and 'Get new plugins' and
install <a href="http://www.mobileread.com/forums/showthread.php?t=259221">FanFicFare</a>. install <a href="http://www.mobileread.com/forums/showthread.php?t=259221">FanFicFare</a>.
+1 -1
View File
@@ -68,7 +68,7 @@
<img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif" <img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif"
alt="Powered by Google App Engine" /> alt="Powered by Google App Engine" />
<br/><br/> <br/><br/>
This is a web front-end to <A href="http://code.google.com/p/fanficdownloader/">FanFicFare</a><br/> This is a web front-end to <a href="https://github.com/JimmXinu/FanFicFare/">FanFicFare</a><br/>
Copyright &copy; FanFicFare team Copyright &copy; FanFicFare team
</div> </div>
+16 -37
View File
@@ -30,20 +30,6 @@ import datetime
import traceback import traceback
from StringIO import StringIO from StringIO import StringIO
## Just to shut up the appengine warning about "You are using the
## default Django version (0.96). The default Django version will
## change in an App Engine release in the near future. Please call
## use_library() to explicitly select a Django version. For more
## information see
## http://code.google.com/appengine/docs/python/tools/libraries.html#Django"
## Note that if you are using the SDK App Engine Launcher and hit an SDK
## Console page first, you will get a django version mismatch error when you
## to go hit one of the application pages. Just change a file again, and
## make sure to hit an app page before the SDK page to clear it.
#os.environ['DJANGO_SETTINGS_MODULE'] = 'settings'
#from google.appengine.dist import use_library
#use_library('django', '1.2')
from google.appengine.ext import db from google.appengine.ext import db
from google.appengine.api import taskqueue from google.appengine.api import taskqueue
from google.appengine.api import users from google.appengine.api import users
@@ -59,11 +45,11 @@ from fanficfare import adapters, writers, exceptions
from fanficfare.configurable import Configuration from fanficfare.configurable import Configuration
class UserConfigServer(webapp2.RequestHandler): class UserConfigServer(webapp2.RequestHandler):
def getUserConfig(self,user,url,fileformat): def getUserConfig(self,user,url,fileformat):
configuration = Configuration(adapters.getConfigSectionsFor(url),fileformat) configuration = Configuration(adapters.getConfigSectionsFor(url),fileformat)
logging.debug('reading defaults.ini config file') logging.debug('reading defaults.ini config file')
configuration.read('fanficfare/defaults.ini') configuration.read('fanficfare/defaults.ini')
@@ -110,7 +96,7 @@ class MainHandler(webapp2.RequestHandler):
template_values = {'login_url' : url, 'authorized': False} template_values = {'login_url' : url, 'authorized': False}
path = os.path.join(os.path.dirname(__file__), 'index.html') path = os.path.join(os.path.dirname(__file__), 'index.html')
template_values['supported_sites'] = '<dl>\n' template_values['supported_sites'] = '<dl>\n'
for (site,examples) in adapters.getSiteExamples(): for (site,examples) in adapters.getSiteExamples():
template_values['supported_sites'] += "<dt>%s</dt>\n<dd>Example Story URLs:<br>"%site template_values['supported_sites'] += "<dt>%s</dt>\n<dd>Example Story URLs:<br>"%site
@@ -153,7 +139,7 @@ class EditConfigServer(UserConfigServer):
self.redirect("/?error=configsaved") self.redirect("/?error=configsaved")
except Exception, e: except Exception, e:
logging.info("Saved Config Failed:%s"%e) logging.info("Saved Config Failed:%s"%e)
self.redirect("/?error=custom&errtext=%s"%urlEscape(unicode(e))) self.redirect("/?error=custom&errtext=%s"%urllib.quote(unicode(e),''))
else: # not update, assume display for edit else: # not update, assume display for edit
if uconfig is not None and uconfig.config: if uconfig is not None and uconfig.config:
config = uconfig.config config = uconfig.config
@@ -254,7 +240,7 @@ class FileStatusServer(webapp2.RequestHandler):
if download: if download:
logging.info("Status url: %s" % download.url) logging.info("Status url: %s" % download.url)
if download.completed and download.format=='epub': if download.completed and download.format=='epub':
escaped_url = urlEscape(self.request.host_url+"/file/"+download.name+"."+download.format+"?id="+fileId+"&fake=file."+download.format) escaped_url = urllib.quote(self.request.host_url+"/file/"+download.name+"."+download.format+"?id="+fileId+"&fake=file."+download.format,'')
else: else:
download = DownloadMeta() download = DownloadMeta()
download.failure = "Download not found" download.failure = "Download not found"
@@ -309,7 +295,7 @@ class RecentFilesServer(webapp2.RequestHandler):
for fic in fics: for fic in fics:
if fic.completed and fic.format == 'epub': if fic.completed and fic.format == 'epub':
fic.escaped_url = urlEscape(self.request.host_url+"/file/"+fic.name+"."+fic.format+"?id="+unicode(fic.key())+"&fake=file."+fic.format) fic.escaped_url = urllib.quote(self.request.host_url+"/file/"+fic.name+"."+fic.format+"?id="+unicode(fic.key())+"&fake=file."+fic.format,'')
template_values = dict(fics = fics, nickname = user.nickname()) template_values = dict(fics = fics, nickname = user.nickname())
path = os.path.join(os.path.dirname(__file__), 'recent.html') path = os.path.join(os.path.dirname(__file__), 'recent.html')
@@ -327,7 +313,7 @@ class AllRecentFilesServer(webapp2.RequestHandler):
q.order('-date') q.order('-date')
else: else:
q.order('-count') q.order('-count')
fics = q.fetch(200) fics = q.fetch(200)
logging.info("Recent fetched %d downloads for user %s."%(len(fics),user.nickname())) logging.info("Recent fetched %d downloads for user %s."%(len(fics),user.nickname()))
@@ -336,7 +322,7 @@ class AllRecentFilesServer(webapp2.RequestHandler):
for fic in fics: for fic in fics:
ficslug = FicSlug(fic) ficslug = FicSlug(fic)
sendslugs.append(ficslug) sendslugs.append(ficslug)
template_values = dict(fics = sendslugs, nickname = user.nickname()) template_values = dict(fics = sendslugs, nickname = user.nickname())
path = os.path.join(os.path.dirname(__file__), 'allrecent.html') path = os.path.join(os.path.dirname(__file__), 'allrecent.html')
self.response.out.write(template.render(path, template_values)) self.response.out.write(template.render(path, template_values))
@@ -347,7 +333,7 @@ class FicSlug():
self.count = savedmeta.count self.count = savedmeta.count
for k, v in savedmeta.meta.iteritems(): for k, v in savedmeta.meta.iteritems():
setattr(self,k,v) setattr(self,k,v)
class FanfictionDownloader(UserConfigServer): class FanfictionDownloader(UserConfigServer):
def get(self): def get(self):
self.post() self.post()
@@ -375,7 +361,7 @@ class FanfictionDownloader(UserConfigServer):
ch_end = mc.group('end') ch_end = mc.group('end')
if ch_begin and not mc.group('comma'): if ch_begin and not mc.group('comma'):
ch_end = ch_begin ch_end = ch_begin
logging.info("Queuing Download: %s" % url) logging.info("Queuing Download: %s" % url)
login = self.request.get('login') login = self.request.get('login')
password = self.request.get('password') password = self.request.get('password')
@@ -390,8 +376,11 @@ class FanfictionDownloader(UserConfigServer):
try: try:
try: try:
configuration = self.getUserConfig(user,url,format) configuration = self.getUserConfig(user,url,format)
except exceptions.UnknownSite:
self.redirect("/?error=custom&errtext=%s"%urllib.quote("Unsupported site in URL (%s). See 'Support sites' list below."%url,''))
return
except Exception, e: except Exception, e:
self.redirect("/?error=custom&errtext=%s"%urlEscape("There's an error in your User Configuration: "+unicode(e))) self.redirect("/?error=custom&errtext=%s"%urllib.quote("There's an error in your User Configuration: "+unicode(e),'')[:2048]) # limited due to Locatton header length limit.
return return
adapter = adapters.getAdapter(configuration,url) adapter = adapters.getAdapter(configuration,url)
@@ -551,7 +540,7 @@ class FanfictionDownloaderTask(UserConfigServer):
# delete existing chunks first # delete existing chunks first
for chunk in download.data_chunks: for chunk in download.data_chunks:
chunk.delete() chunk.delete()
index=0 index=0
while( len(data) > 0 ): while( len(data) > 0 ):
# logging.info("len(data): %s" % len(data)) # logging.info("len(data): %s" % len(data))
@@ -577,7 +566,7 @@ class FanfictionDownloaderTask(UserConfigServer):
smeta.meta = allmeta smeta.meta = allmeta
smeta.date = datetime.datetime.now() smeta.date = datetime.datetime.now()
smeta.put() smeta.put()
logging.info("Download finished OK") logging.info("Download finished OK")
del data del data
@@ -630,16 +619,6 @@ def getDownloadMeta(id=None,url=None,user=None,format=None,new=False):
return download return download
def toPercentDecimal(match):
"Return the %decimal number for the character for url escaping"
s = match.group(1)
return "%%%02x" % ord(s)
def urlEscape(data):
"Escape text, including unicode, for use in URLs"
p = re.compile(r'([^\w])')
return p.sub(toPercentDecimal, data.encode("utf-8"))
logging.getLogger().setLevel(logging.DEBUG) logging.getLogger().setLevel(logging.DEBUG)
app = webapp2.WSGIApplication([('/', MainHandler), app = webapp2.WSGIApplication([('/', MainHandler),
('/fdowntask', FanfictionDownloaderTask), ('/fdowntask', FanfictionDownloaderTask),
+1 -1
View File
@@ -45,7 +45,7 @@
<img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif" <img src="http://code.google.com/appengine/images/appengine-silver-120x30.gif"
alt="Powered by Google App Engine" /> alt="Powered by Google App Engine" />
<br/><br/> <br/><br/>
This is a web front-end to <A href="http://code.google.com/p/fanficdownloader/">FanFicFare</a><br/> This is a web front-end to <a href="https://github.com/JimmXinu/FanFicFare/">FanFicFare</a><br/>
Copyright &copy; FanFicFare team Copyright &copy; FanFicFare team
</div> </div>