Compare commits

...
25 Commits
Author SHA1 Message Date
Jim Miller f090485369 Show download count in Jobs list. 2016-02-14 10:18:03 -06:00
Jim Miller 2af3a44e11 Detected no-story for number, raise StoryDoesNotExist(SIYE). 2016-02-14 10:09:38 -06:00
Jim Miller 70d7253dac Fix some ini and highlighting issues. 2016-02-13 09:29:15 -06:00
Jim Miller be4e3610d5 Remove TtH authsoup debug dump. 2016-02-10 13:03:43 -06:00
Jim Miller 69507816a4 Allow Pairing w/o Centered category -> chars in TtH. 2016-02-10 12:46:56 -06:00
Jim Miller 20a789566f Add new AccessDenied exception for one line output in CLI. 2016-02-10 12:19:02 -06:00
Jim Miller 138de3ac3a Adding Incomplete status state to adapter_storiesonlinenet 2016-02-09 17:22:12 -06:00
Jim Miller f2960a8db4 Update translations 2016-02-09 10:35:29 -06:00
Jim Miller 3dd6550882 Fetch updated 'mi' from DB before generating covers. 2016-02-09 10:34:09 -06:00
Jim Miller 9f511dad8d Fix 'In Progress' to 'In-Progress' like all the others. 2016-02-09 10:25:43 -06:00
Jim Miller 7a08e3afdd Add automatic adding of unrecognized metadata in base_efiction. For tgstorytime.com. 2016-02-06 13:22:51 -06:00
Jim Miller a644beea94 Add (partial) translations for Estonian and Norwegian Bokmål 2016-02-05 11:50:27 -06:00
Jim Miller b275007393 Fix for portkey.org--Don't use cache on first hit in case added adult cookie. 2016-02-05 11:46:53 -06:00
Jim Miller 13602b023d Update translations. 2016-02-02 10:03:38 -06:00
Jim Miller 1be99aa95c Fix for replace_br_with_p(htmlheuristics) when author includes <>. 2016-02-02 09:58:53 -06:00
Jim Miller 4bf3399e35 Correct outdated ini comment re *_filename. 2016-02-01 14:24:01 -06:00
Jim Miller 35c066ea65 Add code for lazyload images in base_xenforoforum. 2016-02-01 14:23:13 -06:00
Jim Miller b25a869185 Fix fictionally.ord description so calibre doesn't <code> it. 2016-01-30 12:29:07 -06:00
Jim Miller 22e916bda9 Fix for html5lib handling noscript oddly, noticed with fictionalley.org. 2016-01-30 12:19:19 -06:00
Jim Miller 784375d15e Adding Word Count post-processing option, like Smarten Punct. 2016-01-29 22:34:07 -06:00
Jim Miller 18fd7d3653 Highlight color for *_format ini option 2016-01-26 14:14:41 -06:00
Jim Miller 34333a1c48 Change deprecated has_key() to has_attr() on BS objects. 2016-01-26 14:06:55 -06:00
Jim Miller 4069b1d15d Apply *_format ini option to date/time types. For calibre_* columns passed in. 2016-01-26 11:43:23 -06:00
Jim Miller b962059e4c Add byline site specific metadata for AO3. 2016-01-25 12:20:02 -06:00
Jim Miller 679fcc9d47 Fix for quotev.com change to story image. 2016-01-22 10:00:58 -06:00
97 changed files with 8444 additions and 3765 deletions
+44 -18
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2015, Jim Miller'
__copyright__ = '2016, Jim Miller'
__docformat__ = 'restructuredtext en'
import logging
@@ -81,8 +81,8 @@ no_trans = { 'pini':'personal.ini',
STD_COLS_SKIP = ['size','cover','news','ondevice','path','series_sort','sort']
from calibre_plugins.fanficfare_plugin.prefs \
import (prefs, PREFS_NAMESPACE, updatecalcover_order, calcover_save_options,
gencalcover_order, SAVE_YES, SAVE_NO)
import (prefs, PREFS_NAMESPACE, prefs_save_options, updatecalcover_order,
gencalcover_order, do_wordcount_order, SAVE_YES, SAVE_NO)
from calibre_plugins.fanficfare_plugin.dialogs \
import (UPDATE, UPDATEALWAYS, collision_order, save_collisions, RejectListDialog,
@@ -261,6 +261,7 @@ class ConfigWidget(QWidget):
prefs['checkforurlchange'] = self.basic_tab.checkforurlchange.isChecked()
prefs['injectseries'] = self.basic_tab.injectseries.isChecked()
prefs['matchtitleauth'] = self.basic_tab.matchtitleauth.isChecked()
prefs['do_wordcount'] = prefs_save_options[unicode(self.basic_tab.do_wordcount.currentText())]
prefs['smarten_punctuation'] = self.basic_tab.smarten_punctuation.isChecked()
prefs['reject_always'] = self.basic_tab.reject_always.isChecked()
@@ -286,10 +287,10 @@ class ConfigWidget(QWidget):
prefs['cal_cols_pass_in'] = self.personalini_tab.cal_cols_pass_in.isChecked()
# Covers tab
prefs['updatecalcover'] = calcover_save_options[unicode(self.calibrecover_tab.updatecalcover.currentText())]
prefs['updatecalcover'] = prefs_save_options[unicode(self.calibrecover_tab.updatecalcover.currentText())]
# for backward compatibility:
prefs['updatecover'] = prefs['updatecalcover'] == SAVE_YES
prefs['gencalcover'] = calcover_save_options[unicode(self.calibrecover_tab.gencalcover.currentText())]
prefs['gencalcover'] = prefs_save_options[unicode(self.calibrecover_tab.gencalcover.currentText())]
prefs['calibre_gen_cover'] = self.calibrecover_tab.calibre_gen_cover.isChecked()
prefs['plugin_gen_cover'] = self.calibrecover_tab.plugin_gen_cover.isChecked()
prefs['gcnewonly'] = self.calibrecover_tab.gcnewonly.isChecked()
@@ -478,6 +479,10 @@ class BasicTab(QWidget):
self.lookforurlinhtml.setChecked(prefs['lookforurlinhtml'])
self.l.addWidget(self.lookforurlinhtml)
proc_gb = groupbox = QGroupBox(_("Post Processing Options"))
self.l = QVBoxLayout()
groupbox.setLayout(self.l)
self.mark = QCheckBox(_("Mark added/updated books when finished?"),self)
self.mark.setToolTip(_("Mark added/updated books when finished. Use with option below.\nYou can also manually search for 'marked:fff_success'.\n'marked:fff_failed' is also available, or search 'marked:fff' for both."))
self.mark.setChecked(prefs['mark'])
@@ -493,6 +498,24 @@ class BasicTab(QWidget):
self.smarten_punctuation.setChecked(prefs['smarten_punctuation'])
self.l.addWidget(self.smarten_punctuation)
tooltip = _("Calculate Word Counts using Calibre internal methods.\n"
"Many sites include Word Count, but many do not.\n"
"This will count the words in each book and include it as if it came from the site.")
horz = QHBoxLayout()
label = QLabel(_('Calculate Word Count:'))
label.setToolTip(tooltip)
horz.addWidget(label)
self.do_wordcount = QComboBox(self)
for i in do_wordcount_order:
self.do_wordcount.addItem(i)
self.do_wordcount.setCurrentIndex(self.do_wordcount.findText(prefs_save_options[prefs['do_wordcount']]))
self.do_wordcount.setToolTip(tooltip)
label.setBuddy(self.do_wordcount)
horz.addWidget(self.do_wordcount)
self.l.addLayout(horz)
self.autoconvert = QCheckBox(_("Automatically Convert new/update books?"),self)
self.autoconvert.setToolTip(_("Automatically call calibre's Convert for new/update books.\nConverts to the current output format as chosen in calibre's\nPreferences->Behavior settings."))
self.autoconvert.setChecked(prefs['autoconvert'])
@@ -564,14 +587,17 @@ class BasicTab(QWidget):
horz = QHBoxLayout()
horz.addWidget(cali_gb)
vertleft = QVBoxLayout()
vertleft.addWidget(cali_gb)
vertleft.addWidget(proc_gb)
vert = QVBoxLayout()
vert.addWidget(gui_gb)
vert.addWidget(misc_gb)
vert.addWidget(rej_gb)
vertright = QVBoxLayout()
vertright.addWidget(gui_gb)
vertright.addWidget(misc_gb)
vertright.addWidget(rej_gb)
horz.addLayout(vert)
horz.addLayout(vertleft)
horz.addLayout(vertright)
topl.addLayout(horz)
topl.insertStretch(-1)
@@ -840,11 +866,11 @@ class CalibreCoverTab(QWidget):
self.updatecalcover.addItem(i)
# back compat. If has own value, use.
if prefs['updatecalcover']:
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(calcover_save_options[prefs['updatecalcover']]))
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(prefs_save_options[prefs['updatecalcover']]))
elif prefs['updatecover']: # doesn't have own val, set YES if old value set.
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(calcover_save_options[SAVE_YES]))
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(prefs_save_options[SAVE_YES]))
else: # doesn't have own value, old value not set, NO.
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(calcover_save_options[SAVE_NO]))
self.updatecalcover.setCurrentIndex(self.updatecalcover.findText(prefs_save_options[SAVE_NO]))
self.updatecalcover.setToolTip(tooltip)
label.setBuddy(self.updatecalcover)
horz.addWidget(self.updatecalcover)
@@ -862,11 +888,11 @@ class CalibreCoverTab(QWidget):
self.gencalcover.addItem(i)
# back compat. If has own value, use.
# if prefs['gencalcover']:
self.gencalcover.setCurrentIndex(self.gencalcover.findText(calcover_save_options[prefs['gencalcover']]))
self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[prefs['gencalcover']]))
# elif prefs['gencover']: # doesn't have own val, set YES if old value set.
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(calcover_save_options[SAVE_YES]))
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_YES]))
# else: # doesn't have own value, old value not set, NO.
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(calcover_save_options[SAVE_NO]))
# self.gencalcover.setCurrentIndex(self.gencalcover.findText(prefs_save_options[SAVE_NO]))
self.gencalcover.setToolTip(tooltip)
label.setBuddy(self.gencalcover)
@@ -990,7 +1016,7 @@ class CalibreCoverTab(QWidget):
## First, cover gen on/off
for e in self.gencov_elements:
e.setEnabled(calcover_save_options[unicode(self.gencalcover.currentText())] != SAVE_NO)
e.setEnabled(prefs_save_options[unicode(self.gencalcover.currentText())] != SAVE_NO)
# next, disable plugin settings when using calibre gen cov.
if not self.plugin_gen_cover.isChecked():
+30 -53
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2015, Jim Miller'
__copyright__ = '2016, Jim Miller'
__docformat__ = 'restructuredtext en'
import traceback, re
@@ -73,55 +73,30 @@ from calibre_plugins.fanficfare_plugin.fanficfare.configurable \
from inihighlighter import IniHighlighter
SKIP=_('Skip')
ADDNEW=_('Add New Book')
UPDATE=_('Update EPUB if New Chapters')
UPDATEALWAYS=_('Update EPUB Always')
OVERWRITE=_('Overwrite if Newer')
OVERWRITEALWAYS=_('Overwrite Always')
CALIBREONLY=_('Update Calibre Metadata from Web Site')
CALIBREONLYSAVECOL=_('Update Calibre Metadata from Saved Metadata Column')
collision_order=[SKIP,
ADDNEW,
UPDATE,
UPDATEALWAYS,
OVERWRITE,
OVERWRITEALWAYS,
CALIBREONLY,
CALIBREONLYSAVECOL,]
# best idea I've had for how to deal with config/pref saving the
# collision name in english.
SAVE_SKIP='Skip'
SAVE_ADDNEW='Add New Book'
SAVE_UPDATE='Update EPUB if New Chapters'
SAVE_UPDATEALWAYS='Update EPUB Always'
SAVE_OVERWRITE='Overwrite if Newer'
SAVE_OVERWRITEALWAYS='Overwrite Always'
SAVE_CALIBREONLY='Update Calibre Metadata Only'
SAVE_CALIBREONLYSAVECOL='Update Calibre Metadata Only(Saved Column)'
save_collisions={
SKIP:SAVE_SKIP,
ADDNEW:SAVE_ADDNEW,
UPDATE:SAVE_UPDATE,
UPDATEALWAYS:SAVE_UPDATEALWAYS,
OVERWRITE:SAVE_OVERWRITE,
OVERWRITEALWAYS:SAVE_OVERWRITEALWAYS,
CALIBREONLY:SAVE_CALIBREONLY,
CALIBREONLYSAVECOL:SAVE_CALIBREONLYSAVECOL,
SAVE_SKIP:SKIP,
SAVE_ADDNEW:ADDNEW,
SAVE_UPDATE:UPDATE,
SAVE_UPDATEALWAYS:UPDATEALWAYS,
SAVE_OVERWRITE:OVERWRITE,
SAVE_OVERWRITEALWAYS:OVERWRITEALWAYS,
SAVE_CALIBREONLY:CALIBREONLY,
SAVE_CALIBREONLYSAVECOL:CALIBREONLYSAVECOL,
}
anthology_collision_order=[UPDATE,
UPDATEALWAYS,
OVERWRITEALWAYS]
## moved to prefs.py so they can be included in jobs.py.
from calibre_plugins.fanficfare_plugin.prefs import \
( SAVE_YES,
SAVE_YES_UNLESS_SITE,
SKIP,
ADDNEW,
UPDATE,
UPDATEALWAYS,
OVERWRITE,
OVERWRITEALWAYS,
CALIBREONLY,
CALIBREONLYSAVECOL,
collision_order,
SAVE_SKIP,
SAVE_ADDNEW,
SAVE_UPDATE,
SAVE_UPDATEALWAYS,
SAVE_OVERWRITE,
SAVE_OVERWRITEALWAYS,
SAVE_CALIBREONLY,
SAVE_CALIBREONLYSAVECOL,
save_collisions,
anthology_collision_order,
)
gpstyle='QGroupBox {border:0; padding-top:10px; padding-bottom:0px; margin-bottom:0px;}' # background-color:red;
@@ -473,8 +448,9 @@ class AddNewDialog(SizePersistedDialog):
'updatemeta': self.updatemeta.isChecked(),
'bgmeta': False, # self.bgmeta.isChecked(),
'updateepubcover': self.updateepubcover.isChecked(),
'smarten_punctuation':self.prefs['smarten_punctuation']
}
'smarten_punctuation':self.prefs['smarten_punctuation'],
'do_wordcount':self.prefs['do_wordcount'],
}
if self.merge:
retval['fileform']=='epub'
@@ -898,7 +874,8 @@ class UpdateExistingDialog(SizePersistedDialog):
'updatemeta': self.updatemeta.isChecked(),
'bgmeta': self.bgmeta.isChecked(),
'updateepubcover': self.updateepubcover.isChecked(),
'smarten_punctuation':self.prefs['smarten_punctuation']
'smarten_punctuation':self.prefs['smarten_punctuation'],
'do_wordcount':self.prefs['do_wordcount'],
}
class StoryListTableWidget(QTableWidget):
+16 -6
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2015, Jim Miller'
__copyright__ = '2016, Jim Miller'
__docformat__ = 'restructuredtext en'
import logging
@@ -476,7 +476,8 @@ class FanFicFarePlugin(InterfaceAction):
'updatemeta': prefs['updatemeta'],
'bgmeta': False,
'updateepubcover': prefs['updateepubcover'],
'smarten_punctuation':prefs['smarten_punctuation']
'smarten_punctuation':prefs['smarten_punctuation'],
'do_wordcount':prefs['do_wordcount'],
},"\n".join(url_list))
else:
self.gui.status_bar.show_message(_('Finished Fetching Story URLs from Email.'),3000)
@@ -1307,11 +1308,14 @@ class FanFicFarePlugin(InterfaceAction):
if collision == OVERWRITE and \
db.has_format(book_id,formmapping[fileform],index_is_id=True):
logger.debug("OVERWRITE file: "+db.format_abspath(book_id, formmapping[fileform], index_is_id=True))
fileupdated=datetime.fromtimestamp(os.stat(db.format_abspath(book_id, formmapping[fileform], index_is_id=True))[8])
book['fileupdated']=fileupdated
logger.debug("OVERWRITE file updated: %s"%fileupdated)
book['updated']=fileupdated
if not bgmeta:
# check make sure incoming is newer.
lastupdated=story.getMetadataRaw('dateUpdated')
logger.debug("OVERWRITE site updated: %s"%lastupdated)
# updated doesn't have time (or is midnight), use dates only.
# updated does have time, use full timestamps.
@@ -1472,7 +1476,7 @@ class FanFicFarePlugin(InterfaceAction):
cpus = self.gui.job_manager.server.pool_size
args = ['calibre_plugins.fanficfare_plugin.jobs', 'do_download_worker',
(book_list, options, cpus, merge)]
desc = _('Download FanFiction Book')
desc = _('Download %s FanFiction Book(s)') % len(filter(lambda x : x['good'], book_list))
job = self.gui.job_manager.run_job(
self.Dispatcher(partial(self.download_list_completed,options=options,merge=merge)),
func, args=args,
@@ -2024,8 +2028,11 @@ class FanFicFarePlugin(InterfaceAction):
cover_generated = False # flag for polish below.
# Yes, should do gencov. Which?
if prefs['calibre_gen_cover'] and HAS_CALGC:
# calibre's builtin, if available.
cdata = cal_generate_cover(mi)
## calibre's builtin, if available. fetch updated mi
## object from database. Additional normalization of
## series (at least) happens
realmi = db.get_metadata(book_id, index_is_id=True)
cdata = cal_generate_cover(realmi)
db.set_cover(book_id, cdata)
cover_generated = True
elif prefs['plugin_gen_cover'] and 'Generate Cover' in self.gui.iactions:
@@ -2072,6 +2079,9 @@ class FanFicFarePlugin(InterfaceAction):
if setting_name:
logger.debug("Running Generate Cover with settings %s."%setting_name)
## fetch updated mi object from
## database. Additional normalization of series
## (at least) happens
realmi = db.get_metadata(book_id, index_is_id=True)
gc_plugin.generate_cover_for_book(realmi,saved_setting_name=setting_name)
cover_generated = True
+17 -4
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2015, Jim Miller, 2011, Grant Drake <grant.drake@gmail.com>'
__copyright__ = '2016, Jim Miller, 2011, Grant Drake <grant.drake@gmail.com>'
__docformat__ = 'restructuredtext en'
import logging
@@ -20,6 +20,9 @@ from calibre.constants import numeric_version as calibre_version
from calibre.utils.date import local_tz
from calibre.library.comments import sanitize_comments_html
from calibre_plugins.fanficfare_plugin.wordcount import get_word_count
from calibre_plugins.fanficfare_plugin.prefs import (SAVE_YES, SAVE_YES_UNLESS_SITE)
# ------------------------------------------------------------------------------
#
# Functions to perform downloads using worker jobs
@@ -148,7 +151,7 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
adapter.setChaptersRange(book['begin'],book['end'])
adapter.load_cookiejar(options['cookiejarfile'])
logger.debug("cookiejar:%s"%adapter.cookiejar)
#logger.debug("cookiejar:%s"%adapter.cookiejar)
adapter.set_pagecache(options['pagecache'])
story = adapter.getStoryMetadataOnly()
@@ -217,6 +220,7 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
logger.info("write to %s"%outfile)
inject_cal_cols(book,story,configuration)
writer.writeStory(outfilename=outfile, forceOverwrite=True)
book['comment'] = 'Download %s completed, %s chapters.'%(options['fileform'],story.getMetadata("numChapters"))
book['all_metadata'] = story.getAllMetadata(removeallentities=True)
if options['savemetacol'] != '':
@@ -271,6 +275,16 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
if options['savemetacol'] != '':
book['savemetacol'] = story.dump_html_metadata()
if options['do_wordcount'] == SAVE_YES or (
options['do_wordcount'] == SAVE_YES_UNLESS_SITE and not story.getMetadataRaw('numWords') ):
wordcount = get_word_count(outfile)
logger.info("get_word_count:%s"%wordcount)
story.setMetadata('numWords',wordcount)
writer.writeStory(outfilename=outfile, forceOverwrite=True)
book['all_metadata'] = story.getAllMetadata(removeallentities=True)
if options['savemetacol'] != '':
book['savemetacol'] = story.dump_html_metadata()
if options['smarten_punctuation'] and options['fileform'] == "epub" \
and calibre_version >= (0, 9, 39):
# for smarten punc
@@ -286,8 +300,7 @@ def do_download_for_worker(book,options,merge,notification=lambda x,y:x):
opts = O(**opts)
log = Log(level=Log.DEBUG)
# report = []
polish({outfile:outfile}, opts, log, logger.info) # report.append
polish({outfile:outfile}, opts, log, logger.info)
except NotGoingToDownload as d:
book['good']=False
+23 -5
View File
@@ -747,7 +747,7 @@ extratags: FanFiction,Testing,HTML
## AO3 adapter defines a few extra metadata entries.
## If there's ever more than 4 series, add series04,series04Url etc.
extra_valid_entries:fandoms,freeformtags,freefromtags,ao3categories,comments,kudos,hits,bookmarks,collections,series00,series01,series02,series03,series00Url,series01Url,series02Url,series03Url,series00HTML,series01HTML,series02HTML,series03HTML
extra_valid_entries:fandoms,freeformtags,freefromtags,ao3categories,comments,kudos,hits,bookmarks,collections,byline,series00,series01,series02,series03,series00Url,series01Url,series02Url,series03Url,series00HTML,series01HTML,series02HTML,series03HTML
fandoms_label:Fandoms
freeformtags_label:Freeform Tags
freefromtags_label:Freeform Tags
@@ -783,7 +783,7 @@ include_in_category:fandoms
include_in_freefromtags:freeformtags
## adds to titlepage_entries instead of replacing it.
#extra_titlepage_entries: fandoms,freeformtags,ao3categories,comments,kudos,hits,bookmarks,series01HTML,series02HTML,series03HTML
#extra_titlepage_entries: fandoms,freeformtags,ao3categories,comments,kudos,hits,bookmarks,series01HTML,series02HTML,series03HTML,byline
## adds to include_subject_tags instead of replacing it.
#extra_subject_tags:fandoms,freeformtags,ao3categories
@@ -1557,6 +1557,24 @@ extracategories:Transgender
## confirm they are adult for adult content.
#is_adult:true
## This site has a number of additional site specific metadata
## entries. This is the first test case of base_efiction 'Auto
## metadata' automatically including unrecognized metadata. Still
## requires entries in extra_valid_entries to be used.
extra_valid_entries:turnedinto,featureditems,locale,motivationforchange,sexualorientation,storytheme,bodymodification,personality,storytype,typeofchange
#add_to_titlepage_entries:,turnedinto,featureditems,locale,motivationforchange,sexualorientation,storytheme,bodymodification,personality,storytype,typeofchange
turnedinto_label:Turned Into
featureditems_label:Featured Items
locale_label:Locale
motivationforchange_label:Motivation for Change
sexualorientation_label:Sexual Orientation
storytheme_label:Story Theme
bodymodification_label:Body Modification
personality_label:Personality
storytype_label:Story Type
typeofchange_label:Type of Change
[thehexfiles.net]
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
@@ -2000,7 +2018,7 @@ extra_valid_entries:stars,reviews,reads,takesplaces,snapeflavours,sitetags
stars_label:Frogs
takesplaces_label:Takes Place
snapeflavours_label:Snape Flavour
sitetags_labels:Site Tags
sitetags_label:Site Tags
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
@@ -2036,8 +2054,8 @@ pages_label:Pages
readers_label:Readers
reads_label:Reads
favorites_label:Favorites
searchtags:Search Tags
comments:Comments
searchtags_label:Search Tags
comments_label:Comments
include_in_category:category,searchtags
+59 -3
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2015, Jim Miller'
__copyright__ = '2016, Jim Miller'
__docformat__ = 'restructuredtext en'
import logging
@@ -15,9 +15,59 @@ import copy
from calibre.utils.config import JSONConfig
from calibre.gui2.ui import get_gui
from calibre_plugins.fanficfare_plugin.dialogs import SAVE_UPDATE
from calibre_plugins.fanficfare_plugin.common_utils import get_library_uuid
SKIP=_('Skip')
ADDNEW=_('Add New Book')
UPDATE=_('Update EPUB if New Chapters')
UPDATEALWAYS=_('Update EPUB Always')
OVERWRITE=_('Overwrite if Newer')
OVERWRITEALWAYS=_('Overwrite Always')
CALIBREONLY=_('Update Calibre Metadata from Web Site')
CALIBREONLYSAVECOL=_('Update Calibre Metadata from Saved Metadata Column')
collision_order=[SKIP,
ADDNEW,
UPDATE,
UPDATEALWAYS,
OVERWRITE,
OVERWRITEALWAYS,
CALIBREONLY,
CALIBREONLYSAVECOL,]
# best idea I've had for how to deal with config/pref saving the
# collision name in english.
SAVE_SKIP='Skip'
SAVE_ADDNEW='Add New Book'
SAVE_UPDATE='Update EPUB if New Chapters'
SAVE_UPDATEALWAYS='Update EPUB Always'
SAVE_OVERWRITE='Overwrite if Newer'
SAVE_OVERWRITEALWAYS='Overwrite Always'
SAVE_CALIBREONLY='Update Calibre Metadata Only'
SAVE_CALIBREONLYSAVECOL='Update Calibre Metadata Only(Saved Column)'
save_collisions={
SKIP:SAVE_SKIP,
ADDNEW:SAVE_ADDNEW,
UPDATE:SAVE_UPDATE,
UPDATEALWAYS:SAVE_UPDATEALWAYS,
OVERWRITE:SAVE_OVERWRITE,
OVERWRITEALWAYS:SAVE_OVERWRITEALWAYS,
CALIBREONLY:SAVE_CALIBREONLY,
CALIBREONLYSAVECOL:SAVE_CALIBREONLYSAVECOL,
SAVE_SKIP:SKIP,
SAVE_ADDNEW:ADDNEW,
SAVE_UPDATE:UPDATE,
SAVE_UPDATEALWAYS:UPDATEALWAYS,
SAVE_OVERWRITE:OVERWRITE,
SAVE_OVERWRITEALWAYS:OVERWRITEALWAYS,
SAVE_CALIBREONLY:CALIBREONLY,
SAVE_CALIBREONLYSAVECOL:CALIBREONLYSAVECOL,
}
anthology_collision_order=[UPDATE,
UPDATEALWAYS,
OVERWRITEALWAYS]
# Show translated strings, but save the same string in prefs so your
# prefs are the same in different languages.
YES=_('Yes, Always')
@@ -26,9 +76,11 @@ YES_IF_IMG=_('Yes, if EPUB has a cover image')
SAVE_YES_IF_IMG='Yes, if img'
YES_UNLESS_IMG=_('Yes, unless FanFicFare found a cover image')
SAVE_YES_UNLESS_IMG='Yes, unless img'
YES_UNLESS_SITE=_('Yes, unless found on site')
SAVE_YES_UNLESS_SITE='Yes, unless site'
NO=_('No')
SAVE_NO='No'
calcover_save_options = {
prefs_save_options = {
YES:SAVE_YES,
SAVE_YES:YES,
YES_IF_IMG:SAVE_YES_IF_IMG,
@@ -37,9 +89,12 @@ calcover_save_options = {
SAVE_YES_UNLESS_IMG:YES_UNLESS_IMG,
NO:SAVE_NO,
SAVE_NO:NO,
YES_UNLESS_SITE:SAVE_YES_UNLESS_SITE,
SAVE_YES_UNLESS_SITE:YES_UNLESS_SITE,
}
updatecalcover_order=[YES,YES_IF_IMG,NO]
gencalcover_order=[YES,YES_UNLESS_IMG,NO]
do_wordcount_order=[YES,YES_UNLESS_SITE,NO]
# if don't have any settings for FanFicFarePlugin, copy from
# predecessor FanFictionDownLoaderPlugin.
@@ -78,6 +133,7 @@ default_prefs['checkforseriesurlid'] = True
default_prefs['checkforurlchange'] = True
default_prefs['injectseries'] = False
default_prefs['matchtitleauth'] = True
default_prefs['do_wordcount'] = SAVE_YES_UNLESS_SITE
default_prefs['smarten_punctuation'] = False
default_prefs['show_est_time'] = False
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+95
View File
@@ -0,0 +1,95 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2016, Jim Miller, 2011, Grant Drake <grant.drake@gmail.com>'
__docformat__ = 'restructuredtext en'
'''
A lot of this is lifted from Count Pages plugin by Grant Drake (with
some changes from davidfor.)
'''
import logging
logger = logging.getLogger(__name__)
import re
from calibre.ebooks.oeb.iterator import EbookIterator
RE_HTML_BODY = re.compile(u'<body[^>]*>(.*)</body>', re.UNICODE | re.DOTALL | re.IGNORECASE)
RE_STRIP_MARKUP = re.compile(u'<[^>]+>', re.UNICODE)
def get_word_count(book_path):
'''
Estimate a word count
'''
from calibre.utils.localization import get_lang
iterator = _open_epub_file(book_path)
lang = iterator.opf.language
lang = get_lang() if not lang else lang
count = _get_epub_standard_word_count(iterator, lang)
return count
def _open_epub_file(book_path, strip_html=False):
'''
Given a path to an EPUB file, read the contents into a giant block of text
'''
iterator = EbookIterator(book_path)
iterator.__enter__(only_input_plugin=True, run_char_count=True,
read_anchor_map=False)
return iterator
def _get_epub_standard_word_count(iterator, lang='en'):
'''
This algorithm counts individual words instead of pages
'''
book_text = _read_epub_contents(iterator, strip_html=True)
try:
from calibre.spell.break_iterator import count_words
wordcount = count_words(book_text, lang)
logger.debug('\tWord count - count_words method:%s'%wordcount)
except:
try: # The above method is new and no-one will have it as of 08/01/2016. Use an older method for a beta.
from calibre.spell.break_iterator import split_into_words_and_positions
wordcount = len(split_into_words_and_positions(book_text, lang))
logger.debug('\tWord count - split_into_words_and_positions method:%s'%wordcount)
except:
from calibre.utils.wordcount import get_wordcount_obj
wordcount = get_wordcount_obj(book_text)
wordcount = wordcount.words
logger.debug('\tWord count - old method:%s'%wordcount)
return wordcount
def _read_epub_contents(iterator, strip_html=False):
'''
Given an iterator for an ePub file, read the contents into a giant block of text
'''
book_files = []
for path in iterator.spine:
with open(path, 'rb') as f:
html = f.read().decode('utf-8', 'replace')
if strip_html:
html = unicode(_extract_body_text(html)).strip()
#print('FOUND HTML:', html)
book_files.append(html)
return ''.join(book_files)
def _extract_body_text(data):
'''
Get the body text of this html content wit any html tags stripped
'''
body = RE_HTML_BODY.findall(data)
if body:
return RE_STRIP_MARKUP.sub('', body[0]).replace('.','. ')
return ''
@@ -190,6 +190,10 @@ class ArchiveOfOurOwnOrgAdapter(BaseSiteAdapter):
self.story.addToList('authorUrl',a['href'])
self.story.addToList('author',a.text)
byline = metasoup.find('h3',{'class':'byline'})
if byline:
self.story.setMetadata('byline',stripHTML(byline))
newestChapter = None
self.newestChapterNum = None # save for comparing during update.
# Scan all chapters to find the oldest and newest, on AO3 it's
@@ -131,7 +131,7 @@ class AshwinderSycophantHexComAdapter(BaseSiteAdapter):
data = self._fetchUrl(url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -119,7 +119,7 @@ class Asr3SlashzoneOrgAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -198,7 +198,7 @@ class BloodTiesFansComAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -106,7 +106,7 @@ class ChaosSycophantHexComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -105,7 +105,7 @@ class CSIForensicsComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -122,7 +122,7 @@ class DestinysGatewayComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -147,7 +147,7 @@ class DokugaComAdapter(BaseSiteAdapter):
soup = self.make_soup(data)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# print data
# Now go hunting for all the meta data and the chapter list.
@@ -161,7 +161,7 @@ class DracoAndGinnyComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -146,7 +146,7 @@ class DramioneOrgAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -91,7 +91,7 @@ class EfictionEstelielDeAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# Now go hunting for all the meta data and the chapter list.
+1 -1
View File
@@ -127,7 +127,7 @@ class EFPFanFicNet(BaseSiteAdapter):
data = self._fetchUrl(url)
# if "Access denied. This story has not been validated by the adminstrators of this site." in data:
# raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -122,7 +122,7 @@ class ErosnSapphoSycophantHexComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -177,7 +177,7 @@ class FanficCastleTVNetAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -170,7 +170,7 @@ class FanfictionJunkiesDeAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -171,7 +171,7 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
if head.find('span',title='Fertiggestellt'):
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In Progress')
self.story.setMetadata('status', 'In-Progress')
#find metadata on the author's page
asoup = self.make_soup(self._fetchUrl("http://"+self.getSiteDomain()+"?a=q&a1=v&t=nickdetailsstories&lbi=stories&ar=0&nick="+self.story.getMetadata('authorId')))
+1 -1
View File
@@ -193,7 +193,7 @@ class FicBookNetAdapter(BaseSiteAdapter):
if table.find('span', {'style' : 'color: green'}):
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In Progress')
self.story.setMetadata('status', 'In-Progress')
tags = table.findAll('b')
@@ -188,6 +188,7 @@ class FictionAlleyOrgSiteAdapter(BaseSiteAdapter):
for small in storydd.findAll('small'):
small.extract() ## removes the <small> tags, leaving only the summary.
storydd.name = 'div' ## change tag name else Calibre treats it oddly.
self.setDescription(url,storydd)
#self.story.setMetadata('description',stripHTML(storydd))
@@ -203,11 +204,14 @@ class FictionAlleyOrgSiteAdapter(BaseSiteAdapter):
# Yes, it's an evil kludge, but what can ya do? Using
# something other than div prevents soup from pairing
# our div with poor html inside the story text.
data = data.replace('<!-- headerend -->','<crazytagstringnobodywouldstumbleonaccidently id="storytext">').replace('<!-- footerstart -->','</crazytagstringnobodywouldstumbleonaccidently>')
crazy = "crazytagstringnobodywouldstumbleonaccidently"
data = data.replace('<!-- headerend -->','<'+crazy+' id="storytext">').replace('<!-- footerstart -->','</'+crazy+'>')
# problems with some stories confusing Soup. This is a nasty
# hack, but it works.
data = data[data.index("<crazytagstringnobodywouldstumbleonaccidently"):]
data = data[data.index('<'+crazy+''):]
# ditto with extra crap at the end.
data = data[:data.index('</'+crazy+'>')+len('</'+crazy+'>')]
soup = self.make_soup(data)
body = soup.findAll('body') ## some stories use a nested body and body
@@ -218,7 +222,7 @@ class FictionAlleyOrgSiteAdapter(BaseSiteAdapter):
text = body[1]
text.name='div' # force to be a div to avoid multiple body tags.
else:
text = soup.find('crazytagstringnobodywouldstumbleonaccidently', {'id' : 'storytext'})
text = soup.find(crazy, {'id' : 'storytext'})
text.name='div' # change to div tag.
if not data or not text:
@@ -126,7 +126,7 @@ class FineStoriesComAdapter(BaseSiteAdapter):
data = self._fetchUrl(url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -173,7 +173,7 @@ class GrangerEnchantedCom(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -86,7 +86,7 @@ class HarryPotterFanFictionComSiteAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -90,7 +90,7 @@ class HLFictionNetAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -91,7 +91,7 @@ class HPFanficArchiveComAdapter(BaseSiteAdapter):
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -173,7 +173,7 @@ class IkEternalNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -161,7 +161,7 @@ class ImagineEFicComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -148,7 +148,7 @@ class KSArchiveComAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -106,7 +106,7 @@ class LumosSycophantHexComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -554,7 +554,6 @@ class Chapter(object):
def _parseRatingFromImage(self, element):
"""Given an image element, try to parse story rating from it."""
# Although deprecated, `has_key()' is required here.
if not element.has_attr('src'):
return
source = element['src']
@@ -161,7 +161,7 @@ class MerlinFicDtwinsCoUk(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -144,7 +144,7 @@ class MidnightwhispersCaAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -184,7 +184,7 @@ class MuggleNetComAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -95,7 +95,7 @@ class NationalLibraryNetAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -94,7 +94,7 @@ class NCISFicComAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -100,7 +100,7 @@ class NCISFictionNetAdapter(BaseSiteAdapter):
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -148,7 +148,7 @@ class NfaCommunityComAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -96,7 +96,7 @@ class NickAndGregNetAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -131,7 +131,7 @@ class OcclumencySycophantHexComAdapter(BaseSiteAdapter):
data = self._fetchUrl(url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -150,7 +150,7 @@ class OneDirectionFanfictionComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -171,7 +171,7 @@ class PommeDeSangComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -127,7 +127,7 @@ class PonyFictionArchiveNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -109,7 +109,7 @@ class PortkeyOrgAdapter(BaseSiteAdapter): # XXX
self.get_cookiejar().set_cookie(cookie)
try:
data = self._fetchUrl(url)
data = self._fetchUrl(url,usecache=False)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
@@ -77,7 +77,7 @@ class PotionsAndSnitchesOrgSiteAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -170,7 +170,7 @@ class PotterHeadsAnonymousComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -125,7 +125,7 @@ class PretenderCenterComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -121,7 +121,7 @@ class PsychFicComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -122,7 +122,7 @@ class QafFicComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+3 -1
View File
@@ -74,7 +74,9 @@ class QuotevComAdapter(BaseSiteAdapter):
self.story.setMetadata('authorId','0')
self.setDescription(self.url, soup.find('div', id='qdesct'))
self.setCoverImage(self.url, urlparse.urljoin(self.url, soup.find('img', {'class': 'logo'})['src']))
imgmeta = soup.find('meta',{'property':"og:image" })
if imgmeta:
self.setCoverImage(self.url, urlparse.urljoin(self.url, imgmeta['content']))
for a in soup.find_all('a', {'href': re.compile(SITE_DOMAIN+'/stories/c/')}):
self.story.addToList('category', a.get_text())
+1 -1
View File
@@ -198,7 +198,7 @@ class SamAndJackNetAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -94,7 +94,7 @@ class SamDeanArchiveNuAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -170,7 +170,7 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -130,7 +130,7 @@ class ScarvesAndCoffeeNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -169,7 +169,7 @@ class SheppardWeirComAdapter(BaseSiteAdapter): # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -121,7 +121,7 @@ class SinfulDesireOrgAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+2
View File
@@ -103,6 +103,8 @@ class SiyeCoUkAdapter(BaseSiteAdapter): # XXX
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+"))
if a is None:
raise exceptions.StoryDoesNotExist(self.url)
self.story.setMetadata('authorId',a['href'].split('=')[1])
self.story.setMetadata('authorUrl','http://'+self.host+'/siye/'+a['href'])
self.story.setMetadata('author',a.string)
@@ -139,7 +139,7 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
data = self._fetchUrl(url+":i",usecache=False)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
elif "Error! The story you're trying to access is being filtered by your choice of contents filtering." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Error! The story you're trying to access is being filtered by your choice of contents filtering.")
@@ -318,7 +318,10 @@ class StoriesOnlineNetAdapter(BaseSiteAdapter):
status = lc4.find('span', {'class' : 'ab'})
if status != None:
self.story.setMetadata('status', 'In-Progress')
if 'Incomplete and Inactive' in status.text:
self.story.setMetadata('status', 'Incomplete')
else:
self.story.setMetadata('status', 'In-Progress')
if "Last Activity" in status.text:
# date is passed as a timestamp and converted in JS.
value = status.findNext('noscript').text
@@ -134,7 +134,7 @@ class TenhawkPresentsComSiteAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -90,7 +90,7 @@ class TheAlphaGateComAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -168,7 +168,7 @@ class TheMasqueNetAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -129,7 +129,7 @@ class ThePetulantPoetessComAdapter(BaseSiteAdapter):
data = self._fetchUrl(url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -122,7 +122,7 @@ class TokraFandomnetComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -165,7 +165,7 @@ class TrekiverseOrgAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+9 -3
View File
@@ -172,17 +172,22 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
descurl=authorurl
authorsoup = self.make_soup(authordata)
# author can have several pages, scan until we find it.
while( not authorsoup.find('a', href=re.compile(r"^/Story-"+self.story.getMetadata('storyId')+'/')) ):
# find('a', href=re.compile(r"^/Story-"+self.story.getMetadata('storyId')+'/')) ):
#logger.info("authsoup:%s"%authorsoup)
while( not authorsoup.find('div', {'id':'st'+self.story.getMetadata('storyId'), 'class':re.compile(r"storylistitem")}) ):
nextarrow = authorsoup.find('a', {'class':'arrowf'})
if not nextarrow:
## if rating is set lower than story, it won't be
## visible on author lists unless. The *story* is
## visible via the url, just not the entry on
## author list.
raise exceptions.AdultCheckRequired(self.url)
logger.info("Story Not Found on Author List--Assuming needs Adult.")
raise exceptions.FailedToDownload("Story Not Found on Author List--Assume needs Adult?")
# raise exceptions.AdultCheckRequired(self.url)
nextpage = 'http://'+self.host+nextarrow['href']
logger.debug("**AUTHOR** nextpage URL: "+nextpage)
authordata = self._fetchUrl(nextpage)
#logger.info("authsoup:%s"%authorsoup)
descurl=nextpage
authorsoup = self.make_soup(authordata)
except urllib2.HTTPError, e:
@@ -258,7 +263,8 @@ class TwistingTheHellmouthSiteAdapter(BaseSiteAdapter):
elif self.getConfig('pairingcat_to_characters_ships') and cat.string.startswith('Pairing: '):
pair = cat.string[len('Pairing: '):]
self.story.addToList('characters',pair)
self.story.addToList('ships',char+'/'+pair)
if char:
self.story.addToList('ships',char+'/'+pair)
elif cat.string not in ['General', 'Non-BtVS/AtS Stories', 'Non-BTVS/AtS Stories', 'BtVS/AtS Non-Crossover', 'Non-BtVS Crossovers']:
# assumed only ship category after Romance cat.
if self.getConfig('romancecat_to_characters_ships') and romance:
@@ -127,7 +127,7 @@ class TheWritersCoffeeShopComSiteAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# problems with some stories, but only in calibre. I suspect
# issues with different SGML parsers in python. This is a
@@ -256,9 +256,9 @@ class TheWritersCoffeeShopComSiteAdapter(BaseSiteAdapter):
found=False
for div in soup.findAll('div'):
if div.has_key('class') and div['class'] == 'notes':
if div.has_attr('class') and div['class'] == 'notes':
chapter.append(div)
if div.has_key('id') and div['id'] == 'story':
if div.has_attr('id') and div['id'] == 'story':
chapter.append(div)
found=True
@@ -90,7 +90,7 @@ class TwilightArchivesComAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -118,7 +118,7 @@ class TwilightedNetSiteAdapter(BaseSiteAdapter):
data = self._fetchUrl(url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# problems with some stories, but only in calibre. I suspect
# issues with different SGML parsers in python. This is a
@@ -106,7 +106,7 @@ class WalkingThePlankOrgAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
@@ -90,7 +90,7 @@ class WolverineAndRogueComAdapter(BaseSiteAdapter):
raise e
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+1 -1
View File
@@ -102,7 +102,7 @@ class WraithBaitComAdapter(BaseSiteAdapter):
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
raise exceptions.AccessDenied(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
+13 -2
View File
@@ -573,7 +573,7 @@ class BaseSiteAdapter(Configurable):
acceptable_attributes.extend(('src','alt','longdesc'))
for img in soup.findAll('img'):
# some pre-existing epubs have img tags that had src stripped off.
if img.has_key('src'):
if img.has_attr('src'):
(img['src'],img['longdesc'])=self.story.addImgUrl(url,img['src'],fetch,
coverexclusion=self.getConfig('cover_exclusion_regexp'))
@@ -640,10 +640,21 @@ class BaseSiteAdapter(Configurable):
Convenience method for getting a bs4 soup. Older and
non-updated adapters call the included bs3 library themselves.
'''
## html5lib handles <noscript> oddly. See:
## https://bugs.launchpad.net/beautifulsoup/+bug/1277464
## This should 'hide' and restore <noscript> tags.
data = data.replace("noscript>","fff_hide_noscript>")
## soup and re-soup because BS4/html5lib is more forgiving of
## incorrectly nested tags that way.
soup = bs4.BeautifulSoup(data,'html5lib')
return bs4.BeautifulSoup(unicode(soup),'html5lib')
soup = bs4.BeautifulSoup(unicode(soup),'html5lib')
for ns in soup.find_all('fff_hide_noscript'):
ns.name = 'noscript'
return soup
def cachedfetch(realfetch,cache,url):
if url in cache:
+7 -1
View File
@@ -314,7 +314,13 @@ class BaseEfictionAdapter(BaseSiteAdapter):
## TODO is not a link in the printable view, so no seriesURL possible
self.story.setMetadata('series', value)
else:
logger.info("Unhandled metadata pair: '%s' : '%s'" % (key, value))
# Any other metadata found, convert label to lower case
# w/o spaces and use as key. Still needs to be in
# extra_valid_entries to be used.
autokey = key.replace(' ','').lower()
for val in re.split("\s*,\s*", value):
self.story.addToList(autokey, val)
logger.debug("Auto metadata: entry:%s %s_label:%s value:%s" % (autokey, autokey, key, value))
def extractChapterUrlsAndMetadata(self):
printUrl = self.url + '&action=printable&textsize=0&chapter='
@@ -310,6 +310,11 @@ class BaseXenForoForumAdapter(BaseSiteAdapter):
for qdiv in bq.find_all('div',{'class':'quoteExpand'}):
qdiv.extract() # Remove <div class="quoteExpand">click to expand</div>
## img alt="[IMG]" class="bbCodeImage LbImage lazyload
## include lazy load images.
for img in bq.find_all('img',{'class':'lazyload'}):
img['src'] = img['data-src']
except Exception as e:
if self.getConfig('continue_on_chapter_error'):
@@ -0,0 +1,345 @@
# -*- coding: utf-8 -*-
# Copyright 2015 FanFicFare team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
import traceback
logger = logging.getLogger(__name__)
import re
import urllib2
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
logger = logging.getLogger(__name__)
class BaseXenForoForumAdapter(BaseSiteAdapter):
def __init__(self, config, url):
#logger.info("init url: "+url)
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["utf8",
"Windows-1252"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2])
# get storyId from url--url validation guarantees query correct
m = re.match(self.getSiteURLPattern(),url)
if m:
#logger.debug("groupdict:%s"%m.groupdict())
if m.group('post'):
self.story.setMetadata('storyId',m.group('post'))
self._setURL(self.getURLPrefix() + '/posts/'+m.group('post')+'/')
else:
self.story.setMetadata('storyId',m.group('id'))
# normalized story URL.
self._setURL(self.getURLPrefix() + '/'+m.group('tp')+'/'+self.story.getMetadata('storyId')+'/')
else:
raise exceptions.InvalidStoryURL(url,
self.getSiteDomain(),
self.getSiteExampleURLs())
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','fsb')
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%b %d, %Y at %I:%M %p"
@classmethod
def getConfigSections(cls):
"Only needs to be overriden if has additional ini sections."
return ['base_xenforoforum',cls.getConfigSection()]
@classmethod
def getURLPrefix(cls):
# The site domain. Does have www here, if it uses it.
return 'https://' + cls.getSiteDomain()
@classmethod
def getSiteExampleURLs(cls):
return cls.getURLPrefix()+"/threads/some-story-name.123456/ "+cls.getURLPrefix()+"/posts/123456/"
def getSiteURLPattern(self):
return r"https?://"+re.escape(self.getSiteDomain())+r"/(?P<tp>threads|posts)/(.+\.)?(?P<id>\d+)/?[^#]*?(#post-(?P<post>\d+))?$"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def performLogin(self):
params = {}
if self.password:
params['login'] = self.username
params['password'] = self.password
else:
params['login'] = self.getConfig("username")
params['password'] = self.getConfig("password")
params['register'] = '0'
params['cookie_check'] = '1'
params['_xfToken'] = ''
params['redirect'] = 'https://' + self.getSiteDomain() + '/'
if not params['password']:
return
## https://forum.questionablequesting.com/login/login
loginUrl = 'https://' + self.getSiteDomain() + '/login/login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['login']))
# soup = self.make_soup(self._fetchUrl(loginUrl))
# params['ctkn']=soup.find('input', {'name':'ctkn'})['value']
# params[soup.find('input', {'id':'password'})['name']] = params['password']
d = self._fetchUrl(loginUrl, params)
if "Log Out" not in d :
logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['login']))
raise exceptions.FailedToLogin(self.url,params['login'])
return False
else:
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
useurl = self.url
logger.info("url: "+useurl)
try:
(data,opened) = self._fetchUrlOpened(useurl)
useurl = opened.geturl()
logger.info("use useurl: "+useurl)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
elif e.code == 403:
self.performLogin()
(data,opened) = self._fetchUrlOpened(useurl)
useurl = opened.geturl()
logger.info("use useurl: "+useurl)
else:
raise
# use BeautifulSoup HTML parser to make everything easier to find.
soup = self.make_soup(data)
a = soup.find('h3',{'class':'userText'}).find('a')
self.story.addToList('authorId',a['href'].split('/')[1])
self.story.addToList('authorUrl',self.getURLPrefix()+'/'+a['href'])
self.story.addToList('author',a.text)
h1 = soup.find('div',{'class':'titleBar'}).h1
self.story.setMetadata('title',stripHTML(h1))
if '#' in useurl:
anchorid = useurl.split('#')[1]
soup = soup.find('li',id=anchorid)
else:
# try threadmarks if no '#' in url
# Use a post specific URL to get first post without using threadmarks.
threadmarksa = soup.find('a',{'class':'threadmarksTrigger'})
if threadmarksa:
soupmarks = self.make_soup(self._fetchUrl(self.getURLPrefix()+'/'+threadmarksa['href']))
markas = []
ol = soupmarks.find('ol',{'class':'overlayScroll'})
if ol:
markas = ol.find_all('a')
else:
## SV changed their threadmarks. Not isolated to
## SV only incase SB or QQ make the same change.
markas = soupmarks.find('div',{'class':'threadmarks'}).find_all('a',{'class':'PreviewTooltip'})
if markas:
self.chapterUrls.append(("First Post",useurl))
for (atag,url,name) in [ (x,x['href'],stripHTML(x)) for x in markas ]:
date = self.make_date(atag.find_next_sibling('div',{'class':'extra'}))
if not self.story.getMetadataRaw('datePublished') or date < self.story.getMetadataRaw('datePublished'):
self.story.setMetadata('datePublished', date)
if not self.story.getMetadataRaw('dateUpdated') or date > self.story.getMetadataRaw('dateUpdated'):
self.story.setMetadata('dateUpdated', date)
self.chapterUrls.append((name,self.getURLPrefix()+'/'+url))
# https://forums.spacebattles.com/threads/aberration-worm-d-d.369992/
if re.match(rr"https?://"+re.escape(self.getSiteDomain())+r"/(?P<tp>threads|posts)/(.+\.)?(?P<id>\d+)/?[^#]*?$",url)
m = re.match(,url)
if url == useurl and 'First Post' == self.chapterUrls[0][0]:
# remove "First Post" if included in list.
logger.debug("delete dup 'First Post' chapter: %s %s"%self.chapterUrls[0])
del self.chapterUrls[0]
## only use tags if threadmarks for chapters.
## a bit arbitrary, but likely.
for tag in soup.findAll('a',{'class':'tag'}):
tstr = stripHTML(tag)
if self.getConfig('capitalize_forumtags'):
tstr = tstr.title()
self.story.addToList('forumtags',tstr)
soup = soup.find('li',{'class':'message'}) # limit first post for date stuff below. ('#' posts above)
# Now go hunting for the 'chapter list'.
bq = soup.find('blockquote') # assume first posting contains TOC urls.
bq.name='div'
for iframe in bq.find_all('iframe'):
iframe.extract() # calibre book reader & editor don't like iframes to youtube.
for qdiv in bq.find_all('div',{'class':'quoteExpand'}):
qdiv.extract() # Remove <div class="quoteExpand">click to expand</div>
self.setDescription(useurl,bq)
# otherwise, use first post links--include first post since
# that's often also the first chapter.
if not self.chapterUrls:
self.chapterUrls.append(("First Post",useurl))
for (url,name) in [ (x['href'],stripHTML(x)) for x in bq.find_all('a') ]:
#logger.debug("found chapurl:%s"%url)
if not url.startswith('http'):
url = self.getURLPrefix()+'/'+url
if ( url.startswith(self.getURLPrefix()) or
url.startswith('http://'+self.getSiteDomain()) or
url.startswith('https://'+self.getSiteDomain()) ) and ('/posts/' in url or '/threads/' in url):
# brute force way to deal with SB's http->https change when hardcoded http urls.
url = url.replace('http://'+self.getSiteDomain(),self.getURLPrefix())
url = re.sub(r'(^[\'"]+|[\'"]+$)','',url) # strip leading or trailing '" from incorrect quoting.
url = re.sub(r'like$','',url) # strip 'like' if incorrect 'like' link instead of proper post URL.
logger.debug("(ch:%s)used chapurl:%s"%(len(self.chapterUrls)+1,url))
self.chapterUrls.append((name,url))
if url == useurl and 'First Post' == self.chapterUrls[0][0]:
# remove "First Post" if included in list.
logger.debug("delete dup 'First Post' chapter: %s %s"%self.chapterUrls[0])
del self.chapterUrls[0]
# Didn't use threadmarks, so take created/updated dates
# from the 'first' posting created and updated.
date = self.make_date(soup.find('a',{'class':'datePermalink'}))
if date:
self.story.setMetadata('datePublished', date)
self.story.setMetadata('dateUpdated', date) # updated overwritten below if found.
date = self.make_date(soup.find('div',{'class':'editDate'}))
if date:
self.story.setMetadata('dateUpdated', date)
self.story.setMetadata('numChapters',len(self.chapterUrls))
def make_date(self,parenttag): # forums use a BS thing where dates
# can appear different if recent.
datestr=None
try:
datetag = parenttag.find('span',{'class':'DateTime'})
if datetag:
datestr = datetag['title']
else:
datetag = parenttag.find('abbr',{'class':'DateTime'})
if datetag:
datestr="%s at %s"%(datetag['data-datestring'],datetag['data-timestring'])
# Apr 24, 2015 at 4:39 AM
# May 1, 2015 at 5:47 AM
datestr = re.sub(r' (\d[^\d])',r' 0\1',datestr) # add leading 0 for single digit day & hours.
return makeDate(datestr, self.dateformat)
except:
logger.debug('No date found in %s'%parenttag)
return None
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
## there's some history of stories with links to the wrong
## page. This changes page#post URLs to perma-link URLs.
## Which will be redirected back to page#posts, but the
## *correct* ones.
# http://forums.sufficientvelocity.com/threads/harry-potter-and-the-not-fatal-at-all-cultural-exchange-program.330/page-4#post-39915
# https://forums.sufficientvelocity.com/posts/39915/
if '#post-' in url:
url = self.getURLPrefix()+'/posts/'+url.split('#post-')[1]+'/'
## Same as above except for for case where author mistakenly
## used the reply link instead of normal link to post.
# "http://forums.spacebattles.com/threads/manager-worm-story-thread-iv.301602/reply?quote=15962513"
# https://forums.spacebattles.com/posts/
if 'reply?quote=' in url:
url = self.getURLPrefix()+'/posts/'+url.split('reply?quote=')[1]+'/'
try:
origurl = url
(data,opened) = self._fetchUrlOpened(url)
url = opened.geturl()
if '#' in origurl and '#' not in url:
url = url + origurl[origurl.index('#'):]
logger.debug("chapter URL redirected to: %s"%url)
soup = self.make_soup(data)
if '#' in url:
anchorid = url.split('#')[1]
soup = soup.find('li',id=anchorid)
bq = soup.find('blockquote')
bq.name='div'
for iframe in bq.find_all('iframe'):
iframe.extract() # calibre book reader & editor don't like iframes to youtube.
for qdiv in bq.find_all('div',{'class':'quoteExpand'}):
qdiv.extract() # Remove <div class="quoteExpand">click to expand</div>
except Exception as e:
if self.getConfig('continue_on_chapter_error'):
bq = self.make_soup("""<div>
<p><b>Error</b></p>
<p>FanFicFare failed to download this chapter. Because you have
<b>continue_on_chapter_error</b> set to <b>true</b> in your personal.ini, the download continued.</p>
<p>Chapter URL:<br>%s</p>
<p>Error:<br><pre>%s</pre></p>
</div>"""%(url,traceback.format_exc()))
else:
raise
# XenForo uses <base href="https://forums.spacebattles.com/" />
return self.utf8FromSoup(self.getURLPrefix()+'/',bq)
def normalize_forum_url(url):
'''
Take URLs of form:
https://forums.spacebattles.com/threads/heredity-ii-worm-au.311819/page-72#post-20211500
"/threads/some-story-name.123456/ "+cls.getURLPrefix()+"/posts/123456/"
'''
+5 -3
View File
@@ -365,12 +365,14 @@ def do_download(arg,
del adapter
except exceptions.InvalidStoryURL, isu:
except exceptions.InvalidStoryURL as isu:
print isu
except exceptions.StoryDoesNotExist, dne:
except exceptions.StoryDoesNotExist as dne:
print dne
except exceptions.UnknownSite, us:
except exceptions.UnknownSite as us:
print us
except exceptions.AccessDenied as ad:
print ad
if __name__ == '__main__':
+15 -6
View File
@@ -116,6 +116,11 @@ def get_valid_list_entries():
])
boollist=['true','false']
base_xenforo_list=['base_xenforoforum',
'forums.spacebattles.com',
'forums.sufficientvelocity.com',
'questionablequesting.com',
]
def get_valid_set_options():
'''
dict() of names of boolean options, but as a tuple with
@@ -155,6 +160,9 @@ def get_valid_set_options():
'non_breaking_spaces':(['fictionmania.tv'],None,boollist),
'universe_as_series':(['storiesonline.net'],None,boollist),
'strip_text_links':(['bloodshedverse.com'],None,boollist),
'centeredcat_to_characters':(['tthfanfic.org'],None,boollist),
'pairingcat_to_characters_ships':(['tthfanfic.org'],None,boollist),
'romancecat_to_characters_ships':(['tthfanfic.org'],None,boollist),
# eFiction Base adapters allow bulk_load
# kept forgetting to add them, so now it's automatic.
@@ -169,11 +177,8 @@ def get_valid_set_options():
'grayscale_images':(None,['epub','html'],boollist),
'no_image_processing':(None,['epub','html'],boollist),
'continue_on_chapter_error':(['base_xenforoforum',
'forums.spacebattles.com',
'forums.sufficientvelocity.com',
'questionablequesting.com',
],None,boollist),
'continue_on_chapter_error':(base_xenforo_list,None,boollist),
'':(base_xenforo_list,None,boollist),
}
return dict(valdict)
@@ -310,6 +315,9 @@ def get_valid_keywords():
'strip_chapter_numbers',
'strip_chapter_numeral',
'strip_text_links',
'centeredcat_to_characters',
'pairingcat_to_characters_ships',
'romancecat_to_characters_ships',
'titlepage_end',
'titlepage_entries',
'titlepage_entry',
@@ -332,11 +340,12 @@ def get_valid_keywords():
'zip_filename',
'zip_output',
'continue_on_chapter_error',
'capitalize_forumtags',
])
# *known* entry keywords -- or rather regexps for them.
def get_valid_entry_keywords():
return list(['%s_label',
return list(['%s_(label|format)',
'(default_value|include_in|join_string|keep_in_order)_%s',])
# Moved here for test_config.
+25 -7
View File
@@ -124,7 +124,7 @@ include_tocpage: true
#website_encodings: auto, utf8, Windows-1252
## python string Template, string with ${title}, ${author} etc, same as titlepage_entries
## Can include directories. ${formatext} will be added if not in filename somewhere.
## Can include directories.
#output_filename: books/${title}-${siteabbrev}_${storyId}${formatext}
#output_filename: books/${formatname}/${siteabbrev}/${authorId}/${title}-${siteabbrev}_${storyId}${formatext}
output_filename: ${title}-${siteabbrev}_${storyId}${formatext}
@@ -140,7 +140,7 @@ make_directories: true
## put output (with output_filename) in a zip file zip_filename.
zip_output: false
## Can include directories. .zip will be added if not in name somewhere
## Can include directories.
zip_filename: ${title}-${siteabbrev}_${storyId}${formatext}.zip
## Normally, try to make the filenames 'safe' by removing invalid
@@ -753,7 +753,7 @@ extratags: FanFiction,Testing,HTML
## AO3 adapter defines a few extra metadata entries.
## If there's ever more than 4 series, add series04,series04Url etc.
extra_valid_entries:fandoms,freeformtags,freefromtags,ao3categories,comments,kudos,hits,bookmarks,collections,series00,series01,series02,series03,series00Url,series01Url,series02Url,series03Url,series00HTML,series01HTML,series02HTML,series03HTML
extra_valid_entries:fandoms,freeformtags,freefromtags,ao3categories,comments,kudos,hits,bookmarks,collections,byline,series00,series01,series02,series03,series00Url,series01Url,series02Url,series03Url,series00HTML,series01HTML,series02HTML,series03HTML
fandoms_label:Fandoms
freeformtags_label:Freeform Tags
freefromtags_label:Freeform Tags
@@ -789,7 +789,7 @@ include_in_category:fandoms
include_in_freefromtags:freeformtags
## adds to titlepage_entries instead of replacing it.
#extra_titlepage_entries: fandoms,freeformtags,ao3categories,comments,kudos,hits,bookmarks,series01HTML,series02HTML,series03HTML
#extra_titlepage_entries: fandoms,freeformtags,ao3categories,comments,kudos,hits,bookmarks,series01HTML,series02HTML,series03HTML,byline
## adds to include_subject_tags instead of replacing it.
#extra_subject_tags:fandoms,freeformtags,ao3categories
@@ -1545,6 +1545,24 @@ extracategories:Transgender
## confirm they are adult for adult content.
#is_adult:true
## This site has a number of additional site specific metadata
## entries. This is the first test case of base_efiction 'Auto
## metadata' automatically including unrecognized metadata. Still
## requires entries in extra_valid_entries to be used.
extra_valid_entries:turnedinto,featureditems,locale,motivationforchange,sexualorientation,storytheme,bodymodification,personality,storytype,typeofchange
#add_to_titlepage_entries:,turnedinto,featureditems,locale,motivationforchange,sexualorientation,storytheme,bodymodification,personality,storytype,typeofchange
turnedinto_label:Turned Into
featureditems_label:Featured Items
locale_label:Locale
motivationforchange_label:Motivation for Change
sexualorientation_label:Sexual Orientation
storytheme_label:Story Theme
bodymodification_label:Body Modification
personality_label:Personality
storytype_label:Story Type
typeofchange_label:Type of Change
[thehexfiles.net]
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
@@ -1982,7 +2000,7 @@ extra_valid_entries:stars,reviews,reads,takesplaces,snapeflavours,sitetags
stars_label:Frogs
takesplaces_label:Takes Place
snapeflavours_label:Snape Flavour
sitetags_labels:Site Tags
sitetags_label:Site Tags
## Site dedicated to these categories/characters/ships
extracategories:Harry Potter
@@ -2018,8 +2036,8 @@ pages_label:Pages
readers_label:Readers
reads_label:Reads
favorites_label:Favorites
searchtags:Search Tags
comments:Comments
searchtags_label:Search Tags
comments_label:Comments
include_in_category:category,searchtags
+7
View File
@@ -24,6 +24,13 @@ class FailedToDownload(Exception):
def __str__(self):
return self.error
class AccessDenied(Exception):
def __init__(self,error):
self.error=error
def __str__(self):
return self.error
class RejectImage(Exception):
def __init__(self,error):
self.error=error
+1 -1
View File
@@ -92,7 +92,7 @@ def get_urls_from_html(data,url=None,configuration=None,normalize=False,restrict
#logger.debug("restrict search:%s"%soup)
for a in soup.findAll('a'):
if a.has_key('href'):
if a.has_attr('href'):
#logger.debug("a['href']:%s"%a['href'])
href = form_url(url,a['href'])
#logger.debug("1 urlhref:%s"%href)
+8 -8
View File
@@ -190,16 +190,16 @@ class _html2text(sgmllib.SGMLParser):
If the set of attributes is not found, returns None
"""
if not attrs.has_key('href'): return None
if not attrs.has_attr('href'): return None
i = -1
for a in self.a:
i += 1
match = 0
if a.has_key('href') and a['href'] == attrs['href']:
if a.has_key('title') or attrs.has_key('title'):
if (a.has_key('title') and attrs.has_key('title') and
if a.has_attr('href') and a['href'] == attrs['href']:
if a.has_attr('title') or attrs.has_attr('title'):
if (a.has_attr('title') and attrs.has_attr('title') and
a['title'] == attrs['title']):
match = True
else:
@@ -249,7 +249,7 @@ class _html2text(sgmllib.SGMLParser):
self.abbr_title = None
self.abbr_data = ''
if attrs.has_key('title'):
if attrs.has_attr('title'):
self.abbr_title = attrs['title']
else:
if self.abbr_title != None:
@@ -262,7 +262,7 @@ class _html2text(sgmllib.SGMLParser):
attrsD = {}
for (x, y) in attrs: attrsD[x] = y
attrs = attrsD
if attrs.has_key('href') and not (SKIP_INTERNAL_LINKS and attrs['href'].startswith('#')):
if attrs.has_attr('href') and not (SKIP_INTERNAL_LINKS and attrs['href'].startswith('#')):
self.astack.append(attrs)
self.o("[")
else:
@@ -285,7 +285,7 @@ class _html2text(sgmllib.SGMLParser):
attrsD = {}
for (x, y) in attrs: attrsD[x] = y
attrs = attrsD
if attrs.has_key('src'):
if attrs.has_attr('src'):
attrs['href'] = attrs['src']
alt = attrs.get('alt', '')
i = self.previousIndex(attrs)
@@ -392,7 +392,7 @@ class _html2text(sgmllib.SGMLParser):
for link in self.a:
if self.outcount > link['outcount']:
self.out(" ["+`link['count']`+"]: " + urlparse.urljoin(self.baseurl, link['href']))
if link.has_key('title'): self.out(" ("+link['title']+")")
if link.has_attr('title'): self.out(" ("+link['title']+")")
self.out("\n")
else:
newa.append(link)
+5 -1
View File
@@ -34,7 +34,7 @@ def replace_br_with_p(body):
return body
# logger.debug(u'---')
# logger.debug(u'BODY start.: ' + body[:250])
# logger.debug(u'BODY start.: ' + body[:4000])
# logger.debug(u'--')
# logger.debug(u'BODY end...: ' + body[-250:])
# logger.debug(u'BODY.......: ' + body)
@@ -49,6 +49,9 @@ def replace_br_with_p(body):
if is_valid_block(body) and body.find('<div') == 0:
body = body[body.index('>')+1:body.rindex('<')].strip()
# BS is doing some BS on entities, meaning &lt; and &gt; are turned into < and >... a **very** bad idea in html.
body = re.sub(r'&(.+?);', r'XAMP;\1;', body)
body = soup_up_div(u'<div>' + body + u'</div>')
body = body[body.index('>')+1:body.rindex('<')]
@@ -238,6 +241,7 @@ def replace_br_with_p(body):
body = re.sub(r'\s*<(\S+)[^>]*>\s*</\1>', r'', body)
body = body.replace(u'{br /}', u'<br />')
body = re.sub(r'XAMP;(.+?);', r'&\1;', body)
body = body.strip()
# re-wrap in div tag.
+5 -4
View File
@@ -646,14 +646,15 @@ class Story(Configurable):
elif self.metadata.has_key(key):
value = self.metadata[key]
if value:
if key == "numWords":
value = commaGroups(value)
if key == "numChapters":
value = commaGroups("%d"%value)
if key in ("numWords","numChapters"):
value = commaGroups(unicode(value))
if key in ("dateCreated"):
value = value.strftime(self.getConfig(key+"_format","%Y-%m-%d %H:%M:%S"))
if key in ("datePublished","dateUpdated"):
value = value.strftime(self.getConfig(key+"_format","%Y-%m-%d"))
if isinstance(value, (datetime.date, datetime.datetime, datetime.time)) and self.hasConfig(key+"_format"):
# logger.info("DATE: %s"%key)
value = value.strftime(self.getConfig(key+"_format"))
if key == "title" and (self.chapter_first or self.chapter_last) and self.getConfig("title_chapter_range_pattern"):
first = self.chapter_first or "1"