Compare commits

..
Author SHA1 Message Date
Jim Miller 557fd14447 Update CLI download zip. 2014-11-03 15:33:14 -06:00
Jim Miller 59800117fa Bump versions. 2014-11-03 15:31:04 -06:00
Jim Miller 8f418a7690 Fixing password only for fimf password protected stories on web service. 2014-11-03 11:57:04 -06:00
Jim Miller 69ee06e2ce Improved error reporting for regular expressions in personal.ini. 2014-11-02 09:45:13 -06:00
Jim Miller 34990ff044 Fix for ficwad login, change [www.thewriterscoffeeshop.com] to [www.twcslibrary.net] in config. 2014-11-01 22:55:07 -05:00
facedeer 511490622e Fixing Fimfiction's password handling. Individual chapters need passwords now. 2014-11-01 20:05:35 -06:00
cryzed 72a5624330 Fixed checking for "src" attribute, preventing a potential bug if the site ever decides to change its image handling 2014-11-01 11:49:36 +01:00
facedeer d7103969e7 Fimfiction adapter: fixing img tag urls missing http: 2014-10-31 23:57:44 -06:00
Jim Miller eac7f4ed4a Change thewriterscoffeeshop.com to twcslibrary.net. 2014-10-30 13:47:33 -05:00
Jim Miller 25c82a3fa8 Change thewriterscoffeeshop.com to twcslibrary.net. 2014-10-30 13:47:11 -05:00
facedeer b06685f97b Merge potterfics.com changes 2014-10-28 14:52:59 -06:00
facedeer 99283c5400 Updating Fimfiction adapter to site changes. Also adding cover source and comment count metadata 2014-10-28 14:27:41 -06:00
Jim Miller ea71818b07 Changes for potterfics.com(Spanish) to login for higher rated stories. 2014-10-28 12:07:34 -05:00
cryzed 22d42ed4c8 Fix for changed login behavior for the fanfiktion.de adapter 2014-10-25 12:49:14 +02:00
Jim Miller 95c64e2e95 Added tag FFDL 2.0.08 for changeset 0ab8b8460599 2014-10-20 21:06:19 -05:00
Jim Miller 3e4b42ad5c Update CLI download zip. 2014-10-20 21:06:12 -05:00
Jim Miller f82bcfeed5 Additional fix for ficwad, bump versions. 2014-10-20 21:04:14 -05:00
Jim Miller bf2684557e Added tag FFDL 2.0.07 for changeset 45e18d7dd684 2014-10-20 17:08:38 -05:00
Jim Miller 49658feb5a Update CLI download zip. 2014-10-20 17:08:25 -05:00
Jim Miller f26dabf956 Bump versions. 2014-10-20 17:06:15 -05:00
Jim Miller 1ada755268 Updates for ficwad sites changes. 2014-10-20 14:08:23 -05:00
Jim Miller 4a8720fe99 Skip AO3 announcement banner. 2014-10-20 14:08:09 -05:00
Jim Miller bd60880586 Merge fimf changes. 2014-10-10 09:42:21 -05:00
Jim Miller fdabf4ed14 Update default user_agent version number. 2014-10-10 09:41:51 -05:00
facedeer 1022f5ee0c Fimfiction updated group and warning metadata, more specific password-needed detection 2014-10-09 23:55:23 -06:00
Jim Miller b04fa75787 Added tag FFDL 2.0.06 for changeset 0001a9e9c4b4 2014-10-06 10:20:25 -05:00
Jim Miller 25264f5222 Update CLI download zip. 2014-10-06 10:20:17 -05:00
Jim Miller 3a3d9959f7 Bump versions. 2014-10-06 10:17:27 -05:00
Jim Miller 55eb39e8b2 Fixes for lotrfanfiction.com. 2014-10-05 11:16:19 -05:00
Jim Miller 3c441e0533 Update defaults for new sites. 2014-10-03 13:10:22 -05:00
Jim Miller fbb7bb9eaa Add adapter_lotrfanfictioncom(Base eFiction). 2014-10-03 12:45:59 -05:00
Jim Miller f297814da2 Fix for category vs categories - eFiction Base 2014-09-30 10:55:25 -05:00
Jim Miller 39e96b9966 Updated and new adapters from scout78. 2014-09-30 10:55:02 -05:00
Jim Miller d28a06d8b2 Added tag FFDL 2.0.05 for changeset 494acdb9b1ca 2014-09-23 14:15:25 -05:00
Jim Miller cb700b0b7b Update CLI download zip. 2014-09-23 14:15:15 -05:00
Jim Miller 002cefc1af Bump versions, sync .po file line numbers. 2014-09-23 14:13:54 -05:00
Jim Miller fd15ad6f4b Changes for storiesonline.net site update, from davidfor. 2014-09-23 10:04:01 -05:00
Jim Miller 46262ae17b Add 'extratags' to AllMetadata so it's available for custom columns. 2014-09-23 10:03:28 -05:00
Jim Miller c848edf0a3 Fix for squidge.org/peja using a story URL for 'Site Map'. 2014-09-18 22:47:55 -05:00
Jim Miller be34b6718f Fix for trying to get story URLs with no books selected. 2014-09-11 11:47:00 -05:00
Jim Miller 79971745a2 Fix for AO3 story list URLs that already have a '?' in them. 2014-09-11 11:46:44 -05:00
Jim Miller 4819ca95b1 Added tag FFDL 2.0.04 for changeset e968a543cce3 2014-09-09 15:57:14 -05:00
Jim Miller d9c6185e97 Update CLI download zip. 2014-09-09 15:57:05 -05:00
Jim Miller e5d0fa7eed Bump versions. 2014-09-09 15:55:29 -05:00
Jim Miller b5180d9020 Don't need to fetch twice like that. 2014-09-08 16:18:17 -05:00
Jim Miller 9e0a3d7afa Clear existing data_chunks when downloading again--web service. 2014-09-08 16:13:52 -05:00
Jim Miller 61ba4da640 Fix for changes to fanfiktion.de and enable caching. 2014-09-08 16:04:35 -05:00
Jim Miller 8bb26fd6f3 Change cache debug output to debug level. 2014-09-07 21:59:22 -05:00
Jim Miller 3151d45010 Fix for nhamagicalworldsus changes. 2014-09-07 21:59:05 -05:00
Jim Miller 81b87246d2 Update translations. 2014-09-05 12:58:18 -05:00
Jim Miller 530d7b0ab5 Extend page caching to AO3, fimf, portkey and buffynfaith.net. 2014-09-05 12:57:30 -05:00
Jim Miller b682d0ba6b Finish correction of getSiteExampleURLs(cls) correctly. 2014-09-05 11:28:17 -05:00
Jim Miller af74529b32 Merge pre-base-efiction into default. trekiverse reverted, fannation and
maplebook using base_efiction.
2014-08-31 13:58:18 -05:00
Jim Miller 667c19ac3c Adding fetched file caching feature and optimizing hits for ffnet in particular. 2014-08-31 13:48:47 -05:00
Jim Miller 5de217a0e3 Correct getSiteExampleURLs method def--it's a classmethod. 2014-08-27 15:57:05 -05:00
Jim Miller 3b62e77c01 Some additional messages for translation, from jobs.py. 2014-08-26 20:17:08 -05:00
Jim Miller e2b6f3f416 Add adapter_sheppardweircom, from scout78. 2014-08-26 20:16:21 -05:00
Jim Miller 2aaa08f923 Fix so autoconvert won't delete FFDL's own output. 2014-08-26 20:15:35 -05:00
Jim Miller c116dc9bf3 New translation for Portuguese (Brazil). 2014-08-21 12:07:57 -05:00
Jim Miller 8ede0411a9 Translations for new strings for Spanish and French. 2014-08-19 20:03:51 -05:00
Jim Miller 48e8716a9a Fix numChapters in adapter_literotica.py. 2014-08-15 19:00:51 -05:00
Jim Miller 25097200ad Merge pre-base-efiction into default. CLI zip does *not* contain base efiction. 2014-08-13 19:44:24 -05:00
Jim Miller 4084763d98 Added tag FFDL 2.0.03 for changeset 02a9eb028955 2014-08-13 19:43:19 -05:00
Jim Miller d52d2f2438 Update CLI download zip. 2014-08-13 19:43:09 -05:00
Jim Miller e47e4bf29a Fix for AO3 authorUrl and authorId. Bump versions. 2014-08-13 19:38:10 -05:00
Jim Miller fc0656ec1e Merge pre-base-efiction into default. CLI zip does *not* contain base_efiction 2014-08-13 13:23:33 -05:00
Jim Miller 445d676d24 Added tag FFDL 2.0.02 for changeset 292fdb288fb0 2014-08-13 13:22:24 -05:00
Jim Miller ef71577b73 Update CLI download zip. 2014-08-13 13:22:14 -05:00
Jim Miller 951fd68ce6 Fix for AO3 authors all coming out as Anonymous. Bump versions. 2014-08-13 13:19:36 -05:00
doe ab01e26526 Merge eFiction-base-adapter 2014-08-12 03:26:18 +02:00
doe e740166ba4 Merge upstream 2014-08-12 02:08:09 +02:00
doe 54616e9892 bdsmgesch: Even more generic chapter parsing 2014-08-12 02:06:04 +02:00
Jim Miller 3fbfa5b56c Merge doe's changes with mine. 2014-08-11 16:11:33 -05:00
Jim Miller 0f166e1e4b Merge with default 2014-08-11 16:08:43 -05:00
Jim Miller a606db85d1 Fix for get URLs from page when urlsfromclip is off. 2014-08-11 16:04:08 -05:00
doe 15abab181f bdsmgesch: Fix for wrongly matched next chapter URL 2014-08-11 17:34:54 +02:00
doe d81b365aba bdsmgesch: Better parsing of next/prev links, config option to guess chapter URLs to speed up metadata retrieval 2014-08-11 16:08:55 +02:00
doe 001f1d5fee Removed getHighestWarningLevel, factored out constants 2014-08-09 03:56:29 +02:00
doe 6b0bebd82f More metadata for trekiverse and fannation
* Romance/Read and Awards/Read
* Encoding set for trekiverse
2014-08-09 03:49:38 +02:00
doe 876255afdb Make bulk loading optional in BaseEfictionAdapter
* Chapters can be downloaded one-by-one or once in extractChapterUrlsAndMetadata
* Both methods deliver the same list of chapter URLs, they are exchangeable
* getHighestWarningLevel obsolete now
* Regexes are compiled once
* Login/Warning logic improved
2014-08-09 03:47:33 +02:00
doe 40c9bcccf7 Added config option 'bulk_load' to config
* Setting this to true will load the complete story as printed
  and keep it cached between extractMetadata & getChapter in
  BaseEfictionAdapter based adapters
2014-08-09 03:40:48 +02:00
doe 8ae585728f eFiction base: proper login / warning handling
* performLogin mostly from TrekiverseOrgAdapter
* handle 'not logged in' and 'be warned' cases
* abstracted strings to eFiction constants
* converted TrekiverseOrgAdapter to use base class
* metadata fields
* self.decode can be overriden with self.getEncoding() (@classmethod)
2014-08-08 15:59:10 +02:00
doe c1915b05be eFiction base: better metadata handling
* made most of the methods @classmethods
* checked the eFiction source to make sure the class uses the right defaults
* added documentation
* adapted implementing classes
2014-08-08 14:36:12 +02:00
doe 1bf21e09a0 Marked all adapters that are efiction based with a comment '# Software: eFiction' 2014-08-06 15:52:32 +02:00
doe 0c51160924 Foundations of an eFiction base adapter
* works for fannation and themaplebookshop
* metadata parsing must be more extensible
* missing documentation
* proper handling of warnings / is_adult checks
* ...
2014-08-06 15:33:21 +02:00
doe 6e897c78f1 themaplebookshelf: URL normalization in __init__ 2014-08-06 02:10:25 +02:00
doe 52ccebf16e tolkienfanfiction: Minor change to avoid un-normalized URIs 2014-08-06 01:28:58 +02:00
doe a0e9123c58 Fixed wrong getSiteExampleURLs format in bdsmgesch 2014-08-06 01:22:38 +02:00
doe 6e93ded2a3 merge upstream 2014-08-06 01:18:49 +02:00
doe d95b96b9c4 Added notice to getSiteExampleURLs docstring on the expected format 2014-08-06 01:18:28 +02:00
doe d14b100d7e bdsmgesch: Normalize storyID/URL properly
doesn't support HTTPS fixed
www. fixed
timer removed
2014-08-06 01:09:53 +02:00
doe 66f9d4f7e0 Cleaned up tolkienfanfiction adapter to normalize storyID/URL properly
also handle optional www
2014-08-06 00:48:46 +02:00
Jim Miller cd342bb352 Fix for getSiteExampleURLs--it's expected to be space separated strings. 2014-08-05 16:43:47 -05:00
doe ae7ffcfc32 Added themaplebookshelf to default/plugin-default.ini 2014-08-05 20:11:55 +02:00
doe 863ee5c44b Adapter for The Maple Bookshelf (themaplebookshelf.com) 2014-08-05 17:01:43 +02:00
doe 8d094fc26e Improved literotica adapter
* for metadata: retrieve only original URL and author URL
* oldest chapter date = release date, newest chapter date = update date
* Description = Concatenation of chapter descriptions
2014-08-05 14:36:31 +02:00
Jim Miller ee85c13e75 Moved tag FFDL 2.0.01 to changeset ffbf432fad82 (from changeset a8a798d540d7) 2014-08-04 13:46:34 -05:00
Jim Miller 367bea316b Update CLI download zip. 2014-08-04 13:46:21 -05:00
Jim Miller e5168d1d98 Moved tag FFDL 2.0.01 to changeset a8a798d540d7 (from changeset 4a13d7fbdf27) 2014-08-04 13:42:01 -05:00
Jim Miller ff4559a8ad Fixes for getSiteExampleURLs. 2014-08-04 13:41:49 -05:00
Jim Miller 3cfa3179c9 Added tag FFDL 2.0.01 for changeset 4a13d7fbdf27 2014-08-04 13:36:28 -05:00
Jim Miller e3970a64de Update CLI download zip. 2014-08-04 13:36:00 -05:00
Jim Miller 2320e118b3 Bump versions. 2014-08-04 13:33:21 -05:00
Jim Miller 53d76052a8 tolkienfanfiction.com - extracategories:Lord of the Rings 2014-08-04 13:32:51 -05:00
Jim Miller 9471a74527 Fix for anthology books, no author in comments if all same. 2014-08-03 11:11:40 -05:00
Jim Miller 7ba9290c7d Add site tolkienfanfiction.com. From doe5716. 2014-08-02 08:45:48 -05:00
Jim Miller 64f60b4540 Fix identifiers:"~ur(i|l)..." search string. 2014-08-01 13:53:18 -05:00
Jim Miller 2adbcdc23e Switch to using Transifix .po files w/o poedit. 2014-08-01 12:51:48 -05:00
Jim Miller d3ab5e2024 Add German language site bdsm-geschichten.net. From doe5716. 2014-08-01 12:18:36 -05:00
Jim Miller 635170f664 Fix for CLI complaining no image libs with no_image_processing on. From doe5716. 2014-08-01 11:25:43 -05:00
Jim Miller 5ac90d3cdb Add other languages for literotica.com, from doe5716. 2014-08-01 11:24:31 -05:00
Jim Miller 394b21ab0e Fix for non-split list replace_metadata. 2014-08-01 11:21:55 -05:00
Jim Miller df6599a9cc bloodshedverse & spikelover added tags in title, etc. Use stripHTML more. 2014-07-24 14:41:52 -05:00
Jim Miller d380f8b05c Moved tag FFDL 2.0.00 to changeset 38ffd50ea87f (from changeset 999abcae72fd) 2014-07-23 14:44:29 -05:00
Jim Miller cf1ecee8e9 Update CLI download zip. - Version reset to 2.0.00. 2014-07-23 14:44:16 -05:00
Jim Miller 5a85524629 Added tag FFDL 2.0.00 for changeset 999abcae72fd 2014-07-23 14:40:50 -05:00
Jim Miller 65e6bce0bc Bump version for cal2(reset for CLI/web), update trans, add es trans. 2014-07-23 14:40:13 -05:00
Jim Miller 12161a8224 Fix for login needed for efpfanfic.net 'red' rated stories. 2014-07-23 09:44:36 -05:00
Jim Miller 19d181a90f Fix for .eml file dropping in add box on qt5. 2014-07-22 13:04:38 -05:00
Jim Miller 159d33f287 Replace don't-sort kludge on Reject lists with less kludgey version. 2014-07-21 22:10:43 -05:00
Jim Miller bbd806ab95 Proper Qt5 fixes. 2014-07-21 11:04:11 -05:00
Jim Miller a4f82bf841 More fixes for Qt5 fixes, storiesonline needing login, default bloodshedverse.com
to Windows-1252, and issue 78, chapterless fimf stories.
2014-07-19 21:41:01 -05:00
Jim Miller 389b658135 Change text "Reject Silently" to "Reject Without Confirmation". 2014-07-16 11:17:00 -05:00
Jim Miller 7bcd4143e5 FFDL Qt5 changes, and Reject Silently option. 2014-07-11 18:41:25 -05:00
Jim Miller e9f010a162 Partial fix for literotica site specific eroticatags. Doesn't always work. 2014-07-02 19:49:30 -05:00
Jim Miller babfc35f7b Add site specific reviews to wraithbait.com, allow ffnet story specific covers. 2014-07-02 19:48:43 -05:00
Jim Miller cefcb9ab96 Apply cover_exclusion_regexp to explicit covers, too. 2014-07-02 19:47:48 -05:00
Jim Miller 7a763a8516 Added tag calibre-plugin-1.8.26 for changeset dbf614a1d6ce 2014-06-25 20:12:15 -05:00
Jim Miller a191521649 Added tag FanFictionDownLoader-4.5.07 for changeset dbf614a1d6ce 2014-06-25 20:12:10 -05:00
Jim Miller e108c2d828 Update CLI download zip. 2014-06-25 20:11:58 -05:00
Jim Miller 7534c03a37 Bump versions. 2014-06-25 19:09:02 -05:00
Jim Miller bf2e71e17f Fixes for int types being put in data, and only sort ships when there are ships. 2014-06-24 18:42:20 -05:00
Jim Miller 110960169a Moved tag calibre-plugin-1.8.25 to changeset 8e76e63420e7 (from changeset 2a5ad32eec54) 2014-06-21 09:57:41 -05:00
Jim Miller fb9d128687 Moved tag FanFictionDownLoader-4.5.06 to changeset 8e76e63420e7 (from changeset 2a5ad32eec54) 2014-06-21 09:57:36 -05:00
137 changed files with 8421 additions and 4582 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
# ffd-retief-hrd fanfictiondownloader
application: fanfictiondownloader
version: 4-5-06
version: 2-0-09
runtime: python27
api_version: 1
threadsafe: true
+2 -2
View File
@@ -42,8 +42,8 @@ class FanFictionDownLoaderBase(InterfaceActionBase):
description = _('UI plugin to download FanFiction stories from various sites.')
supported_platforms = ['windows', 'osx', 'linux']
author = 'Jim Miller'
version = (1, 8, 25)
minimum_calibre_version = (1, 13, 0)
version = (2, 0, 9)
minimum_calibre_version = (1, 48, 0)
#: This field defines the GUI plugin class that contains all the code
#: that actually does something. Its format is module_path:class_name
+13 -6
View File
@@ -8,12 +8,19 @@ __copyright__ = '2011, Grant Drake <grant.drake@gmail.com>'
__docformat__ = 'restructuredtext en'
import os
from PyQt4 import QtGui
from PyQt4.Qt import (Qt, QIcon, QPixmap, QLabel, QDialog, QHBoxLayout,
QTableWidgetItem, QFont, QLineEdit, QComboBox,
QVBoxLayout, QDialogButtonBox, QStyledItemDelegate, QDateTime,
QTextEdit,
QListWidget, QAbstractItemView)
try:
from PyQt5 import QtWidgets as QtGui
from PyQt5.Qt import (Qt, QIcon, QPixmap, QLabel, QDialog, QHBoxLayout,
QTableWidgetItem, QFont, QLineEdit, QComboBox,
QVBoxLayout, QDialogButtonBox, QStyledItemDelegate, QDateTime,
QTextEdit, QListWidget, QAbstractItemView)
except ImportError as e:
from PyQt4 import QtGui
from PyQt4.Qt import (Qt, QIcon, QPixmap, QLabel, QDialog, QHBoxLayout,
QTableWidgetItem, QFont, QLineEdit, QComboBox,
QVBoxLayout, QDialogButtonBox, QStyledItemDelegate, QDateTime,
QTextEdit, QListWidget, QAbstractItemView)
from calibre.constants import iswindows
from calibre.gui2 import gprefs, error_dialog, UNDEFINED_QDATETIME, info_dialog
from calibre.gui2.actions import menu_action_unique_name
+48 -21
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2012, Jim Miller'
__copyright__ = '2014, Jim Miller'
__docformat__ = 'restructuredtext en'
import logging
@@ -13,10 +13,31 @@ logger = logging.getLogger(__name__)
import traceback, copy, threading
from collections import OrderedDict
from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QLabel,
QLineEdit, QFont, QWidget, QTextEdit, QComboBox,
QCheckBox, QPushButton, QTabWidget, QVariant, QScrollArea,
QDialogButtonBox, QGroupBox )
try:
from PyQt5.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QLabel,
QLineEdit, QFont, QWidget, QTextEdit, QComboBox,
QCheckBox, QPushButton, QTabWidget, QScrollArea,
QDialogButtonBox, QGroupBox )
except ImportError as e:
from PyQt4.Qt import (QDialog, QWidget, QVBoxLayout, QHBoxLayout, QLabel,
QLineEdit, QFont, QWidget, QTextEdit, QComboBox,
QCheckBox, QPushButton, QTabWidget, QScrollArea,
QDialogButtonBox, QGroupBox )
try:
from calibre.gui2 import QVariant
del QVariant
except ImportError:
is_qt4 = False
convert_qvariant = lambda x: x
else:
is_qt4 = True
def convert_qvariant(x):
vt = x.type()
if vt == x.String:
return unicode(x.toString())
if vt == x.List:
return [convert_qvariant(i) for i in x.toList()]
return x.toPyObject()
from calibre.gui2.ui import get_gui
from calibre.gui2 import dynamic, info_dialog
@@ -60,7 +81,7 @@ from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.adapters \
from calibre_plugins.fanfictiondownloader_plugin.common_utils \
import ( KeyboardConfigDialog, PrefsViewerDialog )
from calibre.gui2.complete import MultiCompleteLineEdit
from calibre.gui2.complete2 import EditWithComplete #MultiCompleteLineEdit
class RejectURLList:
def __init__(self,prefs):
@@ -220,6 +241,7 @@ class ConfigWidget(QWidget):
prefs['checkforurlchange'] = self.basic_tab.checkforurlchange.isChecked()
prefs['injectseries'] = self.basic_tab.injectseries.isChecked()
prefs['smarten_punctuation'] = self.basic_tab.smarten_punctuation.isChecked()
prefs['reject_always'] = self.basic_tab.reject_always.isChecked()
if self.readinglist_tab:
# lists
@@ -243,7 +265,7 @@ class ConfigWidget(QWidget):
prefs['gcnewonly'] = self.generatecover_tab.gcnewonly.isChecked()
gc_site_settings = {}
for (site,combo) in self.generatecover_tab.gc_dropdowns.iteritems():
val = unicode(combo.itemData(combo.currentIndex()).toString())
val = unicode(convert_qvariant(combo.itemData(combo.currentIndex())))
if val != 'none':
gc_site_settings[site] = val
#print("gc_site_settings[%s]:%s"%(site,gc_site_settings[site]))
@@ -275,12 +297,12 @@ class ConfigWidget(QWidget):
# Custom Columns tab
# error column
prefs['errorcol'] = unicode(self.cust_columns_tab.errorcol.itemData(self.cust_columns_tab.errorcol.currentIndex()).toString())
prefs['errorcol'] = unicode(convert_qvariant(self.cust_columns_tab.errorcol.itemData(self.cust_columns_tab.errorcol.currentIndex())))
# cust cols tab
colsmap = {}
for (col,combo) in self.cust_columns_tab.custcol_dropdowns.iteritems():
val = unicode(combo.itemData(combo.currentIndex()).toString())
val = unicode(convert_qvariant(combo.itemData(combo.currentIndex())))
if val != 'none':
colsmap[col] = val
#print("colsmap[%s]:%s"%(col,colsmap[col]))
@@ -482,6 +504,11 @@ class BasicTab(QWidget):
self.reject_reasons.clicked.connect(self.show_reject_reasons)
self.l.addWidget(self.reject_reasons)
self.reject_always = QCheckBox(_('Reject Without Confirmation?'),self)
self.reject_always.setToolTip(_("Always reject URLs on the Reject List without stopping and asking."))
self.reject_always.setChecked(prefs['reject_always'])
self.l.addWidget(self.reject_always)
topl.addWidget(defs_gb)
horz = QHBoxLayout()
@@ -649,7 +676,7 @@ class ReadingListTab(QWidget):
label = QLabel(_('"Send to Device" Reading Lists'))
label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
horz.addWidget(label)
self.send_lists_box = MultiCompleteLineEdit(self)
self.send_lists_box = EditWithComplete(self)
self.send_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
self.send_lists_box.update_items_cache(reading_lists)
self.send_lists_box.setText(prefs['send_lists'])
@@ -665,7 +692,7 @@ class ReadingListTab(QWidget):
label = QLabel(_('"To Read" Reading Lists'))
label.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
horz.addWidget(label)
self.read_lists_box = MultiCompleteLineEdit(self)
self.read_lists_box = EditWithComplete(self)
self.read_lists_box.setToolTip(_("When enabled, new/updated stories will be automatically added to these lists."))
self.read_lists_box.update_items_cache(reading_lists)
self.read_lists_box.setText(prefs['read_lists'])
@@ -727,17 +754,17 @@ class GenerateCoverTab(QWidget):
horz.addWidget(label)
dropdown = QComboBox(self)
dropdown.setToolTip(s)
dropdown.addItem('',QVariant('none'))
dropdown.addItem('','none')
for setting in gc_settings:
dropdown.addItem(setting,QVariant(setting))
dropdown.addItem(setting,setting)
if site == _("Default"):
self.gc_dropdowns["Default"] = dropdown
if 'Default' in prefs['gc_site_settings']:
dropdown.setCurrentIndex(dropdown.findData(QVariant(prefs['gc_site_settings']['Default'])))
dropdown.setCurrentIndex(dropdown.findData(prefs['gc_site_settings']['Default']))
else:
self.gc_dropdowns[site] = dropdown
if site in prefs['gc_site_settings']:
dropdown.setCurrentIndex(dropdown.findData(QVariant(prefs['gc_site_settings'][site])))
dropdown.setCurrentIndex(dropdown.findData(prefs['gc_site_settings'][site]))
horz.addWidget(dropdown)
self.sl.addLayout(horz)
@@ -966,12 +993,12 @@ class CustomColumnsTab(QWidget):
label.setToolTip(_("Update this %s column(%s) with...")%(key,column['datatype']))
horz.addWidget(label)
dropdown = QComboBox(self)
dropdown.addItem('',QVariant('none'))
dropdown.addItem('','none')
for md in permitted_values[column['datatype']]:
dropdown.addItem(titleLabels[md],QVariant(md))
dropdown.addItem(titleLabels[md],md)
self.custcol_dropdowns[key] = dropdown
if key in prefs['custom_cols']:
dropdown.setCurrentIndex(dropdown.findData(QVariant(prefs['custom_cols'][key])))
dropdown.setCurrentIndex(dropdown.findData(prefs['custom_cols'][key]))
if column['datatype'] == 'enumeration':
dropdown.setToolTip(_("Metadata values valid for this type of column.")+"\n"+_("Values that aren't valid for this enumeration column will be ignored."))
else:
@@ -1007,11 +1034,11 @@ class CustomColumnsTab(QWidget):
horz.addWidget(label)
self.errorcol = QComboBox(self)
self.errorcol.setToolTip(tooltip)
self.errorcol.addItem('',QVariant('none'))
self.errorcol.addItem('','none')
for key, column in custom_columns.iteritems():
if column['datatype'] in ('text','comments'):
self.errorcol.addItem(column['name'],QVariant(key))
self.errorcol.setCurrentIndex(self.errorcol.findData(QVariant(prefs['errorcol'])))
self.errorcol.addItem(column['name'],key)
self.errorcol.setCurrentIndex(self.errorcol.findData(prefs['errorcol']))
horz.addWidget(self.errorcol)
self.l.addLayout(horz)
+41 -40
View File
@@ -19,12 +19,36 @@ logger = logging.getLogger(__name__)
import urllib
import email
from PyQt4 import QtGui
from PyQt4.Qt import (QDialog, QTableWidget, QVBoxLayout, QHBoxLayout, QGridLayout,
QPushButton, QString, QLabel, QCheckBox, QIcon, QLineEdit,
QComboBox, QVariant, QProgressDialog, QTimer, QDialogButtonBox,
QPixmap, Qt, QAbstractItemView, SIGNAL, QTextEdit, pyqtSignal,
QGroupBox, QFrame)
try:
from PyQt5 import QtWidgets as QtGui
from PyQt5.Qt import (QDialog, QTableWidget, QVBoxLayout, QHBoxLayout, QGridLayout,
QPushButton, QLabel, QCheckBox, QIcon, QLineEdit,
QComboBox, QProgressDialog, QTimer, QDialogButtonBox,
QPixmap, Qt, QAbstractItemView, QTextEdit, pyqtSignal,
QGroupBox, QFrame)
except ImportError as e:
from PyQt4 import QtGui
from PyQt4.Qt import (QDialog, QTableWidget, QVBoxLayout, QHBoxLayout, QGridLayout,
QPushButton, QLabel, QCheckBox, QIcon, QLineEdit,
QComboBox, QProgressDialog, QTimer, QDialogButtonBox,
QPixmap, Qt, QAbstractItemView, QTextEdit, pyqtSignal,
QGroupBox, QFrame)
try:
from calibre.gui2 import QVariant
del QVariant
except ImportError:
is_qt4 = False
convert_qvariant = lambda x: x
else:
is_qt4 = True
def convert_qvariant(x):
vt = x.type()
if vt == x.String:
return unicode(x.toString())
if vt == x.List:
return [convert_qvariant(i) for i in x.toList()]
return x.toPyObject()
from calibre.gui2.dialogs.confirm_delete import confirm
from calibre.gui2.complete2 import EditWithComplete
@@ -146,19 +170,6 @@ class RejectUrlEntry:
return retval
# This is a more than slightly kludgey way to get
# EditWithComplete to *not* alpha-order the reasons, but leave
# them in the order entered. If
# calibre.gui2.complete2.CompleteModel.set_items ever changes,
# this function will need to also.
def complete_model_set_items_kludge(self, items):
items = [unicode(x.strip()) for x in items]
items = [x for x in items if x]
items = tuple(items)
self.all_items = self.current_items = items
self.current_prefix = ''
self.reset()
class NotGoingToDownload(Exception):
def __init__(self,error,icon='dialog_error.png'):
self.error=error
@@ -196,9 +207,9 @@ class DroppableQTextEdit(QTextEdit):
urllist.extend(get_urls_from_text(part.get_payload(decode=True)))
else:
urllist.extend(get_urls_from_text("%s"%msg))
if urllist:
self.append("\n".join(urllist))
return None
return QTextEdit.dropEvent(self,event)
def canInsertFromMimeData(self, source):
@@ -559,7 +570,7 @@ class LoopProgressDialog(QProgressDialog):
status_prefix=_("Fetched metadata for")):
QProgressDialog.__init__(self,
init_label,
QString(), 0, len(book_list), gui)
_('Cancel'), 0, len(book_list), gui)
self.setWindowTitle(win_title)
self.setMinimumWidth(500)
self.book_list = book_list
@@ -829,11 +840,11 @@ class StoryListTableWidget(QTableWidget):
icon = get_icon(book['icon'])
status_cell = IconWidgetItem(None,icon,val)
status_cell.setData(Qt.UserRole, QVariant(val))
status_cell.setData(Qt.UserRole, val)
self.setItem(row, 0, status_cell)
title_cell = ReadOnlyTableWidgetItem(book['title'])
title_cell.setData(Qt.UserRole, QVariant(row))
title_cell.setData(Qt.UserRole, row)
self.setItem(row, 1, title_cell)
self.setItem(row, 2, AuthorTableWidgetItem(", ".join(book['author']), ", ".join(book['author_sort'])))
@@ -848,7 +859,7 @@ class StoryListTableWidget(QTableWidget):
books = []
#print("=========================\nbooks:%s"%self.books)
for row in range(self.rowCount()):
rnum = self.item(row, 1).data(Qt.UserRole).toPyObject()
rnum = convert_qvariant(self.item(row, 1).data(Qt.UserRole))
book = self.books[rnum]
books.append(book)
return books
@@ -914,15 +925,12 @@ class RejectListTableWidget(QTableWidget):
def populate_table_row(self, row, rej):
url_cell = ReadOnlyTableWidgetItem(rej.url)
url_cell.setData(Qt.UserRole, QVariant(rej.book_id))
url_cell.setData(Qt.UserRole, rej.book_id)
self.setItem(row, 0, url_cell)
self.setItem(row, 1, ReadOnlyTableWidgetItem(rej.title))
self.setItem(row, 2, ReadOnlyTableWidgetItem(rej.auth))
note_cell = EditWithComplete(self)
note_cell.lineEdit().mcompleter.model().set_items = \
partial(complete_model_set_items_kludge,
note_cell.lineEdit().mcompleter.model())
note_cell = EditWithComplete(self,sort_func=lambda x:1)
items = [rej.note]+self.rejectreasons
note_cell.update_items_cache(items)
@@ -992,10 +1000,7 @@ class RejectListDialog(SizePersistedDialog):
button_layout.addItem(spacerItem1)
if show_all_reasons:
self.reason_edit = EditWithComplete(self)
self.reason_edit.lineEdit().mcompleter.model().set_items = \
partial(complete_model_set_items_kludge,
self.reason_edit.lineEdit().mcompleter.model())
self.reason_edit = EditWithComplete(self,sort_func=lambda x:1)
items = ['']+rejectreasons
self.reason_edit.update_items_cache(items)
@@ -1037,7 +1042,7 @@ class RejectListDialog(SizePersistedDialog):
rejectrows = []
for row in range(self.rejects_table.rowCount()):
url = unicode(self.rejects_table.item(row, 0).text()).strip()
book_id = self.rejects_table.item(row, 0).data(Qt.UserRole).toPyObject()
book_id =convert_qvariant(self.rejects_table.item(row, 0).data(Qt.UserRole))
title = unicode(self.rejects_table.item(row, 1).text()).strip()
auth = unicode(self.rejects_table.item(row, 2).text()).strip()
note = unicode(self.rejects_table.cellWidget(row, 3).currentText()).strip()
@@ -1047,7 +1052,7 @@ class RejectListDialog(SizePersistedDialog):
def get_reject_list_ids(self):
rejectrows = []
for row in range(self.rejects_table.rowCount()):
book_id = self.rejects_table.item(row, 0).data(Qt.UserRole).toPyObject()
book_id = convert_qvariant(self.rejects_table.item(row, 0).data(Qt.UserRole))
if book_id:
rejectrows.append(book_id)
return rejectrows
@@ -1089,11 +1094,7 @@ class EditTextDialog(QDialog):
self.textedit.setToolTip(tooltip)
if rejectreasons or reasonslabel:
self.reason_edit = EditWithComplete(self)
self.reason_edit.lineEdit().mcompleter.model().set_items = \
partial(complete_model_set_items_kludge,
self.reason_edit.lineEdit().mcompleter.model())
self.reason_edit = EditWithComplete(self,sort_func=lambda x:1)
items = ['']+rejectreasons
self.reason_edit.update_items_cache(items)
+60 -19
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2012, Jim Miller'
__copyright__ = '2014, Jim Miller'
__docformat__ = 'restructuredtext en'
import logging
@@ -19,10 +19,12 @@ import urllib
import email
import traceback
from PyQt4.Qt import (QApplication, QMenu, QToolButton, QTimer)
from PyQt4.Qt import QPixmap, Qt
from PyQt4.QtCore import QBuffer
try:
from PyQt5.Qt import (QApplication, QMenu, QTimer)
from PyQt5.QtCore import QBuffer
except ImportError as e:
from PyQt4.Qt import (QApplication, QMenu, QTimer)
from PyQt4.QtCore import QBuffer
from calibre.constants import numeric_version as calibre_version
@@ -207,7 +209,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
#print("text/plain:%s"%event.mimeData().data(mimetype))
urllist.extend(get_urls_from_text(event.mimeData().data(mimetype)))
#print("urllist:%s\ndropped_ids:%s"%(urllist,dropped_ids))
# print("urllist:%s\ndropped_ids:%s"%(urllist,dropped_ids))
if urllist or dropped_ids:
QTimer.singleShot(1, partial(self.do_drop,
dropped_ids=dropped_ids,
@@ -385,6 +387,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
def get_urls_from_page_menu(self):
urltxt = ""
if prefs['urlsfromclip']:
try:
urltxt = self.get_urls_clip(storyurls=False)[0]
@@ -417,7 +420,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
def list_story_urls(self):
'''Get list of URLs from existing books.'''
if self.gui.current_view().selectionModel().selectedRows() == 0 :
if not self.gui.current_view().selectionModel().selectedRows() :
self.gui.status_bar.show_message(_('No Selected Books to Get URLs From'),
3000)
return
@@ -712,6 +715,11 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
# No need to do anything with perfs here, but we could.
prefs
def make_id_searchstr(self,url):
# older idents can be uri vs url and have | instead of : after
# http, plus many sites are now switching to https.
return 'identifiers:"~ur(i|l):~^%s$"'%re.sub(r'https?\\\:','https?(\:|\|)',re.escape(url))
def prep_downloads(self, options, books, merge=False, extrapayload=None):
'''Fetch metadata for stories from servers, launch BG job when done.'''
@@ -724,6 +732,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
options['version'] = self.version
logger.debug(self.version)
options['personal.ini'] = get_ffdl_personalini()
#print("prep_downloads:%s"%books)
@@ -764,7 +773,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
if not merge: # skip reject list when merging.
if rejecturllist.check(url):
rejnote = rejecturllist.get_full_note(url)
if question_dialog(self.gui, _('Reject URL?'),'''
if prefs['reject_always'] or question_dialog(self.gui, _('Reject URL?'),'''
<h3>%s</h3>
<p>%s</p>
<p>"<b>%s</b>"</p>
@@ -817,8 +826,16 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
skip_date_update = False
options['personal.ini'] = get_ffdl_personalini()
adapter = get_ffdl_adapter(url,fileform)
## save and share cookiejar and pagecache between all
## downloads.
if 'pagecache' not in options:
options['pagecache'] = adapter.get_empty_pagecache()
adapter.set_pagecache(options['pagecache'])
if 'cookiejar' not in options:
options['cookiejar'] = adapter.get_empty_cookiejar()
adapter.set_cookiejar(options['cookiejar'])
# reduce foreground sleep time for ffnet when few books.
if 'ffnetcount' in options and \
adapter.getConfig('tweak_fg_sleep') and \
@@ -836,7 +853,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
## or a couple tries of one or the other
for x in range(0,2):
try:
adapter.getStoryMetadataOnly()
adapter.getStoryMetadataOnly(get_cover=False)
except exceptions.FailedToLogin, f:
logger.warn("Login Failed, Need Username/Password.")
userpass = UserPassDialog(self.gui,url,f)
@@ -852,12 +869,12 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
adapter.is_adult=True
# let other exceptions percolate up.
story = adapter.getStoryMetadataOnly()
story = adapter.getStoryMetadataOnly(get_cover=False)
series = story.getMetadata('series')
if not merge and series and prefs['checkforseriesurlid']:
# try to find *series anthology* by *seriesUrl* identifier url or uri first.
searchstr = 'identifiers:"~ur(i|l):~^%s$"'%re.sub(r'https?\\:','https?(\:|\|)',re.escape(story.getMetadata('seriesUrl')))
searchstr = self.make_id_searchstr(story.getMetadata('seriesUrl'))
identicalbooks = db.search_getting_ids(searchstr, None)
# print("searchstr:%s"%searchstr)
# print("identicalbooks:%s"%identicalbooks)
@@ -934,7 +951,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
logger.debug("from URL(%s)"%url)
# try to find by identifier url or uri first.
searchstr = 'identifiers:"~ur(i|l):~^%s$"'%re.sub(r'https?\:','https?(\:|\|)',url)
searchstr = self.make_id_searchstr(url)
identicalbooks = db.search_getting_ids(searchstr, None)
# print("searchstr:%s"%searchstr)
# print("identicalbooks:%s"%identicalbooks)
@@ -1080,7 +1097,18 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
dir=options['tdir'])
logger.debug("title:"+book['title'])
logger.debug("outfile:"+tmp.name)
book['outfile'] = tmp.name
book['outfile'] = tmp.name
# cookiejar = PersistentTemporaryFile(prefix=story.formatFileName("${title}-${author}-",allowunsafefilename=False)[:100],
# suffix='.cookiejar',
# dir=options['tdir'])
# adapter.save_cookiejar(cookiejar.name)
# book['cookiejar'] = cookiejar.name
# pagecache = PersistentTemporaryFile(prefix=story.formatFileName("${title}-${author}-",allowunsafefilename=False)[:100],
# suffix='.pagecache',
# dir=options['tdir'])
# adapter.save_pagecache(pagecache.name)
# book['pagecache'] = pagecache.name
return
@@ -1137,7 +1165,15 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
_('FFDL log'), _('FFDL download ended'), msg,
show_copy_button=False)
return
cookiejarfile = PersistentTemporaryFile(suffix='.cookiejar',
dir=options['tdir'])
options['cookiejar'].save(cookiejarfile.name,
ignore_discard=True,
ignore_expires=True)
options['cookiejarfile']=cookiejarfile.name
del options['cookiejar'] ## can't be pickled.
func = 'arbitrary_n'
cpus = self.gui.job_manager.server.pool_size
args = ['calibre_plugins.fanfictiondownloader_plugin.jobs', 'do_download_worker',
@@ -1797,6 +1833,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
book['url'] = ''
book['site'] = ''
book['added'] = False
book['pubdate'] = None
return book
def convert_urls_to_books(self, urls):
@@ -1962,9 +1999,9 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
# fill from first of each if not already present:
for k in ('pubdate', 'timestamp', 'updatedate'):
if k not in b: # not in this book? Skip it.
if k not in b or not b[k]: # not in this book? Skip it.
continue
if k not in book: # first is good enough for publisher.
if k not in book or not book[k]: # first is good enough for publisher.
book[k]=b[k]
# Do these even on first to get the all_metadata settings.
@@ -2013,8 +2050,12 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
book['comments'] = existingbook['comments']
else:
book['title'] = deftitle = book_list[0]['title']
book['comments'] = _("Anthology containing:")+"\n" + \
"\n".join([ _("%s by %s")%(b['title'],', '.join(b['author'])) for b in book_list ])
if len(book['author']) > 1:
book['comments'] = _("Anthology containing:")+"\n" + \
"\n".join([ _("%s by %s")%(b['title'],', '.join(b['author'])) for b in book_list ])
else:
book['comments'] = _("Anthology containing:")+"\n" + \
"\n".join([ b['title'] for b in book_list ])
# book['all_metadata']['description']
# if all same series, use series for name. But only if all and not previous named
+20 -25
View File
@@ -4,7 +4,7 @@ from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2012, Jim Miller'
__copyright__ = '2014, Jim Miller'
__copyright__ = '2011, Grant Drake <grant.drake@gmail.com>'
__docformat__ = 'restructuredtext en'
@@ -19,6 +19,11 @@ from calibre.utils.ipc.server import Server
from calibre.utils.ipc.job import ParallelJob
from calibre.constants import numeric_version as calibre_version
# for smarten punc
from calibre.ebooks.oeb.polish.main import polish, ALL_OPTS
from calibre.utils.logging import Log
from collections import namedtuple
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (NotGoingToDownload,
OVERWRITE, OVERWRITEALWAYS, UPDATE, UPDATEALWAYS, ADDNEW, SKIP, CALIBREONLY)
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
@@ -58,10 +63,6 @@ def do_download_worker(book_list, options,
done=None,
args=args)
job._book = book
# job._book_id = book_id
# job._title = title
# job._modified_date = modified_date
# job._existing_isbn = existing_isbn
server.add_job(job)
else:
# was already bad before the subprocess ever started.
@@ -69,7 +70,7 @@ def do_download_worker(book_list, options,
# This server is an arbitrary_n job, so there is a notifier available.
# Set the % complete to a small number to avoid the 'unavailable' indicator
notification(0.01, 'Downloading FanFiction Stories')
notification(0.01, _('Downloading FanFiction Stories'))
# dequeue the job results as they arrive, saving the results
count = 0
@@ -81,24 +82,19 @@ def do_download_worker(book_list, options,
if not job.is_finished:
continue
# A job really finished. Get the information.
output_book = job.result
#print("output_book:%s"%output_book)
book_list.remove(job._book)
book_list.append(job.result)
book_id = job._book['calibre_id']
#title = job._title
count = count + 1
notification(float(count)/total, '%d of %d stories finished downloading'%(count,total))
# Add this job's output to the current log
logger.info('Logfile for book ID %s (%s)'%(book_id, job._book['title']))
logger.info(job.details)
if count >= total:
logger.info("\nSuccessful:\n%s\n"%("\n".join([book['url'] for book in
logger.info("\n"+_("Successful:")+"\n%s\n"%("\n".join([book['url'] for book in
filter(lambda x: x['good'], book_list) ] ) ) )
logger.info("\nUnsuccessful:\n%s\n"%("\n".join([book['url'] for book in
logger.info("\n"+_("Unsuccessful:")+"\n%s\n"%("\n".join([book['url'] for book in
filter(lambda x: not x['good'], book_list) ] ) ) )
break
@@ -109,11 +105,10 @@ def do_download_worker(book_list, options,
def do_download_for_worker(book,options,notification=lambda x,y:x):
'''
Child job, to extract isbn from formats for this specific book,
when run as a worker job
Child job, to download story when run as a worker job
'''
try:
book['comment'] = 'Download started...'
book['comment'] = _('Download started...')
configuration = get_ffdl_config(book['url'],
options['fileform'],
@@ -122,8 +117,8 @@ def do_download_for_worker(book,options,notification=lambda x,y:x):
if not options['updateepubcover'] and 'epub_for_update' in book and options['collision'] in (UPDATE, UPDATEALWAYS):
configuration.set("overrides","never_make_cover","true")
# images only for epub, even if the user mistakenly turned it
# on else where.
# images only for epub, html, even if the user mistakenly
# turned it on else where.
if options['fileform'] not in ("epub","html"):
configuration.set("overrides","include_images","false")
@@ -133,6 +128,10 @@ def do_download_for_worker(book,options,notification=lambda x,y:x):
adapter.password = book['password']
adapter.setChaptersRange(book['begin'],book['end'])
adapter.load_cookiejar(options['cookiejarfile'])
logger.debug("cookiejar:%s"%adapter.cookiejar)
adapter.set_pagecache(options['pagecache'])
story = adapter.getStoryMetadataOnly()
if 'calibre_series' in book:
adapter.setSeries(book['calibre_series'][0],book['calibre_series'][1])
@@ -191,13 +190,13 @@ def do_download_for_worker(book,options,notification=lambda x,y:x):
# dup handling from ffdl_plugin needed for anthology updates.
if options['collision'] == UPDATE:
if chaptercount == urlchaptercount:
book['comment']="Already contains %d chapters. Reuse as is."%chaptercount
book['comment']=_("Already contains %d chapters. Reuse as is.")%chaptercount
book['outfile'] = book['epub_for_update'] # for anthology merge ops.
return book
# dup handling from ffdl_plugin needed for anthology updates.
if chaptercount > urlchaptercount:
raise NotGoingToDownload("Existing epub contains %d chapters, web site only has %d. Use Overwrite to force update." % (chaptercount,urlchaptercount),'dialog_error.png')
raise NotGoingToDownload(_("Existing epub contains %d chapters, web site only has %d. Use Overwrite to force update.") % (chaptercount,urlchaptercount),'dialog_error.png')
if not (options['collision'] == UPDATEALWAYS and chaptercount == urlchaptercount) \
and adapter.getConfig("do_update_hook"):
@@ -208,16 +207,12 @@ def do_download_for_worker(book,options,notification=lambda x,y:x):
writer.writeStory(outfilename=outfile, forceOverwrite=True)
book['comment'] = 'Update %s completed, added %s chapters for %s total.'%\
book['comment'] = _('Update %s completed, added %s chapters for %s total.')%\
(options['fileform'],(urlchaptercount-chaptercount),urlchaptercount)
if options['smarten_punctuation'] and options['fileform'] == "epub" \
and calibre_version >= (0, 9, 39):
# do smarten_punctuation from calibre's polish feature
from calibre.ebooks.oeb.polish.main import polish, ALL_OPTS
from calibre.utils.logging import Log
from collections import namedtuple
data = {'smarten_punctuation':True}
opts = ALL_OPTS.copy()
opts.update(data)
+1
View File
@@ -25,6 +25,7 @@ default_prefs['rejecturls'] = ''
default_prefs['rejectreasons'] = '''Sucked
Boring
Dup from another site'''
default_prefs['reject_always'] = False
default_prefs['updatemeta'] = True
default_prefs['updatecover'] = False
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+139 -10
View File
@@ -321,7 +321,7 @@ sort_ships:false
#keep_in_order_author:true
## User-agent
user_agent:FFDL/1.7
user_agent:FFDL/2.0
## Each output format has a section that overrides [defaults]
[html]
@@ -626,6 +626,18 @@ extracategories:The Sentinel
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[bdsm-geschichten.net]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
## This site offers no index page so we can either guess the chapter URLs
## by dec/incrementing numbers ('guess') or walk all the chapters in the metadata
## parsing state ('parse'). Since guessing can lead to errors for non-standard
## story URLs, the default is to parse
#find_chapters:guess
[bloodshedverse.com]
## website encoding(s) In theory, each website reports the character
## encoding they use for each page. In practice, some sites report it
@@ -634,7 +646,7 @@ extracategories:The Sentinel
## explicitly set the encoding and order if you need to. The special
## value 'auto' will call chardet and use the encoding it reports if
## it has +90% confidence. 'auto' is not reliable.
website_encodings:ISO-8859-1,auto
website_encodings:Windows-1252,ISO-8859-1,auto
## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them.
@@ -700,6 +712,20 @@ extracategories:Castle
## cover image. This lets you exclude them.
cover_exclusion_regexp:/images/.*?ribbon.gif
[csi-forensics.com]
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Virtually all eFiction-based sites allow downloading the whole story in
## bulk using the 'Print' feature. If 'bulk_load' is set to 'true', both
## metadata and chapters can be loaded in one step
bulk_load:true
extra_valid_entries: readings
readings_label: Readings
[dark-solace.org]
## Site dedicated to these categories/characters/ships
## Some sites require login (or login for some rated stories) The
@@ -872,6 +898,24 @@ extraships:Harry Potter/Hermione Granger
#username:YourName
#password:yourpassword
[fannation.shades-of-moonlight.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Virtually all eFiction-based sites allow downloading the whole story in
## bulk using the 'Print' feature. If 'bulk_load' is set to 'true', both
## metadata and chapters can be loaded in one step
bulk_load:true
extra_valid_entries: readings,romance
extra_titlepage_entries: readings,romance
readings_label: Readings
romance_label: Romance
[ficwad.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -999,7 +1043,19 @@ extraships:Kirk/Spock
[literotica.com]
extra_valid_entries:eroticatags
eroticatags_label:Erotica Tags
#extra_titlepage_entries: eroticatags
extra_titlepage_entries: eroticatags
[lotrfanfiction.com]
## Virtually all eFiction-based sites allow downloading the whole story in
## bulk using the 'Print' feature. If 'bulk_load' is set to 'true', both
## metadata and chapters can be loaded in one step
bulk_load:true
extra_valid_entries: readings
readings_label: Readings
## Site dedicated to these categories/characters/ships
extracategories:Lord of the Rings
[lumos.sycophanthex.com]
## Some sites do not require a login, but do require the user to
@@ -1123,6 +1179,24 @@ extracategories:My Little Pony: Friendship is Magic
## Site dedicated to these categories/characters/ships
extracategories:The Pretender
[samandjack.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1
extracharacters:Sam,Jack
extraships:Sam/Jack
[samdean.archive.nu]
## Site dedicated to these categories/characters/ships
extracategories:Supernatural
@@ -1151,6 +1225,24 @@ extracategories:Harry Potter
## this should go in your personal.ini, not defaults.ini.
#is_adult:true
[sheppardweir.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis
extracharacters:John Sheppard,Elizabeth Weir
extraships:John Sheppard/Elizabeth Weir
[spikeluver.com]
## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them.
@@ -1230,6 +1322,22 @@ extraships:Harry Potter/Draco Malfoy
## Site dedicated to these categories/characters/ships
extracategories:Criminal Minds
[themaplebookshelf.com]
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
#is_adult:true
## Virtually all eFiction-based sites allow downloading the whole story in
## bulk using the 'Print' feature. If 'bulk_load' is set to 'true', both
## metadata and chapters can be loaded in one step
bulk_load:true
extra_valid_entries: readings,challenge
extra_titlepage_entries: readings,challenge
challenge_label: Challenge
readings_label: Readings
[themasque.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -1268,6 +1376,10 @@ extracategories:Harry Potter
## Site dedicated to these categories/characters/ships
extracategories:Stargate: SG-1
[tolkienfanfiction.com]
## Site dedicated to these categories/characters/ships
extracategories:Lord of the Rings
[trekiverse.org]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -1276,6 +1388,11 @@ extracategories:Stargate: SG-1
#username:YourName
#password:yourpassword
## Virtually all eFiction-based sites allow downloading the whole story in
## bulk using the 'Print' feature. If 'bulk_load' is set to 'true', both
## metadata and chapters can be loaded in one step
bulk_load:true
## Some sites also require the user to confirm they are adult for
## adult content. In commandline version, this should go in your
## personal.ini, not defaults.ini.
@@ -1284,8 +1401,10 @@ extracategories:Stargate: SG-1
## Site dedicated to these categories/characters/ships
extracategories:Star Trek
extra_valid_entries:awards
extra_valid_entries:readings,awards
extra_titlepage_entries:readings,awards
awards_label:Awards
readings_label:Readings
cover_exclusion_regexp:art/.*Awards.jpg
@@ -1383,8 +1502,11 @@ type_label:Type of Couple
[www.fanfiction.net]
user_agent:
## fanfiction.net's 'cover' images are really just tiny thumbnails.
## Change this to false to use them anyway.
never_make_cover: true
## Set this to true to never use them.
#never_make_cover: false
## fanfiction.net shows the user's
cover_exclusion_regexp:/imageu/
## fanfiction.net is blocking people more aggressively. If you
## download fewer stories less often you can likely get by with
@@ -1476,7 +1598,7 @@ extracategories:My Little Pony: Friendship is Magic
## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them.
extra_valid_entries:likes,dislikes,views,total_views,short_description,groups,groupsUrl,groupsHTML,prequel,prequelUrl,prequelHTML,sequels,sequelsUrl,sequelsHTML
extra_valid_entries:likes,dislikes,views,total_views,short_description,groups,groupsUrl,groupsHTML,prequel,prequelUrl,prequelHTML,sequels,sequelsUrl,sequelsHTML,comment_count,coverSource,coverSourceUrl,coverSourceHTML
likes_label:Likes
dislikes_label:Dislikes
views_label:Highest Single Chapter Views
@@ -1491,6 +1613,10 @@ prequelHTML_label:Prequel
sequels_label:Sequels
sequelsUrl_label:Sequel URLs
sequelsHTML_label:Sequels
comment_count_label:Comment Count
coverSource_label:Cover Source
coverSourceUrl_label:Cover Source URL
coverSourceHTML_label:Cover Source
keep_in_order_sequels:true
keep_in_order_sequelsUrl:true
@@ -1499,7 +1625,7 @@ keep_in_order_groupsUrl:true
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:prequel,sequels,groups
make_linkhtml_entries:prequel,sequels,groups,coverSource
[www.harrypotterfanfiction.com]
## Some sites do not require a login, but do require the user to
@@ -1707,7 +1833,7 @@ extracategories:Lord of the Rings
#username:YourName
#password:yourpassword
[www.thewriterscoffeeshop.com]
[www.twcslibrary.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
@@ -1720,7 +1846,7 @@ extracategories:Lord of the Rings
## personal.ini, not defaults.ini.
#is_adult:true
## thewriterscoffeeshop.com (ab)uses series as personal reading lists.
## twcslibrary.net (ab)uses series as personal reading lists.
collect_series: false
[www.tthfanfic.org]
@@ -1808,6 +1934,9 @@ extracharacters:Wolverine,Rogue
## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis
extra_valid_entries:reviews
reviews_label:Reviews
[overrides]
## It may sometimes be useful to override all of the specific format,
## site and site:format sections in your private configuration. For
+3 -3
View File
@@ -227,9 +227,9 @@ def main(argv,
except:
options.update = False
pass
## Check for include_images and absence of PIL, give warning.
if adapter.getConfig('include_images'):
## Check for include_images without no_image_processing. In absence of PIL, give warning.
if adapter.getConfig('include_images') and not adapter.getConfig('no_image_processing'):
try:
from calibre.utils.magick import Image
logging.debug("Using calibre.utils.magick")
Binary file not shown.
+13 -5
View File
@@ -39,7 +39,7 @@ import adapter_mediaminerorg
import adapter_potionsandsnitchesnet
import adapter_tenhawkpresentscom
import adapter_adastrafanficcom
import adapter_thewriterscoffeeshopcom
import adapter_twcslibrarynet
import adapter_tthfanficorg
import adapter_twilightednet
import adapter_twiwritenet
@@ -130,11 +130,19 @@ import adapter_nocturnallightnet
import adapter_fanfichu
import adapter_fanfictioncsodaidokhu
import adapter_fictionmaniatv
import adapter_bdsmgeschichten
import adapter_tolkienfanfiction
import adapter_themaplebookshelf
import adapter_fannation
import adapter_sheppardweircom
import adapter_samandjacknet
import adapter_csiforensicscom
import adapter_lotrfanfictioncom
## This bit of complexity allows adapters to be added by just adding
## importing. It eliminates the long if/else clauses we used to need
## to pick out the adapter.
## List of registered site adapters.
__class_list = []
__domain_map = {}
@@ -202,7 +210,7 @@ def getConfigSectionFor(url):
(cls,fixedurl) = getClassFor(url)
if cls:
return cls.getConfigSection()
# No adapter found.
raise exceptions.UnknownSite( url, [cls.getSiteDomain() for cls in __class_list] )
@@ -234,9 +242,9 @@ def getClassFor(url):
if cls:
fixedurl = cls.stripURLParameters(fixedurl)
return (cls,fixedurl)
def getClassFromList(domain):
try:
return __domain_map[domain]
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -53,12 +54,19 @@ class AdAstraFanficComSiteAdapter(BaseSiteAdapter):
return 'www.adastrafanfic.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def extractChapterUrlsAndMetadata(self):
if self.is_adult or self.getConfig("is_adult"):
@@ -78,8 +78,8 @@ class ArchiveOfOurOwnOrgAdapter(BaseSiteAdapter):
return 'archiveofourown.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/works/123456 http://"+self.getSiteDomain()+"/collections/Some_Archive/works/123456 http://"+self.getSiteDomain()+"/works/123456/chapters/78901"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/works/123456 http://"+cls.getSiteDomain()+"/collections/Some_Archive/works/123456 http://"+cls.getSiteDomain()+"/works/123456/chapters/78901"
def getSiteURLPattern(self):
# http://archiveofourown.org/collections/Smallville_Slash_Archive/works/159770
@@ -123,6 +123,13 @@ class ArchiveOfOurOwnOrgAdapter(BaseSiteAdapter):
else:
return True
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
@@ -155,29 +162,33 @@ class ArchiveOfOurOwnOrgAdapter(BaseSiteAdapter):
if self.needToLoginCheck(data):
# need to log in for this one.
self.performLogin(url,data)
data = self._fetchUrl(url)
meta = self._fetchUrl(metaurl)
data = self._fetchUrl(url,usecache=False)
meta = self._fetchUrl(metaurl,usecache=False)
# use BeautifulSoup HTML parser to make everything easier to find.
soup = bs.BeautifulSoup(data)
for tag in soup.findAll('div',id='admin-banner'):
tag.extract()
metasoup = bs.BeautifulSoup(meta)
for tag in metasoup.findAll('div',id='admin-banner'):
tag.extract()
# Now go hunting for all the meta data and the chapter list.
## Title
a = soup.find('a', href=re.compile(r"^/works/\d+$"))
a = soup.find('a', href=re.compile(r"/works/\d+$"))
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
alist = soup.findAll('a', href=re.compile(r"^/users/\w+/pseuds/\w+"))
alist = soup.findAll('a', href=re.compile(r"/users/\w+/pseuds/\w+"))
if len(alist) < 1: # ao3 allows for author 'Anonymous' with no author link.
self.story.setMetadata('author','Anonymous')
self.story.setMetadata('authorUrl','http://archiveofourown.org/')
self.story.setMetadata('authorId','0')
else:
for a in alist:
self.story.addToList('authorId',a['href'].split('/')[2])
self.story.addToList('authorUrl','http://'+self.host+a['href'])
self.story.addToList('authorId',a['href'].split('/')[-1])
self.story.addToList('authorUrl',a['href'])
self.story.addToList('author',a.text)
newestChapter = None
@@ -71,7 +71,7 @@ class ArchiveSkyeHawkeComAdapter(BaseSiteAdapter):
return ['archive.skyehawke.com','www.skyehawke.com']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://archive.skyehawke.com/story.php?no=1234 http://www.skyehawke.com/archive/story.php?no=1234 http://skyehawke.com/archive/story.php?no=1234"
def getSiteURLPattern(self):
@@ -91,8 +91,6 @@ class ArchiveSkyeHawkeComAdapter(BaseSiteAdapter):
else:
raise e
data = self._fetchUrl(url)
# use BeautifulSoup HTML parser to make everything easier to find.
soup = bs.BeautifulSoup(data)
# print data
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class AshwinderSycophantHexComAdapter(BaseSiteAdapter):
return 'ashwinder.sycophanthex.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -64,8 +65,8 @@ class Asr3SlashzoneOrgAdapter(BaseSiteAdapter):
return 'asr3.slashzone.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/archive/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=")+r"\d+$"
@@ -0,0 +1,346 @@
# -*- coding: utf-8 -*-
# Copyright 2014 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
import urlparse
import time
from .. import BeautifulSoup as bs
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
def _translate_date_german_english(date):
fullmon = {"Januar":"01",
"Februar":"02",
u"März":"03",
"April":"04",
"Mai":"05",
"Juni":"06",
"Juli":"07",
"August":"08",
"September":"09",
"Oktober":"10",
"November":"11",
"Dezember":"12"}
for (name,num) in fullmon.items():
date = date.replace(name,num)
return date
_REGEX_TRAILING_DIGIT = re.compile("(\d+)$")
_REGEX_DASH_TO_END = re.compile("-[^-]+$")
_REGEX_CHAPTER_TITLE = re.compile(ur"""
\s*
[\u2013-]?
\s*
([\dIVX-]+)?
\.?
\s*
[\[\(]?
\s*
(Teil|Kapitel|Tag)?
\s*
([\dIVX-]+)?
\s*
[\]\)]?
\s*
$
""", re.VERBOSE)
_INITIAL_STEP = 5
class BdsmGeschichtenAdapter(BaseSiteAdapter):
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["utf8", "Windows-1252"]
self.story.setMetadata('siteabbrev','bdsmgesch')
# Replace possible chapter numbering
chapterMatch = _REGEX_TRAILING_DIGIT.search(url)
if chapterMatch is None:
self.maxChapter = 1
else:
self.maxChapter = int(chapterMatch.group(1))
# url = re.sub(_REGEX_TRAILING_DIGIT, "1", url)
# set storyId
self.story.setMetadata('storyId', re.compile(self.getSiteURLPattern()).match(url).group('storyId'))
# normalize URL
self._setURL('http://%s/%s' % (self.getSiteDomain(), self.story.getMetadata('storyId')))
self.dateformat = '%d. %m %Y - %H:%M'
@staticmethod
def getSiteDomain():
return 'bdsm-geschichten.net'
@classmethod
def getAcceptDomains(cls):
return ['www.bdsm-geschichten.net', 'www.bdsm-geschichten.net']
@classmethod
def getSiteExampleURLs(cls):
return "http://www.bdsm-geschichten.net/title-of-story-1 http://bdsm-geschichten.net/title-of-story-1"
def getSiteURLPattern(self):
return r"http://(www\.)?bdsm-geschichten.net/(?P<storyId>[a-zA-Z0-9_-]+)"
def extractChapterUrlsAndMetadata(self):
if not (self.is_adult or self.getConfig("is_adult")):
raise exceptions.AdultCheckRequired(self.url)
try:
data1 = self._fetchUrl(self.url)
soup = bs.BeautifulSoup(data1)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
#strip comments from soup
[comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, bs.Comment))]
# Cache the soups so we won't have to redownload in getChapterText later
self.soupsCache = {}
self.soupsCache[self.url] = soup
# author
authorDiv = soup.find("div", "author-pane-line author-name")
authorId = authorDiv.string.strip()
self.story.setMetadata('authorId', authorId)
self.story.setMetadata('author', authorId)
# TODO not really true need to be loggedin for this to work or fetch userid
self.story.setMetadata('authorUrl','http://'+self.host+'/'+authorId)
# TODO better metadata
date = soup.find("div", {"class": "submitted"}).string.strip()
date = re.sub(" &#151;.*", "", date)
date = _translate_date_german_english(date)
self.story.setMetadata('datePublished', makeDate(date, self.dateformat))
title1 = soup.find("h1", {'class': 'title'}).string
for tagLink in soup.find("ul", "taxonomy").findAll("a"):
self.story.addToList('category', tagLink.string)
## Retrieve chapter soups
if self.getConfig('find_chapters') == 'guess':
self.chapterUrls = []
self._find_chapters_by_guessing(title1)
else:
self._find_chapters_by_parsing(soup)
firstChapterUrl = self.chapterUrls[0][1]
if firstChapterUrl in self.soupsCache:
firstChapterSoup = self.soupsCache[firstChapterUrl]
h1 = firstChapterSoup.find("h1").text
else:
h1 = soup.find("h1").text
h1 = re.sub(_REGEX_CHAPTER_TITLE, "", h1)
self.story.setMetadata('title', h1)
self.story.setMetadata('numChapters', len(self.chapterUrls))
return
def _find_chapters_by_parsing(self, soup):
# store original soup
origSoup = soup
#
# find first chapter
#
firstLink = None
firstLinkDiv = soup.find("div", "field-field-erster-teil")
if firstLinkDiv is not None:
firstLink = "http://%s%s" % (self.getSiteDomain(), firstLinkDiv.findNext("a")['href'])
logger.debug("Found first chapter right away <%s>" % firstLink)
try:
soup = bs.BeautifulSoup(self._fetchUrl(firstLink))
self.soupsCache[firstLink] = soup
self.chapterUrls.insert(0, (soup.find("h1").text, firstLink))
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise exceptions.StoryDoesNotExist(firstLink)
else:
logger.debug("DIDN'T find first chapter right away")
# parse previous Link until first
while True:
prevLink = None
prevLinkDiv = soup.find("div", "field-field-vorheriger-teil")
if prevLinkDiv is not None:
prevLink = prevLinkDiv.find("a")
if prevLink is None:
prevLink = soup.find("a", text=re.compile("&lt;&lt;&lt;")) # <<<
if prevLink is None:
logger.debug("Couldn't find prev part")
break
else:
logger.debug("Previous Chapter <%s>" % prevLink)
if type(prevLink) != bs.Tag or prevLink.name != "a":
prevLink = prevLink.findParent("a")
if prevLink is None or '#' in prevLink['href']:
logger.debug("Couldn't find prev part (false positive) <%s>" % prevLink)
break
prevLink = prevLink['href']
try:
soup = bs.BeautifulSoup(self._fetchUrl(prevLink))
self.soupsCache[prevLink] = soup
prevTtitle = soup.find("h1", {'class': 'title'}).string
self.chapterUrls.insert(0, (prevTtitle, prevLink))
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(nextLink)
else:
raise e
firstLink = prevLink
# if first chapter couldn't be determined, assume the URL originally
# passed is the first chapter
if firstLink is None:
logger.debug("Couldn't set first chapter")
firstLink = self.url
self.chapterUrls.insert(0, (soup.find("h1").text, firstLink))
# set first URL
logger.debug("Set first link: %s" % firstLink)
self._setURL(firstLink)
self.story.setMetadata('storyId', re.compile(self.getSiteURLPattern()).match(firstLink).group('storyId'))
#
# Parse next chapters
#
while True:
nextLink = None
nextLinkDiv = soup.find("div", "field-field-naechster-teil")
if nextLinkDiv is not None:
nextLink = nextLinkDiv.find("a")
if nextLink is None:
nextLink = soup.find("a", text=re.compile("&gt;&gt;&gt;"))
if nextLink is None:
nextLink = soup.find("a", text=re.compile("Fortsetzung"))
if nextLink is None:
logger.debug("Couldn't find next part")
break
else:
if type(nextLink) != bs.Tag or nextLink.name != "a":
nextLink = nextLink.findParent("a")
if nextLink is None or '#' in nextLink['href']:
logger.debug("Couldn't find next part (false positive) <%s>" % nextLink)
break
nextLink = nextLink['href']
if not nextLink.startswith('http:'):
nextLink = 'http://' + self.getSiteDomain() + nextLink
for loadedChapter in self.chapterUrls:
if loadedChapter[0] == nextLink:
logger.debug("ERROR: Repeating chapter <%s> Try to fix it" % nextLink)
nextLinkMatch = _REGEX_TRAILING_DIGIT.match(nextLink)
if nextLinkMatch is not None:
curChap = nextLinkMatch.group(1)
nextLink = re.sub(_REGEX_TRAILING_DIGIT, str(int(curChap) + 1), nextLink)
else:
break
try:
data = self._fetchUrl(nextLink)
soup = bs.BeautifulSoup(data)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(nextLink)
else:
raise e
title2 = soup.find("h1", {'class': 'title'}).string
self.chapterUrls.append((title2, nextLink))
logger.debug("Grabbing next chapter URL " + nextLink)
self.soupsCache[nextLink] = soup
# [comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, bs.Comment))]
logger.debug("Chapters: %s" % self.chapterUrls)
def _find_chapters_by_guessing(self, title1):
step = _INITIAL_STEP
curMax = self.maxChapter + step
lastHit = True
while True:
nextChapterUrl = re.sub(_REGEX_TRAILING_DIGIT, str(curMax), self.url)
if nextChapterUrl == self.url:
logger.debug("Unable to guess next chapter because URL doesn't end in numbers")
break;
try:
logger.debug("Trying chapter URL " + nextChapterUrl)
data = self._fetchUrl(nextChapterUrl)
hit = True
except urllib2.HTTPError, e:
if e.code == 404:
hit = False
else:
raise e
if hit:
logger.debug("Found chapter URL " + nextChapterUrl)
self.maxChapter = curMax
self.soupsCache[nextChapterUrl] = bs.BeautifulSoup(data)
if not lastHit:
break
lastHit = curMax
curMax += step
else:
lastHit = False
curMax -= 1
logger.debug(curMax)
for i in xrange(1, self.maxChapter):
nextChapterUrl = re.sub(_REGEX_TRAILING_DIGIT, str(i), self.url)
nextChapterTitle = re.sub("1", str(i), title1)
self.chapterUrls.append((nextChapterTitle, nextChapterUrl))
def getChapterText(self, url):
if url in self.soupsCache:
logger.debug('Getting chapter <%s> from cache' % url)
soup = self.soupsCache[url]
else:
logger.debug('Downloading chapter <%s>' % url)
data1 = self._fetchUrl(url)
soup = bs.BeautifulSoup(data1)
#strip comments from soup
[comment.extract() for comment in soup.findAll(text=lambda text:isinstance(text, bs.Comment))]
# get story text
storyDiv1 = bs.Tag(soup, "div")
for para in soup.find("div", "full-node").find('div', 'content').findAll("p"):
storyDiv1.append(para)
storyDiv1.append('<br />')
storytext = self.utf8FromSoup(url,storyDiv1)
return storytext
def getClass():
return BdsmGeschichtenAdapter
@@ -4,6 +4,7 @@ import urllib2
import urlparse
from .. import BeautifulSoup
from ..htmlcleanup import stripHTML
from base_adapter import BaseSiteAdapter, makeDate
from .. import exceptions
@@ -74,11 +75,11 @@ class BloodshedverseComAdapter(BaseSiteAdapter):
# Since no 404 error code we have to raise the exception ourselves.
# A title that is just 'by' indicates that there is no author name
# and no story title available.
if soup.title.string.strip() == 'by':
if stripHTML(soup.title) == 'by':
raise exceptions.StoryDoesNotExist(self.url)
for option in soup.find('select', {'name': 'chapter'}):
title = option.string.strip()
title = stripHTML(option)
url = self.READ_URL_TEMPLATE % option['value']
self.chapterUrls.append((title, url))
@@ -101,15 +102,15 @@ class BloodshedverseComAdapter(BaseSiteAdapter):
raise exceptions.FailedToDownload(self.url)
title_anchor = list_box.find('a', {'class': 'fictitle'})
self.story.setMetadata('title', title_anchor.string.strip())
self.story.setMetadata('title', stripHTML(title_anchor))
author_anchor = title_anchor.findNextSibling('a')
self.story.setMetadata('author', author_anchor.string.strip())
self.story.setMetadata('author', stripHTML(author_anchor))
self.story.setMetadata('authorId', _get_query_data(author_anchor['href'])['who'])
self.story.setMetadata('authorUrl', urlparse.urljoin(self.url, author_anchor['href']))
list_review = list_box.find('div', {'class': 'list_review'})
reviews = list_review.a.string.strip().split(' ', 1)[0]
reviews = stripHTML(list_review.a).split(' ', 1)[0]
self.story.setMetadata('reviews', reviews)
summary_div = list_box.find('div', {'class': 'list_summary'})
@@ -122,7 +123,7 @@ class BloodshedverseComAdapter(BaseSiteAdapter):
# I'm assuming this to be the category, not sure what else it could be
first_listinfo = list_box.find('div', {'class': 'list_info'})
self.story.addToList('category', first_listinfo.a.string.strip())
self.story.addToList('category', stripHTML(first_listinfo.a))
for list_info in first_listinfo.findNextSiblings('div', {'class': 'list_info'}):
for b_tag in list_info('b'):
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -89,8 +90,8 @@ class BloodTiesFansComAdapter(BaseSiteAdapter): # XXX
return 'bloodties-fans.com' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fiction/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fiction/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fiction/viewstory.php?sid=")+r"\d+$"
@@ -90,8 +90,8 @@ class BuffyNFaithNetAdapter(BaseSiteAdapter):
self.opener.addheaders.append(('Referer', 'http://'+self.getSiteDomain()+'/'))
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234 http://buffynfaith.net/fanfictions/index.php?act=ovr&id=1234 http://buffynfaith.net/fanfictions/index.php?act=vie&id=1234&ch=2"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=ovr&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234&ch=2"
def getSiteURLPattern(self):
#http://buffynfaith.net/fanfictions/index.php?act=vie&id=963
@@ -101,6 +101,13 @@ class BuffyNFaithNetAdapter(BaseSiteAdapter):
r"(vie|ovr)&id=(?P<id>\d+)(&ch=(?P<ch>\d+))?$"
return p
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def extractChapterUrlsAndMetadata(self):
dateformat = "%d %B %Y"
@@ -109,7 +116,6 @@ class BuffyNFaithNetAdapter(BaseSiteAdapter):
#set a cookie to get past adult check
if self.is_adult or self.getConfig("is_adult"):
cookieproc = urllib2.HTTPCookieProcessor()
cookie = cl.Cookie(version=0, name='my_age', value='yes',
port=None, port_specified=False,
domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False,
@@ -121,8 +127,7 @@ class BuffyNFaithNetAdapter(BaseSiteAdapter):
comment_url=None,
rest={'HttpOnly': None},
rfc2109=False)
cookieproc.cookiejar.set_cookie(cookie)
self.opener = urllib2.build_opener(cookieproc)
self.cookiejar.set_cookie(cookie)
self.setHeader()
try:
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -91,8 +92,8 @@ class CastleFansOrgAdapter(BaseSiteAdapter): # XXX
return 'castlefans.org' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfic/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfic/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fanfic/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class ChaosSycophantHexComAdapter(BaseSiteAdapter):
return 'chaos.sycophanthex.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -65,8 +65,8 @@ class CheckmatedComAdapter(BaseSiteAdapter):
return 'www.checkmated.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/story.php?story=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/story.php?story=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/story.php?story=")+r"\d+$"
@@ -0,0 +1,56 @@
# -*- coding: utf-8 -*-
# Copyright 2014 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
import re
from base_efiction_adapter import BaseEfictionAdapter
class CSIForensicsComAdapter(BaseEfictionAdapter):
@staticmethod
def getSiteDomain():
return 'csi-forensics.com'
@classmethod
def getPathToArchive(self):
return ''
@classmethod
def getSiteAbbrev(self):
return 'csiforensics'
@classmethod
def getDateFormat(self):
return "%d %b %Y"
def handleMetadataPair(self, key, value):
if key == 'Warnings':
for val in re.split("\s*,\s*", value):
if value == 'None':
return
else:
self.story.addToList('warnings', val)
elif 'Categories' in key:
for val in re.split("\s*>\s*", value):
self.story.addToList('category', val)
else:
super(CSIForensicsComAdapter, self).handleMetadataPair(key, value)
def getClass():
return CSIForensicsComAdapter
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -70,8 +71,8 @@ class DarkSolaceOrgAdapter(BaseSiteAdapter):
return ['www.dark-solace.org','dark-solace.org']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/elysian/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/elysian/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/elysian/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class DestinysGatewayComAdapter(BaseSiteAdapter):
return 'www.destinysgateway.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -76,8 +76,8 @@ class DokugaComAdapter(BaseSiteAdapter):
return 'www.dokuga.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfiction/story/1234/1 http://"+self.getSiteDomain()+"/spark/story/1234/1"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/1 http://"+cls.getSiteDomain()+"/spark/story/1234/1"
def getSiteURLPattern(self):
return r"http://"+self.getSiteDomain()+"/(fanfiction|spark)?/story/\d+/?\d+?$"
@@ -66,8 +66,8 @@ class DotMoonNetAdapter(BaseSiteAdapter):
return 'www.dotmoon.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/library_view.php?storyid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/library_view.php?storyid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/library_view.php?storyid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class DracoAndGinnyComAdapter(BaseSiteAdapter):
return 'www.dracoandginny.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class DramioneOrgAdapter(BaseSiteAdapter):
return 'dramione.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class EfictionEstelielDeAdapter(BaseSiteAdapter):
return 'efiction.esteliel.de'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,15 +67,16 @@ class EFPFanFicNet(BaseSiteAdapter):
return 'www.efpfanfic.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data):
if 'Fai il login e leggi la storia!' in data:
if( 'Fai il login e leggi la storia!' in data or
'Questa storia presenta contenuti non adatti ai minori' in data ):
return True
else:
return False
@@ -203,8 +205,8 @@ class EFPFanFicNet(BaseSiteAdapter):
noteblock = storyblock.find('div', {'class':'notebloc'})
#print("%s"%noteblock)
notetext = ("%s" % noteblock).replace("<br />"," |")
# <div class="notebloc">Autore: <a href="viewuser.php?uid=243036">Cendrillon89</a> | Pubblicata: 23/10/12 | Aggiornata: 30/10/12 | Rating: Arancione | Genere: Drammatico, Sentimentale | Capitoli: 10 | Completa<br />
# Tipo di coppia: Het | Personaggi: Akasuna no Sasori , Akatsuki, Nuovo Personaggio | Note: OOC | Avvertimenti: Tematiche delicate<br />
# <div class="notebloc">Autore: <a href="viewuser.php?uid=243036">Cendrillon89</a> | Pubblicata: 23/10/12 | Aggiornata: 30/10/12 | Rating: Arancione | Genere: Drammatico, Sentimentale | Capitoli: 10 | Completa<br />
# Tipo di coppia: Het | Personaggi: Akasuna no Sasori , Akatsuki, Nuovo Personaggio | Note: OOC | Avvertimenti: Tematiche delicate<br />
# Categoria: <a href="categories.php?catid=1&amp;parentcatid=1">Anime & Manga</a> > <a href="categories.php?catid=108&amp;parentcatid=108">Naruto</a> | Contesto: Naruto Shippuuden | Leggi le <a href="reviews.php?sid=1331275&amp;a=">3</a> recensioni</div>
cats = noteblock.findAll('a',href=re.compile(r'browse.php\?type=categories'))
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class ErosnSapphoSycophantHexComAdapter(BaseSiteAdapter):
return 'erosnsappho.sycophanthex.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -52,6 +52,8 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
# latest chapter yet and going back to chapter 1 to pull the
# chapter list doesn't get the latest. So save and use the
# original URL given to pull chapter list & metadata.
# Not used by plugin because URL gets normalized first for
# eliminating duplicate story urls.
self.origurl = url
if "https://m." in self.origurl:
## accept m(mobile)url, but use www.
@@ -68,20 +70,29 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
return ['www.fanfiction.net','m.fanfiction.net']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "https://www.fanfiction.net/s/1234/1/ https://www.fanfiction.net/s/1234/12/ http://www.fanfiction.net/s/1234/1/Story_Title http://m.fanfiction.net/s/1234/1/"
def getSiteURLPattern(self):
return r"https?://(www|m)?\.fanfiction\.net/s/\d+(/\d+)?(/|/[^/]+)?/?$"
def _fetchUrl(self,url):
time.sleep(1.0) ## ffnet(and, I assume, fpcom) tends to fail
## more if hit too fast. This is in
## additional to what ever the
## slow_down_sleep_time setting is.
return BaseSiteAdapter._fetchUrl(self,url)
def _fetchUrl(self,url,parameters=None,extrasleep=1.0):
# time.sleep(1.0) ## ffnet(and, I assume, fpcom) tends to fail
# ## more if hit too fast. This is in
# ## additional to what ever the
# ## slow_down_sleep_time setting is.
return BaseSiteAdapter._fetchUrl(self,url,
parameters=parameters,
extrasleep=extrasleep)
def extractChapterUrlsAndMetadata(self):
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def doExtractChapterUrlsAndMetadata(self,get_cover=True):
# fetch the chapter. From that we will get almost all the
# metadata and chapter list
@@ -256,14 +267,15 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
else:
self.story.setMetadata('status', 'In-Progress')
# Try the larger image first.
try:
img = soup.find('img',{'class':'lazy cimage'})
self.setCoverImage(url,img['data-original'])
except:
img = soup.find('img',{'class':'cimage'})
if img:
self.setCoverImage(url,img['src'])
if get_cover:
# Try the larger image first.
try:
img = soup.find('img',{'class':'lazy cimage'})
self.setCoverImage(url,img['data-original'])
except:
img = soup.find('img',{'class':'cimage'})
if img:
self.setCoverImage(url,img['src'])
# Find the chapter selector
select = soup.find('select', { 'name' : 'chapter' } )
@@ -287,12 +299,12 @@ class FanFictionNetSiteAdapter(BaseSiteAdapter):
return
def getChapterText(self, url):
time.sleep(4.0) ## ffnet(and, I assume, fpcom) tends to fail
## more if hit too fast. This is in
## additional to what ever the
## slow_down_sleep_time setting is.
# time.sleep(4.0) ## ffnet(and, I assume, fpcom) tends to fail
# ## more if hit too fast. This is in
# ## additional to what ever the
# ## slow_down_sleep_time setting is.
logger.debug('Getting chapter text from: %s' % url)
data = self._fetchUrl(url)
data = self._fetchUrl(url,extrasleep=4.0)
if "Please email this error message in full to <a href='mailto:support@fanfiction.com'>support@fanfiction.com</a>" in data:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! FanFiction.net Site Error!" % url)
@@ -68,12 +68,19 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
return 'www.fanfiktion.de'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/s/46ccbef30000616306614050 http://"+self.getSiteDomain()+"/s/46ccbef30000616306614050/1 http://"+self.getSiteDomain()+"/s/46ccbef30000616306614050/1/story-name"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/s/46ccbef30000616306614050 http://"+cls.getSiteDomain()+"/s/46ccbef30000616306614050/1 http://"+cls.getSiteDomain()+"/s/46ccbef30000616306614050/1/story-name"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/s/")+r"\w+(/\d+)?"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data):
if 'Diese Geschichte wurde als entwicklungsbeeintr' in data \
@@ -99,9 +106,8 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
loginUrl = 'https://ssl.fanfiktion.de/'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['nickname']))
d = self._postUrl(loginUrl,params)
if "Login erfolgreich" not in d : #Member Account
soup = bs.BeautifulSoup(self._postUrl(loginUrl,params))
if not soup.find('a', title='Logout'):
logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['nickname']))
raise exceptions.FailedToLogin(url,params['nickname'])
@@ -126,7 +132,7 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
if self.needToLoginCheck(data):
# need to log in for this one.
self.performLogin(url)
data = self._fetchUrl(url)
data = self._fetchUrl(url,usecache=False)
if "Uhr ist diese Geschichte nur nach einer" in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Auserhalb der Zeit von 23:00 Uhr bis 04:00 Uhr ist diese Geschichte nur nach einer erfolgreichen Altersverifikation zuganglich.")
@@ -142,7 +148,7 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
head = soup.find('div', {'class' : 'story-metadata-left-top'})
head = soup.find('div', {'class' : 'story-left'})
a = head.find('a')
self.story.setMetadata('authorId',a['href'].split('/')[2])
self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href'])
@@ -156,14 +162,18 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
self.story.setMetadata('language','German')
#find metadata on the story page
self.story.setMetadata('datePublished', makeDate(head.text.split('erstellt: ')[1].split('\n')[0], self.dateformat))
headtext = stripHTML(head)
self.story.setMetadata('datePublished', makeDate(headtext.split('erstellt: ')[1].split('\n')[0], self.dateformat))
self.story.setMetadata('dateUpdated', makeDate(head.text.split('letztes Update: ')[1].split('\n')[0], self.dateformat))
for genre in head.text.split('&nbsp;&nbsp;&nbsp;')[3].split('/')[0].split(', '):
self.story.addToList('genre',genre)
self.story.setMetadata('dateUpdated', makeDate(headtext.split('aktualisiert: ')[1].split('\n')[0], self.dateformat))
# second colspan=3 td in head.
genres=stripHTML(head.findAll('td',{'colspan':'3'})[1])
self.story.extendList('genre',genres[:genres.index('(')].split(', '))
# for genre in head.text.split('&nbsp;&nbsp;&nbsp;')[3].split('/')[0].split(', '):
# self.story.addToList('genre',genre)
if 'fertiggestellt' in head.text:
if 'fertiggestellt' in headtext:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In Progress')
@@ -178,9 +188,9 @@ class FanFiktionDeAdapter(BaseSiteAdapter):
self.setDescription(url,a['onmouseover'].split("', '")[1])
td = tr[i].findAll('td')
self.story.addToList('category',stripHTML(td[1]))
self.story.setMetadata('rating', stripHTML(td[4]))
self.story.setMetadata('numWords', stripHTML(td[5]))
self.story.addToList('category',stripHTML(td[2]))
self.story.setMetadata('rating', stripHTML(td[5]))
self.story.setMetadata('numWords', stripHTML(td[6]))
# grab the text for an individual chapter.
@@ -0,0 +1,44 @@
# -*- coding: utf-8 -*-
# Copyright 2014 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
import re
from base_efiction_adapter import BaseEfictionAdapter
class FanNationAdapter(BaseEfictionAdapter):
@staticmethod
def getSiteDomain():
return 'fannation.shades-of-moonlight.com'
@classmethod
def getPathToArchive(self):
return '/archive'
@classmethod
def getSiteAbbrev(self):
return 'fannation'
def handleMetadataPair(self, key, value):
if key == 'Romance':
for val in re.split("\s*,\s*", value):
self.story.addToList('romance', val)
else:
super(FanNationAdapter, self).handleMetadataPair(key, value)
def getClass():
return FanNationAdapter
@@ -70,8 +70,8 @@ class FicBookNetAdapter(BaseSiteAdapter):
return 'www.ficbook.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/readfic/12345 http://"+self.getSiteDomain()+"/readfic/93626/246417#part_content"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/readfic/12345 http://"+cls.getSiteDomain()+"/readfic/93626/246417#part_content"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/readfic/")+r"\d+"
@@ -58,8 +58,8 @@ class FictionAlleyOrgSiteAdapter(BaseSiteAdapter):
return 'www.fictionalley.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/authors/drt/DA.html http://"+self.getSiteDomain()+"/authors/drt/JOTP01a.html"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/authors/drt/DA.html http://"+cls.getSiteDomain()+"/authors/drt/JOTP01a.html"
def getSiteURLPattern(self):
# http://www.fictionalley.org/authors/drt/DA.html
@@ -57,7 +57,7 @@ class FictionPadSiteAdapter(BaseSiteAdapter):
return 'fictionpad.com'
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "https://fictionpad.com/author/Author/stories/1234/Some-Title"
def getSiteURLPattern(self):
@@ -40,7 +40,7 @@ class FictionPressComSiteAdapter(FanFictionNetSiteAdapter):
return ['www.fictionpress.com','m.fictionpress.com']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "https://www.fictionpress.com/s/1234/1/ https://www.fictionpress.com/s/1234/12/ http://www.fictionpress.com/s/1234/1/Story_Title http://m.fictionpress.com/s/1234/1/"
def getSiteURLPattern(self):
+29 -16
View File
@@ -46,7 +46,7 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
return 'ficwad.com'
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://ficwad.com/story/1234"
def getSiteURLPattern(self):
@@ -65,7 +65,7 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
loginUrl = 'http://' + self.getSiteDomain() + '/account/login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['username']))
d = self._postUrl(loginUrl,params)
d = self._postUrl(loginUrl,params,usecache=False)
if "Login attempt failed..." in d:
logger.info("Failed to login to URL %s as %s" % (loginUrl,
@@ -75,6 +75,13 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
else:
return True
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def extractChapterUrlsAndMetadata(self):
# fetch the chapter. From that we will get almost all the
@@ -96,9 +103,15 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
else:
raise e
h3 = soup.find('h3')
storya = h3.find('a',href=re.compile("^/story/\d+$"))
if storya : # if there's a story link in the h3 header, this is a chapter page.
# if blocked, attempt login.
if soup.find("div",{"class":"blocked"}):
if self.performLogin(url): # performLogin raises
# FailedToLogin if it fails.
soup = bs.BeautifulSoup(self._fetchUrl(url,usecache=False))
divstory = soup.find('div',id='story')
storya = divstory.find('a',href=re.compile("^/story/\d+$"))
if storya : # if there's a story link in the divstory header, this is a chapter page.
# normalize story URL on chapter list.
self.story.setMetadata('storyId',storya['href'].split('/',)[2])
url = "http://"+self.getSiteDomain()+storya['href']
@@ -113,17 +126,17 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
raise e
# if blocked, attempt login.
if soup.find("li",{"class":"blocked"}):
if soup.find("div",{"class":"blocked"}):
if self.performLogin(url): # performLogin raises
# FailedToLogin if it fails.
soup = bs.BeautifulSoup(self._fetchUrl(url))
soup = bs.BeautifulSoup(self._fetchUrl(url,usecache=False))
# title - first h4 tag will be title.
titleh4 = soup.find('h4')
titleh4 = soup.find('div',{'class':'storylist'}).find('h4')
self.story.setMetadata('title', stripHTML(titleh4.a))
# Find authorid and URL from... author url.
a = soup.find('a', href=re.compile(r"^/author/\d+"))
a = soup.find('span',{'class':'author'}).find('a', href=re.compile(r"^/author/\d+"))
self.story.setMetadata('authorId',a['href'].split('/')[2])
self.story.setMetadata('authorUrl','http://'+self.host+a['href'])
self.story.setMetadata('author',a.string)
@@ -139,7 +152,7 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
# warnings
# <span class="req"><a href="/help/38" title="Medium Spoilers">[!!] </a> <a href="/help/38" title="Rape/Sexual Violence">[R] </a> <a href="/help/38" title="Violence">[V] </a> <a href="/help/38" title="Child/Underage Sex">[Y] </a></span>
spanreq = metap.find("span",{"class":"req"})
spanreq = metap.find("span",{"class":"story-warnings"})
if spanreq: # can be no warnings.
for a in spanreq.findAll("a"):
self.story.addToList('warnings',a['title'])
@@ -165,16 +178,16 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
if g:
self.story.addToList('characters',g)
m = re.match(r".*?Published: ([0-9/]+?) -.*?",metastr)
m = re.match(r".*?Published: ([0-9-]+?) -.*?",metastr)
if m:
self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y/%m/%d"))
self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y-%m-%d"))
# Updated can have more than one space after it. <shrug>
m = re.match(r".*?Updated: ([0-9/]+?) +-.*?",metastr)
m = re.match(r".*?Updated: ([0-9-]+?) +-.*?",metastr)
if m:
self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y/%m/%d"))
self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y-%m-%d"))
m = re.match(r".*? - ([0-9/]+?) words.*?",metastr)
m = re.match(r".*? - ([0-9,]+?) words.*?",metastr)
if m:
self.story.setMetadata('numWords',m.group(1))
@@ -185,7 +198,7 @@ class FicwadComSiteAdapter(BaseSiteAdapter):
# get the chapter list first this time because that's how we
# detect the need to login.
storylistul = soup.find('ul',{'id':'storylist'})
storylistul = soup.find('ul',{'class':'storylist'})
if not storylistul:
# no list found, so it's a one-chapter story.
self.chapterUrls.append((self.story.getMetadata('title'),url))
+142 -119
View File
@@ -55,16 +55,22 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
return ['www.fimfiction.net','mobile.fimfiction.net', 'www.fimfiction.com', 'mobile.fimfiction.com']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://www.fimfiction.net/story/1234/story-title-here http://www.fimfiction.net/story/1234/ http://www.fimfiction.com/story/1234/1/ http://mobile.fimfiction.net/story/1234/1/story-title-here/chapter-title-here"
def getSiteURLPattern(self):
return r"https?://(www|mobile)\.fimfiction\.(net|com)/story/\d+/?.*"
def extractChapterUrlsAndMetadata(self):
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
def doExtractChapterUrlsAndMetadata(self,get_cover=True):
if self.is_adult or self.getConfig("is_adult"):
cookieproc = urllib2.HTTPCookieProcessor()
cookie = cl.Cookie(version=0, name='view_mature', value='true',
port=None, port_specified=False,
domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False,
@@ -76,15 +82,12 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
comment_url=None,
rest={'HttpOnly': None},
rfc2109=False)
cookieproc.cookiejar.set_cookie(cookie)
self.opener = urllib2.build_opener(cookieproc)
self.cookiejar.set_cookie(cookie)
##---------------------------------------------------------------------------------------------------
## Get the story's title page. Check if it exists.
try:
apiResponse = urllib2.urlopen("http://www.fimfiction.net/api/story.php?story=%s" % (self.story.getMetadata("storyId"))).read()
apiData = json.loads(apiResponse)
# Unfortunately, we still need to load the story index
# page to parse the characters. And chapters, now, too.
data = self.do_fix_blockquotes(self._fetchUrl(self.url))
soup = bs.BeautifulSoup(data)
except urllib2.HTTPError, e:
@@ -96,109 +99,97 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
if "Warning: mysql_fetch_array(): supplied argument is not a valid MySQL result resource" in data:
raise exceptions.StoryDoesNotExist(self.url)
# Can cause problems if a missing story is referenced in a comment.
# Shouldn't be needed anyway.
# if "/images/missing_story.png" in data:
# raise exceptions.StoryDoesNotExist(self.url)
if "This story has been marked as having adult content. Please click below to confirm you are of legal age to view adult material in your country." in data:
raise exceptions.AdultCheckRequired(self.url)
if self.password:
params = {}
params['password'] = self.password
data = self._postUrl(self.url,params)
data = self._postUrl(self.url, params)
soup = bs.BeautifulSoup(data)
if "Enter the password the author set for this story to view it." in data:
if not (soup.find('form', {'id' : 'password_form'}) == None):
if self.getConfig('fail_on_password'):
raise exceptions.FailedToDownload("%s requires story password and fail_on_password is true."%self.url)
else:
raise exceptions.FailedToLogin(self.url,"Story requires individual password",passwdonly=True)
if "Invalid story id" in apiData.values():
raise exceptions.StoryDoesNotExist(self.url)
storyMetadata = apiData["story"]
## Title
a = soup.find('a', href=re.compile(r'^/story/'+self.story.getMetadata('storyId')))
self.story.setMetadata('title',stripHTML(a))
# self.story.setMetadata("title", storyMetadata["title"])
# if not storyMetadata["title"]:
# raise exceptions.FailedToDownload("%s doesn't have a title in the API. This is a known fimfiction.net bug with titles containing ."%self.url)
self.story.setMetadata("author", storyMetadata["author"]["name"])
self.story.setMetadata("authorId", storyMetadata["author"]["id"])
self.story.setMetadata("authorUrl", "http://%s/user/%s" % (self.getSiteDomain(), storyMetadata["author"]["name"]))
# chapters = [{"chapterTitle": chapter["title"], "chapterURL": chapter["link"]} for chapter in storyMetadata["chapters"]]
# ## this is bit of a kludge based on the assumption all the
# ## 'bad' chapters will be at the end.
# ## limit down to the number of chapters reported by chapter_count.
# chapters = chapters[:storyMetadata["chapter_count"]]
# for chapter in chapters:
# self.chapterUrls.append((chapter["chapterTitle"], chapter["chapterURL"]))
# self.story.setMetadata("numChapters", len(self.chapterUrls))
##----------------------------------------------------------------------------------------------------
## Extract metadata
for chapter in soup.findAll('a',{'class':'chapter_link'}):
storyContentBox = soup.find('div', {'class':'story_content_box'})
# Title
title = storyContentBox.find('a', {'class':re.compile(r'.*\bstory_name\b.*')})
self.story.setMetadata('title',stripHTML(title))
# Author
author = storyContentBox.find('span', {'class':'author'})
self.story.setMetadata("author", stripHTML(author))
#No longer seems to be a way to access Fimfiction's internal author ID
self.story.setMetadata("authorId", self.story.getMetadata("author"))
self.story.setMetadata("authorUrl", "http://%s/user/%s" % (self.getSiteDomain(), stripHTML(author)))
#Rating text is replaced with full words for historical compatibility after the site changed
#on 2014-10-27
rating = stripHTML(storyContentBox.find('a', {'class':re.compile(r'.*\bcontent-rating-.*')}))
rating = rating.replace("E", "Everyone").replace("T", "Teen").replace("M", "Mature")
self.story.setMetadata("rating", rating)
# Chapters
for chapter in storyContentBox.findAll('a',{'class':'chapter_link'}):
self.chapterUrls.append((stripHTML(chapter), 'http://'+self.host+chapter['href']))
self.story.setMetadata('numChapters',len(self.chapterUrls))
# In the case of fimfiction.net, possible statuses are 'Completed', 'Incomplete', 'On Hiatus' and 'Cancelled'
# Status
# In the case of Fimfiction, possible statuses are 'Completed', 'Incomplete', 'On Hiatus' and 'Cancelled'
# For the sake of bringing it in line with the other adapters, 'Incomplete' becomes 'In-Progress'
# and 'Complete' beomes 'Completed'. 'Cancelled' seems an important enough (not to mention more strictly true)
# status to leave unchanged.
# Nov2012 - 'On Hiatus' is now passed, too. It's easy now for users to change/remove if they want
# with replace_metadata
status = storyMetadata["status"].replace("Incomplete", "In-Progress").replace("Complete", "Completed")
# and 'Complete' becomes 'Completed'. 'Cancelled' and 'On Hiatus' are passed through, it's easy now for users
# to change/remove if they want with replace_metadata
status = stripHTML(storyContentBox.find('span', {'class':re.compile(r'.*\bcompleted-status-.*')}))
status = status.replace("Incomplete", "In-Progress").replace("Complete", "Completed")
self.story.setMetadata("status", status)
self.story.setMetadata("rating", storyMetadata["content_rating_text"])
## Warnings aren't included in the API.
bottomli = soup.find('li',{'class':'bottom'})
if bottomli:
bottomspans = bottomli.findAll('span')
# the first span in bottom is the rating, obtained above.
if bottomspans and len(bottomspans) > 1:
for warning in bottomspans[1:]:
self.story.addToList('warnings',warning.string)
for category in storyMetadata["categories"]:
if storyMetadata["categories"][category]:
self.story.addToList("genre", category)
self.story.setMetadata("numWords", str(storyMetadata["words"]))
# fimfic is the first site with an explicit cover image.
if "image" in storyMetadata.keys():
if "full_image" in storyMetadata:
coverurl = storyMetadata["full_image"]
# Genres and Warnings
# warnings were folded into general categories in the 2014-10-27 site update
categories = storyContentBox.findAll('a', {'class':re.compile(r'.*\bstory_category\b.*')})
for category in categories:
category = stripHTML(category)
if category == "Gore" or category == "Sex":
self.story.addToList('warnings', category)
else:
coverurl = storyMetadata["image"]
self.story.addToList('genre', category)
# Word count
wordCountText = stripHTML(storyContentBox.find('li', {'class':'bottom'}).find('div', {'class':'word_count'}))
self.story.setMetadata("numWords", re.sub(r'[^0-9]', '', wordCountText))
# Cover image
storyImage = storyContentBox.find('div', {'class':'story_image'})
if storyImage:
coverurl = storyImage.find('a')['href']
if coverurl.startswith('//'): # fix for img urls missing 'http:'
coverurl = "http:"+coverurl
if get_cover:
self.setCoverImage(self.url,coverurl)
self.setCoverImage(self.url,coverurl)
coverSource = storyImage.find('a', {'class':'source'})
if coverSource:
self.story.setMetadata('coverSourceUrl', coverSource['href'])
#There's no text associated with the cover source link, so just
#reuse the URL. Makes it clear it's an external link leading
#outside of the fanfic site, at least.
self.story.setMetadata('coverSource', coverSource['href'])
# fimf has started including extra stuff inside the description div.
descdivstr = u"%s"%soup.find("div", {"class":"description"})
descdivstr = u"%s"%storyContentBox.find("div", {"class":"description"})
hrstr=u"<hr />"
descdivstr = u'<div class="description">'+descdivstr[descdivstr.index(hrstr)+len(hrstr):]
self.setDescription(self.url,descdivstr)
# Can't trust dates from API anymore I'm told.
# Dates are in Unix time
# Take the publish date from the first chapter posted
# rawDatePublished = storyMetadata["chapters"][0]["date_modified"]
# self.story.setMetadata("datePublished", datetime.fromtimestamp(rawDatePublished))
# rawDateUpdated = storyMetadata["date_modified"]
# self.story.setMetadata("dateUpdated", datetime.fromtimestamp(rawDateUpdated))
# Find the newest and oldest chapter dates
storyData = storyContentBox.find('div', {'class':'story_data'})
oldestChapter = None
newestChapter = None
self.newestChapterNum = None # save for comparing during update.
@@ -206,18 +197,21 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
# FiMFiction it's possible for authors to insert new chapters
# out-of-order or change the dates of earlier ones by editing
# them--That WILL break epub update.
for index, chapterDate in enumerate(soup.findAll('span', {'class':'date'})):
date=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",chapterDate.contents[1].strip())
chapterDate = makeDate(date,self.dateformat)
for index, chapterDate in enumerate(storyData.findAll('span', {'class':'date'})):
dateString=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",chapterDate.contents[1].strip())
chapterDate = makeDate(dateString,self.dateformat)
if oldestChapter == None or chapterDate < oldestChapter:
oldestChapter = chapterDate
if newestChapter == None or chapterDate > newestChapter:
newestChapter = chapterDate
self.newestChapterNum = index
# Date updated
self.story.setMetadata("dateUpdated", newestChapter)
pubdatetag = soup.find('span', {'class':'date_approved'})
# Date published
# falls back to oldest chapter date for stories that haven't been officially published yet
pubdatetag = storyContentBox.find('span', {'class':'date_approved'})
if pubdatetag is None:
self.story.setMetadata("datePublished", oldestChapter)
else:
@@ -225,37 +219,48 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
datestripped=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",pubdateraw.strip())
pubDate = makeDate(datestripped,self.dateformat)
self.story.setMetadata("datePublished", pubDate)
chars = soup.find("div", {"class":"inner_data"})
# fimfic stopped putting the char name on or around the char
# icon now for some reason. Pull it from the image name with
# some heuristics.
for character in [character_icon["src"] for character_icon in chars.findAll("img", {"class":"character_icon"})]:
# //static.fimfiction.net/images/characters/twilight_sparkle.png
# 5th split /, remove last four, replace _, capitolize every word(title())
char = character.split('/')[5][:-4].replace('_',' ').title()
if char == 'Oc':
char = "OC"
if char == 'Cmc':
char = "Cutie Mark Crusaders"
self.story.addToList("characters", char)
# extra site specific metadata
extralist = ["likes","dislikes","views","total_views","short_description"]
for metakey in extralist:
if metakey in storyMetadata:
value = storyMetadata[metakey]
if not isinstance(value,basestring):
value = unicode(value)
self.story.setMetadata(metakey, value)
## Groups and sequels code from FaceDeer
allGroupLists = soup.findAll('ul', {'id':'story_group_list'})
for groupList in allGroupLists:
for groupName in groupList.findAll('a', {'href':re.compile('^/group/')}):
# Characters
chars = storyContentBox.find("div", {"class":"extra_story_data"})
for character in chars.findAll("a", {"class":"character_icon"}):
self.story.addToList("characters", character['title'])
# Likes and dislikes
storyToolbar = soup.find('div', {'class':'story-toolbar'})
likes = storyToolbar.find('span', {'class':'likes'})
if not likes is None:
self.story.setMetadata("likes", stripHTML(likes))
dislikes = storyToolbar.find('span', {'class':'dislikes'})
if not dislikes is None:
self.story.setMetadata("dislikes", stripHTML(dislikes))
# Highest view for a chapter and total views
viewSpan = storyToolbar.find('span', {'title':re.compile(r'.*\btotal views\b.*')})
self.story.setMetadata("views", re.sub(r'[^0-9]', '', stripHTML(viewSpan)))
self.story.setMetadata("total_views", re.sub(r'[^0-9]', '', viewSpan['title']))
# Comment count
commentSpan = storyToolbar.find('span', {'title':re.compile(r'.*\bcomments\b.*')})
self.story.setMetadata("comment_count", re.sub(r'[^0-9]', '', stripHTML(commentSpan)))
# Short description
descriptionMeta = soup.find('meta', {'property':'og:description'})
self.story.setMetadata("short_description", stripHTML(descriptionMeta['content']))
#groups
if soup.find('button', {'id':'button-view-all-groups'}):
groupResponse = self._fetchUrl("http://www.fimfiction.net/ajax/groups/story_groups_list.php?story=%s" % (self.story.getMetadata("storyId")))
groupData = json.loads(groupResponse)
groupList = bs.BeautifulSoup(groupData["content"])
else:
groupList = soup.find('ul', {'id':'story-groups-list'})
if not (groupList == None):
for groupName in groupList.findAll('a'):
self.story.addToList("groupsUrl", 'http://'+self.host+groupName["href"])
self.story.addToList("groups",stripHTML(groupName).replace(',', ';'))
#sequels
sequelStoryHeader = soup.find('h1', {'class':'header-stories'}, text="Sequels")
if not sequelStoryHeader == None:
sequelContainer = sequelStoryHeader.parent.parent
@@ -296,9 +301,27 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
data = self.do_fix_blockquotes(self._fetchUrl(url))
data = self._fetchUrl(url)
soup = bs.BeautifulSoup(data)
if not (soup.find('form', {'id' : 'password_form'}) == None):
if self.password:
params = {}
params['password'] = self.password
data = self._postUrl(url, params)
else:
print("Chapter %s needed password but no password was present" % url)
data = self.do_fix_blockquotes(data)
soup = bs.BeautifulSoup(data,selfClosingTags=('br','hr')).find('div', {'class' : 'chapter_content'})
if soup == None:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
# fix for img urls missing 'http:'
images = soup.findAll('img')
for imagetag in images:
if 'src' in imagetag.attrs and imagetag['src'].startswith('//'):
imagetag['src'] = "http:"+imagetag['src']
return self.utf8FromSoup(url,soup)
@@ -62,8 +62,8 @@ class FineStoriesComAdapter(BaseSiteAdapter):
return 'finestories.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/s/1234 http://"+self.getSiteDomain()+"/s/1234:4010 http://"+self.getSiteDomain()+"/library/storyInfo.php?id=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/s/1234 http://"+cls.getSiteDomain()+"/s/1234:4010 http://"+cls.getSiteDomain()+"/library/storyInfo.php?id=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain())+r"/(s|library)?/(storyInfo.php\?id=)?\d+(:\d+)?(;\d+)?$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -75,7 +76,7 @@ class GrangerEnchantedCom(BaseSiteAdapter):
return ['grangerenchanted.com','malfoymanor.grangerenchanted.com']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://grangerenchanted.com/enchant/viewstory.php?sid=1234 http://malfoymanor.grangerenchanted.com/themanor/viewstory.php?sid=1234"
def getSiteURLPattern(self):
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -57,7 +58,7 @@ class HarryPotterFanFictionComSiteAdapter(BaseSiteAdapter):
return ['www.harrypotterfanfiction.com','harrypotterfanfiction.com']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://www.harrypotterfanfiction.com/viewstory.php?psid=1234"
def getSiteURLPattern(self):
@@ -66,8 +66,8 @@ class HennethAnnunNetAdapter(BaseSiteAdapter):
return 'www.henneth-annun.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/stories/chapter.cfm?stid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/stories/chapter.cfm?stid=1234"
def getSiteURLPattern(self):
return "http://"+self.getSiteDomain()+"/stories/chapter(_view)?.cfm\?stid="+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class HLFictionNetAdapter(BaseSiteAdapter):
return 'hlfiction.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -71,8 +72,8 @@ class HPFandomNetAdapterAdapter(BaseSiteAdapter): # XXX
return 'www.hpfandom.net' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/eff/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/eff/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/eff/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class HPFanficArchiveComAdapter(BaseSiteAdapter):
return 'www.hpfanficarchive.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/stories/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/stories/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/stories/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class IkEternalNetAdapter(BaseSiteAdapter):
return 'www.ik-eternal.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class ImagineEFicComAdapter(BaseSiteAdapter):
return 'imagine.e-fic.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -73,8 +73,8 @@ class InDeathNetAdapter(BaseSiteAdapter):
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/blog/archive/123-story-in-death/"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/blog/archive/123-story-in-death/"
def getSiteURLPattern(self):
# http://www.indeath.net/blog/archive/169-ransom-in-death/
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -75,8 +76,8 @@ class KSArchiveComAdapter(BaseSiteAdapter): # XXX
return 'ksarchive.com' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class LibraryOfMoriaComAdapter(BaseSiteAdapter):
return 'www.libraryofmoria.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/a/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/a/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/a/viewstory.php?sid=")+r"\d+$"
+110 -104
View File
@@ -39,64 +39,97 @@ class LiteroticaSiteAdapter(BaseSiteAdapter):
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.story.setMetadata('siteabbrev','litero')
# normalize to first chapter. Not sure if they ever have more than 2 digits.
storyid = self.parsedUrl.path.split('/',)[2]
if re.match(r'-ch\d\d$',storyid):
storyid = storyid[:-2]+'01'
self.story.setMetadata('storyId',storyid)
self.origurl = url
if "//www.i." in self.origurl:
## accept m(mobile)url, but use www.
self.origurl = self.origurl.replace("//www.i.","//www.")
storyId = self.parsedUrl.path.split('/',)[2]
# replace later chapters with first chapter but don't remove numbers
# from the URL that disambiguate stories with the same title.
storyId = re.sub("-ch-?\d\d", "", storyId)
self.story.setMetadata('storyId', storyId)
# normalized story URL.
self._setURL(url[:url.index('//')+2]+self.getSiteDomain()\
+"/s/"+self.story.getMetadata('storyId'))
## accept m(mobile)url, but use www.
url = re.sub("^(www|german|spanish|french|dutch|italian|romanian|portuguese|other)\.i",
"\1",
url)
## strip ?page=...
url = re.sub("\?page=.*$", "", url)
## set url
self._setURL(url)
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = '%m/%d/%y'
@staticmethod
def getSiteDomain():
return 'www.literotica.com'
return 'literotica.com'
@classmethod
def getAcceptDomains(cls):
return ['www.literotica.com', 'www.i.literotica.com']
return ['www.literotica.com',
'www.i.literotica.com',
'german.literotica.com',
'german.i.literotica.com',
'spanish.literotica.com',
'spanish.i.literotica.com',
'french.literotica.com',
'french.i.literotica.com',
'dutch.literotica.com',
'dutch.i.literotica.com',
'italian.literotica.com',
'italian.i.literotica.com',
'romanian.literotica.com',
'romanian.i.literotica.com',
'portuguese.literotica.com',
'portuguese.i.literotica.com',
'other.literotica.com',
'other.i.literotica.com']
@classmethod
def getSiteExampleURLs(self):
#return "http://www.literotica.com/s/story-title http://www.literotica.com/stories/showstory.php?id=1234 http://www.i.literotica.com/stories/showstory.php?id=1234"
return "http://www.literotica.com/s/story-title https://www.literotica.com/s/story-title"
def getSiteExampleURLs(cls):
return "http://www.literotica.com/s/story-title https://www.literotica.com/s/story-title http://portuguese.literotica.com/s/story-title http://german.literotica.com/s/story-title"
def getSiteURLPattern(self):
return r"https?://www(\.i)?\.literotica\.com/s/([a-zA-Z0-9_-]+)"
return r"https?://(www|german|spanish|french|dutch|italian|romanian|portuguese|other)(\.i)?\.literotica\.com/s/([a-zA-Z0-9_-]+)"
def extractChapterUrlsAndMetadata(self):
"""
NOTE: Some stories can have versions,
e.g. /my-story-ch-05-version-10
NOTE: If two stories share the same title, a running index is added,
e.g.: /my-story-ch-02-1
Strategy:
* Go to author's page, search for the current story link,
* If it's in a tr.root-story => One-part story
* , get metadata and be done
* If it's in a tr.sl => Chapter in series
* Search up from there until we find a tr.ser-ttl (this is the
story)
* Gather metadata
* Search down from there for all tr.sl until the next
tr.ser-ttl, foreach
* Chapter link is there
"""
if not (self.is_adult or self.getConfig("is_adult")):
raise exceptions.AdultCheckRequired(self.url)
url1 = self.origurl
logger.debug("first page URL: "+url1)
logger.debug("Chapter/Story URL: <%s> " % self.url)
try:
data1 = self._fetchUrl(url1)
data1 = self._fetchUrl(self.url)
soup1 = bs.BeautifulSoup(data1)
#strip comments from soup
[comment.extract() for comment in soup1.findAll(text=lambda text:isinstance(text, bs.Comment))]
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(url1)
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
#strip comments from soup
[comment.extract() for comment in soup1.findAll(text=lambda text:isinstance(text, bs.Comment))]
# author
a = soup1.find("span", "b-story-user-y")
self.story.setMetadata('authorId', urlparse.parse_qs(a.a['href'].split('?')[1])['uid'][0])
@@ -110,105 +143,78 @@ class LiteroticaSiteAdapter(BaseSiteAdapter):
try:
dataAuth = self._fetchUrl(authorurl)
soupAuth = bs.BeautifulSoup(dataAuth)
#strip comments from soup
[comment.extract() for comment in soupAuth.findAll(text=lambda text:isinstance(text, bs.Comment))]
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(authorurl)
else:
raise e
## Find link to url in author's page
## site has started using //domain.name/asdf urls remove https?: from front
storyLink = soupAuth.find('a', href=url1[url1.index(':')+1:])
storyLink = soupAuth.find('a', href=self.url[self.url.index(':')+1:])
if storyLink is not None:
# pull the published date from the author page
# default values from single link. Updated below if multiple chapter.
date = storyLink.parent.parent.findAll('td')[-1].text
urlTr = storyLink.parent.parent
if urlTr['class'] == "sl":
isSingleStory = False
else:
isSingleStory = True
else:
raise exceptions.FailedToDownload("Couldn't find story <%s> on author's page <%s>" % (url, authorurl))
if isSingleStory:
self.story.setMetadata('title', storyLink.text)
self.story.setMetadata('description', urlTr.findAll("td")[1].text)
self.story.addToList('eroticatags', urlTr.findAll("td")[2].text)
date = urlTr.findAll('td')[-1].text
self.story.setMetadata('datePublished', makeDate(date, self.dateformat))
self.story.setMetadata('dateUpdated',makeDate(date, self.dateformat))
# find num of pages
# find a "3 Pages:" string on the page and parse it
pgs = soup1.find("span", "b-pager-caption-t r-d45").string.split(' ')[0]
# If there are multiple pages, find and request the last page
if "1" != pgs:
logger.debug("last page number: "+pgs)
try:
data2 = self._fetchUrl(url1, {'page': pgs})
soup2 = bs.BeautifulSoup(data2)
[comment.extract() for comment in soup2.findAll(text=lambda text:isinstance(text, bs.Comment))]
except urllib2.HTTPError, e:
if e.code == 404:
# TODO: Probably should reformat this
raise exceptions.StoryDoesNotExist(url1, {'page': pgs})
else:
raise e
self.chapterUrls = [(storyLink.text, self.url)]
else:
#If we're already on the last page, copy the soup
soup2 = soup1
seriesTr = urlTr.previousSibling
while seriesTr['class'] != 'ser-ttl':
seriesTr = seriesTr.previousSibling
m = re.match("^(?P<title>.*?):\s(?P<numChapters>\d+)\sPart\sSeries$", seriesTr.find("strong").text)
self.story.setMetadata('title', m.group('title'))
# parse out the list of chapters
chaps = soup2.find('div', id='b-series')
if chaps: # may be one post only
#self.chapterUrls = [(ch.a.text, ch.a['href']) for ch in chaps.findAll('li')]
# if there are chapters, lets pull them and title from the
# author page because *this* chapter is omitted from the
# list on the last page.
row = storyLink.parent.parent.previousSibling
while row['class'] != 'ser-ttl':
row = row.previousSibling
seriesTitle = stripHTML(row)
if seriesTitle:
# this regex is deliberately greedy. We want to get the biggest match before a ':'
self.story.setMetadata('title', re.match('(.*):[^:]*$', seriesTitle).group(1))
else:
self.story.setMetadata('title', soup1.h1.string)
# now chapter list. Assumed oldest to newest.
## Walk the chapters
chapterTr = seriesTr.nextSibling
self.chapterUrls = []
row = row.nextSibling
self.story.setMetadata('datePublished',makeDate(stripHTML(row.find('td',{'class':'dt'})), self.dateformat))
while row['class'] == 'sl':
# pages include full URLs.
chapurl = row.a['href']
if chapurl.startswith('//'):
chapurl = self.parsedUrl.scheme+':'+chapurl
self.chapterUrls.append((row.a.string,chapurl))
if not row.nextSibling:
break
row = row.nextSibling
dates = []
descriptions = []
while chapterTr is not None and chapterTr['class'] == 'sl':
descriptions.append(chapterTr.findAll("td")[1].text)
chapterLink = chapterTr.find("td", "fc").find("a")
self.chapterUrls.append((chapterLink.text, "http:" + chapterLink["href"]))
self.story.addToList('eroticatags', chapterTr.findAll("td")[2].text)
dates.append(makeDate(chapterTr.findAll('td')[-1].text, self.dateformat))
chapterTr = chapterTr.nextSibling
row = row.previousSibling
self.story.setMetadata('dateUpdated',makeDate(stripHTML(row.find('td',{'class':'dt'})), self.dateformat))
else: # if one post only
self.chapterUrls = [(soup1.h1.string, url1)]
self.story.setMetadata('title', soup1.h1.string)
## Set description to joint chapter descriptions
self.story.setMetadata('description', " / ".join(descriptions))
# normalize on first chapter URL.
self._setURL(self.chapterUrls[0][1])
## Set the oldest date as publication date, the newest as update date
dates.sort()
self.story.setMetadata('datePublished', dates[0])
self.story.setMetadata('dateUpdated', dates[-1])
# reset storyId to first chapter.
self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2])
# normalize on first chapter URL.
self._setURL(self.chapterUrls[0][1])
self.story.setMetadata('numChapters', len(self.chapterUrls))
self.story.setMetadata('category', soup1.find('div', 'b-breadcrumbs').findAll('a')[1].string)
# deliberately not self.setDescription() because it's never HTML.
self.story.setMetadata('description', soup1.find('meta', {'name': 'description'})['content'])
# li tags inside div class b-s-story-tag-list
for li in soup1.find('div', {'class':'b-s-story-tag-list'}).findAll('a'):
self.story.addToList('eroticatags',stripHTML(li))
# set storyId to 'title-author' to avoid duplicates
# self.story.setMetadata('storyId',
# re.sub("[^a-z0-9]", "", self.story.getMetadata('title').lower())
# + "-"
# + re.sub("[^a-z0-9]", "", self.story.getMetadata('author').lower()))
return
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
time.sleep(0.5)
logger.debug('Getting chapter text from <%s>' % url)
data1 = self._fetchUrl(url)
soup1 = bs.BeautifulSoup(data1)
@@ -0,0 +1,36 @@
# -*- coding: utf-8 -*-
# Copyright 2014 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Software: eFiction
from base_efiction_adapter import BaseEfictionAdapter
class TheLOTRFanFictionSiteAdapter(BaseEfictionAdapter):
@staticmethod
def getSiteDomain():
return 'lotrfanfiction.com'
@classmethod
def getSiteAbbrev(seluuf):
return 'lotrff'
@classmethod
def getDateFormat(self):
return "%d/%m/%y"
def getClass():
return TheLOTRFanFictionSiteAdapter
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class LumosSycophantHexComAdapter(BaseSiteAdapter):
return 'lumos.sycophanthex.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -56,8 +56,8 @@ class MediaMinerOrgSiteAdapter(BaseSiteAdapter):
return 'www.mediaminer.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfic/view_st.php/123456 http://"+self.getSiteDomain()+"/fanfic/view_ch.php/1234123/123444#fic_c"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfic/view_st.php/123456 http://"+cls.getSiteDomain()+"/fanfic/view_ch.php/1234123/123444#fic_c"
def getSiteURLPattern(self):
## http://www.mediaminer.org/fanfic/view_st.php/76882
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class MerlinFicDtwinsCoUk(BaseSiteAdapter):
return 'merlinfic.dtwins.co.uk'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -71,8 +72,8 @@ class MidnightwhispersCaAdapter(BaseSiteAdapter): # XXX
return 'www.midnightwhispers.ca' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -74,8 +75,8 @@ class MuggleNetComAdapter(BaseSiteAdapter): # XXX
return ['fanfiction.mugglenet.com','fanfic.mugglenet.com']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+r"fanfic(tion)?\.mugglenet\.com"+re.escape("/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -69,9 +70,9 @@ class NationalLibraryNetAdapter(BaseSiteAdapter):
return ['www.national-library.net','national-library.net']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
# ONLY the stories archived on or after June 17, 2006 and that are hosted on the website:
return "http://"+self.getSiteDomain()+"/viewstory.php?storyid=1234"
return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -69,8 +70,8 @@ class NCISFicComAdapter(BaseSiteAdapter):
return ['www.ncisfic.com','ncisfic.com']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?storyid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$"
@@ -75,8 +75,8 @@ class NCISFictionNetAdapter(BaseSiteAdapter):
return ['www.ncisfiction.net','www.ncisfiction.com']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/story.php?stid=01234 http://"+self.getSiteDomain()+"/chapters.php?stid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/story.php?stid=01234 http://"+cls.getSiteDomain()+"/chapters.php?stid=1234"
def getSiteURLPattern(self):
return r'http://www\.ncisfiction\.(net|com)/(chapters|story)?.php\?stid=\d+'
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -64,8 +65,8 @@ class NetRaptorOrgAdapter(BaseSiteAdapter):
return 'netraptor.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfiction/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfiction/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fanfiction/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -75,8 +76,8 @@ class NfaCommunityComAdapter(BaseSiteAdapter): # XXX
return 'nfacommunity.com' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -59,7 +60,7 @@ class NHAMagicalWorldsUsAdapter(BaseSiteAdapter):
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = " %m/%d/%y"
self.dateformat = " %d/%m/%y"
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
@@ -67,8 +68,8 @@ class NHAMagicalWorldsUsAdapter(BaseSiteAdapter):
return 'nha.magical-worlds.us'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -86,6 +87,28 @@ class NHAMagicalWorldsUsAdapter(BaseSiteAdapter):
else:
raise e
m = re.search(r"'viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data)
if m != None:
if self.is_adult or self.getConfig("is_adult"):
# We tried the default and still got a warning, so
# let's pull the warning number from the 'continue'
# link and reload data.
addurl = m.group(1)
# correct stupid &amp; error in url.
addurl = addurl.replace("&amp;","&")
url = self.url+'&index=1'+addurl
logger.debug("URL 2nd try: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
else:
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class NickAndGregNetAdapter(BaseSiteAdapter):
return 'www.nickandgreg.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/desert_archive/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/desert_archive/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/desert_archive/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -67,8 +68,8 @@ class OcclumencySycophantHexComAdapter(BaseSiteAdapter):
return 'occlumency.sycophanthex.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -71,8 +72,8 @@ class OneDirectionFanfictionComAdapter(BaseSiteAdapter):
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -66,8 +66,8 @@ class PhoenixSongNetAdapter(BaseSiteAdapter):
return 'www.phoenixsong.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfiction/story/1234/"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fanfiction/story/")+r"\d+/?$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -76,8 +77,8 @@ class PommeDeSangComAdapter(BaseSiteAdapter):
return 'pommedesang.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/efiction/viewstory.php?sid=1234 http://"+self.getSiteDomain()+"/sds/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/efiction/viewstory.php?sid=1234 http://"+cls.getSiteDomain()+"/sds/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return r"http://"+self.getSiteDomain()+"/(efiction|sds)?/viewstory.php\?sid=\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -70,8 +71,8 @@ class PonyFictionArchiveNetAdapter(BaseSiteAdapter):
return ['www.ponyfictionarchive.net','ponyfictionarchive.net','explicit.ponyfictionarchive.net']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234 http://explicit."+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234 http://explicit."+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.|explicit\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -73,12 +73,19 @@ class PortkeyOrgAdapter(BaseSiteAdapter): # XXX
return 'fanfiction.portkey.org' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/story/1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/story/1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/story/")+r"\d+(/\d+)?$"
def use_pagecache(self):
'''
adapters that will work with the page cache need to implement
this and change it to True.
'''
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
@@ -88,7 +95,6 @@ class PortkeyOrgAdapter(BaseSiteAdapter): # XXX
# portkey screws around with using a different URL to set the
# cookie and it's a pain. So... cheat!
if self.is_adult or self.getConfig("is_adult"):
cookieproc = urllib2.HTTPCookieProcessor()
cookie = cl.Cookie(version=0, name='verify17', value='1',
port=None, port_specified=False,
domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False,
@@ -99,9 +105,8 @@ class PortkeyOrgAdapter(BaseSiteAdapter): # XXX
comment=None,
comment_url=None,
rest={'HttpOnly': None},
rfc2109=False)
cookieproc.cookiejar.set_cookie(cookie)
self.opener = urllib2.build_opener(cookieproc)
rfc2109=False)
self.cookiejar.set_cookie(cookie)
try:
data = self._fetchUrl(url)
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -56,7 +57,7 @@ class PotionsAndSnitchesNetSiteAdapter(BaseSiteAdapter):
return ['www.potionsandsnitches.net','potionsandsnitches.net']
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://www.potionsandsnitches.net/fanfiction/viewstory.php?sid=1234"
def getSiteURLPattern(self):
@@ -73,7 +73,7 @@ class PotterFicsComAdapter(BaseSiteAdapter):
return 'www.potterfics.com'
@classmethod
def getSiteExampleURLs(self):
def getSiteExampleURLs(cls):
return "http://www.potterfics.com/historias/12345 http://www.potterfics.com/historias/12345/capitulo-1 "
def getSiteURLPattern(self):
@@ -86,6 +86,40 @@ class PotterFicsComAdapter(BaseSiteAdapter):
r"(?P<id>\d+)(/capitulo-(?P<ch>\d+))?/?$"
return p
def needToLoginCheck(self, data):
# partials used to avoid having to figure out what was wrong
# with included utf8 higher chars.
if 'Para ver esta historia, por favor inicia tu sesi' in data \
or '<script>alert("El nombre de usuario o contrase' in data:
return True
else:
return False
def performLogin(self,url):
params = {}
if self.password:
params['login_usuario'] = self.username
params['login_password'] = self.password
else:
params['login_usuario'] = self.getConfig("username")
params['login_password'] = self.getConfig("password")
params['login_ck'] = '1'
loginUrl = 'http://www.potterfics.com/secciones/usuarios/login.php'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['login_usuario']))
d = self._postUrl(loginUrl,params)
#print("d:%s"%d)
if '<script>alert("El nombre de usuario o contrase' in d:
logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['login_usuario']))
raise exceptions.FailedToLogin(url,params['login_usuario'])
return False
else:
return True
def extractChapterUrlsAndMetadata(self):
#this converts '/historias/12345' to 'http://www.potterfics.com/historias/12345'
@@ -123,8 +157,12 @@ class PotterFicsComAdapter(BaseSiteAdapter):
#print data
#deal with adult content warnings - doesn't seem to apply to this site
#deal with adult content login
if self.needToLoginCheck(data):
# need to log in for this one.
self.performLogin(url)
data = self._fetchUrl(url,usecache=False)
#set constant meta for this site:
#Set Language = Spanish
self.story.setMetadata('language', 'Spanish')
@@ -144,8 +182,8 @@ class PotterFicsComAdapter(BaseSiteAdapter):
#within that, we want the second row, first cell
cell = table.tr.findNextSibling('tr').td
#find first metadata block
mb = cell.div.findNextSibling('div')
#find first metadata block--isn't first if logged in
mb = cell.div.findNextSibling('div',{'align':'left'})
#Get meta...
self.story.setMetadata('title', stripHTML(mb.b))
#strip out brackets on rating
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class PotterHeadsAnonymousComAdapter(BaseSiteAdapter):
return 'fanfic.potterheadsanonymous.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -70,8 +71,8 @@ class PretenderCenterComAdapter(BaseSiteAdapter):
return ['www.pretendercentre.com','pretendercentre.com']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/missingpieces/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/missingpieces/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/missingpieces/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class PsychFicComAdapter(BaseSiteAdapter):
return 'www.psychfic.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class QafFicComAdapter(BaseSiteAdapter):
return 'www.qaf-fic.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/atp/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/atp/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/atp/viewstory.php?sid=")+r"\d+$"
@@ -69,8 +69,8 @@ class RestrictedSectionOrgSiteAdapter(BaseSiteAdapter):
return 'www.restrictedsection.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/story.php?story=1234 http://"+self.getSiteDomain()+"/file.php?file=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/story.php?story=1234 http://"+cls.getSiteDomain()+"/file.php?file=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain())+r"/(?P<filestory>file|story).php\?(file|story)=(?P<id>\d+)$"
@@ -0,0 +1,342 @@
# -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
# By virtue of being recent and requiring both is_adult and user/pass,
# adapter_fanficcastletvnet.py is the best choice for learning to
# write adapters--especially for sites that use the eFiction system.
# Most sites that have ".../viewstory.php?sid=123" in the story URL
# are eFiction.
# For non-eFiction sites, it can be considerably more complex, but
# this is still a good starting point.
# In general an 'adapter' needs to do these five things:
# - 'Register' correctly with the downloader
# - Site Login (if needed)
# - 'Are you adult?' check (if needed--some do one, some the other, some both)
# - Grab the chapter list
# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page)
# - Grab the chapter texts
# Search for XXX comments--that's where things are most likely to need changing.
# This function is called by the downloader in all adapter_*.py files
# in this dir to register the adapter class. So it needs to be
# updated to reflect the class below it. That, plus getSiteDomain()
# take care of 'Registering'.
def getClass():
return SamAndJackNetAdapter # XXX
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class SamAndJackNetAdapter(BaseSiteAdapter): # XXX
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["Windows-1252",
"utf8"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
# normalized story URL.
# XXX Most sites don't have the /fanfic part. Replace all to remove it usually.
self._setURL('http://' + self.getSiteDomain() + '/fanfics/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','sjn') # XXX
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%b %d, %Y" # XXX
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'samandjack.net' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfics/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fanfics/viewstory.php?sid=")+r"\d+$"
## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data):
if 'Registered Users Only' in data \
or 'There is no such account on our website' in data \
or "That password doesn't match the one in our database" in data:
return True
else:
return False
def performLogin(self, url):
params = {}
if self.password:
params['penname'] = self.username
params['password'] = self.password
else:
params['penname'] = self.getConfig("username")
params['password'] = self.getConfig("password")
params['cookiecheck'] = '1'
params['submit'] = 'Submit'
loginUrl = 'http://' + self.getSiteDomain() + '/fanfics/user.php?action=login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['penname']))
d = self._fetchUrl(loginUrl, params)
if "Member Account" not in d : #Member Account
logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['penname']))
raise exceptions.FailedToLogin(url,params['penname'])
return False
else:
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
if self.is_adult or self.getConfig("is_adult"):
# Weirdly, different sites use different warning numbers.
# If the title search below fails, there's a good chance
# you need a different number. print data at that point
# and see what the 'click here to continue' url says.
# Furthermore, there's a couple sites now with more than
# one warning level for different ratings. And they're
# fussy about it. midnightwhispers has three: 10, 3 & 5.
# we'll try 5 first.
addurl = "&ageconsent=ok&warning=5" # XXX
else:
addurl=""
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
url = self.url+'&index=1'+addurl
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
# The actual text that is used to announce you need to be an
# adult varies from site to site. Again, print data before
# the title search to troubleshoot.
# Since the warning text can change by warning level, let's
# look for the warning pass url. nfacommunity uses
# &amp;warning= -- actually, so do other sites. Must be an
# eFiction book.
# viewstory.php?sid=1882&amp;warning=4
# viewstory.php?sid=1654&amp;ageconsent=ok&amp;warning=5
#print data
#m = re.search(r"'viewstory.php\?sid=1882(&amp;warning=4)'",data)
m = re.search(r"'viewstory.php\?sid=\d+((?:&amp;ageconsent=ok)?&amp;warning=\d+)'",data)
if m != None:
if self.is_adult or self.getConfig("is_adult"):
# We tried the default and still got a warning, so
# let's pull the warning number from the 'continue'
# link and reload data.
addurl = m.group(1)
# correct stupid &amp; error in url.
addurl = addurl.replace("&amp;","&")
url = self.url+'&index=1'+addurl
logger.debug("URL 2nd try: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
else:
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = bs.BeautifulSoup(data)
# print data
# Now go hunting for all the meta data and the chapter list.
pagetitle = soup.find('div',{'id':'pagetitle'})
## Title
a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
# (fetch multiple authors)
alist = soup.findAll('a', href=re.compile(r"viewuser.php\?uid=\d+"))
for a in alist:
self.story.addToList('authorId',a['href'].split('=')[1])
self.story.addToList('authorUrl','http://'+self.host+'/fanfics/'+a['href'])
self.story.addToList('author',a.string)
# Reviews
reviewdata = soup.find('div', {'id' : 'sort'})
a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one.
self.story.setMetadata('reviews',stripHTML(a))
# Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfics/'+chapter['href']+addurl))
self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly.
# utility method
def defaultGetattr(d,k):
try:
return d[k]
except:
return ""
# <span class="label">Rated:</span> NC-17<br /> etc
labels = soup.findAll('span',{'class':'label'})
for labelspan in labels:
value = labelspan.nextSibling
label = labelspan.string
if 'Summary' in label:
self.setDescription(url,value)
if 'Rated' in label:
self.story.setMetadata('rating', value)
if 'Word count' in label:
self.story.setMetadata('numWords', value)
if 'Categories' in label:
cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))
catstext = [cat.string for cat in cats]
for cat in catstext:
self.story.addToList('category',cat.string)
if 'Characters' in label:
chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters'))
charstext = [char.string for char in chars]
for char in charstext:
self.story.addToList('characters',char.string)
## Not all sites use Genre, but there's no harm to
## leaving it in. Check to make sure the type_id number
## is correct, though--it's site specific.
if 'Genre' in label:
genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX
genrestext = [genre.string for genre in genres]
self.genre = ', '.join(genrestext)
for genre in genrestext:
self.story.addToList('genre',genre.string)
## Not all sites use Warnings, but there's no harm to
## leaving it in. Check to make sure the type_id number
## is correct, though--it's site specific.
if 'Warnings' in label:
warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX
warningstext = [warning.string for warning in warnings]
self.warning = ', '.join(warningstext)
for warning in warningstext:
self.story.addToList('warnings',warning.string)
if 'Completed' in label:
if 'Yes' in value:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
if 'Published' in label:
value=value.replace(' | ','')
self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat))
if 'Updated' in label:
# there's a stray [ at the end.
#value = value[0:-1]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat))
try:
# Find Series name from series URL.
a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+"))
series_name = a.string
series_url = 'http://'+self.host+'/fanfics/'+a['href']
# use BeautifulSoup HTML parser to make everything easier to find.
seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url))
storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$'))
i=1
for a in storyas:
if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')):
self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl',series_url)
break
i+=1
except:
# I find it hard to care if the series parsing fails
pass
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = bs.BeautifulStoneSoup(self._fetchUrl(url),
selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags.
div = soup.find('div', {'id' : 'story'})
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div)
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -70,8 +71,8 @@ class SamDeanArchiveNuAdapter(BaseSiteAdapter):
return ['www.samdean.archive.nu','samdean.archive.nu']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+r"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class ScarHeadNetAdapter(BaseSiteAdapter):
return 'scarhead.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class ScarvesAndCoffeeNetAdapter(BaseSiteAdapter):
return 'www.scarvesandcoffee.net'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -75,8 +76,8 @@ class SG1HeliopolisComAdapter(BaseSiteAdapter):
return 'sg1-heliopolis.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=1234 http://"+self.getSiteDomain()+"/adult/viewstory.php?sid=1234 http://"+self.getSiteDomain()+"/atlantis/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/archive/viewstory.php?sid=1234 http://"+cls.getSiteDomain()+"/adult/viewstory.php?sid=1234 http://"+cls.getSiteDomain()+"/atlantis/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return r"http://sg1-heliopolis.com/(archive|adult|atlantis)?/viewstory.php\?sid=\d+$"
@@ -0,0 +1,317 @@
# -*- coding: utf-8 -*-
# Copyright 2011 Fanficdownloader team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
import time
import logging
logger = logging.getLogger(__name__)
import re
import urllib2
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
from .. import exceptions as exceptions
from base_adapter import BaseSiteAdapter, makeDate
# By virtue of being recent and requiring both is_adult and user/pass,
# adapter_fanficcastletvnet.py is the best choice for learning to
# write adapters--especially for sites that use the eFiction system.
# Most sites that have ".../viewstory.php?sid=123" in the story URL
# are eFiction.
# For non-eFiction sites, it can be considerably more complex, but
# this is still a good starting point.
# In general an 'adapter' needs to do these five things:
# - 'Register' correctly with the downloader
# - Site Login (if needed)
# - 'Are you adult?' check (if needed--some do one, some the other, some both)
# - Grab the chapter list
# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page)
# - Grab the chapter texts
# Search for XXX comments--that's where things are most likely to need changing.
# This function is called by the downloader in all adapter_*.py files
# in this dir to register the adapter class. So it needs to be
# updated to reflect the class below it. That, plus getSiteDomain()
# take care of 'Registering'.
def getClass():
return SheppardWeirComAdapter # XXX
# Class name has to be unique. Our convention is camel case the
# sitename with Adapter at the end. www is skipped.
class SheppardWeirComAdapter(BaseSiteAdapter): # XXX
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
self.decode = ["Windows-1252",
"utf8"] # 1252 is a superset of iso-8859-1.
# Most sites that claim to be
# iso-8859-1 (and some that claim to be
# utf8) are really windows-1252.
self.username = "NoneGiven" # if left empty, site doesn't return any message at all.
self.password = ""
self.is_adult=False
# get storyId from url--url validation guarantees query is only sid=1234
self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])
# normalized story URL.
# XXX Most sites don't have the /fanfic part. Replace all to remove it usually.
self._setURL('http://' + self.getSiteDomain() + '/fanfics/viewstory.php?sid='+self.story.getMetadata('storyId'))
# Each adapter needs to have a unique site abbreviation.
self.story.setMetadata('siteabbrev','swf') # XXX
# The date format will vary from site to site.
# http://docs.python.org/library/datetime.html#strftime-strptime-behavior
self.dateformat = "%B %d, %Y" # XXX
@staticmethod # must be @staticmethod, don't remove it.
def getSiteDomain():
# The site domain. Does have www here, if it uses it.
return 'sheppardweir.com' # XXX
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/fanfics/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/fanfics/viewstory.php?sid=")+r"\d+$"
## Login seems to be reasonably standard across eFiction sites.
def needToLoginCheck(self, data):
if 'Registered Users Only' in data \
or 'There is no such account on our website' in data \
or "That password doesn't match the one in our database" in data:
return True
else:
return False
def performLogin(self, url):
params = {}
if self.password:
params['penname'] = self.username
params['password'] = self.password
else:
params['penname'] = self.getConfig("username")
params['password'] = self.getConfig("password")
params['cookiecheck'] = '1'
params['submit'] = 'Submit'
loginUrl = 'http://' + self.getSiteDomain() + '/fanfics/user.php?action=login'
logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,
params['penname']))
d = self._fetchUrl(loginUrl, params)
if "Member Account" not in d : #Member Account
logger.info("Failed to login to URL %s as %s" % (loginUrl,
params['penname']))
raise exceptions.FailedToLogin(url,params['penname'])
return False
else:
return True
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
if self.is_adult or self.getConfig("is_adult"):
# Weirdly, different sites use different warning numbers.
# If the title search below fails, there's a good chance
# you need a different number. print data at that point
# and see what the 'click here to continue' url says.
addurl = "&ageconsent=ok&warning=4" # XXX
else:
addurl=""
# index=1 makes sure we see the story chapter index. Some
# sites skip that for one-chapter stories.
url = self.url+'&index=1'+addurl
logger.debug("URL: "+url)
try:
data = self._fetchUrl(url)
except urllib2.HTTPError, e:
if e.code == 404:
raise exceptions.StoryDoesNotExist(self.url)
else:
raise e
if self.needToLoginCheck(data):
# need to log in for this one.
self.performLogin(url)
data = self._fetchUrl(url)
# The actual text that is used to announce you need to be an
# adult varies from site to site. Again, print data before
# the title search to troubleshoot.
if "Age Consent Required" in data: # XXX
raise exceptions.AdultCheckRequired(self.url)
if "Access denied. This story has not been validated by the adminstrators of this site." in data:
raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")
# use BeautifulSoup HTML parser to make everything easier to find.
soup = bs.BeautifulSoup(data)
# print data
# Now go hunting for all the meta data and the chapter list.
pagetitle = soup.find('div',{'id':'pagetitle'})
## Title
a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))
self.story.setMetadata('title',stripHTML(a))
# Find authorid and URL from... author url.
# (fetch multiple authors)
alist = soup.findAll('a', href=re.compile(r"viewuser.php\?uid=\d+"))
for a in alist:
self.story.addToList('authorId',a['href'].split('=')[1])
self.story.addToList('authorUrl','http://'+self.host+'/fanfics/'+a['href'])
self.story.addToList('author',a.string)
# Reviews
reviewdata = soup.find('div', {'id' : 'sort'})
a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one.
self.story.setMetadata('reviews',stripHTML(a))
# Find the chapters:
for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):
# just in case there's tags, like <i> in chapter titles.
self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfics/'+chapter['href']+addurl))
self.story.setMetadata('numChapters',len(self.chapterUrls))
# eFiction sites don't help us out a lot with their meta data
# formating, so it's a little ugly.
# utility method
def defaultGetattr(d,k):
try:
return d[k]
except:
return ""
# Summary
summarydata = unicode(soup.find('div',{'class':'content'}))
start='<span class="label">Summary: </span>'
end='</div>'
summarydata = summarydata[summarydata.index(start)+len(start):summarydata.rindex(end)]
self.setDescription(url,bs.BeautifulSoup(summarydata))
# <span class="label">Rated:</span> NC-17<br /> etc
labels = soup.findAll('span',{'class':'label'})
for labelspan in labels:
value = labelspan.nextSibling
label = labelspan.string
if 'Rated' in label:
self.story.setMetadata('rating', value)
if 'Word count' in label:
self.story.setMetadata('numWords', value)
if 'Categories' in label:
cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))
catstext = [cat.string for cat in cats]
for cat in catstext:
self.story.addToList('category',cat.string)
if 'Characters' in label:
chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters'))
charstext = [char.string for char in chars]
for char in charstext:
self.story.addToList('characters',char.string)
## Not all sites use Genre, but there's no harm to
## leaving it in. Check to make sure the type_id number
## is correct, though--it's site specific.
if 'Genre' in label:
genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX
genrestext = [genre.string for genre in genres]
self.genre = ', '.join(genrestext)
for genre in genrestext:
self.story.addToList('genre',genre.string)
## Not all sites use Warnings, but there's no harm to
## leaving it in. Check to make sure the type_id number
## is correct, though--it's site specific.
if 'Warnings' in label:
warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX
warningstext = [warning.string for warning in warnings]
self.warning = ', '.join(warningstext)
for warning in warningstext:
self.story.addToList('warnings',warning.string)
if 'Completed' in label:
if 'Yes' in value:
self.story.setMetadata('status', 'Completed')
else:
self.story.setMetadata('status', 'In-Progress')
if 'Published' in label:
value=value.replace(' - ','')
self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat))
if 'Updated' in label:
# there's a stray [ at the end.
#value = value[0:-1]
self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat))
try:
# Find Series name from series URL.
a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+"))
series_name = a.string
series_url = 'http://'+self.host+'/fanfics/'+a['href']
# use BeautifulSoup HTML parser to make everything easier to find.
seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url))
storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$'))
i=1
for a in storyas:
if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')):
self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl',series_url)
break
i+=1
except:
# I find it hard to care if the series parsing fails
pass
# grab the text for an individual chapter.
def getChapterText(self, url):
logger.debug('Getting chapter text from: %s' % url)
soup = bs.BeautifulStoneSoup(self._fetchUrl(url),
selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags.
div = soup.find('div', {'id' : 'story'})
if None == div:
raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url)
return self.utf8FromSoup(url,div)
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class SimplyUndeniableComAdapter(BaseSiteAdapter):
return 'www.simplyundeniable.com'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -66,8 +67,8 @@ class SinfulDesireOrgAdapter(BaseSiteAdapter):
return 'www.sinful-desire.org'
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/archive/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=")+r"\d+$"
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -71,8 +72,8 @@ class SiyeCoUkAdapter(BaseSiteAdapter): # XXX
return ['www.siye.co.uk','siye.co.uk']
@classmethod
def getSiteExampleURLs(self):
return "http://"+self.getSiteDomain()+"/siye/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "http://"+cls.getSiteDomain()+"/siye/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return re.escape("http://")+r"(www\.)?siye\.co\.uk/(siye/)?"+re.escape("viewstory.php?sid=")+r"\d+$"
@@ -1,8 +1,10 @@
# Software: eFiction
import re
import urllib2
import urlparse
from .. import BeautifulSoup
from ..htmlcleanup import stripHTML
from base_adapter import BaseSiteAdapter, makeDate
from .. import exceptions
@@ -88,25 +90,25 @@ class SpikeluverComAdapter(BaseSiteAdapter):
soup = self._customized_fetch_url(url)
pagetitle_div = soup.find('div', id='pagetitle')
self.story.setMetadata('title', pagetitle_div.a.string.strip())
self.story.setMetadata('title', stripHTML(pagetitle_div.a))
author_anchor = pagetitle_div.a.findNextSibling('a')
url = urlparse.urljoin(self.BASE_URL, author_anchor['href'])
components = urlparse.urlparse(url)
query_data = urlparse.parse_qs(components.query)
self.story.setMetadata('author', author_anchor.string.strip())
self.story.setMetadata('author', stripHTML(author_anchor))
self.story.setMetadata('authorId', query_data['uid'])
self.story.setMetadata('authorUrl', url)
sort_div = soup.find('div', id='sort')
self.story.setMetadata('reviews', sort_div('a')[1].string.strip())
self.story.setMetadata('reviews', stripHTML(sort_div('a')[1]))
listbox_tag = soup.find('div', {'class': 'listbox'})
for span_tag in listbox_tag('span'):
key = span_tag.string.strip(' :')
try:
value = span_tag.nextSibling.string.strip()
value = stripHTML(span_tag.nextSibling)
# This can happen with some fancy markup in the summary. Just
# ignore this error and set value to None, the summary parsing
# takes care of this
@@ -145,27 +147,27 @@ class SpikeluverComAdapter(BaseSiteAdapter):
if sibling.name == 'br':
break
self.story.addToList('category', sibling.string.strip())
self.story.addToList('category', stripHTML(sibling))
# Seems to be always "None" for some reason
elif key == 'Characters':
for sibling in span_tag.findNextSiblings(['a', 'br']):
if sibling.name == 'br':
break
self.story.addToList('characters', sibling.string.strip())
self.story.addToList('characters', stripHTML(sibling))
elif key == 'Genres':
for sibling in span_tag.findNextSiblings(['a', 'br']):
if sibling.name == 'br':
break
self.story.addToList('genre', sibling.string.strip())
self.story.addToList('genre', stripHTML(sibling))
elif key == 'Warnings':
for sibling in span_tag.findNextSiblings(['a', 'br']):
if sibling.name == 'br':
break
self.story.addToList('warnings', sibling.string.strip())
self.story.addToList('warnings', stripHTML(sibling))
# Challenges
@@ -173,7 +175,7 @@ class SpikeluverComAdapter(BaseSiteAdapter):
a = span_tag.findNextSibling('a')
if not a:
continue
self.story.setMetadata('series', a.string.strip())
self.story.setMetadata('series', stripHTML(a))
self.story.setMetadata('seriesUrl', urlparse.urljoin(self.BASE_URL, a['href']))
elif key == 'Chapters':
@@ -196,7 +198,7 @@ class SpikeluverComAdapter(BaseSiteAdapter):
if not chapter_anchor:
continue
title = chapter_anchor.string.strip()
title = stripHTML(chapter_anchor)
url = urlparse.urljoin(self.BASE_URL, chapter_anchor['href'])
self.chapterUrls.append((title, url))
@@ -15,6 +15,7 @@
# limitations under the License.
#
# Software: eFiction
import time
import logging
logger = logging.getLogger(__name__)
@@ -82,11 +83,12 @@ class SquidgeOrgPejaAdapter(BaseSiteAdapter):
return cls.getSiteDomain()+'/peja'
@classmethod
def getSiteExampleURLs(self):
return "https://"+self.getSiteDomain()+"/peja/cgi-bin/viewstory.php?sid=1234"
def getSiteExampleURLs(cls):
return "https://"+cls.getSiteDomain()+"/peja/cgi-bin/viewstory.php?sid=1234"
def getSiteURLPattern(self):
return r"https?"+re.escape("://"+self.getSiteDomain()+"/")+r"~?"+re.escape("peja/cgi-bin/viewstory.php?sid=")+r"\d+$"
# but not https://www.squidge.org/peja/cgi-bin/viewstory.php?sid=47746 -- that's the 'Site Map' negative look aead
return r"https?"+re.escape("://"+self.getSiteDomain()+"/")+r"~?"+re.escape("peja/cgi-bin/viewstory.php?sid=")+r"(?!47746)\d+$"
## Getting the chapter list and the meta data, plus 'is adult' checking.
def extractChapterUrlsAndMetadata(self):
@@ -218,7 +220,9 @@ class SquidgeOrgPejaAdapter(BaseSiteAdapter):
self.setSeries(series_name, i)
self.story.setMetadata('seriesUrl',series_url)
break
i+=1
# don't count the 'site map' story. See the url pattern method.
if '47746' not in a['href']:
i+=1
except:
# I find it hard to care if the series parsing fails

Some files were not shown because too many files have changed in this diff Show More