Compare commits

...
Author SHA1 Message Date
Jim Miller e108c2d828 Update CLI download zip. 2014-06-25 20:11:58 -05:00
Jim Miller 7534c03a37 Bump versions. 2014-06-25 19:09:02 -05:00
Jim Miller bf2e71e17f Fixes for int types being put in data, and only sort ships when there are ships. 2014-06-24 18:42:20 -05:00
Jim Miller 110960169a Moved tag calibre-plugin-1.8.25 to changeset 8e76e63420e7 (from changeset 2a5ad32eec54) 2014-06-21 09:57:41 -05:00
Jim Miller fb9d128687 Moved tag FanFictionDownLoader-4.5.06 to changeset 8e76e63420e7 (from changeset 2a5ad32eec54) 2014-06-21 09:57:36 -05:00
Jim Miller e99c3d6ea6 Update CLI download zip. 2014-06-21 09:57:26 -05:00
Jim Miller 1172322446 Added tag calibre-plugin-1.8.25 for changeset 2a5ad32eec54 2014-06-21 09:56:05 -05:00
Jim Miller 2f4ce5c40e Added tag FanFictionDownLoader-4.5.06 for changeset 2a5ad32eec54 2014-06-21 09:55:56 -05:00
Jim Miller 73c7ffeff6 Bump versions. 2014-06-21 09:55:43 -05:00
Jim Miller 0a78a0c044 Switch to transifex.com translations, still using poedit.exe to make .mo files. 2014-06-21 09:54:52 -05:00
Jim Miller dc5adb7f4a Add a literotica.com comment. 2014-06-19 21:53:53 -05:00
Jim Miller 4abfbdf462 Site specific metadata 'eroticatags' for literotica.com. 2014-06-19 21:48:18 -05:00
Jim Miller d1b73b9a6a Auto-add http: to URLs starting with //. 2014-06-18 21:28:42 -05:00
Jim Miller e477a9870d Fix for literotica with split-capable internals. 2014-06-18 21:28:21 -05:00
Jim Miller 2dbbb0f13d Fix for utf8 descs on fimf. 2014-06-18 21:27:47 -05:00
cryzed 3d1d3f4e26 Added adapter for http://fictionmania.tv/ 2014-06-17 22:50:51 +02:00
Jim Miller 7a5d77975a Fix for calibre-injected series--don't treat series as a list, it isn't. 2014-06-16 20:06:34 -05:00
Jim Miller c3911a279b Changes to '\,' better implmt split list feature to avoid infinite recursions. 2014-06-15 20:36:29 -05:00
Jim Miller 8cf6f210e6 Change to with/without www. code to work with https also. 2014-06-15 20:30:35 -05:00
Jim Miller 539426b41d Fix for 'a' flag on custom_columns_settings not working as intended. 2014-06-14 20:58:18 -05:00
Jim Miller 78efbb3e1e Added tag FanFictionDownLoader-4.5.05 for changeset 5de0389d525f 2014-06-14 10:42:49 -05:00
Jim Miller 4878837805 Added tag calibre-plugin-1.8.24 for changeset 5de0389d525f 2014-06-14 10:42:39 -05:00
16 changed files with 463 additions and 159 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
# ffd-retief-hrd fanfictiondownloader
application: fanfictiondownloader
version: 4-5-05
version: 4-5-07
runtime: python27
api_version: 1
threadsafe: true
+1 -1
View File
@@ -42,7 +42,7 @@ class FanFictionDownLoaderBase(InterfaceActionBase):
description = _('UI plugin to download FanFiction stories from various sites.')
supported_platforms = ['windows', 'osx', 'linux']
author = 'Jim Miller'
version = (1, 8, 24)
version = (1, 8, 26)
minimum_calibre_version = (1, 13, 0)
#: This field defines the GUI plugin class that contains all the code
+2 -2
View File
@@ -889,7 +889,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
# all_metadata duplicates some data, but also includes extra_entries, etc.
book['all_metadata'] = story.getAllMetadata(removeallentities=True)
book['title'] = story.getMetadata("title", removeallentities=True)
book['author_sort'] = book['author'] = story.getList("author", removeallentities=True)
book['publisher'] = story.getMetadata("site")
@@ -1617,7 +1617,7 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
pass
if val:
vallist = [val]
vallist.append(val)
db.set_custom(book_id, ", ".join(vallist), label=label, commit=False)
+13 -14
View File
@@ -1,18 +1,22 @@
# SOME DESCRIPTIVE TITLE.
# Copyright (C) YEAR ORGANIZATION
# FIRST AUTHOR <EMAIL@ADDRESS>, YEAR.
#
# Translators:
# Simon_Schuette <simonschuette@arcor.de>, 2014
msgid ""
msgstr ""
"Project-Id-Version: FanFictionDownLoader 1.8\n"
"Project-Id-Version: calibre-plugins\n"
"POT-Creation-Date: 2014-02-15 09:39+Central Standard Time\n"
"PO-Revision-Date: 2014-02-15 09:41-0600\n"
"PO-Revision-Date: 2014-06-20 17:28-0600\n"
"Last-Translator: Jim Miller <RetiefJimm@gmail.com>\n"
"Language-Team: \n"
"Language-Team: German (http://www.transifex.com/projects/p/calibre-plugins/"
"language/de/)\n"
"MIME-Version: 1.0\n"
"Content-Type: text/plain; charset=UTF-8\n"
"Content-Transfer-Encoding: 8bit\n"
"Generated-By: pygettext.py 1.5\n"
"Language: de\n"
"Plural-Forms: nplurals=2; plural=(n != 1);\n"
"X-Generator: Poedit 1.5.7\n"
#: __init__.py:42
@@ -497,7 +501,6 @@ msgstr "Plugin Standardeinstellungen"
msgid "OK"
msgstr "OK"
# %(rl)s = Reading List. Keep as is.
#: config.py:638
msgid ""
"These settings provide integration with the %(rl)s Plugin. %(rl)s can "
@@ -514,7 +517,6 @@ msgid "Add new/updated stories to \"Send to Device\" Reading List(s)."
msgstr ""
"Neue/aktualiserte Stories auf \"ans Gerät senden\" Leseliste(n) hinzufügen."
# %(rl)s = Reading List. Keep as is.
#: config.py:644
msgid ""
"Automatically add new/updated stories to these lists in the %(rl)s plugin."
@@ -537,7 +539,6 @@ msgstr ""
msgid "Add new/updated stories to \"To Read\" Reading List(s)."
msgstr "Neue/aktualiserte Stories auf \"zu lesen\" Leseliste(n) hinzufügen."
# %(rl)s = Reading List. Keep as is.
#: config.py:660
msgid ""
"Automatically add new/updated stories to these lists in the %(rl)s plugin.\n"
@@ -568,7 +569,6 @@ msgstr ""
"gleichzeitig die Stories zurück auf die \"ans Gerät senden\" Leseliste(n) "
"fügen."
# %(gc)s = Generate Cover. Keep as is.
#: config.py:698
msgid ""
"The %(gc)s plugin can create cover images for books using various metadata "
@@ -584,7 +584,6 @@ msgstr ""
msgid "Default"
msgstr "Standardeinstellung"
# %(gc)s = Generate Cover. Keep as is.
#: config.py:721
msgid ""
"On Metadata update, run %(gc)s with this setting, if not selected for "
@@ -594,14 +593,12 @@ msgstr ""
"verwendet werden, wenn für diese bestimmte Seite keine Auswahl festgelegt "
"wurde."
# %(gc)s = Generate Cover. Keep as is.
#: config.py:724
msgid "On Metadata update, run %(gc)s with this setting for %(site)s stories."
msgstr ""
"Beim Aktualisieren der Metadaten soll %(gc)s für die %(site)s Stories diese "
"Einstellung verwenden."
# %(gc)s = Generate Cover. Keep as is.
#: config.py:747
msgid "Run %(gc)s Only on New Books"
msgstr "%(gc)s nur bei neuen Bücher anwenden"
@@ -631,12 +628,17 @@ msgstr ""
#: config.py:757
msgid "Use calibre's Polish feature to inject/update the cover"
msgstr ""
"Verwenden Sie Calibre´s \"Bücher Perfektionieren\" Funktion zum Einfügen/"
"Aktualisieren des Coverbildes."
#: config.py:758
msgid ""
"Calibre's Polish feature will be used to inject or update the generated "
"cover into the ebook, EPUB only."
msgstr ""
"Calibre´s \"Bücher Perfektionieren\" Funktion wird verwendet, um das "
"erstellte Cover zu aktualisieren oder in das Buch einzufügen. Es werden nur "
"AZW3 und EPUB unterstützt."
#: config.py:772
msgid ""
@@ -1222,7 +1224,6 @@ msgstr ""
msgid "Download FanFiction stories from various web sites"
msgstr "FanFiction-Stories von verschiedenen Web-Seiten herunterladen"
# This is what appears on the plugin button/menu when added to calibre's main toolbar or menu.
#: ffdl_plugin.py:117
msgid "FanFictionDL"
msgstr "FanFictionDL"
@@ -1367,7 +1368,6 @@ msgstr ""
"Ablehnung der FFDL URL´s: Keines der ausgewählten Bücher hat eine FanFiction-"
"URL."
# %s = EpubMerge
#: ffdl_plugin.py:543
msgid "Cannot Make Anthologys without %s"
msgstr "Sammelbände können nicht ohne %s erstellt werden."
@@ -1853,7 +1853,6 @@ msgstr "Bad URL"
msgid "Anthology containing:"
msgstr "Sammelband enthält:"
# title by author
#: ffdl_plugin.py:1991
msgid "%s by %s"
msgstr "%s von %s"
+7 -19
View File
@@ -1,24 +1,22 @@
# SOME DESCRIPTIVE TITLE.
# Copyright (C) YEAR ORGANIZATION
# FIRST AUTHOR <EMAIL@ADDRESS>, YEAR.
#
# Translators:
msgid ""
msgstr ""
"Project-Id-Version: FanFictionDownLoader 1.8\n"
"Project-Id-Version: calibre-plugins\n"
"POT-Creation-Date: 2014-02-15 09:39+Central Standard Time\n"
"PO-Revision-Date: 2014-02-15 09:41-0600\n"
"PO-Revision-Date: 2014-06-20 17:28-0600\n"
"Last-Translator: Jim Miller <RetiefJimm@gmail.com>\n"
"Language-Team: Ptitprince <Leporello1791@gmail.com>\n"
"Language: fr_FR\n"
"Language-Team: French (http://www.transifex.com/projects/p/calibre-plugins/"
"language/fr/)\n"
"MIME-Version: 1.0\n"
"Content-Type: text/plain; charset=UTF-8\n"
"Content-Transfer-Encoding: 8bit\n"
"Generated-By: pygettext.py 1.5\n"
"X-Generator: Poedit 1.5.7\n"
"Language: fr\n"
"Plural-Forms: nplurals=2; plural=(n > 1);\n"
"X-Poedit-Bookmarks: -1,-1,-1,231,-1,-1,-1,-1,-1,-1\n"
"X-Poedit-Basepath: D:\\catalogue po\n"
"X-Poedit-SearchPath-0: D:\\catalogue po\n"
"X-Generator: Poedit 1.5.7\n"
#: __init__.py:42
msgid "UI plugin to download FanFiction stories from various sites."
@@ -500,7 +498,6 @@ msgstr "Paramètres par défaut du greffon"
msgid "OK"
msgstr "OK"
# %(rl)s = Reading List. Keep as is.
#: config.py:638
msgid ""
"These settings provide integration with the %(rl)s Plugin. %(rl)s can "
@@ -517,7 +514,6 @@ msgstr ""
"Ajouter des récits nouveaux/mis à jour à la/aux liste(s) de lecture de "
"\"Envoyer vers le dispositif\"."
# %(rl)s = Reading List. Keep as is.
#: config.py:644
msgid ""
"Automatically add new/updated stories to these lists in the %(rl)s plugin."
@@ -542,7 +538,6 @@ msgstr ""
"Ajouter des récits nouveaux/mis à jour à la/aux liste(s) de lecture \" A "
"lire \"."
# %(rl)s = Reading List. Keep as is.
#: config.py:660
msgid ""
"Automatically add new/updated stories to these lists in the %(rl)s plugin.\n"
@@ -572,7 +567,6 @@ msgstr ""
"Option du menu pour retirer des listes de \"A lire\" ajoutera également à "
"nouveau des récits à/aux Liste(s) de lecture \"Envoyer vers le dispositif\""
# %(gc)s = Generate Cover. Keep as is.
#: config.py:698
msgid ""
"The %(gc)s plugin can create cover images for books using various metadata "
@@ -588,7 +582,6 @@ msgstr ""
msgid "Default"
msgstr "Par défaut"
# %(gc)s = Generate Cover. Keep as is.
#: config.py:721
msgid ""
"On Metadata update, run %(gc)s with this setting, if not selected for "
@@ -597,14 +590,12 @@ msgstr ""
"A la mise à jour des métadonnées, exécute %(gc)s avec ces paramètres, s'ils "
"ne sont pas sélectionnés pour un site spécifique."
# %(gc)s = Generate Cover. Keep as is.
#: config.py:724
msgid "On Metadata update, run %(gc)s with this setting for %(site)s stories."
msgstr ""
"A la mise à jour de métadonnées, exécute %(gc)s avec ces paramètres pour des "
"%(site)s de récits."
# %(gc)s = Generate Cover. Keep as is.
#: config.py:747
msgid "Run %(gc)s Only on New Books"
msgstr "Exécuter %(gc)s uniquement pour les nouveaux livres"
@@ -1221,7 +1212,6 @@ msgstr ""
msgid "Download FanFiction stories from various web sites"
msgstr "Télécharger des récits FanFiction de différents sites"
# This is what appears on the plugin button/menu when added to calibre's main toolbar or menu.
#: ffdl_plugin.py:117
msgid "FanFictionDL"
msgstr "FanFictionDL"
@@ -1367,7 +1357,6 @@ msgstr ""
"Occupé de rejeter les URLs FFDL : aucun des livres sélectionnés n'ont d'URLs "
"FanFiction"
# %s = EpubMerge
#: ffdl_plugin.py:543
msgid "Cannot Make Anthologys without %s"
msgstr "Ne peut faire d'Anthologies sans %s"
@@ -1848,7 +1837,6 @@ msgstr "Mauvaise URL"
msgid "Anthology containing:"
msgstr "Anthologie contenant : "
# title by author
#: ffdl_plugin.py:1991
msgid "%s by %s"
msgstr "%s par %s"
+74 -30
View File
@@ -880,6 +880,45 @@ extraships:Harry Potter/Hermione Granger
#username:YourName
#password:yourpassword
[fictionmania.tv]
## website encoding(s) In theory, each website reports the character
## encoding they use for each page. In practice, some sites report it
## incorrectly. Each adapter has a default list, usually "utf8,
## Windows-1252" or "Windows-1252, utf8", but this will let you
## explicitly set the encoding and order if you need to. The special
## value 'auto' will call chardet and use the encoding it reports if
## it has +90% confidence. 'auto' is not reliable.
website_encodings:ISO-8859-1,auto
## items to include in the log page Empty metadata entries, or those
## that haven't changed since the last update, will *not* appear, even
## if in the list. You can include extra text or HTML that will be
## included as-is in each log entry. Eg: logpage_entries: ...,<br />,
## summary,<br />,...
## Don't include numChapters since all stories are a single "chapter", there's
## no way to reliably find the next chapter
logpage_entries: dateCreated,datePublished,dateUpdated,numChapters,numWords,status,series,title,author,description,category,genre,rating,warnings
## items to include in the title page
## Empty metadata entries will *not* appear, even if in the list.
## You can include extra text or HTML that will be included as-is in
## the title page. Eg: titlepage_entries: ...,<br />,summary,<br />,...
## All current formats already include title and author.
## Don't include numChapters since all stories are a single "chapter", there's
## no way to reliably find the next chapter
titlepage_entries: seriesHTML,category,genre,language,characters,ships,status,datePublished,dateUpdated,dateCreated,rating,warnings,numWords,site,description
## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them.
extra_valid_entries:fileName,fileSize,oldName,newName,keyWords,mainCharactersAge,readings
## Turns all space characters into "&nbsp" HTML entities to forcefully preserve
## formatting with spaces. Enabling this will blow up the filesize quite a bit
## and is probably not a good idea, unless you absolutely need the story
## formatting.
## Specific to fictionmania.tv
non_breaking_spaces:false
[fictionpad.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -905,36 +944,6 @@ dislikes_label:Dislikes
#username:YourName
#password:yourpassword
[storiesonline.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Clear FanFiction from defaults, site is original fiction.
extratags:
extra_valid_entries:size,universe,universeUrl,universeHTML,codes,notice
#extra_titlepage_entries:size,universeHTML,codes,notice
size_label:Size
universe_label:Universe
universeUrl_label:Universe URL
universeHTML_label:Universe
codes_label:Codes
notice_label:Notice
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:universe
## storiesonline.net stories can be in a series or a universe, but not
## both. By default, universe will be populated in 'series' with
## index=0
universe_as_series: true
[grangerenchanted.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -987,6 +996,11 @@ extracategories:Star Trek
extracharacters:Kirk,Spock
extraships:Kirk/Spock
[literotica.com]
extra_valid_entries:eroticatags
eroticatags_label:Erotica Tags
#extra_titlepage_entries: eroticatags
[lumos.sycophanthex.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
@@ -1151,6 +1165,36 @@ extracategories:Buffy the Vampire Slayer
## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis
[storiesonline.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Clear FanFiction from defaults, site is original fiction.
extratags:
extra_valid_entries:size,universe,universeUrl,universeHTML,codes,notice
#extra_titlepage_entries:size,universeHTML,codes,notice
size_label:Size
universe_label:Universe
universeUrl_label:Universe URL
universeHTML_label:Universe
codes_label:Codes
notice_label:Notice
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:universe
## storiesonline.net stories can be in a series or a universe, but not
## both. By default, universe will be populated in 'series' with
## index=0
universe_as_series: true
[svufiction.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
Binary file not shown.
+5 -2
View File
@@ -129,6 +129,7 @@ import adapter_bloodshedversecom
import adapter_nocturnallightnet
import adapter_fanfichu
import adapter_fanfictioncsodaidokhu
import adapter_fictionmaniatv
## This bit of complexity allows adapters to be added by just adding
## importing. It eliminates the long if/else clauses we used to need
@@ -208,6 +209,8 @@ def getConfigSectionFor(url):
def getClassFor(url):
## fix up leading protocol.
fixedurl = re.sub(r"(?i)^[htp]+(s?)[:/]+",r"http\1://",url.strip())
if fixedurl.startswith("//"):
fixedurl = "http:%s"%url
if not fixedurl.startswith("http"):
fixedurl = "http://%s"%url
## remove any trailing '#' locations.
@@ -223,11 +226,11 @@ def getClassFor(url):
domain = domain.replace("www.","")
#logger.debug("trying site:without www: "+domain)
cls = getClassFromList(domain)
fixedurl = fixedurl.replace("http://www.","http://")
fixedurl = re.sub(r"^http(s?)://www\.",r"http\1://",fixedurl)
if not cls:
#logger.debug("trying site:www."+domain)
cls = getClassFromList("www."+domain)
fixedurl = fixedurl.replace("http://","http://www.")
fixedurl = re.sub(r"^http(s?)://",r"http\1://www.",fixedurl)
if cls:
fixedurl = cls.stripURLParameters(fixedurl)
@@ -0,0 +1,178 @@
import re
import urllib2
import urlparse
from .. import BeautifulSoup
from ..BeautifulSoup import NavigableString
from base_adapter import BaseSiteAdapter, makeDate
from .. import exceptions
def getClass():
return FictionManiaTVAdapter
def _get_query_data(url):
components = urlparse.urlparse(url)
query_data = urlparse.parse_qs(components.query)
return dict((key, data[0]) for key, data in query_data.items())
# yields Tag _and_ NavigableString siblings from the given tag. The
# BeautifulSoup findNextSiblings() method for some reasons only returns either
# NavigableStrings _or_ Tag objects, not both.
def _yield_next_siblings(tag):
sibling = tag.nextSibling
while sibling:
yield sibling
sibling = sibling.nextSibling
class FictionManiaTVAdapter(BaseSiteAdapter):
SITE_ABBREVIATION = 'fmt'
SITE_DOMAIN = 'fictionmania.tv'
BASE_URL = 'http://' + SITE_DOMAIN + '/stories/'
READ_TEXT_STORY_URL_TEMPLATE = BASE_URL + 'readtextstory.html?storyID=%s'
DETAILS_URL_TEMPLATE = BASE_URL + 'details.html?storyID=%s'
DATETIME_FORMAT = '%m/%d/%Y'
ALTERNATIVE_DATETIME_FORMAT = '%m/%d/%y'
def __init__(self, config, url):
BaseSiteAdapter.__init__(self, config, url)
query_data = urlparse.parse_qs(self.parsedUrl.query)
story_id = query_data['storyID'][0]
self.story.setMetadata('storyId', story_id)
self._setURL(self.READ_TEXT_STORY_URL_TEMPLATE % story_id)
self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION)
# Always single chapters, probably should use the Anthology feature to
# merge chapters of a story
self.story.setMetadata('numChapters', 1)
def _customized_fetch_url(self, url, exception=None, parameters=None):
if exception:
try:
data = self._fetchUrl(url, parameters)
except urllib2.HTTPError:
raise exception(self.url)
# Just let self._fetchUrl throw the exception, don't catch and
# customize it.
else:
data = self._fetchUrl(url, parameters)
return BeautifulSoup.BeautifulSoup(data)
@staticmethod
def getSiteDomain():
return FictionManiaTVAdapter.SITE_DOMAIN
@classmethod
def getSiteExampleURLs(cls):
return cls.READ_TEXT_STORY_URL_TEMPLATE % 1234
def getSiteURLPattern(self):
return re.escape(self.BASE_URL) + '(readtextstory|details)\.html\?storyID=\d+$'
def extractChapterUrlsAndMetadata(self):
url = self.DETAILS_URL_TEMPLATE % self.story.getMetadata('storyId')
soup = self._customized_fetch_url(url)
keep_summary_html = self.getConfig('keep_summary_html')
for row in soup.find('table')('tr'):
cells = row('td')
key = cells[0].b.string.strip(':')
try:
value = cells[1].string
except AttributeError:
value = None
if key == 'Story Name-Title':
self.story.setMetadata('title', value)
self.chapterUrls.append((value, self.url))
elif key == 'File Name':
self.story.setMetadata('fileName', value)
elif key == 'File Size':
self.story.setMetadata('fileSize', value)
elif key == 'Author':
element = cells[1].a
self.story.setMetadata('author', element.string)
query_data = _get_query_data(element['href'])
self.story.setMetadata('authorId', query_data['word'])
self.story.setMetadata('authorUrl', urlparse.urljoin(url, element['href']))
elif key == 'Date Added':
try:
date = makeDate(value, self.DATETIME_FORMAT)
except ValueError:
date = makeDate(value, self.ALTERNATIVE_DATETIME_FORMAT)
self.story.setMetadata('datePublished', date)
elif key == 'Old Name':
self.story.setMetadata('oldName', value)
elif key == 'New Name':
self.story.setMetadata('newName', value)
elif key == 'Other Key Names':
for name in value.split(', '):
self.story.addToList('characters', name)
# I have no clue how the rating system works, if you are reading
# transgender fanfiction, you are probably an adult.
elif key == 'Rating':
self.story.setMetadata('rating', value)
elif key == 'Complete':
self.story.setMetadata('status', 'Complete' if value == 'Complete' else 'In-Progress')
elif key == 'Categories':
for element in cells[1]('a'):
self.story.addToList('category', element.string)
elif key == 'Key Words':
for element in cells[1]('a'):
self.story.addToList('keyWords', element.string)
elif key == 'Main Characters Age':
element = cells[1].a
self.story.setMetadata('mainCharactersAge', element.string)
elif key == 'Synopsis':
element = cells[1]
# Replace td with div to avoid possible strange formatting in
# the ebook later on
element.name = 'div'
if keep_summary_html:
self.story.setMetadata('description', unicode(element))
else:
self.story.setMetadata('description', ''.join(element(text=True)))
elif key == 'Reads':
self.story.setMetadata('readings', value)
def getChapterText(self, url):
soup = self._customized_fetch_url(url)
element = soup.find('pre')
element.name = 'div'
# The story's content is contained in a <pre> tag, probably taken 1:1
# from the source text file. A simple replacement of all newline
# characters with a break line tag should take care of formatting.
# While wrapping in paragraphs would be possible, it's too much work,
# I'd rather display the story 1:1 like it was found in the pre tag.
content = unicode(element)
content = content.replace('\n', '<br />')
if self.getConfig('non_breaking_spaces'):
content = content.replace(' ', '&nbsp;')
return content
@@ -186,9 +186,9 @@ class FimFictionNetSiteAdapter(BaseSiteAdapter):
self.setCoverImage(self.url,coverurl)
# fimf has started including extra stuff inside the description div.
descdivstr = "%s"%soup.find("div", {"class":"description"})
hrstr="<hr />"
descdivstr = '<div class="description">'+descdivstr[descdivstr.index(hrstr)+len(hrstr):]
descdivstr = u"%s"%soup.find("div", {"class":"description"})
hrstr=u"<hr />"
descdivstr = u'<div class="description">'+descdivstr[descdivstr.index(hrstr)+len(hrstr):]
self.setDescription(self.url,descdivstr)
# Can't trust dates from API anymore I'm told.
@@ -99,7 +99,7 @@ class LiteroticaSiteAdapter(BaseSiteAdapter):
# author
a = soup1.find("span", "b-story-user-y")
self.story.setMetadata('authorId', urlparse.parse_qs(a.a['href'].split('?')[1])['uid'])
self.story.setMetadata('authorId', urlparse.parse_qs(a.a['href'].split('?')[1])['uid'][0])
authorurl = a.a['href']
if authorurl.startswith('//'):
authorurl = self.parsedUrl.scheme+':'+authorurl
@@ -197,7 +197,12 @@ class LiteroticaSiteAdapter(BaseSiteAdapter):
self.story.setMetadata('numChapters', len(self.chapterUrls))
self.story.setMetadata('category', soup1.find('div', 'b-breadcrumbs').findAll('a')[1].string)
# deliberately not self.setDescription() because it's never HTML.
self.story.setMetadata('description', soup1.find('meta', {'name': 'description'})['content'])
# li tags inside div class b-s-story-tag-list
for li in soup1.find('div', {'class':'b-s-story-tag-list'}).findAll('a'):
self.story.addToList('eroticatags',stripHTML(li))
return
+6 -3
View File
@@ -219,15 +219,18 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
elif idstr == '81':
self.story.addToList('category',u'Pitch Perfect')
self.story.addToList('characters','Chloe B.')
elif idstr == '82':
self.story.addToList('characters','Henry (Once Upon a Time)')
self.story.addToList('category',u'Once Upon a Time (TV)')
elif idstr == '83':
self.story.addToList('category',u'Rizzoli &amp; Isles')
self.story.addToList('characters','J. Rizzoli')
self.story.addToList('category',u'Pitch Perfect')
self.story.addToList('characters','Chloe B.')
self.story.addToList('ships','Chloe B. &amp; J. Rizzoli')
elif idstr == '82':
self.story.addToList('characters','Henry (Once Upon a Time)')
self.story.addToList('category',u'Once Upon a Time (TV)')
elif idstr == '90':
self.story.setMetadata('characters','Henry (Once Upon a Time)')
self.story.setMetadata('category',u'Once Upon a Time (TV)')
else:
self.story.addToList('category','Harry Potter')
self.story.addToList('category','Furbie')
+19 -10
View File
@@ -53,46 +53,52 @@ class Configuration(ConfigParser.SafeConfigParser):
self.addConfigSection(sitewithout+":"+fileform)
self.addConfigSection("overrides")
self.validEntries = [
self.listTypeEntries = [
'category',
'genre',
'language',
'characters',
'ships',
'warnings',
'extratags',
'author',
'authorId',
'authorUrl',
'lastupdate',
]
self.validEntries = self.listTypeEntries + [
'series',
'seriesUrl',
'language',
'status',
'datePublished',
'dateUpdated',
'dateCreated',
'rating',
'warnings',
'numChapters',
'numWords',
'site',
'storyId',
'authorId',
'extratags',
'title',
'storyUrl',
'description',
'author',
'authorUrl',
'formatname',
'formatext',
'siteabbrev',
'version',
# internal stuff.
'langcode',
'output_css',
'authorHTML',
'seriesHTML',
'lastupdate'
'langcode',
'output_css',
]
def addConfigSection(self,section):
self.sectionslist.insert(0,section)
def isListType(self,key):
return key in self.listTypeEntries or self.hasConfig("include_in_"+key)
def isValidMetaEntry(self, key):
return key in self.getValidMetaList()
@@ -162,6 +168,9 @@ class Configurable(object):
def __init__(self, configuration):
self.configuration = configuration
def isListType(self,key):
return self.configuration.isListType(key)
def isValidMetaEntry(self, key):
return self.configuration.isValidMetaEntry(key)
+71 -38
View File
@@ -323,18 +323,28 @@ class Story(Configurable):
iel = []
self.in_ex_cludes[ie] = self.set_in_ex_clude(ies)
def join_list(self, key, vallist):
return self.getConfig("join_string_"+key,u", ").replace(SPACE_REPLACE,' ').join(map(unicode, vallist))
def setMetadata(self, key, value, condremoveentities=True):
## still keeps &lt; &lt; and &amp;
if condremoveentities:
self.metadata[key]=conditionalRemoveEntities(value)
# keep as list type, but set as only value.
if self.isList(key):
self.addToList(key,value,condremoveentities=condremoveentities,clear=True)
else:
self.metadata[key]=value
## still keeps &lt; &lt; and &amp;
if condremoveentities:
self.metadata[key]=conditionalRemoveEntities(value)
else:
self.metadata[key]=value
if key == "language":
try:
# getMetadata not just self.metadata[] to do replace_metadata.
self.metadata['langcode'] = langs[self.getMetadata(key)]
self.setMetadata('langcode',langs[self.getMetadata(key)])
except:
self.metadata['langcode'] = 'en'
self.setMetadata('langcode','en')
if key == 'dateUpdated':
# Last Update tags for Bill.
self.addToList('lastupdate',value.strftime("Last Update Year/Month: %Y/%m"))
@@ -421,11 +431,16 @@ class Story(Configurable):
replacement=replacement.replace(SPACE_REPLACE,' ')
self.replacements.append([metakeys,regexp,replacement,condkey,condregexp])
def doReplacements(self,value,key):
def doReplacements(self,value,key,return_list=False,seen_list=[]):
value = self.do_in_ex_clude('include_metadata_pre',value,key)
value = self.do_in_ex_clude('exclude_metadata_pre',value,key)
for (metakeys,regexp,replacement,condkey,condregexp) in self.replacements:
retlist = [value]
for replaceline in self.replacements:
if replaceline in seen_list: # recursion on pattern, bail
# print("bailing on %s"%replaceline)
continue
(metakeys,regexp,replacement,condkey,condregexp) = replaceline
if (metakeys == None or key in metakeys) \
and isinstance(value,basestring) \
and regexp.search(value):
@@ -435,21 +450,32 @@ class Story(Configurable):
doreplace = condval != None and condregexp.search(condval)
if doreplace:
# split into more than one list entry if list and
# split into more than one list entry if
# SPLIT_META present in replacement string. Split
# first, then regex sub.
if self.isList(key) and SPLIT_META in replacement:
repllist = replacement.split(SPLIT_META)
for repl in repllist[1:]:
self.addToList(key,regexp.sub(repl,value))
value = regexp.sub(repllist[0],value)
else:
value = regexp.sub(replacement,value)
# first, then regex sub, then recurse call replace
# on each. Break out of loop, each split element
# handled individually by recursion call.
if SPLIT_META in replacement:
retlist = []
for splitrepl in replacement.split(SPLIT_META):
retlist.extend(self.doReplacements(regexp.sub(splitrepl,value),
key,
return_list=True,
seen_list=seen_list+[replaceline]))
break
else:
retlist = [regexp.sub(replacement,value)]
for val in retlist:
retlist = map(partial(self.do_in_ex_clude,'include_metadata_post',key=key),retlist)
retlist = map(partial(self.do_in_ex_clude,'exclude_metadata_post',key=key),retlist)
# value = self.do_in_ex_clude('include_metadata_post',value,key)
# value = self.do_in_ex_clude('exclude_metadata_post',value,key)
value = self.do_in_ex_clude('include_metadata_post',value,key)
value = self.do_in_ex_clude('exclude_metadata_post',value,key)
return value
if return_list:
return retlist
else:
return self.join_list(key,retlist)
def getMetadataRaw(self,key):
if self.isValidMetaEntry(key) and self.metadata.has_key(key):
@@ -463,8 +489,9 @@ class Story(Configurable):
return value
if self.isList(key):
join_string = self.getConfig("join_string_"+key,u", ").replace(SPACE_REPLACE,' ')
value = join_string.join(self.getList(key, removeallentities, doreplacements=True))
# join_string = self.getConfig("join_string_"+key,u", ").replace(SPACE_REPLACE,' ')
# value = join_string.join(self.getList(key, removeallentities, doreplacements=True))
value = self.join_list(key,self.getList(key, removeallentities, doreplacements=True))
if doreplacements:
value = self.doReplacements(value,key+"_LIST")
return value
@@ -494,7 +521,8 @@ class Story(Configurable):
doreplacements=True,
keeplists=False):
'''
All single value *and* list value metadata as strings (unless keeplists=True, then keep lists).
All single value *and* list value metadata as strings (unless
keeplists=True, then keep lists).
'''
allmetadata = {}
@@ -514,16 +542,16 @@ class Story(Configurable):
auth=removeAllEntities(auth)
htmllist.append(linkhtml%('author',aurl,auth))
join_string = self.getConfig("join_string_authorHTML",u", ").replace(SPACE_REPLACE,' ')
self.setMetadata('authorHTML',join_string.join(htmllist))
# join_string = self.getConfig("join_string_authorHTML",u", ").replace(SPACE_REPLACE,' ')
self.setMetadata('authorHTML',self.join_list("join_string_authorHTML",htmllist))
else:
self.setMetadata('authorHTML',linkhtml%('author',self.getMetadata('authorUrl', removeallentities, doreplacements),
self.getMetadata('author', removeallentities, doreplacements)))
if self.getMetadataRaw('seriesUrl') != None:
if self.getMetadataRaw('seriesUrl'):
self.setMetadata('seriesHTML',linkhtml%('series',self.getMetadata('seriesUrl', removeallentities, doreplacements),
self.getMetadata('series', removeallentities, doreplacements)))
elif self.getMetadataRaw('series') != None:
elif self.getMetadataRaw('series'):
self.setMetadata('seriesHTML',self.getMetadataRaw('series'))
# logger.debug("make_linkhtml_entries:%s"%self.getConfig('make_linkhtml_entries'))
@@ -546,8 +574,8 @@ class Story(Configurable):
v=removeAllEntities(v)
htmllist.append(linkhtml%(k,url,v))
join_string = self.getConfig("join_string_"+k+"HTML",u", ").replace(SPACE_REPLACE,' ')
self.setMetadata(k+'HTML',join_string.join(htmllist))
# join_string = self.getConfig("join_string_"+k+"HTML",u", ").replace(SPACE_REPLACE,' ')
self.setMetadata(k+'HTML',self.join_list("join_string_"+k+"HTML",htmllist))
for k in self.getValidMetaList():
if self.isList(k) and keeplists:
@@ -562,11 +590,12 @@ class Story(Configurable):
for v in l:
self.addToList(listname,v.strip())
def addToList(self,listname,value):
def addToList(self,listname,value,condremoveentities=True,clear=False):
if value==None:
return
value = conditionalRemoveEntities(value)
if not self.isList(listname) or not listname in self.metadata:
if condremoveentities:
value = conditionalRemoveEntities(value)
if clear or not self.isList(listname) or not listname in self.metadata:
# Calling addToList to a non-list meta will overwrite it.
self.metadata[listname]=[]
# prevent duplicates.
@@ -578,7 +607,7 @@ class Story(Configurable):
def isList(self,listname):
'Everything set with an include_in_* is considered a list.'
return self.hasConfig("include_in_"+listname) or \
return self.isListType(listname) or \
( self.isValidMetaEntry(listname) and self.metadata.has_key(listname) \
and isinstance(self.metadata[listname],list) )
@@ -607,16 +636,20 @@ class Story(Configurable):
if retlist:
if doreplacements:
retlist = filter( lambda x : x!=None and x!='' ,
map(partial(self.doReplacements,key=listname),retlist) )
newretlist = []
for val in retlist:
newretlist.extend(self.doReplacements(val,listname,return_list=True))
retlist = newretlist
if removeallentities:
retlist = filter( lambda x : x!=None and x!='' ,
map(removeAllEntities,retlist) )
retlist = map(removeAllEntities,retlist)
retlist = filter( lambda x : x!=None and x!='' ,retlist)
# reorder ships so b/a and c/b/a become a/b and a/b/c. Only on '/',
# use replace_metadata to change separator first if needed.
# ships=>[ ]*(/|&amp;|&)[ ]*=>/
if listname == 'ships' and self.getConfig('sort_ships'):
if listname == 'ships' and self.getConfig('sort_ships') and retlist:
retlist = [ '/'.join(sorted(x.split('/'))) for x in retlist ]
if retlist:
+3 -5
View File
@@ -60,10 +60,8 @@
</p>
<p>
<ul>
<li>New site: nocturnal-light.net -- Thanks, cryzed!</li>
<li>New site: fanfic.hu (Hungarian language) -- Thanks, cryzed!</li>
<li>New site: fanfiction.csodaidok.hu (Hungarian language) -- Thanks, cryzed!</li>
<li>Improvements &amp; fixes for recently added sites -- Thanks, cryzed!</li>
<li>Fixes for some metadata problems on various sites.</li>
<li>Known problem: specific metadata 'eroticatags' for literotica.com doesn't work on all stories.</li>
</ul>
</p>
<p>
@@ -74,7 +72,7 @@
If you have any problems with this application, please
report them in
the <a href="http://groups.google.com/group/fanfic-downloader">FanFictionDownLoader Google Group</a>. The
<a href="http://4-5-03.fanfictiondownloader.appspot.com">Previous Version</a> is also available for you to use if necessary.
<a href="http://4-5-06.fanfictiondownloader.appspot.com">Previous Version</a> is also available for you to use if necessary.
</p>
<div id='error'>
{{ error_message }}
+74 -30
View File
@@ -874,6 +874,45 @@ extraships:Harry Potter/Hermione Granger
#username:YourName
#password:yourpassword
[fictionmania.tv]
## website encoding(s) In theory, each website reports the character
## encoding they use for each page. In practice, some sites report it
## incorrectly. Each adapter has a default list, usually "utf8,
## Windows-1252" or "Windows-1252, utf8", but this will let you
## explicitly set the encoding and order if you need to. The special
## value 'auto' will call chardet and use the encoding it reports if
## it has +90% confidence. 'auto' is not reliable.
website_encodings:ISO-8859-1,auto
## items to include in the log page Empty metadata entries, or those
## that haven't changed since the last update, will *not* appear, even
## if in the list. You can include extra text or HTML that will be
## included as-is in each log entry. Eg: logpage_entries: ...,<br />,
## summary,<br />,...
## Don't include numChapters since all stories are a single "chapter", there's
## no way to reliably find the next chapter
logpage_entries: dateCreated,datePublished,dateUpdated,numChapters,numWords,status,series,title,author,description,category,genre,rating,warnings
## items to include in the title page
## Empty metadata entries will *not* appear, even if in the list.
## You can include extra text or HTML that will be included as-is in
## the title page. Eg: titlepage_entries: ...,<br />,summary,<br />,...
## All current formats already include title and author.
## Don't include numChapters since all stories are a single "chapter", there's
## no way to reliably find the next chapter
titlepage_entries: seriesHTML,category,genre,language,characters,ships,status,datePublished,dateUpdated,dateCreated,rating,warnings,numWords,site,description
## Extra metadata that this adapter knows about. See [dramione.org]
## for examples of how to use them.
extra_valid_entries:fileName,fileSize,oldName,newName,keyWords,mainCharactersAge,readings
## Turns all space characters into "&nbsp" HTML entities to forcefully preserve
## formatting with spaces. Enabling this will blow up the filesize quite a bit
## and is probably not a good idea, unless you absolutely need the story
## formatting.
## Specific to fictionmania.tv
non_breaking_spaces:false
[fictionpad.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -899,36 +938,6 @@ dislikes_label:Dislikes
#username:YourName
#password:yourpassword
[storiesonline.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Clear FanFiction from defaults, site is original fiction.
extratags:
extra_valid_entries:size,universe,universeUrl,universeHTML,codes,notice
#extra_titlepage_entries:size,universeHTML,codes,notice
size_label:Size
universe_label:Universe
universeUrl_label:Universe URL
universeHTML_label:Universe
codes_label:Codes
notice_label:Notice
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:universe
## storiesonline.net stories can be in a series or a universe, but not
## both. By default, universe will be populated in 'series' with
## index=0
universe_as_series: true
[grangerenchanted.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
@@ -981,6 +990,11 @@ extracategories:Star Trek
extracharacters:Kirk,Spock
extraships:Kirk/Spock
[literotica.com]
extra_valid_entries:eroticatags
eroticatags_label:Erotica Tags
#extra_titlepage_entries: eroticatags
[lumos.sycophanthex.com]
## Some sites do not require a login, but do require the user to
## confirm they are adult for adult content. In commandline version,
@@ -1145,6 +1159,36 @@ extracategories:Buffy the Vampire Slayer
## Site dedicated to these categories/characters/ships
extracategories:Stargate: Atlantis
[storiesonline.net]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In
## commandline version, this should go in your personal.ini, not
## defaults.ini.
#username:YourName
#password:yourpassword
## Clear FanFiction from defaults, site is original fiction.
extratags:
extra_valid_entries:size,universe,universeUrl,universeHTML,codes,notice
#extra_titlepage_entries:size,universeHTML,codes,notice
size_label:Size
universe_label:Universe
universeUrl_label:Universe URL
universeHTML_label:Universe
codes_label:Codes
notice_label:Notice
## Assume entryUrl, apply to "<a class='%slink' href='%s'>%s</a>" to
## make entryHTML.
make_linkhtml_entries:universe
## storiesonline.net stories can be in a series or a universe, but not
## both. By default, universe will be populated in 'series' with
## index=0
universe_as_series: true
[svufiction.com]
## Some sites require login (or login for some rated stories) The
## program can prompt you, or you can save it in config. In