diff --git a/defaults.ini b/defaults.ini index 36c0c0e..bb643a2 100644 --- a/defaults.ini +++ b/defaults.ini @@ -790,6 +790,19 @@ extracategories:Harry Potter ## cover image. This lets you exclude them. cover_exclusion_regexp:/images/.*?ribbon.gif +[fanfic.hu] +## website encoding(s) In theory, each website reports the character +## encoding they use for each page. In practice, some sites report it +## incorrectly. Each adapter has a default list, usually "utf8, +## Windows-1252" or "Windows-1252, utf8", but this will let you +## explicitly set the encoding and order if you need to. The special +## value 'auto' will call chardet and use the encoding it reports if +## it has +90% confidence. 'auto' is not reliable. +website_encodings:ISO-8859-1,auto + +## Site dedicated to these categories/characters/ships +extracategories:Harry Potter + [fanfiction.mugglenet.com] ## Some sites do not require a login, but do require the user to ## confirm they are adult for adult content. In commandline version, diff --git a/fanficdownloader/adapters/__init__.py b/fanficdownloader/adapters/__init__.py index 64d683d..94f2de2 100644 --- a/fanficdownloader/adapters/__init__.py +++ b/fanficdownloader/adapters/__init__.py @@ -127,6 +127,7 @@ import adapter_voracity2eficcom import adapter_spikeluvercom import adapter_bloodshedversecom import adapter_nocturnallightnet +import adapter_fanfichu ## This bit of complexity allows adapters to be added by just adding ## importing. It eliminates the long if/else clauses we used to need diff --git a/fanficdownloader/adapters/adapter_fanfichu.py b/fanficdownloader/adapters/adapter_fanfichu.py new file mode 100644 index 0000000..c4d132c --- /dev/null +++ b/fanficdownloader/adapters/adapter_fanfichu.py @@ -0,0 +1,179 @@ +# coding=utf-8 + +import re +import urllib2 +import urlparse + +from .. import BeautifulSoup + +from base_adapter import BaseSiteAdapter, makeDate +from .. import exceptions + + +_SOURCE_CODE_ENCODING = 'utf-8' + + +def getClass(): + return FanficHuAdapter + + +def _get_query_data(url): + components = urlparse.urlparse(url) + query_data = urlparse.parse_qs(components.query) + return dict((key, data[0]) for key, data in query_data.items()) + + +class FanficHuAdapter(BaseSiteAdapter): + SITE_ABBREVIATION = 'ffh' + SITE_DOMAIN = 'fanfic.hu' + + BASE_URL = 'http://' + SITE_DOMAIN + '/merengo/' + VIEW_STORY_URL_TEMPLATE = BASE_URL + 'viewstory.php?sid=%s' + + DATE_FORMAT = '%m/%d/%Y' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + query_data = urlparse.parse_qs(self.parsedUrl.query) + story_id = query_data['sid'][0] + + self.story.setMetadata('storyId', story_id) + self._setURL(self.VIEW_STORY_URL_TEMPLATE % story_id) + self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return BeautifulSoup.BeautifulSoup(data) + + @staticmethod + def getSiteDomain(): + return FanficHuAdapter.SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls.VIEW_STORY_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return re.escape(self.VIEW_STORY_URL_TEMPLATE[:-2]) + r'\d+$' + + def extractChapterUrlsAndMetadata(self): + soup = self._customized_fetch_url(self.url + '&i=1') + + if soup.title.string.encode(_SOURCE_CODE_ENCODING).strip(' :') == 'írta': + raise exceptions.StoryDoesNotExist(self.url) + + chapter_options = soup.find('form', action='viewstory.php').select('option') + # Remove redundant "Fejezetek" option + chapter_options.pop(0) + + # If there is still more than one entry remove chapter overview entry + if len(chapter_options) > 1: + chapter_options.pop(0) + + for option in chapter_options: + url = urlparse.urljoin(self.url, option['value']) + self.chapterUrls.append((option.string, url)) + + author_url = urlparse.urljoin(self.BASE_URL, soup.find('a', href=lambda href: href and href.startswith('viewuser.php?uid='))['href']) + soup = self._customized_fetch_url(author_url) + + story_id = self.story.getMetadata('storyId') + for table in soup('table', {'class': 'mainnav'}): + title_anchor = table.find('span', {'class': 'storytitle'}).a + query_data = _get_query_data(title_anchor['href']) + + if query_data['sid'] == story_id: + break + + title = title_anchor.string + self.story.setMetadata('title', title) + if not chapter_options: + self.chapterUrls.append((title, self.url)) + + rows = table('tr') + anchors = rows[0].div('a') + + author_anchor = anchors[1] + query_data = _get_query_data(author_anchor['href']) + self.story.setMetadata('author', author_anchor.string) + self.story.setMetadata('authorId', query_data['uid']) + self.story.setMetadata('authorUrl', urlparse.urljoin(self.BASE_URL, author_anchor['href'])) + self.story.setMetadata('reviews', anchors[3].string) + + if self.getConfig('keep_summary_html'): + self.story.setMetadata('description', self.utf8FromSoup(author_url, rows[1].td)) + else: + self.story.setMetadata('description', ''.join(rows[1].td(text=True))) + + for row in rows[3:]: + index = 0 + cells = row('td') + + while index < len(cells): + cell = cells[index] + key = cell.b.string.encode(_SOURCE_CODE_ENCODING).strip(':') + try: + value = cells[index+1].string.encode(_SOURCE_CODE_ENCODING) + except AttributeError: + value = None + + if key == 'Kategória': + for anchor in cells[index+1]('a'): + self.story.addToList('category', anchor.string) + + elif key == 'Szereplõk': + if cells[index+1].string: + for name in cells[index+1].string.split(', '): + self.story.addToList('character', name) + + elif key == 'Korhatár': + if value != 'nem korhatáros': + self.story.setMetadata('rating', value) + + elif key == 'Figyelmeztetések': + for b_tag in cells[index+1]('b'): + self.story.addToList('warnings', b_tag.string) + + elif key == 'Jellemzõk': + for genre in cells[index+1].string.split(', '): + self.story.addToList('genre', genre) + + elif key == 'Fejezetek': + self.story.setMetadata('numChapters', int(value)) + + elif key == 'Megjelenés': + self.story.setMetadata('datePublished', makeDate(value, self.DATE_FORMAT)) + + elif key == 'Frissítés': + self.story.setMetadata('dateUpdated', makeDate(value, self.DATE_FORMAT)) + + elif key == 'Szavak': + self.story.setMetadata('words', value) + + elif key == 'Befejezett': + self.story.setMetadata('status', 'Completed' if value == 'Nem' else 'In-Progress') + + index += 2 + + if self.story.getMetadata('rating') == '18': + if not (self.is_adult or self.getConfig('is_adult')): + raise exceptions.AdultCheckRequired(self.url) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + story_cell = soup.find('form', action='viewstory.php').parent.parent + + for div in story_cell('div'): + div.extract() + + return self.utf8FromSoup(url, story_cell) diff --git a/plugin-defaults.ini b/plugin-defaults.ini index fee34e7..91627f0 100644 --- a/plugin-defaults.ini +++ b/plugin-defaults.ini @@ -781,6 +781,19 @@ extracategories:Harry Potter ## cover image. This lets you exclude them. cover_exclusion_regexp:/images/.*?ribbon.gif +[fanfic.hu] +## website encoding(s) In theory, each website reports the character +## encoding they use for each page. In practice, some sites report it +## incorrectly. Each adapter has a default list, usually "utf8, +## Windows-1252" or "Windows-1252, utf8", but this will let you +## explicitly set the encoding and order if you need to. The special +## value 'auto' will call chardet and use the encoding it reports if +## it has +90% confidence. 'auto' is not reliable. +website_encodings:ISO-8859-1,auto + +## Site dedicated to these categories/characters/ships +extracategories:Harry Potter + [fanfiction.mugglenet.com] ## Some sites do not require a login, but do require the user to ## confirm they are adult for adult content. In commandline version,