From b3e72a271c16ccbd186c8f8af093f42c1c9ec904 Mon Sep 17 00:00:00 2001 From: Jim Miller Date: Wed, 8 Apr 2015 13:49:36 -0500 Subject: [PATCH] Move fff_internals package to fanficfare, share defaults.ini/example.ini between cli and web service again. --- .gitignore | 3 + calibre-plugin/config.py | 2 +- calibre-plugin/dialogs.py | 6 +- calibre-plugin/ffdl_plugin.py | 6 +- calibre-plugin/ffdl_util.py | 4 +- calibre-plugin/jobs.py | 4 +- .../BeautifulSoup.py | 0 {fff_internals => fanficfare}/HtmlTagStack.py | 112 +- {fff_internals => fanficfare}/__init__.py | 0 .../adapters/__init__.py | 510 ++--- .../adapters/adapter_adastrafanficcom.py | 476 ++-- .../adapters/adapter_archiveofourownorg.py | 0 .../adapters/adapter_archiveskyehawkecom.py | 382 ++-- .../adapter_ashwindersycophanthexcom.py | 508 ++--- .../adapters/adapter_asr3slashzoneorg.py | 454 ++-- .../adapters/adapter_bdsmgeschichten.py | 0 .../adapters/adapter_bloodshedversecom.py | 386 ++-- .../adapters/adapter_bloodtiesfancom.py | 674 +++--- .../adapters/adapter_buffynfaithnet.py | 592 ++--- .../adapters/adapter_chaossycophanthexcom.py | 476 ++-- .../adapters/adapter_checkmatedcom.py | 476 ++-- .../adapters/adapter_csiforensicscom.py | 472 ++-- .../adapters/adapter_darksolaceorg.py | 672 +++--- .../adapters/adapter_destinysgatewaycom.py | 488 ++-- .../adapters/adapter_devianthearts.py | 106 +- .../adapters/adapter_dokugacom.py | 556 ++--- .../adapters/adapter_dotmoonnet.py | 434 ++-- .../adapters/adapter_dracoandginnycom.py | 604 ++--- .../adapters/adapter_dramioneorg.py | 622 ++--- .../adapters/adapter_efictionestelielde.py | 448 ++-- .../adapters/adapter_efpfanficnet.py | 632 +++--- .../adapter_erosnsapphosycophanthexcom.py | 512 ++--- .../adapters/adapter_fanficcastletvnet.py | 644 +++--- .../adapters/adapter_fanfichu.py | 370 +-- .../adapters/adapter_fanfictioncsodaidokhu.py | 436 ++-- .../adapters/adapter_fanfictionjunkiesde.py | 582 ++--- .../adapters/adapter_fanfictionnet.py | 682 +++--- .../adapters/adapter_fanfiktionde.py | 412 ++-- .../adapters/adapter_fannation.py | 0 .../adapters/adapter_fhsarchivecom.py | 0 .../adapters/adapter_ficbooknet.py | 2 +- .../adapters/adapter_fictionalleyorg.py | 482 ++-- .../adapters/adapter_fictionmaniatv.py | 332 +-- .../adapters/adapter_fictionpadcom.py | 388 ++-- .../adapters/adapter_fictionpresscom.py | 102 +- .../adapters/adapter_ficwadcom.py | 466 ++-- .../adapters/adapter_fimfictionnet.py | 714 +++--- .../adapters/adapter_finestoriescom.py | 576 ++--- .../adapters/adapter_grangerenchantedcom.py | 622 ++--- .../adapter_harrypotterfanfictioncom.py | 406 ++-- .../adapters/adapter_hennethannunnet.py | 344 +-- .../adapters/adapter_hlfictionnet.py | 464 ++-- .../adapters/adapter_hpfandomnet.py | 466 ++-- .../adapters/adapter_hpfanficarchivecom.py | 446 ++-- .../adapters/adapter_iketernalnet.py | 566 ++--- .../adapters/adapter_imagineeficcom.py | 580 ++--- .../adapters/adapter_indeathnet.py | 400 ++-- .../adapters/adapter_ksarchivecom.py | 660 +++--- .../adapters/adapter_libraryofmoriacom.py | 502 ++--- .../adapters/adapter_literotica.py | 0 .../adapters/adapter_lotrfanfictioncom.py | 72 +- .../adapters/adapter_lumossycophanthexcom.py | 476 ++-- .../adapters/adapter_mediaminerorg.py | 474 ++-- .../adapters/adapter_merlinficdtwinscouk.py | 588 ++--- .../adapters/adapter_midnightwhispersca.py | 580 ++--- .../adapters/adapter_mugglenetcom.py | 672 +++--- .../adapters/adapter_nationallibrarynet.py | 426 ++-- .../adapters/adapter_ncisficcom.py | 438 ++-- .../adapters/adapter_ncisfictionnet.py | 420 ++-- .../adapters/adapter_netraptororg.py | 426 ++-- .../adapters/adapter_nfacommunitycom.py | 580 ++--- .../adapters/adapter_nhamagicalworldsus.py | 474 ++-- .../adapters/adapter_nickandgregnet.py | 354 +-- .../adapters/adapter_nocturnallightnet.py | 354 +-- .../adapter_occlumencysycophanthexcom.py | 526 ++--- .../adapter_onedirectionfanfictioncom.py | 540 ++--- .../adapters/adapter_phoenixsongnet.py | 480 ++-- .../adapters/adapter_pommedesangcom.py | 602 ++--- .../adapters/adapter_ponyfictionarchivenet.py | 498 ++-- .../adapters/adapter_portkeyorg.py | 564 ++--- .../adapters/adapter_potionsandsnitches.py | 422 ++-- .../adapters/adapter_potterficscom.py | 562 ++--- .../adapter_potterheadsanonymouscom.py | 598 ++--- .../adapters/adapter_pretendercentrecom.py | 508 ++--- .../adapters/adapter_psychficcom.py | 498 ++-- .../adapters/adapter_qafficcom.py | 530 ++--- .../adapters/adapter_restrictedsectionorg.py | 530 ++--- .../adapters/adapter_samandjacknet.py | 684 +++--- .../adapters/adapter_samdeanarchivenu.py | 466 ++-- .../adapters/adapter_scarheadnet.py | 600 ++--- .../adapters/adapter_scarvesandcoffeenet.py | 496 ++-- .../adapters/adapter_sg1heliopoliscom.py | 518 ++--- .../adapters/adapter_sheppardweircom.py | 634 +++--- .../adapters/adapter_simplyundeniablecom.py | 440 ++-- .../adapters/adapter_sinfuldesireorg.py | 504 ++--- .../adapters/adapter_siyecouk.py | 494 ++-- .../adapters/adapter_spikeluvercom.py | 416 ++-- .../adapters/adapter_squidgeorgpeja.py | 512 ++--- .../adapters/adapter_stargateatlantisorg.py | 460 ++-- .../adapters/adapter_storiesofardacom.py | 322 +-- .../adapters/adapter_storiesonlinenet.py | 844 +++---- .../adapters/adapter_tenhawkpresentscom.py | 514 ++--- .../adapters/adapter_test1.py | 740 +++--- .../adapters/adapter_thealphagatecom.py | 430 ++-- .../adapters/adapter_thehexfilesnet.py | 414 ++-- .../adapters/adapter_thehookupzonenet.py | 618 ++--- .../adapters/adapter_themaplebookshelf.py | 80 +- .../adapters/adapter_themasquenet.py | 548 ++--- .../adapters/adapter_thepetulantpoetesscom.py | 484 ++-- .../adapters/adapter_thequidditchpitchorg.py | 584 ++--- .../adapters/adapter_tokrafandomnetcom.py | 476 ++-- .../adapters/adapter_tolkienfanfiction.py | 0 .../adapters/adapter_trekiverseorg.py | 650 +++--- .../adapters/adapter_tthfanficorg.py | 620 ++--- .../adapters/adapter_twcslibrarynet.py | 544 ++--- .../adapters/adapter_twilightarchivescom.py | 376 ++-- .../adapters/adapter_twilightednet.py | 506 ++--- .../adapters/adapter_twiwritenet.py | 562 ++--- .../adapters/adapter_voracity2eficcom.py | 466 ++-- .../adapters/adapter_walkingtheplankorg.py | 464 ++-- .../adapters/adapter_whoficcom.py | 476 ++-- .../adapters/adapter_wizardtalesnet.py | 606 ++--- .../adapters/adapter_wolverineandroguecom.py | 438 ++-- .../adapters/adapter_wraithbaitcom.py | 468 ++-- .../adapters/base_adapter.py | 1224 +++++----- .../adapters/base_efiction_adapter.py | 852 +++---- {fff_internals => fanficfare}/cli.py | 18 +- {fff_internals => fanficfare}/configurable.py | 1212 +++++----- {fff_internals => fanficfare}/defaults.ini | 2 +- {fff_internals => fanficfare}/epubutils.py | 388 ++-- {fff_internals => fanficfare}/example.ini | 0 {fff_internals => fanficfare}/exceptions.py | 186 +- {fff_internals => fanficfare}/geturls.py | 466 ++-- {fff_internals => fanficfare}/gziphttp.py | 76 +- {fff_internals => fanficfare}/html.py | 0 {fff_internals => fanficfare}/html2text.py | 0 {fff_internals => fanficfare}/htmlcleanup.py | 970 ++++---- .../htmlheuristics.py | 724 +++--- {fff_internals => fanficfare}/mobi.py | 0 {fff_internals => fanficfare}/story.py | 1848 +++++++-------- {fff_internals => fanficfare}/translit.py | 112 +- .../writers/__init__.py | 76 +- .../writers/base_writer.py | 570 ++--- .../writers/writer_epub.py | 1380 ++++++------ .../writers/writer_html.py | 288 +-- .../writers/writer_mobi.py | 394 ++-- .../writers/writer_txt.py | 380 ++-- makeplugin.py | 2 +- setup.py | 8 +- webservice/Readme.txt | 7 +- webservice/defaults.ini | 2003 ----------------- webservice/example.ini | 103 - webservice/main.py | 10 +- 153 files changed, 32550 insertions(+), 34656 deletions(-) rename {fff_internals => fanficfare}/BeautifulSoup.py (100%) rename {fff_internals => fanficfare}/HtmlTagStack.py (95%) rename {fff_internals => fanficfare}/__init__.py (100%) rename {fff_internals => fanficfare}/adapters/__init__.py (96%) rename {fff_internals => fanficfare}/adapters/adapter_adastrafanficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_archiveofourownorg.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_archiveskyehawkecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ashwindersycophanthexcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_asr3slashzoneorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_bdsmgeschichten.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_bloodshedversecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_bloodtiesfancom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_buffynfaithnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_chaossycophanthexcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_checkmatedcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_csiforensicscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_darksolaceorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_destinysgatewaycom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_devianthearts.py (96%) rename {fff_internals => fanficfare}/adapters/adapter_dokugacom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_dotmoonnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_dracoandginnycom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_dramioneorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_efictionestelielde.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_efpfanficnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_erosnsapphosycophanthexcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanficcastletvnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanfichu.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanfictioncsodaidokhu.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanfictionjunkiesde.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanfictionnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fanfiktionde.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fannation.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_fhsarchivecom.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_ficbooknet.py (99%) rename {fff_internals => fanficfare}/adapters/adapter_fictionalleyorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fictionmaniatv.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fictionpadcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fictionpresscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ficwadcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_fimfictionnet.py (98%) rename {fff_internals => fanficfare}/adapters/adapter_finestoriescom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_grangerenchantedcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_harrypotterfanfictioncom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_hennethannunnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_hlfictionnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_hpfandomnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_hpfanficarchivecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_iketernalnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_imagineeficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_indeathnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ksarchivecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_libraryofmoriacom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_literotica.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_lotrfanfictioncom.py (96%) rename {fff_internals => fanficfare}/adapters/adapter_lumossycophanthexcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_mediaminerorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_merlinficdtwinscouk.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_midnightwhispersca.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_mugglenetcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_nationallibrarynet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ncisficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ncisfictionnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_netraptororg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_nfacommunitycom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_nhamagicalworldsus.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_nickandgregnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_nocturnallightnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_occlumencysycophanthexcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_onedirectionfanfictioncom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_phoenixsongnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_pommedesangcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_ponyfictionarchivenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_portkeyorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_potionsandsnitches.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_potterficscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_potterheadsanonymouscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_pretendercentrecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_psychficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_qafficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_restrictedsectionorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_samandjacknet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_samdeanarchivenu.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_scarheadnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_scarvesandcoffeenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_sg1heliopoliscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_sheppardweircom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_simplyundeniablecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_sinfuldesireorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_siyecouk.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_spikeluvercom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_squidgeorgpeja.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_stargateatlantisorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_storiesofardacom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_storiesonlinenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_tenhawkpresentscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_test1.py (98%) rename {fff_internals => fanficfare}/adapters/adapter_thealphagatecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_thehexfilesnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_thehookupzonenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_themaplebookshelf.py (96%) rename {fff_internals => fanficfare}/adapters/adapter_themasquenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_thepetulantpoetesscom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_thequidditchpitchorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_tokrafandomnetcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_tolkienfanfiction.py (100%) rename {fff_internals => fanficfare}/adapters/adapter_trekiverseorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_tthfanficorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_twcslibrarynet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_twilightarchivescom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_twilightednet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_twiwritenet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_voracity2eficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_walkingtheplankorg.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_whoficcom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_wizardtalesnet.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_wolverineandroguecom.py (97%) rename {fff_internals => fanficfare}/adapters/adapter_wraithbaitcom.py (97%) rename {fff_internals => fanficfare}/adapters/base_adapter.py (97%) rename {fff_internals => fanficfare}/adapters/base_efiction_adapter.py (97%) rename {fff_internals => fanficfare}/cli.py (94%) rename {fff_internals => fanficfare}/configurable.py (97%) rename {fff_internals => fanficfare}/defaults.ini (99%) rename {fff_internals => fanficfare}/epubutils.py (97%) rename {fff_internals => fanficfare}/example.ini (100%) rename {fff_internals => fanficfare}/exceptions.py (96%) rename {fff_internals => fanficfare}/geturls.py (97%) rename {fff_internals => fanficfare}/gziphttp.py (97%) rename {fff_internals => fanficfare}/html.py (100%) rename {fff_internals => fanficfare}/html2text.py (100%) rename {fff_internals => fanficfare}/htmlcleanup.py (96%) rename {fff_internals => fanficfare}/htmlheuristics.py (97%) rename {fff_internals => fanficfare}/mobi.py (100%) rename {fff_internals => fanficfare}/story.py (97%) rename {fff_internals => fanficfare}/translit.py (97%) rename {fff_internals => fanficfare}/writers/__init__.py (97%) rename {fff_internals => fanficfare}/writers/base_writer.py (97%) rename {fff_internals => fanficfare}/writers/writer_epub.py (97%) rename {fff_internals => fanficfare}/writers/writer_html.py (96%) rename {fff_internals => fanficfare}/writers/writer_mobi.py (97%) rename {fff_internals => fanficfare}/writers/writer_txt.py (96%) delete mode 100644 webservice/defaults.ini delete mode 100644 webservice/example.ini diff --git a/.gitignore b/.gitignore index 33f800d..3743024 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,9 @@ # Windows batch files *.bat +# usually perl -pi.back -e edits. +*.back + cleanup.sh FanFictionDownLoader.zip *.epub diff --git a/calibre-plugin/config.py b/calibre-plugin/config.py index e3face4..2b3f267 100644 --- a/calibre-plugin/config.py +++ b/calibre-plugin/config.py @@ -75,7 +75,7 @@ from calibre_plugins.fanficfare_plugin.dialogs \ import (UPDATE, UPDATEALWAYS, collision_order, save_collisions, RejectListDialog, EditTextDialog, IniTextDialog, RejectUrlEntry) -from calibre_plugins.fanficfare_plugin.fff_internals.adapters \ +from calibre_plugins.fanficfare_plugin.fanficfare.adapters \ import getConfigSections from calibre_plugins.fanficfare_plugin.common_utils \ diff --git a/calibre-plugin/dialogs.py b/calibre-plugin/dialogs.py index 43d5ee9..c16733a 100644 --- a/calibre-plugin/dialogs.py +++ b/calibre-plugin/dialogs.py @@ -64,10 +64,10 @@ from calibre_plugins.fanficfare_plugin.common_utils \ import (ReadOnlyTableWidgetItem, ReadOnlyTextIconWidgetItem, SizePersistedDialog, ImageTitleLayout, get_icon) -from calibre_plugins.fanficfare_plugin.fff_internals.geturls import get_urls_from_html, get_urls_from_text -from calibre_plugins.fanficfare_plugin.fff_internals.adapters import getNormalStoryURL +from calibre_plugins.fanficfare_plugin.fanficfare.geturls import get_urls_from_html, get_urls_from_text +from calibre_plugins.fanficfare_plugin.fanficfare.adapters import getNormalStoryURL -from calibre_plugins.fanficfare_plugin.fff_internals.configurable \ +from calibre_plugins.fanficfare_plugin.fanficfare.configurable \ import (get_valid_sections, get_valid_entries, get_valid_keywords, get_valid_entry_keywords) diff --git a/calibre-plugin/ffdl_plugin.py b/calibre-plugin/ffdl_plugin.py index 4b46c99..d25431a 100644 --- a/calibre-plugin/ffdl_plugin.py +++ b/calibre-plugin/ffdl_plugin.py @@ -51,9 +51,9 @@ except NameError: from calibre_plugins.fanficfare_plugin.common_utils import (set_plugin_icon_resources, get_icon, create_menu_action_unique, get_library_uuid) -from calibre_plugins.fanficfare_plugin.fff_internals import adapters, exceptions -from calibre_plugins.fanficfare_plugin.fff_internals.epubutils import get_dcsource, get_dcsource_chaptercount, get_story_url_from_html -from calibre_plugins.fanficfare_plugin.fff_internals.geturls import get_urls_from_page, get_urls_from_html, get_urls_from_text, get_urls_from_imap +from calibre_plugins.fanficfare_plugin.fanficfare import adapters, exceptions +from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import get_dcsource, get_dcsource_chaptercount, get_story_url_from_html +from calibre_plugins.fanficfare_plugin.fanficfare.geturls import get_urls_from_page, get_urls_from_html, get_urls_from_text, get_urls_from_imap from calibre_plugins.fanficfare_plugin.ffdl_util import (get_ffdl_adapter, get_ffdl_config, get_ffdl_personalini) from calibre_plugins.fanficfare_plugin.config import (permitted_values, rejecturllist) diff --git a/calibre-plugin/ffdl_util.py b/calibre-plugin/ffdl_util.py index 2fbbac8..7a161de 100644 --- a/calibre-plugin/ffdl_util.py +++ b/calibre-plugin/ffdl_util.py @@ -13,8 +13,8 @@ from ConfigParser import ParsingError import logging logger = logging.getLogger(__name__) -from calibre_plugins.fanficfare_plugin.fff_internals import adapters, exceptions -from calibre_plugins.fanficfare_plugin.fff_internals.configurable import Configuration +from calibre_plugins.fanficfare_plugin.fanficfare import adapters, exceptions +from calibre_plugins.fanficfare_plugin.fanficfare.configurable import Configuration from calibre_plugins.fanficfare_plugin.prefs import prefs def get_ffdl_personalini(): diff --git a/calibre-plugin/jobs.py b/calibre-plugin/jobs.py index e2ef343..2805df3 100644 --- a/calibre-plugin/jobs.py +++ b/calibre-plugin/jobs.py @@ -101,8 +101,8 @@ def do_download_for_worker(book,options,notification=lambda x,y:x): from calibre_plugins.fanficfare_plugin.dialogs import (NotGoingToDownload, OVERWRITE, OVERWRITEALWAYS, UPDATE, UPDATEALWAYS, ADDNEW, SKIP, CALIBREONLY) - from calibre_plugins.fanficfare_plugin.fff_internals import adapters, writers, exceptions - from calibre_plugins.fanficfare_plugin.fff_internals.epubutils import get_update_data + from calibre_plugins.fanficfare_plugin.fanficfare import adapters, writers, exceptions + from calibre_plugins.fanficfare_plugin.fanficfare.epubutils import get_update_data from calibre_plugins.fanficfare_plugin.ffdl_util import (get_ffdl_adapter, get_ffdl_config) diff --git a/fff_internals/BeautifulSoup.py b/fanficfare/BeautifulSoup.py similarity index 100% rename from fff_internals/BeautifulSoup.py rename to fanficfare/BeautifulSoup.py diff --git a/fff_internals/HtmlTagStack.py b/fanficfare/HtmlTagStack.py similarity index 95% rename from fff_internals/HtmlTagStack.py rename to fanficfare/HtmlTagStack.py index f166ff3..3a9e703 100644 --- a/fff_internals/HtmlTagStack.py +++ b/fanficfare/HtmlTagStack.py @@ -1,57 +1,57 @@ -# coding: utf-8 - -import re -import codecs - -stack = [] - -def get_end_tag(tag): - if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: - return re.sub(r'.*<([^\ >]+).*', r'', tag) - return u'' - -def get_tag_name(tag): - if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: - return re.sub(r']+).*', r'\1', tag) - return u'' - -def push(tag): - if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: - stack.append(tag) - -def pop(): - if len(stack) > 0: - return stack.pop() - return u'' - -def pop_end_tag(): - return unicode(get_end_tag(pop())) - -def spool_end(): - html = u'' - for tag in reversed(stack): - html += get_end_tag(tag) - return html - -def spool_start(): - html = u'' - for item in stack: - html += item - return html - -def has_elements(): - return len(stack) > 0 - -def get_last(): - # t = pop() - # push(t) - # return t - if len(stack) > 0: - return stack[len(stack)-1] - return u'' - -def flush(): - del stack[:] - -def get_stack(): +# coding: utf-8 + +import re +import codecs + +stack = [] + +def get_end_tag(tag): + if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: + return re.sub(r'.*<([^\ >]+).*', r'', tag) + return u'' + +def get_tag_name(tag): + if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: + return re.sub(r']+).*', r'\1', tag) + return u'' + +def push(tag): + if len(tag) > 0 and tag.find(u'<') > -1 and tag.rfind(u'>') > -1: + stack.append(tag) + +def pop(): + if len(stack) > 0: + return stack.pop() + return u'' + +def pop_end_tag(): + return unicode(get_end_tag(pop())) + +def spool_end(): + html = u'' + for tag in reversed(stack): + html += get_end_tag(tag) + return html + +def spool_start(): + html = u'' + for item in stack: + html += item + return html + +def has_elements(): + return len(stack) > 0 + +def get_last(): + # t = pop() + # push(t) + # return t + if len(stack) > 0: + return stack[len(stack)-1] + return u'' + +def flush(): + del stack[:] + +def get_stack(): return stack \ No newline at end of file diff --git a/fff_internals/__init__.py b/fanficfare/__init__.py similarity index 100% rename from fff_internals/__init__.py rename to fanficfare/__init__.py diff --git a/fff_internals/adapters/__init__.py b/fanficfare/adapters/__init__.py similarity index 96% rename from fff_internals/adapters/__init__.py rename to fanficfare/adapters/__init__.py index b335d19..6e200fd 100644 --- a/fff_internals/adapters/__init__.py +++ b/fanficfare/adapters/__init__.py @@ -1,255 +1,255 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import os, re, sys, glob, types -from os.path import dirname, basename, normpath -import logging -import urlparse as up - -logger = logging.getLogger(__name__) - -from .. import exceptions as exceptions -from ..configurable import Configuration - -## must import each adapter here. - -import adapter_test1 -import adapter_fanfictionnet -import adapter_fanficcastletvnet -import adapter_fictionalleyorg -import adapter_fictionpresscom -import adapter_ficwadcom -import adapter_fimfictionnet -import adapter_harrypotterfanfictioncom -import adapter_mediaminerorg -import adapter_potionsandsnitches -import adapter_tenhawkpresentscom -import adapter_adastrafanficcom -import adapter_twcslibrarynet -import adapter_tthfanficorg -import adapter_twilightednet -import adapter_twiwritenet -import adapter_whoficcom -import adapter_siyecouk -import adapter_archiveofourownorg -import adapter_ficbooknet -import adapter_portkeyorg -import adapter_mugglenetcom -import adapter_hpfandomnet -import adapter_thequidditchpitchorg -import adapter_nfacommunitycom -import adapter_midnightwhispersca -import adapter_ksarchivecom -import adapter_archiveskyehawkecom -import adapter_squidgeorgpeja -import adapter_libraryofmoriacom -import adapter_wraithbaitcom -import adapter_checkmatedcom -import adapter_chaossycophanthexcom -import adapter_dramioneorg -import adapter_erosnsapphosycophanthexcom -import adapter_lumossycophanthexcom -import adapter_occlumencysycophanthexcom -import adapter_phoenixsongnet -import adapter_walkingtheplankorg -import adapter_ashwindersycophanthexcom -import adapter_thehexfilesnet -import adapter_dokugacom -import adapter_iketernalnet -import adapter_onedirectionfanfictioncom -import adapter_storiesofardacom -import adapter_samdeanarchivenu -import adapter_destinysgatewaycom -import adapter_ncisfictionnet -import adapter_stargateatlantisorg -import adapter_thealphagatecom -import adapter_fanfiktionde -import adapter_ponyfictionarchivenet -import adapter_sg1heliopoliscom -import adapter_ncisficcom -import adapter_nationallibrarynet -import adapter_themasquenet -import adapter_pretendercentrecom -import adapter_darksolaceorg -import adapter_finestoriescom -import adapter_hpfanficarchivecom -import adapter_twilightarchivescom -import adapter_wizardtalesnet -import adapter_nhamagicalworldsus -import adapter_hlfictionnet -import adapter_grangerenchantedcom -import adapter_dracoandginnycom -import adapter_scarvesandcoffeenet -import adapter_thepetulantpoetesscom -import adapter_wolverineandroguecom -import adapter_sinfuldesireorg -import adapter_merlinficdtwinscouk -import adapter_thehookupzonenet -import adapter_bloodtiesfancom -import adapter_indeathnet -import adapter_qafficcom -import adapter_efpfanficnet -import adapter_potterficscom -import adapter_efictionestelielde -import adapter_dotmoonnet -import adapter_pommedesangcom -import adapter_restrictedsectionorg -import adapter_imagineeficcom -import adapter_buffynfaithnet -import adapter_psychficcom -import adapter_hennethannunnet -import adapter_tokrafandomnetcom -import adapter_netraptororg -import adapter_asr3slashzoneorg -import adapter_nickandgregnet -import adapter_potterheadsanonymouscom -import adapter_simplyundeniablecom -import adapter_scarheadnet -import adapter_fictionpadcom -import adapter_storiesonlinenet -import adapter_trekiverseorg -import adapter_literotica -import adapter_voracity2eficcom -import adapter_spikeluvercom -import adapter_bloodshedversecom -import adapter_nocturnallightnet -import adapter_fanfichu -import adapter_fanfictioncsodaidokhu -import adapter_fictionmaniatv -import adapter_bdsmgeschichten -import adapter_tolkienfanfiction -import adapter_themaplebookshelf -import adapter_fannation -import adapter_sheppardweircom -import adapter_samandjacknet -import adapter_csiforensicscom -import adapter_lotrfanfictioncom -import adapter_fhsarchivecom -import adapter_fanfictionjunkiesde -import adapter_devianthearts - -## This bit of complexity allows adapters to be added by just adding -## importing. It eliminates the long if/else clauses we used to need -## to pick out the adapter. - -## List of registered site adapters. -__class_list = [] -__domain_map = {} - -def imports(): - for name, val in globals().items(): - if isinstance(val, types.ModuleType): - yield val.__name__ - -for x in imports(): - if "fff_internals.adapters.adapter_" in x: - #print x - cls = sys.modules[x].getClass() - __class_list.append(cls) - for site in cls.getAcceptDomains(): - __domain_map[site]=cls - -def getNormalStoryURL(url): - r = getNormalStoryURLSite(url) - if r: - return r[0] - else: - return None - -def getNormalStoryURLSite(url): - if not getNormalStoryURL.__dummyconfig: - getNormalStoryURL.__dummyconfig = Configuration("test1.com","EPUB") - # pulling up an adapter is pretty low over-head. If - # it fails, it's a bad url. - try: - adapter = getAdapter(getNormalStoryURL.__dummyconfig,url) - url = adapter.url - site = adapter.getSiteDomain() - del adapter - return (url,site) - except: - return None - -# kludgey function static/singleton -getNormalStoryURL.__dummyconfig = None - -def getAdapter(config,url,anyurl=False): - - #logger.debug("trying url:"+url) - (cls,fixedurl) = getClassFor(url) - #logger.debug("fixedurl:"+fixedurl) - if cls: - if anyurl: - fixedurl = cls.getSiteExampleURLs().split()[0] - adapter = cls(config,fixedurl) # raises InvalidStoryURL - return adapter - # No adapter found. - raise exceptions.UnknownSite( url, [cls.getSiteDomain() for cls in __class_list] ) - -def getConfigSections(): - return [cls.getConfigSection() for cls in __class_list] - -def getSiteExamples(): - l=[] - for cls in sorted(__class_list, key=lambda x : x.getConfigSection()): - l.append((cls.getConfigSection(),cls.getSiteExampleURLs().split())) - return l - -def getConfigSectionFor(url): - (cls,fixedurl) = getClassFor(url) - if cls: - return cls.getConfigSection() - - # No adapter found. - raise exceptions.UnknownSite( url, [cls.getSiteDomain() for cls in __class_list] ) - -def getClassFor(url): - ## fix up leading protocol. - fixedurl = re.sub(r"(?i)^[htp]+(s?)[:/]+",r"http\1://",url.strip()) - if fixedurl.startswith("//"): - fixedurl = "http:%s"%url - if not fixedurl.startswith("http"): - fixedurl = "http://%s"%url - ## remove any trailing '#' locations. - fixedurl = re.sub(r"#.*$","",fixedurl) - - parsedUrl = up.urlparse(fixedurl) - domain = parsedUrl.netloc.lower() - if( domain != parsedUrl.netloc ): - fixedurl = fixedurl.replace(parsedUrl.netloc,domain) - - cls = getClassFromList(domain) - if not cls and domain.startswith("www."): - domain = domain.replace("www.","") - #logger.debug("trying site:without www: "+domain) - cls = getClassFromList(domain) - fixedurl = re.sub(r"^http(s?)://www\.",r"http\1://",fixedurl) - if not cls: - #logger.debug("trying site:www."+domain) - cls = getClassFromList("www."+domain) - fixedurl = re.sub(r"^http(s?)://",r"http\1://www.",fixedurl) - - if cls: - fixedurl = cls.stripURLParameters(fixedurl) - - return (cls,fixedurl) - -def getClassFromList(domain): - try: - return __domain_map[domain] - except KeyError: - pass # return none. +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import os, re, sys, glob, types +from os.path import dirname, basename, normpath +import logging +import urlparse as up + +logger = logging.getLogger(__name__) + +from .. import exceptions as exceptions +from ..configurable import Configuration + +## must import each adapter here. + +import adapter_test1 +import adapter_fanfictionnet +import adapter_fanficcastletvnet +import adapter_fictionalleyorg +import adapter_fictionpresscom +import adapter_ficwadcom +import adapter_fimfictionnet +import adapter_harrypotterfanfictioncom +import adapter_mediaminerorg +import adapter_potionsandsnitches +import adapter_tenhawkpresentscom +import adapter_adastrafanficcom +import adapter_twcslibrarynet +import adapter_tthfanficorg +import adapter_twilightednet +import adapter_twiwritenet +import adapter_whoficcom +import adapter_siyecouk +import adapter_archiveofourownorg +import adapter_ficbooknet +import adapter_portkeyorg +import adapter_mugglenetcom +import adapter_hpfandomnet +import adapter_thequidditchpitchorg +import adapter_nfacommunitycom +import adapter_midnightwhispersca +import adapter_ksarchivecom +import adapter_archiveskyehawkecom +import adapter_squidgeorgpeja +import adapter_libraryofmoriacom +import adapter_wraithbaitcom +import adapter_checkmatedcom +import adapter_chaossycophanthexcom +import adapter_dramioneorg +import adapter_erosnsapphosycophanthexcom +import adapter_lumossycophanthexcom +import adapter_occlumencysycophanthexcom +import adapter_phoenixsongnet +import adapter_walkingtheplankorg +import adapter_ashwindersycophanthexcom +import adapter_thehexfilesnet +import adapter_dokugacom +import adapter_iketernalnet +import adapter_onedirectionfanfictioncom +import adapter_storiesofardacom +import adapter_samdeanarchivenu +import adapter_destinysgatewaycom +import adapter_ncisfictionnet +import adapter_stargateatlantisorg +import adapter_thealphagatecom +import adapter_fanfiktionde +import adapter_ponyfictionarchivenet +import adapter_sg1heliopoliscom +import adapter_ncisficcom +import adapter_nationallibrarynet +import adapter_themasquenet +import adapter_pretendercentrecom +import adapter_darksolaceorg +import adapter_finestoriescom +import adapter_hpfanficarchivecom +import adapter_twilightarchivescom +import adapter_wizardtalesnet +import adapter_nhamagicalworldsus +import adapter_hlfictionnet +import adapter_grangerenchantedcom +import adapter_dracoandginnycom +import adapter_scarvesandcoffeenet +import adapter_thepetulantpoetesscom +import adapter_wolverineandroguecom +import adapter_sinfuldesireorg +import adapter_merlinficdtwinscouk +import adapter_thehookupzonenet +import adapter_bloodtiesfancom +import adapter_indeathnet +import adapter_qafficcom +import adapter_efpfanficnet +import adapter_potterficscom +import adapter_efictionestelielde +import adapter_dotmoonnet +import adapter_pommedesangcom +import adapter_restrictedsectionorg +import adapter_imagineeficcom +import adapter_buffynfaithnet +import adapter_psychficcom +import adapter_hennethannunnet +import adapter_tokrafandomnetcom +import adapter_netraptororg +import adapter_asr3slashzoneorg +import adapter_nickandgregnet +import adapter_potterheadsanonymouscom +import adapter_simplyundeniablecom +import adapter_scarheadnet +import adapter_fictionpadcom +import adapter_storiesonlinenet +import adapter_trekiverseorg +import adapter_literotica +import adapter_voracity2eficcom +import adapter_spikeluvercom +import adapter_bloodshedversecom +import adapter_nocturnallightnet +import adapter_fanfichu +import adapter_fanfictioncsodaidokhu +import adapter_fictionmaniatv +import adapter_bdsmgeschichten +import adapter_tolkienfanfiction +import adapter_themaplebookshelf +import adapter_fannation +import adapter_sheppardweircom +import adapter_samandjacknet +import adapter_csiforensicscom +import adapter_lotrfanfictioncom +import adapter_fhsarchivecom +import adapter_fanfictionjunkiesde +import adapter_devianthearts + +## This bit of complexity allows adapters to be added by just adding +## importing. It eliminates the long if/else clauses we used to need +## to pick out the adapter. + +## List of registered site adapters. +__class_list = [] +__domain_map = {} + +def imports(): + for name, val in globals().items(): + if isinstance(val, types.ModuleType): + yield val.__name__ + +for x in imports(): + if "fanficfare.adapters.adapter_" in x: + #print x + cls = sys.modules[x].getClass() + __class_list.append(cls) + for site in cls.getAcceptDomains(): + __domain_map[site]=cls + +def getNormalStoryURL(url): + r = getNormalStoryURLSite(url) + if r: + return r[0] + else: + return None + +def getNormalStoryURLSite(url): + if not getNormalStoryURL.__dummyconfig: + getNormalStoryURL.__dummyconfig = Configuration("test1.com","EPUB") + # pulling up an adapter is pretty low over-head. If + # it fails, it's a bad url. + try: + adapter = getAdapter(getNormalStoryURL.__dummyconfig,url) + url = adapter.url + site = adapter.getSiteDomain() + del adapter + return (url,site) + except: + return None + +# kludgey function static/singleton +getNormalStoryURL.__dummyconfig = None + +def getAdapter(config,url,anyurl=False): + + #logger.debug("trying url:"+url) + (cls,fixedurl) = getClassFor(url) + #logger.debug("fixedurl:"+fixedurl) + if cls: + if anyurl: + fixedurl = cls.getSiteExampleURLs().split()[0] + adapter = cls(config,fixedurl) # raises InvalidStoryURL + return adapter + # No adapter found. + raise exceptions.UnknownSite( url, [cls.getSiteDomain() for cls in __class_list] ) + +def getConfigSections(): + return [cls.getConfigSection() for cls in __class_list] + +def getSiteExamples(): + l=[] + for cls in sorted(__class_list, key=lambda x : x.getConfigSection()): + l.append((cls.getConfigSection(),cls.getSiteExampleURLs().split())) + return l + +def getConfigSectionFor(url): + (cls,fixedurl) = getClassFor(url) + if cls: + return cls.getConfigSection() + + # No adapter found. + raise exceptions.UnknownSite( url, [cls.getSiteDomain() for cls in __class_list] ) + +def getClassFor(url): + ## fix up leading protocol. + fixedurl = re.sub(r"(?i)^[htp]+(s?)[:/]+",r"http\1://",url.strip()) + if fixedurl.startswith("//"): + fixedurl = "http:%s"%url + if not fixedurl.startswith("http"): + fixedurl = "http://%s"%url + ## remove any trailing '#' locations. + fixedurl = re.sub(r"#.*$","",fixedurl) + + parsedUrl = up.urlparse(fixedurl) + domain = parsedUrl.netloc.lower() + if( domain != parsedUrl.netloc ): + fixedurl = fixedurl.replace(parsedUrl.netloc,domain) + + cls = getClassFromList(domain) + if not cls and domain.startswith("www."): + domain = domain.replace("www.","") + #logger.debug("trying site:without www: "+domain) + cls = getClassFromList(domain) + fixedurl = re.sub(r"^http(s?)://www\.",r"http\1://",fixedurl) + if not cls: + #logger.debug("trying site:www."+domain) + cls = getClassFromList("www."+domain) + fixedurl = re.sub(r"^http(s?)://",r"http\1://www.",fixedurl) + + if cls: + fixedurl = cls.stripURLParameters(fixedurl) + + return (cls,fixedurl) + +def getClassFromList(domain): + try: + return __domain_map[domain] + except KeyError: + pass # return none. diff --git a/fff_internals/adapters/adapter_adastrafanficcom.py b/fanficfare/adapters/adapter_adastrafanficcom.py similarity index 97% rename from fff_internals/adapters/adapter_adastrafanficcom.py rename to fanficfare/adapters/adapter_adastrafanficcom.py index 1b5fb74..9262e8a 100644 --- a/fff_internals/adapters/adapter_adastrafanficcom.py +++ b/fanficfare/adapters/adapter_adastrafanficcom.py @@ -1,238 +1,238 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -class AdAstraFanficComSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','aaff') - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - - @staticmethod - def getSiteDomain(): - return 'www.adastrafanfic.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - addurl = "&warning=5" - else: - addurl="" - - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Content is only suitable for mature adults. May contain explicit language and adult themes. Equivalent of NC-17." in data: - raise exceptions.AdultCheckRequired(self.url) - - # problems with some stories, but only in calibre. I suspect - # issues with different SGML parsers in python. This is a - # nasty hack, but it works. - data = data[data.index(" in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - ## - ## Summary, strangely, is in the content attr of a tag - ## which is escaped HTML. Unfortunately, we can't use it because they don't - ## escape (') chars in the desc, breakin the tag. - #meta_desc = soup.find('meta',{'name':'description'}) - #metasoup = bs.BeautifulStoneSoup(meta_desc['content']) - #self.story.setMetadata('description',stripHTML(metasoup)) - - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = '' - while value and not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - # sometimes poorly formated desc (

w/o

) leads - # to all labels being included. - svalue=svalue[:svalue.find('')] - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(value.strip(), "%d %b %Y")) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(value.strip(), "%d %b %Y")) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - # problems with some stories, but only in calibre. I suspect - # issues with different SGML parsers in python. This is a - # nasty hack, but it works. - data = data[data.index(" in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + ## + ## Summary, strangely, is in the content attr of a tag + ## which is escaped HTML. Unfortunately, we can't use it because they don't + ## escape (') chars in the desc, breakin the tag. + #meta_desc = soup.find('meta',{'name':'description'}) + #metasoup = bs.BeautifulStoneSoup(meta_desc['content']) + #self.story.setMetadata('description',stripHTML(metasoup)) + + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = '' + while value and not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + # sometimes poorly formated desc (

w/o

) leads + # to all labels being included. + svalue=svalue[:svalue.find('')] + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(value.strip(), "%d %b %Y")) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(value.strip(), "%d %b %Y")) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + # problems with some stories, but only in calibre. I suspect + # issues with different SGML parsers in python. This is a + # nasty hack, but it works. + data = data[data.index(" in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d): - try: - return d.name - except: - return "" - - cats = info.findAll('a',href=re.compile('categories.php')) - for cat in cats: - self.story.addToList('category',cat.string) - - a = info.find('a', href=re.compile(r'reviews.php\?sid='+self.story.getMetadata('storyId'))) - val = a.nextSibling - svalue = "" - while not defaultGetattr(val) == 'br': - val = val.nextSibling - val = val.nextSibling - while not defaultGetattr(val) == 'table': - svalue += str(val) - val = val.nextSibling - self.setDescription(url,svalue) - - # Rated: NC-17
etc - labels = info.findAll('b') - for labelspan in labels: - value = labelspan.nextSibling - label = stripHTML(labelspan) - - if 'Rating' in label: - self.story.setMetadata('rating', value) - - if 'Word Count' in label: - self.story.setMetadata('numWords', value) - - if 'Genres' in label: - genres = value.string.split(', ') - for genre in genres: - if genre != 'none': - self.story.addToList('genre',genre) - - if 'Warnings' in label: - warnings = value.string.split(', ') - for warning in warnings: - if warning != ' none': - self.story.addToList('warnings',warning) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - - soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr','span','center')) # some chapters seem to be hanging up on those tags, so it is safer to close them - - story = soup.find('div', {"align" : "left"}) - - if None == story: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,story) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return AshwinderSycophantHexComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class AshwinderSycophantHexComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','asph') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'ashwinder.sycophanthex.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'This story contains adult content and/or themes.' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['rememberme'] = '1' + params['sid'] = '' + params['intent'] = '' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Logout" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + try: + # in case link points somewhere other than the first chapter + a = soup.findAll('option')[1]['value'] + self.story.setMetadata('storyId',a.split('=',)[1]) + url = 'http://'+self.host+'/'+a + soup = bs.BeautifulSoup(self._fetchUrl(url)) + except: + pass + + for info in asoup.findAll('table', {'width' : '100%', 'bordercolor' : re.compile(r'#')}): + a = info.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + if a != None: + self.story.setMetadata('title',stripHTML(a)) + break + + + # Find the chapters: + chapters=soup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+&i=1$')) + if len(chapters) == 0: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d): + try: + return d.name + except: + return "" + + cats = info.findAll('a',href=re.compile('categories.php')) + for cat in cats: + self.story.addToList('category',cat.string) + + a = info.find('a', href=re.compile(r'reviews.php\?sid='+self.story.getMetadata('storyId'))) + val = a.nextSibling + svalue = "" + while not defaultGetattr(val) == 'br': + val = val.nextSibling + val = val.nextSibling + while not defaultGetattr(val) == 'table': + svalue += str(val) + val = val.nextSibling + self.setDescription(url,svalue) + + # Rated: NC-17
etc + labels = info.findAll('b') + for labelspan in labels: + value = labelspan.nextSibling + label = stripHTML(labelspan) + + if 'Rating' in label: + self.story.setMetadata('rating', value) + + if 'Word Count' in label: + self.story.setMetadata('numWords', value) + + if 'Genres' in label: + genres = value.string.split(', ') + for genre in genres: + if genre != 'none': + self.story.addToList('genre',genre) + + if 'Warnings' in label: + warnings = value.string.split(', ') + for warning in warnings: + if warning != ' none': + self.story.addToList('warnings',warning) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + + soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr','span','center')) # some chapters seem to be hanging up on those tags, so it is safer to close them + + story = soup.find('div', {"align" : "left"}) + + if None == story: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,story) diff --git a/fff_internals/adapters/adapter_asr3slashzoneorg.py b/fanficfare/adapters/adapter_asr3slashzoneorg.py similarity index 97% rename from fff_internals/adapters/adapter_asr3slashzoneorg.py rename to fanficfare/adapters/adapter_asr3slashzoneorg.py index 74e1d6c..3e91a58 100644 --- a/fff_internals/adapters/adapter_asr3slashzoneorg.py +++ b/fanficfare/adapters/adapter_asr3slashzoneorg.py @@ -1,227 +1,227 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return Asr3SlashzoneOrgAdapter - -class Asr3SlashzoneOrgAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/archive/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','asr3') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'asr3.slashzone.org' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/archive/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=3" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - #print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/archive/'+a['href']) - self.story.setMetadata('author',a.string) - - # Rating - rate = stripHTML(soup.find('div',{'id':'pagetitle'})) - rate = rate[rate.rindex('[')+1:rate.rindex(']')] - self.story.setMetadata('rating', rate) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/archive/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - metadiv = soup.find('div',{'class':'content'}) - smalldiv = metadiv.find('div',{'class':'small'}) - - categorys = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for category in categorys: - self.story.addToList('category',category.string) - - chars = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - ships = smalldiv.parent.findAll('a',href=re.compile(r'browse\.php\?type=class&type_id=2&classid=1')) - for ship in ships: - self.story.addToList('ships',ship.string) - - metatext = stripHTML(smalldiv) - - if 'Completed: Yes' in metatext: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - wordstart=metatext.rindex('Word count:')+12 - words = metatext[wordstart:metatext.index(' ',wordstart)] - self.story.setMetadata('numWords', words) - - datesdiv = soup.find('div',{'class':'bottom'}) - dates = stripHTML(datesdiv).split() - # Published: 04/26/2011 Updated: 03/06/2013 - self.story.setMetadata('datePublished', makeDate(dates[1], self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(dates[3], self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/archive/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # remove 'small' leaving only summary. - smalldiv.extract() - self.setDescription(url,metadiv) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return Asr3SlashzoneOrgAdapter + +class Asr3SlashzoneOrgAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/archive/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','asr3') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'asr3.slashzone.org' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/archive/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/archive/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=3" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + #print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/archive/'+a['href']) + self.story.setMetadata('author',a.string) + + # Rating + rate = stripHTML(soup.find('div',{'id':'pagetitle'})) + rate = rate[rate.rindex('[')+1:rate.rindex(']')] + self.story.setMetadata('rating', rate) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/archive/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + metadiv = soup.find('div',{'class':'content'}) + smalldiv = metadiv.find('div',{'class':'small'}) + + categorys = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for category in categorys: + self.story.addToList('category',category.string) + + chars = smalldiv.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + ships = smalldiv.parent.findAll('a',href=re.compile(r'browse\.php\?type=class&type_id=2&classid=1')) + for ship in ships: + self.story.addToList('ships',ship.string) + + metatext = stripHTML(smalldiv) + + if 'Completed: Yes' in metatext: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + wordstart=metatext.rindex('Word count:')+12 + words = metatext[wordstart:metatext.index(' ',wordstart)] + self.story.setMetadata('numWords', words) + + datesdiv = soup.find('div',{'class':'bottom'}) + dates = stripHTML(datesdiv).split() + # Published: 04/26/2011 Updated: 03/06/2013 + self.story.setMetadata('datePublished', makeDate(dates[1], self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(dates[3], self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/archive/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # remove 'small' leaving only summary. + smalldiv.extract() + self.setDescription(url,metadiv) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_bdsmgeschichten.py b/fanficfare/adapters/adapter_bdsmgeschichten.py similarity index 100% rename from fff_internals/adapters/adapter_bdsmgeschichten.py rename to fanficfare/adapters/adapter_bdsmgeschichten.py diff --git a/fff_internals/adapters/adapter_bloodshedversecom.py b/fanficfare/adapters/adapter_bloodshedversecom.py similarity index 97% rename from fff_internals/adapters/adapter_bloodshedversecom.py rename to fanficfare/adapters/adapter_bloodshedversecom.py index a47db27..03fd517 100644 --- a/fff_internals/adapters/adapter_bloodshedversecom.py +++ b/fanficfare/adapters/adapter_bloodshedversecom.py @@ -1,193 +1,193 @@ -from datetime import timedelta -import re -import urllib2 -import urlparse - -from .. import BeautifulSoup -from ..htmlcleanup import stripHTML - -from base_adapter import BaseSiteAdapter, makeDate -from .. import exceptions - - -def getClass(): - return BloodshedverseComAdapter - - -def _get_query_data(url): - components = urlparse.urlparse(url) - query_data = urlparse.parse_qs(components.query) - return dict((key, data[0]) for key, data in query_data.items()) - - -class BloodshedverseComAdapter(BaseSiteAdapter): - SITE_ABBREVIATION = 'bvc' - SITE_DOMAIN = 'bloodshedverse.com' - - BASE_URL = 'http://' + SITE_DOMAIN + '/' - READ_URL_TEMPLATE = BASE_URL + 'stories.php?go=read&no=%s' - - STARTED_DATETIME_FORMAT = '%m/%d/%Y' - UPDATED_DATETIME_FORMAT = '%m/%d/%Y %I:%M' - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - query_data = urlparse.parse_qs(self.parsedUrl.query) - story_no = query_data['no'][0] - - self.story.setMetadata('storyId', story_no) - self._setURL(self.READ_URL_TEMPLATE % story_no) - self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) - - def _customized_fetch_url(self, url, exception=None, parameters=None): - if exception: - try: - data = self._fetchUrl(url, parameters) - except urllib2.HTTPError: - raise exception(self.url) - # Just let self._fetchUrl throw the exception, don't catch and - # customize it. - else: - data = self._fetchUrl(url, parameters) - - return BeautifulSoup.BeautifulSoup(data) - - @staticmethod - def getSiteDomain(): - return BloodshedverseComAdapter.SITE_DOMAIN - - @classmethod - def getSiteExampleURLs(cls): - return cls.READ_URL_TEMPLATE % 1234 - - def getSiteURLPattern(self): - return re.escape(self.BASE_URL + 'stories.php?go=') + r'(read|chapters)\&no=\d+$' - - # Override stripURLParameters so the "no" parameter won't get stripped - @classmethod - def stripURLParameters(cls, url): - return url - - def extractChapterUrlsAndMetadata(self): - soup = self._customized_fetch_url(self.url) - - # Since no 404 error code we have to raise the exception ourselves. - # A title that is just 'by' indicates that there is no author name - # and no story title available. - if stripHTML(soup.title) == 'by': - raise exceptions.StoryDoesNotExist(self.url) - - for option in soup.find('select', {'name': 'chapter'}): - title = stripHTML(option) - url = self.READ_URL_TEMPLATE % option['value'] - self.chapterUrls.append((title, url)) - - # Get the URL to the author's page and find the correct story entry to - # scrape the metadata - author_url = urlparse.urljoin(self.url, soup.find('a', {'class': 'headline'})['href']) - soup = self._customized_fetch_url(author_url) - - story_no = self.story.getMetadata('storyId') - # Ignore first list_box div, it only contains the author information - for list_box in soup('div', {'class': 'list_box'})[1:]: - url = list_box.find('a', {'class': 'fictitle'})['href'] - query_data = _get_query_data(url) - - # Found the div containing the story's metadata; break the loop and - # parse the element - if query_data['no'] == story_no: - break - else: - raise exceptions.FailedToDownload(self.url) - - title_anchor = list_box.find('a', {'class': 'fictitle'}) - self.story.setMetadata('title', stripHTML(title_anchor)) - - author_anchor = title_anchor.findNextSibling('a') - self.story.setMetadata('author', stripHTML(author_anchor)) - self.story.setMetadata('authorId', _get_query_data(author_anchor['href'])['who']) - self.story.setMetadata('authorUrl', urlparse.urljoin(self.url, author_anchor['href'])) - - list_review = list_box.find('div', {'class': 'list_review'}) - reviews = stripHTML(list_review.a).split(' ', 1)[0] - self.story.setMetadata('reviews', reviews) - - summary_div = list_box.find('div', {'class': 'list_summary'}) - if not self.getConfig('keep_summary_html'): - summary = ''.join(summary_div(text=True)) - else: - summary = self.utf8FromSoup(author_url, summary_div) - - self.story.setMetadata('description', summary) - - # I'm assuming this to be the category, not sure what else it could be - first_listinfo = list_box.find('div', {'class': 'list_info'}) - self.story.addToList('category', stripHTML(first_listinfo.a)) - - for list_info in first_listinfo.findNextSiblings('div', {'class': 'list_info'}): - for b_tag in list_info('b'): - key = b_tag.string.strip(': ') - # Strip colons from the beginning, superfluous spaces and minus - # characters from the end, and possibly trailing commas from - # the warnings if only one is present - value = b_tag.nextSibling.string.strip(': -,') - - if key == 'Genre': - for genre in value.split(', '): - # Ignore the "none" genre - if not genre == 'none': - self.story.addToList('genre', genre) - - elif key == 'Rating': - self.story.setMetadata('rating', value) - - elif key == 'Complete': - self.story.setMetadata('status', 'Completed' if value == 'Yes' else 'In-Progress') - - elif key == 'Warning': - for warning in value.split(', '): - # The string here starts with ", " before the actual list - # of values sometimes, so check for an empty warning - # and ignore the "none" warning. - if not warning or warning == 'none': - continue - - self.story.addToList('warnings', warning) - - elif key == 'Chapters': - self.story.setMetadata('numChapters', int(value)) - - elif key == 'Words': - # Apparently only numChapters need to be an integer for - # some strange reason. Remove possible ',' characters as to - # not confuse the codebase down the line - self.story.setMetadata('numWords', value.replace(',', '')) - - elif key == 'Started': - self.story.setMetadata('datePublished', makeDate(value, self.STARTED_DATETIME_FORMAT)) - - elif key == 'Updated': - date_string, period = value.rsplit(' ', 1) - date = makeDate(date_string, self.UPDATED_DATETIME_FORMAT) - - # Rather ugly hack to work around Calibre's changing of - # Python's locale setting, causing am/pm to not be properly - # parsed by strptime() when using a non-english locale - if period == 'pm': - date += timedelta(hours=12) - self.story.setMetadata('dateUpdated', date) - - if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')): - raise exceptions.AdultCheckRequired(self.url) - - def getChapterText(self, url): - soup = self._customized_fetch_url(url) - storytext_div = soup.find('div', {'class': 'storytext'}) - - if self.getConfig('strip_text_links'): - for anchor in storytext_div('a', {'class': 'FAtxtL'}): - navigable_string = BeautifulSoup.NavigableString(anchor.string) - anchor.replaceWith(navigable_string) - - return self.utf8FromSoup(url, storytext_div) +from datetime import timedelta +import re +import urllib2 +import urlparse + +from .. import BeautifulSoup +from ..htmlcleanup import stripHTML + +from base_adapter import BaseSiteAdapter, makeDate +from .. import exceptions + + +def getClass(): + return BloodshedverseComAdapter + + +def _get_query_data(url): + components = urlparse.urlparse(url) + query_data = urlparse.parse_qs(components.query) + return dict((key, data[0]) for key, data in query_data.items()) + + +class BloodshedverseComAdapter(BaseSiteAdapter): + SITE_ABBREVIATION = 'bvc' + SITE_DOMAIN = 'bloodshedverse.com' + + BASE_URL = 'http://' + SITE_DOMAIN + '/' + READ_URL_TEMPLATE = BASE_URL + 'stories.php?go=read&no=%s' + + STARTED_DATETIME_FORMAT = '%m/%d/%Y' + UPDATED_DATETIME_FORMAT = '%m/%d/%Y %I:%M' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + query_data = urlparse.parse_qs(self.parsedUrl.query) + story_no = query_data['no'][0] + + self.story.setMetadata('storyId', story_no) + self._setURL(self.READ_URL_TEMPLATE % story_no) + self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return BeautifulSoup.BeautifulSoup(data) + + @staticmethod + def getSiteDomain(): + return BloodshedverseComAdapter.SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls.READ_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return re.escape(self.BASE_URL + 'stories.php?go=') + r'(read|chapters)\&no=\d+$' + + # Override stripURLParameters so the "no" parameter won't get stripped + @classmethod + def stripURLParameters(cls, url): + return url + + def extractChapterUrlsAndMetadata(self): + soup = self._customized_fetch_url(self.url) + + # Since no 404 error code we have to raise the exception ourselves. + # A title that is just 'by' indicates that there is no author name + # and no story title available. + if stripHTML(soup.title) == 'by': + raise exceptions.StoryDoesNotExist(self.url) + + for option in soup.find('select', {'name': 'chapter'}): + title = stripHTML(option) + url = self.READ_URL_TEMPLATE % option['value'] + self.chapterUrls.append((title, url)) + + # Get the URL to the author's page and find the correct story entry to + # scrape the metadata + author_url = urlparse.urljoin(self.url, soup.find('a', {'class': 'headline'})['href']) + soup = self._customized_fetch_url(author_url) + + story_no = self.story.getMetadata('storyId') + # Ignore first list_box div, it only contains the author information + for list_box in soup('div', {'class': 'list_box'})[1:]: + url = list_box.find('a', {'class': 'fictitle'})['href'] + query_data = _get_query_data(url) + + # Found the div containing the story's metadata; break the loop and + # parse the element + if query_data['no'] == story_no: + break + else: + raise exceptions.FailedToDownload(self.url) + + title_anchor = list_box.find('a', {'class': 'fictitle'}) + self.story.setMetadata('title', stripHTML(title_anchor)) + + author_anchor = title_anchor.findNextSibling('a') + self.story.setMetadata('author', stripHTML(author_anchor)) + self.story.setMetadata('authorId', _get_query_data(author_anchor['href'])['who']) + self.story.setMetadata('authorUrl', urlparse.urljoin(self.url, author_anchor['href'])) + + list_review = list_box.find('div', {'class': 'list_review'}) + reviews = stripHTML(list_review.a).split(' ', 1)[0] + self.story.setMetadata('reviews', reviews) + + summary_div = list_box.find('div', {'class': 'list_summary'}) + if not self.getConfig('keep_summary_html'): + summary = ''.join(summary_div(text=True)) + else: + summary = self.utf8FromSoup(author_url, summary_div) + + self.story.setMetadata('description', summary) + + # I'm assuming this to be the category, not sure what else it could be + first_listinfo = list_box.find('div', {'class': 'list_info'}) + self.story.addToList('category', stripHTML(first_listinfo.a)) + + for list_info in first_listinfo.findNextSiblings('div', {'class': 'list_info'}): + for b_tag in list_info('b'): + key = b_tag.string.strip(': ') + # Strip colons from the beginning, superfluous spaces and minus + # characters from the end, and possibly trailing commas from + # the warnings if only one is present + value = b_tag.nextSibling.string.strip(': -,') + + if key == 'Genre': + for genre in value.split(', '): + # Ignore the "none" genre + if not genre == 'none': + self.story.addToList('genre', genre) + + elif key == 'Rating': + self.story.setMetadata('rating', value) + + elif key == 'Complete': + self.story.setMetadata('status', 'Completed' if value == 'Yes' else 'In-Progress') + + elif key == 'Warning': + for warning in value.split(', '): + # The string here starts with ", " before the actual list + # of values sometimes, so check for an empty warning + # and ignore the "none" warning. + if not warning or warning == 'none': + continue + + self.story.addToList('warnings', warning) + + elif key == 'Chapters': + self.story.setMetadata('numChapters', int(value)) + + elif key == 'Words': + # Apparently only numChapters need to be an integer for + # some strange reason. Remove possible ',' characters as to + # not confuse the codebase down the line + self.story.setMetadata('numWords', value.replace(',', '')) + + elif key == 'Started': + self.story.setMetadata('datePublished', makeDate(value, self.STARTED_DATETIME_FORMAT)) + + elif key == 'Updated': + date_string, period = value.rsplit(' ', 1) + date = makeDate(date_string, self.UPDATED_DATETIME_FORMAT) + + # Rather ugly hack to work around Calibre's changing of + # Python's locale setting, causing am/pm to not be properly + # parsed by strptime() when using a non-english locale + if period == 'pm': + date += timedelta(hours=12) + self.story.setMetadata('dateUpdated', date) + + if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')): + raise exceptions.AdultCheckRequired(self.url) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + storytext_div = soup.find('div', {'class': 'storytext'}) + + if self.getConfig('strip_text_links'): + for anchor in storytext_div('a', {'class': 'FAtxtL'}): + navigable_string = BeautifulSoup.NavigableString(anchor.string) + anchor.replaceWith(navigable_string) + + return self.utf8FromSoup(url, storytext_div) diff --git a/fff_internals/adapters/adapter_bloodtiesfancom.py b/fanficfare/adapters/adapter_bloodtiesfancom.py similarity index 97% rename from fff_internals/adapters/adapter_bloodtiesfancom.py rename to fanficfare/adapters/adapter_bloodtiesfancom.py index 6da5260..e399229 100644 --- a/fff_internals/adapters/adapter_bloodtiesfancom.py +++ b/fanficfare/adapters/adapter_bloodtiesfancom.py @@ -1,337 +1,337 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# By virtue of being recent and requiring both is_adult and user/pass, -# adapter_fanficcastletvnet.py is the best choice for learning to -# write adapters--especially for sites that use the eFiction system. -# Most sites that have ".../viewstory.php?sid=123" in the story URL -# are eFiction. - -# For non-eFiction sites, it can be considerably more complex, but -# this is still a good starting point. - -# In general an 'adapter' needs to do these five things: - -# - 'Register' correctly with the downloader -# - Site Login (if needed) -# - 'Are you adult?' check (if needed--some do one, some the other, some both) -# - Grab the chapter list -# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) -# - Grab the chapter texts - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return BloodTiesFansComAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class BloodTiesFansComAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/fiction/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','btf') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d, %Y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'bloodties-fans.com' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fiction/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/fiction/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/fiction/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - - # Furthermore, there's a couple sites now with more than - # one warning level for different ratings. And they're - # fussy about it. midnightwhispers has three: 4, 2 & 1. - # we'll try 1 first. - addurl = "&ageconsent=ok&warning=4" # XXX - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. nfacommunity uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=561&warning=4 - # viewstory.php?sid=561&warning=1 - # viewstory.php?sid=561&warning=2 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/fiction/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fiction/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/fiction/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# By virtue of being recent and requiring both is_adult and user/pass, +# adapter_fanficcastletvnet.py is the best choice for learning to +# write adapters--especially for sites that use the eFiction system. +# Most sites that have ".../viewstory.php?sid=123" in the story URL +# are eFiction. + +# For non-eFiction sites, it can be considerably more complex, but +# this is still a good starting point. + +# In general an 'adapter' needs to do these five things: + +# - 'Register' correctly with the downloader +# - Site Login (if needed) +# - 'Are you adult?' check (if needed--some do one, some the other, some both) +# - Grab the chapter list +# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) +# - Grab the chapter texts + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return BloodTiesFansComAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class BloodTiesFansComAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/fiction/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','btf') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d, %Y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'bloodties-fans.com' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fiction/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/fiction/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/fiction/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + + # Furthermore, there's a couple sites now with more than + # one warning level for different ratings. And they're + # fussy about it. midnightwhispers has three: 4, 2 & 1. + # we'll try 1 first. + addurl = "&ageconsent=ok&warning=4" # XXX + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. nfacommunity uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=561&warning=4 + # viewstory.php?sid=561&warning=1 + # viewstory.php?sid=561&warning=2 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/fiction/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fiction/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/fiction/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_buffynfaithnet.py b/fanficfare/adapters/adapter_buffynfaithnet.py similarity index 97% rename from fff_internals/adapters/adapter_buffynfaithnet.py rename to fanficfare/adapters/adapter_buffynfaithnet.py index 9a91bee..6823602 100644 --- a/fff_internals/adapters/adapter_buffynfaithnet.py +++ b/fanficfare/adapters/adapter_buffynfaithnet.py @@ -1,296 +1,296 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import cookielib as cl - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return BuffyNFaithNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class BuffyNFaithNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.setHeader() - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query correct - m = re.match(self.getSiteURLPattern(),url) - if m: - self.story.setMetadata('storyId',m.group('id')) - - # normalized story URL. gets rid of chapter if there, left with ch 1 URL on this site - nurl = "http://"+self.getSiteDomain()+"/fanfictions/index.php?act=vie&id="+self.story.getMetadata('storyId') - self._setURL(nurl) - #argh, this mangles the ampersands I need on metadata['storyUrl'] - #will set it this way - self.story.setMetadata('storyUrl',nurl,condremoveentities=False) - else: - raise exceptions.InvalidStoryURL(url, - self.getSiteDomain(), - self.getSiteExampleURLs()) - - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','bnfnet') - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'buffynfaith.net' - - @classmethod - def stripURLParameters(cls,url): - "Only needs to be overriden if URL contains more than one parameter" - ## This adapter needs at least two parameters left on the URL, act and id - return re.sub(r"(\?act=(vie|ovr)&id=\d+)&.*$",r"\1",url) - - def setHeader(self): - "buffynfaith.net wants a Referer for images. Used both above and below(after cookieproc added)" - self.opener.addheaders.append(('Referer', 'http://'+self.getSiteDomain()+'/')) - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=ovr&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234&ch=2" - - def getSiteURLPattern(self): - #http://buffynfaith.net/fanfictions/index.php?act=vie&id=963 - #http://buffynfaith.net/fanfictions/index.php?act=vie&id=949 - #http://buffynfaith.net/fanfictions/index.php?act=vie&id=949&ch=2 - p = re.escape("http://"+self.getSiteDomain()+"/fanfictions/index.php?act=")+\ - r"(vie|ovr)&id=(?P\d+)(&ch=(?P\d+))?$" - return p - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - def extractChapterUrlsAndMetadata(self): - - dateformat = "%d %B %Y" - url = self.url - logger.debug("URL: "+url) - - #set a cookie to get past adult check - if self.is_adult or self.getConfig("is_adult"): - cookie = cl.Cookie(version=0, name='my_age', value='yes', - port=None, port_specified=False, - domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, - path='/', path_specified=True, - secure=False, - expires=time.time()+10000, - discard=False, - comment=None, - comment_url=None, - rest={'HttpOnly': None}, - rfc2109=False) - self.cookiejar.set_cookie(cookie) - self.setHeader() - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - #print data - - if "ADULT CONTENT WARNING" in data: - raise exceptions.AdultCheckRequired(self.url) - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - # Now go hunting for all the meta data and the chapter list. - - #stuff in : description - svalue = soup.head.find('meta',attrs={'name':'description'})['content'] - #self.story.setMetadata('description',svalue) - self.setDescription(url,svalue) - - #useful stuff in rest of doc, all contained in this: - doc = soup.body.find('div', id='my_wrapper') - - #first the site category (more of a genre to me, meh) and title, in this element: - mt = doc.find('div',attrs={'class':'maintitle'}) - self.story.addToList('genre',mt.findAll('a')[1].string) - self.story.setMetadata('title',mt.findAll('a')[1].nextSibling[len(' » '):]) - del mt - - #the actual category, for me, is 'Buffy: The Vampire Slayer' - #self.story.addToList('category','Buffy: The Vampire Slayer') - #No need to do it here, it is better to set it in in plugin-defaults.ini and defaults.ini - - #then a block that sits in a table cell like so: - #(contains a lot of metadata) - mblock = doc.find('td', align='left', width = '70%').contents - while len(mblock) > 0: - i = mblock.pop(0) - if 'Author:' in i.string: - #drop empty space - mblock.pop(0) - #get author link - a = mblock.pop(0) - authre = re.escape('./index.php?act=bio&id=')+'(?P\d+)' - m = re.match(authre,a['href']) - self.story.setMetadata('author',a.string) - self.story.setMetadata('authorId',m.group('authid')) - authurl = u'http://%s/fanfictions/index.php?act=bio&id=%s' % ( self.getSiteDomain(), - self.story.getMetadata('authorId')) - self.story.setMetadata('authorUrl',authurl,condremoveentities=False) - #drop empty space - mblock.pop(0) - if 'Rating:' in i.string: - self.story.setMetadata('rating',mblock.pop(0).strip()) - if 'Published:' in i.string: - date = mblock.pop(0).strip() - #get rid of 'st', 'nd', 'rd', 'th' after day number - date = date[0:2]+date[4:] - self.story.setMetadata('datePublished',makeDate(date, dateformat)) - if 'Last Updated:' in i.string: - date = mblock.pop(0).strip() - #get rid of 'st', 'nd', 'rd', 'th' after day number - date = date[0:2]+date[4:] - self.story.setMetadata('dateUpdated',makeDate(date, dateformat)) - if 'Genre:' in i.string: - genres = mblock.pop(0).strip() - genres = genres.split('/') - for genre in genres: self.story.addToList('genre',genre) - #end ifs - #end while - - # Find the chapter selector - select = soup.find('select', { 'name' : 'ch' } ) - - if select is None: - # no selector found, so it's a one-chapter story. - #self.chapterUrls.append((self.story.getMetadata('title'),url)) - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - allOptions = select.findAll('option') - for o in allOptions: - url = u'http://%s/fanfictions/index.php?act=vie&id=%s&ch=%s' % ( self.getSiteDomain(), - self.story.getMetadata('storyId'), - o['value']) - title = u"%s" % o - title = stripHTML(title) - ts = title.split(' ',1) - title = ts[0]+'. '+ts[1] - self.chapterUrls.append((title,url)) - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - ## Go scrape the rest of the metadata from the author's page. - data = self._fetchUrl(self.story.getMetadata('authorUrl')) - soup = bs.BeautifulSoup(data) - #find the story link and its parent div - storya = soup.find('a',{'href':self.story.getMetadata('storyUrl')}) - storydiv = storya.parent - #warnings come under a tag. Never seen that before... - #appears to just be a line of freeform text, not necessarily a list - #optional - spawn = storydiv.find('spawn',{'id':'warnings'}) - if spawn is not None: - warns = spawn.nextSibling.strip() - self.story.addToList('warnings',warns) - #some meta in spans - this should get all, even the ones jammed in a table - spans = storydiv.findAll('span') - for s in spans: - if s.string == 'Ship:': - list = s.nextSibling.strip().split() - self.story.extendList('ships',list) - if s.string == 'Characters:': - list = s.nextSibling.strip().split(',') - self.story.extendList('characters',list) - if s.string == 'Status:': - st = s.nextSibling.strip() - self.story.setMetadata('status',st) - if s.string == 'Words:': - st = s.nextSibling.strip() - self.story.setMetadata('numWords',st) - - #reviews - is this worth having? - #ffnet adapter gathers it, don't know if anything else does - #or if it's ever going to be used! - a = storydiv.find('a',{'id':'bold-blue'}) - if a: - revs = a.nextSibling.strip()[1:-1] - self.story.setMetadata('reviews',st) - else: - revs = '0' - self.story.setMetadata('reviews',st) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'fanfiction'}) - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - #remove all the unnecessary bookmark tags - [s.extract() for s in div('div',{'class':"tiny_box2"})] - - #is there a review link? - r = div.find('a',href=re.compile(re.escape("./index.php?act=irv")+".*$")) - if r is not None: - #remove the review link and its parent div - r.parent.extract() - - #There might also be a link to the sequel on the last chapter - #I'm inclined to keep it in, but the URL needs to be changed from relative to absolute - #Shame there isn't proper series metadata available - #(I couldn't find it anyway) - s = div.find('a',href=re.compile(re.escape("./index.php?act=ovr")+".*$")) - if s is not None: - s['href'] = 'http://'+self.getSiteDomain()+'/fanfictions'+s['href'][1:] - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import cookielib as cl + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return BuffyNFaithNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class BuffyNFaithNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.setHeader() + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query correct + m = re.match(self.getSiteURLPattern(),url) + if m: + self.story.setMetadata('storyId',m.group('id')) + + # normalized story URL. gets rid of chapter if there, left with ch 1 URL on this site + nurl = "http://"+self.getSiteDomain()+"/fanfictions/index.php?act=vie&id="+self.story.getMetadata('storyId') + self._setURL(nurl) + #argh, this mangles the ampersands I need on metadata['storyUrl'] + #will set it this way + self.story.setMetadata('storyUrl',nurl,condremoveentities=False) + else: + raise exceptions.InvalidStoryURL(url, + self.getSiteDomain(), + self.getSiteExampleURLs()) + + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','bnfnet') + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'buffynfaith.net' + + @classmethod + def stripURLParameters(cls,url): + "Only needs to be overriden if URL contains more than one parameter" + ## This adapter needs at least two parameters left on the URL, act and id + return re.sub(r"(\?act=(vie|ovr)&id=\d+)&.*$",r"\1",url) + + def setHeader(self): + "buffynfaith.net wants a Referer for images. Used both above and below(after cookieproc added)" + self.opener.addheaders.append(('Referer', 'http://'+self.getSiteDomain()+'/')) + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=ovr&id=1234 http://"+cls.getSiteDomain()+"/fanfictions/index.php?act=vie&id=1234&ch=2" + + def getSiteURLPattern(self): + #http://buffynfaith.net/fanfictions/index.php?act=vie&id=963 + #http://buffynfaith.net/fanfictions/index.php?act=vie&id=949 + #http://buffynfaith.net/fanfictions/index.php?act=vie&id=949&ch=2 + p = re.escape("http://"+self.getSiteDomain()+"/fanfictions/index.php?act=")+\ + r"(vie|ovr)&id=(?P\d+)(&ch=(?P\d+))?$" + return p + + def use_pagecache(self): + ''' + adapters that will work with the page cache need to implement + this and change it to True. + ''' + return True + + def extractChapterUrlsAndMetadata(self): + + dateformat = "%d %B %Y" + url = self.url + logger.debug("URL: "+url) + + #set a cookie to get past adult check + if self.is_adult or self.getConfig("is_adult"): + cookie = cl.Cookie(version=0, name='my_age', value='yes', + port=None, port_specified=False, + domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, + path='/', path_specified=True, + secure=False, + expires=time.time()+10000, + discard=False, + comment=None, + comment_url=None, + rest={'HttpOnly': None}, + rfc2109=False) + self.cookiejar.set_cookie(cookie) + self.setHeader() + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + #print data + + if "ADULT CONTENT WARNING" in data: + raise exceptions.AdultCheckRequired(self.url) + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + # Now go hunting for all the meta data and the chapter list. + + #stuff in : description + svalue = soup.head.find('meta',attrs={'name':'description'})['content'] + #self.story.setMetadata('description',svalue) + self.setDescription(url,svalue) + + #useful stuff in rest of doc, all contained in this: + doc = soup.body.find('div', id='my_wrapper') + + #first the site category (more of a genre to me, meh) and title, in this element: + mt = doc.find('div',attrs={'class':'maintitle'}) + self.story.addToList('genre',mt.findAll('a')[1].string) + self.story.setMetadata('title',mt.findAll('a')[1].nextSibling[len(' » '):]) + del mt + + #the actual category, for me, is 'Buffy: The Vampire Slayer' + #self.story.addToList('category','Buffy: The Vampire Slayer') + #No need to do it here, it is better to set it in in plugin-defaults.ini and defaults.ini + + #then a block that sits in a table cell like so: + #(contains a lot of metadata) + mblock = doc.find('td', align='left', width = '70%').contents + while len(mblock) > 0: + i = mblock.pop(0) + if 'Author:' in i.string: + #drop empty space + mblock.pop(0) + #get author link + a = mblock.pop(0) + authre = re.escape('./index.php?act=bio&id=')+'(?P\d+)' + m = re.match(authre,a['href']) + self.story.setMetadata('author',a.string) + self.story.setMetadata('authorId',m.group('authid')) + authurl = u'http://%s/fanfictions/index.php?act=bio&id=%s' % ( self.getSiteDomain(), + self.story.getMetadata('authorId')) + self.story.setMetadata('authorUrl',authurl,condremoveentities=False) + #drop empty space + mblock.pop(0) + if 'Rating:' in i.string: + self.story.setMetadata('rating',mblock.pop(0).strip()) + if 'Published:' in i.string: + date = mblock.pop(0).strip() + #get rid of 'st', 'nd', 'rd', 'th' after day number + date = date[0:2]+date[4:] + self.story.setMetadata('datePublished',makeDate(date, dateformat)) + if 'Last Updated:' in i.string: + date = mblock.pop(0).strip() + #get rid of 'st', 'nd', 'rd', 'th' after day number + date = date[0:2]+date[4:] + self.story.setMetadata('dateUpdated',makeDate(date, dateformat)) + if 'Genre:' in i.string: + genres = mblock.pop(0).strip() + genres = genres.split('/') + for genre in genres: self.story.addToList('genre',genre) + #end ifs + #end while + + # Find the chapter selector + select = soup.find('select', { 'name' : 'ch' } ) + + if select is None: + # no selector found, so it's a one-chapter story. + #self.chapterUrls.append((self.story.getMetadata('title'),url)) + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + allOptions = select.findAll('option') + for o in allOptions: + url = u'http://%s/fanfictions/index.php?act=vie&id=%s&ch=%s' % ( self.getSiteDomain(), + self.story.getMetadata('storyId'), + o['value']) + title = u"%s" % o + title = stripHTML(title) + ts = title.split(' ',1) + title = ts[0]+'. '+ts[1] + self.chapterUrls.append((title,url)) + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + ## Go scrape the rest of the metadata from the author's page. + data = self._fetchUrl(self.story.getMetadata('authorUrl')) + soup = bs.BeautifulSoup(data) + #find the story link and its parent div + storya = soup.find('a',{'href':self.story.getMetadata('storyUrl')}) + storydiv = storya.parent + #warnings come under a tag. Never seen that before... + #appears to just be a line of freeform text, not necessarily a list + #optional + spawn = storydiv.find('spawn',{'id':'warnings'}) + if spawn is not None: + warns = spawn.nextSibling.strip() + self.story.addToList('warnings',warns) + #some meta in spans - this should get all, even the ones jammed in a table + spans = storydiv.findAll('span') + for s in spans: + if s.string == 'Ship:': + list = s.nextSibling.strip().split() + self.story.extendList('ships',list) + if s.string == 'Characters:': + list = s.nextSibling.strip().split(',') + self.story.extendList('characters',list) + if s.string == 'Status:': + st = s.nextSibling.strip() + self.story.setMetadata('status',st) + if s.string == 'Words:': + st = s.nextSibling.strip() + self.story.setMetadata('numWords',st) + + #reviews - is this worth having? + #ffnet adapter gathers it, don't know if anything else does + #or if it's ever going to be used! + a = storydiv.find('a',{'id':'bold-blue'}) + if a: + revs = a.nextSibling.strip()[1:-1] + self.story.setMetadata('reviews',st) + else: + revs = '0' + self.story.setMetadata('reviews',st) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'fanfiction'}) + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + #remove all the unnecessary bookmark tags + [s.extract() for s in div('div',{'class':"tiny_box2"})] + + #is there a review link? + r = div.find('a',href=re.compile(re.escape("./index.php?act=irv")+".*$")) + if r is not None: + #remove the review link and its parent div + r.parent.extract() + + #There might also be a link to the sequel on the last chapter + #I'm inclined to keep it in, but the URL needs to be changed from relative to absolute + #Shame there isn't proper series metadata available + #(I couldn't find it anyway) + s = div.find('a',href=re.compile(re.escape("./index.php?act=ovr")+".*$")) + if s is not None: + s['href'] = 'http://'+self.getSiteDomain()+'/fanfictions'+s['href'][1:] + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_chaossycophanthexcom.py b/fanficfare/adapters/adapter_chaossycophanthexcom.py similarity index 97% rename from fff_internals/adapters/adapter_chaossycophanthexcom.py rename to fanficfare/adapters/adapter_chaossycophanthexcom.py index b5b945f..806b9b0 100644 --- a/fff_internals/adapters/adapter_chaossycophanthexcom.py +++ b/fanficfare/adapters/adapter_chaossycophanthexcom.py @@ -1,238 +1,238 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return ChaosSycophantHexComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class ChaosSycophantHexComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','csph') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'chaos.sycophanthex.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=19" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "Age Consent Required" in data: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - pt = soup.find('div', {'id' : 'pagetitle'}) - a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - rating=pt.text.split('(')[1].split(')')[0] - self.story.setMetadata('rating', rating) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - - labels = soup.findAll('span',{'class':'label'}) - - value = labels[0].previousSibling - svalue = "" - while value != None: - val = value - value = value.previousSibling - while not defaultGetattr(val,'class') == 'label': - svalue += str(val) - val = val.nextSibling - self.setDescription(url,svalue) - - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Word count' in label: - self.story.setMetadata('numWords', value.split(' -')[0]) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Complete' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return ChaosSycophantHexComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class ChaosSycophantHexComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','csph') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'chaos.sycophanthex.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=19" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "Age Consent Required" in data: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + pt = soup.find('div', {'id' : 'pagetitle'}) + a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + rating=pt.text.split('(')[1].split(')')[0] + self.story.setMetadata('rating', rating) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + + labels = soup.findAll('span',{'class':'label'}) + + value = labels[0].previousSibling + svalue = "" + while value != None: + val = value + value = value.previousSibling + while not defaultGetattr(val,'class') == 'label': + svalue += str(val) + val = val.nextSibling + self.setDescription(url,svalue) + + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Word count' in label: + self.story.setMetadata('numWords', value.split(' -')[0]) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Complete' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_checkmatedcom.py b/fanficfare/adapters/adapter_checkmatedcom.py similarity index 97% rename from fff_internals/adapters/adapter_checkmatedcom.py rename to fanficfare/adapters/adapter_checkmatedcom.py index 7772995..662eb40 100644 --- a/fff_internals/adapters/adapter_checkmatedcom.py +++ b/fanficfare/adapters/adapter_checkmatedcom.py @@ -1,238 +1,238 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - - -def getClass(): - return CheckmatedComAdapter - - -class CheckmatedComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - self._setURL('http://' + self.getSiteDomain() + '/story.php?story='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','chm') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.checkmated.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/story.php?story=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/story.php?story=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. This story is in The Bedchamber - def needToLoginCheck(self, data): - if 'This story is in The Bedchamber' in data \ - or 'That username is not in our database' in data \ - or "That password is not correct, please try again" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['name'] = self.username - params['pass'] = self.password - else: - params['name'] = self.getConfig("username") - params['pass'] = self.getConfig("password") - params['login'] = 'yes' - params['submit'] = 'login' - - loginUrl = 'http://' + self.getSiteDomain()+'/login.php' - d = self._fetchUrl(loginUrl,params) - e = self._fetchUrl(url) - - if "Welcome back," not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['name'])) - raise exceptions.FailedToLogin(url,params['name']) - return False - elif "This story is in The Bedchamber" in e: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Your account does not have sufficient priviliges to read this story.") - return False - else: - return True - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('span', {'class' : 'storytitle'}) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = a.parent.find('a', href=re.compile(r"authors.php\?name\=\w+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - a = soup.find('select', {'name' : 'chapter'}) - if a == None: - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - for chapter in a.findAll('option'): - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/story.php?story='+self.story.getMetadata('storyId')+'&chapter='+chapter['value'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # website does not keep track of word count, and there is no convenient way to calculate it - - summary = soup.find('fieldset') - summary.find('legend').extract() - summary.name='div' - self.setDescription(url,summary) - - - # Rated: NC-17
etc - table = soup.findAll('div', {'class' : 'text'})[1] - for labels in table.findAll('tr'): - value = labels.findAll('td')[1] - label = labels.findAll('td')[0] - - - if 'Rating' in stripHTML(label): - self.story.setMetadata('rating', stripHTML(value)) - - if 'Ship' in stripHTML(label): - if value.string != "none/none": - self.story.addToList('ships',value.string) - for char in value.string.split('/'): - if char != 'none': - self.story.addToList('characters',char) - - if 'Status' in stripHTML(label): - if value.find('img', {'src' : 'img/incomplete.gif'}) == None: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in stripHTML(label): - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in stripHTML(label): - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - a = self._fetchUrl(self.story.getMetadata('authorUrl')+'&cat=stories') - for story in bs.BeautifulSoup(a).findAll('table', {'class' : 'storyinfo'}): - a = story.find('a', href=re.compile(r"review.php\?s\="+self.story.getMetadata('storyId')+'&act=view')) - if a != None: - for labels in story.findAll('tr'): - value = labels.findAll('td')[1] - label = labels.findAll('td')[0] - if 'genre' in stripHTML(label): - for genre in value.findAll('img'): - self.story.addToList('genre',genre['title']) - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'resizeableText'}) - div.find('div', {'class' : 'storyTools'}).extract() - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + + +def getClass(): + return CheckmatedComAdapter + + +class CheckmatedComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + self._setURL('http://' + self.getSiteDomain() + '/story.php?story='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','chm') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.checkmated.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/story.php?story=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/story.php?story=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. This story is in The Bedchamber + def needToLoginCheck(self, data): + if 'This story is in The Bedchamber' in data \ + or 'That username is not in our database' in data \ + or "That password is not correct, please try again" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['name'] = self.username + params['pass'] = self.password + else: + params['name'] = self.getConfig("username") + params['pass'] = self.getConfig("password") + params['login'] = 'yes' + params['submit'] = 'login' + + loginUrl = 'http://' + self.getSiteDomain()+'/login.php' + d = self._fetchUrl(loginUrl,params) + e = self._fetchUrl(url) + + if "Welcome back," not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['name'])) + raise exceptions.FailedToLogin(url,params['name']) + return False + elif "This story is in The Bedchamber" in e: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Your account does not have sufficient priviliges to read this story.") + return False + else: + return True + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('span', {'class' : 'storytitle'}) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = a.parent.find('a', href=re.compile(r"authors.php\?name\=\w+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + a = soup.find('select', {'name' : 'chapter'}) + if a == None: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + for chapter in a.findAll('option'): + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/story.php?story='+self.story.getMetadata('storyId')+'&chapter='+chapter['value'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # website does not keep track of word count, and there is no convenient way to calculate it + + summary = soup.find('fieldset') + summary.find('legend').extract() + summary.name='div' + self.setDescription(url,summary) + + + # Rated: NC-17
etc + table = soup.findAll('div', {'class' : 'text'})[1] + for labels in table.findAll('tr'): + value = labels.findAll('td')[1] + label = labels.findAll('td')[0] + + + if 'Rating' in stripHTML(label): + self.story.setMetadata('rating', stripHTML(value)) + + if 'Ship' in stripHTML(label): + if value.string != "none/none": + self.story.addToList('ships',value.string) + for char in value.string.split('/'): + if char != 'none': + self.story.addToList('characters',char) + + if 'Status' in stripHTML(label): + if value.find('img', {'src' : 'img/incomplete.gif'}) == None: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in stripHTML(label): + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in stripHTML(label): + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + a = self._fetchUrl(self.story.getMetadata('authorUrl')+'&cat=stories') + for story in bs.BeautifulSoup(a).findAll('table', {'class' : 'storyinfo'}): + a = story.find('a', href=re.compile(r"review.php\?s\="+self.story.getMetadata('storyId')+'&act=view')) + if a != None: + for labels in story.findAll('tr'): + value = labels.findAll('td')[1] + label = labels.findAll('td')[0] + if 'genre' in stripHTML(label): + for genre in value.findAll('img'): + self.story.addToList('genre',genre['title']) + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'resizeableText'}) + div.find('div', {'class' : 'storyTools'}).extract() + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_csiforensicscom.py b/fanficfare/adapters/adapter_csiforensicscom.py similarity index 97% rename from fff_internals/adapters/adapter_csiforensicscom.py rename to fanficfare/adapters/adapter_csiforensicscom.py index b1154d7..51bc2fa 100644 --- a/fff_internals/adapters/adapter_csiforensicscom.py +++ b/fanficfare/adapters/adapter_csiforensicscom.py @@ -1,237 +1,237 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - - -def getClass(): - return CSIForensicsComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class CSIForensicsComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','csiforensics') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d %b %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'csi-forensics.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=5&skin=elegantcsi" - else: - addurl="&skin=elegantcsi" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "This story is rated NC-17, and therefore is not suitable for minors. If you are below the age required to view such material in your locality, please return from whence you came." in data: # XXX - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - - pt = soup.find('div', {'id' : 'pagetitle'}) - a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',a.string) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Rating - rate = stripHTML(soup.find('div',{'id':'pagetitle'})) - rate = rate[rate.rindex('[')+1:rate.rindex(']')] - self.story.setMetadata('rating', rate) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - smalldiv = soup.find('div', {'class' : 'small'}) - - - chars = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - metatext = stripHTML(smalldiv) - - if 'Completed: Yes' in metatext: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - word=soup.find(text=re.compile("Word count:")).split(':') - self.story.setMetadata('numWords', word[1]) - - cats = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - warnings = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=class(&)type_id=2(&)classid=\d+')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - date=soup.find('div',{'class' : 'bottom'}) - pd=date.find(text=re.compile("Published:")).string.split(': ') - self.story.setMetadata('datePublished', makeDate(stripHTML(pd[1].split(' U')[0]), self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(stripHTML(pd[2]), self.dateformat)) - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - pub=0 - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Genres' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - smalldiv.extract() - - # Summary - summary = soup.find('div', {'class' : 'content'}) - self.setDescription(url,summary) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + + +def getClass(): + return CSIForensicsComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class CSIForensicsComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','csiforensics') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d %b %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'csi-forensics.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=5&skin=elegantcsi" + else: + addurl="&skin=elegantcsi" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "This story is rated NC-17, and therefore is not suitable for minors. If you are below the age required to view such material in your locality, please return from whence you came." in data: # XXX + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + + pt = soup.find('div', {'id' : 'pagetitle'}) + a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',a.string) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Rating + rate = stripHTML(soup.find('div',{'id':'pagetitle'})) + rate = rate[rate.rindex('[')+1:rate.rindex(']')] + self.story.setMetadata('rating', rate) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + smalldiv = soup.find('div', {'class' : 'small'}) + + + chars = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + metatext = stripHTML(smalldiv) + + if 'Completed: Yes' in metatext: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + word=soup.find(text=re.compile("Word count:")).split(':') + self.story.setMetadata('numWords', word[1]) + + cats = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + warnings = smalldiv.findAll('a',href=re.compile(r'browse.php\?type=class(&)type_id=2(&)classid=\d+')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + date=soup.find('div',{'class' : 'bottom'}) + pd=date.find(text=re.compile("Published:")).string.split(': ') + self.story.setMetadata('datePublished', makeDate(stripHTML(pd[1].split(' U')[0]), self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(stripHTML(pd[2]), self.dateformat)) + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + pub=0 + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Genres' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + smalldiv.extract() + + # Summary + summary = soup.find('div', {'class' : 'content'}) + self.setDescription(url,summary) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) \ No newline at end of file diff --git a/fff_internals/adapters/adapter_darksolaceorg.py b/fanficfare/adapters/adapter_darksolaceorg.py similarity index 97% rename from fff_internals/adapters/adapter_darksolaceorg.py rename to fanficfare/adapters/adapter_darksolaceorg.py index 10b0fe6..b52d7fe 100644 --- a/fff_internals/adapters/adapter_darksolaceorg.py +++ b/fanficfare/adapters/adapter_darksolaceorg.py @@ -1,336 +1,336 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DarkSolaceOrgAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DarkSolaceOrgAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/elysian/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','dksl') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'dark-solace.org' - - @classmethod - def getAcceptDomains(cls): - return ['www.dark-solace.org','dark-solace.org'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/elysian/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/elysian/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'This story contains adult content not suitable for children' in data \ - or "That password doesn't match the one in our database" in data \ - or "Registered Users Only" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['action'] = 'login' - params['submit'] = 'Submit' - - loginUrl = 'http://www.' + self.getSiteDomain() + '/elysian/user.php' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._postUrl(loginUrl, params) - - if "Member Account" not in d : #User Account Page - logger.info("Failed to login to URL %s as %s, or have no authorization to access the story" % (loginUrl, params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=5" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title and author - div = soup.find('div', {'id' : 'pagetitle'}) - - aut = div.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',aut['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/elysian/'+aut['href']) - self.story.setMetadata('author',aut.string) - aut.extract() - - # first a tag in pagetitle is title - self.story.setMetadata('title',stripHTML(div.find('a'))) - div.find('a').extract() - # only thing left in div(pagetitle) now should be 'by' and rating. - rating = stripHTML(div) - if '[' in rating: - self.story.setMetadata('rating', rating[rating.index('[')+1:-1]) - - for chapa in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+ - self.story.getMetadata('storyId')+'&chapter=\d+')): - self.chapterUrls.append((stripHTML(chapa),'http://'+self.host+'/elysian/'+chapa['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - storylink = asoup.find('a', href=re.compile(r'viewstory.php\?sid='+ - self.story.getMetadata('storyId')+'($|[^\d])')) - # author's story list is paginated if there's a pagelinks div. - # Only need to look in it if the story wasn't on the first page. - pagelinks = asoup.find('div',{'id':'pagelinks'}) - if pagelinks and storylink==None: - authpageslist = pagelinks.findAll('a',href=re.compile(r'action=storiesby')) - for page in authpageslist[1:]: # skip first, already checked above. - asoup = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+'/elysian/'+page['href'])) - storylink = asoup.find('a', href=re.compile(r'viewstory.php\?sid='+ - self.story.getMetadata('storyId')+'($|[^\d])')) - if storylink: - break - - if not storylink: - raise exceptions.FailedToDownload("Unable to find story metadata on author's page(s)") - - metalist = storylink.parent.parent - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - labels = metalist.findAll('span', {'class' : 'label'}) - for labelspan in labels: - label = labelspan.text - value = labelspan.nextSibling - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while value and not (defaultGetattr(value,'class') == 'label' or "Chapters: " in stripHTML(value)): - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = metalist.find('a', href=re.compile(r"series.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/elysian/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storylink = seriessoup.find('a', href=re.compile(r'viewstory.php\?sid='+ - self.story.getMetadata('storyId')+'($|[^\d])')) - if storylink and storylink.parent and storylink.parent['class'] != 'title': # in case of links inside story summaries. - storylink = None - - offset = 0 - # series story list is paginated if there's a pagelinks div. - # Only need to look in it if the story wasn't on the first page. - pagelinks = seriessoup.find('div',{'id':'pagelinks'}) - if pagelinks and storylink==None: - authpageslist = pagelinks.findAll('a',href=re.compile(r'offset=')) - for page in authpageslist[1:]: # skip first, already checked above. - seriessoup = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+'/elysian/'+page['href'])) - storylink = seriessoup.find('a', href=re.compile(r'viewstory.php\?sid='+ - self.story.getMetadata('storyId')+'($|[^\d])')) - if storylink and storylink.parent and storylink.parent['class'] != 'title': # in case of links inside story summaries. - storylink = None - if storylink: - offset = int(page['href'].split('=')[-1]) # offset is last. - break - - # for reasons I don't understand, searching for story - # links by regex wasn't working reliably. It was missing - # the javascript links sometimes. This is cleaner anyway. - for i, div in enumerate(seriessoup.findAll('div', {'class':'title'})): - a = div.find('a') # first a is story link. - # skip 'report this' and 'TOC' links - if a == storylink: - self.setSeries(series_name, 1+i+offset) - self.story.setMetadata('seriesUrl',series_url) - break - - except Exception, e: - logger.debug("Series parsing failed: %s"%e) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DarkSolaceOrgAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DarkSolaceOrgAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/elysian/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','dksl') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'dark-solace.org' + + @classmethod + def getAcceptDomains(cls): + return ['www.dark-solace.org','dark-solace.org'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/elysian/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/elysian/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'This story contains adult content not suitable for children' in data \ + or "That password doesn't match the one in our database" in data \ + or "Registered Users Only" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['action'] = 'login' + params['submit'] = 'Submit' + + loginUrl = 'http://www.' + self.getSiteDomain() + '/elysian/user.php' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._postUrl(loginUrl, params) + + if "Member Account" not in d : #User Account Page + logger.info("Failed to login to URL %s as %s, or have no authorization to access the story" % (loginUrl, params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=5" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title and author + div = soup.find('div', {'id' : 'pagetitle'}) + + aut = div.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',aut['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/elysian/'+aut['href']) + self.story.setMetadata('author',aut.string) + aut.extract() + + # first a tag in pagetitle is title + self.story.setMetadata('title',stripHTML(div.find('a'))) + div.find('a').extract() + # only thing left in div(pagetitle) now should be 'by' and rating. + rating = stripHTML(div) + if '[' in rating: + self.story.setMetadata('rating', rating[rating.index('[')+1:-1]) + + for chapa in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+ + self.story.getMetadata('storyId')+'&chapter=\d+')): + self.chapterUrls.append((stripHTML(chapa),'http://'+self.host+'/elysian/'+chapa['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + storylink = asoup.find('a', href=re.compile(r'viewstory.php\?sid='+ + self.story.getMetadata('storyId')+'($|[^\d])')) + # author's story list is paginated if there's a pagelinks div. + # Only need to look in it if the story wasn't on the first page. + pagelinks = asoup.find('div',{'id':'pagelinks'}) + if pagelinks and storylink==None: + authpageslist = pagelinks.findAll('a',href=re.compile(r'action=storiesby')) + for page in authpageslist[1:]: # skip first, already checked above. + asoup = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+'/elysian/'+page['href'])) + storylink = asoup.find('a', href=re.compile(r'viewstory.php\?sid='+ + self.story.getMetadata('storyId')+'($|[^\d])')) + if storylink: + break + + if not storylink: + raise exceptions.FailedToDownload("Unable to find story metadata on author's page(s)") + + metalist = storylink.parent.parent + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + labels = metalist.findAll('span', {'class' : 'label'}) + for labelspan in labels: + label = labelspan.text + value = labelspan.nextSibling + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while value and not (defaultGetattr(value,'class') == 'label' or "Chapters: " in stripHTML(value)): + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = metalist.find('a', href=re.compile(r"series.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/elysian/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storylink = seriessoup.find('a', href=re.compile(r'viewstory.php\?sid='+ + self.story.getMetadata('storyId')+'($|[^\d])')) + if storylink and storylink.parent and storylink.parent['class'] != 'title': # in case of links inside story summaries. + storylink = None + + offset = 0 + # series story list is paginated if there's a pagelinks div. + # Only need to look in it if the story wasn't on the first page. + pagelinks = seriessoup.find('div',{'id':'pagelinks'}) + if pagelinks and storylink==None: + authpageslist = pagelinks.findAll('a',href=re.compile(r'offset=')) + for page in authpageslist[1:]: # skip first, already checked above. + seriessoup = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+'/elysian/'+page['href'])) + storylink = seriessoup.find('a', href=re.compile(r'viewstory.php\?sid='+ + self.story.getMetadata('storyId')+'($|[^\d])')) + if storylink and storylink.parent and storylink.parent['class'] != 'title': # in case of links inside story summaries. + storylink = None + if storylink: + offset = int(page['href'].split('=')[-1]) # offset is last. + break + + # for reasons I don't understand, searching for story + # links by regex wasn't working reliably. It was missing + # the javascript links sometimes. This is cleaner anyway. + for i, div in enumerate(seriessoup.findAll('div', {'class':'title'})): + a = div.find('a') # first a is story link. + # skip 'report this' and 'TOC' links + if a == storylink: + self.setSeries(series_name, 1+i+offset) + self.story.setMetadata('seriesUrl',series_url) + break + + except Exception, e: + logger.debug("Series parsing failed: %s"%e) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_destinysgatewaycom.py b/fanficfare/adapters/adapter_destinysgatewaycom.py similarity index 97% rename from fff_internals/adapters/adapter_destinysgatewaycom.py rename to fanficfare/adapters/adapter_destinysgatewaycom.py index 3551860..1a5f4b1 100644 --- a/fff_internals/adapters/adapter_destinysgatewaycom.py +++ b/fanficfare/adapters/adapter_destinysgatewaycom.py @@ -1,244 +1,244 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DestinysGatewayComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DestinysGatewayComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','dgrfa') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.destinysgateway.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&warning=4" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DestinysGatewayComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DestinysGatewayComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','dgrfa') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.destinysgateway.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&warning=4" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_devianthearts.py b/fanficfare/adapters/adapter_devianthearts.py similarity index 96% rename from fff_internals/adapters/adapter_devianthearts.py rename to fanficfare/adapters/adapter_devianthearts.py index be541ae..a6c1ea3 100644 --- a/fff_internals/adapters/adapter_devianthearts.py +++ b/fanficfare/adapters/adapter_devianthearts.py @@ -1,53 +1,53 @@ -# -*- coding: utf-8 -*- - -# Copyright 2015 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import re -from base_efiction_adapter import BaseEfictionAdapter - -class DeviantHeartsAdapter(BaseEfictionAdapter): - - @staticmethod - def getSiteDomain(): - return 'devianthearts.com' - - @classmethod - def getSiteAbbrev(self): - return 'devhrt' - - @classmethod - def getDateFormat(self): - return "%m/%d/%y" - - # def handleMetadataPair(self, key, value): - # if key == 'Warnings': - # for val in re.split("\s*,\s*", value): - # if value == 'None': - # return - # else: - # # toss numbers only. - # self.story.addToList('warnings', filter(lambda x : not x.isdigit() , val)) - - # # elif 'Categories' in key: - # # for val in re.split("\s*>\s*", value): - # # self.story.addToList('category', val) - # else: - # super(FHSArchiveComAdapter, self).handleMetadataPair(key, value) - -def getClass(): - return DeviantHeartsAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2015 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import re +from base_efiction_adapter import BaseEfictionAdapter + +class DeviantHeartsAdapter(BaseEfictionAdapter): + + @staticmethod + def getSiteDomain(): + return 'devianthearts.com' + + @classmethod + def getSiteAbbrev(self): + return 'devhrt' + + @classmethod + def getDateFormat(self): + return "%m/%d/%y" + + # def handleMetadataPair(self, key, value): + # if key == 'Warnings': + # for val in re.split("\s*,\s*", value): + # if value == 'None': + # return + # else: + # # toss numbers only. + # self.story.addToList('warnings', filter(lambda x : not x.isdigit() , val)) + + # # elif 'Categories' in key: + # # for val in re.split("\s*>\s*", value): + # # self.story.addToList('category', val) + # else: + # super(FHSArchiveComAdapter, self).handleMetadataPair(key, value) + +def getClass(): + return DeviantHeartsAdapter + diff --git a/fff_internals/adapters/adapter_dokugacom.py b/fanficfare/adapters/adapter_dokugacom.py similarity index 97% rename from fff_internals/adapters/adapter_dokugacom.py rename to fanficfare/adapters/adapter_dokugacom.py index 43e5a70..7c1d501 100644 --- a/fff_internals/adapters/adapter_dokugacom.py +++ b/fanficfare/adapters/adapter_dokugacom.py @@ -1,278 +1,278 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DokugaComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DokugaComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[3]) - - - # www.dokuga.com has two 'sections', shown in URL as - # 'fanfiction' and 'spark' that change how things should be - # handled. - # http://www.dokuga.com/fanfiction/story/7528/1 - # http://www.dokuga.com/spark/story/7299/1 - self.section=self.parsedUrl.path.split('/',)[1] - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/'+self.parsedUrl.path.split('/',)[1]+'/story/'+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','dkg') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - if 'fanfiction' in self.section: - self.dateformat = "%d %b %Y" - else: - self.dateformat = "%m-%d-%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.dokuga.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/1 http://"+cls.getSiteDomain()+"/spark/story/1234/1" - - def getSiteURLPattern(self): - return r"http://"+self.getSiteDomain()+"/(fanfiction|spark)?/story/\d+/?\d+?$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'The author has disabled anonymous viewing for this story.' in data: - return True - else: - return False - - def performLogin(self, url,soup): - params = {} - - if self.password: - params['username'] = self.username - params['passwd'] = self.password - else: - params['username'] = self.getConfig("username") - params['passwd'] = self.getConfig("password") - params['Submit'] = 'Submit' - - # copy all hidden input tags to pick up appropriate tokens. - for tag in soup.findAll('input',{'type':'hidden'}): - params[tag['name']] = tag['value'] - - loginUrl = 'http://' + self.getSiteDomain() + '/fanfiction' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['username'])) - - d = self._postUrl(loginUrl, params) - - if "Your session has expired. Please log in again." in d: - d = self._postUrl(loginUrl, params) - - if "Logout" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['username'])) - raise exceptions.FailedToLogin(url,params['username']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url,soup) - data = self._fetchUrl(url) - soup = bs.BeautifulSoup(data) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title and author - a = soup.find('div', {'align' : 'center'}).find('h3') - - # Find authorid and URL from... author url. - aut = a.find('a') - self.story.setMetadata('authorId',aut['href'].split('=')[1]) - alink='http://'+self.host+aut['href'] - self.story.setMetadata('authorUrl','http://'+self.host+aut['href']) - self.story.setMetadata('author',aut.string) - aut.extract() - - a = a.string[:(len(a.string)-4)] - self.story.setMetadata('title',stripHTML(a)) - - # Find the chapters: - chapters = soup.find('select').findAll('option') - if len(chapters)==1: - self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/'+self.section+'/story/'+self.story.getMetadata('storyId')+'/1')) - else: - for chapter in chapters: - # just in case there's tags, like in chapter titles. /fanfiction/story/7406/1 - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/story/'+self.story.getMetadata('storyId')+'/'+chapter['value'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - asoup = bs.BeautifulSoup(self._fetchUrl(alink)) - - if 'fanfiction' in self.section: - asoup=asoup.find('div', {'id' : 'cb_tabid_52'}).find('div') - - #grab the rest of the metadata from the author's page - for div in asoup.findAll('div'): - nav=div.find('a', href=re.compile(r'/fanfiction/story/'+self.story.getMetadata('storyId')+"/1$")) - if nav != None: - break - div=div.nextSibling - self.setDescription(url,div) - - div=div.nextSibling - - a=div.text.split('Rating: ') - if len(a) == 2: self.story.setMetadata('rating', a[1].split('&')[0]) - - a=div.text.split('Status: ') - if len(a)==2: - iscomp=a[1].split('&')[0] - if 'Complete' in iscomp: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - a=div.text.split('Category: ') - if len(a) == 2: self.story.addToList('category', a[1].split('&')[0]) - - a=div.text.split('Created: ') - if len(a) == 2: self.story.setMetadata('datePublished', makeDate(stripHTML(a[1].split('&')[0]), self.dateformat)) - - a=div.text.split('Updated: ') - if len(a) == 2: self.story.setMetadata('dateUpdated', makeDate(stripHTML(a[1]), self.dateformat)) - - div=div.nextSibling.nextSibling - a=div.text.split('Words: ') - if len(a) == 2: self.story.setMetadata('numWords', a[1].split('&')[0]) - - a=div.text.split('Genre: ') - if len(a) == 2: - for genre in a[1].split('&')[0].split(', '): - self.story.addToList('genre',genre) - - else: - asoup=asoup.find('div', {'id' : 'maincol'}).find('div', {'class' : 'padding'}) - for div in asoup.findAll('div'): - nav=div.find('a', href=re.compile(r'/spark/story/'+self.story.getMetadata('storyId')+"/1$")) - if nav != None: - break - - div=div.nextSibling.nextSibling - self.setDescription(url,div) - self.story.addToList('category', 'Spark') - - div=div.nextSibling.nextSibling - a=div.text.split('Rating: ') - if len(a) == 2: self.story.setMetadata('rating', a[1].split(' - ')[0]) - - a=div.text.split('Status: ') - if len(a)==2: - iscomp=a[1].split(' - ')[0] - if 'Complete' in iscomp: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - a=div.text.split('Genre: ') - if len(a)==2: - for genre in a[1].split(' - ')[0].split('/'): - self.story.addToList('genre',genre) - - div=div.nextSibling.nextSibling - - a=div.text.split('Updated: ') - if len(a)==2: - date=a[1].split(' -')[0] - self.story.setMetadata('dateUpdated', makeDate(date, self.dateformat)) - - # does not have published date anywhere - self.story.setMetadata('datePublished', makeDate(date, self.dateformat)) - - a=div.text.split('Words ') - if len(a)==2: self.story.setMetadata('numWords', a[1]) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'chtext'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DokugaComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DokugaComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[3]) + + + # www.dokuga.com has two 'sections', shown in URL as + # 'fanfiction' and 'spark' that change how things should be + # handled. + # http://www.dokuga.com/fanfiction/story/7528/1 + # http://www.dokuga.com/spark/story/7299/1 + self.section=self.parsedUrl.path.split('/',)[1] + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/'+self.parsedUrl.path.split('/',)[1]+'/story/'+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','dkg') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + if 'fanfiction' in self.section: + self.dateformat = "%d %b %Y" + else: + self.dateformat = "%m-%d-%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.dokuga.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/1 http://"+cls.getSiteDomain()+"/spark/story/1234/1" + + def getSiteURLPattern(self): + return r"http://"+self.getSiteDomain()+"/(fanfiction|spark)?/story/\d+/?\d+?$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'The author has disabled anonymous viewing for this story.' in data: + return True + else: + return False + + def performLogin(self, url,soup): + params = {} + + if self.password: + params['username'] = self.username + params['passwd'] = self.password + else: + params['username'] = self.getConfig("username") + params['passwd'] = self.getConfig("password") + params['Submit'] = 'Submit' + + # copy all hidden input tags to pick up appropriate tokens. + for tag in soup.findAll('input',{'type':'hidden'}): + params[tag['name']] = tag['value'] + + loginUrl = 'http://' + self.getSiteDomain() + '/fanfiction' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['username'])) + + d = self._postUrl(loginUrl, params) + + if "Your session has expired. Please log in again." in d: + d = self._postUrl(loginUrl, params) + + if "Logout" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['username'])) + raise exceptions.FailedToLogin(url,params['username']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url,soup) + data = self._fetchUrl(url) + soup = bs.BeautifulSoup(data) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title and author + a = soup.find('div', {'align' : 'center'}).find('h3') + + # Find authorid and URL from... author url. + aut = a.find('a') + self.story.setMetadata('authorId',aut['href'].split('=')[1]) + alink='http://'+self.host+aut['href'] + self.story.setMetadata('authorUrl','http://'+self.host+aut['href']) + self.story.setMetadata('author',aut.string) + aut.extract() + + a = a.string[:(len(a.string)-4)] + self.story.setMetadata('title',stripHTML(a)) + + # Find the chapters: + chapters = soup.find('select').findAll('option') + if len(chapters)==1: + self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/'+self.section+'/story/'+self.story.getMetadata('storyId')+'/1')) + else: + for chapter in chapters: + # just in case there's tags, like in chapter titles. /fanfiction/story/7406/1 + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/story/'+self.story.getMetadata('storyId')+'/'+chapter['value'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + asoup = bs.BeautifulSoup(self._fetchUrl(alink)) + + if 'fanfiction' in self.section: + asoup=asoup.find('div', {'id' : 'cb_tabid_52'}).find('div') + + #grab the rest of the metadata from the author's page + for div in asoup.findAll('div'): + nav=div.find('a', href=re.compile(r'/fanfiction/story/'+self.story.getMetadata('storyId')+"/1$")) + if nav != None: + break + div=div.nextSibling + self.setDescription(url,div) + + div=div.nextSibling + + a=div.text.split('Rating: ') + if len(a) == 2: self.story.setMetadata('rating', a[1].split('&')[0]) + + a=div.text.split('Status: ') + if len(a)==2: + iscomp=a[1].split('&')[0] + if 'Complete' in iscomp: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + a=div.text.split('Category: ') + if len(a) == 2: self.story.addToList('category', a[1].split('&')[0]) + + a=div.text.split('Created: ') + if len(a) == 2: self.story.setMetadata('datePublished', makeDate(stripHTML(a[1].split('&')[0]), self.dateformat)) + + a=div.text.split('Updated: ') + if len(a) == 2: self.story.setMetadata('dateUpdated', makeDate(stripHTML(a[1]), self.dateformat)) + + div=div.nextSibling.nextSibling + a=div.text.split('Words: ') + if len(a) == 2: self.story.setMetadata('numWords', a[1].split('&')[0]) + + a=div.text.split('Genre: ') + if len(a) == 2: + for genre in a[1].split('&')[0].split(', '): + self.story.addToList('genre',genre) + + else: + asoup=asoup.find('div', {'id' : 'maincol'}).find('div', {'class' : 'padding'}) + for div in asoup.findAll('div'): + nav=div.find('a', href=re.compile(r'/spark/story/'+self.story.getMetadata('storyId')+"/1$")) + if nav != None: + break + + div=div.nextSibling.nextSibling + self.setDescription(url,div) + self.story.addToList('category', 'Spark') + + div=div.nextSibling.nextSibling + a=div.text.split('Rating: ') + if len(a) == 2: self.story.setMetadata('rating', a[1].split(' - ')[0]) + + a=div.text.split('Status: ') + if len(a)==2: + iscomp=a[1].split(' - ')[0] + if 'Complete' in iscomp: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + a=div.text.split('Genre: ') + if len(a)==2: + for genre in a[1].split(' - ')[0].split('/'): + self.story.addToList('genre',genre) + + div=div.nextSibling.nextSibling + + a=div.text.split('Updated: ') + if len(a)==2: + date=a[1].split(' -')[0] + self.story.setMetadata('dateUpdated', makeDate(date, self.dateformat)) + + # does not have published date anywhere + self.story.setMetadata('datePublished', makeDate(date, self.dateformat)) + + a=div.text.split('Words ') + if len(a)==2: self.story.setMetadata('numWords', a[1]) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'chtext'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_dotmoonnet.py b/fanficfare/adapters/adapter_dotmoonnet.py similarity index 97% rename from fff_internals/adapters/adapter_dotmoonnet.py rename to fanficfare/adapters/adapter_dotmoonnet.py index f41d3f3..a80ee77 100644 --- a/fff_internals/adapters/adapter_dotmoonnet.py +++ b/fanficfare/adapters/adapter_dotmoonnet.py @@ -1,217 +1,217 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DotMoonNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DotMoonNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. www.dotmoon.net/library_view.php?storyid=3 - self._setURL('http://' + self.getSiteDomain() + '/library_view.php?storyid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','dotm') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%Y-%m-%d" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.dotmoon.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/library_view.php?storyid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/library_view.php?storyid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'You must be logged in to read adult-rated stories' in data \ - or 'Password incorrect' in data \ - or "That username does not exist" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['user'] = self.username - params['passwrd'] = self.password - else: - params['user'] = self.getConfig("username") - params['passwrd'] = self.getConfig("password") - - loginUrl = 'http://' + self.getSiteDomain() + '/board/index.php' - - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['user'])) - - d = self._fetchUrl(loginUrl+'?action=login2&user='+params['user']+'&passwrd='+params['passwrd']) - d = self._fetchUrl(loginUrl) - - if "Show unread posts since last visit" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['user'])) - raise exceptions.FailedToLogin(url,params['user']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - if "Invalid story ID" in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Invalid story ID.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - body=soup.findAll('body')[1] - body.find('table').extract() - - ## Title - a = body.find('b') - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. http://www.dotmoon.net/board/index.php?action=profile;u=1' - a = body.find('a', href=re.compile(r"index.php\?action=profile;u=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[2]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: 'library_storyview.php?chapterid=3 - chapters=body.findAll('a', href=re.compile(r"library_storyview.php\?chapterid=\d+$")) - if len(chapters)==0: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: No php/html chapters found.") - if len(chapters)==1: - self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/'+chapters[0]['href'])) - else: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # other tags - - labels = body.find('table', {'width':'390'}).findAll('td') - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if label != None: - if 'Fandom' in label: - self.story.addToList('category',value.string) - - if 'Setting' in label: - self.story.addToList('genre',value.string) - - if 'Genre' in label: - self.story.addToList('genre',value.string) - - if 'Style' in label: - self.story.addToList('genre',value.string) - - if 'Rating' in label: - self.story.addToList('rating',value.string) - - if 'Created' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - if 'Status' in label: - if 'Completed' in value.string: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - table=body.findAll('table', {'width':'400'})[1].find('td') - self.setDescription(url,stripHTML(table).split('Summary: ')[1]) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('blockquote') - div.name='div' - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DotMoonNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DotMoonNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. www.dotmoon.net/library_view.php?storyid=3 + self._setURL('http://' + self.getSiteDomain() + '/library_view.php?storyid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','dotm') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y-%m-%d" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.dotmoon.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/library_view.php?storyid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/library_view.php?storyid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'You must be logged in to read adult-rated stories' in data \ + or 'Password incorrect' in data \ + or "That username does not exist" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['user'] = self.username + params['passwrd'] = self.password + else: + params['user'] = self.getConfig("username") + params['passwrd'] = self.getConfig("password") + + loginUrl = 'http://' + self.getSiteDomain() + '/board/index.php' + + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['user'])) + + d = self._fetchUrl(loginUrl+'?action=login2&user='+params['user']+'&passwrd='+params['passwrd']) + d = self._fetchUrl(loginUrl) + + if "Show unread posts since last visit" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['user'])) + raise exceptions.FailedToLogin(url,params['user']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + if "Invalid story ID" in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Invalid story ID.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + body=soup.findAll('body')[1] + body.find('table').extract() + + ## Title + a = body.find('b') + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. http://www.dotmoon.net/board/index.php?action=profile;u=1' + a = body.find('a', href=re.compile(r"index.php\?action=profile;u=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[2]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: 'library_storyview.php?chapterid=3 + chapters=body.findAll('a', href=re.compile(r"library_storyview.php\?chapterid=\d+$")) + if len(chapters)==0: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: No php/html chapters found.") + if len(chapters)==1: + self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/'+chapters[0]['href'])) + else: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # other tags + + labels = body.find('table', {'width':'390'}).findAll('td') + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if label != None: + if 'Fandom' in label: + self.story.addToList('category',value.string) + + if 'Setting' in label: + self.story.addToList('genre',value.string) + + if 'Genre' in label: + self.story.addToList('genre',value.string) + + if 'Style' in label: + self.story.addToList('genre',value.string) + + if 'Rating' in label: + self.story.addToList('rating',value.string) + + if 'Created' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + if 'Status' in label: + if 'Completed' in value.string: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + table=body.findAll('table', {'width':'400'})[1].find('td') + self.setDescription(url,stripHTML(table).split('Summary: ')[1]) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('blockquote') + div.name='div' + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_dracoandginnycom.py b/fanficfare/adapters/adapter_dracoandginnycom.py similarity index 97% rename from fff_internals/adapters/adapter_dracoandginnycom.py rename to fanficfare/adapters/adapter_dracoandginnycom.py index 1a73283..1444ed7 100644 --- a/fff_internals/adapters/adapter_dracoandginnycom.py +++ b/fanficfare/adapters/adapter_dracoandginnycom.py @@ -1,302 +1,302 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DracoAndGinnyComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DracoAndGinnyComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','dcagn') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.dracoandginny.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=2" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - content=soup.find('div',{'class' : 'listbox'}) - - self.setDescription(url,content.find('blockquote')) - - for genre in content.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')): - self.story.addToList('genre',genre.string) - - for warning in content.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')): - self.story.addToList('warnings',warning.string) - - labels = content.findAll('b') - - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - - if 'Word count' in label: - self.story.setMetadata('numWords', value.split(' |')[0]) - - if 'Rating' in label: - self.story.setMetadata('rating', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value.nextSibling).split(' |')[0], self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'class' : 'listbox'}) - - if None == div: - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DracoAndGinnyComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DracoAndGinnyComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','dcagn') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.dracoandginny.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=2" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + content=soup.find('div',{'class' : 'listbox'}) + + self.setDescription(url,content.find('blockquote')) + + for genre in content.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')): + self.story.addToList('genre',genre.string) + + for warning in content.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')): + self.story.addToList('warnings',warning.string) + + labels = content.findAll('b') + + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + + if 'Word count' in label: + self.story.setMetadata('numWords', value.split(' |')[0]) + + if 'Rating' in label: + self.story.setMetadata('rating', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value.nextSibling).split(' |')[0], self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'class' : 'listbox'}) + + if None == div: + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_dramioneorg.py b/fanficfare/adapters/adapter_dramioneorg.py similarity index 97% rename from fff_internals/adapters/adapter_dramioneorg.py rename to fanficfare/adapters/adapter_dramioneorg.py index 3b3768f..65b831a 100644 --- a/fff_internals/adapters/adapter_dramioneorg.py +++ b/fanficfare/adapters/adapter_dramioneorg.py @@ -1,311 +1,311 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return DramioneOrgAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class DramioneOrgAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','drmn') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d %B %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'dramione.org' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&warning=5" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "Stories that are suitable for ages 16 and older" in data: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Use banner as cover if found - coverurl = '' - img = soup.find('img',{'class':'banner'}) - if img: - coverurl = img['src'] - #print "Cover: "+coverurl - a = soup.find(text="This story has a banner; click to view.") - if a: - #print "A: "+ ', '.join("(%s, %s)" %tup for tup in a.parent.attrs) - coverurl = a.parent['href'] - #print "Cover: "+coverurl - if coverurl: - self.setCoverImage(url,coverurl) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - genres=soup.findAll('a', {'class' : "tag-1"}) - for genre in genres: - self.story.addToList('genre',genre.string) - - warnings=soup.findAll('a', {'class' : "tag-2"}) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - themes=soup.findAll('a', {'class' : "tag-3"}) - for theme in themes: - self.story.addToList('themes',theme.string) - - hermiones=soup.findAll('a', {'class' : "tag-4"}) - for hermione in hermiones: - self.story.addToList('hermiones',hermione.string) - - dracos=soup.findAll('a', {'class' : "tag-5"}) - for draco in dracos: - self.story.addToList('dracos',draco.string) - - timelines=soup.findAll('a', {'class' : "tag-6"}) - for timeline in timelines: - self.story.addToList('timeline',timeline.string) - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Read' in label: - self.story.setMetadata('read', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - value=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",value) - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - value=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",value) - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - try: - self.story.setMetadata('reviews', - stripHTML(soup.find('h2',{'id':'pagetitle'}). - findAll('a', href=re.compile(r'^reviews.php'))[1])) - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return DramioneOrgAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class DramioneOrgAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','drmn') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d %B %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'dramione.org' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&warning=5" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "Stories that are suitable for ages 16 and older" in data: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Use banner as cover if found + coverurl = '' + img = soup.find('img',{'class':'banner'}) + if img: + coverurl = img['src'] + #print "Cover: "+coverurl + a = soup.find(text="This story has a banner; click to view.") + if a: + #print "A: "+ ', '.join("(%s, %s)" %tup for tup in a.parent.attrs) + coverurl = a.parent['href'] + #print "Cover: "+coverurl + if coverurl: + self.setCoverImage(url,coverurl) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + genres=soup.findAll('a', {'class' : "tag-1"}) + for genre in genres: + self.story.addToList('genre',genre.string) + + warnings=soup.findAll('a', {'class' : "tag-2"}) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + themes=soup.findAll('a', {'class' : "tag-3"}) + for theme in themes: + self.story.addToList('themes',theme.string) + + hermiones=soup.findAll('a', {'class' : "tag-4"}) + for hermione in hermiones: + self.story.addToList('hermiones',hermione.string) + + dracos=soup.findAll('a', {'class' : "tag-5"}) + for draco in dracos: + self.story.addToList('dracos',draco.string) + + timelines=soup.findAll('a', {'class' : "tag-6"}) + for timeline in timelines: + self.story.addToList('timeline',timeline.string) + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Read' in label: + self.story.setMetadata('read', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + value=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",value) + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + value=re.sub(r"(\d+)(st|nd|rd|th)",r"\1",value) + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + try: + self.story.setMetadata('reviews', + stripHTML(soup.find('h2',{'id':'pagetitle'}). + findAll('a', href=re.compile(r'^reviews.php'))[1])) + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_efictionestelielde.py b/fanficfare/adapters/adapter_efictionestelielde.py similarity index 97% rename from fff_internals/adapters/adapter_efictionestelielde.py rename to fanficfare/adapters/adapter_efictionestelielde.py index 2c50505..01d7064 100644 --- a/fff_internals/adapters/adapter_efictionestelielde.py +++ b/fanficfare/adapters/adapter_efictionestelielde.py @@ -1,224 +1,224 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return EfictionEstelielDeAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class EfictionEstelielDeAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','eesd') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'efiction.esteliel.de' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1' - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # Now go hunting for all the meta data and the chapter list. - - ## Title and author - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - pagetitle = soup.find('div',{'id':'pagetitle'}) - ## Title - a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - list = soup.find('div', {'class':'listbox'}) - labelspan=list.find('span',{'class':'label'}) - value = labelspan.nextSibling - label = labelspan.string - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - labels = list.findAll('b') - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while 'Rating' not in str(value): - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rating' in label: - self.story.setMetadata('rating', value) - - if 'Words' in label: - self.story.setMetadata('numWords', value) - - if 'Category' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - if list.find('a', href=re.compile(r"series.php")) != None: - for series in asoup.findAll('a', href=re.compile(r"series.php\?seriesid=\d+")): - # Find Series name from series URL. - series_url = 'http://'+self.host+'/'+series['href'] - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - name=seriessoup.find('div', {'id' : 'pagetitle'}) - name.find('a').extract() - self.setSeries(name.text.split(' by[')[0], i) - self.story.setMetadata('seriesUrl',series_url) - i=0 - break - i+=1 - if i == 0: - break - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return EfictionEstelielDeAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class EfictionEstelielDeAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','eesd') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'efiction.esteliel.de' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1' + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # Now go hunting for all the meta data and the chapter list. + + ## Title and author + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + pagetitle = soup.find('div',{'id':'pagetitle'}) + ## Title + a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + list = soup.find('div', {'class':'listbox'}) + labelspan=list.find('span',{'class':'label'}) + value = labelspan.nextSibling + label = labelspan.string + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + labels = list.findAll('b') + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while 'Rating' not in str(value): + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rating' in label: + self.story.setMetadata('rating', value) + + if 'Words' in label: + self.story.setMetadata('numWords', value) + + if 'Category' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + if list.find('a', href=re.compile(r"series.php")) != None: + for series in asoup.findAll('a', href=re.compile(r"series.php\?seriesid=\d+")): + # Find Series name from series URL. + series_url = 'http://'+self.host+'/'+series['href'] + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + name=seriessoup.find('div', {'id' : 'pagetitle'}) + name.find('a').extract() + self.setSeries(name.text.split(' by[')[0], i) + self.story.setMetadata('seriesUrl',series_url) + i=0 + break + i+=1 + if i == 0: + break + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_efpfanficnet.py b/fanficfare/adapters/adapter_efpfanficnet.py similarity index 97% rename from fff_internals/adapters/adapter_efpfanficnet.py rename to fanficfare/adapters/adapter_efpfanficnet.py index 26df765..b08d12c 100644 --- a/fff_internals/adapters/adapter_efpfanficnet.py +++ b/fanficfare/adapters/adapter_efpfanficnet.py @@ -1,316 +1,316 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return EFPFanFicNet - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class EFPFanFicNet(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','efp') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.efpfanfic.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if( 'Fai il login e leggi la storia!' in data or - 'Questa storia presenta contenuti non adatti ai minori' in data ): - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Invia' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?sid='+self.story.getMetadata('storyId') - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if '' in d : # register for new account link - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # if "Access denied. This story has not been validated by the adminstrators of this site." in data: - # raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapter selector - select = soup.find('select', { 'name' : 'sid' } ) - - if select is None: - # no selector found, so it's a one-chapter story. - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - allOptions = select.findAll('option', {'value' : re.compile(r'viewstory')}) - for o in allOptions: - url = u'http://%s/%s' % ( self.getSiteDomain(), - o['value']) - # just in case there's tags, like in chapter titles. - title = stripHTML(o) - self.chapterUrls.append((title,url)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - self.story.setMetadata('language','Italian') - - # normalize story URL to first chapter if later chapter URL was given: - url = self.chapterUrls[0][1].replace('&i=1','') - logger.debug("Normalizing to URL: "+url) - self._setURL(url) - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - storya = None - authsoup = None - storyblock = None - authurl = self.story.getMetadata('authorUrl') - - ## author can have more than one page of stories. - while storyblock == None: - - # no storya, but do have authsoup--we're looping on author pages. - if authsoup != None: - # last author link with offset should be the 'next' link. - authurl = u'http://%s/%s' % ( self.getSiteDomain(), - authsoup.findAll('a',href=re.compile(r'viewuser\.php\?uid=\d+&catid=&offset='))[-1]['href'] ) - - # Need author page for most of the metadata. - logger.debug("fetching author page: (%s)"%authurl) - authsoup = bs.BeautifulSoup(self._fetchUrl(authurl)) - #print("authsoup:%s"%authsoup) - - storyas = authsoup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+r'&i=1$')) - for storya in storyas: - #print("======storya:%s"%storya) - storyblock = storya.findParent('div',{'class':'storybloc'}) - #print("======storyblock:%s"%storyblock) - if storyblock != None: - continue - - self.setDescription(url,storyblock.find('div', {'class':'introbloc'})) - - noteblock = storyblock.find('div', {'class':'notebloc'}) - #print("%s"%noteblock) - notetext = ("%s" % noteblock).replace("
"," |") - #
Autore: Cendrillon89 | Pubblicata: 23/10/12 | Aggiornata: 30/10/12 | Rating: Arancione | Genere: Drammatico, Sentimentale | Capitoli: 10 | Completa
- # Tipo di coppia: Het | Personaggi: Akasuna no Sasori , Akatsuki, Nuovo Personaggio | Note: OOC | Avvertimenti: Tematiche delicate
- # Categoria: Anime & Manga > Naruto | Contesto: Naruto Shippuuden | Leggi le 3 recensioni
- - cats = noteblock.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - for item in notetext.split("|"): - if ":" in item: - (label,value) = item.split(":") - label=label.strip() - value=value.strip() - else: - label=value=item.strip() - - if 'Pubblicata' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Aggiornata' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - if label == "Completa": - self.story.setMetadata('status', 'Completed') - - if label == "In corso": - self.story.setMetadata('status', 'In-Progress') - - if 'Rating' in label: - self.story.setMetadata('rating', value) - - if 'Personaggi' in label: - for val in value.split(","): - self.story.addToList('characters',val) - - if 'Genere' in label: - for val in value.split(","): - self.story.addToList('genre',val) - - if 'Coppie' in label: - for val in value.split(","): - self.story.addToList('ships',val) - - if 'Avvertimenti' in label: - for val in value.split(","): - if val != "None": - self.story.addToList('warnings',val) - - # 'extra' metadata for this adapter: - - if 'Tipo di coppia' in label: - for val in value.split(","): - self.story.addToList('type',val) - - if 'Note' in label: - for val in value.split(","): - if val != "None": - self.story.addToList('notes',val) - - if 'Contesto' in label: - self.story.setMetadata('context', value) - - ## Note--efp doesn't provide word count. - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?ssid=\d+&i=1")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+&i=1')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId'))+'&i=1': - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) - - div = soup.find('div', {'class' : 'storia'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - # remove any header and 'o:p' tags. - for tag in div.findAll("head") + div.findAll("o:p"): - tag.extract() - - # change any html and body tags to div. - for tag in div.findAll("html") + div.findAll("body"): - tag.name='div' - - # remove extra bogus doctype. - # - return re.sub(r"]+>","",self.utf8FromSoup(url,div)) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return EFPFanFicNet + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class EFPFanFicNet(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','efp') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.efpfanfic.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if( 'Fai il login e leggi la storia!' in data or + 'Questa storia presenta contenuti non adatti ai minori' in data ): + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Invia' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?sid='+self.story.getMetadata('storyId') + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if '' in d : # register for new account link + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # if "Access denied. This story has not been validated by the adminstrators of this site." in data: + # raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapter selector + select = soup.find('select', { 'name' : 'sid' } ) + + if select is None: + # no selector found, so it's a one-chapter story. + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + allOptions = select.findAll('option', {'value' : re.compile(r'viewstory')}) + for o in allOptions: + url = u'http://%s/%s' % ( self.getSiteDomain(), + o['value']) + # just in case there's tags, like in chapter titles. + title = stripHTML(o) + self.chapterUrls.append((title,url)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + self.story.setMetadata('language','Italian') + + # normalize story URL to first chapter if later chapter URL was given: + url = self.chapterUrls[0][1].replace('&i=1','') + logger.debug("Normalizing to URL: "+url) + self._setURL(url) + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + storya = None + authsoup = None + storyblock = None + authurl = self.story.getMetadata('authorUrl') + + ## author can have more than one page of stories. + while storyblock == None: + + # no storya, but do have authsoup--we're looping on author pages. + if authsoup != None: + # last author link with offset should be the 'next' link. + authurl = u'http://%s/%s' % ( self.getSiteDomain(), + authsoup.findAll('a',href=re.compile(r'viewuser\.php\?uid=\d+&catid=&offset='))[-1]['href'] ) + + # Need author page for most of the metadata. + logger.debug("fetching author page: (%s)"%authurl) + authsoup = bs.BeautifulSoup(self._fetchUrl(authurl)) + #print("authsoup:%s"%authsoup) + + storyas = authsoup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+r'&i=1$')) + for storya in storyas: + #print("======storya:%s"%storya) + storyblock = storya.findParent('div',{'class':'storybloc'}) + #print("======storyblock:%s"%storyblock) + if storyblock != None: + continue + + self.setDescription(url,storyblock.find('div', {'class':'introbloc'})) + + noteblock = storyblock.find('div', {'class':'notebloc'}) + #print("%s"%noteblock) + notetext = ("%s" % noteblock).replace("
"," |") + #
Autore: Cendrillon89 | Pubblicata: 23/10/12 | Aggiornata: 30/10/12 | Rating: Arancione | Genere: Drammatico, Sentimentale | Capitoli: 10 | Completa
+ # Tipo di coppia: Het | Personaggi: Akasuna no Sasori , Akatsuki, Nuovo Personaggio | Note: OOC | Avvertimenti: Tematiche delicate
+ # Categoria: Anime & Manga > Naruto | Contesto: Naruto Shippuuden | Leggi le 3 recensioni
+ + cats = noteblock.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + for item in notetext.split("|"): + if ":" in item: + (label,value) = item.split(":") + label=label.strip() + value=value.strip() + else: + label=value=item.strip() + + if 'Pubblicata' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Aggiornata' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + if label == "Completa": + self.story.setMetadata('status', 'Completed') + + if label == "In corso": + self.story.setMetadata('status', 'In-Progress') + + if 'Rating' in label: + self.story.setMetadata('rating', value) + + if 'Personaggi' in label: + for val in value.split(","): + self.story.addToList('characters',val) + + if 'Genere' in label: + for val in value.split(","): + self.story.addToList('genre',val) + + if 'Coppie' in label: + for val in value.split(","): + self.story.addToList('ships',val) + + if 'Avvertimenti' in label: + for val in value.split(","): + if val != "None": + self.story.addToList('warnings',val) + + # 'extra' metadata for this adapter: + + if 'Tipo di coppia' in label: + for val in value.split(","): + self.story.addToList('type',val) + + if 'Note' in label: + for val in value.split(","): + if val != "None": + self.story.addToList('notes',val) + + if 'Contesto' in label: + self.story.setMetadata('context', value) + + ## Note--efp doesn't provide word count. + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?ssid=\d+&i=1")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+&i=1')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId'))+'&i=1': + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + div = soup.find('div', {'class' : 'storia'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + # remove any header and 'o:p' tags. + for tag in div.findAll("head") + div.findAll("o:p"): + tag.extract() + + # change any html and body tags to div. + for tag in div.findAll("html") + div.findAll("body"): + tag.name='div' + + # remove extra bogus doctype. + # + return re.sub(r"]+>","",self.utf8FromSoup(url,div)) diff --git a/fff_internals/adapters/adapter_erosnsapphosycophanthexcom.py b/fanficfare/adapters/adapter_erosnsapphosycophanthexcom.py similarity index 97% rename from fff_internals/adapters/adapter_erosnsapphosycophanthexcom.py rename to fanficfare/adapters/adapter_erosnsapphosycophanthexcom.py index d57ca0b..61cf08f 100644 --- a/fff_internals/adapters/adapter_erosnsapphosycophanthexcom.py +++ b/fanficfare/adapters/adapter_erosnsapphosycophanthexcom.py @@ -1,256 +1,256 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return ErosnSapphoSycophantHexComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class ErosnSapphoSycophantHexComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','essph') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'erosnsappho.sycophanthex.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=18" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - pt = soup.find('div', {'id' : 'pagetitle'}) - a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - rating=pt.text.split('(')[1].split(')')[0] - self.story.setMetadata('rating', rating) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - - labels = soup.findAll('span',{'class':'label'}) - - value = labels[0].previousSibling - svalue = "" - while value != None: - val = value - value = value.previousSibling - while not defaultGetattr(val,'class') == 'label': - svalue += str(val) - val = val.nextSibling - self.setDescription(url,svalue) - - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Word count' in label: - self.story.setMetadata('numWords', value.split(' -')[0]) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Complete' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return ErosnSapphoSycophantHexComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class ErosnSapphoSycophantHexComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','essph') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'erosnsappho.sycophanthex.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=18" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + pt = soup.find('div', {'id' : 'pagetitle'}) + a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + rating=pt.text.split('(')[1].split(')')[0] + self.story.setMetadata('rating', rating) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + + labels = soup.findAll('span',{'class':'label'}) + + value = labels[0].previousSibling + svalue = "" + while value != None: + val = value + value = value.previousSibling + while not defaultGetattr(val,'class') == 'label': + svalue += str(val) + val = val.nextSibling + self.setDescription(url,svalue) + + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Word count' in label: + self.story.setMetadata('numWords', value.split(' -')[0]) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Complete' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_fanficcastletvnet.py b/fanficfare/adapters/adapter_fanficcastletvnet.py similarity index 97% rename from fff_internals/adapters/adapter_fanficcastletvnet.py rename to fanficfare/adapters/adapter_fanficcastletvnet.py index f40fdd5..5d20207 100644 --- a/fff_internals/adapters/adapter_fanficcastletvnet.py +++ b/fanficfare/adapters/adapter_fanficcastletvnet.py @@ -1,322 +1,322 @@ -# -*- coding: utf-8 -*- - -# Copyright 2014 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# In general an 'adapter' needs to do these five things: - -# - 'Register' correctly with the downloader -# - Site Login (if needed) -# - 'Are you adult?' check (if needed--some do one, some the other, some both) -# - Grab the chapter list -# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) -# - Grab the chapter texts - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return FanficCastleTVNetAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class FanficCastleTVNetAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','csltv') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d, %Y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'fanfic.castletv.net' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=3" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - pagetitle = soup.find('div',{'id':'pagetitle'}) - ## Title - a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Reviews - reviewdata = soup.find('div', {'id' : 'sort'}) - a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one. - self.story.setMetadata('reviews',stripHTML(a)) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2014 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# In general an 'adapter' needs to do these five things: + +# - 'Register' correctly with the downloader +# - Site Login (if needed) +# - 'Are you adult?' check (if needed--some do one, some the other, some both) +# - Grab the chapter list +# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) +# - Grab the chapter texts + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return FanficCastleTVNetAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class FanficCastleTVNetAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','csltv') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d, %Y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'fanfic.castletv.net' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=3" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + pagetitle = soup.find('div',{'id':'pagetitle'}) + ## Title + a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Reviews + reviewdata = soup.find('div', {'id' : 'sort'}) + a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one. + self.story.setMetadata('reviews',stripHTML(a)) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_fanfichu.py b/fanficfare/adapters/adapter_fanfichu.py similarity index 97% rename from fff_internals/adapters/adapter_fanfichu.py rename to fanficfare/adapters/adapter_fanfichu.py index 4de2d29..05461e9 100644 --- a/fff_internals/adapters/adapter_fanfichu.py +++ b/fanficfare/adapters/adapter_fanfichu.py @@ -1,185 +1,185 @@ -# coding=utf-8 - -import re -import urllib2 -import urlparse - -from .. import BeautifulSoup - -from base_adapter import BaseSiteAdapter, makeDate -from .. import exceptions - - -_SOURCE_CODE_ENCODING = 'utf-8' - - -def getClass(): - return FanficHuAdapter - - -def _get_query_data(url): - components = urlparse.urlparse(url) - query_data = urlparse.parse_qs(components.query) - return dict((key, data[0]) for key, data in query_data.items()) - - -class FanficHuAdapter(BaseSiteAdapter): - SITE_ABBREVIATION = 'ffh' - SITE_DOMAIN = 'fanfic.hu' - SITE_LANGUAGE = 'Hungarian' - - BASE_URL = 'http://' + SITE_DOMAIN + '/merengo/' - VIEW_STORY_URL_TEMPLATE = BASE_URL + 'viewstory.php?sid=%s' - - DATE_FORMAT = '%m/%d/%Y' - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - query_data = urlparse.parse_qs(self.parsedUrl.query) - story_id = query_data['sid'][0] - - self.story.setMetadata('storyId', story_id) - self._setURL(self.VIEW_STORY_URL_TEMPLATE % story_id) - self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) - self.story.setMetadata('language', self.SITE_LANGUAGE) - - def _customized_fetch_url(self, url, exception=None, parameters=None): - if exception: - try: - data = self._fetchUrl(url, parameters) - except urllib2.HTTPError: - raise exception(self.url) - # Just let self._fetchUrl throw the exception, don't catch and - # customize it. - else: - data = self._fetchUrl(url, parameters) - - return BeautifulSoup.BeautifulSoup(data) - - @staticmethod - def getSiteDomain(): - return FanficHuAdapter.SITE_DOMAIN - - @classmethod - def getSiteExampleURLs(cls): - return cls.VIEW_STORY_URL_TEMPLATE % 1234 - - def getSiteURLPattern(self): - return re.escape(self.VIEW_STORY_URL_TEMPLATE[:-2]) + r'\d+$' - - def extractChapterUrlsAndMetadata(self): - soup = self._customized_fetch_url(self.url + '&i=1') - - if soup.title.string.encode(_SOURCE_CODE_ENCODING).strip(' :') == 'írta': - raise exceptions.StoryDoesNotExist(self.url) - - chapter_options = soup.find('form', action='viewstory.php').select('option') - # Remove redundant "Fejezetek" option - chapter_options.pop(0) - - # If there is still more than one entry remove chapter overview entry - if len(chapter_options) > 1: - chapter_options.pop(0) - - for option in chapter_options: - url = urlparse.urljoin(self.url, option['value']) - self.chapterUrls.append((option.string, url)) - - author_url = urlparse.urljoin(self.BASE_URL, soup.find('a', href=lambda href: href and href.startswith('viewuser.php?uid='))['href']) - soup = self._customized_fetch_url(author_url) - - story_id = self.story.getMetadata('storyId') - for table in soup('table', {'class': 'mainnav'}): - title_anchor = table.find('span', {'class': 'storytitle'}).a - href = title_anchor['href'] - if href.startswith('javascript:'): - href = href.rsplit(' ', 1)[1].strip("'") - query_data = _get_query_data(href) - - if query_data['sid'] == story_id: - break - else: - # This should never happen, the story must be found on the author's - # page. - raise exceptions.FailedToDownload(self.url) - - self.story.setMetadata('title', title_anchor.string) - - rows = table('tr') - - anchors = rows[0].div('a') - author_anchor = anchors[1] - query_data = _get_query_data(author_anchor['href']) - self.story.setMetadata('author', author_anchor.string) - self.story.setMetadata('authorId', query_data['uid']) - self.story.setMetadata('authorUrl', urlparse.urljoin(self.BASE_URL, author_anchor['href'])) - self.story.setMetadata('reviews', anchors[3].string) - - if self.getConfig('keep_summary_html'): - self.story.setMetadata('description', self.utf8FromSoup(author_url, rows[1].td)) - else: - self.story.setMetadata('description', ''.join(rows[1].td(text=True))) - - for row in rows[3:]: - index = 0 - cells = row('td') - - while index < len(cells): - cell = cells[index] - key = cell.b.string.encode(_SOURCE_CODE_ENCODING).strip(':') - try: - value = cells[index+1].string.encode(_SOURCE_CODE_ENCODING) - except AttributeError: - value = None - - if key == 'Kategória': - for anchor in cells[index+1]('a'): - self.story.addToList('category', anchor.string) - - elif key == 'Szereplõk': - if cells[index+1].string: - for name in cells[index+1].string.split(', '): - self.story.addToList('character', name) - - elif key == 'Korhatár': - if value != 'nem korhatáros': - self.story.setMetadata('rating', value) - - elif key == 'Figyelmeztetések': - for b_tag in cells[index+1]('b'): - self.story.addToList('warnings', b_tag.string) - - elif key == 'Jellemzõk': - for genre in cells[index+1].string.split(', '): - self.story.addToList('genre', genre) - - elif key == 'Fejezetek': - self.story.setMetadata('numChapters', int(value)) - - elif key == 'Megjelenés': - self.story.setMetadata('datePublished', makeDate(value, self.DATE_FORMAT)) - - elif key == 'Frissítés': - self.story.setMetadata('dateUpdated', makeDate(value, self.DATE_FORMAT)) - - elif key == 'Szavak': - self.story.setMetadata('numWords', value) - - elif key == 'Befejezett': - self.story.setMetadata('status', 'Completed' if value == 'Nem' else 'In-Progress') - - index += 2 - - if self.story.getMetadata('rating') == '18': - if not (self.is_adult or self.getConfig('is_adult')): - raise exceptions.AdultCheckRequired(self.url) - - def getChapterText(self, url): - soup = self._customized_fetch_url(url) - story_cell = soup.find('form', action='viewstory.php').parent.parent - - for div in story_cell('div'): - div.extract() - - return self.utf8FromSoup(url, story_cell) +# coding=utf-8 + +import re +import urllib2 +import urlparse + +from .. import BeautifulSoup + +from base_adapter import BaseSiteAdapter, makeDate +from .. import exceptions + + +_SOURCE_CODE_ENCODING = 'utf-8' + + +def getClass(): + return FanficHuAdapter + + +def _get_query_data(url): + components = urlparse.urlparse(url) + query_data = urlparse.parse_qs(components.query) + return dict((key, data[0]) for key, data in query_data.items()) + + +class FanficHuAdapter(BaseSiteAdapter): + SITE_ABBREVIATION = 'ffh' + SITE_DOMAIN = 'fanfic.hu' + SITE_LANGUAGE = 'Hungarian' + + BASE_URL = 'http://' + SITE_DOMAIN + '/merengo/' + VIEW_STORY_URL_TEMPLATE = BASE_URL + 'viewstory.php?sid=%s' + + DATE_FORMAT = '%m/%d/%Y' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + query_data = urlparse.parse_qs(self.parsedUrl.query) + story_id = query_data['sid'][0] + + self.story.setMetadata('storyId', story_id) + self._setURL(self.VIEW_STORY_URL_TEMPLATE % story_id) + self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) + self.story.setMetadata('language', self.SITE_LANGUAGE) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return BeautifulSoup.BeautifulSoup(data) + + @staticmethod + def getSiteDomain(): + return FanficHuAdapter.SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls.VIEW_STORY_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return re.escape(self.VIEW_STORY_URL_TEMPLATE[:-2]) + r'\d+$' + + def extractChapterUrlsAndMetadata(self): + soup = self._customized_fetch_url(self.url + '&i=1') + + if soup.title.string.encode(_SOURCE_CODE_ENCODING).strip(' :') == 'írta': + raise exceptions.StoryDoesNotExist(self.url) + + chapter_options = soup.find('form', action='viewstory.php').select('option') + # Remove redundant "Fejezetek" option + chapter_options.pop(0) + + # If there is still more than one entry remove chapter overview entry + if len(chapter_options) > 1: + chapter_options.pop(0) + + for option in chapter_options: + url = urlparse.urljoin(self.url, option['value']) + self.chapterUrls.append((option.string, url)) + + author_url = urlparse.urljoin(self.BASE_URL, soup.find('a', href=lambda href: href and href.startswith('viewuser.php?uid='))['href']) + soup = self._customized_fetch_url(author_url) + + story_id = self.story.getMetadata('storyId') + for table in soup('table', {'class': 'mainnav'}): + title_anchor = table.find('span', {'class': 'storytitle'}).a + href = title_anchor['href'] + if href.startswith('javascript:'): + href = href.rsplit(' ', 1)[1].strip("'") + query_data = _get_query_data(href) + + if query_data['sid'] == story_id: + break + else: + # This should never happen, the story must be found on the author's + # page. + raise exceptions.FailedToDownload(self.url) + + self.story.setMetadata('title', title_anchor.string) + + rows = table('tr') + + anchors = rows[0].div('a') + author_anchor = anchors[1] + query_data = _get_query_data(author_anchor['href']) + self.story.setMetadata('author', author_anchor.string) + self.story.setMetadata('authorId', query_data['uid']) + self.story.setMetadata('authorUrl', urlparse.urljoin(self.BASE_URL, author_anchor['href'])) + self.story.setMetadata('reviews', anchors[3].string) + + if self.getConfig('keep_summary_html'): + self.story.setMetadata('description', self.utf8FromSoup(author_url, rows[1].td)) + else: + self.story.setMetadata('description', ''.join(rows[1].td(text=True))) + + for row in rows[3:]: + index = 0 + cells = row('td') + + while index < len(cells): + cell = cells[index] + key = cell.b.string.encode(_SOURCE_CODE_ENCODING).strip(':') + try: + value = cells[index+1].string.encode(_SOURCE_CODE_ENCODING) + except AttributeError: + value = None + + if key == 'Kategória': + for anchor in cells[index+1]('a'): + self.story.addToList('category', anchor.string) + + elif key == 'Szereplõk': + if cells[index+1].string: + for name in cells[index+1].string.split(', '): + self.story.addToList('character', name) + + elif key == 'Korhatár': + if value != 'nem korhatáros': + self.story.setMetadata('rating', value) + + elif key == 'Figyelmeztetések': + for b_tag in cells[index+1]('b'): + self.story.addToList('warnings', b_tag.string) + + elif key == 'Jellemzõk': + for genre in cells[index+1].string.split(', '): + self.story.addToList('genre', genre) + + elif key == 'Fejezetek': + self.story.setMetadata('numChapters', int(value)) + + elif key == 'Megjelenés': + self.story.setMetadata('datePublished', makeDate(value, self.DATE_FORMAT)) + + elif key == 'Frissítés': + self.story.setMetadata('dateUpdated', makeDate(value, self.DATE_FORMAT)) + + elif key == 'Szavak': + self.story.setMetadata('numWords', value) + + elif key == 'Befejezett': + self.story.setMetadata('status', 'Completed' if value == 'Nem' else 'In-Progress') + + index += 2 + + if self.story.getMetadata('rating') == '18': + if not (self.is_adult or self.getConfig('is_adult')): + raise exceptions.AdultCheckRequired(self.url) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + story_cell = soup.find('form', action='viewstory.php').parent.parent + + for div in story_cell('div'): + div.extract() + + return self.utf8FromSoup(url, story_cell) diff --git a/fff_internals/adapters/adapter_fanfictioncsodaidokhu.py b/fanficfare/adapters/adapter_fanfictioncsodaidokhu.py similarity index 97% rename from fff_internals/adapters/adapter_fanfictioncsodaidokhu.py rename to fanficfare/adapters/adapter_fanfictioncsodaidokhu.py index 76fc051..bade4c9 100644 --- a/fff_internals/adapters/adapter_fanfictioncsodaidokhu.py +++ b/fanficfare/adapters/adapter_fanfictioncsodaidokhu.py @@ -1,218 +1,218 @@ -# coding=utf-8 - -import re -import urllib2 -import urlparse - -from .. import BeautifulSoup - -from base_adapter import BaseSiteAdapter, makeDate -from .. import exceptions - - -_SOURCE_CODE_ENCODING = 'utf-8' - - -def getClass(): - return FanfictionCsodaidokHuAdapter - - -def _get_query_data(url): - components = urlparse.urlparse(url) - query_data = urlparse.parse_qs(components.query) - return dict((key, data[0]) for key, data in query_data.items()) - - -# yields Tag _and_ NavigableString siblings from the given tag. The -# BeautifulSoup findNextSiblings() method for some reasons only returns either -# NavigableStrings _or_ Tag objects, not both. -def _yield_next_siblings(tag): - sibling = tag.nextSibling - while sibling: - yield sibling - sibling = sibling.nextSibling - - -class FanfictionCsodaidokHuAdapter(BaseSiteAdapter): - _SITE_DOMAIN = 'fanfiction.csodaidok.hu' - _BASE_URL = 'http://' + _SITE_DOMAIN + '/' - _VIEW_STORY_URL_TEMPLATE = _BASE_URL + 'viewstory.php?sid=%s' - _VIEW_CHAPTER_URL_TEMPLATE = _VIEW_STORY_URL_TEMPLATE + '&chapter=%s' - - _STORY_DOES_NOT_EXIST_PAGE_TITLE = 'Cím: Szerző:' - _DATE_FORMAT = '%Y.%m.%d' - _SITE_LANGUAGE = 'Hungarian' - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - query_data = urlparse.parse_qs(self.parsedUrl.query) - story_id = query_data['sid'][0] - - self.story.setMetadata('storyId', story_id) - self._setURL(self._VIEW_STORY_URL_TEMPLATE % story_id) - self.story.setMetadata('siteabbrev', self._SITE_DOMAIN) - self.story.setMetadata('language', self._SITE_LANGUAGE) - - def _customized_fetch_url(self, url, exception=None, parameters=None): - if exception: - try: - data = self._fetchUrl(url, parameters) - except urllib2.HTTPError: - raise exception(self.url) - # Just let self._fetchUrl throw the exception, don't catch and - # customize it. - else: - data = self._fetchUrl(url, parameters) - - return BeautifulSoup.BeautifulSoup(data) - - @staticmethod - def getSiteDomain(): - return FanfictionCsodaidokHuAdapter._SITE_DOMAIN - - @classmethod - def getSiteExampleURLs(cls): - return cls._VIEW_STORY_URL_TEMPLATE % 1234 - - def getSiteURLPattern(self): - return re.escape(self._VIEW_STORY_URL_TEMPLATE[:-2]) + r'\d+$' - - def extractChapterUrlsAndMetadata(self): - soup = self._customized_fetch_url(self.url + '&chapter=1') - - element = soup.find('div', id='pagetitle') - page_title = ''.join(element(text=True)).encode(_SOURCE_CODE_ENCODING) - if page_title == self._STORY_DOES_NOT_EXIST_PAGE_TITLE: - raise exceptions.StoryDoesNotExist(self.url) - - author_url = urlparse.urljoin(self.url, element.a['href']) - - story_id = self.story.getMetadata('storyId') - element = soup.find('select', {'name': 'chapter'}) - if element: - for option in element('option'): - title = option.string - url = self._VIEW_CHAPTER_URL_TEMPLATE % (story_id, option['value']) - self.chapterUrls.append((title, url)) - - soup = self._customized_fetch_url(author_url) - story_id = self.story.getMetadata('storyId') - - for listbox_div in soup('div', {'class': lambda klass: klass and 'listbox' in klass}): - a = listbox_div.div.a - if not a['href'].startswith('viewstory.php?sid='): - continue - - query_data = _get_query_data(a['href']) - if query_data['sid'] == story_id: - break - else: - raise exceptions.FailedToDownload(self.url) - - title = ''.join(a(text=True)) - self.story.setMetadata('title', title) - if not self.chapterUrls: - self.chapterUrls.append((title, self.url)) - - element = a.findNextSibling('a') - self.story.setMetadata('author', element.string) - query_data = _get_query_data(element['href']) - self.story.setMetadata('authorId', query_data['uid']) - self.story.setMetadata('authorUrl', author_url) - - element = element.findNextSibling('span') - rating = element.nextSibling.strip(' [') - - if rating.encode(_SOURCE_CODE_ENCODING) != 'Korhatár nélkül': - self.story.setMetadata('rating', rating) - - if rating == '18': - raise exceptions.AdultCheckRequired(self.url) - - element = element.findNextSiblings('a')[1] - self.story.setMetadata('reviews', element.string) - - sections = listbox_div('div', {'class': lambda klass: klass and klass in ['content', 'tail']}) - for section in sections: - for element in section('span', {'class': 'classification'}): - key = element.string.encode(_SOURCE_CODE_ENCODING).strip(' :') - try: - value = element.nextSibling.string.encode(_SOURCE_CODE_ENCODING).strip() - except AttributeError: - value = None - - if key == 'Tartalom': - contents = [] - keep_summary_html = self.getConfig('keep_summary_html') - - for sibling in _yield_next_siblings(element): - if isinstance(sibling, BeautifulSoup.Tag): - if sibling.name == 'span' and sibling.get('class', None) == 'classification': - break - - if keep_summary_html: - contents.append(self.utf8FromSoup(author_url, sibling)) - else: - contents.append(''.join(sibling(text=True))) - else: - contents.append(sibling) - self.story.setMetadata('description', ''.join(contents)) - - elif key == 'Kategória': - for sibling in element.findNextSiblings(['a', 'span']): - if sibling.name == 'span': - break - - self.story.addToList('category', sibling.string) - - elif key == 'Szereplők': - for name in value.split(', '): - self.story.addToList('characters', name) - - elif key == 'Műfaj': - if value != 'Nincs': - self.story.setMetadata('genre', value) - - elif key == 'Figyelmeztetés': - if value != 'Nincs': - for warning in value.split(', '): - self.story.addToList('warnings', warning) - - elif key == 'Kihívás': - if value != 'Nincs': - self.story.setMetadata('challenge', value) - - elif key == 'Sorozat': - if value != 'Nincs': - self.story.setMetadata('series', value) - - elif key == 'Fejezetek': - self.story.setMetadata('numChapters', int(value)) - - elif key == 'Befejezett': - self.story.setMetadata('status', 'Completed' if value == 'Nem' else 'In-Progress') - - elif key == 'Szavak száma': - self.story.setMetadata('numWords', value) - - elif key == 'Feltöltve': - self.story.setMetadata('datePublished', makeDate(value, self._DATE_FORMAT)) - - elif key == 'Frissítve': - self.story.setMetadata('dateUpdated', makeDate(value, self._DATE_FORMAT)) - - def getChapterText(self, url): - soup = self._customized_fetch_url(url) - contents = [] - - notes_div = soup.find('div', id='notes') - if notes_div: - contents.append(self.utf8FromSoup(url, notes_div)) - story_div = notes_div.findNextSibling('div') - else: - element = soup.find('div', {'class': 'jumpmenu'}) - story_div = element.findNextSibling('div') - - contents.append(self.utf8FromSoup(url, story_div.span)) - return ''.join(contents) +# coding=utf-8 + +import re +import urllib2 +import urlparse + +from .. import BeautifulSoup + +from base_adapter import BaseSiteAdapter, makeDate +from .. import exceptions + + +_SOURCE_CODE_ENCODING = 'utf-8' + + +def getClass(): + return FanfictionCsodaidokHuAdapter + + +def _get_query_data(url): + components = urlparse.urlparse(url) + query_data = urlparse.parse_qs(components.query) + return dict((key, data[0]) for key, data in query_data.items()) + + +# yields Tag _and_ NavigableString siblings from the given tag. The +# BeautifulSoup findNextSiblings() method for some reasons only returns either +# NavigableStrings _or_ Tag objects, not both. +def _yield_next_siblings(tag): + sibling = tag.nextSibling + while sibling: + yield sibling + sibling = sibling.nextSibling + + +class FanfictionCsodaidokHuAdapter(BaseSiteAdapter): + _SITE_DOMAIN = 'fanfiction.csodaidok.hu' + _BASE_URL = 'http://' + _SITE_DOMAIN + '/' + _VIEW_STORY_URL_TEMPLATE = _BASE_URL + 'viewstory.php?sid=%s' + _VIEW_CHAPTER_URL_TEMPLATE = _VIEW_STORY_URL_TEMPLATE + '&chapter=%s' + + _STORY_DOES_NOT_EXIST_PAGE_TITLE = 'Cím: Szerző:' + _DATE_FORMAT = '%Y.%m.%d' + _SITE_LANGUAGE = 'Hungarian' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + query_data = urlparse.parse_qs(self.parsedUrl.query) + story_id = query_data['sid'][0] + + self.story.setMetadata('storyId', story_id) + self._setURL(self._VIEW_STORY_URL_TEMPLATE % story_id) + self.story.setMetadata('siteabbrev', self._SITE_DOMAIN) + self.story.setMetadata('language', self._SITE_LANGUAGE) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return BeautifulSoup.BeautifulSoup(data) + + @staticmethod + def getSiteDomain(): + return FanfictionCsodaidokHuAdapter._SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls._VIEW_STORY_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return re.escape(self._VIEW_STORY_URL_TEMPLATE[:-2]) + r'\d+$' + + def extractChapterUrlsAndMetadata(self): + soup = self._customized_fetch_url(self.url + '&chapter=1') + + element = soup.find('div', id='pagetitle') + page_title = ''.join(element(text=True)).encode(_SOURCE_CODE_ENCODING) + if page_title == self._STORY_DOES_NOT_EXIST_PAGE_TITLE: + raise exceptions.StoryDoesNotExist(self.url) + + author_url = urlparse.urljoin(self.url, element.a['href']) + + story_id = self.story.getMetadata('storyId') + element = soup.find('select', {'name': 'chapter'}) + if element: + for option in element('option'): + title = option.string + url = self._VIEW_CHAPTER_URL_TEMPLATE % (story_id, option['value']) + self.chapterUrls.append((title, url)) + + soup = self._customized_fetch_url(author_url) + story_id = self.story.getMetadata('storyId') + + for listbox_div in soup('div', {'class': lambda klass: klass and 'listbox' in klass}): + a = listbox_div.div.a + if not a['href'].startswith('viewstory.php?sid='): + continue + + query_data = _get_query_data(a['href']) + if query_data['sid'] == story_id: + break + else: + raise exceptions.FailedToDownload(self.url) + + title = ''.join(a(text=True)) + self.story.setMetadata('title', title) + if not self.chapterUrls: + self.chapterUrls.append((title, self.url)) + + element = a.findNextSibling('a') + self.story.setMetadata('author', element.string) + query_data = _get_query_data(element['href']) + self.story.setMetadata('authorId', query_data['uid']) + self.story.setMetadata('authorUrl', author_url) + + element = element.findNextSibling('span') + rating = element.nextSibling.strip(' [') + + if rating.encode(_SOURCE_CODE_ENCODING) != 'Korhatár nélkül': + self.story.setMetadata('rating', rating) + + if rating == '18': + raise exceptions.AdultCheckRequired(self.url) + + element = element.findNextSiblings('a')[1] + self.story.setMetadata('reviews', element.string) + + sections = listbox_div('div', {'class': lambda klass: klass and klass in ['content', 'tail']}) + for section in sections: + for element in section('span', {'class': 'classification'}): + key = element.string.encode(_SOURCE_CODE_ENCODING).strip(' :') + try: + value = element.nextSibling.string.encode(_SOURCE_CODE_ENCODING).strip() + except AttributeError: + value = None + + if key == 'Tartalom': + contents = [] + keep_summary_html = self.getConfig('keep_summary_html') + + for sibling in _yield_next_siblings(element): + if isinstance(sibling, BeautifulSoup.Tag): + if sibling.name == 'span' and sibling.get('class', None) == 'classification': + break + + if keep_summary_html: + contents.append(self.utf8FromSoup(author_url, sibling)) + else: + contents.append(''.join(sibling(text=True))) + else: + contents.append(sibling) + self.story.setMetadata('description', ''.join(contents)) + + elif key == 'Kategória': + for sibling in element.findNextSiblings(['a', 'span']): + if sibling.name == 'span': + break + + self.story.addToList('category', sibling.string) + + elif key == 'Szereplők': + for name in value.split(', '): + self.story.addToList('characters', name) + + elif key == 'Műfaj': + if value != 'Nincs': + self.story.setMetadata('genre', value) + + elif key == 'Figyelmeztetés': + if value != 'Nincs': + for warning in value.split(', '): + self.story.addToList('warnings', warning) + + elif key == 'Kihívás': + if value != 'Nincs': + self.story.setMetadata('challenge', value) + + elif key == 'Sorozat': + if value != 'Nincs': + self.story.setMetadata('series', value) + + elif key == 'Fejezetek': + self.story.setMetadata('numChapters', int(value)) + + elif key == 'Befejezett': + self.story.setMetadata('status', 'Completed' if value == 'Nem' else 'In-Progress') + + elif key == 'Szavak száma': + self.story.setMetadata('numWords', value) + + elif key == 'Feltöltve': + self.story.setMetadata('datePublished', makeDate(value, self._DATE_FORMAT)) + + elif key == 'Frissítve': + self.story.setMetadata('dateUpdated', makeDate(value, self._DATE_FORMAT)) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + contents = [] + + notes_div = soup.find('div', id='notes') + if notes_div: + contents.append(self.utf8FromSoup(url, notes_div)) + story_div = notes_div.findNextSibling('div') + else: + element = soup.find('div', {'class': 'jumpmenu'}) + story_div = element.findNextSibling('div') + + contents.append(self.utf8FromSoup(url, story_div.span)) + return ''.join(contents) diff --git a/fff_internals/adapters/adapter_fanfictionjunkiesde.py b/fanficfare/adapters/adapter_fanfictionjunkiesde.py similarity index 97% rename from fff_internals/adapters/adapter_fanfictionjunkiesde.py rename to fanficfare/adapters/adapter_fanfictionjunkiesde.py index 616e143..8991f90 100644 --- a/fff_internals/adapters/adapter_fanfictionjunkiesde.py +++ b/fanficfare/adapters/adapter_fanfictionjunkiesde.py @@ -1,291 +1,291 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# By virtue of being recent and requiring both is_adult and user/pass, -# adapter_fanficcastletvnet.py is the best choice for learning to -# write adapters--especially for sites that use the eFiction system. -# Most sites that have ".../viewstory.php?sid=123" in the story URL -# are eFiction. - -# For non-eFiction sites, it can be considerably more complex, but -# this is still a good starting point. - -# In general an 'adapter' needs to do these five things: - -# - 'Register' correctly with the downloader -# - Site Login (if needed) -# - 'Are you adult?' check (if needed--some do one, some the other, some both) -# - Grab the chapter list -# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) -# - Grab the chapter texts - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return FanfictionJunkiesDeAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class FanfictionJunkiesDeAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/efiction/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ffjde') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'fanfiction-junkies.de' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/efiction/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/efiction/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/efiction/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=1" # XXX - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "For adults only " in data: # XXX - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - pagetitle = soup.find('h4') - ## Title - a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',a.string) - - # Find authorid and URL from... author url. - a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/efiction/'+a['href']) - self.story.setMetadata('author',a.string) - - # Reviews - reviewdata = soup.find('div', {'id' : 'sort'}) - a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one. - self.story.setMetadata('reviews',stripHTML(a)) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/efiction/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - list = soup.find('div', {'class':'listbox'}) - - - labels = list.findAll('b') - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Zusammenfassung' in label: - self.setDescription(url,value) - - if 'Eingestuft' in label: - self.story.setMetadata('rating', value) - - if 'Wörter' in label: - self.story.setMetadata('numWords', value) - - if 'Kategorie' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Charaktere' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Abgeschlossen' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Veröffentlicht' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Aktualisiert' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/efiction/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# By virtue of being recent and requiring both is_adult and user/pass, +# adapter_fanficcastletvnet.py is the best choice for learning to +# write adapters--especially for sites that use the eFiction system. +# Most sites that have ".../viewstory.php?sid=123" in the story URL +# are eFiction. + +# For non-eFiction sites, it can be considerably more complex, but +# this is still a good starting point. + +# In general an 'adapter' needs to do these five things: + +# - 'Register' correctly with the downloader +# - Site Login (if needed) +# - 'Are you adult?' check (if needed--some do one, some the other, some both) +# - Grab the chapter list +# - Grab the story meta-data (some (non-eFiction) adapters have to get it from the author page) +# - Grab the chapter texts + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return FanfictionJunkiesDeAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class FanfictionJunkiesDeAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/efiction/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ffjde') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'fanfiction-junkies.de' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/efiction/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/efiction/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/efiction/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=1" # XXX + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "For adults only " in data: # XXX + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + pagetitle = soup.find('h4') + ## Title + a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',a.string) + + # Find authorid and URL from... author url. + a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/efiction/'+a['href']) + self.story.setMetadata('author',a.string) + + # Reviews + reviewdata = soup.find('div', {'id' : 'sort'}) + a = reviewdata.findAll('a', href=re.compile(r'reviews.php\?type=ST&(amp;)?item='+self.story.getMetadata('storyId')+"$"))[1] # second one. + self.story.setMetadata('reviews',stripHTML(a)) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/efiction/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + list = soup.find('div', {'class':'listbox'}) + + + labels = list.findAll('b') + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Zusammenfassung' in label: + self.setDescription(url,value) + + if 'Eingestuft' in label: + self.story.setMetadata('rating', value) + + if 'Wörter' in label: + self.story.setMetadata('numWords', value) + + if 'Kategorie' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Charaktere' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Abgeschlossen' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Veröffentlicht' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Aktualisiert' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/efiction/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_fanfictionnet.py b/fanficfare/adapters/adapter_fanfictionnet.py similarity index 97% rename from fff_internals/adapters/adapter_fanfictionnet.py rename to fanficfare/adapters/adapter_fanfictionnet.py index baac7c5..df612b1 100644 --- a/fff_internals/adapters/adapter_fanfictionnet.py +++ b/fanficfare/adapters/adapter_fanfictionnet.py @@ -1,341 +1,341 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -from datetime import datetime -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -from urllib import unquote_plus -import time - -from .. import exceptions as exceptions -from ..htmlcleanup import stripHTML - -from base_adapter import BaseSiteAdapter, makeDate - -ffnetgenres=["Adventure", "Angst", "Crime", "Drama", "Family", "Fantasy", "Friendship", "General", - "Horror", "Humor", "Hurt-Comfort", "Mystery", "Parody", "Poetry", "Romance", "Sci-Fi", - "Spiritual", "Supernatural", "Suspense", "Tragedy", "Western"] - -class FanFictionNetSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','ffnet') - - # get storyId from url--url validation guarantees second part is storyId - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2]) - - # normalized story URL. - self._setURL("https://"+self.getSiteDomain()\ - +"/s/"+self.story.getMetadata('storyId')+"/1/") - - # ffnet update emails have the latest chapter URL. - # Frequently, when they arrive, not all the servers have the - # latest chapter yet and going back to chapter 1 to pull the - # chapter list doesn't get the latest. So save and use the - # original URL given to pull chapter list & metadata. - # Not used by plugin because URL gets normalized first for - # eliminating duplicate story urls. - self.origurl = url - if "https://m." in self.origurl: - ## accept m(mobile)url, but use www. - self.origurl = self.origurl.replace("https://m.","https://www.") - - self.opener.addheaders.append(('Referer',self.origurl)) - - @staticmethod - def getSiteDomain(): - return 'www.fanfiction.net' - - @classmethod - def getAcceptDomains(cls): - return ['www.fanfiction.net','m.fanfiction.net'] - - @classmethod - def getSiteExampleURLs(cls): - return "https://www.fanfiction.net/s/1234/1/ https://www.fanfiction.net/s/1234/12/ http://www.fanfiction.net/s/1234/1/Story_Title http://m.fanfiction.net/s/1234/1/" - - def getSiteURLPattern(self): - return r"https?://(www|m)?\.fanfiction\.net/s/\d+(/\d+)?(/|/[^/]+)?/?$" - - def _fetchUrl(self,url,parameters=None,extrasleep=1.0,usecache=True): - ## ffnet(and, I assume, fpcom) tends to fail more if hit too - ## fast. This is in additional to what ever the - ## slow_down_sleep_time setting is. - return BaseSiteAdapter._fetchUrl(self,url, - parameters=parameters, - extrasleep=extrasleep, - usecache=usecache) - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - def doExtractChapterUrlsAndMetadata(self,get_cover=True): - - # fetch the chapter. From that we will get almost all the - # metadata and chapter list - - url = self.origurl - logger.debug("URL: "+url) - - # use BeautifulSoup HTML parser to make everything easier to find. - try: - data = self._fetchUrl(url) - #logger.debug("\n===================\n%s\n===================\n"%data) - soup = self.make_soup(data) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(url) - else: - raise e - - if "Unable to locate story" in data: - raise exceptions.StoryDoesNotExist(url) - - # some times "Chapter not found...", sometimes "Chapter text not found..." - if "not found. Please check to see you are not using an outdated url." in data: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! 'Chapter not found. Please check to see you are not using an outdated url.'" % url) - - if self.getConfig('check_next_chapter'): - try: - ## ffnet used to have a tendency to send out update - ## notices in email before all their servers were - ## showing the update on the first chapter. It - ## generates another server request and doesn't seem - ## to be needed lately, so now default it to off. - try: - chapcount = len(soup.find('select', { 'name' : 'chapter' } ).findAll('option')) - # get chapter part of url. - except: - chapcount = 1 - chapter = url.split('/',)[5] - tryurl = "https://%s/s/%s/%d/"%(self.getSiteDomain(), - self.story.getMetadata('storyId'), - chapcount+1) - logger.debug('=Trying newer chapter: %s' % tryurl) - newdata = self._fetchUrl(tryurl) - if "not found. Please check to see you are not using an outdated url." \ - not in newdata: - logger.debug('=======Found newer chapter: %s' % tryurl) - soup = self.make_soup(newdata) - except: - pass - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"^/u/\d+")) - self.story.setMetadata('authorId',a['href'].split('/')[2]) - self.story.setMetadata('authorUrl','https://'+self.host+a['href']) - self.story.setMetadata('author',a.string) - - ## Pull some additional data from html. - - ## ffnet shows category two ways - ## 1) class(Book, TV, Game,etc) >> category(Harry Potter, Sailor Moon, etc) - ## 2) cat1_cat2_Crossover - ## For 1, use the second link. - ## For 2, fetch the crossover page and pull the two categories from there. - - categories = soup.find('div',{'id':'pre_story_links'}).findAll('a',{'class':'xcontrast_txt'}) - #print("xcontrast_txt a:%s"%categories) - if len(categories) > 1: - # Strangely, the ones with *two* links are the - # non-crossover categories. Each is in a category itself - # of Book, Movie, etc. - self.story.addToList('category',stripHTML(categories[1])) - elif 'Crossover' in categories[0]['href']: - caturl = "https://%s%s"%(self.getSiteDomain(),categories[0]['href']) - catsoup = self.make_soup(self._fetchUrl(caturl)) - for a in catsoup.findAll('a',href=re.compile(r"^/crossovers/.+?/\d+/")): - self.story.addToList('category',stripHTML(a)) - else: - # Fall back. I ran across a story with a Crossver - # category link to a broken page once. - # http://www.fanfiction.net/s/2622060/1/ - # Naruto + Harry Potter Crossover - logger.info("Fall back category collection") - for c in stripHTML(categories[0]).replace(" Crossover","").split(' + '): - self.story.addToList('category',c) - - - - a = soup.find('a', href=re.compile(r'https?://www\.fictionratings\.com/')) - rating = a.string - if 'Fiction' in rating: # if rating has 'Fiction ', strip that out for consistency with past. - rating = rating[8:] - - self.story.setMetadata('rating',rating) - - # after Rating, the same bit of text containing id:123456 contains - # Complete--if completed. - gui_table1i = soup.find('div',{'id':'content_wrapper_inner'}) - - self.story.setMetadata('title', stripHTML(gui_table1i.find('b'))) # title appears to be only(or at least first) bold tag in gui_table1i - - summarydiv = gui_table1i.find('div',{'style':'margin-top:2px'}) - if summarydiv: - self.setDescription(url,stripHTML(summarydiv)) - - - grayspan = gui_table1i.find('span', {'class':'xgray xcontrast_txt'}) - # for b in grayspan.findAll('button'): - # b.extract() - metatext = stripHTML(grayspan).replace('Hurt/Comfort','Hurt-Comfort') - #logger.debug("metatext:(%s)"%metatext) - metalist = metatext.split(" - ") - #logger.debug("metalist:(%s)"%metalist) - - # Rated: Fiction K - English - Words: 158,078 - Published: 02-04-11 - # Rated: Fiction T - English - Adventure/Sci-Fi - Naruto U. - Chapters: 22 - Words: 114,414 - Reviews: 395 - Favs: 779 - Follows: 835 - Updated: 03-21-13 - Published: 04-28-12 - id: 8067258 - - # rating is obtained above more robustly. - if metalist[0].startswith('Rated:'): - metalist=metalist[1:] - - # next is assumed to be language. - self.story.setMetadata('language',metalist[0]) - metalist=metalist[1:] - - # next might be genre. - genrelist = metalist[0].split('/') # Hurt/Comfort already changed above. - goodgenres=True - for g in genrelist: - #logger.debug("g:(%s)"%g) - if g.strip() not in ffnetgenres: - #logger.info("g not in ffnetgenres") - goodgenres=False - if goodgenres: - self.story.extendList('genre',genrelist) - metalist=metalist[1:] - - # Updated: 5/8 - Published: 7/12/2010 - # Published: 8m ago - dates = soup.findAll('span',{'data-xutime':re.compile(r'^\d+$')}) - if len(dates) > 1 : - # updated get set to the same as published upstream if not found. - self.story.setMetadata('dateUpdated',datetime.fromtimestamp(float(dates[0]['data-xutime']))) - self.story.setMetadata('datePublished',datetime.fromtimestamp(float(dates[-1]['data-xutime']))) - - donechars = False - while len(metalist) > 0: - if metalist[0].startswith('Chapters') or metalist[0].startswith('Status') or metalist[0].startswith('id:') or metalist[0].startswith('Updated:') or metalist[0].startswith('Published:'): - pass - elif metalist[0].startswith('Reviews'): - self.story.setMetadata('reviews',metalist[0].split(':')[1].strip()) - elif metalist[0].startswith('Favs:'): - self.story.setMetadata('favs',metalist[0].split(':')[1].strip()) - elif metalist[0].startswith('Follows:'): - self.story.setMetadata('follows',metalist[0].split(':')[1].strip()) - elif metalist[0].startswith('Words'): - self.story.setMetadata('numWords',metalist[0].split(':')[1].strip()) - elif not donechars: - # with 'pairing' support, pairings are bracketed w/o comma after - # [Caspian X, Lucy Pevensie] Edmund Pevensie, Peter Pevensie - self.story.extendList('characters',metalist[0].replace('[','').replace(']',',').split(',')) - - l = metalist[0] - while '[' in l: - self.story.addToList('ships',l[l.index('[')+1:l.index(']')].replace(', ','/')) - l = l[l.index(']')+1:] - - donechars = True - metalist=metalist[1:] - - if 'Status: Complete' in metatext: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if get_cover: - # Try the larger image first. - try: - img = soup.find('img',{'class':'lazy cimage'}) - self.setCoverImage(url,img['data-original']) - except: - img = soup.find('img',{'class':'cimage'}) - if img: - self.setCoverImage(url,img['src']) - - # Find the chapter selector - select = soup.find('select', { 'name' : 'chapter' } ) - - if select is None: - # no selector found, so it's a one-chapter story. - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - allOptions = select.findAll('option') - for o in allOptions: - url = u'https://%s/s/%s/%s/' % ( self.getSiteDomain(), - self.story.getMetadata('storyId'), - o['value']) - # just in case there's tags, like in chapter titles. - title = u"%s" % o - title = re.sub(r'<[^>]+>','',title) - self.chapterUrls.append((title,url)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - return - - def getChapterText(self, url): - # time.sleep(4.0) ## ffnet(and, I assume, fpcom) tends to fail - # ## more if hit too fast. This is in - # ## additional to what ever the - # ## slow_down_sleep_time setting is. - logger.debug('Getting chapter text from: %s' % url) - data = self._fetchUrl(url,extrasleep=4.0) - - if "Please email this error message in full to support@fanfiction.com" in data: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! FanFiction.net Site Error!" % url) - - # some ancient stories have body tags inside them that cause - # soup parsing to discard the content. For story text we - # don't care about anything before "
> category(Harry Potter, Sailor Moon, etc) + ## 2) cat1_cat2_Crossover + ## For 1, use the second link. + ## For 2, fetch the crossover page and pull the two categories from there. + + categories = soup.find('div',{'id':'pre_story_links'}).findAll('a',{'class':'xcontrast_txt'}) + #print("xcontrast_txt a:%s"%categories) + if len(categories) > 1: + # Strangely, the ones with *two* links are the + # non-crossover categories. Each is in a category itself + # of Book, Movie, etc. + self.story.addToList('category',stripHTML(categories[1])) + elif 'Crossover' in categories[0]['href']: + caturl = "https://%s%s"%(self.getSiteDomain(),categories[0]['href']) + catsoup = self.make_soup(self._fetchUrl(caturl)) + for a in catsoup.findAll('a',href=re.compile(r"^/crossovers/.+?/\d+/")): + self.story.addToList('category',stripHTML(a)) + else: + # Fall back. I ran across a story with a Crossver + # category link to a broken page once. + # http://www.fanfiction.net/s/2622060/1/ + # Naruto + Harry Potter Crossover + logger.info("Fall back category collection") + for c in stripHTML(categories[0]).replace(" Crossover","").split(' + '): + self.story.addToList('category',c) + + + + a = soup.find('a', href=re.compile(r'https?://www\.fictionratings\.com/')) + rating = a.string + if 'Fiction' in rating: # if rating has 'Fiction ', strip that out for consistency with past. + rating = rating[8:] + + self.story.setMetadata('rating',rating) + + # after Rating, the same bit of text containing id:123456 contains + # Complete--if completed. + gui_table1i = soup.find('div',{'id':'content_wrapper_inner'}) + + self.story.setMetadata('title', stripHTML(gui_table1i.find('b'))) # title appears to be only(or at least first) bold tag in gui_table1i + + summarydiv = gui_table1i.find('div',{'style':'margin-top:2px'}) + if summarydiv: + self.setDescription(url,stripHTML(summarydiv)) + + + grayspan = gui_table1i.find('span', {'class':'xgray xcontrast_txt'}) + # for b in grayspan.findAll('button'): + # b.extract() + metatext = stripHTML(grayspan).replace('Hurt/Comfort','Hurt-Comfort') + #logger.debug("metatext:(%s)"%metatext) + metalist = metatext.split(" - ") + #logger.debug("metalist:(%s)"%metalist) + + # Rated: Fiction K - English - Words: 158,078 - Published: 02-04-11 + # Rated: Fiction T - English - Adventure/Sci-Fi - Naruto U. - Chapters: 22 - Words: 114,414 - Reviews: 395 - Favs: 779 - Follows: 835 - Updated: 03-21-13 - Published: 04-28-12 - id: 8067258 + + # rating is obtained above more robustly. + if metalist[0].startswith('Rated:'): + metalist=metalist[1:] + + # next is assumed to be language. + self.story.setMetadata('language',metalist[0]) + metalist=metalist[1:] + + # next might be genre. + genrelist = metalist[0].split('/') # Hurt/Comfort already changed above. + goodgenres=True + for g in genrelist: + #logger.debug("g:(%s)"%g) + if g.strip() not in ffnetgenres: + #logger.info("g not in ffnetgenres") + goodgenres=False + if goodgenres: + self.story.extendList('genre',genrelist) + metalist=metalist[1:] + + # Updated: 5/8 - Published: 7/12/2010 + # Published: 8m ago + dates = soup.findAll('span',{'data-xutime':re.compile(r'^\d+$')}) + if len(dates) > 1 : + # updated get set to the same as published upstream if not found. + self.story.setMetadata('dateUpdated',datetime.fromtimestamp(float(dates[0]['data-xutime']))) + self.story.setMetadata('datePublished',datetime.fromtimestamp(float(dates[-1]['data-xutime']))) + + donechars = False + while len(metalist) > 0: + if metalist[0].startswith('Chapters') or metalist[0].startswith('Status') or metalist[0].startswith('id:') or metalist[0].startswith('Updated:') or metalist[0].startswith('Published:'): + pass + elif metalist[0].startswith('Reviews'): + self.story.setMetadata('reviews',metalist[0].split(':')[1].strip()) + elif metalist[0].startswith('Favs:'): + self.story.setMetadata('favs',metalist[0].split(':')[1].strip()) + elif metalist[0].startswith('Follows:'): + self.story.setMetadata('follows',metalist[0].split(':')[1].strip()) + elif metalist[0].startswith('Words'): + self.story.setMetadata('numWords',metalist[0].split(':')[1].strip()) + elif not donechars: + # with 'pairing' support, pairings are bracketed w/o comma after + # [Caspian X, Lucy Pevensie] Edmund Pevensie, Peter Pevensie + self.story.extendList('characters',metalist[0].replace('[','').replace(']',',').split(',')) + + l = metalist[0] + while '[' in l: + self.story.addToList('ships',l[l.index('[')+1:l.index(']')].replace(', ','/')) + l = l[l.index(']')+1:] + + donechars = True + metalist=metalist[1:] + + if 'Status: Complete' in metatext: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if get_cover: + # Try the larger image first. + try: + img = soup.find('img',{'class':'lazy cimage'}) + self.setCoverImage(url,img['data-original']) + except: + img = soup.find('img',{'class':'cimage'}) + if img: + self.setCoverImage(url,img['src']) + + # Find the chapter selector + select = soup.find('select', { 'name' : 'chapter' } ) + + if select is None: + # no selector found, so it's a one-chapter story. + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + allOptions = select.findAll('option') + for o in allOptions: + url = u'https://%s/s/%s/%s/' % ( self.getSiteDomain(), + self.story.getMetadata('storyId'), + o['value']) + # just in case there's tags, like in chapter titles. + title = u"%s" % o + title = re.sub(r'<[^>]+>','',title) + self.chapterUrls.append((title,url)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + return + + def getChapterText(self, url): + # time.sleep(4.0) ## ffnet(and, I assume, fpcom) tends to fail + # ## more if hit too fast. This is in + # ## additional to what ever the + # ## slow_down_sleep_time setting is. + logger.debug('Getting chapter text from: %s' % url) + data = self._fetchUrl(url,extrasleep=4.0) + + if "Please email this error message in full to support@fanfiction.com" in data: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! FanFiction.net Site Error!" % url) + + # some ancient stories have body tags inside them that cause + # soup parsing to discard the content. For story text we + # don't care about anything before "
[a-zA-Z0-9_]+)/(?P[a-zA-Z0-9_]+)\.html" - - def _postFetchWithIAmOld(self,url): - if self.is_adult or self.getConfig("is_adult"): - params={'iamold':'Yes', - 'action':'ageanswer'} - logger.info("Attempting to get cookie for %s" % url) - ## posting on list doesn't work, but doesn't hurt, either. - data = self._postUrl(url,params) - else: - data = self._fetchUrl(url) - return data - - def extractChapterUrlsAndMetadata(self): - - ## could be either chapter list page or one-shot text page. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._postFetchWithIAmOld(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - chapterdata = data - # If chapter list page, get the first chapter to look for adult check - chapterlinklist = soup.findAll('a',{'class':'chapterlink'}) - if chapterlinklist: - chapterdata = self._postFetchWithIAmOld(chapterlinklist[0]['href']) - - if "Are you over seventeen years old" in chapterdata: - raise exceptions.AdultCheckRequired(self.url) - - if not chapterlinklist: - # no chapter list, chapter URL: change to list link. - # second a tag inside div breadcrumbs - storya = soup.find('div',{'class':'breadcrumbs'}).findAll('a')[1] - self._setURL(storya['href']) - url=self.url - logger.debug("Normalizing to URL: "+url) - ## title's right there... - self.story.setMetadata('title',stripHTML(storya)) - data = self._fetchUrl(url) - soup = bs.BeautifulSoup(data) - chapterlinklist = soup.findAll('a',{'class':'chapterlink'}) - else: - ## still need title from somewhere. If chapterlinklist, - ## then chapterdata contains a chapter, find title the - ## same way. - chapsoup = bs.BeautifulSoup(chapterdata) - storya = chapsoup.find('div',{'class':'breadcrumbs'}).findAll('a')[1] - self.story.setMetadata('title',stripHTML(storya)) - del chapsoup - - del chapterdata - - ## authorid already set. - ##

Just Off The Platform II by DrT

- authora=soup.find('h1',{'class':'title'}).find('a') - self.story.setMetadata('author',authora.string) - self.story.setMetadata('authorUrl',authora['href']) - - if len(chapterlinklist) == 1: - self.chapterUrls.append((self.story.getMetadata('title'),chapterlinklist[0]['href'])) - else: - # Find the chapters: - for chapter in chapterlinklist: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - ## Go scrape the rest of the metadata from the author's page. - data = self._fetchUrl(self.story.getMetadata('authorUrl')) - soup = bs.BeautifulSoup(data) - - #
- # [Rid] The Magical Hottiez by Aafro Man Ziegod
- #
- # Chaos ensues after Witch Weekly, seeking to increase readers, decides to create a boyband out of five seemingly talentless wizards: Harry Potter, Draco Malfoy, Ron Weasley, Neville Longbottom, and Oliver "Toss Your Knickers Here" Wood.
- # - #
- - storya = soup.find('a',{'href':self.story.getMetadata('storyUrl')}) - storydd = storya.findNext('dd') - - # Rating: PG - Spoilers: None - 2525 hits - 736 words - # Genre: Humor - Main character(s): H, R - Ships: None - Era: Multiple Eras - # Harry and Ron are back at it again! They reeeeeeally don't want to be back, because they know what's awaiting them. "VH1 Goes Inside..." is back! Why? 'Cos there are soooo many more couples left to pick on. - # Published: September 25, 2004 (between Order of Phoenix and Half-Blood Prince) - Updated: September 25, 2004 - - ## change to text and regexp find. - metastr = stripHTML(storydd).replace('\n',' ').replace('\t',' ') - - m = re.match(r".*?Rating: (.+?) -.*?",metastr) - if m: - self.story.setMetadata('rating', m.group(1)) - - m = re.match(r".*?Genre: (.+?) -.*?",metastr) - if m: - for g in m.group(1).split(','): - self.story.addToList('genre',g) - - m = re.match(r".*?Published: ([a-zA-Z]+ \d\d?, \d\d\d\d).*?",metastr) - if m: - self.story.setMetadata('datePublished',makeDate(m.group(1), "%B %d, %Y")) - - m = re.match(r".*?Updated: ([a-zA-Z]+ \d\d?, \d\d\d\d).*?",metastr) - if m: - self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%B %d, %Y")) - - m = re.match(r".*? (\d+) words Genre.*?",metastr) - if m: - self.story.setMetadata('numWords', m.group(1)) - - for small in storydd.findAll('small'): - small.extract() ## removes the tags, leaving only the summary. - self.setDescription(url,storydd) - #self.story.setMetadata('description',stripHTML(storydd)) - - return - - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - # find & and - # replaced with matching div pair for easier parsing. - # Yes, it's an evil kludge, but what can ya do? Using - # something other than div prevents soup from pairing - # our div with poor html inside the story text. - data = data.replace('','').replace('','') - - # problems with some stories confusing Soup. This is a nasty - # hack, but it works. - data = data[data.index("1: - text = body[1] - text.name='div' # force to be a div to avoid multiple body tags. - else: - text = soup.find('crazytagstringnobodywouldstumbleonaccidently', {'id' : 'storytext'}) - text.name='div' # change to div tag. - - if not data or not text: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - # not sure how, but we can get html, etc tags still in some - # stories. That breaks later updates because it confuses - # epubutils.py - for tag in text.findAll('head'): - tag.extract() - - for tag in text.findAll('body') + text.findAll('html'): - tag.name = 'div' - - return self.utf8FromSoup(url,text) - -def getClass(): - return FictionAlleyOrgSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +class FictionAlleyOrgSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','fa') + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query correct + m = re.match(self.getSiteURLPattern(),url) + if m: + self.story.setMetadata('authorId',m.group('auth')) + self.story.setMetadata('storyId',m.group('id')) + + # normalized story URL. + self._setURL(url) + else: + raise exceptions.InvalidStoryURL(url, + self.getSiteDomain(), + self.getSiteExampleURLs()) + + @staticmethod + def getSiteDomain(): + return 'www.fictionalley.org' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/authors/drt/DA.html http://"+cls.getSiteDomain()+"/authors/drt/JOTP01a.html" + + def getSiteURLPattern(self): + # http://www.fictionalley.org/authors/drt/DA.html + # http://www.fictionalley.org/authors/drt/JOTP01a.html + return re.escape("http://"+self.getSiteDomain())+"/authors/(?P[a-zA-Z0-9_]+)/(?P[a-zA-Z0-9_]+)\.html" + + def _postFetchWithIAmOld(self,url): + if self.is_adult or self.getConfig("is_adult"): + params={'iamold':'Yes', + 'action':'ageanswer'} + logger.info("Attempting to get cookie for %s" % url) + ## posting on list doesn't work, but doesn't hurt, either. + data = self._postUrl(url,params) + else: + data = self._fetchUrl(url) + return data + + def extractChapterUrlsAndMetadata(self): + + ## could be either chapter list page or one-shot text page. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._postFetchWithIAmOld(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + chapterdata = data + # If chapter list page, get the first chapter to look for adult check + chapterlinklist = soup.findAll('a',{'class':'chapterlink'}) + if chapterlinklist: + chapterdata = self._postFetchWithIAmOld(chapterlinklist[0]['href']) + + if "Are you over seventeen years old" in chapterdata: + raise exceptions.AdultCheckRequired(self.url) + + if not chapterlinklist: + # no chapter list, chapter URL: change to list link. + # second a tag inside div breadcrumbs + storya = soup.find('div',{'class':'breadcrumbs'}).findAll('a')[1] + self._setURL(storya['href']) + url=self.url + logger.debug("Normalizing to URL: "+url) + ## title's right there... + self.story.setMetadata('title',stripHTML(storya)) + data = self._fetchUrl(url) + soup = bs.BeautifulSoup(data) + chapterlinklist = soup.findAll('a',{'class':'chapterlink'}) + else: + ## still need title from somewhere. If chapterlinklist, + ## then chapterdata contains a chapter, find title the + ## same way. + chapsoup = bs.BeautifulSoup(chapterdata) + storya = chapsoup.find('div',{'class':'breadcrumbs'}).findAll('a')[1] + self.story.setMetadata('title',stripHTML(storya)) + del chapsoup + + del chapterdata + + ## authorid already set. + ##

Just Off The Platform II by DrT

+ authora=soup.find('h1',{'class':'title'}).find('a') + self.story.setMetadata('author',authora.string) + self.story.setMetadata('authorUrl',authora['href']) + + if len(chapterlinklist) == 1: + self.chapterUrls.append((self.story.getMetadata('title'),chapterlinklist[0]['href'])) + else: + # Find the chapters: + for chapter in chapterlinklist: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + ## Go scrape the rest of the metadata from the author's page. + data = self._fetchUrl(self.story.getMetadata('authorUrl')) + soup = bs.BeautifulSoup(data) + + #
+ # [Rid] The Magical Hottiez by Aafro Man Ziegod
+ #
+ # Chaos ensues after Witch Weekly, seeking to increase readers, decides to create a boyband out of five seemingly talentless wizards: Harry Potter, Draco Malfoy, Ron Weasley, Neville Longbottom, and Oliver "Toss Your Knickers Here" Wood.
+ # + #
+ + storya = soup.find('a',{'href':self.story.getMetadata('storyUrl')}) + storydd = storya.findNext('dd') + + # Rating: PG - Spoilers: None - 2525 hits - 736 words + # Genre: Humor - Main character(s): H, R - Ships: None - Era: Multiple Eras + # Harry and Ron are back at it again! They reeeeeeally don't want to be back, because they know what's awaiting them. "VH1 Goes Inside..." is back! Why? 'Cos there are soooo many more couples left to pick on. + # Published: September 25, 2004 (between Order of Phoenix and Half-Blood Prince) - Updated: September 25, 2004 + + ## change to text and regexp find. + metastr = stripHTML(storydd).replace('\n',' ').replace('\t',' ') + + m = re.match(r".*?Rating: (.+?) -.*?",metastr) + if m: + self.story.setMetadata('rating', m.group(1)) + + m = re.match(r".*?Genre: (.+?) -.*?",metastr) + if m: + for g in m.group(1).split(','): + self.story.addToList('genre',g) + + m = re.match(r".*?Published: ([a-zA-Z]+ \d\d?, \d\d\d\d).*?",metastr) + if m: + self.story.setMetadata('datePublished',makeDate(m.group(1), "%B %d, %Y")) + + m = re.match(r".*?Updated: ([a-zA-Z]+ \d\d?, \d\d\d\d).*?",metastr) + if m: + self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%B %d, %Y")) + + m = re.match(r".*? (\d+) words Genre.*?",metastr) + if m: + self.story.setMetadata('numWords', m.group(1)) + + for small in storydd.findAll('small'): + small.extract() ## removes the tags, leaving only the summary. + self.setDescription(url,storydd) + #self.story.setMetadata('description',stripHTML(storydd)) + + return + + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + # find & and + # replaced with matching div pair for easier parsing. + # Yes, it's an evil kludge, but what can ya do? Using + # something other than div prevents soup from pairing + # our div with poor html inside the story text. + data = data.replace('','').replace('','') + + # problems with some stories confusing Soup. This is a nasty + # hack, but it works. + data = data[data.index("1: + text = body[1] + text.name='div' # force to be a div to avoid multiple body tags. + else: + text = soup.find('crazytagstringnobodywouldstumbleonaccidently', {'id' : 'storytext'}) + text.name='div' # change to div tag. + + if not data or not text: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + # not sure how, but we can get html, etc tags still in some + # stories. That breaks later updates because it confuses + # epubutils.py + for tag in text.findAll('head'): + tag.extract() + + for tag in text.findAll('body') + text.findAll('html'): + tag.name = 'div' + + return self.utf8FromSoup(url,text) + +def getClass(): + return FictionAlleyOrgSiteAdapter + diff --git a/fff_internals/adapters/adapter_fictionmaniatv.py b/fanficfare/adapters/adapter_fictionmaniatv.py similarity index 97% rename from fff_internals/adapters/adapter_fictionmaniatv.py rename to fanficfare/adapters/adapter_fictionmaniatv.py index 7456299..861b44a 100644 --- a/fff_internals/adapters/adapter_fictionmaniatv.py +++ b/fanficfare/adapters/adapter_fictionmaniatv.py @@ -1,166 +1,166 @@ -import re -import urllib2 -import urlparse - -from base_adapter import BaseSiteAdapter, makeDate - - -def getClass(): - return FictionManiaTVAdapter - - -def _get_query_data(url): - components = urlparse.urlparse(url) - query_data = urlparse.parse_qs(components.query) - return dict((key, data[0]) for key, data in query_data.items()) - - -class FictionManiaTVAdapter(BaseSiteAdapter): - SITE_ABBREVIATION = 'fmt' - SITE_DOMAIN = 'fictionmania.tv' - - BASE_URL = 'http://' + SITE_DOMAIN + '/stories/' - READ_TEXT_STORY_URL_TEMPLATE = BASE_URL + 'readtextstory.html?storyID=%s' - DETAILS_URL_TEMPLATE = BASE_URL + 'details.html?storyID=%s' - - DATETIME_FORMAT = '%m/%d/%Y' - ALTERNATIVE_DATETIME_FORMAT = '%m/%d/%y' - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - query_data = urlparse.parse_qs(self.parsedUrl.query) - story_id = query_data['storyID'][0] - - self.story.setMetadata('storyId', story_id) - self._setURL(self.READ_TEXT_STORY_URL_TEMPLATE % story_id) - self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) - - # Always single chapters, probably should use the Anthology feature to - # merge chapters of a story - self.story.setMetadata('numChapters', 1) - - def _customized_fetch_url(self, url, exception=None, parameters=None): - if exception: - try: - data = self._fetchUrl(url, parameters) - except urllib2.HTTPError: - raise exception(self.url) - # Just let self._fetchUrl throw the exception, don't catch and - # customize it. - else: - data = self._fetchUrl(url, parameters) - - return self.make_soup(data) - - @staticmethod - def getSiteDomain(): - return FictionManiaTVAdapter.SITE_DOMAIN - - @classmethod - def getSiteExampleURLs(cls): - return cls.READ_TEXT_STORY_URL_TEMPLATE % 1234 - - def getSiteURLPattern(self): - return 'https?' + re.escape(self.BASE_URL[len('http'):]) + '(readtextstory|readxstory|details)\.html\?storyID=\d+$' - - def extractChapterUrlsAndMetadata(self): - url = self.DETAILS_URL_TEMPLATE % self.story.getMetadata('storyId') - soup = self._customized_fetch_url(url) - - keep_summary_html = self.getConfig('keep_summary_html') - for row in soup.find('table')('tr'): - cells = row('td') - key = cells[0].b.string.strip(':') - try: - value = cells[1].string - except AttributeError: - value = None - - if key == 'Title': - self.story.setMetadata('title', value) - self.chapterUrls.append((value, self.url)) - - elif key == 'File Name': - self.story.setMetadata('fileName', value) - - elif key == 'File Size': - self.story.setMetadata('fileSize', value) - - elif key == 'Author': - element = cells[1].a - self.story.setMetadata('author', element.string) - query_data = _get_query_data(element['href']) - self.story.setMetadata('authorId', query_data['word']) - self.story.setMetadata('authorUrl', urlparse.urljoin(url, element['href'])) - - elif key == 'Date Added': - try: - date = makeDate(value, self.DATETIME_FORMAT) - except ValueError: - date = makeDate(value, self.ALTERNATIVE_DATETIME_FORMAT) - self.story.setMetadata('datePublished', date) - - elif key == 'Old Name': - self.story.setMetadata('oldName', value) - - elif key == 'New Name': - self.story.setMetadata('newName', value) - - elif key == 'Other Names': - for name in value.split(', '): - self.story.addToList('characters', name) - - # I have no clue how the rating system works, if you are reading - # transgender fanfiction, you are probably an adult. - elif key == 'Rating': - self.story.setMetadata('rating', value) - - elif key == 'Complete': - self.story.setMetadata('status', 'Complete' if value == 'Complete' else 'In-Progress') - - elif key == 'Categories': - for element in cells[1]('a'): - self.story.addToList('category', element.string) - - elif key == 'Key Words': - for element in cells[1]('a'): - self.story.addToList('keyWords', element.string) - - elif key == 'Age': - element = cells[1].a - self.story.setMetadata('mainCharactersAge', element.string) - - elif key == 'Synopsis': - element = cells[1] - - # Replace td with div to avoid possible strange formatting in - # the ebook later on - element.name = 'div' - - if keep_summary_html: - self.story.setMetadata('description', unicode(element)) - else: - self.story.setMetadata('description', element.get_text(strip=True)) - - elif key == 'Reads': - self.story.setMetadata('readings', value) - - def getChapterText(self, url): - soup = self._customized_fetch_url(url) - element = soup.find('pre') - element.name = 'div' - - # The story's content is contained in a
 tag, probably taken 1:1
-        # from the source text file. A simple replacement of all newline
-        # characters with a break line tag should take care of formatting.
-
-        # While wrapping in paragraphs would be possible, it's too much work,
-        # I'd rather display the story 1:1 like it was found in the pre tag.
-        content = unicode(element)
-        content = content.replace('\n', '
') - - if self.getConfig('non_breaking_spaces'): - return content.replace(' ', ' ') - - return content +import re +import urllib2 +import urlparse + +from base_adapter import BaseSiteAdapter, makeDate + + +def getClass(): + return FictionManiaTVAdapter + + +def _get_query_data(url): + components = urlparse.urlparse(url) + query_data = urlparse.parse_qs(components.query) + return dict((key, data[0]) for key, data in query_data.items()) + + +class FictionManiaTVAdapter(BaseSiteAdapter): + SITE_ABBREVIATION = 'fmt' + SITE_DOMAIN = 'fictionmania.tv' + + BASE_URL = 'http://' + SITE_DOMAIN + '/stories/' + READ_TEXT_STORY_URL_TEMPLATE = BASE_URL + 'readtextstory.html?storyID=%s' + DETAILS_URL_TEMPLATE = BASE_URL + 'details.html?storyID=%s' + + DATETIME_FORMAT = '%m/%d/%Y' + ALTERNATIVE_DATETIME_FORMAT = '%m/%d/%y' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + query_data = urlparse.parse_qs(self.parsedUrl.query) + story_id = query_data['storyID'][0] + + self.story.setMetadata('storyId', story_id) + self._setURL(self.READ_TEXT_STORY_URL_TEMPLATE % story_id) + self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) + + # Always single chapters, probably should use the Anthology feature to + # merge chapters of a story + self.story.setMetadata('numChapters', 1) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return self.make_soup(data) + + @staticmethod + def getSiteDomain(): + return FictionManiaTVAdapter.SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls.READ_TEXT_STORY_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return 'https?' + re.escape(self.BASE_URL[len('http'):]) + '(readtextstory|readxstory|details)\.html\?storyID=\d+$' + + def extractChapterUrlsAndMetadata(self): + url = self.DETAILS_URL_TEMPLATE % self.story.getMetadata('storyId') + soup = self._customized_fetch_url(url) + + keep_summary_html = self.getConfig('keep_summary_html') + for row in soup.find('table')('tr'): + cells = row('td') + key = cells[0].b.string.strip(':') + try: + value = cells[1].string + except AttributeError: + value = None + + if key == 'Title': + self.story.setMetadata('title', value) + self.chapterUrls.append((value, self.url)) + + elif key == 'File Name': + self.story.setMetadata('fileName', value) + + elif key == 'File Size': + self.story.setMetadata('fileSize', value) + + elif key == 'Author': + element = cells[1].a + self.story.setMetadata('author', element.string) + query_data = _get_query_data(element['href']) + self.story.setMetadata('authorId', query_data['word']) + self.story.setMetadata('authorUrl', urlparse.urljoin(url, element['href'])) + + elif key == 'Date Added': + try: + date = makeDate(value, self.DATETIME_FORMAT) + except ValueError: + date = makeDate(value, self.ALTERNATIVE_DATETIME_FORMAT) + self.story.setMetadata('datePublished', date) + + elif key == 'Old Name': + self.story.setMetadata('oldName', value) + + elif key == 'New Name': + self.story.setMetadata('newName', value) + + elif key == 'Other Names': + for name in value.split(', '): + self.story.addToList('characters', name) + + # I have no clue how the rating system works, if you are reading + # transgender fanfiction, you are probably an adult. + elif key == 'Rating': + self.story.setMetadata('rating', value) + + elif key == 'Complete': + self.story.setMetadata('status', 'Complete' if value == 'Complete' else 'In-Progress') + + elif key == 'Categories': + for element in cells[1]('a'): + self.story.addToList('category', element.string) + + elif key == 'Key Words': + for element in cells[1]('a'): + self.story.addToList('keyWords', element.string) + + elif key == 'Age': + element = cells[1].a + self.story.setMetadata('mainCharactersAge', element.string) + + elif key == 'Synopsis': + element = cells[1] + + # Replace td with div to avoid possible strange formatting in + # the ebook later on + element.name = 'div' + + if keep_summary_html: + self.story.setMetadata('description', unicode(element)) + else: + self.story.setMetadata('description', element.get_text(strip=True)) + + elif key == 'Reads': + self.story.setMetadata('readings', value) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + element = soup.find('pre') + element.name = 'div' + + # The story's content is contained in a
 tag, probably taken 1:1
+        # from the source text file. A simple replacement of all newline
+        # characters with a break line tag should take care of formatting.
+
+        # While wrapping in paragraphs would be possible, it's too much work,
+        # I'd rather display the story 1:1 like it was found in the pre tag.
+        content = unicode(element)
+        content = content.replace('\n', '
') + + if self.getConfig('non_breaking_spaces'): + return content.replace(' ', ' ') + + return content diff --git a/fff_internals/adapters/adapter_fictionpadcom.py b/fanficfare/adapters/adapter_fictionpadcom.py similarity index 97% rename from fff_internals/adapters/adapter_fictionpadcom.py rename to fanficfare/adapters/adapter_fictionpadcom.py index 36f82bf..a890276 100644 --- a/fff_internals/adapters/adapter_fictionpadcom.py +++ b/fanficfare/adapters/adapter_fictionpadcom.py @@ -1,194 +1,194 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import time -import json - -from .. import BeautifulSoup as bs -#from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -class FictionPadSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','fpad') - self.dateformat = "%Y-%m-%dT%H:%M:%SZ" - self.is_adult=False - self.username = None - self.password = None - # get storyId from url--url validation guarantees query correct - m = re.match(self.getSiteURLPattern(),url) - if m: - self.story.setMetadata('storyId',m.group('id')) - - # normalized story URL. - self._setURL("https://"+self.getSiteDomain() - +"/author/"+m.group('author') - +"/stories/"+self.story.getMetadata('storyId')) - else: - raise exceptions.InvalidStoryURL(url, - self.getSiteDomain(), - self.getSiteExampleURLs()) - - @staticmethod - def getSiteDomain(): - return 'fictionpad.com' - - @classmethod - def getSiteExampleURLs(cls): - return "https://fictionpad.com/author/Author/stories/1234/Some-Title" - - def getSiteURLPattern(self): - # http://fictionpad.com/author/Serdd/stories/4275 - return r"http(s)?://(www\.)?fictionpad\.com/author/(?P[^/]+)/stories/(?P\d+)" - -#
-# -# -# or with FictionPad -# -# -# -# -# -#

-# Forgot your password? -#

-#
- def performLogin(self): - params = {} - - if self.password: - params['login'] = self.username - params['password'] = self.password - else: - params['login'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['remember'] = '1' - - loginUrl = 'http://' + self.getSiteDomain() + '/signin' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['login'])) - - ## need to pull empty login page first to get authenticity_token - soup = bs.BeautifulSoup(self._fetchUrl(loginUrl)) - params['authenticity_token']=soup.find('input', {'name':'authenticity_token'})['value'] - - data = self._postUrl(loginUrl, params) - - if "Invalid email/pseudonym and password combination." in data: - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['login'])) - raise exceptions.FailedToLogin(loginUrl,params['login']) - - - def extractChapterUrlsAndMetadata(self): - # fetch the chapter. From that we will get almost all the - # metadata and chapter list - - url=self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - if "This is a mature story. Please sign in to read it." in data: - self.performLogin() - data = self._fetchUrl(url) - - find = "wordyarn.config.page = " - data = data[data.index(find)+len(find):] - data = data[:data.index("")] - data = data[:data.rindex(";")] - data = data.replace('tables:','"tables":') - tables = json.loads(data)['tables'] - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(url) - else: - raise e - - # looks like only one author per story allowed. - author = tables['users'][0] - story = tables['stories'][0] - story_ver = tables['story_versions'][0] - logger.debug("story:%s"%story) - - self.story.setMetadata('authorId',author['id']) - self.story.setMetadata('author',author['display_name']) - self.story.setMetadata('authorUrl','https://'+self.host+'/author/'+author['display_name']+'/stories') - - self.story.setMetadata('title',story_ver['title']) - self.setDescription(url,story_ver['description']) - - if not ('assets/story_versions/covers' in story_ver['profile_image_url@2x']): - self.setCoverImage(url,story_ver['profile_image_url@2x']) - - self.story.setMetadata('datePublished',makeDate(story['published_at'], self.dateformat)) - self.story.setMetadata('dateUpdated',makeDate(story['published_at'], self.dateformat)) - - self.story.setMetadata('followers',story['followers_count']) - self.story.setMetadata('comments',story['comments_count']) - self.story.setMetadata('views',story['views_count']) - self.story.setMetadata('likes',int(story['likes'])) # no idea why they floated these. - if 'dislikes' in story: - self.story.setMetadata('dislikes',int(story['dislikes'])) - - if story_ver['is_complete']: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - self.story.setMetadata('rating', story_ver['maturity_level']) - self.story.setMetadata('numWords', unicode(story_ver['word_count'])) - - for i in tables['fandoms']: - self.story.addToList('category',i['name']) - - for i in tables['genres']: - self.story.addToList('genre',i['name']) - - for i in tables['characters']: - self.story.addToList('characters',i['name']) - - for c in tables['chapters']: - chtitle = "Chapter %d"%c['number'] - if c['title']: - chtitle += " - %s"%c['title'] - self.chapterUrls.append((chtitle,c['body_url'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - def getChapterText(self, url): - logger.debug('Getting chapter text from: %s' % url) - if not url: - data = u"This chapter has no text." - else: - data = self._fetchUrl(url) - soup = bs.BeautifulSoup(u"
"+data+u"
") - return self.utf8FromSoup(url,soup) - -def getClass(): - return FictionPadSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import time +import json + +from .. import BeautifulSoup as bs +#from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +class FictionPadSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','fpad') + self.dateformat = "%Y-%m-%dT%H:%M:%SZ" + self.is_adult=False + self.username = None + self.password = None + # get storyId from url--url validation guarantees query correct + m = re.match(self.getSiteURLPattern(),url) + if m: + self.story.setMetadata('storyId',m.group('id')) + + # normalized story URL. + self._setURL("https://"+self.getSiteDomain() + +"/author/"+m.group('author') + +"/stories/"+self.story.getMetadata('storyId')) + else: + raise exceptions.InvalidStoryURL(url, + self.getSiteDomain(), + self.getSiteExampleURLs()) + + @staticmethod + def getSiteDomain(): + return 'fictionpad.com' + + @classmethod + def getSiteExampleURLs(cls): + return "https://fictionpad.com/author/Author/stories/1234/Some-Title" + + def getSiteURLPattern(self): + # http://fictionpad.com/author/Serdd/stories/4275 + return r"http(s)?://(www\.)?fictionpad\.com/author/(?P[^/]+)/stories/(?P\d+)" + +#
+# +# +# or with FictionPad +# +# +# +# +# +#

+# Forgot your password? +#

+#
+ def performLogin(self): + params = {} + + if self.password: + params['login'] = self.username + params['password'] = self.password + else: + params['login'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['remember'] = '1' + + loginUrl = 'http://' + self.getSiteDomain() + '/signin' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['login'])) + + ## need to pull empty login page first to get authenticity_token + soup = bs.BeautifulSoup(self._fetchUrl(loginUrl)) + params['authenticity_token']=soup.find('input', {'name':'authenticity_token'})['value'] + + data = self._postUrl(loginUrl, params) + + if "Invalid email/pseudonym and password combination." in data: + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['login'])) + raise exceptions.FailedToLogin(loginUrl,params['login']) + + + def extractChapterUrlsAndMetadata(self): + # fetch the chapter. From that we will get almost all the + # metadata and chapter list + + url=self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + if "This is a mature story. Please sign in to read it." in data: + self.performLogin() + data = self._fetchUrl(url) + + find = "wordyarn.config.page = " + data = data[data.index(find)+len(find):] + data = data[:data.index("")] + data = data[:data.rindex(";")] + data = data.replace('tables:','"tables":') + tables = json.loads(data)['tables'] + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(url) + else: + raise e + + # looks like only one author per story allowed. + author = tables['users'][0] + story = tables['stories'][0] + story_ver = tables['story_versions'][0] + logger.debug("story:%s"%story) + + self.story.setMetadata('authorId',author['id']) + self.story.setMetadata('author',author['display_name']) + self.story.setMetadata('authorUrl','https://'+self.host+'/author/'+author['display_name']+'/stories') + + self.story.setMetadata('title',story_ver['title']) + self.setDescription(url,story_ver['description']) + + if not ('assets/story_versions/covers' in story_ver['profile_image_url@2x']): + self.setCoverImage(url,story_ver['profile_image_url@2x']) + + self.story.setMetadata('datePublished',makeDate(story['published_at'], self.dateformat)) + self.story.setMetadata('dateUpdated',makeDate(story['published_at'], self.dateformat)) + + self.story.setMetadata('followers',story['followers_count']) + self.story.setMetadata('comments',story['comments_count']) + self.story.setMetadata('views',story['views_count']) + self.story.setMetadata('likes',int(story['likes'])) # no idea why they floated these. + if 'dislikes' in story: + self.story.setMetadata('dislikes',int(story['dislikes'])) + + if story_ver['is_complete']: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + self.story.setMetadata('rating', story_ver['maturity_level']) + self.story.setMetadata('numWords', unicode(story_ver['word_count'])) + + for i in tables['fandoms']: + self.story.addToList('category',i['name']) + + for i in tables['genres']: + self.story.addToList('genre',i['name']) + + for i in tables['characters']: + self.story.addToList('characters',i['name']) + + for c in tables['chapters']: + chtitle = "Chapter %d"%c['number'] + if c['title']: + chtitle += " - %s"%c['title'] + self.chapterUrls.append((chtitle,c['body_url'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + def getChapterText(self, url): + logger.debug('Getting chapter text from: %s' % url) + if not url: + data = u"This chapter has no text." + else: + data = self._fetchUrl(url) + soup = bs.BeautifulSoup(u"
"+data+u"
") + return self.utf8FromSoup(url,soup) + +def getClass(): + return FictionPadSiteAdapter + diff --git a/fff_internals/adapters/adapter_fictionpresscom.py b/fanficfare/adapters/adapter_fictionpresscom.py similarity index 97% rename from fff_internals/adapters/adapter_fictionpresscom.py rename to fanficfare/adapters/adapter_fictionpresscom.py index 795ff94..e7f973e 100644 --- a/fff_internals/adapters/adapter_fictionpresscom.py +++ b/fanficfare/adapters/adapter_fictionpresscom.py @@ -1,51 +1,51 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import time - -## They're from the same people and pretty much identical. -from adapter_fanfictionnet import FanFictionNetSiteAdapter - -class FictionPressComSiteAdapter(FanFictionNetSiteAdapter): - - def __init__(self, config, url): - FanFictionNetSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','fpcom') - - @staticmethod - def getSiteDomain(): - return 'www.fictionpress.com' - - @classmethod - def getAcceptDomains(cls): - return ['www.fictionpress.com','m.fictionpress.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "https://www.fictionpress.com/s/1234/1/ https://www.fictionpress.com/s/1234/12/ http://www.fictionpress.com/s/1234/1/Story_Title http://m.fictionpress.com/s/1234/1/" - - def getSiteURLPattern(self): - return r"https?://(www|m)?\.fictionpress\.com/s/\d+(/\d+)?(/|/[a-zA-Z0-9_-]+)?/?$" - -def getClass(): - return FictionPressComSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import time + +## They're from the same people and pretty much identical. +from adapter_fanfictionnet import FanFictionNetSiteAdapter + +class FictionPressComSiteAdapter(FanFictionNetSiteAdapter): + + def __init__(self, config, url): + FanFictionNetSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','fpcom') + + @staticmethod + def getSiteDomain(): + return 'www.fictionpress.com' + + @classmethod + def getAcceptDomains(cls): + return ['www.fictionpress.com','m.fictionpress.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "https://www.fictionpress.com/s/1234/1/ https://www.fictionpress.com/s/1234/12/ http://www.fictionpress.com/s/1234/1/Story_Title http://m.fictionpress.com/s/1234/1/" + + def getSiteURLPattern(self): + return r"https?://(www|m)?\.fictionpress\.com/s/\d+(/\d+)?(/|/[a-zA-Z0-9_-]+)?/?$" + +def getClass(): + return FictionPressComSiteAdapter + diff --git a/fff_internals/adapters/adapter_ficwadcom.py b/fanficfare/adapters/adapter_ficwadcom.py similarity index 97% rename from fff_internals/adapters/adapter_ficwadcom.py rename to fanficfare/adapters/adapter_ficwadcom.py index 7934a13..ff0012e 100644 --- a/fff_internals/adapters/adapter_ficwadcom.py +++ b/fanficfare/adapters/adapter_ficwadcom.py @@ -1,233 +1,233 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import time -import httplib, urllib - -from .. import exceptions as exceptions -from ..htmlcleanup import stripHTML - -from base_adapter import BaseSiteAdapter, makeDate - -class FicwadComSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','fw') - - # get storyId from url--url validation guarantees second part is storyId - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2]) - - self.username = "NoneGiven" - self.password = "" - - @staticmethod - def getSiteDomain(): - return 'ficwad.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://ficwad.com/story/1234" - - def getSiteURLPattern(self): - return re.escape(r"http://"+self.getSiteDomain())+"/story/\d+?$" - - def performLogin(self,url): - params = {} - - if self.password: - params['username'] = self.username - params['password'] = self.password - else: - params['username'] = self.getConfig("username") - params['password'] = self.getConfig("password") - - loginUrl = 'http://' + self.getSiteDomain() + '/account/login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['username'])) - d = self._postUrl(loginUrl,params,usecache=False) - - if "Login attempt failed..." in d: - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['username'])) - raise exceptions.FailedToLogin(url,params['username']) - return False - else: - return True - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - def extractChapterUrlsAndMetadata(self): - - # fetch the chapter. From that we will get almost all the - # metadata and chapter list - - url = self.url - logger.debug("URL: "+url) - - # use BeautifulSoup HTML parser to make everything easier to find. - try: - data = self._fetchUrl(url) - # non-existent/removed story urls get thrown to the front page. - if "

Welcome to FicWad

" in data: - raise exceptions.StoryDoesNotExist(self.url) - soup = self.make_soup(data) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # if blocked, attempt login. - if soup.find("div",{"class":"blocked"}) or soup.find("li",{"class":"blocked"}): - if self.performLogin(url): # performLogin raises - # FailedToLogin if it fails. - soup = self.make_soup(self._fetchUrl(url,usecache=False)) - - divstory = soup.find('div',id='story') - storya = divstory.find('a',href=re.compile("^/story/\d+$")) - if storya : # if there's a story link in the divstory header, this is a chapter page. - # normalize story URL on chapter list. - self.story.setMetadata('storyId',storya['href'].split('/',)[2]) - url = "http://"+self.getSiteDomain()+storya['href'] - logger.debug("Normalizing to URL: "+url) - self._setURL(url) - try: - soup = self.make_soup(self._fetchUrl(url)) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # if blocked, attempt login. - if soup.find("div",{"class":"blocked"}) or soup.find("li",{"class":"blocked"}): - if self.performLogin(url): # performLogin raises - # FailedToLogin if it fails. - soup = self.make_soup(self._fetchUrl(url,usecache=False)) - - # title - first h4 tag will be title. - titleh4 = soup.find('div',{'class':'storylist'}).find('h4') - self.story.setMetadata('title', stripHTML(titleh4.a)) - - # Find authorid and URL from... author url. - a = soup.find('span',{'class':'author'}).find('a', href=re.compile(r"^/author/\d+")) - self.story.setMetadata('authorId',a['href'].split('/')[2]) - self.story.setMetadata('authorUrl','http://'+self.host+a['href']) - self.story.setMetadata('author',a.string) - - # description - storydiv = soup.find("div",{"id":"story"}) - self.setDescription(url,storydiv.find("blockquote",{'class':'summary'}).p) - #self.story.setMetadata('description', storydiv.find("blockquote",{'class':'summary'}).p.string) - - # most of the meta data is here: - metap = storydiv.find("p",{"class":"meta"}) - self.story.addToList('category',metap.find("a",href=re.compile(r"^/category/\d+")).string) - - # warnings - # [!!] [R] [V] [Y] - spanreq = metap.find("span",{"class":"story-warnings"}) - if spanreq: # can be no warnings. - for a in spanreq.findAll("a"): - self.story.addToList('warnings',a['title']) - - ## perhaps not the most efficient way to parse this, using - ## regexps for each rather than something more complex, but - ## IMO, it's more readable and amenable to change. - metastr = stripHTML(str(metap)).replace('\n',' ').replace('\t',' ').replace(u'\u00a0',' ') - - m = re.match(r".*?Rating: (.+?) -.*?",metastr) - if m: - self.story.setMetadata('rating', m.group(1)) - - m = re.match(r".*?Genres: (.+?) -.*?",metastr) - if m: - for g in m.group(1).split(','): - self.story.addToList('genre',g) - - m = re.match(r".*?Characters: (.*?) -.*?",metastr) - if m: - for g in m.group(1).split(','): - if g: - self.story.addToList('characters',g) - - m = re.match(r".*?Published: ([0-9-]+?) -.*?",metastr) - if m: - self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y-%m-%d")) - - # Updated can have more than one space after it. - m = re.match(r".*?Updated: ([0-9-]+?) +-.*?",metastr) - if m: - self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y-%m-%d")) - - m = re.match(r".*? - ([0-9,]+?) words.*?",metastr) - if m: - self.story.setMetadata('numWords',m.group(1)) - - if metastr.endswith("Complete"): - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - # get the chapter list first this time because that's how we - # detect the need to login. - storylistul = soup.find('ul',{'class':'storylist'}) - if not storylistul: - # no list found, so it's a one-chapter story. - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - chapterlistlis = storylistul.findAll('li') - for chapterli in chapterlistlis: - if "blocked" in chapterli['class']: - # paranoia check. We should already be logged in by now. - raise exceptions.FailedToLogin(url,self.username) - else: - #print "chapterli.h4.a (%s)"%chapterli.h4.a - self.chapterUrls.append((chapterli.h4.a.string, - u'http://%s%s'%(self.getSiteDomain(), - chapterli.h4.a['href']))) - #print "self.chapterUrls:%s"%self.chapterUrls - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - return - - - def getChapterText(self, url): - logger.debug('Getting chapter text from: %s' % url) - soup = self.make_soup(self._fetchUrl(url)) - - span = soup.find('div', {'id' : 'storytext'}) - - if None == span: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,span) - -def getClass(): - return FicwadComSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import time +import httplib, urllib + +from .. import exceptions as exceptions +from ..htmlcleanup import stripHTML + +from base_adapter import BaseSiteAdapter, makeDate + +class FicwadComSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','fw') + + # get storyId from url--url validation guarantees second part is storyId + self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2]) + + self.username = "NoneGiven" + self.password = "" + + @staticmethod + def getSiteDomain(): + return 'ficwad.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://ficwad.com/story/1234" + + def getSiteURLPattern(self): + return re.escape(r"http://"+self.getSiteDomain())+"/story/\d+?$" + + def performLogin(self,url): + params = {} + + if self.password: + params['username'] = self.username + params['password'] = self.password + else: + params['username'] = self.getConfig("username") + params['password'] = self.getConfig("password") + + loginUrl = 'http://' + self.getSiteDomain() + '/account/login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['username'])) + d = self._postUrl(loginUrl,params,usecache=False) + + if "Login attempt failed..." in d: + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['username'])) + raise exceptions.FailedToLogin(url,params['username']) + return False + else: + return True + + def use_pagecache(self): + ''' + adapters that will work with the page cache need to implement + this and change it to True. + ''' + return True + + def extractChapterUrlsAndMetadata(self): + + # fetch the chapter. From that we will get almost all the + # metadata and chapter list + + url = self.url + logger.debug("URL: "+url) + + # use BeautifulSoup HTML parser to make everything easier to find. + try: + data = self._fetchUrl(url) + # non-existent/removed story urls get thrown to the front page. + if "

Welcome to FicWad

" in data: + raise exceptions.StoryDoesNotExist(self.url) + soup = self.make_soup(data) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # if blocked, attempt login. + if soup.find("div",{"class":"blocked"}) or soup.find("li",{"class":"blocked"}): + if self.performLogin(url): # performLogin raises + # FailedToLogin if it fails. + soup = self.make_soup(self._fetchUrl(url,usecache=False)) + + divstory = soup.find('div',id='story') + storya = divstory.find('a',href=re.compile("^/story/\d+$")) + if storya : # if there's a story link in the divstory header, this is a chapter page. + # normalize story URL on chapter list. + self.story.setMetadata('storyId',storya['href'].split('/',)[2]) + url = "http://"+self.getSiteDomain()+storya['href'] + logger.debug("Normalizing to URL: "+url) + self._setURL(url) + try: + soup = self.make_soup(self._fetchUrl(url)) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # if blocked, attempt login. + if soup.find("div",{"class":"blocked"}) or soup.find("li",{"class":"blocked"}): + if self.performLogin(url): # performLogin raises + # FailedToLogin if it fails. + soup = self.make_soup(self._fetchUrl(url,usecache=False)) + + # title - first h4 tag will be title. + titleh4 = soup.find('div',{'class':'storylist'}).find('h4') + self.story.setMetadata('title', stripHTML(titleh4.a)) + + # Find authorid and URL from... author url. + a = soup.find('span',{'class':'author'}).find('a', href=re.compile(r"^/author/\d+")) + self.story.setMetadata('authorId',a['href'].split('/')[2]) + self.story.setMetadata('authorUrl','http://'+self.host+a['href']) + self.story.setMetadata('author',a.string) + + # description + storydiv = soup.find("div",{"id":"story"}) + self.setDescription(url,storydiv.find("blockquote",{'class':'summary'}).p) + #self.story.setMetadata('description', storydiv.find("blockquote",{'class':'summary'}).p.string) + + # most of the meta data is here: + metap = storydiv.find("p",{"class":"meta"}) + self.story.addToList('category',metap.find("a",href=re.compile(r"^/category/\d+")).string) + + # warnings + # [!!] [R] [V] [Y] + spanreq = metap.find("span",{"class":"story-warnings"}) + if spanreq: # can be no warnings. + for a in spanreq.findAll("a"): + self.story.addToList('warnings',a['title']) + + ## perhaps not the most efficient way to parse this, using + ## regexps for each rather than something more complex, but + ## IMO, it's more readable and amenable to change. + metastr = stripHTML(str(metap)).replace('\n',' ').replace('\t',' ').replace(u'\u00a0',' ') + + m = re.match(r".*?Rating: (.+?) -.*?",metastr) + if m: + self.story.setMetadata('rating', m.group(1)) + + m = re.match(r".*?Genres: (.+?) -.*?",metastr) + if m: + for g in m.group(1).split(','): + self.story.addToList('genre',g) + + m = re.match(r".*?Characters: (.*?) -.*?",metastr) + if m: + for g in m.group(1).split(','): + if g: + self.story.addToList('characters',g) + + m = re.match(r".*?Published: ([0-9-]+?) -.*?",metastr) + if m: + self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y-%m-%d")) + + # Updated can have more than one space after it. + m = re.match(r".*?Updated: ([0-9-]+?) +-.*?",metastr) + if m: + self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y-%m-%d")) + + m = re.match(r".*? - ([0-9,]+?) words.*?",metastr) + if m: + self.story.setMetadata('numWords',m.group(1)) + + if metastr.endswith("Complete"): + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + # get the chapter list first this time because that's how we + # detect the need to login. + storylistul = soup.find('ul',{'class':'storylist'}) + if not storylistul: + # no list found, so it's a one-chapter story. + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + chapterlistlis = storylistul.findAll('li') + for chapterli in chapterlistlis: + if "blocked" in chapterli['class']: + # paranoia check. We should already be logged in by now. + raise exceptions.FailedToLogin(url,self.username) + else: + #print "chapterli.h4.a (%s)"%chapterli.h4.a + self.chapterUrls.append((chapterli.h4.a.string, + u'http://%s%s'%(self.getSiteDomain(), + chapterli.h4.a['href']))) + #print "self.chapterUrls:%s"%self.chapterUrls + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + return + + + def getChapterText(self, url): + logger.debug('Getting chapter text from: %s' % url) + soup = self.make_soup(self._fetchUrl(url)) + + span = soup.find('div', {'id' : 'storytext'}) + + if None == span: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,span) + +def getClass(): + return FicwadComSiteAdapter + diff --git a/fff_internals/adapters/adapter_fimfictionnet.py b/fanficfare/adapters/adapter_fimfictionnet.py similarity index 98% rename from fff_internals/adapters/adapter_fimfictionnet.py rename to fanficfare/adapters/adapter_fimfictionnet.py index 469e931..74c29c5 100644 --- a/fff_internals/adapters/adapter_fimfictionnet.py +++ b/fanficfare/adapters/adapter_fimfictionnet.py @@ -1,357 +1,357 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -from datetime import date -from datetime import timedelta -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import cookielib as cl -import json - -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return FimFictionNetSiteAdapter - -class FimFictionNetSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','fimficnet') - self.story.setMetadata('storyId', self.parsedUrl.path.split('/',)[2]) - self._setURL("http://"+self.getSiteDomain()+"/story/"+self.story.getMetadata('storyId')+"/") - self.is_adult = False - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d %b %Y" - - @staticmethod - def getSiteDomain(): - return 'www.fimfiction.net' - - @classmethod - def getAcceptDomains(cls): - # mobile.fimifction.com isn't actually a valid domain, but we can still get the story id from URLs anyway - return ['www.fimfiction.net','mobile.fimfiction.net', 'www.fimfiction.com', 'mobile.fimfiction.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://www.fimfiction.net/story/1234/story-title-here http://www.fimfiction.net/story/1234/ http://www.fimfiction.com/story/1234/1/ http://mobile.fimfiction.net/story/1234/1/story-title-here/chapter-title-here" - - def getSiteURLPattern(self): - return r"https?://(www|mobile)\.fimfiction\.(net|com)/story/\d+/?.*" - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - def doExtractChapterUrlsAndMetadata(self,get_cover=True): - - if self.is_adult or self.getConfig("is_adult"): - cookie = cl.Cookie(version=0, name='view_mature', value='true', - port=None, port_specified=False, - domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, - path='/story', path_specified=True, - secure=False, - expires=time.time()+10000, - discard=False, - comment=None, - comment_url=None, - rest={'HttpOnly': None}, - rfc2109=False) - self.cookiejar.set_cookie(cookie) - - ##--------------------------------------------------------------------------------------------------- - ## Get the story's title page. Check if it exists. - - try: - # don't use cache if manual is_adult--should only happen - # if it's an adult story and they don't have is_adult in ini. - data = self.do_fix_blockquotes(self._fetchUrl(self.url, - usecache=(not self.is_adult))) - soup = self.make_soup(data) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Warning: mysql_fetch_array(): supplied argument is not a valid MySQL result resource" in data: - raise exceptions.StoryDoesNotExist(self.url) - - if "This story has been marked as having adult content. Please click below to confirm you are of legal age to view adult material in your country." in data: - raise exceptions.AdultCheckRequired(self.url) - - if self.password: - params = {} - params['password'] = self.password - data = self._postUrl(self.url, params) - soup = self.make_soup(data) - - if not (soup.find('form', {'id' : 'password_form'}) == None): - if self.getConfig('fail_on_password'): - raise exceptions.FailedToDownload("%s requires story password and fail_on_password is true."%self.url) - else: - raise exceptions.FailedToLogin(self.url,"Story requires individual password",passwdonly=True) - - ##---------------------------------------------------------------------------------------------------- - ## Extract metadata - - storyContentBox = soup.find('div', {'class':'story_content_box'}) - - # Title - title = storyContentBox.find('a', {'class':re.compile(r'.*\bstory_name\b.*')}) - self.story.setMetadata('title',stripHTML(title)) - - # Author - author = storyContentBox.find('div', {'class':'author'}).find('a') - self.story.setMetadata("author", stripHTML(author)) - #No longer seems to be a way to access Fimfiction's internal author ID - self.story.setMetadata("authorId", self.story.getMetadata("author")) - self.story.setMetadata("authorUrl", "http://%s/user/%s" % (self.getSiteDomain(), stripHTML(author))) - - #Rating text is replaced with full words for historical compatibility after the site changed - #on 2014-10-27 - rating = stripHTML(storyContentBox.find('a', {'class':re.compile(r'.*\bcontent-rating-.*')})) - rating = rating.replace("E", "Everyone").replace("T", "Teen").replace("M", "Mature") - self.story.setMetadata("rating", rating) - - # Chapters - for chapter in storyContentBox.find_all('a',{'class':'chapter_link'}): - self.chapterUrls.append((stripHTML(chapter), 'http://'+self.host+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # Status - # In the case of Fimfiction, possible statuses are 'Completed', 'Incomplete', 'On Hiatus' and 'Cancelled' - # For the sake of bringing it in line with the other adapters, 'Incomplete' becomes 'In-Progress' - # and 'Complete' becomes 'Completed'. 'Cancelled' and 'On Hiatus' are passed through, it's easy now for users - # to change/remove if they want with replace_metadata - status = stripHTML(storyContentBox.find('span', {'class':re.compile(r'.*\bcompleted-status-.*')})) - status = status.replace("Incomplete", "In-Progress").replace("Complete", "Completed") - self.story.setMetadata("status", status) - - # Genres and Warnings - # warnings were folded into general categories in the 2014-10-27 site update - categories = storyContentBox.find_all('a', {'class':re.compile(r'.*\bstory_category\b.*')}) - for category in categories: - category = stripHTML(category) - if category == "Gore" or category == "Sex": - self.story.addToList('warnings', category) - else: - self.story.addToList('genre', category) - - # Word count - wordCountText = stripHTML(storyContentBox.find('li', {'class':'bottom'}).find('div', {'class':'word_count'})) - self.story.setMetadata("numWords", re.sub(r'[^0-9]', '', wordCountText)) - - # Cover image - storyImage = storyContentBox.find('div', {'class':'story_image'}) - if storyImage: - coverurl = storyImage.find('a')['href'] - if coverurl.startswith('//'): # fix for img urls missing 'http:' - coverurl = "http:"+coverurl - if get_cover: - self.setCoverImage(self.url,coverurl) - - coverSource = storyImage.find('a', {'class':'source'}) - if coverSource: - self.story.setMetadata('coverSourceUrl', coverSource['href']) - #There's no text associated with the cover source link, so just - #reuse the URL. Makes it clear it's an external link leading - #outside of the fanfic site, at least. - self.story.setMetadata('coverSource', coverSource['href']) - - # fimf has started including extra stuff inside the description div. - descdivstr = u"%s"%storyContentBox.find("div", {"class":"description"}) - hrstr=u"
" - descdivstr = u'
'+descdivstr[descdivstr.index(hrstr)+len(hrstr):] - self.setDescription(self.url,descdivstr) - - # Find the newest and oldest chapter dates - storyData = storyContentBox.find('div', {'class':'story_data'}) - oldestChapter = None - newestChapter = None - self.newestChapterNum = None # save for comparing during update. - # Scan all chapters to find the oldest and newest, on - # FiMFiction it's possible for authors to insert new chapters - # out-of-order or change the dates of earlier ones by editing - # them--That WILL break epub update. - for index, chapterDate in enumerate(storyData.find_all('span', {'class':'date'})): - chapterDate = self.ordinal_date_string_to_date(chapterDate.contents[1]) - if oldestChapter == None or chapterDate < oldestChapter: - oldestChapter = chapterDate - if newestChapter == None or chapterDate > newestChapter: - newestChapter = chapterDate - self.newestChapterNum = index - - if newestChapter is None: - #this will only be true when updating metadata for stories that have 0 chapters - #there is a "last modified" date given on the page, extract it and use that. - moddatetag = storyContentBox.find('span', {'class':'last_modified'}) - if not moddatetag is None: - newestChapter = self.ordinal_date_string_to_date(moddatetag('span')[1].text) - - # Date updated - self.story.setMetadata("dateUpdated", newestChapter) - - # Date published - # falls back to oldest chapter date for stories that haven't been officially published yet - pubdatetag = storyContentBox.find('span', {'class':'date_approved'}) - if pubdatetag is None: - if oldestChapter is None: - #this will only be true when updating metadata for stories that have 0 chapters - #and that have never been officially published - a rare occurrence. Fall back to last - #modified date as the publication date, it's all that we've got. - self.story.setMetadata("datePublished", newestChapter) - else: - self.story.setMetadata("datePublished", oldestChapter) - else: - pubDate = self.ordinal_date_string_to_date(pubdatetag('span')[1].text) - self.story.setMetadata("datePublished", pubDate) - - # Characters - chars = storyContentBox.find("div", {"class":"extra_story_data"}) - for character in chars.find_all("a", {"class":"character_icon"}): - self.story.addToList("characters", character['title']) - - # Likes and dislikes - storyToolbar = soup.find('div', {'class':'story-toolbar'}) - likes = storyToolbar.find('span', {'class':'likes'}) - if not likes is None: - self.story.setMetadata("likes", stripHTML(likes)) - dislikes = storyToolbar.find('span', {'class':'dislikes'}) - if not dislikes is None: - self.story.setMetadata("dislikes", stripHTML(dislikes)) - - # Highest view for a chapter and total views - viewSpan = storyToolbar.find('span', {'title':re.compile(r'.*\btotal views\b.*')}) - self.story.setMetadata("views", re.sub(r'[^0-9]', '', stripHTML(viewSpan))) - self.story.setMetadata("total_views", re.sub(r'[^0-9]', '', viewSpan['title'])) - - # Comment count - commentSpan = storyToolbar.find('span', {'title':re.compile(r'.*\bcomments\b.*')}) - self.story.setMetadata("comment_count", re.sub(r'[^0-9]', '', stripHTML(commentSpan))) - - # Short description - descriptionMeta = soup.find('meta', {'property':'og:description'}) - self.story.setMetadata("short_description", stripHTML(descriptionMeta['content'])) - - #groups - if soup.find('button', {'id':'button-view-all-groups'}): - groupResponse = self._fetchUrl("http://www.fimfiction.net/ajax/groups/story_groups_list.php?story=%s" % (self.story.getMetadata("storyId"))) - groupData = json.loads(groupResponse) - groupList = self.make_soup(groupData["content"]) - else: - groupList = soup.find('ul', {'id':'story-groups-list'}) - - if not (groupList == None): - for groupName in groupList.find_all('a'): - self.story.addToList("groupsUrl", 'http://'+self.host+groupName["href"]) - self.story.addToList("groups",stripHTML(groupName).replace(',', ';')) - - #sequels - for header in soup.find_all('h1', {'class':'header-stories'}): - # I don't know why using text=re.compile with find() wouldn't work, but it didn't. - if header.text.startswith('Sequels'): - sequelContainer = header.parent - for sequel in sequelContainer.find_all('a', {'class':'story_link'}): - self.story.addToList("sequelsUrl", 'http://'+self.host+sequel["href"]) - self.story.addToList("sequels", stripHTML(sequel).replace(',', ';')) - - #author last login - userPageHeader = soup.find('div', {'class':re.compile(r'\buser-page-header\b')}) - if not userPageHeader == None: - infoContainer = userPageHeader.find('div', {'class':re.compile(r'\binfo-container\b')}) - listItems = infoContainer.find_all('li') - lastLoginString = stripHTML(listItems[1]) - lastLogin = None - if "online" in lastLoginString: - lastLogin = date.today() - elif "offline" in lastLoginString: - #this regex extracts the number of weeks and the number of days from the last login string. - #durations under a day are ignored. - #group 1 is weeks, group 2 is days - durationGroups = re.match(r"(?:[^0-9]*(\d+?)w)?[^0-9]*(?:(\d+?)d)?", lastLoginString) - lastLogin = date.today() - timedelta(days=int(durationGroups.group(2) or 0), weeks=int(durationGroups.group(1) or 0)) - self.story.setMetadata("authorLastLogin", lastLogin) - - #The link to the prequel is embedded in the description text, so erring - #on the side of caution and wrapping this whole thing in a try block. - #If anything goes wrong this probably wasn't a valid prequel link. - try: - description = soup.find('div', {'class':'description'}) - firstHR = description.find("hr") - nextSib = firstHR.nextSibling - if "This story is a sequel to" in nextSib.string: - link = nextSib.nextSibling - if link.name == "a": - self.story.setMetadata("prequelUrl", 'http://'+self.host+link["href"]) - self.story.setMetadata("prequel", stripHTML(link)) - except: - pass - - def ordinal_date_string_to_date(self, datestring): - datestripped=re.sub(r"(\d+)(st|nd|rd|th)", r"\1", datestring.strip()) - return makeDate(datestripped, self.dateformat) - - def hookForUpdates(self,chaptercount): - if self.oldchapters and len(self.oldchapters) > self.newestChapterNum: - logger.info("Existing epub has %s chapters\nNewest chapter is %s. Discarding old chapters from there on."%(len(self.oldchapters), self.newestChapterNum+1)) - self.oldchapters = self.oldchapters[:self.newestChapterNum] - return len(self.oldchapters) - - def do_fix_blockquotes(self,data): - if self.getConfig('fix_fimf_blockquotes'): - #

- #

- # include > in re groups so there's always something in the group. - data = re.sub(r']*>\s*)]*>)',r'\s*)

',r'',data) - return data - - def getChapterText(self, url): - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - - soup = self.make_soup(data) - if not (soup.find('form', {'id' : 'password_form'}) == None): - if self.password: - params = {} - params['password'] = self.password - data = self._postUrl(url, params) - else: - logger.error("Chapter %s needed password but no password was present" % url) - - data = self.do_fix_blockquotes(data) - - soup = self.make_soup(data).find('div', {'class' : 'chapter_content'}) - if soup == None: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,soup) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +from datetime import date +from datetime import timedelta +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import cookielib as cl +import json + +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return FimFictionNetSiteAdapter + +class FimFictionNetSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','fimficnet') + self.story.setMetadata('storyId', self.parsedUrl.path.split('/',)[2]) + self._setURL("http://"+self.getSiteDomain()+"/story/"+self.story.getMetadata('storyId')+"/") + self.is_adult = False + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d %b %Y" + + @staticmethod + def getSiteDomain(): + return 'www.fimfiction.net' + + @classmethod + def getAcceptDomains(cls): + # mobile.fimifction.com isn't actually a valid domain, but we can still get the story id from URLs anyway + return ['www.fimfiction.net','mobile.fimfiction.net', 'www.fimfiction.com', 'mobile.fimfiction.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://www.fimfiction.net/story/1234/story-title-here http://www.fimfiction.net/story/1234/ http://www.fimfiction.com/story/1234/1/ http://mobile.fimfiction.net/story/1234/1/story-title-here/chapter-title-here" + + def getSiteURLPattern(self): + return r"https?://(www|mobile)\.fimfiction\.(net|com)/story/\d+/?.*" + + def use_pagecache(self): + ''' + adapters that will work with the page cache need to implement + this and change it to True. + ''' + return True + + def doExtractChapterUrlsAndMetadata(self,get_cover=True): + + if self.is_adult or self.getConfig("is_adult"): + cookie = cl.Cookie(version=0, name='view_mature', value='true', + port=None, port_specified=False, + domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, + path='/story', path_specified=True, + secure=False, + expires=time.time()+10000, + discard=False, + comment=None, + comment_url=None, + rest={'HttpOnly': None}, + rfc2109=False) + self.cookiejar.set_cookie(cookie) + + ##--------------------------------------------------------------------------------------------------- + ## Get the story's title page. Check if it exists. + + try: + # don't use cache if manual is_adult--should only happen + # if it's an adult story and they don't have is_adult in ini. + data = self.do_fix_blockquotes(self._fetchUrl(self.url, + usecache=(not self.is_adult))) + soup = self.make_soup(data) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Warning: mysql_fetch_array(): supplied argument is not a valid MySQL result resource" in data: + raise exceptions.StoryDoesNotExist(self.url) + + if "This story has been marked as having adult content. Please click below to confirm you are of legal age to view adult material in your country." in data: + raise exceptions.AdultCheckRequired(self.url) + + if self.password: + params = {} + params['password'] = self.password + data = self._postUrl(self.url, params) + soup = self.make_soup(data) + + if not (soup.find('form', {'id' : 'password_form'}) == None): + if self.getConfig('fail_on_password'): + raise exceptions.FailedToDownload("%s requires story password and fail_on_password is true."%self.url) + else: + raise exceptions.FailedToLogin(self.url,"Story requires individual password",passwdonly=True) + + ##---------------------------------------------------------------------------------------------------- + ## Extract metadata + + storyContentBox = soup.find('div', {'class':'story_content_box'}) + + # Title + title = storyContentBox.find('a', {'class':re.compile(r'.*\bstory_name\b.*')}) + self.story.setMetadata('title',stripHTML(title)) + + # Author + author = storyContentBox.find('div', {'class':'author'}).find('a') + self.story.setMetadata("author", stripHTML(author)) + #No longer seems to be a way to access Fimfiction's internal author ID + self.story.setMetadata("authorId", self.story.getMetadata("author")) + self.story.setMetadata("authorUrl", "http://%s/user/%s" % (self.getSiteDomain(), stripHTML(author))) + + #Rating text is replaced with full words for historical compatibility after the site changed + #on 2014-10-27 + rating = stripHTML(storyContentBox.find('a', {'class':re.compile(r'.*\bcontent-rating-.*')})) + rating = rating.replace("E", "Everyone").replace("T", "Teen").replace("M", "Mature") + self.story.setMetadata("rating", rating) + + # Chapters + for chapter in storyContentBox.find_all('a',{'class':'chapter_link'}): + self.chapterUrls.append((stripHTML(chapter), 'http://'+self.host+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # Status + # In the case of Fimfiction, possible statuses are 'Completed', 'Incomplete', 'On Hiatus' and 'Cancelled' + # For the sake of bringing it in line with the other adapters, 'Incomplete' becomes 'In-Progress' + # and 'Complete' becomes 'Completed'. 'Cancelled' and 'On Hiatus' are passed through, it's easy now for users + # to change/remove if they want with replace_metadata + status = stripHTML(storyContentBox.find('span', {'class':re.compile(r'.*\bcompleted-status-.*')})) + status = status.replace("Incomplete", "In-Progress").replace("Complete", "Completed") + self.story.setMetadata("status", status) + + # Genres and Warnings + # warnings were folded into general categories in the 2014-10-27 site update + categories = storyContentBox.find_all('a', {'class':re.compile(r'.*\bstory_category\b.*')}) + for category in categories: + category = stripHTML(category) + if category == "Gore" or category == "Sex": + self.story.addToList('warnings', category) + else: + self.story.addToList('genre', category) + + # Word count + wordCountText = stripHTML(storyContentBox.find('li', {'class':'bottom'}).find('div', {'class':'word_count'})) + self.story.setMetadata("numWords", re.sub(r'[^0-9]', '', wordCountText)) + + # Cover image + storyImage = storyContentBox.find('div', {'class':'story_image'}) + if storyImage: + coverurl = storyImage.find('a')['href'] + if coverurl.startswith('//'): # fix for img urls missing 'http:' + coverurl = "http:"+coverurl + if get_cover: + self.setCoverImage(self.url,coverurl) + + coverSource = storyImage.find('a', {'class':'source'}) + if coverSource: + self.story.setMetadata('coverSourceUrl', coverSource['href']) + #There's no text associated with the cover source link, so just + #reuse the URL. Makes it clear it's an external link leading + #outside of the fanfic site, at least. + self.story.setMetadata('coverSource', coverSource['href']) + + # fimf has started including extra stuff inside the description div. + descdivstr = u"%s"%storyContentBox.find("div", {"class":"description"}) + hrstr=u"
" + descdivstr = u'
'+descdivstr[descdivstr.index(hrstr)+len(hrstr):] + self.setDescription(self.url,descdivstr) + + # Find the newest and oldest chapter dates + storyData = storyContentBox.find('div', {'class':'story_data'}) + oldestChapter = None + newestChapter = None + self.newestChapterNum = None # save for comparing during update. + # Scan all chapters to find the oldest and newest, on + # FiMFiction it's possible for authors to insert new chapters + # out-of-order or change the dates of earlier ones by editing + # them--That WILL break epub update. + for index, chapterDate in enumerate(storyData.find_all('span', {'class':'date'})): + chapterDate = self.ordinal_date_string_to_date(chapterDate.contents[1]) + if oldestChapter == None or chapterDate < oldestChapter: + oldestChapter = chapterDate + if newestChapter == None or chapterDate > newestChapter: + newestChapter = chapterDate + self.newestChapterNum = index + + if newestChapter is None: + #this will only be true when updating metadata for stories that have 0 chapters + #there is a "last modified" date given on the page, extract it and use that. + moddatetag = storyContentBox.find('span', {'class':'last_modified'}) + if not moddatetag is None: + newestChapter = self.ordinal_date_string_to_date(moddatetag('span')[1].text) + + # Date updated + self.story.setMetadata("dateUpdated", newestChapter) + + # Date published + # falls back to oldest chapter date for stories that haven't been officially published yet + pubdatetag = storyContentBox.find('span', {'class':'date_approved'}) + if pubdatetag is None: + if oldestChapter is None: + #this will only be true when updating metadata for stories that have 0 chapters + #and that have never been officially published - a rare occurrence. Fall back to last + #modified date as the publication date, it's all that we've got. + self.story.setMetadata("datePublished", newestChapter) + else: + self.story.setMetadata("datePublished", oldestChapter) + else: + pubDate = self.ordinal_date_string_to_date(pubdatetag('span')[1].text) + self.story.setMetadata("datePublished", pubDate) + + # Characters + chars = storyContentBox.find("div", {"class":"extra_story_data"}) + for character in chars.find_all("a", {"class":"character_icon"}): + self.story.addToList("characters", character['title']) + + # Likes and dislikes + storyToolbar = soup.find('div', {'class':'story-toolbar'}) + likes = storyToolbar.find('span', {'class':'likes'}) + if not likes is None: + self.story.setMetadata("likes", stripHTML(likes)) + dislikes = storyToolbar.find('span', {'class':'dislikes'}) + if not dislikes is None: + self.story.setMetadata("dislikes", stripHTML(dislikes)) + + # Highest view for a chapter and total views + viewSpan = storyToolbar.find('span', {'title':re.compile(r'.*\btotal views\b.*')}) + self.story.setMetadata("views", re.sub(r'[^0-9]', '', stripHTML(viewSpan))) + self.story.setMetadata("total_views", re.sub(r'[^0-9]', '', viewSpan['title'])) + + # Comment count + commentSpan = storyToolbar.find('span', {'title':re.compile(r'.*\bcomments\b.*')}) + self.story.setMetadata("comment_count", re.sub(r'[^0-9]', '', stripHTML(commentSpan))) + + # Short description + descriptionMeta = soup.find('meta', {'property':'og:description'}) + self.story.setMetadata("short_description", stripHTML(descriptionMeta['content'])) + + #groups + if soup.find('button', {'id':'button-view-all-groups'}): + groupResponse = self._fetchUrl("http://www.fimfiction.net/ajax/groups/story_groups_list.php?story=%s" % (self.story.getMetadata("storyId"))) + groupData = json.loads(groupResponse) + groupList = self.make_soup(groupData["content"]) + else: + groupList = soup.find('ul', {'id':'story-groups-list'}) + + if not (groupList == None): + for groupName in groupList.find_all('a'): + self.story.addToList("groupsUrl", 'http://'+self.host+groupName["href"]) + self.story.addToList("groups",stripHTML(groupName).replace(',', ';')) + + #sequels + for header in soup.find_all('h1', {'class':'header-stories'}): + # I don't know why using text=re.compile with find() wouldn't work, but it didn't. + if header.text.startswith('Sequels'): + sequelContainer = header.parent + for sequel in sequelContainer.find_all('a', {'class':'story_link'}): + self.story.addToList("sequelsUrl", 'http://'+self.host+sequel["href"]) + self.story.addToList("sequels", stripHTML(sequel).replace(',', ';')) + + #author last login + userPageHeader = soup.find('div', {'class':re.compile(r'\buser-page-header\b')}) + if not userPageHeader == None: + infoContainer = userPageHeader.find('div', {'class':re.compile(r'\binfo-container\b')}) + listItems = infoContainer.find_all('li') + lastLoginString = stripHTML(listItems[1]) + lastLogin = None + if "online" in lastLoginString: + lastLogin = date.today() + elif "offline" in lastLoginString: + #this regex extracts the number of weeks and the number of days from the last login string. + #durations under a day are ignored. + #group 1 is weeks, group 2 is days + durationGroups = re.match(r"(?:[^0-9]*(\d+?)w)?[^0-9]*(?:(\d+?)d)?", lastLoginString) + lastLogin = date.today() - timedelta(days=int(durationGroups.group(2) or 0), weeks=int(durationGroups.group(1) or 0)) + self.story.setMetadata("authorLastLogin", lastLogin) + + #The link to the prequel is embedded in the description text, so erring + #on the side of caution and wrapping this whole thing in a try block. + #If anything goes wrong this probably wasn't a valid prequel link. + try: + description = soup.find('div', {'class':'description'}) + firstHR = description.find("hr") + nextSib = firstHR.nextSibling + if "This story is a sequel to" in nextSib.string: + link = nextSib.nextSibling + if link.name == "a": + self.story.setMetadata("prequelUrl", 'http://'+self.host+link["href"]) + self.story.setMetadata("prequel", stripHTML(link)) + except: + pass + + def ordinal_date_string_to_date(self, datestring): + datestripped=re.sub(r"(\d+)(st|nd|rd|th)", r"\1", datestring.strip()) + return makeDate(datestripped, self.dateformat) + + def hookForUpdates(self,chaptercount): + if self.oldchapters and len(self.oldchapters) > self.newestChapterNum: + logger.info("Existing epub has %s chapters\nNewest chapter is %s. Discarding old chapters from there on."%(len(self.oldchapters), self.newestChapterNum+1)) + self.oldchapters = self.oldchapters[:self.newestChapterNum] + return len(self.oldchapters) + + def do_fix_blockquotes(self,data): + if self.getConfig('fix_fimf_blockquotes'): + #

+ #

+ # include > in re groups so there's always something in the group. + data = re.sub(r']*>\s*)]*>)',r'\s*)

',r'',data) + return data + + def getChapterText(self, url): + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + + soup = self.make_soup(data) + if not (soup.find('form', {'id' : 'password_form'}) == None): + if self.password: + params = {} + params['password'] = self.password + data = self._postUrl(url, params) + else: + logger.error("Chapter %s needed password but no password was present" % url) + + data = self.do_fix_blockquotes(data) + + soup = self.make_soup(data).find('div', {'class' : 'chapter_content'}) + if soup == None: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,soup) diff --git a/fff_internals/adapters/adapter_finestoriescom.py b/fanficfare/adapters/adapter_finestoriescom.py similarity index 97% rename from fff_internals/adapters/adapter_finestoriescom.py rename to fanficfare/adapters/adapter_finestoriescom.py index 6c99f9d..1902494 100644 --- a/fff_internals/adapters/adapter_finestoriescom.py +++ b/fanficfare/adapters/adapter_finestoriescom.py @@ -1,288 +1,288 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return FineStoriesComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class FineStoriesComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2].split(':')[0]) - if 'storyInfo' in self.story.getMetadata('storyId'): - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/s/storyInfo.php?id='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','fnst') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%Y-%m-%d" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'finestories.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/s/1234 http://"+cls.getSiteDomain()+"/s/1234:4010 http://"+cls.getSiteDomain()+"/library/storyInfo.php?id=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain())+r"/(s|library)?/(storyInfo.php\?id=)?\d+(:\d+)?(;\d+)?$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Free Registration' in data \ - or "Invalid Password!" in data \ - or "Invalid User Name!" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['theusername'] = self.username - params['thepassword'] = self.password - else: - params['theusername'] = self.getConfig("username") - params['thepassword'] = self.getConfig("password") - params['rememberMe'] = '1' - params['page'] = 'http://'+self.getSiteDomain()+'/' - params['submit'] = 'Login' - - loginUrl = 'http://' + self.getSiteDomain() + '/login.php' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['theusername'])) - - d = self._fetchUrl(loginUrl, params) - - if "My Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['theusername'])) - raise exceptions.FailedToLogin(url,params['theusername']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'/s/'+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"/a/\w+")) - self.story.setMetadata('authorId',a['href'].split('/')[2]) - self.story.setMetadata('authorUrl','http://'+self.host+a['href']) - self.story.setMetadata('author',a.text) - - # Find the chapters: - chapters = soup.findAll('a', href=re.compile(r'/s/'+self.story.getMetadata('storyId')+":\d+$")) - if len(chapters) != 0: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+chapter['href'])) - else: - self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/s/'+self.story.getMetadata('storyId'))) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # surprisingly, the detailed page does not give enough details, so go to author's page - - skip=0 - i=0 - while i == 0: - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl')+"&skip="+str(skip))) - - a = asoup.findAll('td', {'class' : 'lc2'}) - for lc2 in a: - if lc2.find('a')['href'] == '/s/'+self.story.getMetadata('storyId'): - i=1 - break - if a[len(a)-1] == lc2: - skip=skip+10 - - for cat in lc2.findAll('div', {'class' : 'typediv'}): - self.story.addToList('category',cat.text) - - self.story.setMetadata('numWords', lc2.findNext('td', {'class' : 'num'}).text) - - lc4 = lc2.findNext('td', {'class' : 'lc4'}) - - - try: - a = lc4.find('a', href=re.compile(r"/library/show_series.php\?id=\d+")) - i = a.parent.text.split('(')[1].split(')')[0] - self.setSeries(a.text, i) - self.story.setMetadata('seriesUrl','http://'+self.host+a['href']) - except: - pass - try: - a = lc4.find('a', href=re.compile(r"/library/universe.php\?id=\d+")) - self.story.addToList("category",a.text) - except: - pass - - for a in lc4.findAll('span', {'class' : 'help'}): - a.extract() - - self.setDescription('http://'+self.host+'/s/'+self.story.getMetadata('storyId'),lc4.text.split('[More Info')[0]) - - for b in lc4.findAll('b'): - label = b.text - value = b.nextSibling - - if 'For Age' in label: - self.story.setMetadata('rating', value) - - if 'Tags' in label: - for genre in value.split(', '): - self.story.addToList('genre',genre) - - if 'Posted' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) - - if 'Concluded' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) - - status = lc4.find('span', {'class' : 'ab'}) - if status != None: - self.story.setMetadata('status', 'In-Progress') - if "Last Activity" in status.text: - self.story.setMetadata('dateUpdated', makeDate(status.text.split('Activity: ')[1].split(')')[0], self.dateformat)) - else: - self.story.setMetadata('status', 'Completed') - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - # some big chapters are split over several pages - pager = div.find('span', {'class' : 'pager'}) - if pager != None: - urls=pager.findAll('a') - urls=urls[:len(urls)-1] - - - for ur in urls: - soup = bs.BeautifulSoup(self._fetchUrl("http://"+self.getSiteDomain()+ur['href']), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div1 = soup.find('div', {'id' : 'story'}) - - # appending next section - last=div.findAll('p') - next=div1.find('span', {'class' : 'conTag'}).nextSibling - - last[len(last)-1]=last[len(last)-1].append(next) - div.append(div1) - - # removing all the left-over stuff - for a in div.findAll('span'): - a.extract() - - for a in div.findAll('h1'): - a.extract() - for a in div.findAll('h2'): - a.extract() - for a in div.findAll('h3'): - a.extract() - for a in div.findAll('h4'): - a.extract() - for a in div.findAll('br'): - a.extract() - for a in div.findAll('div', {'class' : 'date'}): - a.extract() - - a = div.find('form') - if a != None: - b = a.nextSibling - while b != None: - a.extract() - a=b - b=b.nextSibling - - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return FineStoriesComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class FineStoriesComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url + self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2].split(':')[0]) + if 'storyInfo' in self.story.getMetadata('storyId'): + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/s/storyInfo.php?id='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','fnst') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y-%m-%d" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'finestories.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/s/1234 http://"+cls.getSiteDomain()+"/s/1234:4010 http://"+cls.getSiteDomain()+"/library/storyInfo.php?id=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain())+r"/(s|library)?/(storyInfo.php\?id=)?\d+(:\d+)?(;\d+)?$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Free Registration' in data \ + or "Invalid Password!" in data \ + or "Invalid User Name!" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['theusername'] = self.username + params['thepassword'] = self.password + else: + params['theusername'] = self.getConfig("username") + params['thepassword'] = self.getConfig("password") + params['rememberMe'] = '1' + params['page'] = 'http://'+self.getSiteDomain()+'/' + params['submit'] = 'Login' + + loginUrl = 'http://' + self.getSiteDomain() + '/login.php' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['theusername'])) + + d = self._fetchUrl(loginUrl, params) + + if "My Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['theusername'])) + raise exceptions.FailedToLogin(url,params['theusername']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'/s/'+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"/a/\w+")) + self.story.setMetadata('authorId',a['href'].split('/')[2]) + self.story.setMetadata('authorUrl','http://'+self.host+a['href']) + self.story.setMetadata('author',a.text) + + # Find the chapters: + chapters = soup.findAll('a', href=re.compile(r'/s/'+self.story.getMetadata('storyId')+":\d+$")) + if len(chapters) != 0: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+chapter['href'])) + else: + self.chapterUrls.append((self.story.getMetadata('title'),'http://'+self.host+'/s/'+self.story.getMetadata('storyId'))) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # surprisingly, the detailed page does not give enough details, so go to author's page + + skip=0 + i=0 + while i == 0: + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl')+"&skip="+str(skip))) + + a = asoup.findAll('td', {'class' : 'lc2'}) + for lc2 in a: + if lc2.find('a')['href'] == '/s/'+self.story.getMetadata('storyId'): + i=1 + break + if a[len(a)-1] == lc2: + skip=skip+10 + + for cat in lc2.findAll('div', {'class' : 'typediv'}): + self.story.addToList('category',cat.text) + + self.story.setMetadata('numWords', lc2.findNext('td', {'class' : 'num'}).text) + + lc4 = lc2.findNext('td', {'class' : 'lc4'}) + + + try: + a = lc4.find('a', href=re.compile(r"/library/show_series.php\?id=\d+")) + i = a.parent.text.split('(')[1].split(')')[0] + self.setSeries(a.text, i) + self.story.setMetadata('seriesUrl','http://'+self.host+a['href']) + except: + pass + try: + a = lc4.find('a', href=re.compile(r"/library/universe.php\?id=\d+")) + self.story.addToList("category",a.text) + except: + pass + + for a in lc4.findAll('span', {'class' : 'help'}): + a.extract() + + self.setDescription('http://'+self.host+'/s/'+self.story.getMetadata('storyId'),lc4.text.split('[More Info')[0]) + + for b in lc4.findAll('b'): + label = b.text + value = b.nextSibling + + if 'For Age' in label: + self.story.setMetadata('rating', value) + + if 'Tags' in label: + for genre in value.split(', '): + self.story.addToList('genre',genre) + + if 'Posted' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) + + if 'Concluded' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value.split('/ (')[0]), self.dateformat)) + + status = lc4.find('span', {'class' : 'ab'}) + if status != None: + self.story.setMetadata('status', 'In-Progress') + if "Last Activity" in status.text: + self.story.setMetadata('dateUpdated', makeDate(status.text.split('Activity: ')[1].split(')')[0], self.dateformat)) + else: + self.story.setMetadata('status', 'Completed') + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + # some big chapters are split over several pages + pager = div.find('span', {'class' : 'pager'}) + if pager != None: + urls=pager.findAll('a') + urls=urls[:len(urls)-1] + + + for ur in urls: + soup = bs.BeautifulSoup(self._fetchUrl("http://"+self.getSiteDomain()+ur['href']), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div1 = soup.find('div', {'id' : 'story'}) + + # appending next section + last=div.findAll('p') + next=div1.find('span', {'class' : 'conTag'}).nextSibling + + last[len(last)-1]=last[len(last)-1].append(next) + div.append(div1) + + # removing all the left-over stuff + for a in div.findAll('span'): + a.extract() + + for a in div.findAll('h1'): + a.extract() + for a in div.findAll('h2'): + a.extract() + for a in div.findAll('h3'): + a.extract() + for a in div.findAll('h4'): + a.extract() + for a in div.findAll('br'): + a.extract() + for a in div.findAll('div', {'class' : 'date'}): + a.extract() + + a = div.find('form') + if a != None: + b = a.nextSibling + while b != None: + a.extract() + a=b + b=b.nextSibling + + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_grangerenchantedcom.py b/fanficfare/adapters/adapter_grangerenchantedcom.py similarity index 97% rename from fff_internals/adapters/adapter_grangerenchantedcom.py rename to fanficfare/adapters/adapter_grangerenchantedcom.py index 831f08f..c73cceb 100644 --- a/fff_internals/adapters/adapter_grangerenchantedcom.py +++ b/fanficfare/adapters/adapter_grangerenchantedcom.py @@ -1,311 +1,311 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return GrangerEnchantedCom - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class GrangerEnchantedCom(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - self.section=self.parsedUrl.path.split('/',)[1] - - # normalized story URL. - if "malfoymanor" in self.parsedUrl.netloc: - self._setURL('http://malfoymanor.' + self.getSiteDomain() + '/themanor/viewstory.php?sid='+self.story.getMetadata('storyId')) - self.story.addToList("category","The Manor") - else: - self._setURL('http://' + self.getSiteDomain() + '/enchant/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','gech') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%b/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'grangerenchanted.com' - - @classmethod - def getAcceptDomains(cls): - return ['grangerenchanted.com','malfoymanor.grangerenchanted.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://grangerenchanted.com/enchant/viewstory.php?sid=1234 http://malfoymanor.grangerenchanted.com/themanor/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return r"http://(malfoymanor.)?grangerenchanted.com/(enchant|themanor)?/viewstory.php\?sid=\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - if "enchant" in self.section: - loginUrl = 'http://grangerenchanted.com/enchant/user.php?action=login' - else: - loginUrl = 'http://malfoymanor.grangerenchanted.com/themanor/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=1" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Read' in label: - self.story.setMetadata('read', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+self.section+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - except: - # I find it hard to care if the series parsing fails - pass - - try: - self.story.setMetadata('reviews', - stripHTML(soup.find('div',{'id':'sort'}). - findAll('a', href=re.compile(r'^reviews.php'))[1])) - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story1'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return GrangerEnchantedCom + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class GrangerEnchantedCom(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + self.section=self.parsedUrl.path.split('/',)[1] + + # normalized story URL. + if "malfoymanor" in self.parsedUrl.netloc: + self._setURL('http://malfoymanor.' + self.getSiteDomain() + '/themanor/viewstory.php?sid='+self.story.getMetadata('storyId')) + self.story.addToList("category","The Manor") + else: + self._setURL('http://' + self.getSiteDomain() + '/enchant/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','gech') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%b/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'grangerenchanted.com' + + @classmethod + def getAcceptDomains(cls): + return ['grangerenchanted.com','malfoymanor.grangerenchanted.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://grangerenchanted.com/enchant/viewstory.php?sid=1234 http://malfoymanor.grangerenchanted.com/themanor/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return r"http://(malfoymanor.)?grangerenchanted.com/(enchant|themanor)?/viewstory.php\?sid=\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + if "enchant" in self.section: + loginUrl = 'http://grangerenchanted.com/enchant/user.php?action=login' + else: + loginUrl = 'http://malfoymanor.grangerenchanted.com/themanor/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=1" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Read' in label: + self.story.setMetadata('read', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+self.section+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + except: + # I find it hard to care if the series parsing fails + pass + + try: + self.story.setMetadata('reviews', + stripHTML(soup.find('div',{'id':'sort'}). + findAll('a', href=re.compile(r'^reviews.php'))[1])) + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story1'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_harrypotterfanfictioncom.py b/fanficfare/adapters/adapter_harrypotterfanfictioncom.py similarity index 97% rename from fff_internals/adapters/adapter_harrypotterfanfictioncom.py rename to fanficfare/adapters/adapter_harrypotterfanfictioncom.py index bddf4b6..1453bd3 100644 --- a/fff_internals/adapters/adapter_harrypotterfanfictioncom.py +++ b/fanficfare/adapters/adapter_harrypotterfanfictioncom.py @@ -1,203 +1,203 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -class HarryPotterFanFictionComSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','hp') - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only psid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?psid='+self.story.getMetadata('storyId')) - - - @staticmethod - def getSiteDomain(): - return 'www.harrypotterfanfiction.com' - - @classmethod - def getAcceptDomains(cls): - return ['www.harrypotterfanfiction.com','harrypotterfanfiction.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://www.harrypotterfanfiction.com/viewstory.php?psid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+r"(www\.)?"+re.escape("harrypotterfanfiction.com/viewstory.php?psid=")+r"\d+$" - - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def extractChapterUrlsAndMetadata(self): - - url = self.url+'&index=1' - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - ## Title - a = soup.find('a', href=re.compile(r'\?psid='+self.story.getMetadata('storyId'))) - self.story.setMetadata('title',stripHTML(a)) - ## javascript:if (confirm('Please note. This story may contain adult themes. By clicking here you are stating that you are over 17. Click cancel if you do not meet this requirement.')) location = '?psid=290995' - if "This story may contain adult themes." in a['href'] and not (self.is_adult or self.getConfig("is_adult")): - raise exceptions.AdultCheckRequired(self.url) - - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?showuid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - ## hpcom doesn't give us total words--but it does give - ## us words/chapter. I'd rather add than fetch and - ## parse another page. - words=0 - for tr in soup.find('table',{'class':'text'}).findAll('tr'): - tdstr = tr.findAll('td')[2].string - if tdstr and tdstr.isdigit(): - words+=int(tdstr) - self.story.setMetadata('numWords',str(words)) - - # Find the chapters: - tablelist = soup.find('table',{'class':'text'}) - for chapter in tablelist.findAll('a', href=re.compile(r'\?chapterid=\d+')): - #javascript:if (confirm('Please note. This story may contain adult themes. By clicking here you are stating that you are over 17. Click cancel if you do not meet this requirement.')) location = '?chapterid=433441&i=1' - # just in case there's tags, like in chapter titles. - chpt=re.sub(r'^.*?(\?chapterid=\d+).*?',r'\1',chapter['href']) - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/viewstory.php'+chpt)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - ## Finding the metadata is a bit of a pain. Desc is the only thing this color. - desctable= soup.find('table',{'bgcolor':'#f0e8e8'}) - self.setDescription(url,desctable) - #self.story.setMetadata('description',stripHTML(desctable)) - - ## Finding the metadata is a bit of a pain. Most of the meta - ## data is in a center.table without a bgcolor. - #for center in soup.findAll('center'): - table = soup.find('table',{'class':'storymaininfo'}) - if table: - metastr = stripHTML(str(table)).replace('\n',' ').replace('\t',' ') - # Rating: 12+ Story Reviews: 3 - # Chapters: 3 - # Characters: Andromeda, Ted, Bellatrix, R. Lestrange, Lucius, Narcissa, OC - # Genre(s): Fluff, Romance, Young Adult Era: OtherPairings: Other Pairing, Lucius/Narcissa - # Status: Completed - # First Published: 2010.09.02 - # Last Published Chapter: 2010.09.28 - # Last Updated: 2010.09.28 - # Favorite Story Of: 1 users - # Warnings: Scenes of a Mild Sexual Nature - - m = re.match(r".*?Status: Completed.*?",metastr) - if m: - self.story.setMetadata('status','Completed') - else: - self.story.setMetadata('status','In-Progress') - - m = re.match(r".*?Rating: (.+?) Story Reviews.*?",metastr) - if m: - self.story.setMetadata('rating', m.group(1)) - - m = re.match(r".*?Genre\(s\): (.+?) Era.*?",metastr) - if m: - for g in m.group(1).split(','): - self.story.addToList('genre',g) - - m = re.match(r".*?Characters: (.+?) Genre.*?",metastr) - if m: - for g in m.group(1).split(','): - self.story.addToList('characters',g) - - m = re.match(r".*?Warnings: (.+).*?",metastr) - if m: - for w in m.group(1).split(','): - if w != 'Now Warnings': - self.story.addToList('warnings',w) - - m = re.match(r".*?First Published: ([0-9\.]+).*?",metastr) - if m: - self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y.%m.%d")) - - # Updated can have more than one space after it. - m = re.match(r".*?Last Updated: ([0-9\.]+).*?",metastr) - if m: - self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y.%m.%d")) - - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - ## most adapters use BeautifulStoneSoup here, but non-Stone - ## allows nested div tags. - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'fluidtext'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) - -def getClass(): - return HarryPotterFanFictionComSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +class HarryPotterFanFictionComSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','hp') + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query is only psid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?psid='+self.story.getMetadata('storyId')) + + + @staticmethod + def getSiteDomain(): + return 'www.harrypotterfanfiction.com' + + @classmethod + def getAcceptDomains(cls): + return ['www.harrypotterfanfiction.com','harrypotterfanfiction.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://www.harrypotterfanfiction.com/viewstory.php?psid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+r"(www\.)?"+re.escape("harrypotterfanfiction.com/viewstory.php?psid=")+r"\d+$" + + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def extractChapterUrlsAndMetadata(self): + + url = self.url+'&index=1' + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + ## Title + a = soup.find('a', href=re.compile(r'\?psid='+self.story.getMetadata('storyId'))) + self.story.setMetadata('title',stripHTML(a)) + ## javascript:if (confirm('Please note. This story may contain adult themes. By clicking here you are stating that you are over 17. Click cancel if you do not meet this requirement.')) location = '?psid=290995' + if "This story may contain adult themes." in a['href'] and not (self.is_adult or self.getConfig("is_adult")): + raise exceptions.AdultCheckRequired(self.url) + + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?showuid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + ## hpcom doesn't give us total words--but it does give + ## us words/chapter. I'd rather add than fetch and + ## parse another page. + words=0 + for tr in soup.find('table',{'class':'text'}).findAll('tr'): + tdstr = tr.findAll('td')[2].string + if tdstr and tdstr.isdigit(): + words+=int(tdstr) + self.story.setMetadata('numWords',str(words)) + + # Find the chapters: + tablelist = soup.find('table',{'class':'text'}) + for chapter in tablelist.findAll('a', href=re.compile(r'\?chapterid=\d+')): + #javascript:if (confirm('Please note. This story may contain adult themes. By clicking here you are stating that you are over 17. Click cancel if you do not meet this requirement.')) location = '?chapterid=433441&i=1' + # just in case there's tags, like in chapter titles. + chpt=re.sub(r'^.*?(\?chapterid=\d+).*?',r'\1',chapter['href']) + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/viewstory.php'+chpt)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + ## Finding the metadata is a bit of a pain. Desc is the only thing this color. + desctable= soup.find('table',{'bgcolor':'#f0e8e8'}) + self.setDescription(url,desctable) + #self.story.setMetadata('description',stripHTML(desctable)) + + ## Finding the metadata is a bit of a pain. Most of the meta + ## data is in a center.table without a bgcolor. + #for center in soup.findAll('center'): + table = soup.find('table',{'class':'storymaininfo'}) + if table: + metastr = stripHTML(str(table)).replace('\n',' ').replace('\t',' ') + # Rating: 12+ Story Reviews: 3 + # Chapters: 3 + # Characters: Andromeda, Ted, Bellatrix, R. Lestrange, Lucius, Narcissa, OC + # Genre(s): Fluff, Romance, Young Adult Era: OtherPairings: Other Pairing, Lucius/Narcissa + # Status: Completed + # First Published: 2010.09.02 + # Last Published Chapter: 2010.09.28 + # Last Updated: 2010.09.28 + # Favorite Story Of: 1 users + # Warnings: Scenes of a Mild Sexual Nature + + m = re.match(r".*?Status: Completed.*?",metastr) + if m: + self.story.setMetadata('status','Completed') + else: + self.story.setMetadata('status','In-Progress') + + m = re.match(r".*?Rating: (.+?) Story Reviews.*?",metastr) + if m: + self.story.setMetadata('rating', m.group(1)) + + m = re.match(r".*?Genre\(s\): (.+?) Era.*?",metastr) + if m: + for g in m.group(1).split(','): + self.story.addToList('genre',g) + + m = re.match(r".*?Characters: (.+?) Genre.*?",metastr) + if m: + for g in m.group(1).split(','): + self.story.addToList('characters',g) + + m = re.match(r".*?Warnings: (.+).*?",metastr) + if m: + for w in m.group(1).split(','): + if w != 'Now Warnings': + self.story.addToList('warnings',w) + + m = re.match(r".*?First Published: ([0-9\.]+).*?",metastr) + if m: + self.story.setMetadata('datePublished',makeDate(m.group(1), "%Y.%m.%d")) + + # Updated can have more than one space after it. + m = re.match(r".*?Last Updated: ([0-9\.]+).*?",metastr) + if m: + self.story.setMetadata('dateUpdated',makeDate(m.group(1), "%Y.%m.%d")) + + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + ## most adapters use BeautifulStoneSoup here, but non-Stone + ## allows nested div tags. + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'fluidtext'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) + +def getClass(): + return HarryPotterFanFictionComSiteAdapter + diff --git a/fff_internals/adapters/adapter_hennethannunnet.py b/fanficfare/adapters/adapter_hennethannunnet.py similarity index 97% rename from fff_internals/adapters/adapter_hennethannunnet.py rename to fanficfare/adapters/adapter_hennethannunnet.py index 4a5391d..bdcc441 100644 --- a/fff_internals/adapters/adapter_hennethannunnet.py +++ b/fanficfare/adapters/adapter_hennethannunnet.py @@ -1,172 +1,172 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return HennethAnnunNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class HennethAnnunNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/stories/chapter.cfm?stid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','htan') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.henneth-annun.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/stories/chapter.cfm?stid=1234" - - def getSiteURLPattern(self): - return "http://"+self.getSiteDomain()+"/stories/chapter(_view)?.cfm\?stid="+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - - if "We're sorry. This story is not available." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: This story is not available.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('h2', {'id':'page_heading'}) - self.story.setMetadata('title',stripHTML(a)) - - # Find the chapters: chapter_view.cfm?stid=6663&spordinal=1" - for chapter in soup.findAll('a', href=re.compile(r'chapter_view.cfm\?stid='+self.story.getMetadata('storyId')+"&spordinal=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/stories/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - self.story.setMetadata('numWords', soup.find('tr', {'class':'foot'}).findAll('td')[1].text) - - self.setDescription(url,soup.find('div', {'id':'summary'})) - - # Rated: NC-17
etc - info = soup.find('div', {'id':'storyinformation'}) - labels=info.findAll('b') - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Completion' in label: - if 'Complete' in value.string: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Rating' in label: - self.story.setMetadata('rating', value.string) - - if 'Era:' in label: - self.story.addToList('category',value.string) - - if 'Genre' in label: - self.story.addToList('genre',value.string) - - labels=info.findAll('strong') - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Author' in label: - value=value.nextSibling - self.story.setMetadata('authorId',value['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+value['href']) - self.story.setMetadata('author',value.string) - - if 'Post' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated:' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - for char in soup.findAll('a', href=re.compile(r"/resources/bios_view.cfm\?scid=\d+")): - self.story.addToList('characters',stripHTML(char)) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'class' : 'block chapter'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return HennethAnnunNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class HennethAnnunNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/stories/chapter.cfm?stid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','htan') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.henneth-annun.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/stories/chapter.cfm?stid=1234" + + def getSiteURLPattern(self): + return "http://"+self.getSiteDomain()+"/stories/chapter(_view)?.cfm\?stid="+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + + if "We're sorry. This story is not available." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: This story is not available.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('h2', {'id':'page_heading'}) + self.story.setMetadata('title',stripHTML(a)) + + # Find the chapters: chapter_view.cfm?stid=6663&spordinal=1" + for chapter in soup.findAll('a', href=re.compile(r'chapter_view.cfm\?stid='+self.story.getMetadata('storyId')+"&spordinal=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/stories/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + self.story.setMetadata('numWords', soup.find('tr', {'class':'foot'}).findAll('td')[1].text) + + self.setDescription(url,soup.find('div', {'id':'summary'})) + + # Rated: NC-17
etc + info = soup.find('div', {'id':'storyinformation'}) + labels=info.findAll('b') + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Completion' in label: + if 'Complete' in value.string: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Rating' in label: + self.story.setMetadata('rating', value.string) + + if 'Era:' in label: + self.story.addToList('category',value.string) + + if 'Genre' in label: + self.story.addToList('genre',value.string) + + labels=info.findAll('strong') + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Author' in label: + value=value.nextSibling + self.story.setMetadata('authorId',value['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+value['href']) + self.story.setMetadata('author',value.string) + + if 'Post' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated:' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + for char in soup.findAll('a', href=re.compile(r"/resources/bios_view.cfm\?scid=\d+")): + self.story.addToList('characters',stripHTML(char)) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'class' : 'block chapter'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_hlfictionnet.py b/fanficfare/adapters/adapter_hlfictionnet.py similarity index 97% rename from fff_internals/adapters/adapter_hlfictionnet.py rename to fanficfare/adapters/adapter_hlfictionnet.py index 96deb95..beabb75 100644 --- a/fff_internals/adapters/adapter_hlfictionnet.py +++ b/fanficfare/adapters/adapter_hlfictionnet.py @@ -1,232 +1,232 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return HLFictionNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class HLFictionNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','hlf') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'hlfiction.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title and author - a = soup.find('div', {'id' : 'pagetitle'}) - - aut = a.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',aut['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+aut['href']) - self.story.setMetadata('author',aut.string) - aut.extract() - - self.story.setMetadata('title',stripHTML(a)[:(len(a.string)-3)]) - - # Find the chapters: - chapters=soup.find('select') - if chapters != None: - for chapter in chapters.findAll('option'): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/viewstory.php?sid='+self.story.getMetadata('storyId')+'&chapter='+chapter['value'])) - else: - self.chapterUrls.append((self.story.getMetadata('title'),url)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - for list in asoup.findAll('div', {'class' : re.compile('listbox\s+')}): - a = list.find('a') - if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: - break - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = list.findAll('span', {'class' : 'classification'}) - for labelspan in labels: - label = labelspan.string - value = labelspan.nextSibling - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'classification': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value[:len(value)-2]) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'categories.php\?catid=\d+')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - for char in value.string.split(', '): - if not 'None' in char: - self.story.addToList('characters',char) - - if 'Genre' in label: - for genre in value.string.split(', '): - if not 'None' in genre: - self.story.addToList('genre',genre) - - if 'Warnings' in label: - for warning in value.string.split(', '): - if not 'None' in warning: - self.story.addToList('warnings',warning) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = list.find('a', href=re.compile(r"series.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return HLFictionNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class HLFictionNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','hlf') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'hlfiction.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title and author + a = soup.find('div', {'id' : 'pagetitle'}) + + aut = a.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',aut['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+aut['href']) + self.story.setMetadata('author',aut.string) + aut.extract() + + self.story.setMetadata('title',stripHTML(a)[:(len(a.string)-3)]) + + # Find the chapters: + chapters=soup.find('select') + if chapters != None: + for chapter in chapters.findAll('option'): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/viewstory.php?sid='+self.story.getMetadata('storyId')+'&chapter='+chapter['value'])) + else: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + for list in asoup.findAll('div', {'class' : re.compile('listbox\s+')}): + a = list.find('a') + if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: + break + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = list.findAll('span', {'class' : 'classification'}) + for labelspan in labels: + label = labelspan.string + value = labelspan.nextSibling + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'classification': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value[:len(value)-2]) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'categories.php\?catid=\d+')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + for char in value.string.split(', '): + if not 'None' in char: + self.story.addToList('characters',char) + + if 'Genre' in label: + for genre in value.string.split(', '): + if not 'None' in genre: + self.story.addToList('genre',genre) + + if 'Warnings' in label: + for warning in value.string.split(', '): + if not 'None' in warning: + self.story.addToList('warnings',warning) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = list.find('a', href=re.compile(r"series.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_hpfandomnet.py b/fanficfare/adapters/adapter_hpfandomnet.py similarity index 97% rename from fff_internals/adapters/adapter_hpfandomnet.py rename to fanficfare/adapters/adapter_hpfandomnet.py index 16a4cfc..77274d6 100644 --- a/fff_internals/adapters/adapter_hpfandomnet.py +++ b/fanficfare/adapters/adapter_hpfandomnet.py @@ -1,233 +1,233 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return HPFandomNetAdapterAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class HPFandomNetAdapterAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /eff part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/eff/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','hpfdm') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%Y.%m.%d" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.hpfandom.net' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/eff/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/eff/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/eff/'+a['href']) - self.story.setMetadata('author',a.string) - - ## Going to get the rest from the author page. - authdata = self._fetchUrl(self.story.getMetadata('authorUrl')) - # fix a typo in the site HTML so I can find the Characters list. - authdata = authdata.replace('','') - - # hpfandom.net only seems to indicate adult-only by javascript on the story/chapter links. - if "javascript:if (confirm('Slash/het fiction which incorporates sexual situations to a somewhat graphic degree and some violence. ')) location = 'viewstory.php?sid=%s'"%self.story.getMetadata('storyId') in authdata \ - and not (self.is_adult or self.getConfig("is_adult")): - raise exceptions.AdultCheckRequired(self.url) - - authsoup = bs.BeautifulSoup(authdata) - - reviewsa = authsoup.find('a', href="reviews.php?sid="+self.story.getMetadata('storyId')+"&a=") - # - # - # - labels = metablock.findAll('td',{'width':'10%'}) - for td in labels: - label = td.string - value = td.nextSibling.string - #print("\nlabel:%s\nvalue:%s\n"%(label,value)) - - if 'Category' in label and value: - cats = td.parent.findAll('a',href=re.compile(r'categories.php')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label and value: # this site can have Character label with no - # values, apparently. Others as a precaution. - for char in value.split(','): - self.story.addToList('characters',char.strip()) - - if 'Genre' in label and value: - for genre in value.split(','): - self.story.addToList('genre',genre.strip()) - - if 'Warnings' in label and value: - for warning in value.split(','): - if warning.strip() != 'none': - self.story.addToList('warnings',warning.strip()) - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - # There's no good wrapper around the chapter text. :-/ - # There are, however, tables with width=100% just above and below the real text. - data = re.sub(r'

- metablock = reviewsa.findParent("table") - #print("metablock:%s"%metablock) - - ## Title - titlea = metablock.find('a', href=re.compile("viewstory.php")) - #print("titlea:%s"%titlea) - if titlea == None: - raise exceptions.FailedToDownload("Story URL (%s) not found on author's page, can't use chapter URLs"%url) - self.story.setMetadata('title',stripHTML(titlea)) - - # Find the chapters: !!! hpfandom.net differs from every other - # eFiction site--the sid on viewstory for chapters is - # *different* for each chapter - for chapter in soup.findAll('a', {'href':re.compile(r"viewstory.php\?sid=\d+&i=\d+")}): - m = re.match(r'.*?(viewstory.php\?sid=\d+&i=\d+).*?',chapter['href']) - # just in case there's tags, like in chapter titles. - #print("====chapter===%s"%m.group(1)) - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/eff/'+m.group(1))) - - if len(self.chapterUrls) == 0: - self.chapterUrls.append((stripHTML(self.story.getMetadata('title')),url)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - summary = metablock.find("td",{"class":"summary"}) - summary.name='span' - self.setDescription(url,summary) - - # words & completed in first row of metablock. - firstrow = stripHTML(metablock.find('tr')) - # A Mother's Love xx Going Grey 1 (G+) by Kiristeen | Reviews - 18 | Words: 27468 | Completed: Yes - m = re.match(r".*?\((?P[^)]+)\).*?Words: (?P\d+).*?Completed: (?PYes|No)",firstrow) - if m != None: - if m.group('rating') != None: - self.story.setMetadata('rating', m.group('rating')) - - if m.group('words') != None: - self.story.setMetadata('numWords', m.group('words')) - - if m.group('status') != None: - if 'Yes' in m.group('status'): - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - - #

Chapters:4Published:2010.09.29
Completed:YesUpdated:2010.10.03
.*?
','
', - data,count=1,flags=re.DOTALL) - - data = re.sub(r'.*?
','
', - data,count=1,flags=re.DOTALL) - - soup = bs.BeautifulStoneSoup(data,selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find("div",{'name':'storybody'}) - #print("\n\ndiv:%s\n\n"%div) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return HPFandomNetAdapterAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class HPFandomNetAdapterAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /eff part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/eff/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','hpfdm') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y.%m.%d" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.hpfandom.net' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/eff/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/eff/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/eff/'+a['href']) + self.story.setMetadata('author',a.string) + + ## Going to get the rest from the author page. + authdata = self._fetchUrl(self.story.getMetadata('authorUrl')) + # fix a typo in the site HTML so I can find the Characters list. + authdata = authdata.replace('','') + + # hpfandom.net only seems to indicate adult-only by javascript on the story/chapter links. + if "javascript:if (confirm('Slash/het fiction which incorporates sexual situations to a somewhat graphic degree and some violence. ')) location = 'viewstory.php?sid=%s'"%self.story.getMetadata('storyId') in authdata \ + and not (self.is_adult or self.getConfig("is_adult")): + raise exceptions.AdultCheckRequired(self.url) + + authsoup = bs.BeautifulSoup(authdata) + + reviewsa = authsoup.find('a', href="reviews.php?sid="+self.story.getMetadata('storyId')+"&a=") + # + # + # + labels = metablock.findAll('td',{'width':'10%'}) + for td in labels: + label = td.string + value = td.nextSibling.string + #print("\nlabel:%s\nvalue:%s\n"%(label,value)) + + if 'Category' in label and value: + cats = td.parent.findAll('a',href=re.compile(r'categories.php')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label and value: # this site can have Character label with no + # values, apparently. Others as a precaution. + for char in value.split(','): + self.story.addToList('characters',char.strip()) + + if 'Genre' in label and value: + for genre in value.split(','): + self.story.addToList('genre',genre.strip()) + + if 'Warnings' in label and value: + for warning in value.split(','): + if warning.strip() != 'none': + self.story.addToList('warnings',warning.strip()) + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + # There's no good wrapper around the chapter text. :-/ + # There are, however, tables with width=100% just above and below the real text. + data = re.sub(r'

+ metablock = reviewsa.findParent("table") + #print("metablock:%s"%metablock) + + ## Title + titlea = metablock.find('a', href=re.compile("viewstory.php")) + #print("titlea:%s"%titlea) + if titlea == None: + raise exceptions.FailedToDownload("Story URL (%s) not found on author's page, can't use chapter URLs"%url) + self.story.setMetadata('title',stripHTML(titlea)) + + # Find the chapters: !!! hpfandom.net differs from every other + # eFiction site--the sid on viewstory for chapters is + # *different* for each chapter + for chapter in soup.findAll('a', {'href':re.compile(r"viewstory.php\?sid=\d+&i=\d+")}): + m = re.match(r'.*?(viewstory.php\?sid=\d+&i=\d+).*?',chapter['href']) + # just in case there's tags, like in chapter titles. + #print("====chapter===%s"%m.group(1)) + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/eff/'+m.group(1))) + + if len(self.chapterUrls) == 0: + self.chapterUrls.append((stripHTML(self.story.getMetadata('title')),url)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + summary = metablock.find("td",{"class":"summary"}) + summary.name='span' + self.setDescription(url,summary) + + # words & completed in first row of metablock. + firstrow = stripHTML(metablock.find('tr')) + # A Mother's Love xx Going Grey 1 (G+) by Kiristeen | Reviews - 18 | Words: 27468 | Completed: Yes + m = re.match(r".*?\((?P[^)]+)\).*?Words: (?P\d+).*?Completed: (?PYes|No)",firstrow) + if m != None: + if m.group('rating') != None: + self.story.setMetadata('rating', m.group('rating')) + + if m.group('words') != None: + self.story.setMetadata('numWords', m.group('words')) + + if m.group('status') != None: + if 'Yes' in m.group('status'): + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + + #

Chapters:4Published:2010.09.29
Completed:YesUpdated:2010.10.03
.*?
','
', + data,count=1,flags=re.DOTALL) + + data = re.sub(r'.*?
','
', + data,count=1,flags=re.DOTALL) + + soup = bs.BeautifulStoneSoup(data,selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find("div",{'name':'storybody'}) + #print("\n\ndiv:%s\n\n"%div) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_hpfanficarchivecom.py b/fanficfare/adapters/adapter_hpfanficarchivecom.py similarity index 97% rename from fff_internals/adapters/adapter_hpfanficarchivecom.py rename to fanficfare/adapters/adapter_hpfanficarchivecom.py index a149dfe..7738015 100644 --- a/fff_internals/adapters/adapter_hpfanficarchivecom.py +++ b/fanficfare/adapters/adapter_hpfanficarchivecom.py @@ -1,223 +1,223 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return HPFanficArchiveComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class HPFanficArchiveComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/stories/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','hpffa') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'hpfanficarchive.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/stories/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/stories/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/stories/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/stories/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - val = labelspan.nextSibling - value = unicode('') - while val and not defaultGetattr(val,'class') == 'label': - value += unicode(val) - val = val.nextSibling - label = labelspan.string - #print("label:%s\nvalue:%s"%(label,value)) - - if 'Summary' in label: - self.setDescription(url,value) - - if 'Rated' in label: - self.story.setMetadata('rating', stripHTML(value)) - - if 'Word count' in label: - self.story.setMetadata('numWords', stripHTML(value)) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Pairing' in label: - ships = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) - for ship in ships: - self.story.addToList('ships',ship.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in stripHTML(value): - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/stories/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return HPFanficArchiveComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class HPFanficArchiveComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/stories/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','hpffa') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'hpfanficarchive.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/stories/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/stories/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/stories/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/stories/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + val = labelspan.nextSibling + value = unicode('') + while val and not defaultGetattr(val,'class') == 'label': + value += unicode(val) + val = val.nextSibling + label = labelspan.string + #print("label:%s\nvalue:%s"%(label,value)) + + if 'Summary' in label: + self.setDescription(url,value) + + if 'Rated' in label: + self.story.setMetadata('rating', stripHTML(value)) + + if 'Word count' in label: + self.story.setMetadata('numWords', stripHTML(value)) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Pairing' in label: + ships = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) + for ship in ships: + self.story.addToList('ships',ship.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in stripHTML(value): + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/stories/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_iketernalnet.py b/fanficfare/adapters/adapter_iketernalnet.py similarity index 97% rename from fff_internals/adapters/adapter_iketernalnet.py rename to fanficfare/adapters/adapter_iketernalnet.py index 846aee7..4600f6f 100644 --- a/fff_internals/adapters/adapter_iketernalnet.py +++ b/fanficfare/adapters/adapter_iketernalnet.py @@ -1,283 +1,283 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return IkEternalNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class IkEternalNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ike') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.ik-eternal.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&warning=1" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. ksarchive uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=1882&warning=4 - # viewstory.php?sid=1654&ageconsent=ok&warning=5 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data,selfClosingTags=('p')) #poor formatting of the paragraphs in the title page - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - asoup = soup.find('div', {'class': 'listbox'}) - for a in asoup.findAll('p'): - a.name='br' - labels = asoup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return IkEternalNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class IkEternalNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ike') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.ik-eternal.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&warning=1" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. ksarchive uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=1882&warning=4 + # viewstory.php?sid=1654&ageconsent=ok&warning=5 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data,selfClosingTags=('p')) #poor formatting of the paragraphs in the title page + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + asoup = soup.find('div', {'class': 'listbox'}) + for a in asoup.findAll('p'): + a.name='br' + labels = asoup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_imagineeficcom.py b/fanficfare/adapters/adapter_imagineeficcom.py similarity index 97% rename from fff_internals/adapters/adapter_imagineeficcom.py rename to fanficfare/adapters/adapter_imagineeficcom.py index f32c584..21e1c39 100644 --- a/fff_internals/adapters/adapter_imagineeficcom.py +++ b/fanficfare/adapters/adapter_imagineeficcom.py @@ -1,290 +1,290 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return ImagineEFicComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class ImagineEFicComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ime') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%Y.%m.%d" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'imagine.e-fic.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=4" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return ImagineEFicComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class ImagineEFicComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ime') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y.%m.%d" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'imagine.e-fic.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=4" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_indeathnet.py b/fanficfare/adapters/adapter_indeathnet.py similarity index 97% rename from fff_internals/adapters/adapter_indeathnet.py rename to fanficfare/adapters/adapter_indeathnet.py index 5c0175e..1616041 100644 --- a/fff_internals/adapters/adapter_indeathnet.py +++ b/fanficfare/adapters/adapter_indeathnet.py @@ -1,200 +1,200 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return InDeathNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class InDeathNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - - # get storyId from url--url validation guarantees query correct - m = re.match(self.getSiteURLPattern(),url) - if m: - self.story.setMetadata('storyId',m.group('id')) - - # normalized story URL. - self._setURL('http://www.' + self.getSiteDomain() + '/blog/archive/'+self.story.getMetadata('storyId')+'-'+m.group('name')+'/') - else: - raise exceptions.InvalidStoryURL(url, - self.getSiteDomain(), - self.getSiteExampleURLs()) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','idn') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d %B %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'indeath.net' - - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/blog/archive/123-story-in-death/" - - def getSiteURLPattern(self): - # http://www.indeath.net/blog/archive/169-ransom-in-death/ - return re.escape("http://")+re.escape(self.getSiteDomain())+r"/blog/(archive/)?(?P\d+)\-(?P[a-z0-9\-]*)/?$" - - - def getDateFromComponents(self, postmonth, postday): - ym = re.search("Entries\ in\ (?PJanuary|February|March|April|May|June|July|August|September|October|November|December)\ (?P\d{4})",postmonth) - d = re.search("(?P\d{2})\ (Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)",postday) - postdate = makeDate(d.group('day')+' '+ym.group('mon')+' '+ym.group('year'),self.dateformat) - return postdate - - def getAuthorData(self): - - mainUrl = self.url.replace("/archive","") - - try: - maindata = self._fetchUrl(mainUrl) - - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.meta) - else: - raise e - - # use BeautifulSoup HTML parser to make everything easier to find. - mainsoup = bs.BeautifulSoup(maindata) - - # find first entry - e = mainsoup.find('div',{'class':"entry"}) - - # get post author as author - d = e.find('div',{'class':"desc"}) - a = d.find('strong') - self.story.setMetadata('author',a.contents[0].string.strip()) - - # Don't seem to be able to get author pages anymore - self.story.setMetadata('authorUrl','http://www.indeath.net/') - self.story.setMetadata('authorId','0') - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - url = self.url - try: - data = self._fetchUrl(url) - - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.meta) - else: - raise e - - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - # Now go hunting for all the meta data and the chapter list. - - ## Title - h = soup.find('a', id="blog_title") - t = h.find('span') - self.story.setMetadata('title',stripHTML(t.contents[0]).strip()) - - s = t.find('div') - if s != None: - self.setDescription(url,s) - - # Get Author from main blog page since it's not reliably on the archive page - self.getAuthorData() - - # Find the chapters: - chapters=soup.findAll('a', title="View entry", href=re.compile(r'http://www.indeath.net/blog/'+self.story.getMetadata('storyId')+"/entry\-(\d+)\-([^/]*)/$")) - - #reverse the list since newest at the top - chapters.reverse() - - # Get date published & updated from first & last entries - posttable=soup.find('div', id="main_column") - - postmonths=posttable.findAll('th', text=re.compile(r'Entries\ in\ ')) - postmonths.reverse() - - postdates=posttable.findAll('span', _class="desc", text=re.compile('\d{2}\ (Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)')) - postdates.reverse() - - self.story.setMetadata('datePublished',self.getDateFromComponents(postmonths[0],postdates[0])) - self.story.setMetadata('dateUpdated',self.getDateFromComponents(postmonths[len(postmonths)-1],postdates[len(postdates)-1])) - - # Process List of Chapters - self.story.setMetadata('numChapters',len(chapters)) - logger.debug("numChapters: (%s)"%self.story.getMetadata('numChapters')) - for x in range(0,len(chapters)): - # just in case there's tags, like in chapter titles. - chapter=chapters[x] - if len(chapters)==1: - self.chapterUrls.append((self.story.getMetadata('title'),chapter['href'])) - else: - ct = stripHTML(chapter) - tnew = re.match("(?i)"+self.story.getMetadata('title')+r" - (?P.*)$",ct) - if tnew: - chaptertitle = tnew.group('newtitle') - else: - chaptertitle = ct - self.chapterUrls.append((chaptertitle,chapter['href'])) - - - - # grab the text for an individual chapter. - def getChapterText(self, url): - logger.debug('Getting chapter text from: %s' % url) - - #chapter=bs.BeautifulSoup('
') - data = self._fetchUrl(url) - soup = bs.BeautifulSoup(data,selfClosingTags=('br','hr','span','center')) - - chapter = soup.find("div", "entry_content") - - if None == chapter: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,chapter) - +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return InDeathNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class InDeathNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + + # get storyId from url--url validation guarantees query correct + m = re.match(self.getSiteURLPattern(),url) + if m: + self.story.setMetadata('storyId',m.group('id')) + + # normalized story URL. + self._setURL('http://www.' + self.getSiteDomain() + '/blog/archive/'+self.story.getMetadata('storyId')+'-'+m.group('name')+'/') + else: + raise exceptions.InvalidStoryURL(url, + self.getSiteDomain(), + self.getSiteExampleURLs()) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','idn') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d %B %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'indeath.net' + + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/blog/archive/123-story-in-death/" + + def getSiteURLPattern(self): + # http://www.indeath.net/blog/archive/169-ransom-in-death/ + return re.escape("http://")+re.escape(self.getSiteDomain())+r"/blog/(archive/)?(?P\d+)\-(?P[a-z0-9\-]*)/?$" + + + def getDateFromComponents(self, postmonth, postday): + ym = re.search("Entries\ in\ (?PJanuary|February|March|April|May|June|July|August|September|October|November|December)\ (?P\d{4})",postmonth) + d = re.search("(?P\d{2})\ (Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)",postday) + postdate = makeDate(d.group('day')+' '+ym.group('mon')+' '+ym.group('year'),self.dateformat) + return postdate + + def getAuthorData(self): + + mainUrl = self.url.replace("/archive","") + + try: + maindata = self._fetchUrl(mainUrl) + + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.meta) + else: + raise e + + # use BeautifulSoup HTML parser to make everything easier to find. + mainsoup = bs.BeautifulSoup(maindata) + + # find first entry + e = mainsoup.find('div',{'class':"entry"}) + + # get post author as author + d = e.find('div',{'class':"desc"}) + a = d.find('strong') + self.story.setMetadata('author',a.contents[0].string.strip()) + + # Don't seem to be able to get author pages anymore + self.story.setMetadata('authorUrl','http://www.indeath.net/') + self.story.setMetadata('authorId','0') + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + url = self.url + try: + data = self._fetchUrl(url) + + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.meta) + else: + raise e + + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + # Now go hunting for all the meta data and the chapter list. + + ## Title + h = soup.find('a', id="blog_title") + t = h.find('span') + self.story.setMetadata('title',stripHTML(t.contents[0]).strip()) + + s = t.find('div') + if s != None: + self.setDescription(url,s) + + # Get Author from main blog page since it's not reliably on the archive page + self.getAuthorData() + + # Find the chapters: + chapters=soup.findAll('a', title="View entry", href=re.compile(r'http://www.indeath.net/blog/'+self.story.getMetadata('storyId')+"/entry\-(\d+)\-([^/]*)/$")) + + #reverse the list since newest at the top + chapters.reverse() + + # Get date published & updated from first & last entries + posttable=soup.find('div', id="main_column") + + postmonths=posttable.findAll('th', text=re.compile(r'Entries\ in\ ')) + postmonths.reverse() + + postdates=posttable.findAll('span', _class="desc", text=re.compile('\d{2}\ (Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)')) + postdates.reverse() + + self.story.setMetadata('datePublished',self.getDateFromComponents(postmonths[0],postdates[0])) + self.story.setMetadata('dateUpdated',self.getDateFromComponents(postmonths[len(postmonths)-1],postdates[len(postdates)-1])) + + # Process List of Chapters + self.story.setMetadata('numChapters',len(chapters)) + logger.debug("numChapters: (%s)"%self.story.getMetadata('numChapters')) + for x in range(0,len(chapters)): + # just in case there's tags, like in chapter titles. + chapter=chapters[x] + if len(chapters)==1: + self.chapterUrls.append((self.story.getMetadata('title'),chapter['href'])) + else: + ct = stripHTML(chapter) + tnew = re.match("(?i)"+self.story.getMetadata('title')+r" - (?P.*)$",ct) + if tnew: + chaptertitle = tnew.group('newtitle') + else: + chaptertitle = ct + self.chapterUrls.append((chaptertitle,chapter['href'])) + + + + # grab the text for an individual chapter. + def getChapterText(self, url): + logger.debug('Getting chapter text from: %s' % url) + + #chapter=bs.BeautifulSoup('
') + data = self._fetchUrl(url) + soup = bs.BeautifulSoup(data,selfClosingTags=('br','hr','span','center')) + + chapter = soup.find("div", "entry_content") + + if None == chapter: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,chapter) + diff --git a/fff_internals/adapters/adapter_ksarchivecom.py b/fanficfare/adapters/adapter_ksarchivecom.py similarity index 97% rename from fff_internals/adapters/adapter_ksarchivecom.py rename to fanficfare/adapters/adapter_ksarchivecom.py index e8073af..92ccabd 100644 --- a/fff_internals/adapters/adapter_ksarchivecom.py +++ b/fanficfare/adapters/adapter_ksarchivecom.py @@ -1,330 +1,330 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return KSArchiveComAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class KSArchiveComAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ksa') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b/%d/%Y" # XXX - - @classmethod - def getAcceptDomains(cls): - return ['www.ksarchive.com','ksarchive.com'] - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'ksarchive.com' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - - # Furthermore, there's a couple sites now with more than - # one warning level for different ratings. And they're - # fussy about it. midnightwhispers has three: 10, 3 & 5. - # we'll try 5 first. - addurl = "&ageconsent=ok&warning=2" # XXX - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. ksarchive uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=1882&warning=4 - # viewstory.php?sid=1654&ageconsent=ok&warning=5 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) # title's inside a tag. - - # Find authorid and URL from... author urls. - pagetitle = soup.find('div',id='pagetitle') - for a in pagetitle.findAll('a', href=re.compile(r"viewuser.php\?uid=\d+")): - self.story.addToList('authorId',a['href'].split('=')[1]) - self.story.addToList('authorUrl','http://'+self.host+'/'+a['href']) - self.story.addToList('author',stripHTML(a)) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = stripHTML(labelspan) - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - # poor HTML(unclosed

for one) can cause run on - # over the next label. - if '' in svalue: - svalue = svalue[0:svalue.find('')] - break - else: - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [stripHTML(cat) for cat in cats] - for cat in catstext: - # ran across one story with an empty - # tag in the desc once. - if cat and cat.strip() in ('Poetry','Essays'): - self.story.addToList('category',stripHTML(cat)) - - if 'Characters' in label: - self.story.addToList('characters','Kirk') - self.story.addToList('characters','Spock') - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [stripHTML(char) for char in chars] - for char in charstext: - self.story.addToList('characters',stripHTML(char)) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX - genrestext = [stripHTML(genre) for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',stripHTML(genre)) - - ## In addition to Genre (which is very site specific) KSA - ## has 'Story Type', which is much more what most sites - ## call genre. - if 'Story Type' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=5')) # XXX - genrestext = [stripHTML(genre) for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',stripHTML(genre)) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [stripHTML(warning) for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',stripHTML(warning)) - - if 'Universe' in label: - universes = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) # XXX - universestext = [stripHTML(universe) for universe in universes] - self.universe = ', '.join(universestext) - for universe in universestext: - self.story.addToList('universe',stripHTML(universe)) - - if 'Crossover Fandom' in label: - crossoverfandoms = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) # XXX - crossoverfandomstext = [stripHTML(crossoverfandom) for crossoverfandom in crossoverfandoms] - self.crossoverfandom = ', '.join(crossoverfandomstext) - for crossoverfandom in crossoverfandomstext: - self.story.addToList('crossoverfandom',stripHTML(crossoverfandom)) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = stripHTML(a) - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - soup = bs.BeautifulStoneSoup(data, - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - if "A fatal MySQL error was encountered" in data: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Database error on the site reported!" % url) - else: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return KSArchiveComAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class KSArchiveComAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ksa') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b/%d/%Y" # XXX + + @classmethod + def getAcceptDomains(cls): + return ['www.ksarchive.com','ksarchive.com'] + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'ksarchive.com' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + + # Furthermore, there's a couple sites now with more than + # one warning level for different ratings. And they're + # fussy about it. midnightwhispers has three: 10, 3 & 5. + # we'll try 5 first. + addurl = "&ageconsent=ok&warning=2" # XXX + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. ksarchive uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=1882&warning=4 + # viewstory.php?sid=1654&ageconsent=ok&warning=5 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) # title's inside a tag. + + # Find authorid and URL from... author urls. + pagetitle = soup.find('div',id='pagetitle') + for a in pagetitle.findAll('a', href=re.compile(r"viewuser.php\?uid=\d+")): + self.story.addToList('authorId',a['href'].split('=')[1]) + self.story.addToList('authorUrl','http://'+self.host+'/'+a['href']) + self.story.addToList('author',stripHTML(a)) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = stripHTML(labelspan) + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + # poor HTML(unclosed

for one) can cause run on + # over the next label. + if '' in svalue: + svalue = svalue[0:svalue.find('')] + break + else: + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [stripHTML(cat) for cat in cats] + for cat in catstext: + # ran across one story with an empty + # tag in the desc once. + if cat and cat.strip() in ('Poetry','Essays'): + self.story.addToList('category',stripHTML(cat)) + + if 'Characters' in label: + self.story.addToList('characters','Kirk') + self.story.addToList('characters','Spock') + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [stripHTML(char) for char in chars] + for char in charstext: + self.story.addToList('characters',stripHTML(char)) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX + genrestext = [stripHTML(genre) for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',stripHTML(genre)) + + ## In addition to Genre (which is very site specific) KSA + ## has 'Story Type', which is much more what most sites + ## call genre. + if 'Story Type' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=5')) # XXX + genrestext = [stripHTML(genre) for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',stripHTML(genre)) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [stripHTML(warning) for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',stripHTML(warning)) + + if 'Universe' in label: + universes = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) # XXX + universestext = [stripHTML(universe) for universe in universes] + self.universe = ', '.join(universestext) + for universe in universestext: + self.story.addToList('universe',stripHTML(universe)) + + if 'Crossover Fandom' in label: + crossoverfandoms = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=4')) # XXX + crossoverfandomstext = [stripHTML(crossoverfandom) for crossoverfandom in crossoverfandoms] + self.crossoverfandom = ', '.join(crossoverfandomstext) + for crossoverfandom in crossoverfandomstext: + self.story.addToList('crossoverfandom',stripHTML(crossoverfandom)) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = stripHTML(a) + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + soup = bs.BeautifulStoneSoup(data, + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + if "A fatal MySQL error was encountered" in data: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Database error on the site reported!" % url) + else: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_libraryofmoriacom.py b/fanficfare/adapters/adapter_libraryofmoriacom.py similarity index 97% rename from fff_internals/adapters/adapter_libraryofmoriacom.py rename to fanficfare/adapters/adapter_libraryofmoriacom.py index 3df41bd..8c01030 100644 --- a/fff_internals/adapters/adapter_libraryofmoriacom.py +++ b/fanficfare/adapters/adapter_libraryofmoriacom.py @@ -1,251 +1,251 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - - -def getClass(): - return LibraryOfMoriaComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class LibraryOfMoriaComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/a/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','lom') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.libraryofmoria.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/a/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/a/viewstory.php?sid=")+r"\d+$" - - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - addurl = "&ageconsent=ok&warning=3" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/a/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/a/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - if 'Type' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warning' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=5')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/a/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + + +def getClass(): + return LibraryOfMoriaComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class LibraryOfMoriaComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/a/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','lom') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.libraryofmoria.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/a/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/a/viewstory.php?sid=")+r"\d+$" + + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + addurl = "&ageconsent=ok&warning=3" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/a/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/a/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + if 'Type' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warning' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=5')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/a/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_literotica.py b/fanficfare/adapters/adapter_literotica.py similarity index 100% rename from fff_internals/adapters/adapter_literotica.py rename to fanficfare/adapters/adapter_literotica.py diff --git a/fff_internals/adapters/adapter_lotrfanfictioncom.py b/fanficfare/adapters/adapter_lotrfanfictioncom.py similarity index 96% rename from fff_internals/adapters/adapter_lotrfanfictioncom.py rename to fanficfare/adapters/adapter_lotrfanfictioncom.py index b20c2ec..4e04ed0 100644 --- a/fff_internals/adapters/adapter_lotrfanfictioncom.py +++ b/fanficfare/adapters/adapter_lotrfanfictioncom.py @@ -1,36 +1,36 @@ -# -*- coding: utf-8 -*- - -# Copyright 2014 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -from base_efiction_adapter import BaseEfictionAdapter - -class TheLOTRFanFictionSiteAdapter(BaseEfictionAdapter): - - @staticmethod - def getSiteDomain(): - return 'lotrfanfiction.com' - - @classmethod - def getSiteAbbrev(seluuf): - return 'lotrff' - - @classmethod - def getDateFormat(self): - return "%d/%m/%y" - -def getClass(): - return TheLOTRFanFictionSiteAdapter +# -*- coding: utf-8 -*- + +# Copyright 2014 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +from base_efiction_adapter import BaseEfictionAdapter + +class TheLOTRFanFictionSiteAdapter(BaseEfictionAdapter): + + @staticmethod + def getSiteDomain(): + return 'lotrfanfiction.com' + + @classmethod + def getSiteAbbrev(seluuf): + return 'lotrff' + + @classmethod + def getDateFormat(self): + return "%d/%m/%y" + +def getClass(): + return TheLOTRFanFictionSiteAdapter diff --git a/fff_internals/adapters/adapter_lumossycophanthexcom.py b/fanficfare/adapters/adapter_lumossycophanthexcom.py similarity index 97% rename from fff_internals/adapters/adapter_lumossycophanthexcom.py rename to fanficfare/adapters/adapter_lumossycophanthexcom.py index 805d752..58fecf8 100644 --- a/fff_internals/adapters/adapter_lumossycophanthexcom.py +++ b/fanficfare/adapters/adapter_lumossycophanthexcom.py @@ -1,238 +1,238 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return LumosSycophantHexComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class LumosSycophantHexComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','lsph') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'lumos.sycophanthex.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=19" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "Age Consent Required" in data: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - pt = soup.find('div', {'id' : 'pagetitle'}) - a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - rating=pt.text.split('(')[1].split(')')[0] - self.story.setMetadata('rating', rating) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - - # Rated: NC-17
etc - - labels = soup.findAll('span',{'class':'label'}) - - value = labels[0].previousSibling - svalue = "" - while value != None: - val = value - value = value.previousSibling - while not defaultGetattr(val,'class') == 'label': - svalue += str(val) - val = val.nextSibling - self.setDescription(url,svalue) - - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Word count' in label: - self.story.setMetadata('numWords', value.split(' -')[0]) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Complete' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return LumosSycophantHexComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class LumosSycophantHexComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','lsph') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'lumos.sycophanthex.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=19" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "Age Consent Required" in data: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + pt = soup.find('div', {'id' : 'pagetitle'}) + a = pt.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pt.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + rating=pt.text.split('(')[1].split(')')[0] + self.story.setMetadata('rating', rating) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + + # Rated: NC-17
etc + + labels = soup.findAll('span',{'class':'label'}) + + value = labels[0].previousSibling + svalue = "" + while value != None: + val = value + value = value.previousSibling + while not defaultGetattr(val,'class') == 'label': + svalue += str(val) + val = val.nextSibling + self.setDescription(url,svalue) + + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Word count' in label: + self.story.setMetadata('numWords', value.split(' -')[0]) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Complete' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value.split(' -')[0]), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_mediaminerorg.py b/fanficfare/adapters/adapter_mediaminerorg.py similarity index 97% rename from fff_internals/adapters/adapter_mediaminerorg.py rename to fanficfare/adapters/adapter_mediaminerorg.py index 261b1f7..4d442d9 100644 --- a/fff_internals/adapters/adapter_mediaminerorg.py +++ b/fanficfare/adapters/adapter_mediaminerorg.py @@ -1,237 +1,237 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -class MediaMinerOrgSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','mm') - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - - # get storyId from url--url validation guarantees query correct - m = re.match(self.getSiteURLPattern(),url) - if m: - self.story.setMetadata('storyId',m.group('id')) - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/fanfic/view_st.php/'+self.story.getMetadata('storyId')) - else: - raise exceptions.InvalidStoryURL(url, - self.getSiteDomain(), - self.getSiteExampleURLs()) - - @staticmethod - def getSiteDomain(): - return 'www.mediaminer.org' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fanfic/view_st.php/123456 http://"+cls.getSiteDomain()+"/fanfic/view_ch.php/1234123/123444#fic_c" - - def getSiteURLPattern(self): - ## http://www.mediaminer.org/fanfic/view_st.php/76882 - ## http://www.mediaminer.org/fanfic/view_ch.php/167618/594087#fic_c - return re.escape("http://"+self.getSiteDomain())+\ - "/fanfic/view_(st|ch)\.php/"+r"(?P\d+)(/\d+(#fic_c)?)?$" - - def extractChapterUrlsAndMetadata(self): - - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - # [ A - All Readers ], strip '[' ']' - ## Above title because we remove the smtxt font to get title. - smtxt = soup.find("font",{"class":"smtxt"}) - if not smtxt: - raise exceptions.StoryDoesNotExist(self.url) - rating = smtxt.string[1:-1] - self.story.setMetadata('rating',rating) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"/fanfic/src.php/u/\d+")) - self.story.setMetadata('authorId',a['href'].split('/')[-1]) - self.story.setMetadata('authorUrl','http://'+self.host+a['href']) - self.story.setMetadata('author',a.string) - - ## Title - Good grief. Title varies by chaptered, 1chapter and 'type=one shot'--and even 'one-shot's can have titled chapter. - ## But, if colspan=2, there's no chapter title. - ## Atmosphere: Chapter 1
[ P - Pre-Teen ] - ## Hearts of Ice [ P - Pre-Teen ] - ## Suzaku no Princess [ P - Pre-Teen ] - ## The Kraut, The Bartender, and The Drunkard: Chapter 1
[ P - Pre-Teen ] - ## Betrayal and Justice: A Cold Heart ( Chapter 1 ) [ A - All Readers ] - ## Question and Answer: Question and Answer ( One-Shot ) [ A - All Readers ] - title = soup.find('td',{'class':'ffh'}) - for font in title.findAll('font'): - font.extract() # removes 'font' tags from inside the td. - if title.has_key('colspan'): - titlet = stripHTML(title) - else: - ## No colspan, it's part chapter title--even if it's a one-shot. - titlet = ':'.join(stripHTML(title).split(':')[:-1]) # strip trailing 'Chapter X' or chapter title - self.story.setMetadata('title',titlet) - ## The story title is difficult to reliably parse from the - ## story pages. Getting it from the author page is, but costs - ## another fetch. - # authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - # titlea = authsoup.find('a',{'href':'/fanfic/view_st.php/'+self.story.getMetadata('storyId')}) - # self.story.setMetadata('title',titlea.text) - - # save date from first for later. - firstdate=None - - # Find the chapters - select = soup.find('select',{'name':'cid'}) - if not select: - self.chapterUrls.append(( self.story.getMetadata('title'),self.url)) - else: - for option in select.findAll("option"): - chapter = stripHTML(option.string) - ## chapter can be: Chapter 7 [Jan 23, 2011] - ## or: Vigilant Moonlight ( Chapter 1 ) [Jan 30, 2004] - ## or even: Prologue ( Prologue ) [Jul 31, 2010] - m = re.match(r'^(.*?) (\( .*? \) )?\[(.*?)\]$',chapter) - chapter = m.group(1) - # save date from first for later. - if not firstdate: - firstdate = m.group(3) - self.chapterUrls.append((chapter,'http://'+self.host+'/fanfic/view_ch.php/'+self.story.getMetadata('storyId')+'/'+option['value'])) - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # category - # Ranma 1/2 - for a in soup.findAll('a',href=re.compile(r"^/fanfic/src.php/a/")): - self.story.addToList('category',a.string) - - # genre - # Ranma 1/2 - for a in soup.findAll('a',href=re.compile(r"^/fanfic/src.php/g/")): - self.story.addToList('genre',a.string) - - # if firstdate, then the block below will only have last updated. - if firstdate: - self.story.setMetadata('datePublished', makeDate(firstdate, "%b %d, %Y")) - # Everything else is in - - metastr = stripHTML(soup.find("tr",{"bgcolor":"#EEEED4"})).replace('\n',' ').replace('\r',' ').replace('\t',' ') - # Latest Revision: August 03, 2010 - m = re.match(r".*?(?:Latest Revision|Uploaded On): ([a-zA-Z]+ \d\d, \d\d\d\d)",metastr) - if m: - self.story.setMetadata('dateUpdated', makeDate(m.group(1), "%B %d, %Y")) - if not firstdate: - self.story.setMetadata('datePublished', - self.story.getMetadataRaw('dateUpdated')) - - else: - self.story.setMetadata('dateUpdated', - self.story.getMetadataRaw('datePublished')) - - # Words: 123456 - m = re.match(r".*?\| Words: (\d+) \|",metastr) - if m: - self.story.setMetadata('numWords', m.group(1)) - - # Summary: .... - m = re.match(r".*?Summary: (.*)$",metastr) - if m: - self.setDescription(url, m.group(1)) - #self.story.setMetadata('description', m.group(1)) - - # completed - m = re.match(r".*?Status: Completed.*?",metastr) - if m: - self.story.setMetadata('status','Completed') - else: - self.story.setMetadata('status','In-Progress') - - return - - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data=self._fetchUrl(url) - soup = bs.BeautifulStoneSoup(data, - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - anchor = soup.find('a',{'name':'fic_c'}) - - if None == anchor: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - ## find divs with align=left, those are paragraphs in newer stories. - divlist = anchor.findAllNext('div',{'align':'left'}) - if divlist: - for div in divlist: - div.name='p' # convert to

mediaminer uses div with - # a margin for paragraphs. - anchor.append(div) # cheat! stuff all the content - # divs into anchor just as a - # holder. - del div['style'] - del div['align'] - anchor.name='div' - return self.utf8FromSoup(url,anchor) - - else: - logger.debug('Using kludgey text find for older mediaminer story.') - ## Some older mediaminer stories are unparsable with BeautifulSoup. - ## Really nasty formatting. Sooo... Cheat! Parse it ourselves a bit first. - ## Story stuff falls between: - data = "

" + data[data.find(''):] +"
" - soup = bs.BeautifulStoneSoup(data, - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - for tag in soup.findAll('td',{'class':'ffh'}) + \ - soup.findAll('div',{'class':'acl'}) + \ - soup.findAll('div',{'class':'footer smtxt'}) + \ - soup.findAll('table',{'class':'tbbrdr'}): - tag.extract() # remove tag from soup. - - return self.utf8FromSoup(url,soup) - - -def getClass(): - return MediaMinerOrgSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +class MediaMinerOrgSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','mm') + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + + # get storyId from url--url validation guarantees query correct + m = re.match(self.getSiteURLPattern(),url) + if m: + self.story.setMetadata('storyId',m.group('id')) + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/fanfic/view_st.php/'+self.story.getMetadata('storyId')) + else: + raise exceptions.InvalidStoryURL(url, + self.getSiteDomain(), + self.getSiteExampleURLs()) + + @staticmethod + def getSiteDomain(): + return 'www.mediaminer.org' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fanfic/view_st.php/123456 http://"+cls.getSiteDomain()+"/fanfic/view_ch.php/1234123/123444#fic_c" + + def getSiteURLPattern(self): + ## http://www.mediaminer.org/fanfic/view_st.php/76882 + ## http://www.mediaminer.org/fanfic/view_ch.php/167618/594087#fic_c + return re.escape("http://"+self.getSiteDomain())+\ + "/fanfic/view_(st|ch)\.php/"+r"(?P\d+)(/\d+(#fic_c)?)?$" + + def extractChapterUrlsAndMetadata(self): + + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + # [ A - All Readers ], strip '[' ']' + ## Above title because we remove the smtxt font to get title. + smtxt = soup.find("font",{"class":"smtxt"}) + if not smtxt: + raise exceptions.StoryDoesNotExist(self.url) + rating = smtxt.string[1:-1] + self.story.setMetadata('rating',rating) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"/fanfic/src.php/u/\d+")) + self.story.setMetadata('authorId',a['href'].split('/')[-1]) + self.story.setMetadata('authorUrl','http://'+self.host+a['href']) + self.story.setMetadata('author',a.string) + + ## Title - Good grief. Title varies by chaptered, 1chapter and 'type=one shot'--and even 'one-shot's can have titled chapter. + ## But, if colspan=2, there's no chapter title. + ## Atmosphere: Chapter 1 [ P - Pre-Teen ] + ## Hearts of Ice [ P - Pre-Teen ] + ## Suzaku no Princess [ P - Pre-Teen ] + ## The Kraut, The Bartender, and The Drunkard: Chapter 1 [ P - Pre-Teen ] + ## Betrayal and Justice: A Cold Heart ( Chapter 1 ) [ A - All Readers ] + ## Question and Answer: Question and Answer ( One-Shot ) [ A - All Readers ] + title = soup.find('td',{'class':'ffh'}) + for font in title.findAll('font'): + font.extract() # removes 'font' tags from inside the td. + if title.has_key('colspan'): + titlet = stripHTML(title) + else: + ## No colspan, it's part chapter title--even if it's a one-shot. + titlet = ':'.join(stripHTML(title).split(':')[:-1]) # strip trailing 'Chapter X' or chapter title + self.story.setMetadata('title',titlet) + ## The story title is difficult to reliably parse from the + ## story pages. Getting it from the author page is, but costs + ## another fetch. + # authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + # titlea = authsoup.find('a',{'href':'/fanfic/view_st.php/'+self.story.getMetadata('storyId')}) + # self.story.setMetadata('title',titlea.text) + + # save date from first for later. + firstdate=None + + # Find the chapters + select = soup.find('select',{'name':'cid'}) + if not select: + self.chapterUrls.append(( self.story.getMetadata('title'),self.url)) + else: + for option in select.findAll("option"): + chapter = stripHTML(option.string) + ## chapter can be: Chapter 7 [Jan 23, 2011] + ## or: Vigilant Moonlight ( Chapter 1 ) [Jan 30, 2004] + ## or even: Prologue ( Prologue ) [Jul 31, 2010] + m = re.match(r'^(.*?) (\( .*? \) )?\[(.*?)\]$',chapter) + chapter = m.group(1) + # save date from first for later. + if not firstdate: + firstdate = m.group(3) + self.chapterUrls.append((chapter,'http://'+self.host+'/fanfic/view_ch.php/'+self.story.getMetadata('storyId')+'/'+option['value'])) + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # category + # Ranma 1/2 + for a in soup.findAll('a',href=re.compile(r"^/fanfic/src.php/a/")): + self.story.addToList('category',a.string) + + # genre + # Ranma 1/2 + for a in soup.findAll('a',href=re.compile(r"^/fanfic/src.php/g/")): + self.story.addToList('genre',a.string) + + # if firstdate, then the block below will only have last updated. + if firstdate: + self.story.setMetadata('datePublished', makeDate(firstdate, "%b %d, %Y")) + # Everything else is in + + metastr = stripHTML(soup.find("tr",{"bgcolor":"#EEEED4"})).replace('\n',' ').replace('\r',' ').replace('\t',' ') + # Latest Revision: August 03, 2010 + m = re.match(r".*?(?:Latest Revision|Uploaded On): ([a-zA-Z]+ \d\d, \d\d\d\d)",metastr) + if m: + self.story.setMetadata('dateUpdated', makeDate(m.group(1), "%B %d, %Y")) + if not firstdate: + self.story.setMetadata('datePublished', + self.story.getMetadataRaw('dateUpdated')) + + else: + self.story.setMetadata('dateUpdated', + self.story.getMetadataRaw('datePublished')) + + # Words: 123456 + m = re.match(r".*?\| Words: (\d+) \|",metastr) + if m: + self.story.setMetadata('numWords', m.group(1)) + + # Summary: .... + m = re.match(r".*?Summary: (.*)$",metastr) + if m: + self.setDescription(url, m.group(1)) + #self.story.setMetadata('description', m.group(1)) + + # completed + m = re.match(r".*?Status: Completed.*?",metastr) + if m: + self.story.setMetadata('status','Completed') + else: + self.story.setMetadata('status','In-Progress') + + return + + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data=self._fetchUrl(url) + soup = bs.BeautifulStoneSoup(data, + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + anchor = soup.find('a',{'name':'fic_c'}) + + if None == anchor: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + ## find divs with align=left, those are paragraphs in newer stories. + divlist = anchor.findAllNext('div',{'align':'left'}) + if divlist: + for div in divlist: + div.name='p' # convert to

mediaminer uses div with + # a margin for paragraphs. + anchor.append(div) # cheat! stuff all the content + # divs into anchor just as a + # holder. + del div['style'] + del div['align'] + anchor.name='div' + return self.utf8FromSoup(url,anchor) + + else: + logger.debug('Using kludgey text find for older mediaminer story.') + ## Some older mediaminer stories are unparsable with BeautifulSoup. + ## Really nasty formatting. Sooo... Cheat! Parse it ourselves a bit first. + ## Story stuff falls between: + data = "

" + data[data.find(''):] +"
" + soup = bs.BeautifulStoneSoup(data, + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + for tag in soup.findAll('td',{'class':'ffh'}) + \ + soup.findAll('div',{'class':'acl'}) + \ + soup.findAll('div',{'class':'footer smtxt'}) + \ + soup.findAll('table',{'class':'tbbrdr'}): + tag.extract() # remove tag from soup. + + return self.utf8FromSoup(url,soup) + + +def getClass(): + return MediaMinerOrgSiteAdapter + diff --git a/fff_internals/adapters/adapter_merlinficdtwinscouk.py b/fanficfare/adapters/adapter_merlinficdtwinscouk.py similarity index 97% rename from fff_internals/adapters/adapter_merlinficdtwinscouk.py rename to fanficfare/adapters/adapter_merlinficdtwinscouk.py index ea518ea..47b8f5d 100644 --- a/fff_internals/adapters/adapter_merlinficdtwinscouk.py +++ b/fanficfare/adapters/adapter_merlinficdtwinscouk.py @@ -1,294 +1,294 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return MerlinFicDtwinsCoUk - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class MerlinFicDtwinsCoUk(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','mrfd') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%b %d, %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'merlinfic.dtwins.co.uk' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=4" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Pairing' in label: - ships = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for ship in ships: - self.story.addToList('ships',ship.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return MerlinFicDtwinsCoUk + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class MerlinFicDtwinsCoUk(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','mrfd') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%b %d, %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'merlinfic.dtwins.co.uk' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=4" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Pairing' in label: + ships = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for ship in ships: + self.story.addToList('ships',ship.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_midnightwhispersca.py b/fanficfare/adapters/adapter_midnightwhispersca.py similarity index 97% rename from fff_internals/adapters/adapter_midnightwhispersca.py rename to fanficfare/adapters/adapter_midnightwhispersca.py index ec884d6..e74ab0c 100644 --- a/fff_internals/adapters/adapter_midnightwhispersca.py +++ b/fanficfare/adapters/adapter_midnightwhispersca.py @@ -1,290 +1,290 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return MidnightwhispersCaAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class MidnightwhispersCaAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','mw') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d, %Y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.midnightwhispers.ca' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - - # Furthermore, there's a couple sites now with more than - # one warning level for different ratings. And they're - # fussy about it. midnightwhispers has three: 10, 3 & 5. - # we'll try 5 first. - addurl = "&ageconsent=ok&warning=5" # XXX - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. nfacommunity uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=1882&warning=4 - # viewstory.php?sid=1654&ageconsent=ok&warning=5 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) # title's inside a tag. - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - soup = bs.BeautifulStoneSoup(data, - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - if "A fatal MySQL error was encountered" in data: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Database error on the site reported!" % url) - else: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return MidnightwhispersCaAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class MidnightwhispersCaAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','mw') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d, %Y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.midnightwhispers.ca' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + + # Furthermore, there's a couple sites now with more than + # one warning level for different ratings. And they're + # fussy about it. midnightwhispers has three: 10, 3 & 5. + # we'll try 5 first. + addurl = "&ageconsent=ok&warning=5" # XXX + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. nfacommunity uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=1882&warning=4 + # viewstory.php?sid=1654&ageconsent=ok&warning=5 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) # title's inside a tag. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + soup = bs.BeautifulStoneSoup(data, + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + if "A fatal MySQL error was encountered" in data: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Database error on the site reported!" % url) + else: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_mugglenetcom.py b/fanficfare/adapters/adapter_mugglenetcom.py similarity index 97% rename from fff_internals/adapters/adapter_mugglenetcom.py rename to fanficfare/adapters/adapter_mugglenetcom.py index 261a623..b37b657 100644 --- a/fff_internals/adapters/adapter_mugglenetcom.py +++ b/fanficfare/adapters/adapter_mugglenetcom.py @@ -1,336 +1,336 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return MuggleNetComAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class MuggleNetComAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','mgln') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. - return 'fanfiction.mugglenet.com' - - @classmethod - def getAcceptDomains(cls): - return ['fanfiction.mugglenet.com','fanfic.mugglenet.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+r"fanfic(tion)?\.mugglenet\.com"+re.escape("/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if "class='errortext'>Registered Users Only" in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login&sid='+self.story.getMetadata('storyId') - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - # http://fanfiction.mugglenet.com/viewstory.php?sid=91079&ageconsent=ok&warning=3 - addurl = "&ageconsent=ok&warning=3" # XXX &warning=5 - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - #print("\nurl:%s\ndata:\n%s\n"%(url,data)) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. nfacommunity uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=1882&warning=4 - # viewstory.php?sid=1654&ageconsent=ok&warning=5 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=%s((?:&ageconsent=ok)?&warning=\d+)'"%self.story.getMetadata('storyId'),data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - - # Not good enough-- content can contain a ('), which ends the content prematurely. - # metadesc = soup.find('meta',{'name':'description'}) - # print("removeAllEntities(metadesc['content']):\n%s\n"%removeAllEntities(metadesc['content'])) - start='Summary: ' - end='Rated:' - summarydata = data[data.index(start)+len(start):data.index(end)] - #print("summarydata:\n%s\n"%summarydata) - self.setDescription(url,bs.BeautifulSoup(summarydata)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - # not good enough--poorly formated summary html will break it. - # if 'Summary' in label: - # ## Everything until the next span class='label' - # svalue = "" - # while not defaultGetattr(value,'class') == 'label': - # svalue += str(value) - # value = value.nextSibling - # self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return MuggleNetComAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class MuggleNetComAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','mgln') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. + return 'fanfiction.mugglenet.com' + + @classmethod + def getAcceptDomains(cls): + return ['fanfiction.mugglenet.com','fanfic.mugglenet.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+r"fanfic(tion)?\.mugglenet\.com"+re.escape("/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if "class='errortext'>Registered Users Only" in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login&sid='+self.story.getMetadata('storyId') + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + # http://fanfiction.mugglenet.com/viewstory.php?sid=91079&ageconsent=ok&warning=3 + addurl = "&ageconsent=ok&warning=3" # XXX &warning=5 + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + #print("\nurl:%s\ndata:\n%s\n"%(url,data)) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. nfacommunity uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=1882&warning=4 + # viewstory.php?sid=1654&ageconsent=ok&warning=5 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=%s((?:&ageconsent=ok)?&warning=\d+)'"%self.story.getMetadata('storyId'),data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + + # Not good enough-- content can contain a ('), which ends the content prematurely. + # metadesc = soup.find('meta',{'name':'description'}) + # print("removeAllEntities(metadesc['content']):\n%s\n"%removeAllEntities(metadesc['content'])) + start='Summary: ' + end='Rated:' + summarydata = data[data.index(start)+len(start):data.index(end)] + #print("summarydata:\n%s\n"%summarydata) + self.setDescription(url,bs.BeautifulSoup(summarydata)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + # not good enough--poorly formated summary html will break it. + # if 'Summary' in label: + # ## Everything until the next span class='label' + # svalue = "" + # while not defaultGetattr(value,'class') == 'label': + # svalue += str(value) + # value = value.nextSibling + # self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_nationallibrarynet.py b/fanficfare/adapters/adapter_nationallibrarynet.py similarity index 97% rename from fff_internals/adapters/adapter_nationallibrarynet.py rename to fanficfare/adapters/adapter_nationallibrarynet.py index c9d7531..f966f09 100644 --- a/fff_internals/adapters/adapter_nationallibrarynet.py +++ b/fanficfare/adapters/adapter_nationallibrarynet.py @@ -1,213 +1,213 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NationalLibraryNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NationalLibraryNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only storyid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?storyid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ntlb') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m-%d-%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - return 'national-library.net' - - @classmethod - def getAcceptDomains(cls): - return ['www.national-library.net','national-library.net'] - - @classmethod - def getSiteExampleURLs(cls): - # ONLY the stories archived on or after June 17, 2006 and that are hosted on the website: - return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('h1') - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"authorresults.php\?author=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for p in soup.findAll('p'): - chapters = p.findAll('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId')+"&chapnum=\d+$")) - if len(chapters) > 0: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - break - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - self.story.setMetadata('status', 'Completed') - - # Rated: NC-17
etc - labels = soup.findAll('b') - for x in range(2,len(labels)): - value = labels[x].nextSibling - label = labels[x].string - - if 'Summary' in label: - self.setDescription(url,value) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rating' in label: - self.story.setMetadata('rating', stripHTML(value.nextSibling)) - - if 'Word Count' in label: - self.story.setMetadata('numWords', value.string) - - if 'Category' in label: - for cat in value.string.split(', '): - self.story.addToList('category',cat) - if 'Crossover Shows' in label: - for cat in value.string.split(', '): - if "No Show" not in cat: - self.story.addToList('category',cat) - - if 'Character' in label: - for char in value.string.split(', '): - self.story.addToList('characters',char) - - if 'Pairing' in label: - for char in value.string.split(', '): - self.story.addToList('ships',char) - - if 'Warnings' in label: - for warning in value.string.split(', '): - self.story.addToList('warnings',warning) - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Series' in label: - self.setSeries(stripHTML(value.nextSibling), value.nextSibling.nextSibling.string[2:]) - self.story.setMetadata('seriesUrl','http://'+self.host+'/'+value.nextSibling['href']) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - story=asoup.find('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId'))) - - a=story.findNext(text=re.compile('Genre')).parent.nextSibling.string.split(', ') - for genre in a: - self.story.setMetadata('genre', genre) - - a=story.findNext(text=re.compile('Archived')) - self.story.setMetadata('datePublished', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div') - - # bit messy since higly inconsistent - for p in soup.findAll('p', {'align' : 'center'}): - p.extract() - p = soup.findAll('p') - for x in range(0,3): - p[x].extract() - if "Chapters: " in stripHTML(p[3]): - p[3].extract() - for x in range(len(p)-2,len(p)-1): - p[x].extract() - - for p in soup.findAll('h1'): - p.extract() - for p in soup.findAll('h3'): - p.extract() - for p in soup.findAll('a'): - p.extract() - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NationalLibraryNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NationalLibraryNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only storyid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?storyid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ntlb') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m-%d-%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + return 'national-library.net' + + @classmethod + def getAcceptDomains(cls): + return ['www.national-library.net','national-library.net'] + + @classmethod + def getSiteExampleURLs(cls): + # ONLY the stories archived on or after June 17, 2006 and that are hosted on the website: + return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('h1') + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"authorresults.php\?author=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for p in soup.findAll('p'): + chapters = p.findAll('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId')+"&chapnum=\d+$")) + if len(chapters) > 0: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + break + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + self.story.setMetadata('status', 'Completed') + + # Rated: NC-17
etc + labels = soup.findAll('b') + for x in range(2,len(labels)): + value = labels[x].nextSibling + label = labels[x].string + + if 'Summary' in label: + self.setDescription(url,value) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rating' in label: + self.story.setMetadata('rating', stripHTML(value.nextSibling)) + + if 'Word Count' in label: + self.story.setMetadata('numWords', value.string) + + if 'Category' in label: + for cat in value.string.split(', '): + self.story.addToList('category',cat) + if 'Crossover Shows' in label: + for cat in value.string.split(', '): + if "No Show" not in cat: + self.story.addToList('category',cat) + + if 'Character' in label: + for char in value.string.split(', '): + self.story.addToList('characters',char) + + if 'Pairing' in label: + for char in value.string.split(', '): + self.story.addToList('ships',char) + + if 'Warnings' in label: + for warning in value.string.split(', '): + self.story.addToList('warnings',warning) + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Series' in label: + self.setSeries(stripHTML(value.nextSibling), value.nextSibling.nextSibling.string[2:]) + self.story.setMetadata('seriesUrl','http://'+self.host+'/'+value.nextSibling['href']) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + story=asoup.find('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId'))) + + a=story.findNext(text=re.compile('Genre')).parent.nextSibling.string.split(', ') + for genre in a: + self.story.setMetadata('genre', genre) + + a=story.findNext(text=re.compile('Archived')) + self.story.setMetadata('datePublished', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div') + + # bit messy since higly inconsistent + for p in soup.findAll('p', {'align' : 'center'}): + p.extract() + p = soup.findAll('p') + for x in range(0,3): + p[x].extract() + if "Chapters: " in stripHTML(p[3]): + p[3].extract() + for x in range(len(p)-2,len(p)-1): + p[x].extract() + + for p in soup.findAll('h1'): + p.extract() + for p in soup.findAll('h3'): + p.extract() + for p in soup.findAll('a'): + p.extract() + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_ncisficcom.py b/fanficfare/adapters/adapter_ncisficcom.py similarity index 97% rename from fff_internals/adapters/adapter_ncisficcom.py rename to fanficfare/adapters/adapter_ncisficcom.py index dbbf3fb..2b3a608 100644 --- a/fff_internals/adapters/adapter_ncisficcom.py +++ b/fanficfare/adapters/adapter_ncisficcom.py @@ -1,219 +1,219 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NCISFicComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NCISFicComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only storyid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?storyid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ncisf') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m-%d-%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - return 'ncisfic.com' - - @classmethod - def getAcceptDomains(cls): - return ['www.ncisfic.com','ncisfic.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('h1') - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"authorresults.php\?author=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for p in soup.findAll('p'): - chapters = p.findAll('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId')+"&chapnum=\d+$")) - if len(chapters) > 0: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - break - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - self.story.setMetadata('status', 'Completed') - - # Rated: NC-17
etc - labels = soup.findAll('b') - for x in range(2,len(labels)): - value = labels[x].nextSibling - label = labels[x].string - - if 'Summary' in label: - self.setDescription(url,value) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rating' in label: - self.story.setMetadata('rating', stripHTML(value.nextSibling)) - - if 'Word Count' in label: - self.story.setMetadata('numWords', value.string) - - if 'Category' in label: - for cat in value.string.split(', '): - self.story.addToList('category',cat) - if 'Crossover Shows' in label: - for cat in value.string.split(', '): - if "No Show" not in cat: - self.story.addToList('category',cat) - - if 'Character' in label: - for char in value.string.split(', '): - self.story.addToList('characters',char) - - if 'Pairing' in label: - for char in value.string.split(', '): - self.story.addToList('ships',char) - - if 'Warnings' in label: - for warning in value.string.split(', '): - self.story.addToList('warnings',warning) - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Series' in label: - if "No Series" not in value.nextSibling.string: - self.setSeries(stripHTML(value.nextSibling), value.nextSibling.nextSibling.string[2:]) - self.story.setMetadata('seriesUrl','http://'+self.host+'/'+value.nextSibling['href']) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - story=asoup.find('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId'))) - - a=story.findNext('font') - if 'Complete' in a.string: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - a=story.findNext(text=re.compile('Genre')).parent.nextSibling.string.split(', ') - for genre in a: - self.story.setMetadata('genre', genre) - - a=story.findNext(text=re.compile('Archived')) - self.story.setMetadata('datePublished', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div') - - # bit messy since higly inconsistent - for p in soup.findAll('p', {'align' : 'center'}): - p.extract() - p = soup.findAll('p') - for x in range(0,3): - p[x].extract() - if "Chapters: " in stripHTML(p[3]): - p[3].extract() - for x in range(len(p)-2,len(p)-1): - p[x].extract() - - for p in soup.findAll('h1'): - p.extract() - for p in soup.findAll('h3'): - p.extract() - for p in soup.findAll('a'): - p.extract() - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NCISFicComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NCISFicComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only storyid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?storyid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ncisf') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m-%d-%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + return 'ncisfic.com' + + @classmethod + def getAcceptDomains(cls): + return ['www.ncisfic.com','ncisfic.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?storyid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?storyid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('h1') + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"authorresults.php\?author=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for p in soup.findAll('p'): + chapters = p.findAll('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId')+"&chapnum=\d+$")) + if len(chapters) > 0: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + break + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + self.story.setMetadata('status', 'Completed') + + # Rated: NC-17
etc + labels = soup.findAll('b') + for x in range(2,len(labels)): + value = labels[x].nextSibling + label = labels[x].string + + if 'Summary' in label: + self.setDescription(url,value) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rating' in label: + self.story.setMetadata('rating', stripHTML(value.nextSibling)) + + if 'Word Count' in label: + self.story.setMetadata('numWords', value.string) + + if 'Category' in label: + for cat in value.string.split(', '): + self.story.addToList('category',cat) + if 'Crossover Shows' in label: + for cat in value.string.split(', '): + if "No Show" not in cat: + self.story.addToList('category',cat) + + if 'Character' in label: + for char in value.string.split(', '): + self.story.addToList('characters',char) + + if 'Pairing' in label: + for char in value.string.split(', '): + self.story.addToList('ships',char) + + if 'Warnings' in label: + for warning in value.string.split(', '): + self.story.addToList('warnings',warning) + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Series' in label: + if "No Series" not in value.nextSibling.string: + self.setSeries(stripHTML(value.nextSibling), value.nextSibling.nextSibling.string[2:]) + self.story.setMetadata('seriesUrl','http://'+self.host+'/'+value.nextSibling['href']) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + story=asoup.find('a', href=re.compile(r'viewstory.php\?storyid='+self.story.getMetadata('storyId'))) + + a=story.findNext('font') + if 'Complete' in a.string: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + a=story.findNext(text=re.compile('Genre')).parent.nextSibling.string.split(', ') + for genre in a: + self.story.setMetadata('genre', genre) + + a=story.findNext(text=re.compile('Archived')) + self.story.setMetadata('datePublished', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(stripHTML(a.parent.nextSibling), self.dateformat)) + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div') + + # bit messy since higly inconsistent + for p in soup.findAll('p', {'align' : 'center'}): + p.extract() + p = soup.findAll('p') + for x in range(0,3): + p[x].extract() + if "Chapters: " in stripHTML(p[3]): + p[3].extract() + for x in range(len(p)-2,len(p)-1): + p[x].extract() + + for p in soup.findAll('h1'): + p.extract() + for p in soup.findAll('h3'): + p.extract() + for p in soup.findAll('a'): + p.extract() + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_ncisfictionnet.py b/fanficfare/adapters/adapter_ncisfictionnet.py similarity index 97% rename from fff_internals/adapters/adapter_ncisfictionnet.py rename to fanficfare/adapters/adapter_ncisfictionnet.py index 58abedf..9b05633 100644 --- a/fff_internals/adapters/adapter_ncisfictionnet.py +++ b/fanficfare/adapters/adapter_ncisfictionnet.py @@ -1,210 +1,210 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NCISFictionNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NCISFictionNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["iso-8859-1", - "Windows-1252"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL("http://"+self.getSiteDomain()\ - +"/chapters.php?stid="+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','ncisfn') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.ncisfiction.net' - - ## Changed from www.ncisfiction.com to www.ncisfiction.net Oct - ## 2012 due to the ncisfiction.com domain expiring. Still accept - ## .com domains for existing updates, etc. - - @classmethod - def getAcceptDomains(cls): - return ['www.ncisfiction.net','www.ncisfiction.com'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/story.php?stid=01234 http://"+cls.getSiteDomain()+"/chapters.php?stid=1234" - - def getSiteURLPattern(self): - return r'http://www\.ncisfiction\.(net|com)/(chapters|story)?.php\?stid=\d+' - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulStoneSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title and author - a = soup.find('div', {'class' : 'main_title'}) - - aut = a.find('a') - self.story.setMetadata('authorId',aut['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+aut['href']) - self.story.setMetadata('author',aut.string) - - aut.extract() - self.story.setMetadata('title',stripHTML(a)[:len(stripHTML(a))-2]) - - # Find the chapters: - i=0 - chapters=soup.findAll('table', {'class' : 'story_table'}) - for chapter in chapters: - ch=chapter.find('a') - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(ch),'http://'+self.host+'/'+ch['href'])) - if i == 0: - self.story.setMetadata('datePublished', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat)) - if i == len(chapters)-1: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat)) - i=i+1 - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - info = soup.find('table', {'class' : 'story_info'}) - - # no convenient way to calculate word count as it is logged differently for stories with and without series - - labels = info.findAll('tr') - for tr in labels: - value = tr.find('td') - label = tr.find('th').string - - if 'Summary' in label: - self.setDescription(url,value) - - if 'Rating' in label: - self.story.setMetadata('rating', value.string) - - if 'Category' in label: - cats = value.findAll('a') - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = value.findAll('a') - for char in chars: - self.story.addToList('characters',char.string) - - if 'Pairing' in label: - ships = value.findAll('a') - for ship in ships: - self.story.addToList('ships',ship.string) - - if 'Genre' in label: - genres = value.findAll('a') - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = value.findAll('a') - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Status' in label: - if 'not completed' in value.text: - self.story.setMetadata('status', 'In-Progress') - else: - self.story.setMetadata('status', 'Completed') - - try: - # Find Series name from series URL. - a = soup.find('div',{'class' : 'sub_header'}) - series_name = a.find('a').string - i = a.text.split('#')[1] - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl','http://'+self.host+'/'+a.find('a')['href']) - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'class' : 'story_text'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NCISFictionNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NCISFictionNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["iso-8859-1", + "Windows-1252"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL("http://"+self.getSiteDomain()\ + +"/chapters.php?stid="+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','ncisfn') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.ncisfiction.net' + + ## Changed from www.ncisfiction.com to www.ncisfiction.net Oct + ## 2012 due to the ncisfiction.com domain expiring. Still accept + ## .com domains for existing updates, etc. + + @classmethod + def getAcceptDomains(cls): + return ['www.ncisfiction.net','www.ncisfiction.com'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/story.php?stid=01234 http://"+cls.getSiteDomain()+"/chapters.php?stid=1234" + + def getSiteURLPattern(self): + return r'http://www\.ncisfiction\.(net|com)/(chapters|story)?.php\?stid=\d+' + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulStoneSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title and author + a = soup.find('div', {'class' : 'main_title'}) + + aut = a.find('a') + self.story.setMetadata('authorId',aut['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+aut['href']) + self.story.setMetadata('author',aut.string) + + aut.extract() + self.story.setMetadata('title',stripHTML(a)[:len(stripHTML(a))-2]) + + # Find the chapters: + i=0 + chapters=soup.findAll('table', {'class' : 'story_table'}) + for chapter in chapters: + ch=chapter.find('a') + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(ch),'http://'+self.host+'/'+ch['href'])) + if i == 0: + self.story.setMetadata('datePublished', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat)) + if i == len(chapters)-1: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(chapter.find('td')).split('Added: ')[1], self.dateformat)) + i=i+1 + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + info = soup.find('table', {'class' : 'story_info'}) + + # no convenient way to calculate word count as it is logged differently for stories with and without series + + labels = info.findAll('tr') + for tr in labels: + value = tr.find('td') + label = tr.find('th').string + + if 'Summary' in label: + self.setDescription(url,value) + + if 'Rating' in label: + self.story.setMetadata('rating', value.string) + + if 'Category' in label: + cats = value.findAll('a') + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = value.findAll('a') + for char in chars: + self.story.addToList('characters',char.string) + + if 'Pairing' in label: + ships = value.findAll('a') + for ship in ships: + self.story.addToList('ships',ship.string) + + if 'Genre' in label: + genres = value.findAll('a') + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = value.findAll('a') + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Status' in label: + if 'not completed' in value.text: + self.story.setMetadata('status', 'In-Progress') + else: + self.story.setMetadata('status', 'Completed') + + try: + # Find Series name from series URL. + a = soup.find('div',{'class' : 'sub_header'}) + series_name = a.find('a').string + i = a.text.split('#')[1] + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl','http://'+self.host+'/'+a.find('a')['href']) + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'class' : 'story_text'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_netraptororg.py b/fanficfare/adapters/adapter_netraptororg.py similarity index 97% rename from fff_internals/adapters/adapter_netraptororg.py rename to fanficfare/adapters/adapter_netraptororg.py index 6d4effb..0fc8b97 100644 --- a/fff_internals/adapters/adapter_netraptororg.py +++ b/fanficfare/adapters/adapter_netraptororg.py @@ -1,213 +1,213 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NetRaptorOrgAdapter - -class NetRaptorOrgAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/fanfiction/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','netrap') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'netraptor.org' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fanfiction/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/fanfiction/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - url = self.url+'&index=1' - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - pagetitle = soup.find('div',{'id':'pagetitle'}) - a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/fanfiction/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfiction/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/fanfiction/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NetRaptorOrgAdapter + +class NetRaptorOrgAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/fanfiction/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','netrap') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'netraptor.org' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fanfiction/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/fanfiction/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + url = self.url+'&index=1' + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + pagetitle = soup.find('div',{'id':'pagetitle'}) + a = pagetitle.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = pagetitle.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/fanfiction/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfiction/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/fanfiction/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_nfacommunitycom.py b/fanficfare/adapters/adapter_nfacommunitycom.py similarity index 97% rename from fff_internals/adapters/adapter_nfacommunitycom.py rename to fanficfare/adapters/adapter_nfacommunitycom.py index 4d428e7..3e91906 100644 --- a/fff_internals/adapters/adapter_nfacommunitycom.py +++ b/fanficfare/adapters/adapter_nfacommunitycom.py @@ -1,290 +1,290 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return NfaCommunityComAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NfaCommunityComAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','nfa') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%Y" # XXX - - @classmethod - def getAcceptDomains(cls): - return ['www.nfacommunity.com','nfacommunity.com'] - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'nfacommunity.com' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - - # Furthermore, there's a couple sites now with more than - # one warning level for different ratings. And they're - # fussy about it. nfacommunity has two: 4 & 5. - # we'll try 5 first. - addurl = "&ageconsent=ok&warning=5" # XXX - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - - # Since the warning text can change by warning level, let's - # look for the warning pass url. nfacommunity uses - # &warning= -- actually, so do other sites. Must be an - # eFiction book. - - # viewstory.php?sid=1882&warning=4 - # viewstory.php?sid=1654&ageconsent=ok&warning=5 - #print data - #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - self.story.addToList('characters',char.string) - - ## Not all sites use Genre, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - warningstext = [warning.string for warning in warnings] - self.warning = ', '.join(warningstext) - for warning in warningstext: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return NfaCommunityComAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NfaCommunityComAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','nfa') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" # XXX + + @classmethod + def getAcceptDomains(cls): + return ['www.nfacommunity.com','nfacommunity.com'] + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'nfacommunity.com' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return "http://(www.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + + # Furthermore, there's a couple sites now with more than + # one warning level for different ratings. And they're + # fussy about it. nfacommunity has two: 4 & 5. + # we'll try 5 first. + addurl = "&ageconsent=ok&warning=5" # XXX + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + + # Since the warning text can change by warning level, let's + # look for the warning pass url. nfacommunity uses + # &warning= -- actually, so do other sites. Must be an + # eFiction book. + + # viewstory.php?sid=1882&warning=4 + # viewstory.php?sid=1654&ageconsent=ok&warning=5 + #print data + #m = re.search(r"'viewstory.php\?sid=1882(&warning=4)'",data) + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + self.story.addToList('characters',char.string) + + ## Not all sites use Genre, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + warningstext = [warning.string for warning in warnings] + self.warning = ', '.join(warningstext) + for warning in warningstext: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_nhamagicalworldsus.py b/fanficfare/adapters/adapter_nhamagicalworldsus.py similarity index 97% rename from fff_internals/adapters/adapter_nhamagicalworldsus.py rename to fanficfare/adapters/adapter_nhamagicalworldsus.py index bd541cc..0feee10 100644 --- a/fff_internals/adapters/adapter_nhamagicalworldsus.py +++ b/fanficfare/adapters/adapter_nhamagicalworldsus.py @@ -1,237 +1,237 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NHAMagicalWorldsUsAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NHAMagicalWorldsUsAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','nha') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = " %d/%m/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'nha.magical-worlds.us' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - try: - # in case link points somewhere other than the first chapter - a = soup.findAll('option')[1]['value'] - self.story.setMetadata('storyId',a.split('=',)[1]) - url = 'http://'+self.host+'/'+a - soup = bs.BeautifulSoup(self._fetchUrl(url)) - except: - pass - - for info in asoup.findAll('table', {'width' : '100%', 'bordercolor' : re.compile(r'#')}): - a = info.find('a') - if 'viewstory.php?sid='+self.story.getMetadata('storyId') == a['href'] or \ - ('viewstory.php?sid='+self.story.getMetadata('storyId')+'&') in a['href']: - self.story.setMetadata('title',stripHTML(a)) - break - - - # Find the chapters: - chapters=soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+'&chapter=\d+$')) - if len(chapters) == 0: - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d): - try: - return d.name - except: - return "" - - cats = info.findAll('a',href=re.compile('categories.php')) - for cat in cats: - self.story.addToList('category',cat.string) - - a = info.find('a', href=re.compile(r'viewuser.php')) - val = a.nextSibling - svalue = "" - while not defaultGetattr(val) == 'br': - val = val.nextSibling - val = val.nextSibling - while not defaultGetattr(val) == 'br': - svalue += unicode(val) - val = val.nextSibling - self.setDescription(url,svalue) - - #does not provide convenient way to get word count - labels = info.findAll('i') - for labelspan in labels: - value = labelspan.nextSibling - label = stripHTML(labelspan) - - if 'Rating' in label: - self.story.setMetadata('rating', value.split(' -')[0]) - - if 'Genres' in label: - genres = value.string.split(', ') - for genre in genres: - if 'None' not in genre: - self.story.addToList('genre',genre.split(' -')[0]) - - if 'Characters' in label: - chars = value.string.split(', ') - for char in chars: - if 'None' not in char: - self.story.addToList('characters',char.split(' -')[0]) - - if 'Warnings' in label: - warnings = value.string.split(', ') - for warning in warnings: - if 'None' not in warning: - self.story.addToList('warnings',warning.split(' -')[0]) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(value.split(' -')[0], self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(value.split(' -')[0], self.dateformat)) - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - - soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr','span','center')) # some chapters seem to be hanging up on those tags, so it is safer to close them - - story = soup.find('div', {"id" : "story"}) - - if None == story: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,story) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NHAMagicalWorldsUsAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NHAMagicalWorldsUsAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','nha') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = " %d/%m/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'nha.magical-worlds.us' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + try: + # in case link points somewhere other than the first chapter + a = soup.findAll('option')[1]['value'] + self.story.setMetadata('storyId',a.split('=',)[1]) + url = 'http://'+self.host+'/'+a + soup = bs.BeautifulSoup(self._fetchUrl(url)) + except: + pass + + for info in asoup.findAll('table', {'width' : '100%', 'bordercolor' : re.compile(r'#')}): + a = info.find('a') + if 'viewstory.php?sid='+self.story.getMetadata('storyId') == a['href'] or \ + ('viewstory.php?sid='+self.story.getMetadata('storyId')+'&') in a['href']: + self.story.setMetadata('title',stripHTML(a)) + break + + + # Find the chapters: + chapters=soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+'&chapter=\d+$')) + if len(chapters) == 0: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d): + try: + return d.name + except: + return "" + + cats = info.findAll('a',href=re.compile('categories.php')) + for cat in cats: + self.story.addToList('category',cat.string) + + a = info.find('a', href=re.compile(r'viewuser.php')) + val = a.nextSibling + svalue = "" + while not defaultGetattr(val) == 'br': + val = val.nextSibling + val = val.nextSibling + while not defaultGetattr(val) == 'br': + svalue += unicode(val) + val = val.nextSibling + self.setDescription(url,svalue) + + #does not provide convenient way to get word count + labels = info.findAll('i') + for labelspan in labels: + value = labelspan.nextSibling + label = stripHTML(labelspan) + + if 'Rating' in label: + self.story.setMetadata('rating', value.split(' -')[0]) + + if 'Genres' in label: + genres = value.string.split(', ') + for genre in genres: + if 'None' not in genre: + self.story.addToList('genre',genre.split(' -')[0]) + + if 'Characters' in label: + chars = value.string.split(', ') + for char in chars: + if 'None' not in char: + self.story.addToList('characters',char.split(' -')[0]) + + if 'Warnings' in label: + warnings = value.string.split(', ') + for warning in warnings: + if 'None' not in warning: + self.story.addToList('warnings',warning.split(' -')[0]) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(value.split(' -')[0], self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(value.split(' -')[0], self.dateformat)) + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + + soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr','span','center')) # some chapters seem to be hanging up on those tags, so it is safer to close them + + story = soup.find('div', {"id" : "story"}) + + if None == story: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,story) diff --git a/fff_internals/adapters/adapter_nickandgregnet.py b/fanficfare/adapters/adapter_nickandgregnet.py similarity index 97% rename from fff_internals/adapters/adapter_nickandgregnet.py rename to fanficfare/adapters/adapter_nickandgregnet.py index cd0fd48..5457463 100644 --- a/fff_internals/adapters/adapter_nickandgregnet.py +++ b/fanficfare/adapters/adapter_nickandgregnet.py @@ -1,177 +1,177 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return NickAndGregNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class NickAndGregNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. - self._setURL('http://' + self.getSiteDomain() + '/desert_archive/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','nag') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%Y/%m/%d" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.nickandgreg.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/desert_archive/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/desert_archive/viewstory.php?sid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&i=1' - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/desert_archive/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - chapters = soup.find('select') - for chapter in chapters.findAll('option'): - if chapter.text != 'Story Index' and chapter.text != 'Chapters': - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/desert_archive/'+chapter['value'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - for div in asoup.findAll('td', {'class' : 'tblborder6'}): - a = div.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - if a != None: - break - - self.setDescription(url,div.find('br').nextSibling) - - a=div.text.split('Rating:') - if len(a) == 2: self.story.setMetadata('rating', a[1].split(' -')[0]) - - a=div.text.split('Characters:') - if len(a) == 2: - for char in a[1].split(' -')[0].split(', '): - self.story.addToList('characters',char) - - a=div.text.split('Genres:') - if len(a) == 2: - for genre in a[1].split(' -')[0].split(', '): - self.story.addToList('genre',genre) - - a=div.text.split('Warnings:') - if len(a) == 2: - for warn in a[1].split(' -')[0].split(', '): - if 'none' not in warn: - self.story.addToList('warnings',warn) - - a=div.text.split('Completed:') - if len(a) ==2: - if 'Yes' in a[1]: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - a=div.text.split('Published:') - if len(a) == 2: self.story.setMetadata('datePublished', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat)) - - a=div.text.split('Updated:') - if len(a) == 2: self.story.setMetadata('dateUpdated', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat)) - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - # wrap a div around it. - divsoup = bs.BeautifulStoneSoup('
', - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - div = divsoup.find('div') - div.append(soup.find('table', {'class' : 'tblborder6'})) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return NickAndGregNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class NickAndGregNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + # XXX Most sites don't have the /fanfic part. Replace all to remove it usually. + self._setURL('http://' + self.getSiteDomain() + '/desert_archive/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','nag') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%Y/%m/%d" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.nickandgreg.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/desert_archive/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/desert_archive/viewstory.php?sid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&i=1' + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/desert_archive/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + chapters = soup.find('select') + for chapter in chapters.findAll('option'): + if chapter.text != 'Story Index' and chapter.text != 'Chapters': + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/desert_archive/'+chapter['value'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + for div in asoup.findAll('td', {'class' : 'tblborder6'}): + a = div.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + if a != None: + break + + self.setDescription(url,div.find('br').nextSibling) + + a=div.text.split('Rating:') + if len(a) == 2: self.story.setMetadata('rating', a[1].split(' -')[0]) + + a=div.text.split('Characters:') + if len(a) == 2: + for char in a[1].split(' -')[0].split(', '): + self.story.addToList('characters',char) + + a=div.text.split('Genres:') + if len(a) == 2: + for genre in a[1].split(' -')[0].split(', '): + self.story.addToList('genre',genre) + + a=div.text.split('Warnings:') + if len(a) == 2: + for warn in a[1].split(' -')[0].split(', '): + if 'none' not in warn: + self.story.addToList('warnings',warn) + + a=div.text.split('Completed:') + if len(a) ==2: + if 'Yes' in a[1]: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + a=div.text.split('Published:') + if len(a) == 2: self.story.setMetadata('datePublished', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat)) + + a=div.text.split('Updated:') + if len(a) == 2: self.story.setMetadata('dateUpdated', makeDate(stripHTML(a[1].split(' -')[0]), self.dateformat)) + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + # wrap a div around it. + divsoup = bs.BeautifulStoneSoup('
', + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + div = divsoup.find('div') + div.append(soup.find('table', {'class' : 'tblborder6'})) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_nocturnallightnet.py b/fanficfare/adapters/adapter_nocturnallightnet.py similarity index 97% rename from fff_internals/adapters/adapter_nocturnallightnet.py rename to fanficfare/adapters/adapter_nocturnallightnet.py index 214e38b..f3aab49 100644 --- a/fff_internals/adapters/adapter_nocturnallightnet.py +++ b/fanficfare/adapters/adapter_nocturnallightnet.py @@ -1,177 +1,177 @@ -import re -import urllib2 -import urlparse - -from .. import BeautifulSoup - -from base_adapter import BaseSiteAdapter, makeDate -from .. import exceptions - - -def getClass(): - return NocturnalLightNetAdapter - - -# yields Tag _and_ NavigableString siblings from the given tag. The -# BeautifulSoup findNextSiblings() method for some reasons only returns either -# NavigableStrings _or_ Tag objects, not both. -def _yield_next_siblings(tag): - sibling = tag.nextSibling - while sibling: - yield sibling - sibling = sibling.nextSibling - - -class NocturnalLightNetAdapter(BaseSiteAdapter): - SITE_ABBREVIATION = 'nln' - SITE_DOMAIN = 'nocturnal-light.net' - BASE_URL = 'http://' + SITE_DOMAIN + '/fanfiction/' - STORY_URL_TEMPLATE = BASE_URL + 'story/%s' - AUTHORS_URL_TEMPLATE = BASE_URL + 'authors/%s' - - DATETIME_FORMAT = '%m-%d-%y' - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - url_tokens = self.parsedUrl.path.split('/') - story_id = url_tokens[url_tokens.index('story') + 1] - - self.story.setMetadata('storyId', story_id) - self._setURL(self.STORY_URL_TEMPLATE % story_id) - self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) - - def _customized_fetch_url(self, url, exception=None, parameters=None): - if exception: - try: - data = self._fetchUrl(url, parameters) - except urllib2.HTTPError: - raise exception(self.url) - # Just let self._fetchUrl throw the exception, don't catch and - # customize it. - else: - data = self._fetchUrl(url, parameters) - - return BeautifulSoup.BeautifulSoup(data) - - @staticmethod - def getSiteDomain(): - return NocturnalLightNetAdapter.SITE_DOMAIN - - @classmethod - def getSiteExampleURLs(cls): - return cls.STORY_URL_TEMPLATE % 1234 - - def getSiteURLPattern(self): - return re.escape(self.STORY_URL_TEMPLATE[:-2]) + r'\d+.*$' - - def extractChapterUrlsAndMetadata(self): - soup = self._customized_fetch_url(self.url) - - # Since no 404 error code we have to raise the exception ourselves. - # A title that is just 'by' indicates that there is no author name - # and no story title available. - if soup.title.string.strip() == 'by': - raise exceptions.StoryDoesNotExist(self.url) - - # "storycontent" is found in a single-chapter story - author_anchor = soup.find('div', id=lambda id: id in ('main', 'storycontent')).h1.a - self.story.setMetadata('author', author_anchor.string) - - url_tokens = author_anchor['href'].split('/') - author_id = url_tokens[url_tokens.index('authors')+1] - self.story.setMetadata('authorId', author_id) - self.story.setMetadata('authorUrl', self.AUTHORS_URL_TEMPLATE % author_id) - - chapter_anchors = soup('a', href=lambda href: href and href.startswith('/fanfiction/story/')) - for chapter_anchor in chapter_anchors: - url = urlparse.urljoin(self.BASE_URL, chapter_anchor['href']) - self.chapterUrls.append((chapter_anchor.string, url)) - - author_url = urlparse.urljoin(self.BASE_URL, author_anchor['href']) - soup = self._customized_fetch_url(author_url) - story_id = self.story.getMetadata('storyId') - for listbox in soup('div', {'class': 'listbox'}): - url_tokens = listbox.a['href'].split('/') - # Found the div containing the story's metadata; break the loop and - # parse the element - if story_id == url_tokens[url_tokens.index('story')+1]: - break - else: - raise exceptions.FailedToDownload(self.url) - - title = listbox.a.string - self.story.setMetadata('title', title) - - # No chapter anchors found in the original story URL, so the story has - # only a single chapter. - if not chapter_anchors: - self.chapterUrls.append((title, self.url)) - - for b_tag in listbox('b'): - key = b_tag.string.strip(':') - try: - value = b_tag.nextSibling.string.replace('•', '').strip(': ') - # This can happen with some fancy markup in the summary. Just - # ignore this error and set value to None, the summary parsing - # takes care of this - except AttributeError: - value = None - - if key == 'Summary': - contents = [] - keep_summary_html = self.getConfig('keep_summary_html') - - for sibling in _yield_next_siblings(b_tag): - if isinstance(sibling, BeautifulSoup.Tag): - if sibling.name == 'b' and sibling.findPreviousSibling().name == 'br': - break - - if keep_summary_html: - contents.append(self.utf8FromSoup(author_url, sibling)) - else: - contents.append(''.join(sibling(text=True))) - else: - contents.append(sibling) - - # Pop last break line tag - contents.pop() - self.story.setMetadata('description', ''.join(contents)) - - elif key == 'Category': - for sibling in b_tag.findNextSiblings(['a', 'b']): - if sibling.name == 'b': - break - - self.story.addToList('category', sibling.string) - - elif key == 'Rating': - self.story.setMetadata('rating', value) - - elif key == 'Chapters': - self.story.setMetadata('numChapters', int(value)) - - # Also parse reviews number which lies right after the chapters - # section - reviews_anchor = b_tag.findNextSibling('a') - reviews = reviews_anchor.string.split(' ')[1].strip('()') - self.story.setMetadata('reviews', reviews) - - elif key == 'Completed': - self.story.setMetadata('status', 'Completed' if value == 'Yes' else 'In-Progress') - - elif key == 'Date Added': - self.story.setMetadata('datePublished', makeDate(value, self.DATETIME_FORMAT)) - - elif key == 'Last Updated': - self.story.setMetadata('dateUpdated', makeDate(value, self.DATETIME_FORMAT)) - - elif key == 'Read': - self.story.setMetadata('readings', value.split()[0]) - - if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')): - raise exceptions.AdultCheckRequired(self.url) - - def getChapterText(self, url): - soup = self._customized_fetch_url(url) - return self.utf8FromSoup(url, soup.find('div', id='storytext')) +import re +import urllib2 +import urlparse + +from .. import BeautifulSoup + +from base_adapter import BaseSiteAdapter, makeDate +from .. import exceptions + + +def getClass(): + return NocturnalLightNetAdapter + + +# yields Tag _and_ NavigableString siblings from the given tag. The +# BeautifulSoup findNextSiblings() method for some reasons only returns either +# NavigableStrings _or_ Tag objects, not both. +def _yield_next_siblings(tag): + sibling = tag.nextSibling + while sibling: + yield sibling + sibling = sibling.nextSibling + + +class NocturnalLightNetAdapter(BaseSiteAdapter): + SITE_ABBREVIATION = 'nln' + SITE_DOMAIN = 'nocturnal-light.net' + BASE_URL = 'http://' + SITE_DOMAIN + '/fanfiction/' + STORY_URL_TEMPLATE = BASE_URL + 'story/%s' + AUTHORS_URL_TEMPLATE = BASE_URL + 'authors/%s' + + DATETIME_FORMAT = '%m-%d-%y' + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + url_tokens = self.parsedUrl.path.split('/') + story_id = url_tokens[url_tokens.index('story') + 1] + + self.story.setMetadata('storyId', story_id) + self._setURL(self.STORY_URL_TEMPLATE % story_id) + self.story.setMetadata('siteabbrev', self.SITE_ABBREVIATION) + + def _customized_fetch_url(self, url, exception=None, parameters=None): + if exception: + try: + data = self._fetchUrl(url, parameters) + except urllib2.HTTPError: + raise exception(self.url) + # Just let self._fetchUrl throw the exception, don't catch and + # customize it. + else: + data = self._fetchUrl(url, parameters) + + return BeautifulSoup.BeautifulSoup(data) + + @staticmethod + def getSiteDomain(): + return NocturnalLightNetAdapter.SITE_DOMAIN + + @classmethod + def getSiteExampleURLs(cls): + return cls.STORY_URL_TEMPLATE % 1234 + + def getSiteURLPattern(self): + return re.escape(self.STORY_URL_TEMPLATE[:-2]) + r'\d+.*$' + + def extractChapterUrlsAndMetadata(self): + soup = self._customized_fetch_url(self.url) + + # Since no 404 error code we have to raise the exception ourselves. + # A title that is just 'by' indicates that there is no author name + # and no story title available. + if soup.title.string.strip() == 'by': + raise exceptions.StoryDoesNotExist(self.url) + + # "storycontent" is found in a single-chapter story + author_anchor = soup.find('div', id=lambda id: id in ('main', 'storycontent')).h1.a + self.story.setMetadata('author', author_anchor.string) + + url_tokens = author_anchor['href'].split('/') + author_id = url_tokens[url_tokens.index('authors')+1] + self.story.setMetadata('authorId', author_id) + self.story.setMetadata('authorUrl', self.AUTHORS_URL_TEMPLATE % author_id) + + chapter_anchors = soup('a', href=lambda href: href and href.startswith('/fanfiction/story/')) + for chapter_anchor in chapter_anchors: + url = urlparse.urljoin(self.BASE_URL, chapter_anchor['href']) + self.chapterUrls.append((chapter_anchor.string, url)) + + author_url = urlparse.urljoin(self.BASE_URL, author_anchor['href']) + soup = self._customized_fetch_url(author_url) + story_id = self.story.getMetadata('storyId') + for listbox in soup('div', {'class': 'listbox'}): + url_tokens = listbox.a['href'].split('/') + # Found the div containing the story's metadata; break the loop and + # parse the element + if story_id == url_tokens[url_tokens.index('story')+1]: + break + else: + raise exceptions.FailedToDownload(self.url) + + title = listbox.a.string + self.story.setMetadata('title', title) + + # No chapter anchors found in the original story URL, so the story has + # only a single chapter. + if not chapter_anchors: + self.chapterUrls.append((title, self.url)) + + for b_tag in listbox('b'): + key = b_tag.string.strip(':') + try: + value = b_tag.nextSibling.string.replace('•', '').strip(': ') + # This can happen with some fancy markup in the summary. Just + # ignore this error and set value to None, the summary parsing + # takes care of this + except AttributeError: + value = None + + if key == 'Summary': + contents = [] + keep_summary_html = self.getConfig('keep_summary_html') + + for sibling in _yield_next_siblings(b_tag): + if isinstance(sibling, BeautifulSoup.Tag): + if sibling.name == 'b' and sibling.findPreviousSibling().name == 'br': + break + + if keep_summary_html: + contents.append(self.utf8FromSoup(author_url, sibling)) + else: + contents.append(''.join(sibling(text=True))) + else: + contents.append(sibling) + + # Pop last break line tag + contents.pop() + self.story.setMetadata('description', ''.join(contents)) + + elif key == 'Category': + for sibling in b_tag.findNextSiblings(['a', 'b']): + if sibling.name == 'b': + break + + self.story.addToList('category', sibling.string) + + elif key == 'Rating': + self.story.setMetadata('rating', value) + + elif key == 'Chapters': + self.story.setMetadata('numChapters', int(value)) + + # Also parse reviews number which lies right after the chapters + # section + reviews_anchor = b_tag.findNextSibling('a') + reviews = reviews_anchor.string.split(' ')[1].strip('()') + self.story.setMetadata('reviews', reviews) + + elif key == 'Completed': + self.story.setMetadata('status', 'Completed' if value == 'Yes' else 'In-Progress') + + elif key == 'Date Added': + self.story.setMetadata('datePublished', makeDate(value, self.DATETIME_FORMAT)) + + elif key == 'Last Updated': + self.story.setMetadata('dateUpdated', makeDate(value, self.DATETIME_FORMAT)) + + elif key == 'Read': + self.story.setMetadata('readings', value.split()[0]) + + if self.story.getMetadata('rating') == 'NC-17' and not (self.is_adult or self.getConfig('is_adult')): + raise exceptions.AdultCheckRequired(self.url) + + def getChapterText(self, url): + soup = self._customized_fetch_url(url) + return self.utf8FromSoup(url, soup.find('div', id='storytext')) diff --git a/fff_internals/adapters/adapter_occlumencysycophanthexcom.py b/fanficfare/adapters/adapter_occlumencysycophanthexcom.py similarity index 97% rename from fff_internals/adapters/adapter_occlumencysycophanthexcom.py rename to fanficfare/adapters/adapter_occlumencysycophanthexcom.py index 87008a5..8099179 100644 --- a/fff_internals/adapters/adapter_occlumencysycophanthexcom.py +++ b/fanficfare/adapters/adapter_occlumencysycophanthexcom.py @@ -1,263 +1,263 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return OcclumencySycophantHexComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class OcclumencySycophantHexComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','osph') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'occlumency.sycophanthex.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'This story contains adult content and/or themes.' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['rememberme'] = '1' - params['sid'] = '' - params['intent'] = '' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Logout" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - try: - # in case link points somewhere other than the first chapter - a = soup.findAll('option')[1]['value'] - self.story.setMetadata('storyId',a.split('=',)[1]) - url = 'http://'+self.host+'/'+a - soup = bs.BeautifulSoup(self._fetchUrl(url)) - except: - pass - - for info in asoup.findAll('table', {'class' : 'border'}): - a = info.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - if a != None: - self.story.setMetadata('title',stripHTML(a)) - break - - - # Find the chapters: - chapters=soup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+&i=1$')) - if len(chapters) == 0: - self.chapterUrls.append((self.story.getMetadata('title'),url)) - else: - for chapter in chapters: - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d): - try: - return d.name - except: - return "" - - cats = info.findAll('a',href=re.compile('categories.php')) - for cat in cats: - self.story.addToList('category',cat.string) - - - a = info.find('a', href=re.compile(r'reviews.php\?sid='+self.story.getMetadata('storyId'))) - val = a.nextSibling - svalue = "" - while not defaultGetattr(val) == 'br': - val = val.nextSibling - val = val.nextSibling - while not defaultGetattr(val) == 'table': - svalue += str(val) - val = val.nextSibling - self.setDescription(url,svalue) - - # Rated: NC-17
etc - labels = info.findAll('b') - for labelspan in labels: - value = labelspan.nextSibling - label = stripHTML(labelspan) - - if 'Rating' in label: - self.story.setMetadata('rating', value) - - if 'Word Count' in label: - self.story.setMetadata('numWords', value) - - if 'Genres' in label: - genres = value.string.split(', ') - for genre in genres: - if genre != 'none': - self.story.addToList('genre',genre) - - if 'Characters' in label: - chars = value.string.split(', ') - for char in chars: - if char != 'none': - self.story.addToList('characters',char) - - if 'Warnings' in label: - warnings = value.string.split(', ') - for warning in warnings: - if warning != ' none': - self.story.addToList('warnings',warning) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - data = self._fetchUrl(url) - data = data.replace('
') - - soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr')) - - story = soup.find('div', {"align" : "left"}) - - if None == story: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,story) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return OcclumencySycophantHexComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class OcclumencySycophantHexComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','osph') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'occlumency.sycophanthex.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'This story contains adult content and/or themes.' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['rememberme'] = '1' + params['sid'] = '' + params['intent'] = '' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Logout" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + try: + # in case link points somewhere other than the first chapter + a = soup.findAll('option')[1]['value'] + self.story.setMetadata('storyId',a.split('=',)[1]) + url = 'http://'+self.host+'/'+a + soup = bs.BeautifulSoup(self._fetchUrl(url)) + except: + pass + + for info in asoup.findAll('table', {'class' : 'border'}): + a = info.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + if a != None: + self.story.setMetadata('title',stripHTML(a)) + break + + + # Find the chapters: + chapters=soup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+&i=1$')) + if len(chapters) == 0: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + else: + for chapter in chapters: + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d): + try: + return d.name + except: + return "" + + cats = info.findAll('a',href=re.compile('categories.php')) + for cat in cats: + self.story.addToList('category',cat.string) + + + a = info.find('a', href=re.compile(r'reviews.php\?sid='+self.story.getMetadata('storyId'))) + val = a.nextSibling + svalue = "" + while not defaultGetattr(val) == 'br': + val = val.nextSibling + val = val.nextSibling + while not defaultGetattr(val) == 'table': + svalue += str(val) + val = val.nextSibling + self.setDescription(url,svalue) + + # Rated: NC-17
etc + labels = info.findAll('b') + for labelspan in labels: + value = labelspan.nextSibling + label = stripHTML(labelspan) + + if 'Rating' in label: + self.story.setMetadata('rating', value) + + if 'Word Count' in label: + self.story.setMetadata('numWords', value) + + if 'Genres' in label: + genres = value.string.split(', ') + for genre in genres: + if genre != 'none': + self.story.addToList('genre',genre) + + if 'Characters' in label: + chars = value.string.split(', ') + for char in chars: + if char != 'none': + self.story.addToList('characters',char) + + if 'Warnings' in label: + warnings = value.string.split(', ') + for warning in warnings: + if warning != ' none': + self.story.addToList('warnings',warning) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + data = self._fetchUrl(url) + data = data.replace('
') + + soup = bs.BeautifulSoup(data, selfClosingTags=('br','hr')) + + story = soup.find('div', {"align" : "left"}) + + if None == story: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,story) diff --git a/fff_internals/adapters/adapter_onedirectionfanfictioncom.py b/fanficfare/adapters/adapter_onedirectionfanfictioncom.py similarity index 97% rename from fff_internals/adapters/adapter_onedirectionfanfictioncom.py rename to fanficfare/adapters/adapter_onedirectionfanfictioncom.py index 5c3efbc..95e074f 100644 --- a/fff_internals/adapters/adapter_onedirectionfanfictioncom.py +++ b/fanficfare/adapters/adapter_onedirectionfanfictioncom.py @@ -1,270 +1,270 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return OneDirectionFanfictionComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class OneDirectionFanfictionComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','odf') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%m/%d/%Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'onedirectionfanfiction.com' - - @classmethod - def getAcceptDomains(cls): - return ['www.onedirectionfanfiction.com','onedirectionfanfiction.com'] - - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=4" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # The actual text that is used to announce you need to be an - # adult varies from site to site. Again, print data before - # the title search to troubleshoot. - if "Age Consent Required" in data: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while value and not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=6')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - # there's a stray [ at the end. - #value = value[0:-1] - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return OneDirectionFanfictionComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class OneDirectionFanfictionComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','odf') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%m/%d/%Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'onedirectionfanfiction.com' + + @classmethod + def getAcceptDomains(cls): + return ['www.onedirectionfanfiction.com','onedirectionfanfiction.com'] + + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+"(www\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=4" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # The actual text that is used to announce you need to be an + # adult varies from site to site. Again, print data before + # the title search to troubleshoot. + if "Age Consent Required" in data: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while value and not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=6')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + # there's a stray [ at the end. + #value = value[0:-1] + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_phoenixsongnet.py b/fanficfare/adapters/adapter_phoenixsongnet.py similarity index 97% rename from fff_internals/adapters/adapter_phoenixsongnet.py rename to fanficfare/adapters/adapter_phoenixsongnet.py index d8918d0..dd4f1d6 100644 --- a/fff_internals/adapters/adapter_phoenixsongnet.py +++ b/fanficfare/adapters/adapter_phoenixsongnet.py @@ -1,240 +1,240 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2, urllib, cookielib - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return PhoenixSongNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class PhoenixSongNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[3]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/fanfiction/story/' +self.story.getMetadata('storyId')+'/') - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','phs') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%B %d %Y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.phoenixsong.net' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/fanfiction/story/")+r"\d+/?$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Please login to continue.' in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['txtusername'] = self.username - params['txtpassword'] = self.password - else: - params['txtusername'] = self.getConfig("username") - params['txtpassword'] = self.getConfig("password") - #params['remember'] = '1' - params['login'] = 'Login' - - loginUrl = 'http://' + self.getSiteDomain() + '/users/processlogin.php' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['txtusername'])) - d = self._fetchUrl(loginUrl, params) - - if 'Please login to continue.' in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['txtusername'])) - raise exceptions.FailedToLogin(url,params['txtusername']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url - logger.debug("URL: "+url) - - try: - if self.getConfig('force_login'): - self.performLogin(url) - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - b = soup.find('div', {'id' : 'nav25'}) - a = b.find('a', href=re.compile(r'fanfiction/story/'+self.story.getMetadata('storyId')+"/$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. /fanfiction/stories.php?psid=125 - a = b.find('a', href=re.compile(r"/fanfiction/stories.php\?psid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - chapters = soup.find('select') - if chapters == None: - self.chapterUrls.append((self.story.getMetadata('title'),url)) - for b in soup.findAll('b'): - if b.text == "Updated": - date = b.nextSibling.string.split(': ')[1].split(',') - self.story.setMetadata('datePublished', makeDate(date[0]+date[1], self.dateformat)) - self.story.setMetadata('dateUpdated', makeDate(date[0]+date[1], self.dateformat)) - else: - i = 0 - chapters = chapters.findAll('option') - for chapter in chapters: - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+chapter['value'])) - if i == 0: - self.story.setMetadata('storyId',chapter['value'].split('/')[3]) - head = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+chapter['value'])).findAll('b') - for b in head: - if b.text == "Updated": - date = b.nextSibling.string.split(': ')[1].split(',') - self.story.setMetadata('datePublished', makeDate(date[0]+date[1], self.dateformat)) - - if i == (len(chapters)-1): - head = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+chapter['value'])).findAll('b') - for b in head: - if b.text == "Updated": - date = b.nextSibling.string.split(': ')[1].split(',') - self.story.setMetadata('dateUpdated', makeDate(date[0]+date[1], self.dateformat)) - i = i+1 - - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - info = asoup.find('a', href=re.compile(r'fanfiction/story/'+self.story.getMetadata('storyId')+"/$")) - while info != None: - info = info.findNext('div') - b = info.find('b') - val = b.nextSibling - - if 'Rating' in b.string: - self.story.setMetadata('rating', val.string.split(': ')[1]) - - if 'Words' in b.string: - self.story.setMetadata('numWords', val.string.split(': ')[1]) - - if 'Setting' in b.string: - self.story.addToList('category', val.string.split(': ')[1]) - - if 'Status' in b.string: - if 'Completed' in val: - val = 'Completed' - else: - val = 'In-Progress' - self.story.setMetadata('status', val) - - if 'Summary' in b.string: - b.extract() - info.find('br').extract() - self.setDescription(url,info) - break - - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - chapter=bs.BeautifulSoup('
') - for p in soup.findAll('p'): - if "This is for problems with the formatting or the layout of the chapter." in stripHTML(p): - break - chapter.append(p) - - for a in chapter.findAll('div'): - a.extract() - for a in chapter.findAll('table'): - a.extract() - for a in chapter.findAll('script'): - a.extract() - for a in chapter.findAll('form'): - a.extract() - for a in chapter.findAll('textarea'): - a.extract() - - - if None == chapter: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,chapter) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2, urllib, cookielib + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return PhoenixSongNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class PhoenixSongNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[3]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/fanfiction/story/' +self.story.getMetadata('storyId')+'/') + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','phs') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%B %d %Y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'www.phoenixsong.net' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/fanfiction/story/1234/" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/fanfiction/story/")+r"\d+/?$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Please login to continue.' in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['txtusername'] = self.username + params['txtpassword'] = self.password + else: + params['txtusername'] = self.getConfig("username") + params['txtpassword'] = self.getConfig("password") + #params['remember'] = '1' + params['login'] = 'Login' + + loginUrl = 'http://' + self.getSiteDomain() + '/users/processlogin.php' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['txtusername'])) + d = self._fetchUrl(loginUrl, params) + + if 'Please login to continue.' in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['txtusername'])) + raise exceptions.FailedToLogin(url,params['txtusername']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url + logger.debug("URL: "+url) + + try: + if self.getConfig('force_login'): + self.performLogin(url) + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + b = soup.find('div', {'id' : 'nav25'}) + a = b.find('a', href=re.compile(r'fanfiction/story/'+self.story.getMetadata('storyId')+"/$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. /fanfiction/stories.php?psid=125 + a = b.find('a', href=re.compile(r"/fanfiction/stories.php\?psid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + chapters = soup.find('select') + if chapters == None: + self.chapterUrls.append((self.story.getMetadata('title'),url)) + for b in soup.findAll('b'): + if b.text == "Updated": + date = b.nextSibling.string.split(': ')[1].split(',') + self.story.setMetadata('datePublished', makeDate(date[0]+date[1], self.dateformat)) + self.story.setMetadata('dateUpdated', makeDate(date[0]+date[1], self.dateformat)) + else: + i = 0 + chapters = chapters.findAll('option') + for chapter in chapters: + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+chapter['value'])) + if i == 0: + self.story.setMetadata('storyId',chapter['value'].split('/')[3]) + head = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+chapter['value'])).findAll('b') + for b in head: + if b.text == "Updated": + date = b.nextSibling.string.split(': ')[1].split(',') + self.story.setMetadata('datePublished', makeDate(date[0]+date[1], self.dateformat)) + + if i == (len(chapters)-1): + head = bs.BeautifulSoup(self._fetchUrl('http://'+self.host+chapter['value'])).findAll('b') + for b in head: + if b.text == "Updated": + date = b.nextSibling.string.split(': ')[1].split(',') + self.story.setMetadata('dateUpdated', makeDate(date[0]+date[1], self.dateformat)) + i = i+1 + + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + asoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + info = asoup.find('a', href=re.compile(r'fanfiction/story/'+self.story.getMetadata('storyId')+"/$")) + while info != None: + info = info.findNext('div') + b = info.find('b') + val = b.nextSibling + + if 'Rating' in b.string: + self.story.setMetadata('rating', val.string.split(': ')[1]) + + if 'Words' in b.string: + self.story.setMetadata('numWords', val.string.split(': ')[1]) + + if 'Setting' in b.string: + self.story.addToList('category', val.string.split(': ')[1]) + + if 'Status' in b.string: + if 'Completed' in val: + val = 'Completed' + else: + val = 'In-Progress' + self.story.setMetadata('status', val) + + if 'Summary' in b.string: + b.extract() + info.find('br').extract() + self.setDescription(url,info) + break + + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + chapter=bs.BeautifulSoup('
') + for p in soup.findAll('p'): + if "This is for problems with the formatting or the layout of the chapter." in stripHTML(p): + break + chapter.append(p) + + for a in chapter.findAll('div'): + a.extract() + for a in chapter.findAll('table'): + a.extract() + for a in chapter.findAll('script'): + a.extract() + for a in chapter.findAll('form'): + a.extract() + for a in chapter.findAll('textarea'): + a.extract() + + + if None == chapter: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,chapter) diff --git a/fff_internals/adapters/adapter_pommedesangcom.py b/fanficfare/adapters/adapter_pommedesangcom.py similarity index 97% rename from fff_internals/adapters/adapter_pommedesangcom.py rename to fanficfare/adapters/adapter_pommedesangcom.py index 4d553a0..8e29666 100644 --- a/fff_internals/adapters/adapter_pommedesangcom.py +++ b/fanficfare/adapters/adapter_pommedesangcom.py @@ -1,301 +1,301 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return PommeDeSangComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class PommeDeSangComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # pommedesang.com has two 'sections', shown in URL as - # 'efiction' and 'sds' that change how things should be - # handled. - # http://pommedesang.com/efiction/viewstory.php?sid=1234 - # http://pommedesang.com/sds/viewstory.php?sid=1234 - self.section=self.parsedUrl.path.split('/',)[1] - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/'+self.section+'/viewstory.php?sid='+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','pmds') - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - if 'efiction' in self.section: - self.dateformat = "%b %d, %Y" - else: - self.dateformat = "%m/%d/%y" - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'pommedesang.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/efiction/viewstory.php?sid=1234 http://"+cls.getSiteDomain()+"/sds/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return r"http://"+self.getSiteDomain()+"/(efiction|sds)?/viewstory.php\?sid=\d+$" - - ## Login seems to be reasonably standard across eFiction sites. - def needToLoginCheck(self, data): - if 'Registered Users Only' in data \ - or 'There is no such account on our website' in data \ - or "That password doesn't match the one in our database" in data: - return True - else: - return False - - def performLogin(self, url): - params = {} - - if self.password: - params['penname'] = self.username - params['password'] = self.password - else: - params['penname'] = self.getConfig("username") - params['password'] = self.getConfig("password") - params['cookiecheck'] = '1' - params['submit'] = 'Submit' - - loginUrl = 'http://' + self.getSiteDomain() + '/'+self.section+'/user.php?action=login' - logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, - params['penname'])) - - d = self._fetchUrl(loginUrl, params) - - if "Member Account" not in d : #Member Account - logger.info("Failed to login to URL %s as %s" % (loginUrl, - params['penname'])) - raise exceptions.FailedToLogin(url,params['penname']) - return False - else: - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&ageconsent=ok&warning=5" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if self.needToLoginCheck(data): - # need to log in for this one. - self.performLogin(url) - data = self._fetchUrl(url) - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile('viewstory.php\?sid=\d+')) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - -# summary, rated, word count, categories, characters, genre, warnings, completed, published, updated, seires - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next span class='label' - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - for cat in cats: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX - for genre in genres: - self.story.addToList('genre',genre.string) - - if 'Warnings' in label: - warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - for warning in warnings: - self.story.addToList('warnings',warning.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+self.section+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile('viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2013 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return PommeDeSangComAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class PommeDeSangComAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # pommedesang.com has two 'sections', shown in URL as + # 'efiction' and 'sds' that change how things should be + # handled. + # http://pommedesang.com/efiction/viewstory.php?sid=1234 + # http://pommedesang.com/sds/viewstory.php?sid=1234 + self.section=self.parsedUrl.path.split('/',)[1] + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/'+self.section+'/viewstory.php?sid='+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','pmds') + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + if 'efiction' in self.section: + self.dateformat = "%b %d, %Y" + else: + self.dateformat = "%m/%d/%y" + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'pommedesang.com' + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/efiction/viewstory.php?sid=1234 http://"+cls.getSiteDomain()+"/sds/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return r"http://"+self.getSiteDomain()+"/(efiction|sds)?/viewstory.php\?sid=\d+$" + + ## Login seems to be reasonably standard across eFiction sites. + def needToLoginCheck(self, data): + if 'Registered Users Only' in data \ + or 'There is no such account on our website' in data \ + or "That password doesn't match the one in our database" in data: + return True + else: + return False + + def performLogin(self, url): + params = {} + + if self.password: + params['penname'] = self.username + params['password'] = self.password + else: + params['penname'] = self.getConfig("username") + params['password'] = self.getConfig("password") + params['cookiecheck'] = '1' + params['submit'] = 'Submit' + + loginUrl = 'http://' + self.getSiteDomain() + '/'+self.section+'/user.php?action=login' + logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl, + params['penname'])) + + d = self._fetchUrl(loginUrl, params) + + if "Member Account" not in d : #Member Account + logger.info("Failed to login to URL %s as %s" % (loginUrl, + params['penname'])) + raise exceptions.FailedToLogin(url,params['penname']) + return False + else: + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&ageconsent=ok&warning=5" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if self.needToLoginCheck(data): + # need to log in for this one. + self.performLogin(url) + data = self._fetchUrl(url) + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile('viewstory.php\?sid=\d+')) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+self.section+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + +# summary, rated, word count, categories, characters, genre, warnings, completed, published, updated, seires + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next span class='label' + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + for cat in cats: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) # XXX + for genre in genres: + self.story.addToList('genre',genre.string) + + if 'Warnings' in label: + warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + for warning in warnings: + self.story.addToList('warnings',warning.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+self.section+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile('viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if ('viewstory.php?sid='+self.story.getMetadata('storyId')) in a['href']: + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_ponyfictionarchivenet.py b/fanficfare/adapters/adapter_ponyfictionarchivenet.py similarity index 97% rename from fff_internals/adapters/adapter_ponyfictionarchivenet.py rename to fanficfare/adapters/adapter_ponyfictionarchivenet.py index eceb332..0c8f7c7 100644 --- a/fff_internals/adapters/adapter_ponyfictionarchivenet.py +++ b/fanficfare/adapters/adapter_ponyfictionarchivenet.py @@ -1,249 +1,249 @@ -# -*- coding: utf-8 -*- - -# Copyright 2012 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -def getClass(): - return PonyFictionArchiveNetAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class PonyFictionArchiveNetAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - # normalized story URL. - if "explicit" in self.parsedUrl.netloc: - self._setURL('http://explicit.' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - self.dateformat = "%d/%b/%y" - else: - self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) - self.dateformat = "%d %b %Y" - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','pffa') - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'ponyfictionarchive.net' - - @classmethod - def getAcceptDomains(cls): - return ['www.ponyfictionarchive.net','ponyfictionarchive.net','explicit.ponyfictionarchive.net'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234 http://explicit."+cls.getSiteDomain()+"/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+"(www\.|explicit\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" - - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - if self.is_adult or self.getConfig("is_adult"): - # Weirdly, different sites use different warning numbers. - # If the title search below fails, there's a good chance - # you need a different number. print data at that point - # and see what the 'click here to continue' url says. - addurl = "&warning=9" - else: - addurl="" - - # index=1 makes sure we see the story chapter index. Some - # sites skip that for one-chapter stories. - url = self.url+'&index=1'+addurl - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - - m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) - if m != None: - if self.is_adult or self.getConfig("is_adult"): - # We tried the default and still got a warning, so - # let's pull the warning number from the 'continue' - # link and reload data. - addurl = m.group(1) - # correct stupid & error in url. - addurl = addurl.replace("&","&") - url = self.url+'&index=1'+addurl - logger.debug("URL 2nd try: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - else: - raise exceptions.AdultCheckRequired(self.url) - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - # print data - - # Now go hunting for all the meta data and the chapter list. - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - genres = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) - for genre in genres: - self.story.addToList('genre',genre.string) - - warnings = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) - for warning in warnings: - self.story.addToList('warnings',warning.string) - - status = soup.find('a',href=re.compile(r'browse.php\?type=class&type_id=2')) - self.story.setMetadata('status',status.string) - - section = soup.findAll('span', {'class' : 'General'})[1] - - self.story.setMetadata('rating', section.previousSibling.previousSibling.string) - - value = section.nextSibling - svalue = "" - while not defaultGetattr(value,'class') == 'label': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - for char in chars: - self.story.addToList('characters',char.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # can't use ^viewstory...$ in case of higher rated stories with javascript href. - storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) - i=1 - for a in storyas: - # skip 'report this' and 'TOC' links - if 'contact.php' not in a['href'] and 'index' not in a['href']: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulSoup(self._fetchUrl(url)) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) +# -*- coding: utf-8 -*- + +# Copyright 2012 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +def getClass(): + return PonyFictionArchiveNetAdapter + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class PonyFictionArchiveNetAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + # normalized story URL. + if "explicit" in self.parsedUrl.netloc: + self._setURL('http://explicit.' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + self.dateformat = "%d/%b/%y" + else: + self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId')) + self.dateformat = "%d %b %Y" + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','pffa') + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'ponyfictionarchive.net' + + @classmethod + def getAcceptDomains(cls): + return ['www.ponyfictionarchive.net','ponyfictionarchive.net','explicit.ponyfictionarchive.net'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/viewstory.php?sid=1234 http://explicit."+cls.getSiteDomain()+"/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+"(www\.|explicit\.)?"+re.escape(self.getSiteDomain()+"/viewstory.php?sid=")+r"\d+$" + + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + if self.is_adult or self.getConfig("is_adult"): + # Weirdly, different sites use different warning numbers. + # If the title search below fails, there's a good chance + # you need a different number. print data at that point + # and see what the 'click here to continue' url says. + addurl = "&warning=9" + else: + addurl="" + + # index=1 makes sure we see the story chapter index. Some + # sites skip that for one-chapter stories. + url = self.url+'&index=1'+addurl + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + + m = re.search(r"'viewstory.php\?sid=\d+((?:&ageconsent=ok)?&warning=\d+)'",data) + if m != None: + if self.is_adult or self.getConfig("is_adult"): + # We tried the default and still got a warning, so + # let's pull the warning number from the 'continue' + # link and reload data. + addurl = m.group(1) + # correct stupid & error in url. + addurl = addurl.replace("&","&") + url = self.url+'&index=1'+addurl + logger.debug("URL 2nd try: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + else: + raise exceptions.AdultCheckRequired(self.url) + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + # print data + + # Now go hunting for all the meta data and the chapter list. + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']+addurl)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + genres = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=1')) + for genre in genres: + self.story.addToList('genre',genre.string) + + warnings = soup.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=3')) + for warning in warnings: + self.story.addToList('warnings',warning.string) + + status = soup.find('a',href=re.compile(r'browse.php\?type=class&type_id=2')) + self.story.setMetadata('status',status.string) + + section = soup.findAll('span', {'class' : 'General'})[1] + + self.story.setMetadata('rating', section.previousSibling.previousSibling.string) + + value = section.nextSibling + svalue = "" + while not defaultGetattr(value,'class') == 'label': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + for char in chars: + self.story.addToList('characters',char.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # can't use ^viewstory...$ in case of higher rated stories with javascript href. + storyas = seriessoup.findAll('a', href=re.compile(r'viewstory.php\?sid=\d+')) + i=1 + for a in storyas: + # skip 'report this' and 'TOC' links + if 'contact.php' not in a['href'] and 'index' not in a['href']: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulSoup(self._fetchUrl(url)) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) diff --git a/fff_internals/adapters/adapter_portkeyorg.py b/fanficfare/adapters/adapter_portkeyorg.py similarity index 97% rename from fff_internals/adapters/adapter_portkeyorg.py rename to fanficfare/adapters/adapter_portkeyorg.py index 92bf17c..419ef39 100644 --- a/fff_internals/adapters/adapter_portkeyorg.py +++ b/fanficfare/adapters/adapter_portkeyorg.py @@ -1,282 +1,282 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 -import cookielib as cl - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -# Search for XXX comments--that's where things are most likely to need changing. - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return PortkeyOrgAdapter # XXX - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class PortkeyOrgAdapter(BaseSiteAdapter): # XXX - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/story/'+self.story.getMetadata('storyId')) - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','prtky') # XXX - - # The date format will vary from site to site. - # http://docs.python.org/library/datetime.html#strftime-strptime-behavior - self.dateformat = "%d/%m/%y" # XXX - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'fanfiction.portkey.org' # XXX - - @classmethod - def getSiteExampleURLs(cls): - return "http://"+cls.getSiteDomain()+"/story/1234" - - def getSiteURLPattern(self): - return re.escape("http://"+self.getSiteDomain()+"/story/")+r"\d+(/\d+)?$" - - def use_pagecache(self): - ''' - adapters that will work with the page cache need to implement - this and change it to True. - ''' - return True - - ## Getting the chapter list and the meta data, plus 'is adult' checking. - def extractChapterUrlsAndMetadata(self): - - url = self.url - logger.debug("URL: "+url) - - # portkey screws around with using a different URL to set the - # cookie and it's a pain. So... cheat! - if self.is_adult or self.getConfig("is_adult"): - cookie = cl.Cookie(version=0, name='verify17', value='1', - port=None, port_specified=False, - domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, - path='/', path_specified=True, - secure=False, - expires=time.time()+10000, - discard=False, - comment=None, - comment_url=None, - rest={'HttpOnly': None}, - rfc2109=False) - self.cookiejar.set_cookie(cookie) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "You must be over 18 years of age to view it" in data: # XXX - raise exceptions.AdultCheckRequired(self.url) - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - #print data - - # Now go hunting for all the meta data and the chapter list. - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"/profile/\d+")) - #print("======a:%s"%a) - self.story.setMetadata('authorId',a['href'].split('/')[-1]) - self.story.setMetadata('authorUrl','http://'+self.host+a['href']) - self.story.setMetadata('author',a.string) - - ## Going to get the rest from the author page. - authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) - - ## Title - titlea = authsoup.find('a', href=re.compile(r'/story/'+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(titlea)) - metablock = titlea.parent - - # Find the chapters: - for chapter in soup.find('select',{'name':'select5'}).findAll('option', {'value':re.compile(r'/story/'+self.story.getMetadata('storyId')+"/\d+$")}): - # just in case there's tags, like in chapter titles. - chtitle = stripHTML(chapter) - if not chtitle: - chtitle = "(Untitled Chapter)" - self.chapterUrls.append((chtitle,'http://'+self.host+chapter['value'])) - - if len(self.chapterUrls) == 0: - self.chapterUrls.append((stripHTML(self.story.getMetadata('title')),url)) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - # eFiction sites don't help us out a lot with their meta data - # formating, so it's a little ugly. - - # utility method - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - # Contents: NC17 - # Published: 12/11/07 - #
- # Description:
A special book helps Harry tap into the power the Dark Lord knows not. Of course it’s a book on sex magic and rituals… but Harry’s not complaining. Spurned on by the ghost of a pervert founder, Harry leads his friends in the hunt for Voldemort’s Horcruxes. - # EROTIC COMEDY! Loads of crude humor and sexual situations! - # - labels = metablock.findAll('span',{'class':'dark-small-bold'}) - for labelspan in labels: - value = labelspan.findNext('span').string - label = stripHTML(labelspan) -# print("\nlabel:%s\nlabel:%s\nvalue:%s\n"%(labelspan,label,value)) - - if 'Description' in label: - self.setDescription(url,value) - - if 'Contents' in label: - self.story.setMetadata('rating', value) - - if 'Words' in label: - self.story.setMetadata('numWords', value) - - # if 'Categories' in label: - # cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - # catstext = [cat.string for cat in cats] - # for cat in catstext: - # self.story.addToList('category',cat.string) - - # if 'Characters' in label: - # chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - # charstext = [char.string for char in chars] - # for char in charstext: - # self.story.addToList('characters',char.string) - - if 'Genre' in label: - # genre is typo'ed on the site--it falls between the - # dark-small-bold label and dark-small-bold content - # spans. - svalue = "" - value = labelspan.nextSibling - while not defaultGetattr(value,'class') == 'dark-small-bold': - svalue += str(value) - value = value.nextSibling - - for genre in svalue.split("/"): - genre = genre.strip() - if genre != 'None': - self.story.addToList('genre',genre) - - ## Not all sites use Warnings, but there's no harm to - ## leaving it in. Check to make sure the type_id number - ## is correct, though--it's site specific. - # if 'Warnings' in label: - # warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX - # warningstext = [warning.string for warning in warnings] - # self.warning = ', '.join(warningstext) - # for warning in warningstext: - # self.story.addToList('warnings',warning.string) - - if 'Status' in label: - if 'Completed' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) - - # try: - # # Find Series name from series URL. - # a = metablock.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - # series_name = a.string - # series_url = 'http://'+self.host+'/'+a['href'] - - # # use BeautifulSoup HTML parser to make everything easier to find. - # seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - # storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - # i=1 - # for a in storyas: - # if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - # self.setSeries(series_name, i) - # break - # i+=1 - # except: - # # I find it hard to care if the series parsing fails - # pass - - # grab the text for an individual chapter. - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - data = self._fetchUrl(url) - - data = data.replace("HTML>","div>") - - soup = bs.BeautifulSoup(data) - - #print("soup:%s"%soup) - tag = soup.find('td', {'class' : 'story'}) - if tag == None and "
Chapter does not exist!
" in data: - logger.error("Chapter is missing at: %s"%url) - return self.utf8FromSoup(url,bs.BeautifulStoneSoup("
"%(url,url))) - tag.name='div' # force to be a div to avoid problems with nook. - - centers = tag.findAll('center') - # first two and last two center tags are some script, 'report - # story', 'report story' and an ad. - centers[0].extract() - centers[1].extract() - centers[-1].extract() - centers[-2].extract() - - if None == tag: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,tag) +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib2 +import cookielib as cl + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +# Search for XXX comments--that's where things are most likely to need changing. + +# This function is called by the downloader in all adapter_*.py files +# in this dir to register the adapter class. So it needs to be +# updated to reflect the class below it. That, plus getSiteDomain() +# take care of 'Registering'. +def getClass(): + return PortkeyOrgAdapter # XXX + +# Class name has to be unique. Our convention is camel case the +# sitename with Adapter at the end. www is skipped. +class PortkeyOrgAdapter(BaseSiteAdapter): # XXX + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + self.username = "NoneGiven" # if left empty, site doesn't return any message at all. + self.password = "" + self.is_adult=False + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.path.split('/',)[2]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/story/'+self.story.getMetadata('storyId')) + + # Each adapter needs to have a unique site abbreviation. + self.story.setMetadata('siteabbrev','prtky') # XXX + + # The date format will vary from site to site. + # http://docs.python.org/library/datetime.html#strftime-strptime-behavior + self.dateformat = "%d/%m/%y" # XXX + + @staticmethod # must be @staticmethod, don't remove it. + def getSiteDomain(): + # The site domain. Does have www here, if it uses it. + return 'fanfiction.portkey.org' # XXX + + @classmethod + def getSiteExampleURLs(cls): + return "http://"+cls.getSiteDomain()+"/story/1234" + + def getSiteURLPattern(self): + return re.escape("http://"+self.getSiteDomain()+"/story/")+r"\d+(/\d+)?$" + + def use_pagecache(self): + ''' + adapters that will work with the page cache need to implement + this and change it to True. + ''' + return True + + ## Getting the chapter list and the meta data, plus 'is adult' checking. + def extractChapterUrlsAndMetadata(self): + + url = self.url + logger.debug("URL: "+url) + + # portkey screws around with using a different URL to set the + # cookie and it's a pain. So... cheat! + if self.is_adult or self.getConfig("is_adult"): + cookie = cl.Cookie(version=0, name='verify17', value='1', + port=None, port_specified=False, + domain=self.getSiteDomain(), domain_specified=False, domain_initial_dot=False, + path='/', path_specified=True, + secure=False, + expires=time.time()+10000, + discard=False, + comment=None, + comment_url=None, + rest={'HttpOnly': None}, + rfc2109=False) + self.cookiejar.set_cookie(cookie) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "You must be over 18 years of age to view it" in data: # XXX + raise exceptions.AdultCheckRequired(self.url) + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + #print data + + # Now go hunting for all the meta data and the chapter list. + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"/profile/\d+")) + #print("======a:%s"%a) + self.story.setMetadata('authorId',a['href'].split('/')[-1]) + self.story.setMetadata('authorUrl','http://'+self.host+a['href']) + self.story.setMetadata('author',a.string) + + ## Going to get the rest from the author page. + authsoup = bs.BeautifulSoup(self._fetchUrl(self.story.getMetadata('authorUrl'))) + + ## Title + titlea = authsoup.find('a', href=re.compile(r'/story/'+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(titlea)) + metablock = titlea.parent + + # Find the chapters: + for chapter in soup.find('select',{'name':'select5'}).findAll('option', {'value':re.compile(r'/story/'+self.story.getMetadata('storyId')+"/\d+$")}): + # just in case there's tags, like in chapter titles. + chtitle = stripHTML(chapter) + if not chtitle: + chtitle = "(Untitled Chapter)" + self.chapterUrls.append((chtitle,'http://'+self.host+chapter['value'])) + + if len(self.chapterUrls) == 0: + self.chapterUrls.append((stripHTML(self.story.getMetadata('title')),url)) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + # eFiction sites don't help us out a lot with their meta data + # formating, so it's a little ugly. + + # utility method + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + # Contents: NC17 + # Published: 12/11/07 + #
+ # Description:
A special book helps Harry tap into the power the Dark Lord knows not. Of course it’s a book on sex magic and rituals… but Harry’s not complaining. Spurned on by the ghost of a pervert founder, Harry leads his friends in the hunt for Voldemort’s Horcruxes. + # EROTIC COMEDY! Loads of crude humor and sexual situations! + # + labels = metablock.findAll('span',{'class':'dark-small-bold'}) + for labelspan in labels: + value = labelspan.findNext('span').string + label = stripHTML(labelspan) +# print("\nlabel:%s\nlabel:%s\nvalue:%s\n"%(labelspan,label,value)) + + if 'Description' in label: + self.setDescription(url,value) + + if 'Contents' in label: + self.story.setMetadata('rating', value) + + if 'Words' in label: + self.story.setMetadata('numWords', value) + + # if 'Categories' in label: + # cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + # catstext = [cat.string for cat in cats] + # for cat in catstext: + # self.story.addToList('category',cat.string) + + # if 'Characters' in label: + # chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + # charstext = [char.string for char in chars] + # for char in charstext: + # self.story.addToList('characters',char.string) + + if 'Genre' in label: + # genre is typo'ed on the site--it falls between the + # dark-small-bold label and dark-small-bold content + # spans. + svalue = "" + value = labelspan.nextSibling + while not defaultGetattr(value,'class') == 'dark-small-bold': + svalue += str(value) + value = value.nextSibling + + for genre in svalue.split("/"): + genre = genre.strip() + if genre != 'None': + self.story.addToList('genre',genre) + + ## Not all sites use Warnings, but there's no harm to + ## leaving it in. Check to make sure the type_id number + ## is correct, though--it's site specific. + # if 'Warnings' in label: + # warnings = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class&type_id=2')) # XXX + # warningstext = [warning.string for warning in warnings] + # self.warning = ', '.join(warningstext) + # for warning in warningstext: + # self.story.addToList('warnings',warning.string) + + if 'Status' in label: + if 'Completed' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + self.story.setMetadata('datePublished', makeDate(stripHTML(value), self.dateformat)) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value), self.dateformat)) + + # try: + # # Find Series name from series URL. + # a = metablock.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + # series_name = a.string + # series_url = 'http://'+self.host+'/'+a['href'] + + # # use BeautifulSoup HTML parser to make everything easier to find. + # seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + # storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + # i=1 + # for a in storyas: + # if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + # self.setSeries(series_name, i) + # break + # i+=1 + # except: + # # I find it hard to care if the series parsing fails + # pass + + # grab the text for an individual chapter. + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + data = self._fetchUrl(url) + + data = data.replace("HTML>","div>") + + soup = bs.BeautifulSoup(data) + + #print("soup:%s"%soup) + tag = soup.find('td', {'class' : 'story'}) + if tag == None and "
Chapter does not exist!
" in data: + logger.error("Chapter is missing at: %s"%url) + return self.utf8FromSoup(url,bs.BeautifulStoneSoup("

Chapter does not exist!

Chapter is missing at: %s

"%(url,url))) + tag.name='div' # force to be a div to avoid problems with nook. + + centers = tag.findAll('center') + # first two and last two center tags are some script, 'report + # story', 'report story' and an ad. + centers[0].extract() + centers[1].extract() + centers[-1].extract() + centers[-2].extract() + + if None == tag: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,tag) diff --git a/fff_internals/adapters/adapter_potionsandsnitches.py b/fanficfare/adapters/adapter_potionsandsnitches.py similarity index 97% rename from fff_internals/adapters/adapter_potionsandsnitches.py rename to fanficfare/adapters/adapter_potionsandsnitches.py index 5673e69..59e5014 100644 --- a/fff_internals/adapters/adapter_potionsandsnitches.py +++ b/fanficfare/adapters/adapter_potionsandsnitches.py @@ -1,211 +1,211 @@ -# -*- coding: utf-8 -*- - -# Copyright 2011 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -# Software: eFiction -import time -import logging -logger = logging.getLogger(__name__) -import re -import urllib -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter, makeDate - -class PotionsAndSnitchesOrgSiteAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - self.story.setMetadata('siteabbrev','pns') - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - - # get storyId from url--url validation guarantees query is only sid=1234 - self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) - - - # normalized story URL. - self._setURL('http://' + self.getSiteDomain() + '/fanfiction/viewstory.php?sid='+self.story.getMetadata('storyId')) - - - @staticmethod - def getSiteDomain(): - return 'www.potionsandsnitches.org' - - @classmethod - def getAcceptDomains(cls): - return ['potionsandsnitches.org','potionsandsnitches.net'] - - @classmethod - def getSiteExampleURLs(cls): - return "http://www.potionsandsnitches.org/fanfiction/viewstory.php?sid=1234" - - def getSiteURLPattern(self): - return re.escape("http://")+r"(www\.)?potionsandsnitches\.(net|org)/fanfiction/viewstory\.php\?sid=\d+$" - - def extractChapterUrlsAndMetadata(self): - - url = self.url+'&index=1' - logger.debug("URL: "+url) - - try: - data = self._fetchUrl(url) - except urllib2.HTTPError, e: - if e.code == 404: - raise exceptions.StoryDoesNotExist(self.url) - else: - raise e - - if "Access denied. This story has not been validated by the adminstrators of this site." in data: - raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") - - # use BeautifulSoup HTML parser to make everything easier to find. - soup = bs.BeautifulSoup(data) - - ## Title - a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) - self.story.setMetadata('title',stripHTML(a)) - - # Find authorid and URL from... author url. - a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) - self.story.setMetadata('authorId',a['href'].split('=')[1]) - self.story.setMetadata('authorUrl','http://'+self.host+'/fanfiction/'+a['href']) - self.story.setMetadata('author',a.string) - - # Find the chapters: - for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): - # just in case there's tags, like in chapter titles. - self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfiction/'+chapter['href'])) - - self.story.setMetadata('numChapters',len(self.chapterUrls)) - - ## - ## Summary, strangely, is in the content attr of a tag - ## which is escaped HTML. Unfortunately, we can't use it because they don't - ## escape (') chars in the desc, breakin the tag. - #meta_desc = soup.find('meta',{'name':'description'}) - #metasoup = bs.BeautifulStoneSoup(meta_desc['content']) - #self.story.setMetadata('description',stripHTML(metasoup)) - - def defaultGetattr(d,k): - try: - return d[k] - except: - return "" - - # Rated: NC-17
etc - labels = soup.findAll('span',{'class':'label'}) - for labelspan in labels: - value = labelspan.nextSibling - label = labelspan.string - - if 'Summary' in label: - ## Everything until the next div class='listbox' - svalue = "" - while not defaultGetattr(value,'class') == 'listbox': - svalue += str(value) - value = value.nextSibling - self.setDescription(url,svalue) - #self.story.setMetadata('description',stripHTML(svalue)) - - if 'Rated' in label: - self.story.setMetadata('rating', value) - - if 'Word count' in label: - self.story.setMetadata('numWords', value) - - if 'Categories' in label: - cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) - catstext = [cat.string for cat in cats] - for cat in catstext: - self.story.addToList('category',cat.string) - - if 'Characters' in label: - chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) - charstext = [char.string for char in chars] - for char in charstext: - if "Snape and Harry (required)" in char: - self.story.addToList('characters',"Snape") - self.story.addToList('characters',"Harry") - else: - self.story.addToList('characters',char.string) - - if 'Genre' in label: - genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class')) - genrestext = [genre.string for genre in genres] - self.genre = ', '.join(genrestext) - for genre in genrestext: - self.story.addToList('genre',genre.string) - - if 'Completed' in label: - if 'Yes' in value: - self.story.setMetadata('status', 'Completed') - else: - self.story.setMetadata('status', 'In-Progress') - - if 'Published' in label: - # limit date values, there's some extra chars. - self.story.setMetadata('datePublished', makeDate(stripHTML(value[:12]), "%d %b %Y")) - - if 'Updated' in label: - self.story.setMetadata('dateUpdated', makeDate(stripHTML(value[:12]), "%d %b %Y")) - - try: - # Find Series name from series URL. - a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) - series_name = a.string - series_url = 'http://'+self.host+'/fanfiction/'+a['href'] - - # use BeautifulSoup HTML parser to make everything easier to find. - seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) - storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) - i=1 - for a in storyas: - if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): - self.setSeries(series_name, i) - self.story.setMetadata('seriesUrl',series_url) - break - i+=1 - - except: - # I find it hard to care if the series parsing fails - pass - - - def getChapterText(self, url): - - logger.debug('Getting chapter text from: %s' % url) - - soup = bs.BeautifulStoneSoup(self._fetchUrl(url), - selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. - - div = soup.find('div', {'id' : 'story'}) - - if None == div: - raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) - - return self.utf8FromSoup(url,div) - -def getClass(): - return PotionsAndSnitchesOrgSiteAdapter - +# -*- coding: utf-8 -*- + +# Copyright 2011 Fanficdownloader team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# Software: eFiction +import time +import logging +logger = logging.getLogger(__name__) +import re +import urllib +import urllib2 + +from .. import BeautifulSoup as bs +from ..htmlcleanup import stripHTML +from .. import exceptions as exceptions + +from base_adapter import BaseSiteAdapter, makeDate + +class PotionsAndSnitchesOrgSiteAdapter(BaseSiteAdapter): + + def __init__(self, config, url): + BaseSiteAdapter.__init__(self, config, url) + self.story.setMetadata('siteabbrev','pns') + self.decode = ["Windows-1252", + "utf8"] # 1252 is a superset of iso-8859-1. + # Most sites that claim to be + # iso-8859-1 (and some that claim to be + # utf8) are really windows-1252. + + # get storyId from url--url validation guarantees query is only sid=1234 + self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1]) + + + # normalized story URL. + self._setURL('http://' + self.getSiteDomain() + '/fanfiction/viewstory.php?sid='+self.story.getMetadata('storyId')) + + + @staticmethod + def getSiteDomain(): + return 'www.potionsandsnitches.org' + + @classmethod + def getAcceptDomains(cls): + return ['potionsandsnitches.org','potionsandsnitches.net'] + + @classmethod + def getSiteExampleURLs(cls): + return "http://www.potionsandsnitches.org/fanfiction/viewstory.php?sid=1234" + + def getSiteURLPattern(self): + return re.escape("http://")+r"(www\.)?potionsandsnitches\.(net|org)/fanfiction/viewstory\.php\?sid=\d+$" + + def extractChapterUrlsAndMetadata(self): + + url = self.url+'&index=1' + logger.debug("URL: "+url) + + try: + data = self._fetchUrl(url) + except urllib2.HTTPError, e: + if e.code == 404: + raise exceptions.StoryDoesNotExist(self.url) + else: + raise e + + if "Access denied. This story has not been validated by the adminstrators of this site." in data: + raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.") + + # use BeautifulSoup HTML parser to make everything easier to find. + soup = bs.BeautifulSoup(data) + + ## Title + a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$")) + self.story.setMetadata('title',stripHTML(a)) + + # Find authorid and URL from... author url. + a = soup.find('a', href=re.compile(r"viewuser.php\?uid=\d+")) + self.story.setMetadata('authorId',a['href'].split('=')[1]) + self.story.setMetadata('authorUrl','http://'+self.host+'/fanfiction/'+a['href']) + self.story.setMetadata('author',a.string) + + # Find the chapters: + for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")): + # just in case there's tags, like in chapter titles. + self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/fanfiction/'+chapter['href'])) + + self.story.setMetadata('numChapters',len(self.chapterUrls)) + + ## + ## Summary, strangely, is in the content attr of a tag + ## which is escaped HTML. Unfortunately, we can't use it because they don't + ## escape (') chars in the desc, breakin the tag. + #meta_desc = soup.find('meta',{'name':'description'}) + #metasoup = bs.BeautifulStoneSoup(meta_desc['content']) + #self.story.setMetadata('description',stripHTML(metasoup)) + + def defaultGetattr(d,k): + try: + return d[k] + except: + return "" + + # Rated: NC-17
etc + labels = soup.findAll('span',{'class':'label'}) + for labelspan in labels: + value = labelspan.nextSibling + label = labelspan.string + + if 'Summary' in label: + ## Everything until the next div class='listbox' + svalue = "" + while not defaultGetattr(value,'class') == 'listbox': + svalue += str(value) + value = value.nextSibling + self.setDescription(url,svalue) + #self.story.setMetadata('description',stripHTML(svalue)) + + if 'Rated' in label: + self.story.setMetadata('rating', value) + + if 'Word count' in label: + self.story.setMetadata('numWords', value) + + if 'Categories' in label: + cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories')) + catstext = [cat.string for cat in cats] + for cat in catstext: + self.story.addToList('category',cat.string) + + if 'Characters' in label: + chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters')) + charstext = [char.string for char in chars] + for char in charstext: + if "Snape and Harry (required)" in char: + self.story.addToList('characters',"Snape") + self.story.addToList('characters',"Harry") + else: + self.story.addToList('characters',char.string) + + if 'Genre' in label: + genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class')) + genrestext = [genre.string for genre in genres] + self.genre = ', '.join(genrestext) + for genre in genrestext: + self.story.addToList('genre',genre.string) + + if 'Completed' in label: + if 'Yes' in value: + self.story.setMetadata('status', 'Completed') + else: + self.story.setMetadata('status', 'In-Progress') + + if 'Published' in label: + # limit date values, there's some extra chars. + self.story.setMetadata('datePublished', makeDate(stripHTML(value[:12]), "%d %b %Y")) + + if 'Updated' in label: + self.story.setMetadata('dateUpdated', makeDate(stripHTML(value[:12]), "%d %b %Y")) + + try: + # Find Series name from series URL. + a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+")) + series_name = a.string + series_url = 'http://'+self.host+'/fanfiction/'+a['href'] + + # use BeautifulSoup HTML parser to make everything easier to find. + seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url)) + storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$')) + i=1 + for a in storyas: + if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')): + self.setSeries(series_name, i) + self.story.setMetadata('seriesUrl',series_url) + break + i+=1 + + except: + # I find it hard to care if the series parsing fails + pass + + + def getChapterText(self, url): + + logger.debug('Getting chapter text from: %s' % url) + + soup = bs.BeautifulStoneSoup(self._fetchUrl(url), + selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags. + + div = soup.find('div', {'id' : 'story'}) + + if None == div: + raise exceptions.FailedToDownload("Error downloading Chapter: %s! Missing required element!" % url) + + return self.utf8FromSoup(url,div) + +def getClass(): + return PotionsAndSnitchesOrgSiteAdapter + diff --git a/fff_internals/adapters/adapter_potterficscom.py b/fanficfare/adapters/adapter_potterficscom.py similarity index 97% rename from fff_internals/adapters/adapter_potterficscom.py rename to fanficfare/adapters/adapter_potterficscom.py index c837927..3d0aab6 100644 --- a/fff_internals/adapters/adapter_potterficscom.py +++ b/fanficfare/adapters/adapter_potterficscom.py @@ -1,281 +1,281 @@ -# -*- coding: utf-8 -*- - -# Copyright 2013 Fanficdownloader team -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import datetime -import logging -logger = logging.getLogger(__name__) -import re -import urllib2 - -from .. import BeautifulSoup as bs -from ..htmlcleanup import stripHTML -from .. import exceptions as exceptions - -from base_adapter import BaseSiteAdapter - -# This function is called by the downloader in all adapter_*.py files -# in this dir to register the adapter class. So it needs to be -# updated to reflect the class below it. That, plus getSiteDomain() -# take care of 'Registering'. -def getClass(): - return PotterFicsComAdapter - -# Class name has to be unique. Our convention is camel case the -# sitename with Adapter at the end. www is skipped. -class PotterFicsComAdapter(BaseSiteAdapter): - - def __init__(self, config, url): - BaseSiteAdapter.__init__(self, config, url) - - self.decode = ["Windows-1252", - "utf8"] # 1252 is a superset of iso-8859-1. - # Most sites that claim to be - # iso-8859-1 (and some that claim to be - # utf8) are really windows-1252. - self.username = "NoneGiven" # if left empty, site doesn't return any message at all. - self.password = "" - self.is_adult=False - - # get storyId from url--url validation guarantees query correct - m = re.match(self.getSiteURLPattern(),url) - if m: - self.story.setMetadata('storyId',m.group('id')) - - # normalized story URL. gets rid of chapter if there, left with chapter index URL - nurl = "http://"+self.getSiteDomain()+"/historias/"+self.story.getMetadata('storyId') - self._setURL(nurl) - else: - raise exceptions.InvalidStoryURL(url, - self.getSiteDomain(), - self.getSiteExampleURLs()) - - - # Each adapter needs to have a unique site abbreviation. - self.story.setMetadata('siteabbrev','potficscom') - - @staticmethod # must be @staticmethod, don't remove it. - def getSiteDomain(): - # The site domain. Does have www here, if it uses it. - return 'www.potterfics.com' - - @classmethod - def getSiteExampleURLs(cls): - return "http://www.potterfics.com/historias/12345 http://www.potterfics.com/historias/12345/capitulo-1 " - - def getSiteURLPattern(self): - #http://www.potterfics.com/historias/127583 - #http://www.potterfics.com/historias/127583/capitulo-1 - #http://www.potterfics.com/historias/127583/capitulo-4 - #http://www.potterfics.com/historias/92810 -> Complete story - #http://www.potterfics.com/historias/111194 -> Complete, single chap - p = re.escape("http://"+self.getSiteDomain()+"/historias/")+\ - r"(?P\d+)(/capitulo-(?P\d+))?/?$" - return p - - def needToLoginCheck(self, data): - # partials used to avoid having to figure out what was wrong - # with included utf8 higher chars. - if 'Para ver esta historia, por favor inicia tu sesi' in data \ - or '