mirror of
https://github.com/wassname/FanFicFare.git
synced 2026-10-04 12:10:26 +08:00
Rewrite of epub update, now handles images.
This commit is contained in:
1 parent
f0445f106c
commit
a9cecbbe4f
11 files changed
+267
-454
No files matched your search
@@ -27,7 +27,7 @@ class FanFictionDownLoaderBase(InterfaceActionBase):
|
||||
description = 'UI plugin to download FanFiction stories from various sites.'
|
||||
supported_platforms = ['windows', 'osx', 'linux']
|
||||
author = 'Jim Miller'
|
||||
version = (1, 4, 6)
|
||||
version = (1, 5, 0)
|
||||
minimum_calibre_version = (0, 8, 30)
|
||||
|
||||
#: This field defines the GUI plugin class that contains all the code
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
#!/usr/bin/env python
|
||||
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
|
||||
from __future__ import (unicode_literals, division, absolute_import,
|
||||
print_function)
|
||||
|
||||
__license__ = 'GPL v3'
|
||||
__copyright__ = '2012, Jim Miller'
|
||||
__docformat__ = 'restructuredtext en'
|
||||
|
||||
from zipfile import ZipFile
|
||||
|
||||
from xml.dom.minidom import parseString
|
||||
|
||||
def get_dcsource(inputio):
|
||||
epub = ZipFile(inputio, 'r')
|
||||
|
||||
## Find the .opf file.
|
||||
container = epub.read("META-INF/container.xml")
|
||||
containerdom = parseString(container)
|
||||
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
|
||||
rootfilename = rootfilenodelist[0].getAttribute("full-path")
|
||||
|
||||
metadom = parseString(epub.read(rootfilename))
|
||||
firstmetadom = metadom.getElementsByTagName("metadata")[0]
|
||||
try:
|
||||
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
|
||||
except:
|
||||
source=None
|
||||
|
||||
return source
|
||||
@@ -35,8 +35,8 @@ from calibre_plugins.fanfictiondownloader_plugin.common_utils import (set_plugin
|
||||
|
||||
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
|
||||
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.htmlcleanup import stripHTML
|
||||
from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
|
||||
from calibre_plugins.fanfictiondownloader_plugin.dcsource import get_dcsource
|
||||
#from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
|
||||
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.epubutils import get_dcsource, get_dcsource_chaptercount
|
||||
|
||||
from calibre_plugins.fanfictiondownloader_plugin.config import (prefs, permitted_values)
|
||||
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (
|
||||
@@ -524,13 +524,15 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
|
||||
# 'book' can exist without epub. If there's no existing epub,
|
||||
# let it go and it will download it.
|
||||
if db.has_format(book_id,fileform,index_is_id=True):
|
||||
toupdateio = StringIO()
|
||||
(epuburl,chaptercount) = doMerge(toupdateio,
|
||||
[StringIO(db.format(book_id,'EPUB',
|
||||
index_is_id=True))],
|
||||
titlenavpoints=False,
|
||||
striptitletoc=True,
|
||||
forceunique=False)
|
||||
#toupdateio = StringIO()
|
||||
(epuburl,chaptercount) = get_dcsource_chaptercount(StringIO(db.format(book_id,'EPUB',
|
||||
index_is_id=True)))
|
||||
# (epuburl,chaptercount) = doMerge(toupdateio,
|
||||
# [StringIO(db.format(book_id,'EPUB',
|
||||
# index_is_id=True))],
|
||||
# titlenavpoints=False,
|
||||
# striptitletoc=True,
|
||||
# forceunique=False)
|
||||
urlchaptercount = int(story.getMetadata('numChapters'))
|
||||
if chaptercount == urlchaptercount:
|
||||
if collision == UPDATE:
|
||||
|
||||
+34
-27
@@ -23,7 +23,8 @@ from calibre.utils.logging import Log
|
||||
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (NotGoingToDownload,
|
||||
OVERWRITE, OVERWRITEALWAYS, UPDATE, UPDATEALWAYS, ADDNEW, SKIP, CALIBREONLY)
|
||||
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
|
||||
from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
|
||||
#from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
|
||||
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.epubutils import get_update_data
|
||||
|
||||
# ------------------------------------------------------------------------------
|
||||
#
|
||||
@@ -136,38 +137,44 @@ def do_download_for_worker(book,options):
|
||||
elif 'epub_for_update' in book and options['collision'] in (UPDATE, UPDATEALWAYS):
|
||||
|
||||
urlchaptercount = int(story.getMetadata('numChapters'))
|
||||
## First, get existing epub with titlepage and tocpage stripped.
|
||||
updateio = StringIO()
|
||||
(epuburl,chaptercount) = doMerge(updateio,
|
||||
[book['epub_for_update']],
|
||||
titlenavpoints=False,
|
||||
striptitletoc=True,
|
||||
forceunique=False)
|
||||
(url,chaptercount,
|
||||
adapter.oldchapters,
|
||||
adapter.oldimgs) = get_update_data(book['epub_for_update'])
|
||||
|
||||
print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount))
|
||||
print("write to %s"%outfile)
|
||||
|
||||
## Get updated title page/metadata by itself in an epub.
|
||||
## Even if the title page isn't included, this carries the metadata.
|
||||
titleio = StringIO()
|
||||
writer.writeStory(outstream=titleio,metaonly=True)
|
||||
writer.writeStory(outfilename=outfile, forceOverwrite=True)
|
||||
|
||||
## First, get existing epub with titlepage and tocpage stripped.
|
||||
# updateio = StringIO()
|
||||
# (epuburl,chaptercount) = doMerge(updateio,
|
||||
# [book['epub_for_update']],
|
||||
# titlenavpoints=False,
|
||||
# striptitletoc=True,
|
||||
# forceunique=False)
|
||||
# ## Get updated title page/metadata by itself in an epub.
|
||||
# ## Even if the title page isn't included, this carries the metadata.
|
||||
# titleio = StringIO()
|
||||
# writer.writeStory(outstream=titleio,metaonly=True)
|
||||
|
||||
newchaptersio = None
|
||||
if urlchaptercount > chaptercount :
|
||||
## Go get the new chapters
|
||||
newchaptersio = StringIO()
|
||||
adapter.setChaptersRange(chaptercount+1,urlchaptercount)
|
||||
# newchaptersio = None
|
||||
# if urlchaptercount > chaptercount :
|
||||
# ## Go get the new chapters
|
||||
# newchaptersio = StringIO()
|
||||
# adapter.setChaptersRange(chaptercount+1,urlchaptercount)
|
||||
|
||||
adapter.config.set("overrides",'include_tocpage','false')
|
||||
adapter.config.set("overrides",'include_titlepage','false')
|
||||
writer.writeStory(outstream=newchaptersio)
|
||||
# adapter.config.set("overrides",'include_tocpage','false')
|
||||
# adapter.config.set("overrides",'include_titlepage','false')
|
||||
# writer.writeStory(outstream=newchaptersio)
|
||||
|
||||
## Merge the three epubs together.
|
||||
doMerge(outfile,
|
||||
[titleio,updateio,newchaptersio],
|
||||
fromfirst=True,
|
||||
titlenavpoints=False,
|
||||
striptitletoc=False,
|
||||
forceunique=False)
|
||||
# ## Merge the three epubs together.
|
||||
# doMerge(outfile,
|
||||
# [titleio,updateio,newchaptersio],
|
||||
# fromfirst=True,
|
||||
# titlenavpoints=False,
|
||||
# striptitletoc=False,
|
||||
# forceunique=False)
|
||||
|
||||
book['comment'] = 'Update %s completed, added %s chapters for %s total.'%\
|
||||
(options['fileform'],(urlchaptercount-chaptercount),urlchaptercount)
|
||||
|
||||
+30
-26
@@ -25,19 +25,16 @@ from StringIO import StringIO
|
||||
from optparse import OptionParser
|
||||
import getpass
|
||||
import string
|
||||
|
||||
import ConfigParser
|
||||
from subprocess import call
|
||||
|
||||
from epubmerge import doMerge
|
||||
from fanficdownloader import adapters,writers,exceptions
|
||||
from fanficdownloader.epubutils import get_dcsource_chaptercount, get_update_data
|
||||
|
||||
if sys.version_info < (2, 5):
|
||||
print "This program requires Python 2.5 or newer."
|
||||
sys.exit(1)
|
||||
|
||||
from fanficdownloader import adapters,writers,exceptions
|
||||
|
||||
import ConfigParser
|
||||
|
||||
def writeStory(config,adapter,writeformat,metaonly=False,outstream=None):
|
||||
writer = writers.getWriter(writeformat,config,adapter)
|
||||
writer.writeStory(outstream=outstream,metaonly=metaonly)
|
||||
@@ -116,12 +113,13 @@ def main():
|
||||
try:
|
||||
## Attempt to update an existing epub.
|
||||
if options.update:
|
||||
updateio = StringIO()
|
||||
(url,chaptercount) = doMerge(updateio,
|
||||
args,
|
||||
titlenavpoints=False,
|
||||
striptitletoc=True,
|
||||
forceunique=False)
|
||||
# updateio = StringIO()
|
||||
# (url,chaptercount) = doMerge(updateio,
|
||||
# args,
|
||||
# titlenavpoints=False,
|
||||
# striptitletoc=True,
|
||||
# forceunique=False)
|
||||
(url,chaptercount) = get_dcsource_chaptercount(args[0])
|
||||
print "Updating %s, URL: %s" % (args[0],url)
|
||||
output_filename = args[0]
|
||||
config.set("overrides","output_filename",args[0])
|
||||
@@ -167,17 +165,23 @@ def main():
|
||||
print "Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount)
|
||||
## Get updated title page/metadata by itself in an epub.
|
||||
## Even if the title page isn't included, this carries the metadata.
|
||||
titleio = StringIO()
|
||||
writeStory(config,adapter,"epub",metaonly=True,outstream=titleio)
|
||||
# titleio = StringIO()
|
||||
# writeStory(config,adapter,"epub",metaonly=True,outstream=titleio)
|
||||
|
||||
newchaptersio = None
|
||||
# newchaptersio = None
|
||||
if not options.metaonly:
|
||||
(url,chaptercount,
|
||||
adapter.oldchapters,
|
||||
adapter.oldimgs) = get_update_data(args[0])
|
||||
|
||||
writeStory(config,adapter,"epub")
|
||||
|
||||
## Go get the new chapters only in another epub.
|
||||
newchaptersio = StringIO()
|
||||
adapter.setChaptersRange(chaptercount+1,urlchaptercount)
|
||||
config.set("overrides",'include_tocpage','false')
|
||||
config.set("overrides",'include_titlepage','false')
|
||||
writeStory(config,adapter,"epub",outstream=newchaptersio)
|
||||
# newchaptersio = StringIO()
|
||||
# adapter.setChaptersRange(chaptercount+1,urlchaptercount)
|
||||
# config.set("overrides",'include_tocpage','false')
|
||||
# config.set("overrides",'include_titlepage','false')
|
||||
# writeStory(config,adapter,"epub",outstream=newchaptersio)
|
||||
|
||||
# out = open("testing/titleio.epub","wb")
|
||||
# out.write(titleio.getvalue())
|
||||
@@ -192,12 +196,12 @@ def main():
|
||||
# out.close()
|
||||
|
||||
## Merge the three epubs together.
|
||||
doMerge(args[0],
|
||||
[titleio,updateio,newchaptersio],
|
||||
fromfirst=True,
|
||||
titlenavpoints=False,
|
||||
striptitletoc=False,
|
||||
forceunique=False)
|
||||
# doMerge(args[0],
|
||||
# [titleio,updateio,newchaptersio],
|
||||
# fromfirst=True,
|
||||
# titlenavpoints=False,
|
||||
# striptitletoc=False,
|
||||
# forceunique=False)
|
||||
|
||||
else:
|
||||
# regular download
|
||||
|
||||
+2
-318
@@ -16,326 +16,10 @@
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import sys
|
||||
import os
|
||||
import re
|
||||
#import StringIO
|
||||
from optparse import OptionParser
|
||||
|
||||
import zlib
|
||||
import zipfile
|
||||
from zipfile import ZipFile, ZIP_STORED, ZIP_DEFLATED
|
||||
from time import time
|
||||
|
||||
from exceptions import KeyError
|
||||
|
||||
from xml.dom.minidom import parse, parseString, getDOMImplementation
|
||||
|
||||
def doMerge(outputio,files,authoropts=[],titleopt=None,descopt=None,
|
||||
fromfirst=False,
|
||||
titlenavpoints=True,
|
||||
striptitletoc=False,
|
||||
forceunique=True):
|
||||
'''
|
||||
outputio = output file name or StringIO.
|
||||
files = list of input file names or StringIOs.
|
||||
authoropts = list of authors to use, otherwise add from all input
|
||||
titleopt = title, otherwise '<first title> Anthology'
|
||||
descopt = description, otherwise '<title> by <author>' list for all input
|
||||
fromfirst if true, take all metadata (including author, title, desc) from first input
|
||||
titlenavpoints if true, put in a new TOC entry for each epub
|
||||
striptitletoc if true, strip out any (title|toc)_page.xhtml files
|
||||
forceunique if true, guarantee uniqueness of contents by adding a dir for each input
|
||||
'''
|
||||
## Python 2.5 ZipFile is rather more primative than later
|
||||
## versions. It can operate on a file, or on a StringIO, but
|
||||
## not on an open stream. OTOH, I suspect we would have had
|
||||
## problems with closing and opening again to change the
|
||||
## compression type anyway.
|
||||
|
||||
filecount=0
|
||||
source=None
|
||||
|
||||
## Write mimetype file, must be first and uncompressed.
|
||||
## Older versions of python(2.4/5) don't allow you to specify
|
||||
## compression by individual file.
|
||||
## Overwrite if existing output file.
|
||||
outputepub = ZipFile(outputio, "w", compression=ZIP_STORED)
|
||||
outputepub.debug = 3
|
||||
outputepub.writestr("mimetype", "application/epub+zip")
|
||||
outputepub.close()
|
||||
|
||||
## Re-open file for content.
|
||||
outputepub = ZipFile(outputio, "a", compression=ZIP_DEFLATED)
|
||||
outputepub.debug = 3
|
||||
|
||||
## Create META-INF/container.xml file. The only thing it does is
|
||||
## point to content.opf
|
||||
containerdom = getDOMImplementation().createDocument(None, "container", None)
|
||||
containertop = containerdom.documentElement
|
||||
containertop.setAttribute("version","1.0")
|
||||
containertop.setAttribute("xmlns","urn:oasis:names:tc:opendocument:xmlns:container")
|
||||
rootfiles = containerdom.createElement("rootfiles")
|
||||
containertop.appendChild(rootfiles)
|
||||
rootfiles.appendChild(newTag(containerdom,"rootfile",{"full-path":"content.opf",
|
||||
"media-type":"application/oebps-package+xml"}))
|
||||
outputepub.writestr("META-INF/container.xml",containerdom.toprettyxml(indent=' ',encoding='utf-8'))
|
||||
|
||||
## Process input epubs.
|
||||
|
||||
items = [] # list of (id, href, type) tuples(all strings) -- From .opfs' manifests
|
||||
items.append(("ncx","toc.ncx","application/x-dtbncx+xml")) ## we'll generate the toc.ncx file,
|
||||
## but it needs to be in the items manifest.
|
||||
itemrefs = [] # list of strings -- idrefs from .opfs' spines
|
||||
navmaps = [] # list of navMap DOM elements -- TOC data for each from toc.ncx files
|
||||
|
||||
booktitles = [] # list of strings -- Each book's title
|
||||
allauthors = [] # list of lists of strings -- Each book's list of authors.
|
||||
|
||||
filelist = []
|
||||
|
||||
booknum=1
|
||||
firstmetadom = None
|
||||
for file in files:
|
||||
if file == None : continue
|
||||
|
||||
book = "%d" % booknum
|
||||
bookdir = ""
|
||||
bookid = ""
|
||||
if forceunique:
|
||||
bookdir = "%d/" % booknum
|
||||
bookid = "a%d" % booknum
|
||||
#print "book %d" % booknum
|
||||
|
||||
epub = ZipFile(file, 'r')
|
||||
|
||||
## Find the .opf file.
|
||||
container = epub.read("META-INF/container.xml")
|
||||
containerdom = parseString(container)
|
||||
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
|
||||
rootfilename = rootfilenodelist[0].getAttribute("full-path")
|
||||
|
||||
## Save the path to the .opf file--hrefs inside it are relative to it.
|
||||
relpath = os.path.dirname(rootfilename)
|
||||
if( len(relpath) > 0 ):
|
||||
relpath=relpath+"/"
|
||||
|
||||
metadom = parseString(epub.read(rootfilename))
|
||||
if booknum==1:
|
||||
firstmetadom = metadom.getElementsByTagName("metadata")[0]
|
||||
try:
|
||||
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
|
||||
except:
|
||||
source=""
|
||||
#print "Source:%s"%source
|
||||
|
||||
## Save indiv book title
|
||||
booktitles.append(metadom.getElementsByTagName("dc:title")[0].firstChild.data)
|
||||
|
||||
## Save authors.
|
||||
authors=[]
|
||||
for creator in metadom.getElementsByTagName("dc:creator"):
|
||||
if( creator.getAttribute("opf:role") == "aut" ):
|
||||
authors.append(creator.firstChild.data)
|
||||
allauthors.append(authors)
|
||||
|
||||
for item in metadom.getElementsByTagName("item"):
|
||||
if( item.getAttribute("media-type") == "application/x-dtbncx+xml" ):
|
||||
# TOC file is only one with this type--as far as I know.
|
||||
# grab the whole navmap, deal with it later.
|
||||
tocdom = parseString(epub.read(relpath+item.getAttribute("href")))
|
||||
|
||||
for navpoint in tocdom.getElementsByTagName("navPoint"):
|
||||
navpoint.setAttribute("id",bookid+navpoint.getAttribute("id"))
|
||||
|
||||
for content in tocdom.getElementsByTagName("content"):
|
||||
content.setAttribute("src",bookdir+relpath+content.getAttribute("src"))
|
||||
|
||||
navmaps.append(tocdom.getElementsByTagName("navMap")[0])
|
||||
else:
|
||||
id=bookid+item.getAttribute("id")
|
||||
href=bookdir+relpath+item.getAttribute("href")
|
||||
href=href.encode('utf8')
|
||||
#print "href:"+href
|
||||
if not striptitletoc or not re.match(r'.*/((title|toc)_page|cover)\.xhtml',
|
||||
item.getAttribute("href")):
|
||||
if href not in filelist:
|
||||
try:
|
||||
outputepub.writestr(href,
|
||||
epub.read(relpath+item.getAttribute("href")))
|
||||
if re.match(r'.*/(file|chapter)\d+\.x?html',href):
|
||||
filecount+=1
|
||||
items.append((id,href,item.getAttribute("media-type")))
|
||||
filelist.append(href)
|
||||
except KeyError, ke:
|
||||
pass # Skip missing files.
|
||||
|
||||
for itemref in metadom.getElementsByTagName("itemref"):
|
||||
|
||||
if not striptitletoc or not re.match(r'((title|toc)_page|cover)', itemref.getAttribute("idref")):
|
||||
itemrefs.append(bookid+itemref.getAttribute("idref"))
|
||||
|
||||
booknum=booknum+1;
|
||||
if not forceunique:
|
||||
# If not forceunique, it's an epub update.
|
||||
# If there's a "calibre_bookmarks.txt", it's from reading
|
||||
# in Calibre and should be preserved.
|
||||
try:
|
||||
fn = "META-INF/calibre_bookmarks.txt"
|
||||
outputepub.writestr(fn,epub.read(fn))
|
||||
except:
|
||||
pass
|
||||
|
||||
|
||||
## create content.opf file.
|
||||
uniqueid="epubmerge-uid-%d" % time() # real sophisticated uid scheme.
|
||||
contentdom = getDOMImplementation().createDocument(None, "package", None)
|
||||
package = contentdom.documentElement
|
||||
if fromfirst and firstmetadom:
|
||||
metadata = firstmetadom
|
||||
firstpackage = firstmetadom.parentNode
|
||||
package.setAttribute("version",firstpackage.getAttribute("version"))
|
||||
package.setAttribute("xmlns",firstpackage.getAttribute("xmlns"))
|
||||
package.setAttribute("unique-identifier",firstpackage.getAttribute("unique-identifier"))
|
||||
else:
|
||||
package.setAttribute("version","2.0")
|
||||
package.setAttribute("xmlns","http://www.idpf.org/2007/opf")
|
||||
package.setAttribute("unique-identifier","epubmerge-id")
|
||||
metadata=newTag(contentdom,"metadata",
|
||||
attrs={"xmlns:dc":"http://purl.org/dc/elements/1.1/",
|
||||
"xmlns:opf":"http://www.idpf.org/2007/opf"})
|
||||
metadata.appendChild(newTag(contentdom,"dc:identifier",text=uniqueid,attrs={"id":"epubmerge-id"}))
|
||||
if( titleopt is None ):
|
||||
titleopt = booktitles[0]+" Anthology"
|
||||
metadata.appendChild(newTag(contentdom,"dc:title",text=titleopt))
|
||||
|
||||
# If cmdline authors, use those instead of those collected from the epubs
|
||||
# (allauthors kept for TOC & description gen below.
|
||||
if( len(authoropts) > 1 ):
|
||||
useauthors=[authoropts]
|
||||
else:
|
||||
useauthors=allauthors
|
||||
|
||||
usedauthors=dict()
|
||||
for authorlist in useauthors:
|
||||
for author in authorlist:
|
||||
if( not usedauthors.has_key(author) ):
|
||||
usedauthors[author]=author
|
||||
metadata.appendChild(newTag(contentdom,"dc:creator",
|
||||
attrs={"opf:role":"aut"},
|
||||
text=author))
|
||||
|
||||
metadata.appendChild(newTag(contentdom,"dc:contributor",text="epubmerge",attrs={"opf:role":"bkp"}))
|
||||
metadata.appendChild(newTag(contentdom,"dc:rights",text="Copyrights as per source stories"))
|
||||
metadata.appendChild(newTag(contentdom,"dc:language",text="en"))
|
||||
|
||||
if not descopt:
|
||||
# created now, but not filled in until TOC generation to save loops.
|
||||
description = newTag(contentdom,"dc:description",text="Anthology containing:\n")
|
||||
else:
|
||||
description = newTag(contentdom,"dc:description",text=descopt)
|
||||
metadata.appendChild(description)
|
||||
|
||||
package.appendChild(metadata)
|
||||
|
||||
manifest = contentdom.createElement("manifest")
|
||||
package.appendChild(manifest)
|
||||
for item in items:
|
||||
(id,href,type)=item
|
||||
manifest.appendChild(newTag(contentdom,"item",
|
||||
attrs={'id':id,
|
||||
'href':href,
|
||||
'media-type':type}))
|
||||
|
||||
spine = newTag(contentdom,"spine",attrs={"toc":"ncx"})
|
||||
package.appendChild(spine)
|
||||
for itemref in itemrefs:
|
||||
spine.appendChild(newTag(contentdom,"itemref",
|
||||
attrs={"idref":itemref,
|
||||
"linear":"yes"}))
|
||||
|
||||
## create toc.ncx file
|
||||
tocncxdom = getDOMImplementation().createDocument(None, "ncx", None)
|
||||
ncx = tocncxdom.documentElement
|
||||
ncx.setAttribute("version","2005-1")
|
||||
ncx.setAttribute("xmlns","http://www.daisy.org/z3986/2005/ncx/")
|
||||
head = tocncxdom.createElement("head")
|
||||
ncx.appendChild(head)
|
||||
head.appendChild(newTag(tocncxdom,"meta",
|
||||
attrs={"name":"dtb:uid", "content":uniqueid}))
|
||||
head.appendChild(newTag(tocncxdom,"meta",
|
||||
attrs={"name":"dtb:depth", "content":"1"}))
|
||||
head.appendChild(newTag(tocncxdom,"meta",
|
||||
attrs={"name":"dtb:totalPageCount", "content":"0"}))
|
||||
head.appendChild(newTag(tocncxdom,"meta",
|
||||
attrs={"name":"dtb:maxPageNumber", "content":"0"}))
|
||||
|
||||
docTitle = tocncxdom.createElement("docTitle")
|
||||
docTitle.appendChild(newTag(tocncxdom,"text",text=titleopt))
|
||||
ncx.appendChild(docTitle)
|
||||
|
||||
tocnavMap = tocncxdom.createElement("navMap")
|
||||
ncx.appendChild(tocnavMap)
|
||||
|
||||
## TOC navPoints can be nested, but this flattens them for
|
||||
## simplicity, plus adds a navPoint for each epub.
|
||||
booknum=0
|
||||
for navmap in navmaps:
|
||||
navpoints = navmap.getElementsByTagName("navPoint")
|
||||
if titlenavpoints:
|
||||
## Copy first navPoint of each epub, give a different id and
|
||||
## text: bookname by authorname
|
||||
newnav = navpoints[0].cloneNode(True)
|
||||
newnav.setAttribute("id","book"+newnav.getAttribute("id"))
|
||||
## For purposes of TOC titling & desc, use first book author
|
||||
newtext = newTag(tocncxdom,"text",text=booktitles[booknum]+" by "+allauthors[booknum][0])
|
||||
text = newnav.getElementsByTagName("text")[0]
|
||||
text.parentNode.replaceChild(newtext,text)
|
||||
tocnavMap.appendChild(newnav)
|
||||
|
||||
if not descopt and not fromfirst:
|
||||
description.appendChild(contentdom.createTextNode(booktitles[booknum]+" by "+allauthors[booknum][0]+"\n"))
|
||||
|
||||
for navpoint in navpoints:
|
||||
#print "navpoint:%s"%navpoint.getAttribute("id")
|
||||
if not striptitletoc or not re.match(r'(title|toc)_page',navpoint.getAttribute("id")):
|
||||
tocnavMap.appendChild(navpoint)
|
||||
booknum=booknum+1;
|
||||
|
||||
## Force strict ordering of playOrder
|
||||
playorder=1
|
||||
for navpoint in tocncxdom.getElementsByTagName("navPoint"):
|
||||
navpoint.setAttribute("playOrder","%d" % playorder)
|
||||
if( not navpoint.getAttribute("id").startswith("book") ):
|
||||
playorder = playorder + 1
|
||||
|
||||
## content.opf written now due to description being filled in
|
||||
## during TOC generation to save loops.
|
||||
outputepub.writestr("content.opf",contentdom.toxml('utf-8'))
|
||||
outputepub.writestr("toc.ncx",tocncxdom.toxml('utf-8'))
|
||||
|
||||
# declares all the files created by Windows. otherwise, when
|
||||
# it runs in appengine, windows unzips the files as 000 perms.
|
||||
for zf in outputepub.filelist:
|
||||
zf.create_system = 0
|
||||
outputepub.close()
|
||||
|
||||
return (source,filecount)
|
||||
|
||||
## Utility method for creating new tags.
|
||||
def newTag(dom,name,attrs=None,text=None):
|
||||
tag = dom.createElement(name)
|
||||
if( attrs is not None ):
|
||||
for attr in attrs.keys():
|
||||
tag.setAttribute(attr,attrs[attr])
|
||||
if( text is not None ):
|
||||
tag.appendChild(dom.createTextNode(text))
|
||||
return tag
|
||||
|
||||
if __name__ == "__main__":
|
||||
print('''
|
||||
This version is only used by fanfictiondownloader now. See:
|
||||
http://code.google.com/p/epubmerge/
|
||||
The this utility has been split out into it's own project.
|
||||
See: http://code.google.com/p/epubmerge/
|
||||
...for a CLI epubmerge.py program and calibre plugin.
|
||||
''')
|
||||
@@ -126,9 +126,9 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
|
||||
('Chapter 4',self.url+"&chapter=5"),
|
||||
('Chapter 5',self.url+"&chapter=6"),
|
||||
('Chapter 6',self.url+"&chapter=6"),
|
||||
('Chapter 7',self.url+"&chapter=6"),
|
||||
('Chapter 8',self.url+"&chapter=6"),
|
||||
('Chapter 9',self.url+"&chapter=6"),
|
||||
# ('Chapter 7',self.url+"&chapter=6"),
|
||||
# ('Chapter 8',self.url+"&chapter=6"),
|
||||
# ('Chapter 9',self.url+"&chapter=6"),
|
||||
# ('Chapter 0',self.url+"&chapter=6"),
|
||||
# ('Chapter a',self.url+"&chapter=6"),
|
||||
# ('Chapter b',self.url+"&chapter=6"),
|
||||
@@ -177,7 +177,7 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
|
||||
else:
|
||||
text=u'''
|
||||
<div>
|
||||
<h3>Chapter</h3>
|
||||
<h3>Chapter title from site</h3>
|
||||
<p><center>Centered text</center></p>
|
||||
<p>Lorem '''+self.crazystring+''' <i>italics</i>, <b>bold</b>, <u>underline</u> consectetur adipisicing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat. Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur. Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit anim id est laborum.</p>
|
||||
br breaks<br><br>
|
||||
|
||||
@@ -22,6 +22,7 @@ import logging
|
||||
import urllib
|
||||
import urllib2 as u2
|
||||
import urlparse as up
|
||||
from functools import partial
|
||||
|
||||
from .. import BeautifulSoup as bs
|
||||
from ..htmlcleanup import stripHTML
|
||||
@@ -86,6 +87,8 @@ class BaseSiteAdapter(Configurable):
|
||||
self.chapterUrls = [] # tuples of (chapter title,chapter url)
|
||||
self.chapterFirst = None
|
||||
self.chapterLast = None
|
||||
self.oldchapters = None
|
||||
self.oldimgs = None
|
||||
## order of preference for decoding.
|
||||
self.decode = ["utf8",
|
||||
"Windows-1252"] # 1252 is a superset of
|
||||
@@ -189,14 +192,21 @@ class BaseSiteAdapter(Configurable):
|
||||
def getStory(self):
|
||||
if not self.storyDone:
|
||||
self.getStoryMetadataOnly()
|
||||
|
||||
for index, (title,url) in enumerate(self.chapterUrls):
|
||||
if (self.chapterFirst!=None and index < self.chapterFirst) or \
|
||||
(self.chapterLast!=None and index > self.chapterLast):
|
||||
self.story.addChapter(removeEntities(title),
|
||||
None)
|
||||
else:
|
||||
if self.oldchapters and index < len(self.oldchapters):
|
||||
data = self.utf8FromSoup(None,
|
||||
self.oldchapters[index],
|
||||
partial(cachedfetch,self._fetchUrlRaw,self.oldimgs))
|
||||
else:
|
||||
data = self.getChapterText(url)
|
||||
self.story.addChapter(removeEntities(title),
|
||||
removeEntities(self.getChapterText(url)))
|
||||
removeEntities(data))
|
||||
self.storyDone = True
|
||||
|
||||
# include image, but no cover from story, add default_cover_image cover.
|
||||
@@ -264,7 +274,9 @@ class BaseSiteAdapter(Configurable):
|
||||
|
||||
# this gives us a unicode object, not just a string containing bytes.
|
||||
# (I gave soup a unicode string, you'd think it could give it back...)
|
||||
def utf8FromSoup(self,url,soup):
|
||||
def utf8FromSoup(self,url,soup,fetch=None):
|
||||
if not fetch:
|
||||
fetch=self._fetchUrlRaw
|
||||
|
||||
acceptable_attributes = ['href','name']
|
||||
#print("include_images:"+self.getConfig('include_images'))
|
||||
@@ -272,7 +284,7 @@ class BaseSiteAdapter(Configurable):
|
||||
acceptable_attributes.extend(('src','alt','origsrc'))
|
||||
for img in soup.findAll('img'):
|
||||
img['origsrc']=img['src']
|
||||
img['src']=self.story.addImgUrl(self,url,img['src'],self._fetchUrlRaw)
|
||||
img['src']=self.story.addImgUrl(self,url,img['src'],fetch)
|
||||
|
||||
for attr in soup._getAttrMap().keys():
|
||||
if attr not in acceptable_attributes:
|
||||
@@ -294,12 +306,21 @@ class BaseSiteAdapter(Configurable):
|
||||
# removes paired, but empty tags.
|
||||
if t.string != None and len(t.string.strip()) == 0 :
|
||||
t.extract()
|
||||
return soup.__str__('utf8').decode('utf-8')
|
||||
# Don't want body tags in chapter html--writers add them.
|
||||
return re.sub(r"</?body>\r?\n?","",soup.__str__('utf8').decode('utf-8'))
|
||||
|
||||
fullmon = {"January":"01", "February":"02", "March":"03", "April":"04", "May":"05",
|
||||
"June":"06","July":"07", "August":"08", "September":"09", "October":"10",
|
||||
"November":"11", "December":"12" }
|
||||
|
||||
def cachedfetch(realfetch,cache,url):
|
||||
if url in cache:
|
||||
print("cache hit")
|
||||
return cache[url]
|
||||
else:
|
||||
return realfetch(url)
|
||||
|
||||
|
||||
def makeDate(string,format):
|
||||
# Surprise! Abstracting this turned out to be more useful than
|
||||
# just saving bytes.
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python
|
||||
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
|
||||
from __future__ import (unicode_literals, division, absolute_import,
|
||||
print_function)
|
||||
|
||||
__license__ = 'GPL v3'
|
||||
__copyright__ = '2012, Jim Miller'
|
||||
__docformat__ = 'restructuredtext en'
|
||||
|
||||
import re, os, traceback
|
||||
from zipfile import ZipFile
|
||||
from xml.dom.minidom import parseString
|
||||
|
||||
from . import BeautifulSoup as bs
|
||||
|
||||
def get_dcsource(inputio):
|
||||
return get_update_data(inputio,getfilecount=False,getsoups=False)[0]
|
||||
|
||||
def get_dcsource_chaptercount(inputio):
|
||||
return get_update_data(inputio,getfilecount=True,getsoups=False)[:2] # (source,filecount)
|
||||
|
||||
def get_update_data(inputio,
|
||||
getfilecount=True,
|
||||
getsoups=True):
|
||||
epub = ZipFile(inputio, 'r')
|
||||
|
||||
## Find the .opf file.
|
||||
container = epub.read("META-INF/container.xml")
|
||||
containerdom = parseString(container)
|
||||
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
|
||||
rootfilename = rootfilenodelist[0].getAttribute("full-path")
|
||||
|
||||
contentdom = parseString(epub.read(rootfilename))
|
||||
firstmetadom = contentdom.getElementsByTagName("metadata")[0]
|
||||
try:
|
||||
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
|
||||
except:
|
||||
source=None
|
||||
|
||||
## Save the path to the .opf file--hrefs inside it are relative to it.
|
||||
relpath = get_path_part(rootfilename)
|
||||
|
||||
filecount = 0
|
||||
soups = [] # list of xhmtl blocks
|
||||
images = {} # dict() origsrc->data
|
||||
if getfilecount:
|
||||
# spin through the manifest--only place there are item tags.
|
||||
for item in contentdom.getElementsByTagName("item"):
|
||||
# First, count the 'chapter' files. FFDL uses file0000.xhtml,
|
||||
# but can also update epubs downloaded from Twisting the
|
||||
# Hellmouth, which uses chapter0.html.
|
||||
if( item.getAttribute("media-type") == "application/xhtml+xml" ):
|
||||
href=relpath+item.getAttribute("href")
|
||||
print("---- item href:%s path part: %s"%(href,get_path_part(href)))
|
||||
if re.match(r'.*/(file|chapter)\d+\.x?html',href):
|
||||
if getsoups:
|
||||
soup = bs.BeautifulSoup(epub.read(href).decode("utf-8"))
|
||||
for img in soup.findAll('img'):
|
||||
try:
|
||||
newsrc=get_path_part(href)+img['src']
|
||||
# remove all .. and the path part above it, if present.
|
||||
# Most for epubs edited by Sigil.
|
||||
newsrc = re.sub(r"([^/]+/\.\./)","",newsrc)
|
||||
origsrc=img['origsrc']
|
||||
data = epub.read(newsrc)
|
||||
images[origsrc] = data
|
||||
img['src'] = img['origsrc']
|
||||
except Exception as e:
|
||||
print("Image %s not found!\n(originally:%s)"%(newsrc,origsrc))
|
||||
print("Exception: %s"%(unicode(e)))
|
||||
traceback.print_exc()
|
||||
soup = soup.find('body')
|
||||
soup.find('h3').extract()
|
||||
soups.append(soup)
|
||||
|
||||
filecount+=1
|
||||
|
||||
for k in images.keys():
|
||||
print("\torigsrc:%s\n\tData len:%s\n"%(k,len(images[k])))
|
||||
return (source,filecount,soups,images)
|
||||
|
||||
def get_path_part(n):
|
||||
relpath = os.path.dirname(n)
|
||||
if( len(relpath) > 0 ):
|
||||
relpath=relpath+"/"
|
||||
return relpath
|
||||
+73
-34
@@ -17,63 +17,73 @@
|
||||
|
||||
import os, re
|
||||
import urlparse
|
||||
from math import floor
|
||||
|
||||
from htmlcleanup import conditionalRemoveEntities, removeAllEntities
|
||||
|
||||
# Create convert_image method depending on which graphics lib we can
|
||||
# load. Preferred: calibre, PIL, none
|
||||
try:
|
||||
from calibre.utils.magick.draw import minify_image
|
||||
from calibre.utils.magick import Image
|
||||
|
||||
def convert_image(url,data,sizes,grayscale):
|
||||
img = minify_image(data, minify_to=sizes)
|
||||
if grayscale:
|
||||
export = False
|
||||
img = Image()
|
||||
img.load(data)
|
||||
|
||||
owidth, oheight = img.size
|
||||
nwidth, nheight = sizes
|
||||
scaled, nwidth, nheight = fit_image(owidth, oheight, nwidth, nheight)
|
||||
if scaled:
|
||||
img.size = (nwidth, nheight)
|
||||
export = True
|
||||
|
||||
if grayscale and img.type != "GrayscaleType":
|
||||
img.type = "GrayscaleType"
|
||||
return (img.export('JPG'),'jpg','image/jpeg')
|
||||
export = True
|
||||
|
||||
if normalize_format_name(img.format) != "jpg":
|
||||
export = True
|
||||
|
||||
if export:
|
||||
return (img.export('JPG'),'jpg','image/jpeg')
|
||||
else:
|
||||
print("image used unchanged")
|
||||
return (data,'jpg','image/jpeg')
|
||||
|
||||
except:
|
||||
|
||||
# No calibre routines, try for PIL for CLI.
|
||||
try:
|
||||
import Image
|
||||
from StringIO import StringIO
|
||||
from math import floor
|
||||
def convert_image(url,data,sizes,grayscale):
|
||||
|
||||
export = False
|
||||
img = Image.open(StringIO(data))
|
||||
outsio = StringIO()
|
||||
|
||||
owidth, oheight = img.size
|
||||
nwidth, nheight = sizes
|
||||
scaled, nwidth, nheight = fit_image(owidth, oheight, nwidth, nheight)
|
||||
if scaled:
|
||||
img = img.resize((nwidth, nheight),Image.ANTIALIAS)
|
||||
export = True
|
||||
|
||||
if grayscale:
|
||||
if grayscale and img.mode != "L":
|
||||
img = img.convert("L")
|
||||
|
||||
img.save(outsio,'JPEG')
|
||||
return (outsio.getvalue(),'jpg','image/jpeg')
|
||||
export = True
|
||||
|
||||
if normalize_format_name(img.format) != "jpg":
|
||||
export = True
|
||||
|
||||
if export:
|
||||
outsio = StringIO()
|
||||
img.save(outsio,'JPEG')
|
||||
return (outsio.getvalue(),'jpg','image/jpeg')
|
||||
else:
|
||||
print("image used unchanged")
|
||||
return (data,'jpg','image/jpeg')
|
||||
|
||||
def fit_image(width, height, pwidth, pheight):
|
||||
'''
|
||||
Fit image in box of width pwidth and height pheight.
|
||||
@param width: Width of image
|
||||
@param height: Height of image
|
||||
@param pwidth: Width of box
|
||||
@param pheight: Height of box
|
||||
@return: scaled, new_width, new_height. scaled is True iff new_width and/or new_height is different from width or height.
|
||||
'''
|
||||
scaled = height > pheight or width > pwidth
|
||||
if height > pheight:
|
||||
corrf = pheight/float(height)
|
||||
width, height = floor(corrf*width), pheight
|
||||
if width > pwidth:
|
||||
corrf = pwidth/float(width)
|
||||
width, height = pwidth, floor(corrf*height)
|
||||
if height > pheight:
|
||||
corrf = pheight/float(height)
|
||||
width, height = floor(corrf*width), pheight
|
||||
|
||||
return scaled, int(width), int(height)
|
||||
except:
|
||||
|
||||
# No calibre or PIL, simple pass through with mimetype.
|
||||
@@ -88,6 +98,35 @@ except:
|
||||
def convert_image(url,data,sizes,grayscale):
|
||||
ext=url[url.rfind('.')+1:].lower()
|
||||
return (data,ext,imagetypes[ext])
|
||||
|
||||
def normalize_format_name(fmt):
|
||||
if fmt:
|
||||
fmt = fmt.lower()
|
||||
if fmt == 'jpeg':
|
||||
fmt = 'jpg'
|
||||
return fmt
|
||||
|
||||
def fit_image(width, height, pwidth, pheight):
|
||||
'''
|
||||
Fit image in box of width pwidth and height pheight.
|
||||
@param width: Width of image
|
||||
@param height: Height of image
|
||||
@param pwidth: Width of box
|
||||
@param pheight: Height of box
|
||||
@return: scaled, new_width, new_height. scaled is True iff new_width and/or new_height is different from width or height.
|
||||
'''
|
||||
scaled = height > pheight or width > pwidth
|
||||
if height > pheight:
|
||||
corrf = pheight/float(height)
|
||||
width, height = floor(corrf*width), pheight
|
||||
if width > pwidth:
|
||||
corrf = pwidth/float(width)
|
||||
width, height = pwidth, floor(corrf*height)
|
||||
if height > pheight:
|
||||
corrf = pheight/float(height)
|
||||
width, height = floor(corrf*width), pheight
|
||||
|
||||
return scaled, int(width), int(height)
|
||||
|
||||
try:
|
||||
# doesn't really matter what, just checking for appengine.
|
||||
@@ -245,9 +284,9 @@ class Story:
|
||||
if is_appengine:
|
||||
return
|
||||
|
||||
if url.startswith("http") or url.startswith("file") :
|
||||
if url.startswith("http") or url.startswith("file") or parenturl == None:
|
||||
imgurl = url
|
||||
elif parenturl != None:
|
||||
else:
|
||||
parsedUrl = urlparse.urlparse(parenturl)
|
||||
if url.startswith("/") :
|
||||
imgurl = urlparse.urlunparse(
|
||||
@@ -271,7 +310,7 @@ class Story:
|
||||
# bit of corner case inefficiency I can live with rather than
|
||||
# scanning all the pre-existing files on update. oldsrc is
|
||||
# being saved on img tags just in case, however.
|
||||
prefix=self.getMetadataRaw('dateCreated').strftime("%Y%m%d%H%M%S")
|
||||
prefix='ffdl' #self.getMetadataRaw('dateCreated').strftime("%Y%m%d%H%M%S")
|
||||
|
||||
if imgurl not in self.imgurls:
|
||||
parsedUrl = urlparse.urlparse(imgurl)
|
||||
|
||||
@@ -294,7 +294,7 @@ ${value}<br />
|
||||
guide = newTag(contentdom,"guide")
|
||||
guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover",
|
||||
"title":"Cover",
|
||||
"href":"cover.xhtml"}))
|
||||
"href":"OEBPS/cover.xhtml"}))
|
||||
|
||||
coverIO = StringIO.StringIO()
|
||||
coverIO.write('''
|
||||
|
||||
Reference in new issue
Block a user