Rewrite of epub update, now handles images.

This commit is contained in:
Jim Miller committed 2012-02-26 20:32:19 -06:00
1 parent f0445f106c
commit a9cecbbe4f
11 files changed
+267 -454

No files matched your search

+1 -1
View File
@@ -27,7 +27,7 @@ class FanFictionDownLoaderBase(InterfaceActionBase):
description = 'UI plugin to download FanFiction stories from various sites.'
supported_platforms = ['windows', 'osx', 'linux']
author = 'Jim Miller'
version = (1, 4, 6)
version = (1, 5, 0)
minimum_calibre_version = (0, 8, 30)
#: This field defines the GUI plugin class that contains all the code
-30
View File
@@ -1,30 +0,0 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2012, Jim Miller'
__docformat__ = 'restructuredtext en'
from zipfile import ZipFile
from xml.dom.minidom import parseString
def get_dcsource(inputio):
epub = ZipFile(inputio, 'r')
## Find the .opf file.
container = epub.read("META-INF/container.xml")
containerdom = parseString(container)
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
rootfilename = rootfilenodelist[0].getAttribute("full-path")
metadom = parseString(epub.read(rootfilename))
firstmetadom = metadom.getElementsByTagName("metadata")[0]
try:
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
except:
source=None
return source
+11 -9
View File
@@ -35,8 +35,8 @@ from calibre_plugins.fanfictiondownloader_plugin.common_utils import (set_plugin
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.htmlcleanup import stripHTML
from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
from calibre_plugins.fanfictiondownloader_plugin.dcsource import get_dcsource
#from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.epubutils import get_dcsource, get_dcsource_chaptercount
from calibre_plugins.fanfictiondownloader_plugin.config import (prefs, permitted_values)
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (
@@ -524,13 +524,15 @@ class FanFictionDownLoaderPlugin(InterfaceAction):
# 'book' can exist without epub. If there's no existing epub,
# let it go and it will download it.
if db.has_format(book_id,fileform,index_is_id=True):
toupdateio = StringIO()
(epuburl,chaptercount) = doMerge(toupdateio,
[StringIO(db.format(book_id,'EPUB',
index_is_id=True))],
titlenavpoints=False,
striptitletoc=True,
forceunique=False)
#toupdateio = StringIO()
(epuburl,chaptercount) = get_dcsource_chaptercount(StringIO(db.format(book_id,'EPUB',
index_is_id=True)))
# (epuburl,chaptercount) = doMerge(toupdateio,
# [StringIO(db.format(book_id,'EPUB',
# index_is_id=True))],
# titlenavpoints=False,
# striptitletoc=True,
# forceunique=False)
urlchaptercount = int(story.getMetadata('numChapters'))
if chaptercount == urlchaptercount:
if collision == UPDATE:
+34 -27
View File
@@ -23,7 +23,8 @@ from calibre.utils.logging import Log
from calibre_plugins.fanfictiondownloader_plugin.dialogs import (NotGoingToDownload,
OVERWRITE, OVERWRITEALWAYS, UPDATE, UPDATEALWAYS, ADDNEW, SKIP, CALIBREONLY)
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader import adapters, writers, exceptions
from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
#from calibre_plugins.fanfictiondownloader_plugin.epubmerge import doMerge
from calibre_plugins.fanfictiondownloader_plugin.fanficdownloader.epubutils import get_update_data
# ------------------------------------------------------------------------------
#
@@ -136,38 +137,44 @@ def do_download_for_worker(book,options):
elif 'epub_for_update' in book and options['collision'] in (UPDATE, UPDATEALWAYS):
urlchaptercount = int(story.getMetadata('numChapters'))
## First, get existing epub with titlepage and tocpage stripped.
updateio = StringIO()
(epuburl,chaptercount) = doMerge(updateio,
[book['epub_for_update']],
titlenavpoints=False,
striptitletoc=True,
forceunique=False)
(url,chaptercount,
adapter.oldchapters,
adapter.oldimgs) = get_update_data(book['epub_for_update'])
print("Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount))
print("write to %s"%outfile)
## Get updated title page/metadata by itself in an epub.
## Even if the title page isn't included, this carries the metadata.
titleio = StringIO()
writer.writeStory(outstream=titleio,metaonly=True)
writer.writeStory(outfilename=outfile, forceOverwrite=True)
## First, get existing epub with titlepage and tocpage stripped.
# updateio = StringIO()
# (epuburl,chaptercount) = doMerge(updateio,
# [book['epub_for_update']],
# titlenavpoints=False,
# striptitletoc=True,
# forceunique=False)
# ## Get updated title page/metadata by itself in an epub.
# ## Even if the title page isn't included, this carries the metadata.
# titleio = StringIO()
# writer.writeStory(outstream=titleio,metaonly=True)
newchaptersio = None
if urlchaptercount > chaptercount :
## Go get the new chapters
newchaptersio = StringIO()
adapter.setChaptersRange(chaptercount+1,urlchaptercount)
# newchaptersio = None
# if urlchaptercount > chaptercount :
# ## Go get the new chapters
# newchaptersio = StringIO()
# adapter.setChaptersRange(chaptercount+1,urlchaptercount)
adapter.config.set("overrides",'include_tocpage','false')
adapter.config.set("overrides",'include_titlepage','false')
writer.writeStory(outstream=newchaptersio)
# adapter.config.set("overrides",'include_tocpage','false')
# adapter.config.set("overrides",'include_titlepage','false')
# writer.writeStory(outstream=newchaptersio)
## Merge the three epubs together.
doMerge(outfile,
[titleio,updateio,newchaptersio],
fromfirst=True,
titlenavpoints=False,
striptitletoc=False,
forceunique=False)
# ## Merge the three epubs together.
# doMerge(outfile,
# [titleio,updateio,newchaptersio],
# fromfirst=True,
# titlenavpoints=False,
# striptitletoc=False,
# forceunique=False)
book['comment'] = 'Update %s completed, added %s chapters for %s total.'%\
(options['fileform'],(urlchaptercount-chaptercount),urlchaptercount)
+30 -26
View File
@@ -25,19 +25,16 @@ from StringIO import StringIO
from optparse import OptionParser
import getpass
import string
import ConfigParser
from subprocess import call
from epubmerge import doMerge
from fanficdownloader import adapters,writers,exceptions
from fanficdownloader.epubutils import get_dcsource_chaptercount, get_update_data
if sys.version_info < (2, 5):
print "This program requires Python 2.5 or newer."
sys.exit(1)
from fanficdownloader import adapters,writers,exceptions
import ConfigParser
def writeStory(config,adapter,writeformat,metaonly=False,outstream=None):
writer = writers.getWriter(writeformat,config,adapter)
writer.writeStory(outstream=outstream,metaonly=metaonly)
@@ -116,12 +113,13 @@ def main():
try:
## Attempt to update an existing epub.
if options.update:
updateio = StringIO()
(url,chaptercount) = doMerge(updateio,
args,
titlenavpoints=False,
striptitletoc=True,
forceunique=False)
# updateio = StringIO()
# (url,chaptercount) = doMerge(updateio,
# args,
# titlenavpoints=False,
# striptitletoc=True,
# forceunique=False)
(url,chaptercount) = get_dcsource_chaptercount(args[0])
print "Updating %s, URL: %s" % (args[0],url)
output_filename = args[0]
config.set("overrides","output_filename",args[0])
@@ -167,17 +165,23 @@ def main():
print "Do update - epub(%d) vs url(%d)" % (chaptercount, urlchaptercount)
## Get updated title page/metadata by itself in an epub.
## Even if the title page isn't included, this carries the metadata.
titleio = StringIO()
writeStory(config,adapter,"epub",metaonly=True,outstream=titleio)
# titleio = StringIO()
# writeStory(config,adapter,"epub",metaonly=True,outstream=titleio)
newchaptersio = None
# newchaptersio = None
if not options.metaonly:
(url,chaptercount,
adapter.oldchapters,
adapter.oldimgs) = get_update_data(args[0])
writeStory(config,adapter,"epub")
## Go get the new chapters only in another epub.
newchaptersio = StringIO()
adapter.setChaptersRange(chaptercount+1,urlchaptercount)
config.set("overrides",'include_tocpage','false')
config.set("overrides",'include_titlepage','false')
writeStory(config,adapter,"epub",outstream=newchaptersio)
# newchaptersio = StringIO()
# adapter.setChaptersRange(chaptercount+1,urlchaptercount)
# config.set("overrides",'include_tocpage','false')
# config.set("overrides",'include_titlepage','false')
# writeStory(config,adapter,"epub",outstream=newchaptersio)
# out = open("testing/titleio.epub","wb")
# out.write(titleio.getvalue())
@@ -192,12 +196,12 @@ def main():
# out.close()
## Merge the three epubs together.
doMerge(args[0],
[titleio,updateio,newchaptersio],
fromfirst=True,
titlenavpoints=False,
striptitletoc=False,
forceunique=False)
# doMerge(args[0],
# [titleio,updateio,newchaptersio],
# fromfirst=True,
# titlenavpoints=False,
# striptitletoc=False,
# forceunique=False)
else:
# regular download
+2 -318
View File
@@ -16,326 +16,10 @@
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import sys
import os
import re
#import StringIO
from optparse import OptionParser
import zlib
import zipfile
from zipfile import ZipFile, ZIP_STORED, ZIP_DEFLATED
from time import time
from exceptions import KeyError
from xml.dom.minidom import parse, parseString, getDOMImplementation
def doMerge(outputio,files,authoropts=[],titleopt=None,descopt=None,
fromfirst=False,
titlenavpoints=True,
striptitletoc=False,
forceunique=True):
'''
outputio = output file name or StringIO.
files = list of input file names or StringIOs.
authoropts = list of authors to use, otherwise add from all input
titleopt = title, otherwise '<first title> Anthology'
descopt = description, otherwise '<title> by <author>' list for all input
fromfirst if true, take all metadata (including author, title, desc) from first input
titlenavpoints if true, put in a new TOC entry for each epub
striptitletoc if true, strip out any (title|toc)_page.xhtml files
forceunique if true, guarantee uniqueness of contents by adding a dir for each input
'''
## Python 2.5 ZipFile is rather more primative than later
## versions. It can operate on a file, or on a StringIO, but
## not on an open stream. OTOH, I suspect we would have had
## problems with closing and opening again to change the
## compression type anyway.
filecount=0
source=None
## Write mimetype file, must be first and uncompressed.
## Older versions of python(2.4/5) don't allow you to specify
## compression by individual file.
## Overwrite if existing output file.
outputepub = ZipFile(outputio, "w", compression=ZIP_STORED)
outputepub.debug = 3
outputepub.writestr("mimetype", "application/epub+zip")
outputepub.close()
## Re-open file for content.
outputepub = ZipFile(outputio, "a", compression=ZIP_DEFLATED)
outputepub.debug = 3
## Create META-INF/container.xml file. The only thing it does is
## point to content.opf
containerdom = getDOMImplementation().createDocument(None, "container", None)
containertop = containerdom.documentElement
containertop.setAttribute("version","1.0")
containertop.setAttribute("xmlns","urn:oasis:names:tc:opendocument:xmlns:container")
rootfiles = containerdom.createElement("rootfiles")
containertop.appendChild(rootfiles)
rootfiles.appendChild(newTag(containerdom,"rootfile",{"full-path":"content.opf",
"media-type":"application/oebps-package+xml"}))
outputepub.writestr("META-INF/container.xml",containerdom.toprettyxml(indent=' ',encoding='utf-8'))
## Process input epubs.
items = [] # list of (id, href, type) tuples(all strings) -- From .opfs' manifests
items.append(("ncx","toc.ncx","application/x-dtbncx+xml")) ## we'll generate the toc.ncx file,
## but it needs to be in the items manifest.
itemrefs = [] # list of strings -- idrefs from .opfs' spines
navmaps = [] # list of navMap DOM elements -- TOC data for each from toc.ncx files
booktitles = [] # list of strings -- Each book's title
allauthors = [] # list of lists of strings -- Each book's list of authors.
filelist = []
booknum=1
firstmetadom = None
for file in files:
if file == None : continue
book = "%d" % booknum
bookdir = ""
bookid = ""
if forceunique:
bookdir = "%d/" % booknum
bookid = "a%d" % booknum
#print "book %d" % booknum
epub = ZipFile(file, 'r')
## Find the .opf file.
container = epub.read("META-INF/container.xml")
containerdom = parseString(container)
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
rootfilename = rootfilenodelist[0].getAttribute("full-path")
## Save the path to the .opf file--hrefs inside it are relative to it.
relpath = os.path.dirname(rootfilename)
if( len(relpath) > 0 ):
relpath=relpath+"/"
metadom = parseString(epub.read(rootfilename))
if booknum==1:
firstmetadom = metadom.getElementsByTagName("metadata")[0]
try:
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
except:
source=""
#print "Source:%s"%source
## Save indiv book title
booktitles.append(metadom.getElementsByTagName("dc:title")[0].firstChild.data)
## Save authors.
authors=[]
for creator in metadom.getElementsByTagName("dc:creator"):
if( creator.getAttribute("opf:role") == "aut" ):
authors.append(creator.firstChild.data)
allauthors.append(authors)
for item in metadom.getElementsByTagName("item"):
if( item.getAttribute("media-type") == "application/x-dtbncx+xml" ):
# TOC file is only one with this type--as far as I know.
# grab the whole navmap, deal with it later.
tocdom = parseString(epub.read(relpath+item.getAttribute("href")))
for navpoint in tocdom.getElementsByTagName("navPoint"):
navpoint.setAttribute("id",bookid+navpoint.getAttribute("id"))
for content in tocdom.getElementsByTagName("content"):
content.setAttribute("src",bookdir+relpath+content.getAttribute("src"))
navmaps.append(tocdom.getElementsByTagName("navMap")[0])
else:
id=bookid+item.getAttribute("id")
href=bookdir+relpath+item.getAttribute("href")
href=href.encode('utf8')
#print "href:"+href
if not striptitletoc or not re.match(r'.*/((title|toc)_page|cover)\.xhtml',
item.getAttribute("href")):
if href not in filelist:
try:
outputepub.writestr(href,
epub.read(relpath+item.getAttribute("href")))
if re.match(r'.*/(file|chapter)\d+\.x?html',href):
filecount+=1
items.append((id,href,item.getAttribute("media-type")))
filelist.append(href)
except KeyError, ke:
pass # Skip missing files.
for itemref in metadom.getElementsByTagName("itemref"):
if not striptitletoc or not re.match(r'((title|toc)_page|cover)', itemref.getAttribute("idref")):
itemrefs.append(bookid+itemref.getAttribute("idref"))
booknum=booknum+1;
if not forceunique:
# If not forceunique, it's an epub update.
# If there's a "calibre_bookmarks.txt", it's from reading
# in Calibre and should be preserved.
try:
fn = "META-INF/calibre_bookmarks.txt"
outputepub.writestr(fn,epub.read(fn))
except:
pass
## create content.opf file.
uniqueid="epubmerge-uid-%d" % time() # real sophisticated uid scheme.
contentdom = getDOMImplementation().createDocument(None, "package", None)
package = contentdom.documentElement
if fromfirst and firstmetadom:
metadata = firstmetadom
firstpackage = firstmetadom.parentNode
package.setAttribute("version",firstpackage.getAttribute("version"))
package.setAttribute("xmlns",firstpackage.getAttribute("xmlns"))
package.setAttribute("unique-identifier",firstpackage.getAttribute("unique-identifier"))
else:
package.setAttribute("version","2.0")
package.setAttribute("xmlns","http://www.idpf.org/2007/opf")
package.setAttribute("unique-identifier","epubmerge-id")
metadata=newTag(contentdom,"metadata",
attrs={"xmlns:dc":"http://purl.org/dc/elements/1.1/",
"xmlns:opf":"http://www.idpf.org/2007/opf"})
metadata.appendChild(newTag(contentdom,"dc:identifier",text=uniqueid,attrs={"id":"epubmerge-id"}))
if( titleopt is None ):
titleopt = booktitles[0]+" Anthology"
metadata.appendChild(newTag(contentdom,"dc:title",text=titleopt))
# If cmdline authors, use those instead of those collected from the epubs
# (allauthors kept for TOC & description gen below.
if( len(authoropts) > 1 ):
useauthors=[authoropts]
else:
useauthors=allauthors
usedauthors=dict()
for authorlist in useauthors:
for author in authorlist:
if( not usedauthors.has_key(author) ):
usedauthors[author]=author
metadata.appendChild(newTag(contentdom,"dc:creator",
attrs={"opf:role":"aut"},
text=author))
metadata.appendChild(newTag(contentdom,"dc:contributor",text="epubmerge",attrs={"opf:role":"bkp"}))
metadata.appendChild(newTag(contentdom,"dc:rights",text="Copyrights as per source stories"))
metadata.appendChild(newTag(contentdom,"dc:language",text="en"))
if not descopt:
# created now, but not filled in until TOC generation to save loops.
description = newTag(contentdom,"dc:description",text="Anthology containing:\n")
else:
description = newTag(contentdom,"dc:description",text=descopt)
metadata.appendChild(description)
package.appendChild(metadata)
manifest = contentdom.createElement("manifest")
package.appendChild(manifest)
for item in items:
(id,href,type)=item
manifest.appendChild(newTag(contentdom,"item",
attrs={'id':id,
'href':href,
'media-type':type}))
spine = newTag(contentdom,"spine",attrs={"toc":"ncx"})
package.appendChild(spine)
for itemref in itemrefs:
spine.appendChild(newTag(contentdom,"itemref",
attrs={"idref":itemref,
"linear":"yes"}))
## create toc.ncx file
tocncxdom = getDOMImplementation().createDocument(None, "ncx", None)
ncx = tocncxdom.documentElement
ncx.setAttribute("version","2005-1")
ncx.setAttribute("xmlns","http://www.daisy.org/z3986/2005/ncx/")
head = tocncxdom.createElement("head")
ncx.appendChild(head)
head.appendChild(newTag(tocncxdom,"meta",
attrs={"name":"dtb:uid", "content":uniqueid}))
head.appendChild(newTag(tocncxdom,"meta",
attrs={"name":"dtb:depth", "content":"1"}))
head.appendChild(newTag(tocncxdom,"meta",
attrs={"name":"dtb:totalPageCount", "content":"0"}))
head.appendChild(newTag(tocncxdom,"meta",
attrs={"name":"dtb:maxPageNumber", "content":"0"}))
docTitle = tocncxdom.createElement("docTitle")
docTitle.appendChild(newTag(tocncxdom,"text",text=titleopt))
ncx.appendChild(docTitle)
tocnavMap = tocncxdom.createElement("navMap")
ncx.appendChild(tocnavMap)
## TOC navPoints can be nested, but this flattens them for
## simplicity, plus adds a navPoint for each epub.
booknum=0
for navmap in navmaps:
navpoints = navmap.getElementsByTagName("navPoint")
if titlenavpoints:
## Copy first navPoint of each epub, give a different id and
## text: bookname by authorname
newnav = navpoints[0].cloneNode(True)
newnav.setAttribute("id","book"+newnav.getAttribute("id"))
## For purposes of TOC titling & desc, use first book author
newtext = newTag(tocncxdom,"text",text=booktitles[booknum]+" by "+allauthors[booknum][0])
text = newnav.getElementsByTagName("text")[0]
text.parentNode.replaceChild(newtext,text)
tocnavMap.appendChild(newnav)
if not descopt and not fromfirst:
description.appendChild(contentdom.createTextNode(booktitles[booknum]+" by "+allauthors[booknum][0]+"\n"))
for navpoint in navpoints:
#print "navpoint:%s"%navpoint.getAttribute("id")
if not striptitletoc or not re.match(r'(title|toc)_page',navpoint.getAttribute("id")):
tocnavMap.appendChild(navpoint)
booknum=booknum+1;
## Force strict ordering of playOrder
playorder=1
for navpoint in tocncxdom.getElementsByTagName("navPoint"):
navpoint.setAttribute("playOrder","%d" % playorder)
if( not navpoint.getAttribute("id").startswith("book") ):
playorder = playorder + 1
## content.opf written now due to description being filled in
## during TOC generation to save loops.
outputepub.writestr("content.opf",contentdom.toxml('utf-8'))
outputepub.writestr("toc.ncx",tocncxdom.toxml('utf-8'))
# declares all the files created by Windows. otherwise, when
# it runs in appengine, windows unzips the files as 000 perms.
for zf in outputepub.filelist:
zf.create_system = 0
outputepub.close()
return (source,filecount)
## Utility method for creating new tags.
def newTag(dom,name,attrs=None,text=None):
tag = dom.createElement(name)
if( attrs is not None ):
for attr in attrs.keys():
tag.setAttribute(attr,attrs[attr])
if( text is not None ):
tag.appendChild(dom.createTextNode(text))
return tag
if __name__ == "__main__":
print('''
This version is only used by fanfictiondownloader now. See:
http://code.google.com/p/epubmerge/
The this utility has been split out into it's own project.
See: http://code.google.com/p/epubmerge/
...for a CLI epubmerge.py program and calibre plugin.
''')
+4 -4
View File
@@ -126,9 +126,9 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
('Chapter 4',self.url+"&chapter=5"),
('Chapter 5',self.url+"&chapter=6"),
('Chapter 6',self.url+"&chapter=6"),
('Chapter 7',self.url+"&chapter=6"),
('Chapter 8',self.url+"&chapter=6"),
('Chapter 9',self.url+"&chapter=6"),
# ('Chapter 7',self.url+"&chapter=6"),
# ('Chapter 8',self.url+"&chapter=6"),
# ('Chapter 9',self.url+"&chapter=6"),
# ('Chapter 0',self.url+"&chapter=6"),
# ('Chapter a',self.url+"&chapter=6"),
# ('Chapter b',self.url+"&chapter=6"),
@@ -177,7 +177,7 @@ Some more longer description. "I suck at summaries!" "Better than it sounds!"
else:
text=u'''
<div>
<h3>Chapter</h3>
<h3>Chapter title from site</h3>
<p><center>Centered text</center></p>
<p>Lorem '''+self.crazystring+''' <i>italics</i>, <b>bold</b>, <u>underline</u> consectetur adipisicing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat. Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur. Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit anim id est laborum.</p>
br breaks<br><br>
+25 -4
View File
@@ -22,6 +22,7 @@ import logging
import urllib
import urllib2 as u2
import urlparse as up
from functools import partial
from .. import BeautifulSoup as bs
from ..htmlcleanup import stripHTML
@@ -86,6 +87,8 @@ class BaseSiteAdapter(Configurable):
self.chapterUrls = [] # tuples of (chapter title,chapter url)
self.chapterFirst = None
self.chapterLast = None
self.oldchapters = None
self.oldimgs = None
## order of preference for decoding.
self.decode = ["utf8",
"Windows-1252"] # 1252 is a superset of
@@ -189,14 +192,21 @@ class BaseSiteAdapter(Configurable):
def getStory(self):
if not self.storyDone:
self.getStoryMetadataOnly()
for index, (title,url) in enumerate(self.chapterUrls):
if (self.chapterFirst!=None and index < self.chapterFirst) or \
(self.chapterLast!=None and index > self.chapterLast):
self.story.addChapter(removeEntities(title),
None)
else:
if self.oldchapters and index < len(self.oldchapters):
data = self.utf8FromSoup(None,
self.oldchapters[index],
partial(cachedfetch,self._fetchUrlRaw,self.oldimgs))
else:
data = self.getChapterText(url)
self.story.addChapter(removeEntities(title),
removeEntities(self.getChapterText(url)))
removeEntities(data))
self.storyDone = True
# include image, but no cover from story, add default_cover_image cover.
@@ -264,7 +274,9 @@ class BaseSiteAdapter(Configurable):
# this gives us a unicode object, not just a string containing bytes.
# (I gave soup a unicode string, you'd think it could give it back...)
def utf8FromSoup(self,url,soup):
def utf8FromSoup(self,url,soup,fetch=None):
if not fetch:
fetch=self._fetchUrlRaw
acceptable_attributes = ['href','name']
#print("include_images:"+self.getConfig('include_images'))
@@ -272,7 +284,7 @@ class BaseSiteAdapter(Configurable):
acceptable_attributes.extend(('src','alt','origsrc'))
for img in soup.findAll('img'):
img['origsrc']=img['src']
img['src']=self.story.addImgUrl(self,url,img['src'],self._fetchUrlRaw)
img['src']=self.story.addImgUrl(self,url,img['src'],fetch)
for attr in soup._getAttrMap().keys():
if attr not in acceptable_attributes:
@@ -294,12 +306,21 @@ class BaseSiteAdapter(Configurable):
# removes paired, but empty tags.
if t.string != None and len(t.string.strip()) == 0 :
t.extract()
return soup.__str__('utf8').decode('utf-8')
# Don't want body tags in chapter html--writers add them.
return re.sub(r"</?body>\r?\n?","",soup.__str__('utf8').decode('utf-8'))
fullmon = {"January":"01", "February":"02", "March":"03", "April":"04", "May":"05",
"June":"06","July":"07", "August":"08", "September":"09", "October":"10",
"November":"11", "December":"12" }
def cachedfetch(realfetch,cache,url):
if url in cache:
print("cache hit")
return cache[url]
else:
return realfetch(url)
def makeDate(string,format):
# Surprise! Abstracting this turned out to be more useful than
# just saving bytes.
+86
View File
@@ -0,0 +1,86 @@
#!/usr/bin/env python
# vim:fileencoding=UTF-8:ts=4:sw=4:sta:et:sts=4:ai
from __future__ import (unicode_literals, division, absolute_import,
print_function)
__license__ = 'GPL v3'
__copyright__ = '2012, Jim Miller'
__docformat__ = 'restructuredtext en'
import re, os, traceback
from zipfile import ZipFile
from xml.dom.minidom import parseString
from . import BeautifulSoup as bs
def get_dcsource(inputio):
return get_update_data(inputio,getfilecount=False,getsoups=False)[0]
def get_dcsource_chaptercount(inputio):
return get_update_data(inputio,getfilecount=True,getsoups=False)[:2] # (source,filecount)
def get_update_data(inputio,
getfilecount=True,
getsoups=True):
epub = ZipFile(inputio, 'r')
## Find the .opf file.
container = epub.read("META-INF/container.xml")
containerdom = parseString(container)
rootfilenodelist = containerdom.getElementsByTagName("rootfile")
rootfilename = rootfilenodelist[0].getAttribute("full-path")
contentdom = parseString(epub.read(rootfilename))
firstmetadom = contentdom.getElementsByTagName("metadata")[0]
try:
source=firstmetadom.getElementsByTagName("dc:source")[0].firstChild.data.encode("utf-8")
except:
source=None
## Save the path to the .opf file--hrefs inside it are relative to it.
relpath = get_path_part(rootfilename)
filecount = 0
soups = [] # list of xhmtl blocks
images = {} # dict() origsrc->data
if getfilecount:
# spin through the manifest--only place there are item tags.
for item in contentdom.getElementsByTagName("item"):
# First, count the 'chapter' files. FFDL uses file0000.xhtml,
# but can also update epubs downloaded from Twisting the
# Hellmouth, which uses chapter0.html.
if( item.getAttribute("media-type") == "application/xhtml+xml" ):
href=relpath+item.getAttribute("href")
print("---- item href:%s path part: %s"%(href,get_path_part(href)))
if re.match(r'.*/(file|chapter)\d+\.x?html',href):
if getsoups:
soup = bs.BeautifulSoup(epub.read(href).decode("utf-8"))
for img in soup.findAll('img'):
try:
newsrc=get_path_part(href)+img['src']
# remove all .. and the path part above it, if present.
# Most for epubs edited by Sigil.
newsrc = re.sub(r"([^/]+/\.\./)","",newsrc)
origsrc=img['origsrc']
data = epub.read(newsrc)
images[origsrc] = data
img['src'] = img['origsrc']
except Exception as e:
print("Image %s not found!\n(originally:%s)"%(newsrc,origsrc))
print("Exception: %s"%(unicode(e)))
traceback.print_exc()
soup = soup.find('body')
soup.find('h3').extract()
soups.append(soup)
filecount+=1
for k in images.keys():
print("\torigsrc:%s\n\tData len:%s\n"%(k,len(images[k])))
return (source,filecount,soups,images)
def get_path_part(n):
relpath = os.path.dirname(n)
if( len(relpath) > 0 ):
relpath=relpath+"/"
return relpath
+73 -34
View File
@@ -17,63 +17,73 @@
import os, re
import urlparse
from math import floor
from htmlcleanup import conditionalRemoveEntities, removeAllEntities
# Create convert_image method depending on which graphics lib we can
# load. Preferred: calibre, PIL, none
try:
from calibre.utils.magick.draw import minify_image
from calibre.utils.magick import Image
def convert_image(url,data,sizes,grayscale):
img = minify_image(data, minify_to=sizes)
if grayscale:
export = False
img = Image()
img.load(data)
owidth, oheight = img.size
nwidth, nheight = sizes
scaled, nwidth, nheight = fit_image(owidth, oheight, nwidth, nheight)
if scaled:
img.size = (nwidth, nheight)
export = True
if grayscale and img.type != "GrayscaleType":
img.type = "GrayscaleType"
return (img.export('JPG'),'jpg','image/jpeg')
export = True
if normalize_format_name(img.format) != "jpg":
export = True
if export:
return (img.export('JPG'),'jpg','image/jpeg')
else:
print("image used unchanged")
return (data,'jpg','image/jpeg')
except:
# No calibre routines, try for PIL for CLI.
try:
import Image
from StringIO import StringIO
from math import floor
def convert_image(url,data,sizes,grayscale):
export = False
img = Image.open(StringIO(data))
outsio = StringIO()
owidth, oheight = img.size
nwidth, nheight = sizes
scaled, nwidth, nheight = fit_image(owidth, oheight, nwidth, nheight)
if scaled:
img = img.resize((nwidth, nheight),Image.ANTIALIAS)
export = True
if grayscale:
if grayscale and img.mode != "L":
img = img.convert("L")
img.save(outsio,'JPEG')
return (outsio.getvalue(),'jpg','image/jpeg')
export = True
if normalize_format_name(img.format) != "jpg":
export = True
if export:
outsio = StringIO()
img.save(outsio,'JPEG')
return (outsio.getvalue(),'jpg','image/jpeg')
else:
print("image used unchanged")
return (data,'jpg','image/jpeg')
def fit_image(width, height, pwidth, pheight):
'''
Fit image in box of width pwidth and height pheight.
@param width: Width of image
@param height: Height of image
@param pwidth: Width of box
@param pheight: Height of box
@return: scaled, new_width, new_height. scaled is True iff new_width and/or new_height is different from width or height.
'''
scaled = height > pheight or width > pwidth
if height > pheight:
corrf = pheight/float(height)
width, height = floor(corrf*width), pheight
if width > pwidth:
corrf = pwidth/float(width)
width, height = pwidth, floor(corrf*height)
if height > pheight:
corrf = pheight/float(height)
width, height = floor(corrf*width), pheight
return scaled, int(width), int(height)
except:
# No calibre or PIL, simple pass through with mimetype.
@@ -88,6 +98,35 @@ except:
def convert_image(url,data,sizes,grayscale):
ext=url[url.rfind('.')+1:].lower()
return (data,ext,imagetypes[ext])
def normalize_format_name(fmt):
if fmt:
fmt = fmt.lower()
if fmt == 'jpeg':
fmt = 'jpg'
return fmt
def fit_image(width, height, pwidth, pheight):
'''
Fit image in box of width pwidth and height pheight.
@param width: Width of image
@param height: Height of image
@param pwidth: Width of box
@param pheight: Height of box
@return: scaled, new_width, new_height. scaled is True iff new_width and/or new_height is different from width or height.
'''
scaled = height > pheight or width > pwidth
if height > pheight:
corrf = pheight/float(height)
width, height = floor(corrf*width), pheight
if width > pwidth:
corrf = pwidth/float(width)
width, height = pwidth, floor(corrf*height)
if height > pheight:
corrf = pheight/float(height)
width, height = floor(corrf*width), pheight
return scaled, int(width), int(height)
try:
# doesn't really matter what, just checking for appengine.
@@ -245,9 +284,9 @@ class Story:
if is_appengine:
return
if url.startswith("http") or url.startswith("file") :
if url.startswith("http") or url.startswith("file") or parenturl == None:
imgurl = url
elif parenturl != None:
else:
parsedUrl = urlparse.urlparse(parenturl)
if url.startswith("/") :
imgurl = urlparse.urlunparse(
@@ -271,7 +310,7 @@ class Story:
# bit of corner case inefficiency I can live with rather than
# scanning all the pre-existing files on update. oldsrc is
# being saved on img tags just in case, however.
prefix=self.getMetadataRaw('dateCreated').strftime("%Y%m%d%H%M%S")
prefix='ffdl' #self.getMetadataRaw('dateCreated').strftime("%Y%m%d%H%M%S")
if imgurl not in self.imgurls:
parsedUrl = urlparse.urlparse(imgurl)
+1 -1
View File
@@ -294,7 +294,7 @@ ${value}<br />
guide = newTag(contentdom,"guide")
guide.appendChild(newTag(contentdom,"reference",attrs={"type":"cover",
"title":"Cover",
"href":"cover.xhtml"}))
"href":"OEBPS/cover.xhtml"}))
coverIO = StringIO.StringIO()
coverIO.write('''