Fix transform_abstracy + tests.

[librarian.git] / librarian / epub.py
diff --git a/librarian/epub.py b/librarian/epub.py

index b079d65..e9670d5 100644 (file)
--- a/librarian/epub.py
+++ b/librarian/epub.py
@@ -3,21 +3,23 @@
  # This file is part of Librarian, licensed under GNU Affero GPLv3 or later.
  # Copyright © Fundacja Nowoczesna Polska. See NOTICE for more information.
  #
  # This file is part of Librarian, licensed under GNU Affero GPLv3 or later.
  # Copyright © Fundacja Nowoczesna Polska. See NOTICE for more information.
  #
-from __future__ import with_statement
+from __future__ import print_function, unicode_literals
  
  import os
  import os.path
  import re
  import subprocess
  
  import os
  import os.path
  import re
  import subprocess
-from StringIO import StringIO
+from six import BytesIO
  from copy import deepcopy
  from copy import deepcopy
+from mimetypes import guess_type
+
  from lxml import etree
  import zipfile
  from tempfile import mkdtemp, NamedTemporaryFile
  from shutil import rmtree
  
  from librarian import RDFNS, WLNS, NCXNS, OPFNS, XHTMLNS, DCNS, OutputFile
  from lxml import etree
  import zipfile
  from tempfile import mkdtemp, NamedTemporaryFile
  from shutil import rmtree
  
  from librarian import RDFNS, WLNS, NCXNS, OPFNS, XHTMLNS, DCNS, OutputFile
-from librarian.cover import DefaultEbookCover
+from librarian.cover import make_cover
  
  from librarian import functions, get_resource
  
  
  from librarian import functions, get_resource
  
@@ -26,58 +28,69 @@ from librarian.hyphenator import Hyphenator
  functions.reg_person_name()
  functions.reg_lang_code_3to2()
  
  functions.reg_person_name()
  functions.reg_lang_code_3to2()
  
+
+def squeeze_whitespace(s):
+    return re.sub(b'\\s+', b' ', s)
+
+
  def set_hyph_language(source_tree):
      def get_short_lng_code(text):
          result = ''
          text = ''.join(text)
          with open(get_resource('res/ISO-639-2_8859-1.txt'), 'rb') as f:
  def set_hyph_language(source_tree):
      def get_short_lng_code(text):
          result = ''
          text = ''.join(text)
          with open(get_resource('res/ISO-639-2_8859-1.txt'), 'rb') as f:
-            for line in f:
+            for line in f.read().decode('latin1').split('\n'):
                  list = line.strip().split('|')
                  if list[0] == text:
                  list = line.strip().split('|')
                  if list[0] == text:
-                    result=list[2]
+                    result = list[2]
          if result == '':
              return text
          else:
              return result
          if result == '':
              return text
          else:
              return result
-    bibl_lng = etree.XPath('//dc:language//text()', namespaces = {'dc':str(DCNS)})(source_tree)
-    short_lng = get_short_lng_code(bibl_lng[0])   
+    bibl_lng = etree.XPath('//dc:language//text()',
+                           namespaces={'dc': str(DCNS)})(source_tree)
+    short_lng = get_short_lng_code(bibl_lng[0])
      try:
      try:
-        return Hyphenator(get_resource('res/hyph-dictionaries/hyph_' + short_lng + '.dic'))
+        return Hyphenator(get_resource('res/hyph-dictionaries/hyph_' +
+                                       short_lng + '.dic'))
      except:
          pass
      except:
          pass
-    
+
+
  def hyphenate_and_fix_conjunctions(source_tree, hyph):
  def hyphenate_and_fix_conjunctions(source_tree, hyph):
-    """ hyphenate only powiesc, opowiadanie and wywiad tag"""
-    if hyph is not None:
-        texts = etree.XPath('//*[self::powiesc|self::opowiadanie|self::wywiad]//text()')(source_tree)
-        for t in texts:
-            parent = t.getparent()
+    texts = etree.XPath('/utwor/*[2]//text()')(source_tree)
+    for t in texts:
+        parent = t.getparent()
+        if hyph is not None:
              newt = ''
              wlist = re.compile(r'\w+|[^\w]', re.UNICODE).findall(t)
              for w in wlist:
              newt = ''
              wlist = re.compile(r'\w+|[^\w]', re.UNICODE).findall(t)
              for w in wlist:
-                newt += hyph.inserted(w, u'\u00AD')       
-            newt = re.sub(r'(?<=\s\w)\s+', u'\u00A0', newt)
-            if t.is_text:
-                parent.text = newt
-            elif t.is_tail:
-                parent.tail = newt
-        
+                newt += hyph.inserted(w, u'\u00AD')
+        else:
+            newt = t
+        newt = re.sub(r'(?<=\s\w)\s+', u'\u00A0', newt)
+        if t.is_text:
+            parent.text = newt
+        elif t.is_tail:
+            parent.tail = newt
+
+
  def inner_xml(node):
      """ returns node's text and children as a string
  
  def inner_xml(node):
      """ returns node's text and children as a string
  
-    >>> print inner_xml(etree.fromstring('<a>x<b>y</b>z</a>'))
+    >>> print(inner_xml(etree.fromstring('<a>x<b>y</b>z</a>')))
      x<b>y</b>z
      """
  
      nt = node.text if node.text is not None else ''
      x<b>y</b>z
      """
  
      nt = node.text if node.text is not None else ''
-    return ''.join([nt] + [etree.tostring(child) for child in node])
+    return ''.join([nt] + [etree.tostring(child, encoding='unicode') for child in node])
+
  
  def set_inner_xml(node, text):
      """ sets node's text and children from a string
  
      >>> e = etree.fromstring('<a>b<b>x</b>x</a>')
      >>> set_inner_xml(e, 'x<b>y</b>z')
  
  def set_inner_xml(node, text):
      """ sets node's text and children from a string
  
      >>> e = etree.fromstring('<a>b<b>x</b>x</a>')
      >>> set_inner_xml(e, 'x<b>y</b>z')
-    >>> print etree.tostring(e)
+    >>> print(etree.tostring(e, encoding='unicode'))
      <a>x<b>y</b>z</a>
      """
  
      <a>x<b>y</b>z</a>
      """
  
@@ -89,7 +102,7 @@ def set_inner_xml(node, text):
  def node_name(node):
      """ Find out a node's name
  
  def node_name(node):
      """ Find out a node's name
  
-    >>> print node_name(etree.fromstring('<a>X<b>Y</b>Z</a>'))
+    >>> print(node_name(etree.fromstring('<a>X<b>Y</b>Z</a>')))
      XYZ
      """
  
      XYZ
      """
  
@@ -104,11 +117,13 @@ def node_name(node):
      return tempnode.text
  
  
      return tempnode.text
  
  
-def xslt(xml, sheet):
+def xslt(xml, sheet, **kwargs):
      if isinstance(xml, etree._Element):
          xml = etree.ElementTree(xml)
      with open(sheet) as xsltf:
      if isinstance(xml, etree._Element):
          xml = etree.ElementTree(xml)
      with open(sheet) as xsltf:
-        return xml.xslt(etree.parse(xsltf))
+        transform = etree.XSLT(etree.parse(xsltf))
+        params = dict((key, transform.strparam(value)) for key, value in kwargs.items())
+        return transform(xml, **params)
  
  
  def replace_characters(node):
  
  
  def replace_characters(node):
@@ -135,7 +150,7 @@ def find_annotations(annotations, source, part_no):
      for child in source:
          if child.tag in ('pe', 'pa', 'pt', 'pr'):
              annotation = deepcopy(child)
      for child in source:
          if child.tag in ('pe', 'pa', 'pt', 'pr'):
              annotation = deepcopy(child)
-            number = str(len(annotations)+1)
+            number = str(len(annotations) + 1)
              annotation.set('number', number)
              annotation.set('part', str(part_no))
              annotation.tail = ''
              annotation.set('number', number)
              annotation.set('part', str(part_no))
              annotation.tail = ''
@@ -157,10 +172,10 @@ class Stanza(object):
  
      >>> s = etree.fromstring("<strofa>a <b>c</b> <b>c</b>/\\nb<x>x/\\ny</x>c/ \\nd</strofa>")
      >>> Stanza(s).versify()
  
      >>> s = etree.fromstring("<strofa>a <b>c</b> <b>c</b>/\\nb<x>x/\\ny</x>c/ \\nd</strofa>")
      >>> Stanza(s).versify()
-    >>> print etree.tostring(s)
-    <strofa><wers_normalny>a <b>c</b> <b>c</b></wers_normalny><wers_normalny>b<x>x/
+    >>> print(etree.tostring(s, encoding='unicode'))
+    <strofa><wers_normalny>a <b>c</b><b>c</b></wers_normalny><wers_normalny>b<x>x/
      y</x>c</wers_normalny><wers_normalny>d</wers_normalny></strofa>
      y</x>c</wers_normalny><wers_normalny>d</wers_normalny></strofa>
-    
+
      """
      def __init__(self, stanza_elem):
          self.stanza = stanza_elem
      """
      def __init__(self, stanza_elem):
          self.stanza = stanza_elem
@@ -175,7 +190,7 @@ class Stanza(object):
          tail = self.stanza.tail
          self.stanza.clear()
          self.stanza.tail = tail
          tail = self.stanza.tail
          self.stanza.clear()
          self.stanza.tail = tail
-        self.stanza.extend(self.verses)
+        self.stanza.extend(verse for verse in self.verses if verse.text or len(verse) > 0)
  
      def open_normal_verse(self):
          self.open_verse = self.stanza.makeelement("wers_normalny")
  
      def open_normal_verse(self):
          self.open_verse = self.stanza.makeelement("wers_normalny")
@@ -192,6 +207,8 @@ class Stanza(object):
          for i, verse_text in enumerate(re.split(r"/\s*\n", text)):
              if i:
                  self.open_normal_verse()
          for i, verse_text in enumerate(re.split(r"/\s*\n", text)):
              if i:
                  self.open_normal_verse()
+            if not verse_text.strip():
+                continue
              verse = self.get_open_verse()
              if len(verse):
                  verse[-1].tail = (verse[-1].tail or "") + verse_text
              verse = self.get_open_verse()
              if len(verse):
                  verse[-1].tail = (verse[-1].tail or "") + verse_text
@@ -222,18 +239,17 @@ def add_to_manifest(manifest, partno):
      """ Adds a node to the manifest section in content.opf file """
  
      partstr = 'part%d' % partno
      """ Adds a node to the manifest section in content.opf file """
  
      partstr = 'part%d' % partno
-    e = manifest.makeelement(OPFNS('item'), attrib={
-                                 'id': partstr,
-                                 'href': partstr + '.html',
-                                 'media-type': 'application/xhtml+xml',
-                             })
+    e = manifest.makeelement(
+        OPFNS('item'), attrib={'id': partstr, 'href': partstr + '.html',
+                               'media-type': 'application/xhtml+xml'}
+    )
      manifest.append(e)
  
  
  def add_to_spine(spine, partno):
      """ Adds a node to the spine section in content.opf file """
  
      manifest.append(e)
  
  
  def add_to_spine(spine, partno):
      """ Adds a node to the spine section in content.opf file """
  
-    e = spine.makeelement(OPFNS('itemref'), attrib={'idref': 'part%d' % partno});
+    e = spine.makeelement(OPFNS('itemref'), attrib={'idref': 'part%d' % partno})
      spine.append(e)
  
  
      spine.append(e)
  
  
@@ -247,7 +263,7 @@ class TOC(object):
      def add(self, name, part_href, level=0, is_part=True, index=None):
          assert level == 0 or index is None
          if level > 0 and self.children:
      def add(self, name, part_href, level=0, is_part=True, index=None):
          assert level == 0 or index is None
          if level > 0 and self.children:
-            return self.children[-1].add(name, part_href, level-1, is_part)
+            return self.children[-1].add(name, part_href, level - 1, is_part)
          else:
              t = TOC(name)
              t.part_href = part_href
          else:
              t = TOC(name)
              t.part_href = part_href
@@ -285,7 +301,10 @@ class TOC(object):
  
              nav_label = nav_map.makeelement(NCXNS('navLabel'))
              text = nav_map.makeelement(NCXNS('text'))
  
              nav_label = nav_map.makeelement(NCXNS('navLabel'))
              text = nav_map.makeelement(NCXNS('text'))
-            text.text = re.sub(r'\n', ' ', child.name)
+            if child.name is not None:
+                text.text = re.sub(r'\n', ' ', child.name)
+            else:
+                text.text = child.name
              nav_label.append(text)
              nav_point.append(nav_label)
  
              nav_label.append(text)
              nav_point.append(nav_label)
  
@@ -302,12 +321,12 @@ class TOC(object):
              texts.append(
                  "<div style='margin-left:%dem;'><a href='%s'>%s</a></div>" %
                  (depth, child.href(), child.name))
              texts.append(
                  "<div style='margin-left:%dem;'><a href='%s'>%s</a></div>" %
                  (depth, child.href(), child.name))
-            texts.append(child.html_part(depth+1))
+            texts.append(child.html_part(depth + 1))
          return "\n".join(texts)
  
      def html(self):
          return "\n".join(texts)
  
      def html(self):
-        with open(get_resource('epub/toc.html')) as f:
-            t = unicode(f.read(), 'utf-8')
+        with open(get_resource('epub/toc.html'), 'rb') as f:
+            t = f.read().decode('utf-8')
          return t % self.html_part()
  
  
          return t % self.html_part()
  
  
@@ -325,10 +344,10 @@ def chop(main_text):
      # prepare a container for each chunk
      part_xml = etree.Element('utwor')
      etree.SubElement(part_xml, 'master')
      # prepare a container for each chunk
      part_xml = etree.Element('utwor')
      etree.SubElement(part_xml, 'master')
-    main_xml_part = part_xml[0] # master
+    main_xml_part = part_xml[0]  # master
  
      last_node_part = False
  
      last_node_part = False
-    
+
      # the below loop are workaround for a problem with epubs in drama ebooks without acts
      is_scene = False
      is_act = False
      # the below loop are workaround for a problem with epubs in drama ebooks without acts
      is_scene = False
      is_act = False
@@ -338,7 +357,7 @@ def chop(main_text):
              is_scene = True
          elif name == 'naglowek_akt':
              is_act = True
              is_scene = True
          elif name == 'naglowek_akt':
              is_act = True
-    
+
      for one_part in main_text:
          name = one_part.tag
          if is_act is False and is_scene is True:
      for one_part in main_text:
          name = one_part.tag
          if is_act is False and is_scene is True:
@@ -362,7 +381,7 @@ def chop(main_text):
                  main_xml_part[:] = [deepcopy(one_part)]
              else:
                  main_xml_part.append(deepcopy(one_part))
                  main_xml_part[:] = [deepcopy(one_part)]
              else:
                  main_xml_part.append(deepcopy(one_part))
-                last_node_part = False            
+                last_node_part = False
      yield part_xml
  
  
      yield part_xml
  
  
@@ -388,39 +407,47 @@ def transform_chunk(chunk_xml, chunk_no, annotations, empty=False, _empty_html_s
          replace_by_verse(chunk_xml)
          html_tree = xslt(chunk_xml, get_resource('epub/xsltScheme.xsl'))
          chars = used_chars(html_tree.getroot())
          replace_by_verse(chunk_xml)
          html_tree = xslt(chunk_xml, get_resource('epub/xsltScheme.xsl'))
          chars = used_chars(html_tree.getroot())
-        output_html = etree.tostring(html_tree, method="html", pretty_print=True)
+        output_html = etree.tostring(
+            html_tree, pretty_print=True, xml_declaration=True,
+            encoding="utf-8",
+            doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" ' +
+                    '"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+        )
      return output_html, toc, chars
  
  
      return output_html, toc, chars
  
  
-def transform(wldoc, verbose=False,
-              style=None, html_toc=False,
-              sample=None, cover=None, flags=None):
+def transform(wldoc, verbose=False, style=None, html_toc=False,
+              sample=None, cover=None, flags=None, hyphenate=False, ilustr_path='', output_type='epub'):
      """ produces a EPUB file
  
      sample=n: generate sample e-book (with at least n paragraphs)
      cover: a cover.Cover factory or True for default
      """ produces a EPUB file
  
      sample=n: generate sample e-book (with at least n paragraphs)
      cover: a cover.Cover factory or True for default
-    flags: less-advertising, without-fonts, working-copy, with-full-fonts
+    flags: less-advertising, without-fonts, working-copy
      """
  
      def transform_file(wldoc, chunk_counter=1, first=True, sample=None):
          """ processes one input file and proceeds to its children """
  
          replace_characters(wldoc.edoc.getroot())
      """
  
      def transform_file(wldoc, chunk_counter=1, first=True, sample=None):
          """ processes one input file and proceeds to its children """
  
          replace_characters(wldoc.edoc.getroot())
-        
-        hyphenator = set_hyph_language(wldoc.edoc.getroot())
+
+        hyphenator = set_hyph_language(wldoc.edoc.getroot()) if hyphenate else None
          hyphenate_and_fix_conjunctions(wldoc.edoc.getroot(), hyphenator)
          hyphenate_and_fix_conjunctions(wldoc.edoc.getroot(), hyphenator)
-        
-        
+
          # every input file will have a TOC entry,
          # pointing to starting chunk
          toc = TOC(wldoc.book_info.title, "part%d.html" % chunk_counter)
          chars = set()
          if first:
              # write book title page
          # every input file will have a TOC entry,
          # pointing to starting chunk
          toc = TOC(wldoc.book_info.title, "part%d.html" % chunk_counter)
          chars = set()
          if first:
              # write book title page
-            html_tree = xslt(wldoc.edoc, get_resource('epub/xsltTitle.xsl'))
+            html_tree = xslt(wldoc.edoc, get_resource('epub/xsltTitle.xsl'), outputtype=output_type)
              chars = used_chars(html_tree.getroot())
              chars = used_chars(html_tree.getroot())
-            zip.writestr('OPS/title.html',
-                 etree.tostring(html_tree, method="html", pretty_print=True))
+            html_string = etree.tostring(
+                html_tree, pretty_print=True, xml_declaration=True,
+                encoding="utf-8",
+                doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN"' +
+                        ' "http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+            )
+            zip.writestr('OPS/title.html', squeeze_whitespace(html_string))
              # add a title page TOC entry
              toc.add(u"Strona tytułowa", "title.html")
          elif wldoc.book_info.parts:
              # add a title page TOC entry
              toc.add(u"Strona tytułowa", "title.html")
          elif wldoc.book_info.parts:
@@ -431,8 +458,13 @@ def transform(wldoc, verbose=False,
              else:
                  html_tree = xslt(wldoc.edoc, get_resource('epub/xsltChunkTitle.xsl'))
                  chars = used_chars(html_tree.getroot())
              else:
                  html_tree = xslt(wldoc.edoc, get_resource('epub/xsltChunkTitle.xsl'))
                  chars = used_chars(html_tree.getroot())
-                html_string = etree.tostring(html_tree, method="html", pretty_print=True)
-            zip.writestr('OPS/part%d.html' % chunk_counter, html_string)
+                html_string = etree.tostring(
+                    html_tree, pretty_print=True, xml_declaration=True,
+                    encoding="utf-8",
+                    doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN"' +
+                            ' "http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+                )
+            zip.writestr('OPS/part%d.html' % chunk_counter, squeeze_whitespace(html_string))
              add_to_manifest(manifest, chunk_counter)
              add_to_spine(spine, chunk_counter)
              chunk_counter += 1
              add_to_manifest(manifest, chunk_counter)
              add_to_spine(spine, chunk_counter)
              chunk_counter += 1
@@ -458,7 +490,7 @@ def transform(wldoc, verbose=False,
  
                  toc.extend(chunk_toc)
                  chars = chars.union(chunk_chars)
  
                  toc.extend(chunk_toc)
                  chars = chars.union(chunk_chars)
-                zip.writestr('OPS/part%d.html' % chunk_counter, chunk_html)
+                zip.writestr('OPS/part%d.html' % chunk_counter, squeeze_whitespace(chunk_html))
                  add_to_manifest(manifest, chunk_counter)
                  add_to_spine(spine, chunk_counter)
                  chunk_counter += 1
                  add_to_manifest(manifest, chunk_counter)
                  add_to_spine(spine, chunk_counter)
                  chunk_counter += 1
@@ -471,7 +503,6 @@ def transform(wldoc, verbose=False,
  
          return toc, chunk_counter, chars, sample
  
  
          return toc, chunk_counter, chars, sample
  
-
      document = deepcopy(wldoc)
      del wldoc
  
      document = deepcopy(wldoc)
      del wldoc
  
@@ -479,9 +510,14 @@ def transform(wldoc, verbose=False,
          for flag in flags:
              document.edoc.getroot().set(flag, 'yes')
  
          for flag in flags:
              document.edoc.getroot().set(flag, 'yes')
  
+    document.clean_ed_note()
+    document.clean_ed_note('abstrakt')
+
      # add editors info
      # add editors info
-    document.edoc.getroot().set('editors', u', '.join(sorted(
-        editor.readable() for editor in document.editors())))
+    editors = document.editors()
+    if editors:
+        document.edoc.getroot().set('editors', u', '.join(sorted(
+            editor.readable() for editor in editors)))
      if document.book_info.funders:
          document.edoc.getroot().set('funders', u', '.join(
              document.book_info.funders))
      if document.book_info.funders:
          document.edoc.getroot().set('funders', u', '.join(
              document.book_info.funders))
@@ -496,28 +532,44 @@ def transform(wldoc, verbose=False,
      output_file = NamedTemporaryFile(prefix='librarian', suffix='.epub', delete=False)
      zip = zipfile.ZipFile(output_file, 'w', zipfile.ZIP_DEFLATED)
  
      output_file = NamedTemporaryFile(prefix='librarian', suffix='.epub', delete=False)
      zip = zipfile.ZipFile(output_file, 'w', zipfile.ZIP_DEFLATED)
  
+    functions.reg_mathml_epub(zip)
+
+    if os.path.isdir(ilustr_path):
+        for i, filename in enumerate(os.listdir(ilustr_path)):
+            file_path = os.path.join(ilustr_path, filename)
+            zip.write(file_path, os.path.join('OPS', filename))
+            image_id = 'image%s' % i
+            manifest.append(etree.fromstring(
+                '<item id="%s" href="%s" media-type="%s" />' % (image_id, filename, guess_type(file_path)[0])))
+
      # write static elements
      mime = zipfile.ZipInfo()
      mime.filename = 'mimetype'
      mime.compress_type = zipfile.ZIP_STORED
      # write static elements
      mime = zipfile.ZipInfo()
      mime.filename = 'mimetype'
      mime.compress_type = zipfile.ZIP_STORED
-    mime.extra = ''
-    zip.writestr(mime, 'application/epub+zip')
-    zip.writestr('META-INF/container.xml', '<?xml version="1.0" ?><container version="1.0" ' \
-                       'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">' \
-                       '<rootfiles><rootfile full-path="OPS/content.opf" ' \
-                       'media-type="application/oebps-package+xml" />' \
-                       '</rootfiles></container>')
-    zip.write(get_resource('res/wl-logo-small.png'), os.path.join('OPS', 'logo_wolnelektury.png'))
-    zip.write(get_resource('res/jedenprocent.png'), os.path.join('OPS', 'jedenprocent.png'))
+    mime.extra = b''
+    zip.writestr(mime, b'application/epub+zip')
+    zip.writestr(
+        'META-INF/container.xml',
+        b'<?xml version="1.0" ?>'
+        b'<container version="1.0" '
+        b'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">'
+        b'<rootfiles><rootfile full-path="OPS/content.opf" '
+        b'media-type="application/oebps-package+xml" />'
+        b'</rootfiles></container>'
+    )
+    zip.write(get_resource('res/wl-logo-small.png'),
+              os.path.join('OPS', 'logo_wolnelektury.png'))
+    zip.write(get_resource('res/jedenprocent.png'),
+              os.path.join('OPS', 'jedenprocent.png'))
      if not style:
          style = get_resource('epub/style.css')
      zip.write(style, os.path.join('OPS', 'style.css'))
  
      if cover:
          if cover is True:
      if not style:
          style = get_resource('epub/style.css')
      zip.write(style, os.path.join('OPS', 'style.css'))
  
      if cover:
          if cover is True:
-            cover = DefaultEbookCover
+            cover = make_cover
  
  
-        cover_file = StringIO()
+        cover_file = BytesIO()
          bound_cover = cover(document.book_info)
          bound_cover.save(cover_file)
          cover_name = 'cover.%s' % bound_cover.ext()
          bound_cover = cover(document.book_info)
          bound_cover.save(cover_file)
          cover_name = 'cover.%s' % bound_cover.ext()
@@ -527,7 +579,11 @@ def transform(wldoc, verbose=False,
          cover_tree = etree.parse(get_resource('epub/cover.html'))
          cover_tree.find('//' + XHTMLNS('img')).set('src', cover_name)
          zip.writestr('OPS/cover.html', etree.tostring(
          cover_tree = etree.parse(get_resource('epub/cover.html'))
          cover_tree.find('//' + XHTMLNS('img')).set('src', cover_name)
          zip.writestr('OPS/cover.html', etree.tostring(
-                        cover_tree, method="html", pretty_print=True))
+            cover_tree, pretty_print=True, xml_declaration=True,
+            encoding="utf-8",
+            doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" ' +
+                    '"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+        ))
  
          if bound_cover.uses_dc_cover:
              if document.book_info.cover_by:
  
          if bound_cover.uses_dc_cover:
              if document.book_info.cover_by:
@@ -543,14 +599,16 @@ def transform(wldoc, verbose=False,
          opf.getroot()[0].append(etree.fromstring('<meta name="cover" content="cover-image"/>'))
          guide.append(etree.fromstring('<reference href="cover.html" type="cover" title="Okładka"/>'))
  
          opf.getroot()[0].append(etree.fromstring('<meta name="cover" content="cover-image"/>'))
          guide.append(etree.fromstring('<reference href="cover.html" type="cover" title="Okładka"/>'))
  
-
      annotations = etree.Element('annotations')
  
      annotations = etree.Element('annotations')
  
-    toc_file = etree.fromstring('<?xml version="1.0" encoding="utf-8"?><!DOCTYPE ncx PUBLIC ' \
-                               '"-//NISO//DTD ncx 2005-1//EN" "http://www.daisy.org/z3986/2005/ncx-2005-1.dtd">' \
-                               '<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" xml:lang="pl" ' \
-                               'version="2005-1"><head></head><docTitle></docTitle><navMap>' \
-                               '</navMap></ncx>')
+    toc_file = etree.fromstring(
+        b'<?xml version="1.0" encoding="utf-8"?><!DOCTYPE ncx PUBLIC '
+        b'"-//NISO//DTD ncx 2005-1//EN" '
+        b'"http://www.daisy.org/z3986/2005/ncx-2005-1.dtd">'
+        b'<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" xml:lang="pl" '
+        b'version="2005-1"><head></head><docTitle></docTitle><navMap>'
+        b'</navMap></ncx>'
+    )
      nav_map = toc_file[-1]
  
      if html_toc:
      nav_map = toc_file[-1]
  
      if html_toc:
@@ -576,28 +634,36 @@ def transform(wldoc, verbose=False,
          html_tree = xslt(annotations, get_resource('epub/xsltAnnotations.xsl'))
          chars = chars.union(used_chars(html_tree.getroot()))
          zip.writestr('OPS/annotations.html', etree.tostring(
          html_tree = xslt(annotations, get_resource('epub/xsltAnnotations.xsl'))
          chars = chars.union(used_chars(html_tree.getroot()))
          zip.writestr('OPS/annotations.html', etree.tostring(
-                            html_tree, method="html", pretty_print=True))
+            html_tree, pretty_print=True, xml_declaration=True,
+            encoding="utf-8",
+            doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" ' +
+                    '"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+        ))
  
      toc.add("Wesprzyj Wolne Lektury", "support.html")
      manifest.append(etree.fromstring(
          '<item id="support" href="support.html" media-type="application/xhtml+xml" />'))
      spine.append(etree.fromstring(
          '<itemref idref="support" />'))
  
      toc.add("Wesprzyj Wolne Lektury", "support.html")
      manifest.append(etree.fromstring(
          '<item id="support" href="support.html" media-type="application/xhtml+xml" />'))
      spine.append(etree.fromstring(
          '<itemref idref="support" />'))
-    html_string = open(get_resource('epub/support.html')).read()
+    html_string = open(get_resource('epub/support.html'), 'rb').read()
      chars.update(used_chars(etree.fromstring(html_string)))
      chars.update(used_chars(etree.fromstring(html_string)))
-    zip.writestr('OPS/support.html', html_string)
+    zip.writestr('OPS/support.html', squeeze_whitespace(html_string))
  
      toc.add("Strona redakcyjna", "last.html")
      manifest.append(etree.fromstring(
          '<item id="last" href="last.html" media-type="application/xhtml+xml" />'))
      spine.append(etree.fromstring(
          '<itemref idref="last" />'))
  
      toc.add("Strona redakcyjna", "last.html")
      manifest.append(etree.fromstring(
          '<item id="last" href="last.html" media-type="application/xhtml+xml" />'))
      spine.append(etree.fromstring(
          '<itemref idref="last" />'))
-    html_tree = xslt(document.edoc, get_resource('epub/xsltLast.xsl'))
+    html_tree = xslt(document.edoc, get_resource('epub/xsltLast.xsl'), outputtype=output_type)
      chars.update(used_chars(html_tree.getroot()))
      chars.update(used_chars(html_tree.getroot()))
-    zip.writestr('OPS/last.html', etree.tostring(
-                        html_tree, method="html", pretty_print=True))
-
-    if not flags or not 'without-fonts' in flags:
+    zip.writestr('OPS/last.html', squeeze_whitespace(etree.tostring(
+        html_tree, pretty_print=True, xml_declaration=True,
+        encoding="utf-8",
+        doctype='<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN" ' +
+                '"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">'
+    )))
+
+    if not flags or 'without-fonts' not in flags:
          # strip fonts
          tmpdir = mkdtemp('-librarian-epub')
          try:
          # strip fonts
          tmpdir = mkdtemp('-librarian-epub')
          try:
@@ -607,23 +673,25 @@ def transform(wldoc, verbose=False,
  
          os.chdir(os.path.join(os.path.dirname(os.path.realpath(__file__)), 'font-optimizer'))
          for fname in 'DejaVuSerif.ttf', 'DejaVuSerif-Bold.ttf', 'DejaVuSerif-Italic.ttf', 'DejaVuSerif-BoldItalic.ttf':
  
          os.chdir(os.path.join(os.path.dirname(os.path.realpath(__file__)), 'font-optimizer'))
          for fname in 'DejaVuSerif.ttf', 'DejaVuSerif-Bold.ttf', 'DejaVuSerif-Italic.ttf', 'DejaVuSerif-BoldItalic.ttf':
-            if not flags or not 'with-full-fonts' in flags:
-                optimizer_call = ['perl', 'subset.pl', '--chars', ''.join(chars).encode('utf-8'),
-                              get_resource('fonts/' + fname), os.path.join(tmpdir, fname)]              
-                if verbose:
-                    print "Running font-optimizer"
-                    subprocess.check_call(optimizer_call)
-                else:
-                    subprocess.check_call(optimizer_call, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
-                    zip.write(os.path.join(tmpdir, fname), os.path.join('OPS', fname))
+            optimizer_call = ['perl', 'subset.pl', '--chars',
+                              ''.join(chars).encode('utf-8'),
+                              get_resource('fonts/' + fname),
+                              os.path.join(tmpdir, fname)]
+            env = {"PERL_USE_UNSAFE_INC": "1"}
+            if verbose:
+                print("Running font-optimizer")
+                subprocess.check_call(optimizer_call, env=env)
              else:
              else:
-                zip.write(get_resource('fonts/' + fname), os.path.join('OPS', fname))
+                dev_null = open(os.devnull, 'w')
+                subprocess.check_call(optimizer_call, stdout=dev_null, stderr=dev_null, env=env)
+            zip.write(os.path.join(tmpdir, fname), os.path.join('OPS', fname))
              manifest.append(etree.fromstring(
                  '<item id="%s" href="%s" media-type="application/x-font-truetype" />' % (fname, fname)))
          rmtree(tmpdir)
          if cwd is not None:
              os.chdir(cwd)
              manifest.append(etree.fromstring(
                  '<item id="%s" href="%s" media-type="application/x-font-truetype" />' % (fname, fname)))
          rmtree(tmpdir)
          if cwd is not None:
              os.chdir(cwd)
-    zip.writestr('OPS/content.opf', etree.tostring(opf, pretty_print=True, xml_declaration = True, encoding='UTF-8'))
+    zip.writestr('OPS/content.opf', etree.tostring(opf, pretty_print=True,
+                 xml_declaration=True, encoding="utf-8"))
      title = document.book_info.title
      attributes = "dtb:uid", "dtb:depth", "dtb:totalPageCount", "dtb:maxPageNumber"
      for st in attributes:
      title = document.book_info.title
      attributes = "dtb:uid", "dtb:depth", "dtb:totalPageCount", "dtb:maxPageNumber"
      for st in attributes:
@@ -640,7 +708,8 @@ def transform(wldoc, verbose=False,
          toc.add(u"Spis treści", "toc.html", index=1)
          zip.writestr('OPS/toc.html', toc.html().encode('utf-8'))
      toc.write_to_xml(nav_map)
          toc.add(u"Spis treści", "toc.html", index=1)
          zip.writestr('OPS/toc.html', toc.html().encode('utf-8'))
      toc.write_to_xml(nav_map)
-    zip.writestr('OPS/toc.ncx', etree.tostring(toc_file, pretty_print=True, xml_declaration = True, encoding='UTF-8'))
+    zip.writestr('OPS/toc.ncx', etree.tostring(toc_file, pretty_print=True,
+                 xml_declaration=True, encoding="utf-8"))
      zip.close()
  
      return OutputFile.from_filename(output_file.name)
      zip.close()
  
      return OutputFile.from_filename(output_file.name)