#1868: check uri pattern in strict mode

[librarian.git] / librarian / parser.py
diff --git a/librarian/parser.py b/librarian/parser.py

index 55b4e4b..2ece72f 100644 (file)
--- a/librarian/parser.py
+++ b/librarian/parser.py
@@ -1,25 +1,32 @@
  # -*- coding: utf-8 -*-
  # -*- coding: utf-8 -*-
-from librarian import ValidationError, NoDublinCore,  ParseError
-from librarian import RDFNS, DCNS
+#
+# This file is part of Librarian, licensed under GNU Affero GPLv3 or later.
+# Copyright © Fundacja Nowoczesna Polska. See NOTICE for more information.
+#
+from librarian import ValidationError, NoDublinCore,  ParseError, NoProvider
+from librarian import RDFNS
  from librarian import dcparser
  
  from xml.parsers.expat import ExpatError
  from lxml import etree
  from lxml.etree import XMLSyntaxError, XSLTApplyError
  
  from librarian import dcparser
  
  from xml.parsers.expat import ExpatError
  from lxml import etree
  from lxml.etree import XMLSyntaxError, XSLTApplyError
  
+import os
  import re
  from StringIO import StringIO
  
  class WLDocument(object):
  import re
  from StringIO import StringIO
  
  class WLDocument(object):
-    LINE_SWAP_EXPR = re.compile(r'/\s', re.MULTILINE | re.UNICODE);
+    LINE_SWAP_EXPR = re.compile(r'/\s', re.MULTILINE | re.UNICODE)
+    provider = None
  
  
-    def __init__(self, edoc, parse_dublincore=True):
+    def __init__(self, edoc, parse_dublincore=True, provider=None, strict=False):
          self.edoc = edoc
          self.edoc = edoc
+        self.provider = provider
  
          root_elem = edoc.getroot()
  
          root_elem = edoc.getroot()
-       
+
          dc_path = './/' + RDFNS('RDF')
          dc_path = './/' + RDFNS('RDF')
-        
+
          if root_elem.tag != 'utwor':
              raise ValidationError("Invalid root element. Found '%s', should be 'utwor'" % root_elem.tag)
  
          if root_elem.tag != 'utwor':
              raise ValidationError("Invalid root element. Found '%s', should be 'utwor'" % root_elem.tag)
  
@@ -28,17 +35,18 @@ class WLDocument(object):
  
              if self.rdf_elem is None:
                  raise NoDublinCore('Document has no DublinCore - which is required.')
  
              if self.rdf_elem is None:
                  raise NoDublinCore('Document has no DublinCore - which is required.')
-            
-            self.book_info = dcparser.BookInfo.from_element(self.rdf_elem)
+
+            self.book_info = dcparser.BookInfo.from_element(
+                    self.rdf_elem, strict=strict)
          else:
              self.book_info = None
          else:
              self.book_info = None
-    
+
      @classmethod
      @classmethod
-    def from_string(cls, xml, swap_endlines=False, parse_dublincore=True):
-        return cls.from_file(StringIO(xml), swap_endlines, parse_dublincore=parse_dublincore)
+    def from_string(cls, xml, *args, **kwargs):
+        return cls.from_file(StringIO(xml), *args, **kwargs)
  
      @classmethod
  
      @classmethod
-    def from_file(cls, xmlfile, swap_endlines=False, parse_dublincore=True):
+    def from_file(cls, xmlfile, parse_dublincore=True, provider=None):
  
          # first, prepare for parsing
          if isinstance(xmlfile, basestring):
  
          # first, prepare for parsing
          if isinstance(xmlfile, basestring):
@@ -53,27 +61,55 @@ class WLDocument(object):
          if not isinstance(data, unicode):
              data = data.decode('utf-8')
  
          if not isinstance(data, unicode):
              data = data.decode('utf-8')
  
-        if swap_endlines:
-            data = cls.LINE_SWAP_EXPR.sub(u'<br />\n', data)
-    
+        data = data.replace(u'\ufeff', '')
+
          try:
          try:
-            parser = etree.XMLParser(remove_blank_text=True)
-            return cls(etree.parse(StringIO(data), parser), parse_dublincore=parse_dublincore)
-        except (ExpatError, XMLSyntaxError, XSLTApplyError), e:
-            raise ParseError(e)                  
+            parser = etree.XMLParser(remove_blank_text=False)
+            tree = etree.parse(StringIO(data.encode('utf-8')), parser)
  
  
-    def part_as_text(self, path):
-        # convert the path to XPath        
-        print "[L] Retrieving part:", path
+            return cls(tree, parse_dublincore=parse_dublincore, provider=provider)
+        except (ExpatError, XMLSyntaxError, XSLTApplyError), e:
+            raise ParseError(e)
+
+    def swap_endlines(self):
+        """Converts line breaks in stanzas into <br/> tags."""
+        # only swap inside stanzas
+        for elem in self.edoc.iter('strofa'):
+            for child in list(elem):
+                if child.tail:
+                    chunks = self.LINE_SWAP_EXPR.split(child.tail)
+                    ins_index = elem.index(child) + 1
+                    while len(chunks) > 1:
+                        ins = etree.Element('br')
+                        ins.tail = chunks.pop()
+                        elem.insert(ins_index, ins)
+                    child.tail = chunks.pop(0)
+            if elem.text:
+                chunks = self.LINE_SWAP_EXPR.split(elem.text)
+                while len(chunks) > 1:
+                    ins = etree.Element('br')
+                    ins.tail = chunks.pop()
+                    elem.insert(0, ins)
+                elem.text = chunks.pop(0)
+
+    def parts(self):
+        if self.provider is None:
+            raise NoProvider('No document provider supplied.')
+        if self.book_info is None:
+            raise NoDublinCore('No Dublin Core in document.')
+        for part_uri in self.book_info.parts:
+            yield self.from_file(self.provider.by_uri(part_uri),
+                    provider=self.provider)
+
+    def chunk(self, path):
+        # convert the path to XPath
+        expr = self.path_to_xpath(path)
+        elems = self.edoc.xpath(expr)
  
  
-        elems = self.edoc.xpath(self.path_to_xpath(path))
-        print "[L] xpath", elems
-        
          if len(elems) == 0:
          if len(elems) == 0:
-            return None        
-
-        return etree.tostring(elems[0], encoding=unicode, pretty_print=True)
-
+            return None
+        else:
+            return elems[0]
  
      def path_to_xpath(self, path):
          parts = []
  
      def path_to_xpath(self, path):
          parts = []
@@ -84,7 +120,7 @@ class WLDocument(object):
                  parts.append(part)
              else:
                  tag, n = match.groups()
                  parts.append(part)
              else:
                  tag, n = match.groups()
-                parts.append("node()[position() = %d and name() = '%s']" % (int(n), tag) )
+                parts.append("*[%d][name() = '%s']" % (int(n)+1, tag) )
  
          if parts[0] == '.':
              parts[0] = ''
  
          if parts[0] == '.':
              parts[0] = ''
@@ -95,8 +131,9 @@ class WLDocument(object):
          return self.edoc.xslt(stylesheet, **options)
  
      def update_dc(self):
          return self.edoc.xslt(stylesheet, **options)
  
      def update_dc(self):
-        parent = self.rdf_elem.getparent()
-        parent.replace( self.rdf_elem, self.book_info.to_etree(parent) )
+        if self.book_info:
+            parent = self.rdf_elem.getparent()
+            parent.replace( self.rdf_elem, self.book_info.to_etree(parent) )
  
      def serialize(self):
          self.update_dc()
  
      def serialize(self):
          self.update_dc()
@@ -108,10 +145,58 @@ class WLDocument(object):
          for key, data in chunk_dict.iteritems():
              try:
                  xpath = self.path_to_xpath(key)
          for key, data in chunk_dict.iteritems():
              try:
                  xpath = self.path_to_xpath(key)
-                node = self.edoc.xpath(xpath)[0]                
-                repl = etree.fromstring(data)
+                node = self.edoc.xpath(xpath)[0]
+                repl = etree.fromstring(u"<%s>%s</%s>" %(node.tag, data, node.tag) )
                  node.getparent().replace(node, repl);
              except Exception, e:
                  unmerged.append( repr( (key, xpath, e) ) )
  
                  node.getparent().replace(node, repl);
              except Exception, e:
                  unmerged.append( repr( (key, xpath, e) ) )
  
-        return unmerged
-\ No newline at end of file
+        return unmerged
+
+    def clean_ed_note(self):
+        """ deletes forbidden tags from nota_red """
+
+        for node in self.edoc.xpath('|'.join('//nota_red//%s' % tag for tag in
+                    ('pa', 'pe', 'pr', 'pt', 'begin', 'end', 'motyw'))):
+            tail = node.tail
+            node.clear()
+            node.tag = 'span'
+            node.tail = tail
+
+    # Converters
+
+    def as_html(self, *args, **kwargs):
+        from librarian import html
+        return html.transform(self, *args, **kwargs)
+
+    def as_text(self, *args, **kwargs):
+        from librarian import text
+        return text.transform(self, *args, **kwargs)
+
+    def as_epub(self, *args, **kwargs):
+        from librarian import epub
+        return epub.transform(self, *args, **kwargs)
+
+    def as_pdf(self, *args, **kwargs):
+        from librarian import pdf
+        return pdf.transform(self, *args, **kwargs)
+
+    def as_mobi(self, *args, **kwargs):
+        from librarian import mobi
+        return mobi.transform(self, *args, **kwargs)
+
+    def save_output_file(self, output_file, output_path=None,
+            output_dir_path=None, make_author_dir=False, ext=None):
+        if output_dir_path:
+            save_path = output_dir_path
+            if make_author_dir:
+                save_path = os.path.join(save_path,
+                        unicode(self.book_info.author).encode('utf-8'))
+            save_path = os.path.join(save_path,
+                                self.book_info.uri.slug)
+            if ext:
+                save_path += '.%s' % ext
+        else:
+            save_path = output_path
+
+        output_file.save_as(save_path)