2014-03-10 01:18:05 -04:00
|
|
|
"""Module for supporting the lxml.etree library. The idea here is to use as much
|
|
|
|
of the native library as possible, without using fragile hacks like custom element
|
|
|
|
names that break between releases. The downside of this is that we cannot represent
|
|
|
|
all possible trees; specifically the following are known to cause problems:
|
|
|
|
|
|
|
|
Text or comments as siblings of the root element
|
|
|
|
Docypes with no name
|
|
|
|
|
|
|
|
When any of these things occur, we emit a DataLossWarning
|
|
|
|
"""
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
from __future__ import absolute_import, division, unicode_literals
|
|
|
|
|
|
|
|
import warnings
|
|
|
|
import re
|
|
|
|
import sys
|
|
|
|
|
|
|
|
from . import _base
|
|
|
|
from ..constants import DataLossWarning
|
|
|
|
from .. import constants
|
|
|
|
from . import etree as etree_builders
|
|
|
|
from .. import ihatexml
|
|
|
|
|
|
|
|
import lxml.etree as etree
|
|
|
|
|
|
|
|
|
|
|
|
fullTree = True
|
|
|
|
tag_regexp = re.compile("{([^}]*)}(.*)")
|
|
|
|
|
|
|
|
comment_type = etree.Comment("asd").tag
|
|
|
|
|
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
class DocumentType(object):
|
|
|
|
def __init__(self, name, publicId, systemId):
|
2014-07-21 19:01:46 -04:00
|
|
|
self.name = name
|
2014-03-10 01:18:05 -04:00
|
|
|
self.publicId = publicId
|
|
|
|
self.systemId = systemId
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
class Document(object):
|
|
|
|
def __init__(self):
|
|
|
|
self._elementTree = None
|
|
|
|
self._childNodes = []
|
|
|
|
|
|
|
|
def appendChild(self, element):
|
|
|
|
self._elementTree.getroot().addnext(element._element)
|
|
|
|
|
|
|
|
def _getChildNodes(self):
|
|
|
|
return self._childNodes
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
childNodes = property(_getChildNodes)
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def testSerializer(element):
|
|
|
|
rv = []
|
|
|
|
finalText = None
|
2014-07-21 19:01:46 -04:00
|
|
|
infosetFilter = ihatexml.InfosetFilter()
|
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def serializeElement(element, indent=0):
|
|
|
|
if not hasattr(element, "tag"):
|
2014-07-21 19:01:46 -04:00
|
|
|
if hasattr(element, "getroot"):
|
|
|
|
# Full tree case
|
2014-03-10 01:18:05 -04:00
|
|
|
rv.append("#document")
|
|
|
|
if element.docinfo.internalDTD:
|
2014-07-21 19:01:46 -04:00
|
|
|
if not (element.docinfo.public_id or
|
2014-03-10 01:18:05 -04:00
|
|
|
element.docinfo.system_url):
|
2014-07-21 19:01:46 -04:00
|
|
|
dtd_str = "<!DOCTYPE %s>" % element.docinfo.root_name
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
dtd_str = """<!DOCTYPE %s "%s" "%s">""" % (
|
|
|
|
element.docinfo.root_name,
|
2014-03-10 01:18:05 -04:00
|
|
|
element.docinfo.public_id,
|
|
|
|
element.docinfo.system_url)
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("|%s%s" % (' ' * (indent + 2), dtd_str))
|
2014-03-10 01:18:05 -04:00
|
|
|
next_element = element.getroot()
|
|
|
|
while next_element.getprevious() is not None:
|
|
|
|
next_element = next_element.getprevious()
|
|
|
|
while next_element is not None:
|
2014-07-21 19:01:46 -04:00
|
|
|
serializeElement(next_element, indent + 2)
|
2014-03-10 01:18:05 -04:00
|
|
|
next_element = next_element.getnext()
|
2014-07-21 19:01:46 -04:00
|
|
|
elif isinstance(element, str) or isinstance(element, bytes):
|
|
|
|
# Text in a fragment
|
|
|
|
assert isinstance(element, str) or sys.version_info.major == 2
|
|
|
|
rv.append("|%s\"%s\"" % (' ' * indent, element))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
# Fragment case
|
2014-03-10 01:18:05 -04:00
|
|
|
rv.append("#document-fragment")
|
|
|
|
for next_element in element:
|
2014-07-21 19:01:46 -04:00
|
|
|
serializeElement(next_element, indent + 2)
|
|
|
|
elif element.tag == comment_type:
|
|
|
|
rv.append("|%s<!-- %s -->" % (' ' * indent, element.text))
|
|
|
|
if hasattr(element, "tail") and element.tail:
|
|
|
|
rv.append("|%s\"%s\"" % (' ' * indent, element.tail))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
assert isinstance(element, etree._Element)
|
2014-03-10 01:18:05 -04:00
|
|
|
nsmatch = etree_builders.tag_regexp.match(element.tag)
|
|
|
|
if nsmatch is not None:
|
|
|
|
ns = nsmatch.group(1)
|
|
|
|
tag = nsmatch.group(2)
|
|
|
|
prefix = constants.prefixes[ns]
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("|%s<%s %s>" % (' ' * indent, prefix,
|
|
|
|
infosetFilter.fromXmlName(tag)))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("|%s<%s>" % (' ' * indent,
|
|
|
|
infosetFilter.fromXmlName(element.tag)))
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
if hasattr(element, "attrib"):
|
|
|
|
attributes = []
|
2014-07-21 19:01:46 -04:00
|
|
|
for name, value in element.attrib.items():
|
2014-03-10 01:18:05 -04:00
|
|
|
nsmatch = tag_regexp.match(name)
|
|
|
|
if nsmatch is not None:
|
|
|
|
ns, name = nsmatch.groups()
|
2014-07-21 19:01:46 -04:00
|
|
|
name = infosetFilter.fromXmlName(name)
|
2014-03-10 01:18:05 -04:00
|
|
|
prefix = constants.prefixes[ns]
|
2014-07-21 19:01:46 -04:00
|
|
|
attr_string = "%s %s" % (prefix, name)
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
attr_string = infosetFilter.fromXmlName(name)
|
2014-03-10 01:18:05 -04:00
|
|
|
attributes.append((attr_string, value))
|
|
|
|
|
|
|
|
for name, value in sorted(attributes):
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append('|%s%s="%s"' % (' ' * (indent + 2), name, value))
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
if element.text:
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("|%s\"%s\"" % (' ' * (indent + 2), element.text))
|
2014-03-10 01:18:05 -04:00
|
|
|
indent += 2
|
2014-07-21 19:01:46 -04:00
|
|
|
for child in element:
|
2014-03-10 01:18:05 -04:00
|
|
|
serializeElement(child, indent)
|
2014-07-21 19:01:46 -04:00
|
|
|
if hasattr(element, "tail") and element.tail:
|
|
|
|
rv.append("|%s\"%s\"" % (' ' * (indent - 2), element.tail))
|
2014-03-10 01:18:05 -04:00
|
|
|
serializeElement(element, 0)
|
|
|
|
|
|
|
|
if finalText is not None:
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("|%s\"%s\"" % (' ' * 2, finalText))
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
return "\n".join(rv)
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def tostring(element):
|
|
|
|
"""Serialize an element and its child nodes to a string"""
|
|
|
|
rv = []
|
|
|
|
finalText = None
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def serializeElement(element):
|
|
|
|
if not hasattr(element, "tag"):
|
|
|
|
if element.docinfo.internalDTD:
|
|
|
|
if element.docinfo.doctype:
|
|
|
|
dtd_str = element.docinfo.doctype
|
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
dtd_str = "<!DOCTYPE %s>" % element.docinfo.root_name
|
2014-03-10 01:18:05 -04:00
|
|
|
rv.append(dtd_str)
|
|
|
|
serializeElement(element.getroot())
|
2014-07-21 19:01:46 -04:00
|
|
|
|
|
|
|
elif element.tag == comment_type:
|
|
|
|
rv.append("<!--%s-->" % (element.text,))
|
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
# This is assumed to be an ordinary element
|
2014-03-10 01:18:05 -04:00
|
|
|
if not element.attrib:
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("<%s>" % (element.tag,))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
attr = " ".join(["%s=\"%s\"" % (name, value)
|
|
|
|
for name, value in element.attrib.items()])
|
|
|
|
rv.append("<%s %s>" % (element.tag, attr))
|
2014-03-10 01:18:05 -04:00
|
|
|
if element.text:
|
|
|
|
rv.append(element.text)
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
for child in element:
|
2014-03-10 01:18:05 -04:00
|
|
|
serializeElement(child)
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("</%s>" % (element.tag,))
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
if hasattr(element, "tail") and element.tail:
|
|
|
|
rv.append(element.tail)
|
|
|
|
|
|
|
|
serializeElement(element)
|
|
|
|
|
|
|
|
if finalText is not None:
|
2014-07-21 19:01:46 -04:00
|
|
|
rv.append("%s\"" % (' ' * 2, finalText))
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
return "".join(rv)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
class TreeBuilder(_base.TreeBuilder):
|
|
|
|
documentClass = Document
|
|
|
|
doctypeClass = DocumentType
|
|
|
|
elementClass = None
|
|
|
|
commentClass = None
|
2014-07-21 19:01:46 -04:00
|
|
|
fragmentClass = Document
|
|
|
|
implementation = etree
|
2014-03-10 01:18:05 -04:00
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
def __init__(self, namespaceHTMLElements, fullTree=False):
|
2014-03-10 01:18:05 -04:00
|
|
|
builder = etree_builders.getETreeModule(etree, fullTree=fullTree)
|
2014-07-21 19:01:46 -04:00
|
|
|
infosetFilter = self.infosetFilter = ihatexml.InfosetFilter()
|
2014-03-10 01:18:05 -04:00
|
|
|
self.namespaceHTMLElements = namespaceHTMLElements
|
|
|
|
|
|
|
|
class Attributes(dict):
|
|
|
|
def __init__(self, element, value={}):
|
|
|
|
self._element = element
|
|
|
|
dict.__init__(self, value)
|
2014-07-21 19:01:46 -04:00
|
|
|
for key, value in self.items():
|
2014-03-10 01:18:05 -04:00
|
|
|
if isinstance(key, tuple):
|
2014-07-21 19:01:46 -04:00
|
|
|
name = "{%s}%s" % (key[2], infosetFilter.coerceAttribute(key[1]))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
name = infosetFilter.coerceAttribute(key)
|
2014-03-10 01:18:05 -04:00
|
|
|
self._element._element.attrib[name] = value
|
|
|
|
|
|
|
|
def __setitem__(self, key, value):
|
|
|
|
dict.__setitem__(self, key, value)
|
|
|
|
if isinstance(key, tuple):
|
2014-07-21 19:01:46 -04:00
|
|
|
name = "{%s}%s" % (key[2], infosetFilter.coerceAttribute(key[1]))
|
2014-03-10 01:18:05 -04:00
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
name = infosetFilter.coerceAttribute(key)
|
2014-03-10 01:18:05 -04:00
|
|
|
self._element._element.attrib[name] = value
|
|
|
|
|
|
|
|
class Element(builder.Element):
|
|
|
|
def __init__(self, name, namespace):
|
2014-07-21 19:01:46 -04:00
|
|
|
name = infosetFilter.coerceElement(name)
|
2014-03-10 01:18:05 -04:00
|
|
|
builder.Element.__init__(self, name, namespace=namespace)
|
|
|
|
self._attributes = Attributes(self)
|
|
|
|
|
|
|
|
def _setName(self, name):
|
2014-07-21 19:01:46 -04:00
|
|
|
self._name = infosetFilter.coerceElement(name)
|
2014-03-10 01:18:05 -04:00
|
|
|
self._element.tag = self._getETreeTag(
|
|
|
|
self._name, self._namespace)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def _getName(self):
|
2014-07-21 19:01:46 -04:00
|
|
|
return infosetFilter.fromXmlName(self._name)
|
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
name = property(_getName, _setName)
|
|
|
|
|
|
|
|
def _getAttributes(self):
|
|
|
|
return self._attributes
|
|
|
|
|
|
|
|
def _setAttributes(self, attributes):
|
|
|
|
self._attributes = Attributes(self, attributes)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
attributes = property(_getAttributes, _setAttributes)
|
|
|
|
|
|
|
|
def insertText(self, data, insertBefore=None):
|
2014-07-21 19:01:46 -04:00
|
|
|
data = infosetFilter.coerceCharacters(data)
|
2014-03-10 01:18:05 -04:00
|
|
|
builder.Element.insertText(self, data, insertBefore)
|
|
|
|
|
|
|
|
def appendChild(self, child):
|
|
|
|
builder.Element.appendChild(self, child)
|
|
|
|
|
|
|
|
class Comment(builder.Comment):
|
|
|
|
def __init__(self, data):
|
2014-07-21 19:01:46 -04:00
|
|
|
data = infosetFilter.coerceComment(data)
|
2014-03-10 01:18:05 -04:00
|
|
|
builder.Comment.__init__(self, data)
|
|
|
|
|
|
|
|
def _setData(self, data):
|
2014-07-21 19:01:46 -04:00
|
|
|
data = infosetFilter.coerceComment(data)
|
2014-03-10 01:18:05 -04:00
|
|
|
self._element.text = data
|
|
|
|
|
|
|
|
def _getData(self):
|
|
|
|
return self._element.text
|
|
|
|
|
|
|
|
data = property(_getData, _setData)
|
|
|
|
|
|
|
|
self.elementClass = Element
|
|
|
|
self.commentClass = builder.Comment
|
2014-07-21 19:01:46 -04:00
|
|
|
# self.fragmentClass = builder.DocumentFragment
|
2014-03-10 01:18:05 -04:00
|
|
|
_base.TreeBuilder.__init__(self, namespaceHTMLElements)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def reset(self):
|
|
|
|
_base.TreeBuilder.reset(self)
|
|
|
|
self.insertComment = self.insertCommentInitial
|
|
|
|
self.initial_comments = []
|
|
|
|
self.doctype = None
|
|
|
|
|
|
|
|
def testSerializer(self, element):
|
|
|
|
return testSerializer(element)
|
|
|
|
|
|
|
|
def getDocument(self):
|
|
|
|
if fullTree:
|
|
|
|
return self.document._elementTree
|
|
|
|
else:
|
|
|
|
return self.document._elementTree.getroot()
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def getFragment(self):
|
|
|
|
fragment = []
|
|
|
|
element = self.openElements[0]._element
|
|
|
|
if element.text:
|
|
|
|
fragment.append(element.text)
|
2014-07-21 19:01:46 -04:00
|
|
|
fragment.extend(list(element))
|
2014-03-10 01:18:05 -04:00
|
|
|
if element.tail:
|
|
|
|
fragment.append(element.tail)
|
|
|
|
return fragment
|
|
|
|
|
|
|
|
def insertDoctype(self, token):
|
|
|
|
name = token["name"]
|
|
|
|
publicId = token["publicId"]
|
|
|
|
systemId = token["systemId"]
|
|
|
|
|
2014-07-21 19:01:46 -04:00
|
|
|
if not name:
|
|
|
|
warnings.warn("lxml cannot represent empty doctype", DataLossWarning)
|
|
|
|
self.doctype = None
|
|
|
|
else:
|
|
|
|
coercedName = self.infosetFilter.coerceElement(name)
|
|
|
|
if coercedName != name:
|
|
|
|
warnings.warn("lxml cannot represent non-xml doctype", DataLossWarning)
|
|
|
|
|
|
|
|
doctype = self.doctypeClass(coercedName, publicId, systemId)
|
|
|
|
self.doctype = doctype
|
2014-03-10 01:18:05 -04:00
|
|
|
|
|
|
|
def insertCommentInitial(self, data, parent=None):
|
|
|
|
self.initial_comments.append(data)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
|
|
|
def insertCommentMain(self, data, parent=None):
|
|
|
|
if (parent == self.document and
|
|
|
|
self.document._elementTree.getroot()[-1].tag == comment_type):
|
|
|
|
warnings.warn("lxml cannot represent adjacent comments beyond the root elements", DataLossWarning)
|
|
|
|
super(TreeBuilder, self).insertComment(data, parent)
|
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
def insertRoot(self, token):
|
|
|
|
"""Create the document root"""
|
2014-07-21 19:01:46 -04:00
|
|
|
# Because of the way libxml2 works, it doesn't seem to be possible to
|
|
|
|
# alter information like the doctype after the tree has been parsed.
|
|
|
|
# Therefore we need to use the built-in parser to create our iniial
|
|
|
|
# tree, after which we can add elements like normal
|
2014-03-10 01:18:05 -04:00
|
|
|
docStr = ""
|
2014-07-21 19:01:46 -04:00
|
|
|
if self.doctype:
|
|
|
|
assert self.doctype.name
|
|
|
|
docStr += "<!DOCTYPE %s" % self.doctype.name
|
|
|
|
if (self.doctype.publicId is not None or
|
|
|
|
self.doctype.systemId is not None):
|
|
|
|
docStr += (' PUBLIC "%s" ' %
|
|
|
|
(self.infosetFilter.coercePubid(self.doctype.publicId or "")))
|
|
|
|
if self.doctype.systemId:
|
|
|
|
sysid = self.doctype.systemId
|
|
|
|
if sysid.find("'") >= 0 and sysid.find('"') >= 0:
|
|
|
|
warnings.warn("DOCTYPE system cannot contain single and double quotes", DataLossWarning)
|
|
|
|
sysid = sysid.replace("'", 'U00027')
|
|
|
|
if sysid.find("'") >= 0:
|
|
|
|
docStr += '"%s"' % sysid
|
|
|
|
else:
|
|
|
|
docStr += "'%s'" % sysid
|
|
|
|
else:
|
|
|
|
docStr += "''"
|
2014-03-10 01:18:05 -04:00
|
|
|
docStr += ">"
|
2014-07-21 19:01:46 -04:00
|
|
|
if self.doctype.name != token["name"]:
|
|
|
|
warnings.warn("lxml cannot represent doctype with a different name to the root element", DataLossWarning)
|
2014-03-10 01:18:05 -04:00
|
|
|
docStr += "<THIS_SHOULD_NEVER_APPEAR_PUBLICLY/>"
|
2014-07-21 19:01:46 -04:00
|
|
|
root = etree.fromstring(docStr)
|
|
|
|
|
|
|
|
# Append the initial comments:
|
2014-03-10 01:18:05 -04:00
|
|
|
for comment_token in self.initial_comments:
|
|
|
|
root.addprevious(etree.Comment(comment_token["data"]))
|
2014-07-21 19:01:46 -04:00
|
|
|
|
|
|
|
# Create the root document and add the ElementTree to it
|
2014-03-10 01:18:05 -04:00
|
|
|
self.document = self.documentClass()
|
|
|
|
self.document._elementTree = root.getroottree()
|
2014-07-21 19:01:46 -04:00
|
|
|
|
2014-03-10 01:18:05 -04:00
|
|
|
# Give the root element the right name
|
|
|
|
name = token["name"]
|
|
|
|
namespace = token.get("namespace", self.defaultNamespace)
|
|
|
|
if namespace is None:
|
|
|
|
etree_tag = name
|
|
|
|
else:
|
2014-07-21 19:01:46 -04:00
|
|
|
etree_tag = "{%s}%s" % (namespace, name)
|
2014-03-10 01:18:05 -04:00
|
|
|
root.tag = etree_tag
|
2014-07-21 19:01:46 -04:00
|
|
|
|
|
|
|
# Add the root element to the internal child/open data structures
|
2014-03-10 01:18:05 -04:00
|
|
|
root_element = self.elementClass(name, namespace)
|
|
|
|
root_element._element = root
|
|
|
|
self.document._childNodes.append(root_element)
|
|
|
|
self.openElements.append(root_element)
|
2014-07-21 19:01:46 -04:00
|
|
|
|
|
|
|
# Reset to the default insert comment function
|
|
|
|
self.insertComment = self.insertCommentMain
|