Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions Doc/library/xml.dom.minidom.rst
Original file line number Diff line number Diff line change
Expand Up @@ -187,13 +187,23 @@ module documentation. This section lists the differences between the API and

The *standalone* argument behaves exactly as in :meth:`writexml`.

No indentation is added inside an element
which is marked with ``xml:space="preserve"``,
which is declared in the DTD as not having element content,
or, in absence of such declaration, which contains text,
because this would change its content.

.. versionchanged:: 3.8
The :meth:`toprettyxml` method now preserves the attribute order specified
by the user.

.. versionchanged:: 3.9
The *standalone* parameter was added.

.. versionchanged:: next
Whitespace is no longer added inside an element with mixed content
or marked with ``xml:space="preserve"``.

.. _dom-example:

DOM Example
Expand Down
8 changes: 8 additions & 0 deletions Doc/library/xml.etree.elementtree.rst
Original file line number Diff line number Diff line change
Expand Up @@ -603,8 +603,16 @@ Functions
characters by default. For indenting partial subtrees inside of an
already indented tree, pass the initial indentation level as *level*.

No whitespace is added inside an element
which is marked with ``xml:space="preserve"``
or which contains text, because this would change its content.

.. versionadded:: 3.9

.. versionchanged:: next
Whitespace is no longer added inside an element with mixed content
or marked with ``xml:space="preserve"``.


.. function:: iselement(element)

Expand Down
17 changes: 17 additions & 0 deletions Doc/whatsnew/3.16.rst
Original file line number Diff line number Diff line change
Expand Up @@ -642,6 +642,14 @@ xml
and :meth:`!Document.createEntityReference`.
(Contributed by Jason Orendorff and Serhiy Storchaka in :gh:`44871`.)

* :meth:`~xml.dom.minidom.Node.toprettyxml` in :mod:`xml.dom.minidom`
and :func:`~xml.etree.ElementTree.indent` in :mod:`xml.etree.ElementTree`
no longer add whitespace inside an element
which is marked with ``xml:space="preserve"`` or which contains text.
:meth:`!toprettyxml` also takes into account
the content model declared in the DTD.
(Contributed by Serhiy Storchaka in :gh:`81623`.)

* Add :meth:`!GetSpecifiedAttributeCount` method
to the :mod:`XML parser <xml.parsers.expat>` objects.
It tells how many of the reported attributes were given in the start tag
Expand Down Expand Up @@ -881,6 +889,15 @@ that may require changes to your code.
Attributes defaulted in the DTD are no longer omitted when parsing.
(Contributed by Jason Orendorff and Serhiy Storchaka in :gh:`44871`.)

* :meth:`~xml.dom.minidom.Node.toprettyxml` in :mod:`xml.dom.minidom`
and :func:`~xml.etree.ElementTree.indent` in :mod:`xml.etree.ElementTree`
no longer add whitespace inside an element
which is marked with ``xml:space="preserve"`` or which contains text,
because this changed the content of the element.
:meth:`!toprettyxml` also takes into account
the content model declared in the DTD.
(Contributed by Serhiy Storchaka in :gh:`81623`.)

* On Windows, seeking a pipe now fails instead of silently appearing to
succeed: :func:`os.lseek` and :meth:`~io.IOBase.seek` raise :exc:`OSError`,
and :meth:`~io.IOBase.seekable` returns ``False``. As a consequence,
Expand Down
84 changes: 76 additions & 8 deletions Lib/test/test_minidom.py
Original file line number Diff line number Diff line change
Expand Up @@ -609,39 +609,107 @@ def testAltNewline(self):
self.assertEqual(domstr, str.replace("\n", "\r\n"))

def test_toprettyxml_with_text_nodes(self):
# see issue #4147, text nodes are not indented
# see gh-48397 and gh-81623,
# the content of an element with text is not changed
decl = '<?xml version="1.0" ?>\n'
self.assertEqual(parseString('<B>A</B>').toprettyxml(),
decl + '<B>A</B>\n')
self.assertEqual(parseString('<C>A<B>A</B></C>').toprettyxml(),
decl + '<C>\n\tA\n\t<B>A</B>\n</C>\n')
decl + '<C>A<B>A</B></C>\n')
self.assertEqual(parseString('<C><B>A</B>A</C>').toprettyxml(),
decl + '<C>\n\t<B>A</B>\n\tA\n</C>\n')
decl + '<C><B>A</B>A</C>\n')
self.assertEqual(parseString('<C><B>A</B><B>A</B></C>').toprettyxml(),
decl + '<C>\n\t<B>A</B>\n\t<B>A</B>\n</C>\n')
self.assertEqual(parseString('<C><B>A</B>A<B>A</B></C>').toprettyxml(),
decl + '<C>\n\t<B>A</B>\n\tA\n\t<B>A</B>\n</C>\n')
decl + '<C><B>A</B>A<B>A</B></C>\n')
# toprettyxml treats whitespace between elements as insignificant
self.assertEqual(parseString('<C> <B>A</B> </C>').toprettyxml(),
decl + '<C>\n\t \n\t<B>A</B>\n\t \n</C>\n')

def test_toprettyxml_with_adjacent_text_nodes(self):
# see issue #4147, adjacent text nodes are indented normally
# see gh-81623, adjacent text nodes are not separated
dom = Document()
elem = dom.createElement('elem')
elem.appendChild(dom.createTextNode('TEXT'))
elem.appendChild(dom.createTextNode('TEXT'))
dom.appendChild(elem)
decl = '<?xml version="1.0" ?>\n'
self.assertEqual(dom.toprettyxml(),
decl + '<elem>\n\tTEXT\n\tTEXT\n</elem>\n')
self.assertEqual(dom.toprettyxml(), decl + '<elem>TEXTTEXT</elem>\n')

def test_toprettyxml_preserve(self):
decl = '<?xml version="1.0" ?>\n'
# xml:space="preserve" applies to the whole subtree
self.assertEqual(
parseString('<C xml:space="preserve"><B>A</B><B>A</B></C>'
).toprettyxml(),
decl + '<C xml:space="preserve"><B>A</B><B>A</B></C>\n')
self.assertEqual(
parseString('<C xml:space="preserve"><B><D/></B></C>'
).toprettyxml(),
decl + '<C xml:space="preserve"><B><D/></B></C>\n')
# other values do not preserve whitespace
self.assertEqual(
parseString('<C xml:space="default"><B>A</B></C>').toprettyxml(),
decl + '<C xml:space="default">\n\t<B>A</B>\n</C>\n')

def test_toprettyxml_with_non_xml_whitespace(self):
# only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
decl = '<?xml version="1.0" ?>\n'
self.assertEqual(parseString('<C>\xa0<B>A</B></C>').toprettyxml(),
decl + '<C>\xa0<B>A</B></C>\n')

def test_toprettyxml_with_dtd(self):
decl = '<?xml version="1.0" ?>\n'
# only whitespace in element content is ignorable
doctype = ('<!DOCTYPE C [<!ELEMENT C (#PCDATA|B)*>'
'<!ELEMENT B (#PCDATA)>]>')
self.assertEqual(
parseString(doctype + '<C><B>A</B><B>A</B></C>').toprettyxml(),
decl + doctype + '\n<C><B>A</B><B>A</B></C>\n')
doctype = '<!DOCTYPE C [<!ELEMENT C (B)*><!ELEMENT B (#PCDATA)>]>'
self.assertEqual(
parseString(doctype + '<C><B>A</B><B>A</B></C>').toprettyxml(),
decl + doctype + '\n<C>\n\t<B>A</B>\n\t<B>A</B>\n</C>\n')

def test_toprettyxml_with_cdata_section(self):
decl = '<?xml version="1.0" ?>\n'
self.assertEqual(
parseString('<C><![CDATA[A]]><B>A</B></C>').toprettyxml(),
decl + '<C><![CDATA[A]]><B>A</B></C>\n')

def test_toprettyxml_preserves_content_of_text_node(self):
# see issue #4147
# see gh-48397
for str in ('<B>A</B>', '<A><B>C</B></A>'):
dom = parseString(str)
dom2 = parseString(dom.toprettyxml())
self.assertEqual(
dom.getElementsByTagName('B')[0].childNodes[0].toxml(),
dom2.getElementsByTagName('B')[0].childNodes[0].toxml())

def test_isWhitespaceInElementContent(self):
# only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
dom = parseString('<!DOCTYPE a [<!ELEMENT a (b)*><!ELEMENT b (#PCDATA)>]>'
'<a> <b>x</b>\xa0</a>')
children = dom.documentElement.childNodes
self.assertTrue(children[0].isWhitespaceInElementContent)
self.assertFalse(children[2].isWhitespaceInElementContent)
dom.unlink()

def test_remove_whitespace_in_element_content(self):
from xml.dom.xmlbuilder import DOMBuilder, DOMInputSource
builder = DOMBuilder()
builder.setFeature("whitespace-in-element-content", False)
source = DOMInputSource()
source.byteStream = io.BytesIO(
b'<!DOCTYPE a [<!ELEMENT a (b)*><!ELEMENT b (#PCDATA)>]>'
b'<a> <b>x</b>\xc2\xa0</a>')
dom = builder.parse(source)
children = dom.documentElement.childNodes
# ignorable whitespace is removed, other characters are not
self.assertEqual([node.nodeName for node in children], ['b', '#text'])
self.assertEqual(children[1].data, '\xa0')
dom.unlink()

def testProcessingInstruction(self):
dom = parseString('<e><?mypi \t\n data \t\n ?></e>')
pi = dom.documentElement.firstChild
Expand Down
47 changes: 46 additions & 1 deletion Lib/test/test_xml_etree.py
Original file line number Diff line number Diff line change
Expand Up @@ -773,9 +773,10 @@ def test_indent(self):
ET.indent(elem)
self.assertEqual(ET.tostring(elem), b'<html>\n <body>text</body>\n</html>')

# an element with mixed content is not indented
elem = ET.XML("<html><body>text</body>tail</html>")
ET.indent(elem)
self.assertEqual(ET.tostring(elem), b'<html>\n <body>text</body>tail</html>')
self.assertEqual(ET.tostring(elem), b'<html><body>text</body>tail</html>')

elem = ET.XML("<html><body><p>par</p>\n<p>text</p>\t<p><br/></p></body></html>")
ET.indent(elem)
Expand Down Expand Up @@ -845,6 +846,45 @@ def test_indent_space_caching(self):
len({id(el.tail) for el in elem.iter()}),
)

def test_indent_non_xml_whitespace(self):
# only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
elem = ET.XML('<html>\xa0<body><p>text</p>\xa0</body></html>')
ET.indent(elem)
self.assertEqual(
ET.tostring(elem),
b'<html>&#160;<body><p>text</p>&#160;</body></html>'
)

def test_indent_preserve(self):
# xml:space="preserve" applies to the whole subtree
elem = ET.XML('<html xml:space="preserve"> <body><p>text</p></body> </html>')
ET.indent(elem)
self.assertEqual(
ET.tostring(elem),
b'<html xml:space="preserve"> <body><p>text</p></body> </html>'
)
# other values do not preserve whitespace
elem = ET.XML('<html xml:space="default"><body><p>text</p></body></html>')
ET.indent(elem)
self.assertEqual(
ET.tostring(elem),
b'<html xml:space="default">\n'
b' <body>\n'
b' <p>text</p>\n'
b' </body>\n'
b'</html>'
)

def test_indent_mixed_content(self):
# whitespace in an element which contains text is significant
elem = ET.XML('<p>hello <b>x</b> <i>y</i></p>')
ET.indent(elem)
self.assertEqual(ET.tostring(elem), b'<p>hello <b>x</b> <i>y</i></p>')
# the subtree of such element is not indented either
elem = ET.XML('<p>hello <b><i>y</i></b></p>')
ET.indent(elem)
self.assertEqual(ET.tostring(elem), b'<p>hello <b><i>y</i></b></p>')

def test_indent_level(self):
elem = ET.XML("<html><body><p>pre<br/>post</p><p>text</p></body></html>")
with self.assertRaises(ValueError):
Expand Down Expand Up @@ -4755,6 +4795,11 @@ def test_simple_roundtrip(self):
xml = '<X xmlns="http://nps/a"><Y xmlns:b="http://nsp/b" b:targets="abc,xyz"></Y></X>'
self.assertEqual(c14n_roundtrip(xml), xml)

def test_c14n_strip_non_xml_whitespace(self):
# only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
self.assertEqual(c14n_roundtrip("<a> \xa0x\xa0 </a>", strip_text=True),
"<a>\xa0x\xa0</a>")

def test_c14n_exclusion(self):
xml = textwrap.dedent("""\
<root xmlns:x="http://example.com/x">
Expand Down
6 changes: 4 additions & 2 deletions Lib/xml/dom/expatbuilder.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,8 @@
from xml.dom import xmlbuilder, minidom, Node
from xml.dom import EMPTY_NAMESPACE, EMPTY_PREFIX, XMLNS_NAMESPACE
from xml.parsers import expat
from xml.dom.minidom import _append_child, _set_attribute_node
from xml.dom.minidom import (_append_child, _set_attribute_node,
_XML_WHITESPACE)
from xml.dom.NodeFilter import NodeFilter

TEXT_NODE = Node.TEXT_NODE
Expand Down Expand Up @@ -413,7 +414,8 @@ def _handle_white_text_nodes(self, node, info):
# whitespace.
L = []
for child in node.childNodes:
if child.nodeType == TEXT_NODE and not child.data.strip():
if (child.nodeType == TEXT_NODE
and not child.data.strip(_XML_WHITESPACE)):
L.append(child)

# Remove ignorable whitespace from the tree.
Expand Down
28 changes: 27 additions & 1 deletion Lib/xml/dom/minidom.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,9 @@
_nodeTypes_with_children = (xml.dom.Node.ELEMENT_NODE,
xml.dom.Node.ENTITY_REFERENCE_NODE)

# The white space characters of the XML specification (see XML 1.0, 2.3).
_XML_WHITESPACE = " \t\r\n"


class Node(xml.dom.Node):
namespaceURI = None # this is non-null only for elements and attributes
Expand Down Expand Up @@ -937,6 +940,10 @@ def writexml(self, writer, indent="", addindent="", newl=""):
self.childNodes[0].nodeType in (
Node.TEXT_NODE, Node.CDATA_SECTION_NODE)):
self.childNodes[0].writexml(writer, '', '', '')
elif self._preserves_whitespace():
# Adding whitespace here would change the content.
for node in self.childNodes:
node.writexml(writer, '', '', '')
else:
writer.write(newl)
for node in self.childNodes:
Expand All @@ -946,6 +953,25 @@ def writexml(self, writer, indent="", addindent="", newl=""):
else:
writer.write("/>%s"%(newl))

def _preserves_whitespace(self):
"""Returns true iff whitespace in the content is significant.

This is the case if the element is marked with xml:space="preserve",
if the DTD declares that its content model is not element content,
or, in absence of such declaration, if it contains text.
"""
if self.getAttribute("xml:space") == "preserve":
return True
doc = self.ownerDocument
info = doc and doc._get_elem_info(self)
if info is not None:
# Only whitespace in element content is ignorable
# (see XML 1.0, 3.2.1).
return not info.isElementContent()
return any(node.nodeType in (Node.TEXT_NODE, Node.CDATA_SECTION_NODE)
and node.data.strip(_XML_WHITESPACE)
for node in self.childNodes)

def _get_attributes(self):
self._ensure_attributes()
return NamedNodeMap(self._attrs, self._attrsNS, self)
Expand Down Expand Up @@ -1209,7 +1235,7 @@ def replaceWholeText(self, content):
return None

def _get_isWhitespaceInElementContent(self):
if self.data.strip():
if self.data.strip(_XML_WHITESPACE):
return False
elem = _get_containing_element(self)
if elem is None:
Expand Down
Loading
Loading