From b944534c46e17da2f4feed3db1b49e44f1f12345 Mon Sep 17 00:00:00 2001 From: Serhiy Storchaka Date: Sun, 30 Aug 2026 16:54:39 +0300 Subject: [PATCH] gh-156658: Only XML white space characters are treated as white space XML defines white space as " \t\r\n" (see XML 1.0, 2.3), but str.strip() also strips other characters, such as U+00A0. Such characters could be lost in ElementTree.indent(), in canonicalize(strip_text=True), and when parsing with the whitespace-in-element-content feature turned off. --- Lib/test/test_minidom.py | 24 +++++++++++++++++++ Lib/test/test_xml_etree.py | 14 +++++++++++ Lib/xml/dom/expatbuilder.py | 6 +++-- Lib/xml/dom/minidom.py | 5 +++- Lib/xml/etree/ElementTree.py | 13 ++++++---- ...-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst | 5 ++++ 6 files changed, 59 insertions(+), 8 deletions(-) create mode 100644 Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst diff --git a/Lib/test/test_minidom.py b/Lib/test/test_minidom.py index 3735a6046891ea9..e204bdc7dc672db 100644 --- a/Lib/test/test_minidom.py +++ b/Lib/test/test_minidom.py @@ -642,6 +642,30 @@ def test_toprettyxml_preserves_content_of_text_node(self): dom.getElementsByTagName('B')[0].childNodes[0].toxml(), dom2.getElementsByTagName('B')[0].childNodes[0].toxml()) + def test_isWhitespaceInElementContent(self): + # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3) + dom = parseString(']>' + ' x\xa0') + children = dom.documentElement.childNodes + self.assertTrue(children[0].isWhitespaceInElementContent) + self.assertFalse(children[2].isWhitespaceInElementContent) + dom.unlink() + + def test_remove_whitespace_in_element_content(self): + from xml.dom.xmlbuilder import DOMBuilder, DOMInputSource + builder = DOMBuilder() + builder.setFeature("whitespace-in-element-content", False) + source = DOMInputSource() + source.byteStream = io.BytesIO( + b']>' + b' x\xc2\xa0') + dom = builder.parse(source) + children = dom.documentElement.childNodes + # ignorable whitespace is removed, other characters are not + self.assertEqual([node.nodeName for node in children], ['b', '#text']) + self.assertEqual(children[1].data, '\xa0') + dom.unlink() + def testProcessingInstruction(self): dom = parseString('') pi = dom.documentElement.firstChild diff --git a/Lib/test/test_xml_etree.py b/Lib/test/test_xml_etree.py index 2af2d1fd64520b1..f87565a24ee4124 100644 --- a/Lib/test/test_xml_etree.py +++ b/Lib/test/test_xml_etree.py @@ -845,6 +845,15 @@ def test_indent_space_caching(self): len({id(el.tail) for el in elem.iter()}), ) + def test_indent_non_xml_whitespace(self): + # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3) + elem = ET.XML('\xa0

text

\xa0') + ET.indent(elem) + self.assertEqual( + ET.tostring(elem), + b' \n

text

 \n' + ) + def test_indent_level(self): elem = ET.XML("

pre
post

text

") with self.assertRaises(ValueError): @@ -4755,6 +4764,11 @@ def test_simple_roundtrip(self): xml = '' self.assertEqual(c14n_roundtrip(xml), xml) + def test_c14n_strip_non_xml_whitespace(self): + # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3) + self.assertEqual(c14n_roundtrip(" \xa0x\xa0 ", strip_text=True), + "\xa0x\xa0") + def test_c14n_exclusion(self): xml = textwrap.dedent("""\ diff --git a/Lib/xml/dom/expatbuilder.py b/Lib/xml/dom/expatbuilder.py index d56b2ddfdb25698..e3917c1cc682880 100644 --- a/Lib/xml/dom/expatbuilder.py +++ b/Lib/xml/dom/expatbuilder.py @@ -30,7 +30,8 @@ from xml.dom import xmlbuilder, minidom, Node from xml.dom import EMPTY_NAMESPACE, EMPTY_PREFIX, XMLNS_NAMESPACE from xml.parsers import expat -from xml.dom.minidom import _append_child, _set_attribute_node +from xml.dom.minidom import (_append_child, _set_attribute_node, + _XML_WHITESPACE) from xml.dom.NodeFilter import NodeFilter TEXT_NODE = Node.TEXT_NODE @@ -413,7 +414,8 @@ def _handle_white_text_nodes(self, node, info): # whitespace. L = [] for child in node.childNodes: - if child.nodeType == TEXT_NODE and not child.data.strip(): + if (child.nodeType == TEXT_NODE + and not child.data.strip(_XML_WHITESPACE)): L.append(child) # Remove ignorable whitespace from the tree. diff --git a/Lib/xml/dom/minidom.py b/Lib/xml/dom/minidom.py index 5fd3911bd3c9ebb..7639fa14c5050fb 100644 --- a/Lib/xml/dom/minidom.py +++ b/Lib/xml/dom/minidom.py @@ -31,6 +31,9 @@ _nodeTypes_with_children = (xml.dom.Node.ELEMENT_NODE, xml.dom.Node.ENTITY_REFERENCE_NODE) +# The white space characters of the XML specification (see XML 1.0, 2.3). +_XML_WHITESPACE = " \t\r\n" + class Node(xml.dom.Node): namespaceURI = None # this is non-null only for elements and attributes @@ -1209,7 +1212,7 @@ def replaceWholeText(self, content): return None def _get_isWhitespaceInElementContent(self): - if self.data.strip(): + if self.data.strip(_XML_WHITESPACE): return False elem = _get_containing_element(self) if elem is None: diff --git a/Lib/xml/etree/ElementTree.py b/Lib/xml/etree/ElementTree.py index 951540eb9f45e90..93b8eceb1571614 100644 --- a/Lib/xml/etree/ElementTree.py +++ b/Lib/xml/etree/ElementTree.py @@ -101,6 +101,9 @@ from . import ElementPath +# The white space characters of the XML specification (see XML 1.0, 2.3). +_XML_WHITESPACE = " \t\r\n" + class ParseError(SyntaxError): """An error when parsing an XML document. @@ -1190,17 +1193,17 @@ def _indent_children(elem, level): child_indentation = indentations[level] + space indentations.append(child_indentation) - if not elem.text or not elem.text.strip(): + if not elem.text or not elem.text.strip(_XML_WHITESPACE): elem.text = child_indentation for child in elem: if len(child): _indent_children(child, child_level) - if not child.tail or not child.tail.strip(): + if not child.tail or not child.tail.strip(_XML_WHITESPACE): child.tail = child_indentation # Dedent after the last child by overwriting the previous indentation. - if not child.tail.strip(): + if not child.tail.strip(_XML_WHITESPACE): child.tail = indentations[level] _indent_children(tree, 0) @@ -1705,7 +1708,7 @@ def _default(self, text): if prefix == ">": self._doctype = None return - text = text.strip() + text = text.strip(_XML_WHITESPACE) if not text: return self._doctype.append(text) @@ -1921,7 +1924,7 @@ def _flush(self, _join_text=''.join): data = _join_text(self._data) del self._data[:] if self._strip_text and not self._preserve_space[-1]: - data = data.strip() + data = data.strip(_XML_WHITESPACE) if self._pending_start is not None: args, self._pending_start = self._pending_start, None qname_text = data if data and _looks_like_prefix_name(data) else None diff --git a/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst b/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst new file mode 100644 index 000000000000000..3f2f105116c1838 --- /dev/null +++ b/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst @@ -0,0 +1,5 @@ +:mod:`xml.dom` and :mod:`xml.etree.ElementTree` no longer treat characters +which are not white space in XML (such as U+00A0) as white space. Previously +they could be lost in :func:`~xml.etree.ElementTree.indent`, +:func:`~xml.etree.ElementTree.canonicalize` with ``strip_text=True``, and when +parsing with the ``whitespace-in-element-content`` feature turned off.