diff --git a/Lib/test/test_minidom.py b/Lib/test/test_minidom.py
index 3735a6046891ea9..e204bdc7dc672db 100644
--- a/Lib/test/test_minidom.py
+++ b/Lib/test/test_minidom.py
@@ -642,6 +642,30 @@ def test_toprettyxml_preserves_content_of_text_node(self):
dom.getElementsByTagName('B')[0].childNodes[0].toxml(),
dom2.getElementsByTagName('B')[0].childNodes[0].toxml())
+ def test_isWhitespaceInElementContent(self):
+ # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
+ dom = parseString(']>'
+ ' x\xa0')
+ children = dom.documentElement.childNodes
+ self.assertTrue(children[0].isWhitespaceInElementContent)
+ self.assertFalse(children[2].isWhitespaceInElementContent)
+ dom.unlink()
+
+ def test_remove_whitespace_in_element_content(self):
+ from xml.dom.xmlbuilder import DOMBuilder, DOMInputSource
+ builder = DOMBuilder()
+ builder.setFeature("whitespace-in-element-content", False)
+ source = DOMInputSource()
+ source.byteStream = io.BytesIO(
+ b']>'
+ b' x\xc2\xa0')
+ dom = builder.parse(source)
+ children = dom.documentElement.childNodes
+ # ignorable whitespace is removed, other characters are not
+ self.assertEqual([node.nodeName for node in children], ['b', '#text'])
+ self.assertEqual(children[1].data, '\xa0')
+ dom.unlink()
+
def testProcessingInstruction(self):
dom = parseString('')
pi = dom.documentElement.firstChild
diff --git a/Lib/test/test_xml_etree.py b/Lib/test/test_xml_etree.py
index 2af2d1fd64520b1..f87565a24ee4124 100644
--- a/Lib/test/test_xml_etree.py
+++ b/Lib/test/test_xml_etree.py
@@ -845,6 +845,15 @@ def test_indent_space_caching(self):
len({id(el.tail) for el in elem.iter()}),
)
+ def test_indent_non_xml_whitespace(self):
+ # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
+ elem = ET.XML('\xa0
text
\xa0')
+ ET.indent(elem)
+ self.assertEqual(
+ ET.tostring(elem),
+ b' \n text
\n'
+ )
+
def test_indent_level(self):
elem = ET.XML("pre
post
text
")
with self.assertRaises(ValueError):
@@ -4755,6 +4764,11 @@ def test_simple_roundtrip(self):
xml = ''
self.assertEqual(c14n_roundtrip(xml), xml)
+ def test_c14n_strip_non_xml_whitespace(self):
+ # only " \t\r\n" are whitespace in XML (see XML 1.0, 2.3)
+ self.assertEqual(c14n_roundtrip(" \xa0x\xa0 ", strip_text=True),
+ "\xa0x\xa0")
+
def test_c14n_exclusion(self):
xml = textwrap.dedent("""\
diff --git a/Lib/xml/dom/expatbuilder.py b/Lib/xml/dom/expatbuilder.py
index d56b2ddfdb25698..e3917c1cc682880 100644
--- a/Lib/xml/dom/expatbuilder.py
+++ b/Lib/xml/dom/expatbuilder.py
@@ -30,7 +30,8 @@
from xml.dom import xmlbuilder, minidom, Node
from xml.dom import EMPTY_NAMESPACE, EMPTY_PREFIX, XMLNS_NAMESPACE
from xml.parsers import expat
-from xml.dom.minidom import _append_child, _set_attribute_node
+from xml.dom.minidom import (_append_child, _set_attribute_node,
+ _XML_WHITESPACE)
from xml.dom.NodeFilter import NodeFilter
TEXT_NODE = Node.TEXT_NODE
@@ -413,7 +414,8 @@ def _handle_white_text_nodes(self, node, info):
# whitespace.
L = []
for child in node.childNodes:
- if child.nodeType == TEXT_NODE and not child.data.strip():
+ if (child.nodeType == TEXT_NODE
+ and not child.data.strip(_XML_WHITESPACE)):
L.append(child)
# Remove ignorable whitespace from the tree.
diff --git a/Lib/xml/dom/minidom.py b/Lib/xml/dom/minidom.py
index 5fd3911bd3c9ebb..7639fa14c5050fb 100644
--- a/Lib/xml/dom/minidom.py
+++ b/Lib/xml/dom/minidom.py
@@ -31,6 +31,9 @@
_nodeTypes_with_children = (xml.dom.Node.ELEMENT_NODE,
xml.dom.Node.ENTITY_REFERENCE_NODE)
+# The white space characters of the XML specification (see XML 1.0, 2.3).
+_XML_WHITESPACE = " \t\r\n"
+
class Node(xml.dom.Node):
namespaceURI = None # this is non-null only for elements and attributes
@@ -1209,7 +1212,7 @@ def replaceWholeText(self, content):
return None
def _get_isWhitespaceInElementContent(self):
- if self.data.strip():
+ if self.data.strip(_XML_WHITESPACE):
return False
elem = _get_containing_element(self)
if elem is None:
diff --git a/Lib/xml/etree/ElementTree.py b/Lib/xml/etree/ElementTree.py
index 951540eb9f45e90..93b8eceb1571614 100644
--- a/Lib/xml/etree/ElementTree.py
+++ b/Lib/xml/etree/ElementTree.py
@@ -101,6 +101,9 @@
from . import ElementPath
+# The white space characters of the XML specification (see XML 1.0, 2.3).
+_XML_WHITESPACE = " \t\r\n"
+
class ParseError(SyntaxError):
"""An error when parsing an XML document.
@@ -1190,17 +1193,17 @@ def _indent_children(elem, level):
child_indentation = indentations[level] + space
indentations.append(child_indentation)
- if not elem.text or not elem.text.strip():
+ if not elem.text or not elem.text.strip(_XML_WHITESPACE):
elem.text = child_indentation
for child in elem:
if len(child):
_indent_children(child, child_level)
- if not child.tail or not child.tail.strip():
+ if not child.tail or not child.tail.strip(_XML_WHITESPACE):
child.tail = child_indentation
# Dedent after the last child by overwriting the previous indentation.
- if not child.tail.strip():
+ if not child.tail.strip(_XML_WHITESPACE):
child.tail = indentations[level]
_indent_children(tree, 0)
@@ -1705,7 +1708,7 @@ def _default(self, text):
if prefix == ">":
self._doctype = None
return
- text = text.strip()
+ text = text.strip(_XML_WHITESPACE)
if not text:
return
self._doctype.append(text)
@@ -1921,7 +1924,7 @@ def _flush(self, _join_text=''.join):
data = _join_text(self._data)
del self._data[:]
if self._strip_text and not self._preserve_space[-1]:
- data = data.strip()
+ data = data.strip(_XML_WHITESPACE)
if self._pending_start is not None:
args, self._pending_start = self._pending_start, None
qname_text = data if data and _looks_like_prefix_name(data) else None
diff --git a/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst b/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst
new file mode 100644
index 000000000000000..3f2f105116c1838
--- /dev/null
+++ b/Misc/NEWS.d/next/Library/2026-08-30-19-30-00.gh-issue-156658.Xq5Nt7.rst
@@ -0,0 +1,5 @@
+:mod:`xml.dom` and :mod:`xml.etree.ElementTree` no longer treat characters
+which are not white space in XML (such as U+00A0) as white space. Previously
+they could be lost in :func:`~xml.etree.ElementTree.indent`,
+:func:`~xml.etree.ElementTree.canonicalize` with ``strip_text=True``, and when
+parsing with the ``whitespace-in-element-content`` feature turned off.