https://github.com/python/cpython/commit/b1e7554ac1e4268eddbb6d9dd7fad512c44d9256
commit: b1e7554ac1e4268eddbb6d9dd7fad512c44d9256
branch: main
author: Serhiy Storchaka <[email protected]>
committer: serhiy-storchaka <[email protected]>
date: 2026-09-12T19:05:42Z
summary:

gh-157366: Remove leftover bytes support from _elementtree (GH-157367)

It was the result of improper translation of the PyString C API used in
Python 2 to PyBytes.  The C implementation now raises the same exceptions
as the Python implementation for bytes paths, tags and event names.

files:
A Misc/NEWS.d/next/Library/2026-09-12-15-50-23.gh-issue-157366.djwD2M.rst
M Modules/_elementtree.c

diff --git 
a/Misc/NEWS.d/next/Library/2026-09-12-15-50-23.gh-issue-157366.djwD2M.rst 
b/Misc/NEWS.d/next/Library/2026-09-12-15-50-23.gh-issue-157366.djwD2M.rst
new file mode 100644
index 000000000000000..c29b132e4b3b198
--- /dev/null
+++ b/Misc/NEWS.d/next/Library/2026-09-12-15-50-23.gh-issue-157366.djwD2M.rst
@@ -0,0 +1,7 @@
+Remove the accidental acceptance of :class:`bytes` in the C implementation of
+:mod:`xml.etree.ElementTree`, a leftover of the Python 2 to Python 3 migration,
+for paths of :meth:`~xml.etree.ElementTree.Element.find`
+and similar methods, for the tag of 
:meth:`~xml.etree.ElementTree.Element.iter`,
+and for the names of events of :class:`~xml.etree.ElementTree.XMLPullParser`
+and :func:`~xml.etree.ElementTree.iterparse`.  It now raises the same
+exceptions as the Python implementation.
diff --git a/Modules/_elementtree.c b/Modules/_elementtree.c
index f6dedeed981c40d..6f51f10b2b22759 100644
--- a/Modules/_elementtree.c
+++ b/Modules/_elementtree.c
@@ -1230,24 +1230,6 @@ checkpath(PyObject* tag)
         }
         return 0;
     }
-    if (PyBytes_Check(tag)) {
-        const char *p = PyBytes_AS_STRING(tag);
-        const Py_ssize_t len = PyBytes_GET_SIZE(tag);
-        if (len >= 3 && p[0] == '{' && (
-                p[1] == '}' || (p[1] == '*' && p[2] == '}'))) {
-            /* wildcard: '{}tag' or '{*}tag' */
-            return 1;
-        }
-        for (i = 0; i < len; i++) {
-            if (p[i] == '{')
-                check = 0;
-            else if (p[i] == '}')
-                check = 1;
-            else if (check && PATHCHAR(p[i]))
-                return 1;
-        }
-        return 0;
-    }
 
     return 1; /* unknown type; might be path expression */
 }
@@ -1552,10 +1534,6 @@ _elementtree_Element_iter_impl(ElementObject *self, 
PyTypeObject *cls,
         if (PyUnicode_GET_LENGTH(tag) == 1 && PyUnicode_READ_CHAR(tag, 0) == 
'*')
             tag = Py_None;
     }
-    else if (PyBytes_Check(tag)) {
-        if (PyBytes_GET_SIZE(tag) == 1 && *PyBytes_AS_STRING(tag) == '*')
-            tag = Py_None;
-    }
 
     elementtreestate *st = get_elementtree_state_by_cls(cls);
     return create_elementiter(st, self, tag, 0);
@@ -2935,17 +2913,7 @@ treebuilder_handle_data(TreeBuilderObject* self, 
PyObject* data)
         self->data = Py_NewRef(data);
     } else {
         /* more than one item; use a list to collect items */
-        if (PyBytes_CheckExact(self->data)
-            && _PyObject_IsUniquelyReferenced(self->data)
-            && PyBytes_CheckExact(data) && PyBytes_GET_SIZE(data) == 1) {
-            /* XXX this code path unused in Python 3? */
-            /* expat often generates single character data sections; handle
-               the most common case by resizing the existing string... */
-            Py_ssize_t size = PyBytes_GET_SIZE(self->data);
-            if (_PyBytes_Resize(&self->data, size + 1) < 0)
-                return NULL;
-            PyBytes_AS_STRING(self->data)[size] = PyBytes_AS_STRING(data)[0];
-        } else if (PyList_CheckExact(self->data)) {
+        if (PyList_CheckExact(self->data)) {
             if (PyList_Append(self->data, data) < 0)
                 return NULL;
         } else {
@@ -4363,18 +4331,14 @@ _elementtree_XMLParser__setevents_impl(XMLParserObject 
*self,
 
     for (i = 0; i < PySequence_Fast_GET_SIZE(events_seq); ++i) {
         PyObject *event_name_obj = PySequence_Fast_GET_ITEM(events_seq, i);
-        const char *event_name = NULL;
-        if (PyUnicode_Check(event_name_obj)) {
-            event_name = PyUnicode_AsUTF8(event_name_obj);
-        } else if (PyBytes_Check(event_name_obj)) {
-            event_name = PyBytes_AS_STRING(event_name_obj);
+        if (!PyUnicode_Check(event_name_obj)) {
+            goto unknown_event;
         }
+        const char *event_name = PyUnicode_AsUTF8(event_name_obj);
         if (event_name == NULL) {
             Py_DECREF(events_seq);
-            PyErr_Format(PyExc_ValueError, "invalid events sequence");
             return NULL;
         }
-
         if (strcmp(event_name, "start") == 0) {
             Py_XSETREF(target->start_event_obj, Py_NewRef(event_name_obj));
         } else if (strcmp(event_name, "end") == 0) {
@@ -4406,7 +4370,8 @@ _elementtree_XMLParser__setevents_impl(XMLParserObject 
*self,
                 (XML_ProcessingInstructionHandler) expat_pi_handler
                 );
         } else {
-            PyErr_Format(PyExc_ValueError, "unknown event '%s'", event_name);
+unknown_event:
+            PyErr_Format(PyExc_ValueError, "unknown event %R", event_name_obj);
             Py_DECREF(events_seq);
             return NULL;
         }

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to