https://github.com/python/cpython/commit/f7586aa88dbaf3f4bd515e972b82f530820795a9
commit: f7586aa88dbaf3f4bd515e972b82f530820795a9
branch: main
author: Petr Viktorin <[email protected]>
committer: encukou <[email protected]>
date: 2026-10-02T15:11:09+02:00
summary:

gh-157675: Add *limit* argument to encodings.idna.nameprep (GH-157682)

Add a *limit* argument to nameprep to allow ToASCII and ToUnicode
to reject extremely large input early.
This replaces the check in #99092, while allowing any number of
harmless "characters mapped to nothing" (RFC 3454 ยง3.1).

The default stays unlimited, to not change behaviour for users
that call nameprep manually (and don't necessarily follow up with
punycode).

Co-authored-by: Stan Ulbrych <[email protected]>

files:
A Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst
M Doc/library/codecs.rst
M Doc/whatsnew/3.16.rst
M Lib/encodings/idna.py
M Lib/test/test_codecs.py

diff --git a/Doc/library/codecs.rst b/Doc/library/codecs.rst
index 311437a67f6e811..ed20a435faf4345 100644
--- a/Doc/library/codecs.rst
+++ b/Doc/library/codecs.rst
@@ -1395,15 +1395,6 @@ encodings.
 |                    |         | :mod:`encodings.idna`.    |
 |                    |         | Only ``errors='strict'``  |
 |                    |         | is supported.             |
-|                    |         |                           |
-|                    |         | .. warning::              |
-|                    |         |                           |
-|                    |         |    This codec builds on   |
-|                    |         |    ``punycode``, whose    |
-|                    |         |    algorithms scale       |
-|                    |         |    poorly, so limit the   |
-|                    |         |    length of untrusted    |
-|                    |         |    input.                 |
 +--------------------+---------+---------------------------+
 | mbcs               | ansi,   | Windows only: Encode the  |
 |                    | dbcs    | operand according to the  |
@@ -1655,11 +1646,6 @@ Applications) and :rfc:`3492` (Nameprep: A Stringprep 
Profile for
 Internationalized Domain Names (IDN)). It builds upon the ``punycode`` encoding
 and :mod:`stringprep`.
 
-.. warning::
-
-   This module builds on ``punycode``, whose algorithms scale poorly, so limit
-   the length of untrusted input.
-
 If you need the IDNA 2008 standard from :rfc:`5891` and :rfc:`5895`, use the
 third-party :pypi:`idna` module.
 
@@ -1697,11 +1683,31 @@ international domain names, and to unify similar 
characters. The nameprep
 functions can be used directly if desired.
 
 
-.. function:: nameprep(label)
+.. function:: nameprep(label, *, limit=None)
 
    Return the nameprepped version of *label*. The implementation currently 
assumes
    query strings, so ``AllowUnassigned`` is true.
 
+   Raise :exc:`UnicodeEncodeError` if the nameprep algorithm emits an error.
+
+   If the *limit* argument is given, it should be set to the maximum size
+   of an encoded A-label (that is, 63 for IDNA).
+   :func:`!nameprep` will raise :exc:`UnicodeEncodeError` if the label is
+   **much** larger than *limit*.
+   Note that this is only a rough check meant to skip expensive processing
+   of extremely large input; the caller should check any exact
+   limits separately.
+
+   .. warning::
+
+      For backwards compatibility, label size is unlimited by default.
+      This may cause issues when processing the result with the
+      ``punycode`` encoding, whose algorithms scale poorly.
+
+   .. versionchanged:: next
+
+      Added the *limit* parameter.
+
 
 .. function:: ToASCII(label)
 
diff --git a/Doc/whatsnew/3.16.rst b/Doc/whatsnew/3.16.rst
index 29bd7dcd749ebc2..16e03e9ef065321 100644
--- a/Doc/whatsnew/3.16.rst
+++ b/Doc/whatsnew/3.16.rst
@@ -397,6 +397,10 @@ encodings
   used for international IMAP4 mailbox names (:rfc:`3501`).
   (Contributed by Serhiy Storchaka in :gh:`66788`.)
 
+* :func:`encodings.idna.nameprep` now takes a *limit* argument that allows
+  rejecting extremely large input early.
+  (Contributed by Petr Viktorin in :gh:`157675`.)
+
 
 gzip
 ----
diff --git a/Lib/encodings/idna.py b/Lib/encodings/idna.py
index c896ffdeadfef7e..98d7d891594c648 100644
--- a/Lib/encodings/idna.py
+++ b/Lib/encodings/idna.py
@@ -1,5 +1,6 @@
 # This module implements the RFCs 3490 (IDNA) and 3491 (Nameprep)
 
+import sys
 import stringprep, re, codecs
 from unicodedata import ucd_3_2_0 as unicodedata
 
@@ -11,7 +12,24 @@
 sace_prefix = "xn--"
 
 # This assumes query strings, so AllowUnassigned is true
-def nameprep(label):  # type: (str) -> str
+def nameprep(label, *, limit=None):  # type: (str) -> str
+    if limit is None:
+        limit = sys.maxsize
+    else:
+        # Protection from gh-98433 and gh-157675 (passing unbounded input to
+        # the quadratic-complexity punycode algorithm).
+        # While the "map" step can remove characters, later steps (in ToASCII
+        # and FromASCII) will not shorten the result *drastically*.
+        # (NFKC normalization can compress e.g. '\u03c9\u0314\u0300\u0345'
+        # to '\u1fa3' -- a 4-fold reduction. Non-ASCII labels then get
+        # longer via prefixing & punycode).
+        # We bail if the number of non-ignored input characters exceeds 8 times
+        # the limit, which gives ample room for future Unicode versions to
+        # include long normalizations, while still preventing us from wasting
+        # time decoding a big thing  that'll just hit the actual <= 63 limit in
+        # ToASCII.
+        limit *= 8
+
     # Map
     newlabel = []
     for c in label:
@@ -19,6 +37,11 @@ def nameprep(label):  # type: (str) -> str
             # Map to nothing
             continue
         newlabel.append(stringprep.map_table_b2(c))
+
+        if len(newlabel) > limit:
+            raise UnicodeEncodeError("idna", label, 0, len(label),
+                                     "label way too long")
+
     label = "".join(newlabel)
 
     # Normalize
@@ -80,7 +103,7 @@ def ToASCII(label):  # type: (str) -> bytes
             raise UnicodeEncodeError("idna", label, 0, len(label), "label too 
long")
 
     # Step 2: nameprep
-    label = nameprep(label)
+    label = nameprep(label, limit=63)
 
     # Step 3: UseSTD3ASCIIRules is false
     # Step 4: try ASCII
@@ -115,18 +138,6 @@ def ToASCII(label):  # type: (str) -> bytes
     raise UnicodeEncodeError("idna", label, 0, len(label), "label too long")
 
 def ToUnicode(label):
-    if len(label) > 1024:
-        # Protection from https://github.com/python/cpython/issues/98433.
-        # https://datatracker.ietf.org/doc/html/rfc5894#section-6
-        # doesn't specify a label size limit prior to NAMEPREP. But having
-        # one makes practical sense.
-        # This leaves ample room for nameprep() to remove Nothing characters
-        # per https://www.rfc-editor.org/rfc/rfc3454#section-3.1 while still
-        # preventing us from wasting time decoding a big thing that'll just
-        # hit the actual <= 63 length limit in Step 6.
-        if isinstance(label, str):
-            label = label.encode("utf-8", errors="backslashreplace")
-        raise UnicodeDecodeError("idna", label, 0, len(label), "label way too 
long")
     # Step 1: Check for ASCII
     if isinstance(label, bytes):
         pure_ascii = True
@@ -139,7 +150,7 @@ def ToUnicode(label):
     if not pure_ascii:
         assert isinstance(label, str)
         # Step 2: Perform nameprep
-        label = nameprep(label)
+        label = nameprep(label, limit=63)
         # It doesn't say this, but apparently, it should be ASCII now
         try:
             label = label.encode("ascii")
@@ -151,6 +162,11 @@ def ToUnicode(label):
     if not label.lower().startswith(ace_prefix):
         return str(label, "ascii")
 
+    # Below in steps 6-7, `label` must match the result of `ToASCII`, so it's
+    # limited to 63 chars. Check before the expensive punycode decode.
+    if len(label) >= 64:
+        raise UnicodeDecodeError("idna", label, 0, len(label), "label too 
long")
+
     # Step 4: Remove ACE prefix
     label1 = label[len(ace_prefix):]
 
diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py
index c4a9ffa5ff1c77e..1916ee5507e9726 100644
--- a/Lib/test/test_codecs.py
+++ b/Lib/test/test_codecs.py
@@ -1646,6 +1646,14 @@ def test_nameprep(self):
                 except Exception as e:
                     raise support.TestFailed("Test 3.%d: %s" % (pos+1, str(e)))
 
+    def test_long_input(self):
+        from encodings.idna import nameprep
+        self.assertEqual(nameprep("x" + "\N{ZWJ}" * 10_000 + "y"), 'xy')
+        self.assertEqual(nameprep("x" + "\N{ZWJ}" * 10_000 + "y", limit=2),
+                         'xy')
+        with self.assertRaises(UnicodeEncodeError):
+            nameprep("x" + "\N{SNAKE}" * 10_000 + "y", limit=10)
+
 
 class IDNACodecTest(unittest.TestCase):
 
@@ -1716,11 +1724,33 @@ def test_builtin_encode_invalid(self):
                 self.assertEqual(exc.end, expected.end)
 
     def test_builtin_decode_length_limit(self):
-        with self.assertRaisesRegex(UnicodeDecodeError, "way too long"):
+        with self.assertRaisesRegex(UnicodeDecodeError, "too long"):
             (b"xn--016c"+b"a"*1100).decode("idna")
         with self.assertRaisesRegex(UnicodeDecodeError, "too long"):
             (b"xn--016c"+b"a"*70).decode("idna")
 
+    def test_builtin_encode_length_limit(self):
+        with self.assertRaisesRegex(UnicodeEncodeError, "too long"):
+            ("x" * 64).encode("idna")
+        with self.assertRaisesRegex(UnicodeEncodeError, "too long"):
+            ("short." + "x" * 64).encode("idna")
+
+        # Test at both sides of the limit (<64 bytes)
+        self.assertEqual(len(("\N{SNAKE}" * 56).encode("idna")), 63)
+        with self.assertRaisesRegex(UnicodeEncodeError, "too long"):
+            ("\N{SNAKE}" * 57).encode("idna")
+        with self.assertRaisesRegex(UnicodeEncodeError, "too long"):
+            ("short." + "\N{SNAKE}" * 57).encode("idna")
+
+        # Very long names are handled
+        with self.assertRaisesRegex(UnicodeEncodeError, "way too long"):
+            ("\N{SNAKE}"*50_000).encode("idna")
+
+        # The limit doesn't apply to ignored characters
+        self.assertEqual(('a' + "\N{ZWSP}"*50_000 + 'b').encode('idna'), b'ab')
+        self.assertEqual(('a' + "\N{ZWSP}"*50_000 + 
'\N{SNAKE}').encode('idna'),
+                         b'xn--a-012s')
+
     def test_stream(self):
         r = codecs.getreader("idna")(io.BytesIO(b"abc"))
         r.read(3)
diff --git 
a/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst 
b/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst
new file mode 100644
index 000000000000000..37f8e3e50c31aec
--- /dev/null
+++ b/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst
@@ -0,0 +1,5 @@
+:func:`encodings.idna.nameprep` now takes a *limit* argument that allows
+rejecting extremely large input early. The ``idna`` encoding and the
+:func:`!encodings.idna.ToASCII` and :func:`!encodings.idna.ToUnicode`
+functions use this to avoid passing unbounded input to quadratic-time
+algorithms.

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to