https://github.com/python/cpython/commit/f7586aa88dbaf3f4bd515e972b82f530820795a9 commit: f7586aa88dbaf3f4bd515e972b82f530820795a9 branch: main author: Petr Viktorin <[email protected]> committer: encukou <[email protected]> date: 2026-10-02T15:11:09+02:00 summary:
gh-157675: Add *limit* argument to encodings.idna.nameprep (GH-157682) Add a *limit* argument to nameprep to allow ToASCII and ToUnicode to reject extremely large input early. This replaces the check in #99092, while allowing any number of harmless "characters mapped to nothing" (RFC 3454 ยง3.1). The default stays unlimited, to not change behaviour for users that call nameprep manually (and don't necessarily follow up with punycode). Co-authored-by: Stan Ulbrych <[email protected]> files: A Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst M Doc/library/codecs.rst M Doc/whatsnew/3.16.rst M Lib/encodings/idna.py M Lib/test/test_codecs.py diff --git a/Doc/library/codecs.rst b/Doc/library/codecs.rst index 311437a67f6e811..ed20a435faf4345 100644 --- a/Doc/library/codecs.rst +++ b/Doc/library/codecs.rst @@ -1395,15 +1395,6 @@ encodings. | | | :mod:`encodings.idna`. | | | | Only ``errors='strict'`` | | | | is supported. | -| | | | -| | | .. warning:: | -| | | | -| | | This codec builds on | -| | | ``punycode``, whose | -| | | algorithms scale | -| | | poorly, so limit the | -| | | length of untrusted | -| | | input. | +--------------------+---------+---------------------------+ | mbcs | ansi, | Windows only: Encode the | | | dbcs | operand according to the | @@ -1655,11 +1646,6 @@ Applications) and :rfc:`3492` (Nameprep: A Stringprep Profile for Internationalized Domain Names (IDN)). It builds upon the ``punycode`` encoding and :mod:`stringprep`. -.. warning:: - - This module builds on ``punycode``, whose algorithms scale poorly, so limit - the length of untrusted input. - If you need the IDNA 2008 standard from :rfc:`5891` and :rfc:`5895`, use the third-party :pypi:`idna` module. @@ -1697,11 +1683,31 @@ international domain names, and to unify similar characters. The nameprep functions can be used directly if desired. -.. function:: nameprep(label) +.. function:: nameprep(label, *, limit=None) Return the nameprepped version of *label*. The implementation currently assumes query strings, so ``AllowUnassigned`` is true. + Raise :exc:`UnicodeEncodeError` if the nameprep algorithm emits an error. + + If the *limit* argument is given, it should be set to the maximum size + of an encoded A-label (that is, 63 for IDNA). + :func:`!nameprep` will raise :exc:`UnicodeEncodeError` if the label is + **much** larger than *limit*. + Note that this is only a rough check meant to skip expensive processing + of extremely large input; the caller should check any exact + limits separately. + + .. warning:: + + For backwards compatibility, label size is unlimited by default. + This may cause issues when processing the result with the + ``punycode`` encoding, whose algorithms scale poorly. + + .. versionchanged:: next + + Added the *limit* parameter. + .. function:: ToASCII(label) diff --git a/Doc/whatsnew/3.16.rst b/Doc/whatsnew/3.16.rst index 29bd7dcd749ebc2..16e03e9ef065321 100644 --- a/Doc/whatsnew/3.16.rst +++ b/Doc/whatsnew/3.16.rst @@ -397,6 +397,10 @@ encodings used for international IMAP4 mailbox names (:rfc:`3501`). (Contributed by Serhiy Storchaka in :gh:`66788`.) +* :func:`encodings.idna.nameprep` now takes a *limit* argument that allows + rejecting extremely large input early. + (Contributed by Petr Viktorin in :gh:`157675`.) + gzip ---- diff --git a/Lib/encodings/idna.py b/Lib/encodings/idna.py index c896ffdeadfef7e..98d7d891594c648 100644 --- a/Lib/encodings/idna.py +++ b/Lib/encodings/idna.py @@ -1,5 +1,6 @@ # This module implements the RFCs 3490 (IDNA) and 3491 (Nameprep) +import sys import stringprep, re, codecs from unicodedata import ucd_3_2_0 as unicodedata @@ -11,7 +12,24 @@ sace_prefix = "xn--" # This assumes query strings, so AllowUnassigned is true -def nameprep(label): # type: (str) -> str +def nameprep(label, *, limit=None): # type: (str) -> str + if limit is None: + limit = sys.maxsize + else: + # Protection from gh-98433 and gh-157675 (passing unbounded input to + # the quadratic-complexity punycode algorithm). + # While the "map" step can remove characters, later steps (in ToASCII + # and FromASCII) will not shorten the result *drastically*. + # (NFKC normalization can compress e.g. '\u03c9\u0314\u0300\u0345' + # to '\u1fa3' -- a 4-fold reduction. Non-ASCII labels then get + # longer via prefixing & punycode). + # We bail if the number of non-ignored input characters exceeds 8 times + # the limit, which gives ample room for future Unicode versions to + # include long normalizations, while still preventing us from wasting + # time decoding a big thing that'll just hit the actual <= 63 limit in + # ToASCII. + limit *= 8 + # Map newlabel = [] for c in label: @@ -19,6 +37,11 @@ def nameprep(label): # type: (str) -> str # Map to nothing continue newlabel.append(stringprep.map_table_b2(c)) + + if len(newlabel) > limit: + raise UnicodeEncodeError("idna", label, 0, len(label), + "label way too long") + label = "".join(newlabel) # Normalize @@ -80,7 +103,7 @@ def ToASCII(label): # type: (str) -> bytes raise UnicodeEncodeError("idna", label, 0, len(label), "label too long") # Step 2: nameprep - label = nameprep(label) + label = nameprep(label, limit=63) # Step 3: UseSTD3ASCIIRules is false # Step 4: try ASCII @@ -115,18 +138,6 @@ def ToASCII(label): # type: (str) -> bytes raise UnicodeEncodeError("idna", label, 0, len(label), "label too long") def ToUnicode(label): - if len(label) > 1024: - # Protection from https://github.com/python/cpython/issues/98433. - # https://datatracker.ietf.org/doc/html/rfc5894#section-6 - # doesn't specify a label size limit prior to NAMEPREP. But having - # one makes practical sense. - # This leaves ample room for nameprep() to remove Nothing characters - # per https://www.rfc-editor.org/rfc/rfc3454#section-3.1 while still - # preventing us from wasting time decoding a big thing that'll just - # hit the actual <= 63 length limit in Step 6. - if isinstance(label, str): - label = label.encode("utf-8", errors="backslashreplace") - raise UnicodeDecodeError("idna", label, 0, len(label), "label way too long") # Step 1: Check for ASCII if isinstance(label, bytes): pure_ascii = True @@ -139,7 +150,7 @@ def ToUnicode(label): if not pure_ascii: assert isinstance(label, str) # Step 2: Perform nameprep - label = nameprep(label) + label = nameprep(label, limit=63) # It doesn't say this, but apparently, it should be ASCII now try: label = label.encode("ascii") @@ -151,6 +162,11 @@ def ToUnicode(label): if not label.lower().startswith(ace_prefix): return str(label, "ascii") + # Below in steps 6-7, `label` must match the result of `ToASCII`, so it's + # limited to 63 chars. Check before the expensive punycode decode. + if len(label) >= 64: + raise UnicodeDecodeError("idna", label, 0, len(label), "label too long") + # Step 4: Remove ACE prefix label1 = label[len(ace_prefix):] diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index c4a9ffa5ff1c77e..1916ee5507e9726 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -1646,6 +1646,14 @@ def test_nameprep(self): except Exception as e: raise support.TestFailed("Test 3.%d: %s" % (pos+1, str(e))) + def test_long_input(self): + from encodings.idna import nameprep + self.assertEqual(nameprep("x" + "\N{ZWJ}" * 10_000 + "y"), 'xy') + self.assertEqual(nameprep("x" + "\N{ZWJ}" * 10_000 + "y", limit=2), + 'xy') + with self.assertRaises(UnicodeEncodeError): + nameprep("x" + "\N{SNAKE}" * 10_000 + "y", limit=10) + class IDNACodecTest(unittest.TestCase): @@ -1716,11 +1724,33 @@ def test_builtin_encode_invalid(self): self.assertEqual(exc.end, expected.end) def test_builtin_decode_length_limit(self): - with self.assertRaisesRegex(UnicodeDecodeError, "way too long"): + with self.assertRaisesRegex(UnicodeDecodeError, "too long"): (b"xn--016c"+b"a"*1100).decode("idna") with self.assertRaisesRegex(UnicodeDecodeError, "too long"): (b"xn--016c"+b"a"*70).decode("idna") + def test_builtin_encode_length_limit(self): + with self.assertRaisesRegex(UnicodeEncodeError, "too long"): + ("x" * 64).encode("idna") + with self.assertRaisesRegex(UnicodeEncodeError, "too long"): + ("short." + "x" * 64).encode("idna") + + # Test at both sides of the limit (<64 bytes) + self.assertEqual(len(("\N{SNAKE}" * 56).encode("idna")), 63) + with self.assertRaisesRegex(UnicodeEncodeError, "too long"): + ("\N{SNAKE}" * 57).encode("idna") + with self.assertRaisesRegex(UnicodeEncodeError, "too long"): + ("short." + "\N{SNAKE}" * 57).encode("idna") + + # Very long names are handled + with self.assertRaisesRegex(UnicodeEncodeError, "way too long"): + ("\N{SNAKE}"*50_000).encode("idna") + + # The limit doesn't apply to ignored characters + self.assertEqual(('a' + "\N{ZWSP}"*50_000 + 'b').encode('idna'), b'ab') + self.assertEqual(('a' + "\N{ZWSP}"*50_000 + '\N{SNAKE}').encode('idna'), + b'xn--a-012s') + def test_stream(self): r = codecs.getreader("idna")(io.BytesIO(b"abc")) r.read(3) diff --git a/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst b/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst new file mode 100644 index 000000000000000..37f8e3e50c31aec --- /dev/null +++ b/Misc/NEWS.d/next/Library/2026-09-17-15-14-08.gh-issue-157675.C4-ze9.rst @@ -0,0 +1,5 @@ +:func:`encodings.idna.nameprep` now takes a *limit* argument that allows +rejecting extremely large input early. The ``idna`` encoding and the +:func:`!encodings.idna.ToASCII` and :func:`!encodings.idna.ToUnicode` +functions use this to avoid passing unbounded input to quadratic-time +algorithms. _______________________________________________ Python-checkins mailing list -- [email protected] To unsubscribe send an email to [email protected] https://mail.python.org/mailman3//lists/python-checkins.python.org Member address: [email protected]
