11# This module implements the RFCs 3490 (IDNA) and 3491 (Nameprep)
22
3+ import sys
34import stringprep , re , codecs
45from unicodedata import ucd_3_2_0 as unicodedata
56
1112sace_prefix = "xn--"
1213
1314# This assumes query strings, so AllowUnassigned is true
14- def nameprep (label ): # type: (str) -> str
15+ def nameprep (label , * , limit = None ): # type: (str) -> str
16+ if limit is None :
17+ limit = sys .maxsize
18+ else :
19+ # Protection from gh-98433 and gh-157675 (passing unbounded input to
20+ # the quadratic-complexity punycode algorithm).
21+ # While the "map" step can remove characters, later steps (in ToASCII
22+ # and FromASCII) will not shorten the result *drastically*.
23+ # (NFKC normalization can compress e.g. '\u03c9\u0314\u0300\u0345'
24+ # to '\u1fa3' -- a 4-fold reduction. Non-ASCII labels then get
25+ # longer via prefixing & punycode).
26+ # We bail if the number of non-ignored input characters exceeds 8 times
27+ # the limit, which gives ample room for future Unicode versions to
28+ # include long normalizations, while still preventing us from wasting
29+ # time decoding a big thing that'll just hit the actual <= 63 limit in
30+ # ToASCII.
31+ limit *= 8
32+
1533 # Map
1634 newlabel = []
1735 for c in label :
1836 if stringprep .in_table_b1 (c ):
1937 # Map to nothing
2038 continue
2139 newlabel .append (stringprep .map_table_b2 (c ))
40+
41+ if len (newlabel ) > limit :
42+ raise UnicodeEncodeError ("idna" , label , 0 , len (label ),
43+ "label way too long" )
44+
2245 label = "" .join (newlabel )
2346
2447 # Normalize
@@ -80,7 +103,7 @@ def ToASCII(label): # type: (str) -> bytes
80103 raise UnicodeEncodeError ("idna" , label , 0 , len (label ), "label too long" )
81104
82105 # Step 2: nameprep
83- label = nameprep (label )
106+ label = nameprep (label , limit = 63 )
84107
85108 # Step 3: UseSTD3ASCIIRules is false
86109 # Step 4: try ASCII
@@ -115,18 +138,6 @@ def ToASCII(label): # type: (str) -> bytes
115138 raise UnicodeEncodeError ("idna" , label , 0 , len (label ), "label too long" )
116139
117140def ToUnicode (label ):
118- if len (label ) > 1024 :
119- # Protection from https://github.com/python/cpython/issues/98433.
120- # https://datatracker.ietf.org/doc/html/rfc5894#section-6
121- # doesn't specify a label size limit prior to NAMEPREP. But having
122- # one makes practical sense.
123- # This leaves ample room for nameprep() to remove Nothing characters
124- # per https://www.rfc-editor.org/rfc/rfc3454#section-3.1 while still
125- # preventing us from wasting time decoding a big thing that'll just
126- # hit the actual <= 63 length limit in Step 6.
127- if isinstance (label , str ):
128- label = label .encode ("utf-8" , errors = "backslashreplace" )
129- raise UnicodeDecodeError ("idna" , label , 0 , len (label ), "label way too long" )
130141 # Step 1: Check for ASCII
131142 if isinstance (label , bytes ):
132143 pure_ascii = True
@@ -139,7 +150,7 @@ def ToUnicode(label):
139150 if not pure_ascii :
140151 assert isinstance (label , str )
141152 # Step 2: Perform nameprep
142- label = nameprep (label )
153+ label = nameprep (label , limit = 63 )
143154 # It doesn't say this, but apparently, it should be ASCII now
144155 try :
145156 label = label .encode ("ascii" )
@@ -151,6 +162,11 @@ def ToUnicode(label):
151162 if not label .lower ().startswith (ace_prefix ):
152163 return str (label , "ascii" )
153164
165+ # Below in steps 6-7, `label` must match the result of `ToASCII`, so it's
166+ # limited to 63 chars. Check before the expensive punycode decode.
167+ if len (label ) >= 64 :
168+ raise UnicodeDecodeError ("idna" , label , 0 , len (label ), "label too long" )
169+
154170 # Step 4: Remove ACE prefix
155171 label1 = label [len (ace_prefix ):]
156172
0 commit comments