@@ -461,6 +461,14 @@ def check_format(expected, format, *args):
461461 check_format ('%abc' ,
462462 b'%%%s' , b'abc' )
463463
464+ # test "%s" with empty string
465+ check_format ('x=' ,
466+ b'x=%s' , b'' )
467+ check_format ('x=' ,
468+ b'x=%0s' , b'' )
469+ check_format ('x=' ,
470+ b'x=%.3s' , b'' )
471+
464472 # truncated string
465473 check_format ('abc' ,
466474 b'%.3s' , b'abcdef' )
@@ -1970,13 +1978,35 @@ def test_ascii(self):
19701978 writer .write_ascii (b"Python! <truncated>" , 6 )
19711979 self .assertEqual (writer .finish (), "Hello Python" )
19721980
1981+ def test_write_latin1 (self ):
1982+ # Test _PyUnicodeWriter_WriteLatin1String()
1983+ writer = self .create_writer (0 )
1984+ # Start with ASCII buffer
1985+ writer .write_latin1 (b"abc IGNORED" , 3 )
1986+ writer .write_latin1 (b"IGNORED" , 0 )
1987+ # Change buffer kind to UCS-1
1988+ writer .write_latin1 (b"\xe9 " , 1 )
1989+ # Change buffer kind to UCS-2
1990+ writer .write_str ('[\u20ac ]' )
1991+ writer .write_latin1 (b"def\xa0 " , 4 )
1992+ # Change buffer kind to UCS-4
1993+ writer .write_str ('[\U0010ffff ]' )
1994+ writer .write_latin1 (b"ghi\xff ." , 5 )
1995+ writer .write_latin1 (b"IGNORED" , 0 )
1996+ self .assertEqual (writer .finish (),
1997+ "abc\xe9 [\u20ac ]def\xa0 [\U0010ffff ]ghi\xff ." )
1998+
19731999 def test_invalid_utf8 (self ):
19742000 writer = self .create_writer (0 )
19752001 with self .assertRaises (UnicodeDecodeError ):
19762002 writer .write_utf8 (b"invalid=\xFF " , - 1 )
19772003
19782004 def test_recover_utf8_error (self ):
1979- # test recovering from PyUnicodeWriter_WriteUTF8() error
2005+ # Recover from PyUnicodeWriter_WriteUTF8() errors. A temporary write
2006+ # changes the buffer kind to UCS-2 before raising UnicodeDecodeError.
2007+ # Then, PyUnicodeWriter_Finish() has to change the buffer kind back to
2008+ # ASCII.
2009+
19802010 writer = self .create_writer (0 )
19812011 writer .write_utf8 (b"value=" , - 1 )
19822012
@@ -1986,12 +2016,13 @@ def test_recover_utf8_error(self):
19862016 with self .assertRaises (UnicodeDecodeError ):
19872017 s = "truncated\u20AC " .encode ()
19882018 writer .write_utf8 (s , len (s ) - 1 )
2019+ with self .assertRaises (UnicodeDecodeError ):
2020+ # Change buffer kind to UCS-2 then raise UnicodeDecodeError
2021+ s = "\u20AC \u20AC " .encode ()
2022+ writer .write_utf8 (s , len (s ) - 1 )
19892023
1990- # retry write with a valid string
19912024 writer .write_utf8 (b"valid" , - 1 )
1992-
1993- self .assertEqual (writer .finish (),
1994- "value=valid" )
2025+ self .assertEqual (writer .finish (), "value=valid" )
19952026
19962027 def test_decode_utf8 (self ):
19972028 # test PyUnicodeWriter_DecodeUTF8Stateful()
@@ -2262,6 +2293,35 @@ def __repr__(self):
22622293 self .assertEqual (writer .get_buffer (), expected )
22632294 self .assertIs (writer .finish (), unique_string )
22642295
2296+ def test_readonly_optim_large_int (self ):
2297+ # Read-only optimization in _PyLong_FormatWriter() for large integer:
2298+ # use _pylong.int_to_decimal_string() result as a read-only string.
2299+ # See pylong_int_to_decimal_string().
2300+
2301+ self .addCleanup (sys .set_int_max_str_digits ,
2302+ sys .get_int_max_str_digits ())
2303+ sys .set_int_max_str_digits (0 )
2304+
2305+ # _PyLong_FormatWriter() calls _pylong.int_to_decimal_string() for
2306+ # integer with Py_SIZE() > 1000.
2307+ large_int = 1 << (sys .int_info .bits_per_digit * 1020 )
2308+ large_int_str = str (large_int )
2309+ expected = (len (large_int_str ), 127 , True )
2310+
2311+ for size in (0 , 123 ):
2312+ with self .subTest (size = size ):
2313+ # Test PyUnicodeWriter_WriteStr()
2314+ writer = self .create_writer (size )
2315+ writer .write_str (large_int )
2316+ self .assertEqual (writer .get_buffer (), expected )
2317+ self .assertEqual (writer .finish (), large_int_str )
2318+
2319+ # Test PyUnicodeWriter_WriteRepr()
2320+ writer = self .create_writer (size )
2321+ writer .write_repr (large_int )
2322+ self .assertEqual (writer .get_buffer (), expected )
2323+ self .assertEqual (writer .finish (), large_int_str )
2324+
22652325
22662326# Test PyUnicodeWriter_Format()
22672327@unittest .skipIf (ctypes is None , 'need ctypes' )
@@ -2299,6 +2359,29 @@ def test_recover_error(self):
22992359
23002360 self .assertEqual (writer .finish (), 'Hello World.' )
23012361
2362+ def test_recheck_maxchar (self ):
2363+ # PyUnicodeWriter_Format() changes buffer kind to UCS-2 before raising
2364+ # an exception. Then, PyUnicodeWriter_Finish() has to change the buffer
2365+ # kind back to ASCII.
2366+ from ctypes import py_object
2367+
2368+ class StrError :
2369+ def __str__ (self ):
2370+ raise RuntimeError ("bug" )
2371+
2372+ writer = self .create_writer (0 )
2373+ # Allocate ASCII buffer
2374+ writer .write_str ('ascii' )
2375+
2376+ obj = StrError ()
2377+ ucs2_utf8 = '\u20ac ' .encode ()
2378+ with self .assertRaises (RuntimeError ):
2379+ # Change buffer kind to UCS-2, but then raise RuntimeError
2380+ self .writer_format (writer , b"%s%S" , ucs2_utf8 , py_object (obj ))
2381+
2382+ writer .write_str ('.' )
2383+ self .assertEqual (writer .finish (), 'ascii.' )
2384+
23022385 def test_readonly_optim (self ):
23032386 # Read-only optimization: if the first and only write is a Python str
23042387 # object and no buffer was allocated yet, return the object unchanged
0 commit comments