Skip to content

Commit 6e7a890

Browse files
authored
Merge branch 'main' into dtoa-precision-limit
2 parents f86460b + 70e6f3c commit 6e7a890

25 files changed

Lines changed: 536 additions & 283 deletions

‎Doc/c-api/unicode.rst‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1945,7 +1945,7 @@ object.
19451945
On success, return ``0``.
19461946
On error, set an exception, leave the writer unchanged, and return ``-1``.
19471947
1948-
.. c:function:: int PyUnicodeWriter_DecodeUTF8Stateful(PyUnicodeWriter *writer, const char *string, Py_ssize_t length, const char *errors, Py_ssize_t *consumed)
1948+
.. c:function:: int PyUnicodeWriter_DecodeUTF8Stateful(PyUnicodeWriter *writer, const char *str, Py_ssize_t size, const char *errors, Py_ssize_t *consumed)
19491949
19501950
Decode the string *str* from UTF-8 with *errors* error handler and write the
19511951
output into *writer*.

‎Include/cpython/unicodeobject.h‎

Lines changed: 27 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -516,8 +516,8 @@ PyAPI_FUNC(int) PyUnicodeWriter_Format(
516516
...);
517517
PyAPI_FUNC(int) PyUnicodeWriter_DecodeUTF8Stateful(
518518
PyUnicodeWriter *writer,
519-
const char *string, /* UTF-8 encoded string */
520-
Py_ssize_t length, /* size of string */
519+
const char *str, /* UTF-8 encoded string */
520+
Py_ssize_t size, /* size of string */
521521
const char *errors, /* error handling */
522522
Py_ssize_t *consumed); /* bytes consumed */
523523

@@ -538,6 +538,9 @@ typedef struct {
538538
/* minimum character (default: 127, ASCII) */
539539
Py_UCS4 min_char;
540540

541+
// If non-zero, _PyUnicodeWriter_Finish() needs to check maxchar.
542+
int recheck_maxchar;
543+
541544
/* If non-zero, overallocate the buffer (default: 0). */
542545
unsigned char overallocate;
543546

@@ -554,25 +557,33 @@ typedef struct {
554557
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(void) _PyUnicodeWriter_Init(
555558
_PyUnicodeWriter *writer);
556559

557-
/* Prepare the buffer to write 'length' characters
558-
with the specified maximum character.
559-
560-
Return 0 on success, raise an exception and return -1 on error. */
561-
#define _PyUnicodeWriter_Prepare(WRITER, LENGTH, MAXCHAR) \
562-
(((MAXCHAR) <= (WRITER)->maxchar \
563-
&& (LENGTH) <= (WRITER)->size - (WRITER)->pos) \
564-
? 0 \
565-
: (((LENGTH) == 0) \
566-
? 0 \
567-
: _PyUnicodeWriter_PrepareInternal((WRITER), (LENGTH), (MAXCHAR))))
568-
569-
/* Don't call this function directly, use the _PyUnicodeWriter_Prepare() macro
570-
instead. */
560+
// Don't call this function directly, use _PyUnicodeWriter_Prepare() instead.
571561
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int) _PyUnicodeWriter_PrepareInternal(
572562
_PyUnicodeWriter *writer,
573563
Py_ssize_t length,
574564
Py_UCS4 maxchar);
575565

566+
// Prepare the buffer to write 'length' characters
567+
// with the specified maximum character.
568+
//
569+
// Return 0 on success. Set an exception and return -1 on error.
570+
_Py_DEPRECATED_EXTERNALLY(3.14) static inline int
571+
_PyUnicodeWriter_Prepare(_PyUnicodeWriter *writer,
572+
Py_ssize_t length, Py_UCS4 maxchar)
573+
{
574+
assert(0 <= length);
575+
if (maxchar <= writer->maxchar && length <= (writer->size - writer->pos)) {
576+
return 0;
577+
}
578+
if (length == 0) {
579+
return 0;
580+
}
581+
_Py_COMP_DIAG_PUSH
582+
_Py_COMP_DIAG_IGNORE_DEPR_DECLS
583+
return _PyUnicodeWriter_PrepareInternal(writer, length, maxchar);
584+
_Py_COMP_DIAG_POP
585+
}
586+
576587
/* Prepare the buffer to have at least the kind KIND.
577588
For example, kind=PyUnicode_2BYTE_KIND ensures that the writer will
578589
support characters in range U+000-U+FFFF.

‎Include/internal/pycore_unicodeobject.h‎

Lines changed: 36 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -130,38 +130,51 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
130130
#endif
131131

132132
static inline void
133-
_PyUnicodeWriter_Update(_PyUnicodeWriter *writer)
133+
_PyUnicodeWriter_SetBuffer(_PyUnicodeWriter *writer, PyObject *buffer)
134134
{
135-
PyObject *buffer = writer->buffer;
136-
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
135+
assert(writer->pos <= PyUnicode_GET_LENGTH(buffer));
136+
137+
// Py_DECREF() the previous buffer (if any)
138+
Py_XSETREF(writer->buffer, buffer);
137139
writer->data = PyUnicode_DATA(buffer);
138140
writer->kind = PyUnicode_KIND(buffer);
141+
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
142+
writer->size = PyUnicode_GET_LENGTH(buffer);
143+
writer->readonly = 0;
144+
}
139145

140-
if (!writer->readonly) {
141-
writer->size = PyUnicode_GET_LENGTH(buffer);
142-
}
143-
else {
144-
/* Copy-on-write mode: set buffer size to 0 so
145-
* _PyUnicodeWriter_Prepare() will copy (and enlarge) the buffer on
146-
* next write. */
147-
writer->size = 0;
148-
}
146+
static inline void
147+
_PyUnicodeWriter_SetReadOnly(_PyUnicodeWriter *writer, PyObject *obj,
148+
Py_ssize_t length)
149+
{
150+
assert(writer->buffer == NULL);
151+
assert(writer->pos == 0);
152+
// Micro-optimization: pass length as a parameter, as it's usually known
153+
// by the caller
154+
assert(length == PyUnicode_GET_LENGTH(obj));
155+
156+
writer->buffer = obj;
157+
writer->data = NULL;
158+
/* Set kind and size to 0 to make sure that the next
159+
* _PyUnicodeWriter_Prepare() call allocates a new buffer and copies
160+
* characters. */
161+
writer->kind = 0;
162+
writer->maxchar = PyUnicode_MAX_CHAR_VALUE(obj);
163+
writer->size = 0;
164+
writer->pos = length;
165+
writer->readonly = 1;
149166
}
150167

151168
static inline int
152169
_PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
153170
{
154-
if (ch > writer->maxchar || 1 > writer->size - writer->pos) {
171+
if (ch > writer->maxchar || 1 > (writer->size - writer->pos)) {
155172
if (writer->buffer == NULL && ch <= 255) {
156173
// If the first write is a Latin1 character, use the singleton
157174
// as a read-only object
158175
PyObject *obj = _Py_LATIN1_CHR(ch);
159-
writer->readonly = 1;
160-
writer->buffer = obj; // Py_NewRef() is not need on immortal object
161-
_PyUnicodeWriter_Update(writer);
162-
assert(writer->pos == 0);
163-
writer->pos = 1;
164-
// The next write will create a new buffer and copy the string
176+
// Py_NewRef() is not needed on immortal object
177+
_PyUnicodeWriter_SetReadOnly(writer, obj, 1);
165178
return 0;
166179
}
167180

@@ -176,6 +189,10 @@ _PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
176189
return 0;
177190
}
178191

192+
extern PyObject* _PyUnicodeWriter_FinishWithSize(
193+
_PyUnicodeWriter *writer,
194+
Py_ssize_t size);
195+
179196
/* --- Unicode API -------------------------------------------------------- */
180197

181198
// Export for '_json' shared extension

‎Lib/test/test_capi/test_unicode.py‎

Lines changed: 88 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -461,6 +461,14 @@ def check_format(expected, format, *args):
461461
check_format('%abc',
462462
b'%%%s', b'abc')
463463

464+
# test "%s" with empty string
465+
check_format('x=',
466+
b'x=%s', b'')
467+
check_format('x=',
468+
b'x=%0s', b'')
469+
check_format('x=',
470+
b'x=%.3s', b'')
471+
464472
# truncated string
465473
check_format('abc',
466474
b'%.3s', b'abcdef')
@@ -1970,13 +1978,35 @@ def test_ascii(self):
19701978
writer.write_ascii(b"Python! <truncated>", 6)
19711979
self.assertEqual(writer.finish(), "Hello Python")
19721980

1981+
def test_write_latin1(self):
1982+
# Test _PyUnicodeWriter_WriteLatin1String()
1983+
writer = self.create_writer(0)
1984+
# Start with ASCII buffer
1985+
writer.write_latin1(b"abc IGNORED", 3)
1986+
writer.write_latin1(b"IGNORED", 0)
1987+
# Change buffer kind to UCS-1
1988+
writer.write_latin1(b"\xe9", 1)
1989+
# Change buffer kind to UCS-2
1990+
writer.write_str('[\u20ac]')
1991+
writer.write_latin1(b"def\xa0", 4)
1992+
# Change buffer kind to UCS-4
1993+
writer.write_str('[\U0010ffff]')
1994+
writer.write_latin1(b"ghi\xff.", 5)
1995+
writer.write_latin1(b"IGNORED", 0)
1996+
self.assertEqual(writer.finish(),
1997+
"abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")
1998+
19731999
def test_invalid_utf8(self):
19742000
writer = self.create_writer(0)
19752001
with self.assertRaises(UnicodeDecodeError):
19762002
writer.write_utf8(b"invalid=\xFF", -1)
19772003

19782004
def test_recover_utf8_error(self):
1979-
# test recovering from PyUnicodeWriter_WriteUTF8() error
2005+
# Recover from PyUnicodeWriter_WriteUTF8() errors. A temporary write
2006+
# changes the buffer kind to UCS-2 before raising UnicodeDecodeError.
2007+
# Then, PyUnicodeWriter_Finish() has to change the buffer kind back to
2008+
# ASCII.
2009+
19802010
writer = self.create_writer(0)
19812011
writer.write_utf8(b"value=", -1)
19822012

@@ -1986,12 +2016,13 @@ def test_recover_utf8_error(self):
19862016
with self.assertRaises(UnicodeDecodeError):
19872017
s = "truncated\u20AC".encode()
19882018
writer.write_utf8(s, len(s) - 1)
2019+
with self.assertRaises(UnicodeDecodeError):
2020+
# Change buffer kind to UCS-2 then raise UnicodeDecodeError
2021+
s = "\u20AC\u20AC".encode()
2022+
writer.write_utf8(s, len(s) - 1)
19892023

1990-
# retry write with a valid string
19912024
writer.write_utf8(b"valid", -1)
1992-
1993-
self.assertEqual(writer.finish(),
1994-
"value=valid")
2025+
self.assertEqual(writer.finish(), "value=valid")
19952026

19962027
def test_decode_utf8(self):
19972028
# test PyUnicodeWriter_DecodeUTF8Stateful()
@@ -2262,6 +2293,35 @@ def __repr__(self):
22622293
self.assertEqual(writer.get_buffer(), expected)
22632294
self.assertIs(writer.finish(), unique_string)
22642295

2296+
def test_readonly_optim_large_int(self):
2297+
# Read-only optimization in _PyLong_FormatWriter() for large integer:
2298+
# use _pylong.int_to_decimal_string() result as a read-only string.
2299+
# See pylong_int_to_decimal_string().
2300+
2301+
self.addCleanup(sys.set_int_max_str_digits,
2302+
sys.get_int_max_str_digits())
2303+
sys.set_int_max_str_digits(0)
2304+
2305+
# _PyLong_FormatWriter() calls _pylong.int_to_decimal_string() for
2306+
# integer with Py_SIZE() > 1000.
2307+
large_int = 1 << (sys.int_info.bits_per_digit * 1020)
2308+
large_int_str = str(large_int)
2309+
expected = (len(large_int_str), 127, True)
2310+
2311+
for size in (0, 123):
2312+
with self.subTest(size=size):
2313+
# Test PyUnicodeWriter_WriteStr()
2314+
writer = self.create_writer(size)
2315+
writer.write_str(large_int)
2316+
self.assertEqual(writer.get_buffer(), expected)
2317+
self.assertEqual(writer.finish(), large_int_str)
2318+
2319+
# Test PyUnicodeWriter_WriteRepr()
2320+
writer = self.create_writer(size)
2321+
writer.write_repr(large_int)
2322+
self.assertEqual(writer.get_buffer(), expected)
2323+
self.assertEqual(writer.finish(), large_int_str)
2324+
22652325

22662326
# Test PyUnicodeWriter_Format()
22672327
@unittest.skipIf(ctypes is None, 'need ctypes')
@@ -2299,6 +2359,29 @@ def test_recover_error(self):
22992359

23002360
self.assertEqual(writer.finish(), 'Hello World.')
23012361

2362+
def test_recheck_maxchar(self):
2363+
# PyUnicodeWriter_Format() changes buffer kind to UCS-2 before raising
2364+
# an exception. Then, PyUnicodeWriter_Finish() has to change the buffer
2365+
# kind back to ASCII.
2366+
from ctypes import py_object
2367+
2368+
class StrError:
2369+
def __str__(self):
2370+
raise RuntimeError("bug")
2371+
2372+
writer = self.create_writer(0)
2373+
# Allocate ASCII buffer
2374+
writer.write_str('ascii')
2375+
2376+
obj = StrError()
2377+
ucs2_utf8 = '\u20ac'.encode()
2378+
with self.assertRaises(RuntimeError):
2379+
# Change buffer kind to UCS-2, but then raise RuntimeError
2380+
self.writer_format(writer, b"%s%S", ucs2_utf8, py_object(obj))
2381+
2382+
writer.write_str('.')
2383+
self.assertEqual(writer.finish(), 'ascii.')
2384+
23022385
def test_readonly_optim(self):
23032386
# Read-only optimization: if the first and only write is a Python str
23042387
# object and no buffer was allocated yet, return the object unchanged

‎Lib/test/test_format.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -240,6 +240,8 @@ def test_common_format(self):
240240
testcommon("%d", 42, "42")
241241
testcommon("%d", -42, "-42")
242242
testcommon("%d", 42.0, "42")
243+
testcommon("%#d", 42, "42")
244+
testcommon("%#d", -42, "-42")
243245
testcommon("%#x", 1, "0x1")
244246
testcommon("%#X", 1, "0X1")
245247
testcommon("%#o", 1, "0o1")
@@ -250,8 +252,12 @@ def test_common_format(self):
250252
testcommon("%#X", 0, "0X0")
251253
testcommon("%x", 0x42, "42")
252254
testcommon("%x", -0x42, "-42")
255+
testcommon("%#x", 0x42, "0x42")
256+
testcommon("%#x", -0x42, "-0x42")
253257
testcommon("%o", 0o42, "42")
254258
testcommon("%o", -0o42, "-42")
259+
testcommon("%#o", 0o42, "0o42")
260+
testcommon("%#o", -0o42, "-0o42")
255261
# alternate float formatting
256262
testcommon('%g', 1.1, '1.1')
257263
testcommon('%#g', 1.1, '1.10000')
@@ -344,12 +350,21 @@ def test_common_format(self):
344350
"format argument 1: %g requires a real number, not str")
345351

346352
def test_str_format(self):
353+
testformat("%s", "abc", "abc")
347354
testformat("%r", "\u0378", "'\\u0378'") # non printable
348355
testformat("%a", "\u0378", "'\\u0378'") # non printable
349356
testformat("%r", "\u0374", "'\u0374'") # printable
350357
testformat("%a", "\u0374", "'\\u0374'") # printable
351358
testformat('%(x)r', {'x': 1}, '1')
352359

360+
# Some small ints
361+
for fmt in ('s', 'r', 'a'):
362+
with self.subTest(fmt=fmt):
363+
testformat("%" + fmt, 42, "42")
364+
testformat("%#" + fmt, 42, "42")
365+
testformat("%" + fmt, -42, "-42")
366+
testformat("%#" + fmt, -42, "-42")
367+
353368
# Test exception for unknown format characters, etc.
354369
if verbose:
355370
print('Testing exceptions')

0 commit comments

Comments
 (0)