Skip to content

Commit 1e03154

Browse files
authored
gh-155742: Use PyMem_Malloc() in decode_unicode_with_escapes() (#157587)
Replace soft deprecated PyBytes_FromStringAndSize(NULL, size) with a simple PyMem_Malloc(). Avoid also "s = buf" to make the code easier to follow (and easier to debug). Add non-ASCII tests to test_string_literals.
1 parent 21403af commit 1e03154

2 files changed

Lines changed: 37 additions & 22 deletions

File tree

‎Lib/test/test_string_literals.py‎

Lines changed: 20 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -54,6 +54,10 @@
5454
assert ord(h) == 0x1d120
5555
i = r'\U0001d120'
5656
assert list(map(ord, i)) == [92, 85, 48, 48, 48, 49, 100, 49, 50, 48]
57+
j = 'ä'
58+
assert list(map(ord, j)) == [228]
59+
k = '\ä'
60+
assert list(map(ord, k)) == [92, 228]
5761
"""
5862

5963

@@ -76,7 +80,7 @@ def test_template(self):
7680
# Check that the template doesn't contain any non-printables
7781
# except for \n.
7882
for c in TEMPLATE:
79-
assert c == '\n' or ' ' <= c <= '~', repr(c)
83+
assert c == '\n' or ' ' <= c <= '~' or c == 'ä', repr(c)
8084

8185
def test_eval_str_normal(self):
8286
self.assertEqual(eval(""" 'x' """), 'x')
@@ -88,6 +92,21 @@ def test_eval_str_normal(self):
8892
self.assertEqual(eval(""" '\u1881' """), chr(0x1881))
8993
self.assertEqual(eval(r""" '\U0001d120' """), chr(0x1d120))
9094
self.assertEqual(eval(""" '\U0001d120' """), chr(0x1d120))
95+
# Lone "\" character at the end
96+
self.assertEqual(eval(r"'abc\\'"), 'abc\\')
97+
98+
def test_eval_str_unicode(self):
99+
for s in (
100+
'ϼўТλФЙ',
101+
'A͏B ﬖ̳AA̝',
102+
'\U00100000\U0010ffff\U0010fffd',
103+
'ä',
104+
'\\ä',
105+
"\\П",
106+
"áàäéèęöő.\\n",
107+
):
108+
with self.subTest(s=s):
109+
self.assertEqual(eval(f"{s!r}"), s)
91110

92111
def test_eval_str_incomplete(self):
93112
self.assertRaises(SyntaxError, eval, r""" '\x' """)

‎Parser/string_parser.c‎

Lines changed: 17 additions & 21 deletions
Original file line numberDiff line numberDiff line change
@@ -135,31 +135,27 @@ static PyObject *
135135
decode_unicode_with_escapes(Parser *parser, const char *s, size_t len, Token *t)
136136
{
137137
PyObject *v;
138-
PyObject *u;
139-
char *buf;
140138
char *p;
141139
const char *end;
142140

143141
/* check for integer overflow */
144142
if (len > (size_t)PY_SSIZE_T_MAX / 6) {
145143
return NULL;
146144
}
147-
/* "ä" (2 bytes) may become "\U000000E4" (10 bytes), or 1:5
148-
"\ä" (3 bytes) may become "\u005c\U000000E4" (16 bytes), or ~1:6 */
149-
u = PyBytes_FromStringAndSize((char *)NULL, (Py_ssize_t)len * 6);
150-
if (u == NULL) {
151-
return NULL;
152-
}
153-
p = buf = PyBytes_AsString(u);
154-
if (p == NULL) {
145+
/* "ä" (2 bytes) may become "\U000000E4" (10 bytes), or 1:5.
146+
* "\ä" (3 bytes) may become "\u005c\U000000E4" (16 bytes), or ~1:6. */
147+
Py_ssize_t alloc = (Py_ssize_t)len * 6;
148+
char *buf = PyMem_Malloc(alloc);
149+
if (buf == NULL) {
155150
return NULL;
156151
}
152+
p = buf;
157153
end = s + len;
158154
while (s < end) {
159155
if (*s == '\\') {
160156
*p++ = *s++;
161157
if (s >= end || *s & 0x80) {
162-
strcpy(p, "u005c");
158+
memcpy(p, "u005c", 5);
163159
p += 5;
164160
if (s >= end) {
165161
break;
@@ -174,46 +170,46 @@ decode_unicode_with_escapes(Parser *parser, const char *s, size_t len, Token *t)
174170
Py_ssize_t i;
175171
w = decode_utf8(&s, end);
176172
if (w == NULL) {
177-
Py_DECREF(u);
173+
PyMem_Free(buf);
178174
return NULL;
179175
}
180176
kind = PyUnicode_KIND(w);
181177
data = PyUnicode_DATA(w);
182178
w_len = PyUnicode_GET_LENGTH(w);
183179
for (i = 0; i < w_len; i++) {
180+
// sprintf() writes a null byte: the buffer is large enough
181+
// for that thanks to the overallocation.
182+
assert((p + 11 - buf) <= alloc);
184183
Py_UCS4 chr = PyUnicode_READ(kind, data, i);
185184
sprintf(p, "\\U%08x", chr);
186185
p += 10;
187186
}
188-
/* Should be impossible to overflow */
189-
assert(p - buf <= PyBytes_GET_SIZE(u));
190187
Py_DECREF(w);
191188
}
192189
else {
193190
*p++ = *s++;
194191
}
195192
}
196193
len = (size_t)(p - buf);
197-
s = buf;
198194

199195
int first_invalid_escape_char;
200196
const char *first_invalid_escape_ptr;
201-
v = _PyUnicode_DecodeUnicodeEscapeInternal2(s, (Py_ssize_t)len, NULL, NULL,
197+
v = _PyUnicode_DecodeUnicodeEscapeInternal2(buf, (Py_ssize_t)len, NULL, NULL,
202198
&first_invalid_escape_char,
203199
&first_invalid_escape_ptr);
204200

205201
// HACK: later we can simply pass the line no, since we don't preserve the tokens
206202
// when we are decoding the string but we preserve the line numbers.
207203
if (v != NULL && first_invalid_escape_ptr != NULL && t != NULL) {
208-
if (warn_invalid_escape_sequence(parser, s, first_invalid_escape_ptr, t) < 0) {
209-
/* We have not decref u before because first_invalid_escape_ptr
210-
points inside u. */
211-
Py_XDECREF(u);
204+
if (warn_invalid_escape_sequence(parser, buf, first_invalid_escape_ptr, t) < 0) {
205+
/* We have not deallocated the buffer before because
206+
* first_invalid_escape_ptr points inside buf. */
207+
PyMem_Free(buf);
212208
Py_DECREF(v);
213209
return NULL;
214210
}
215211
}
216-
Py_XDECREF(u);
212+
PyMem_Free(buf);
217213
return v;
218214
}
219215

0 commit comments

Comments
 (0)