@@ -135,31 +135,27 @@ static PyObject *
135135decode_unicode_with_escapes (Parser * parser , const char * s , size_t len , Token * t )
136136{
137137 PyObject * v ;
138- PyObject * u ;
139- char * buf ;
140138 char * p ;
141139 const char * end ;
142140
143141 /* check for integer overflow */
144142 if (len > (size_t )PY_SSIZE_T_MAX / 6 ) {
145143 return NULL ;
146144 }
147- /* "ä" (2 bytes) may become "\U000000E4" (10 bytes), or 1:5
148- "\ä" (3 bytes) may become "\u005c\U000000E4" (16 bytes), or ~1:6 */
149- u = PyBytes_FromStringAndSize ((char * )NULL , (Py_ssize_t )len * 6 );
150- if (u == NULL ) {
151- return NULL ;
152- }
153- p = buf = PyBytes_AsString (u );
154- if (p == NULL ) {
145+ /* "ä" (2 bytes) may become "\U000000E4" (10 bytes), or 1:5.
146+ * "\ä" (3 bytes) may become "\u005c\U000000E4" (16 bytes), or ~1:6. */
147+ Py_ssize_t alloc = (Py_ssize_t )len * 6 ;
148+ char * buf = PyMem_Malloc (alloc );
149+ if (buf == NULL ) {
155150 return NULL ;
156151 }
152+ p = buf ;
157153 end = s + len ;
158154 while (s < end ) {
159155 if (* s == '\\' ) {
160156 * p ++ = * s ++ ;
161157 if (s >= end || * s & 0x80 ) {
162- strcpy (p , "u005c" );
158+ memcpy (p , "u005c" , 5 );
163159 p += 5 ;
164160 if (s >= end ) {
165161 break ;
@@ -174,46 +170,46 @@ decode_unicode_with_escapes(Parser *parser, const char *s, size_t len, Token *t)
174170 Py_ssize_t i ;
175171 w = decode_utf8 (& s , end );
176172 if (w == NULL ) {
177- Py_DECREF ( u );
173+ PyMem_Free ( buf );
178174 return NULL ;
179175 }
180176 kind = PyUnicode_KIND (w );
181177 data = PyUnicode_DATA (w );
182178 w_len = PyUnicode_GET_LENGTH (w );
183179 for (i = 0 ; i < w_len ; i ++ ) {
180+ // sprintf() writes a null byte: the buffer is large enough
181+ // for that thanks to the overallocation.
182+ assert ((p + 11 - buf ) <= alloc );
184183 Py_UCS4 chr = PyUnicode_READ (kind , data , i );
185184 sprintf (p , "\\U%08x" , chr );
186185 p += 10 ;
187186 }
188- /* Should be impossible to overflow */
189- assert (p - buf <= PyBytes_GET_SIZE (u ));
190187 Py_DECREF (w );
191188 }
192189 else {
193190 * p ++ = * s ++ ;
194191 }
195192 }
196193 len = (size_t )(p - buf );
197- s = buf ;
198194
199195 int first_invalid_escape_char ;
200196 const char * first_invalid_escape_ptr ;
201- v = _PyUnicode_DecodeUnicodeEscapeInternal2 (s , (Py_ssize_t )len , NULL , NULL ,
197+ v = _PyUnicode_DecodeUnicodeEscapeInternal2 (buf , (Py_ssize_t )len , NULL , NULL ,
202198 & first_invalid_escape_char ,
203199 & first_invalid_escape_ptr );
204200
205201 // HACK: later we can simply pass the line no, since we don't preserve the tokens
206202 // when we are decoding the string but we preserve the line numbers.
207203 if (v != NULL && first_invalid_escape_ptr != NULL && t != NULL ) {
208- if (warn_invalid_escape_sequence (parser , s , first_invalid_escape_ptr , t ) < 0 ) {
209- /* We have not decref u before because first_invalid_escape_ptr
210- points inside u . */
211- Py_XDECREF ( u );
204+ if (warn_invalid_escape_sequence (parser , buf , first_invalid_escape_ptr , t ) < 0 ) {
205+ /* We have not deallocated the buffer before because
206+ * first_invalid_escape_ptr points inside buf . */
207+ PyMem_Free ( buf );
212208 Py_DECREF (v );
213209 return NULL ;
214210 }
215211 }
216- Py_XDECREF ( u );
212+ PyMem_Free ( buf );
217213 return v ;
218214}
219215
0 commit comments