Skip to content

Commit da5cb33

Browse files
committed
gh-158445: Reject invalid UCS4 in PyUnicode_FromKindAndData()
PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) and PyUnicodeWriter_WriteUCS4() now raise an exception if a character is not in the [U+0000; U+10ffff] range, instead of creating an invalid str object. * Add _testinternalcapi._Py_MAX_UNICODE. * Add unicode_invalid_character() helper function.
1 parent dc0b1f8 commit da5cb33

5 files changed

Lines changed: 88 additions & 24 deletions

File tree

‎Doc/c-api/unicode.rst‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1896,7 +1896,7 @@ object.
18961896
18971897
.. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const Py_UCS4 *str, Py_ssize_t size)
18981898
1899-
Writer the UCS4 string *str* into *writer*.
1899+
Write the UCS4 string *str* into *writer*.
19001900
19011901
*size* is a number of UCS4 characters.
19021902

‎Lib/test/test_capi/test_unicode.py‎

Lines changed: 39 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -19,6 +19,11 @@
1919
from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T
2020

2121

22+
MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE
23+
# The first invalid character after MAX_UNICODE
24+
INVALID_CHAR = MAX_UNICODE + 1
25+
# Maximum invalid character which fits into 32-bit Py_UCS4
26+
MAX_INVALID_CHAR = 0xFFFF_FFFF
2227
NULL = None
2328

2429
class Str(str):
@@ -73,14 +78,14 @@ def test_new(self):
7378
self.assertEqual(new(0, maxchar), '')
7479
self.assertEqual(new(5, maxchar), chr(maxchar)*5)
7580
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar)
76-
self.assertEqual(new(0, 0x110000), '')
81+
self.assertEqual(new(0, INVALID_CHAR), '')
7782
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60)
7883
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60)
7984
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600)
8085
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600)
8186
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600)
8287
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600)
83-
self.assertRaises(SystemError, new, 5, 0x110000)
88+
self.assertRaises(SystemError, new, 5, INVALID_CHAR)
8489
self.assertRaises(SystemError, new, -1, 0)
8590
self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0)
8691

@@ -115,7 +120,7 @@ def test_fill(self):
115120
s = strings[0]
116121
self.assertRaises(IndexError, fill, s, -1, 0, 0x78)
117122
self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78)
118-
self.assertRaises(ValueError, fill, s, 0, 0, 0x110000)
123+
self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR)
119124
self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78)
120125
self.assertRaises(SystemError, fill, [], 0, 0, 0x78)
121126
# CRASHES fill(s, 0, NULL, 0, 0)
@@ -129,7 +134,7 @@ def _test_writechar(self, writechar, *, check):
129134
'\U0001f600\U0001f601\U0001f602'
130135
]
131136
# one character for every kind + out of range code
132-
chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000]
137+
chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR]
133138
for i, s in enumerate(strings):
134139
for j, c in enumerate(chars):
135140
if j <= i:
@@ -301,7 +306,17 @@ def test_fromkindanddata(self):
301306
self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1)
302307
self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN)
303308
# CRASHES fromkindanddata(1, NULL, 1)
304-
# CRASHES fromkindanddata(4, b'\xff\xff\xff\xff')
309+
310+
# Test invalid UCS-4 string
311+
for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
312+
with self.subTest(invalid_char=invalid_char):
313+
# Test single character
314+
ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
315+
self.assertRaises(ValueError, fromkindanddata, 4, ucs4_char)
316+
317+
# Test multiple characters
318+
s = 'valid'.encode(enc4) + ucs4_char
319+
self.assertRaises(ValueError, fromkindanddata, 4, s)
305320

306321
def test_substring(self):
307322
"""Test PyUnicode_Substring()"""
@@ -446,7 +461,7 @@ def check_format(expected, format, *args):
446461
check_format('\U0010ffff',
447462
b'%c', c_int(0x10ffff))
448463
with self.assertRaises(OverflowError):
449-
PyUnicode_FromFormat(b'%c', c_int(0x110000))
464+
PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR))
450465
# Issue #18183
451466
check_format('\U00010000\U00100000',
452467
b'%c%c', c_int(0x10000), c_int(0x100000))
@@ -1015,7 +1030,7 @@ def test_fromordinal(self):
10151030
self.assertEqual(fromordinal(0x20ac), '\u20ac')
10161031
self.assertEqual(fromordinal(0x1f600), '\U0001f600')
10171032

1018-
self.assertRaises(ValueError, fromordinal, 0x110000)
1033+
self.assertRaises(ValueError, fromordinal, INVALID_CHAR)
10191034
self.assertRaises(ValueError, fromordinal, -1)
10201035

10211036
def test_asutf8(self):
@@ -1367,8 +1382,8 @@ def test_findchar(self):
13671382
self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str), -1), i)
13681383

13691384
str = "!>_<!"
1370-
self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), 1), -1)
1371-
self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), -1), -1)
1385+
self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), 1), -1)
1386+
self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), -1), -1)
13721387
# start < end
13731388
self.assertEqual(unicode_findchar(str, ord('!'), 1, len(str)+1, 1), 4)
13741389
self.assertEqual(unicode_findchar(str, ord('!'), 1, PY_SSIZE_T_MAX, 1), 4)
@@ -1757,7 +1772,7 @@ def test_max_char_value(self):
17571772
self.assertEqual(max_char_value('ascii'), 0x7f)
17581773
self.assertEqual(max_char_value('latin1:\xe9'), 0xff)
17591774
self.assertEqual(max_char_value('bmp:\u20ac'), 0xffff)
1760-
self.assertEqual(max_char_value('\U0010ffff'), 0x10_ffff)
1775+
self.assertEqual(max_char_value(chr(0x10_0000)), 0x10_ffff)
17611776

17621777
# CRASHES max_char_value(NULL)
17631778

@@ -1935,8 +1950,8 @@ def test_write_char(self):
19351950
writer.write_char(ord('$'))
19361951
writer.write_char(0x20ac)
19371952
writer.write_char(0x10_ffff)
1938-
self.assertRaises(ValueError, writer.write_char, 0x11_0000)
1939-
self.assertRaises(ValueError, writer.write_char, 0xFFFF_FFFF)
1953+
self.assertRaises(ValueError, writer.write_char, INVALID_CHAR)
1954+
self.assertRaises(ValueError, writer.write_char, MAX_INVALID_CHAR)
19401955
self.assertEqual(writer.finish(),
19411956
"\0$\u20AC\U0010FFFF")
19421957

@@ -2106,11 +2121,20 @@ def test_ucs4(self):
21062121
writer.write_ucs4("pair\uD83D\uDC0D".encode(encoding, 'surrogatepass'))
21072122
writer.write_char(ord("-"))
21082123
writer.write_ucs4("null[\0]".encode(encoding), 7)
2109-
invalid = (b'\x00\x00\x11\x00' if sys.byteorder == 'little' else
2110-
b'\x00\x11\x00\x00')
2111-
# CRASHES writer.write_ucs4("invalid".encode(encoding) + invalid)
21122124
writer.write_ucs4(NULL, 0)
21132125
# CRASHES writer.write_ucs4(NULL, 1)
2126+
2127+
# Invalid UCS-4 strings
2128+
for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
2129+
with self.subTest(invalid_char=invalid_char):
2130+
# Test single character
2131+
ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
2132+
self.assertRaises(ValueError, writer.write_ucs4, ucs4_char)
2133+
2134+
# Test multiple characters
2135+
s = 'valid'.encode(encoding) + ucs4_char
2136+
self.assertRaises(ValueError, writer.write_ucs4, s)
2137+
21142138
self.assertEqual(writer.finish(),
21152139
"lone\udc80-pair\ud83d\udc0d-null[\x00]")
21162140

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,4 @@
1+
:c:func:`PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND)
2+
<PyUnicode_FromKindAndData>` and :c:func:`PyUnicodeWriter_WriteUCS4` now raise
3+
an exception if a character is not in the [U+0000; U+10ffff] range, instead of
4+
creating an invalid str object. Patch by Victor Stinner.

‎Modules/_testinternalcapi.c‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3544,6 +3544,10 @@ module_exec(PyObject *module)
35443544
}
35453545
PyModule_AddObject(module, "SelfInterruptingContextManager", (PyObject *)&SelfInterruptingContextManager_Type);
35463546

3547+
if (PyModule_AddIntMacro(module, _Py_MAX_UNICODE) < 0) {
3548+
return 1;
3549+
}
3550+
35473551
return 0;
35483552
}
35493553

‎Objects/unicodeobject.c‎

Lines changed: 40 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -1592,6 +1592,14 @@ PyUnicode_CopyCharacters(PyObject *to, Py_ssize_t to_start,
15921592
return how_many;
15931593
}
15941594

1595+
static void
1596+
unicode_invalid_character(Py_UCS4 ch)
1597+
{
1598+
PyErr_Format(PyExc_ValueError,
1599+
"character U+%x is not in range [U+0000; U+%x]",
1600+
ch, MAX_UNICODE);
1601+
}
1602+
15951603
/* Find the maximum code point and count the number of surrogate pairs so a
15961604
correct string length can be computed before converting a string to UCS4.
15971605
This function counts single surrogates as a character and not as a pair.
@@ -1627,9 +1635,7 @@ find_maxchar_surrogates(const wchar_t *begin, const wchar_t *end,
16271635
if (ch > *maxchar) {
16281636
*maxchar = ch;
16291637
if (*maxchar > MAX_UNICODE) {
1630-
PyErr_Format(PyExc_ValueError,
1631-
"character U+%x is not in range [U+0000; U+%x]",
1632-
ch, MAX_UNICODE);
1638+
unicode_invalid_character(ch);
16331639
return -1;
16341640
}
16351641
}
@@ -2224,15 +2230,29 @@ static PyObject*
22242230
_PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size)
22252231
{
22262232
PyObject *res;
2227-
Py_UCS4 max_char;
22282233

22292234
if (size == 0)
22302235
_Py_RETURN_UNICODE_EMPTY();
22312236
assert(size > 0);
2232-
if (size == 1)
2237+
2238+
// ucs4lib_find_max_char() cannot be used, it ignores limit greater
2239+
// than MAX_UNICODE
2240+
Py_UCS4 max_char = 127;
2241+
for (Py_ssize_t i = 0; i < size; i++) {
2242+
Py_UCS4 ch = u[i];
2243+
if (ch > max_char) {
2244+
if (ch > MAX_UNICODE) {
2245+
unicode_invalid_character(ch);
2246+
return NULL;
2247+
}
2248+
max_char = ch;
2249+
}
2250+
}
2251+
2252+
if (size == 1) {
22332253
return unicode_char(u[0]);
2254+
}
22342255

2235-
max_char = ucs4lib_find_max_char(u, u + size);
22362256
res = PyUnicode_New(size, max_char);
22372257
if (!res)
22382258
return NULL;
@@ -2266,9 +2286,21 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer,
22662286
return 0;
22672287
}
22682288

2269-
Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size);
2289+
// ucs4lib_find_max_char() cannot be used, it ignores limit greater
2290+
// than MAX_UNICODE
2291+
Py_UCS4 maxchar = 127;
2292+
for (Py_ssize_t i = 0; i < size; i++) {
2293+
Py_UCS4 ch = str[i];
2294+
if (ch > maxchar) {
2295+
if (ch > MAX_UNICODE) {
2296+
unicode_invalid_character(ch);
2297+
return -1;
2298+
}
2299+
maxchar = ch;
2300+
}
2301+
}
22702302

2271-
if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) {
2303+
if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) {
22722304
return -1;
22732305
}
22742306
assert(_PyUnicodeWriter_CanWrite(writer));

0 commit comments

Comments
 (0)