From ea1f7c58fb5a5ba79bec137d328ca6ad5b227ce6 Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Fri, 18 Sep 2026 12:09:24 +0100 Subject: [PATCH 1/4] Fix numeric values in the Unicode 3.2 database Backport CPython 3e0322ff16: use numeric_changed and its sentinel instead of decimal_changed. Add the two upstream historical-value assertions. --- CHANGELOG.md | 1 + tests/test_unicodedata2.py | 2 ++ unicodedata2/unicodedata.c | 4 ++-- 3 files changed, 5 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bf3d3c3..99c6cb9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,7 @@ - Backport algorithmic character names from CPython, including Tangut, Jurchen, and Small Seal. - Backport CPython's faster canonical ordering, also fixing Unicode normalization on PyPy. - Require Python 3.9 or newer; drop Python 3.8 support. + - Fix historical numeric values in the Unicode 3.2 database. ## 17.0.0 - Upgrade to Unicode 17.0.0 diff --git a/tests/test_unicodedata2.py b/tests/test_unicodedata2.py index 4575456..6695031 100644 --- a/tests/test_unicodedata2.py +++ b/tests/test_unicodedata2.py @@ -149,12 +149,14 @@ def test_numeric(self): self.assertEqual(self.db.numeric('\U0001012A', None), 9000) # Changed in 4.1.0 self.assertEqual(self.db.numeric('\u5793', None), None) + self.assertEqual(self.db.ucd_3_2_0.numeric('\u5793', None), 1e20) # New in 5.0.0 self.assertEqual(self.db.numeric('\u07c0', None), 0.0) # New in 5.1.0 self.assertEqual(self.db.numeric('\ua627', None), 7.0) # Changed in 5.2.0 self.assertEqual(self.db.numeric('\u09f6'), 3/16) + self.assertEqual(self.db.ucd_3_2_0.numeric('\u09f6'), 3.0) # New in 6.0.0 self.assertEqual(self.db.numeric('\u0b72', None), 0.25) # New in 12.0.0 diff --git a/unicodedata2/unicodedata.c b/unicodedata2/unicodedata.c index db9d4aa..4e69b72 100644 --- a/unicodedata2/unicodedata.c +++ b/unicodedata2/unicodedata.c @@ -233,9 +233,9 @@ unicodedata_UCD_numeric_impl(PyObject *self, int chr, have_old = 1; rc = -1.0; } - else if (old->decimal_changed != 0xFF) { + else if (old->numeric_changed != 0.0) { have_old = 1; - rc = old->decimal_changed; + rc = old->numeric_changed; } } From 8bc148dfabf1f9e4e9fa00560869e8990e23e634 Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Fri, 18 Sep 2026 12:09:24 +0100 Subject: [PATCH 2/4] Make Hangul syllable lookup case-insensitive Backport CPython e66f4a5a9c using the existing PyPy-safe name_startswith helper. Port the upstream Hangul lookup cases from test_ucn. --- CHANGELOG.md | 1 + tests/test_unicodedata2.py | 20 ++++++++++++++++++++ unicodedata2/unicodedata.c | 26 +++++++++++++------------- 3 files changed, 34 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 99c6cb9..ec8520e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ - Backport CPython's faster canonical ordering, also fixing Unicode normalization on PyPy. - Require Python 3.9 or newer; drop Python 3.8 support. - Fix historical numeric values in the Unicode 3.2 database. + - Make Hangul syllable lookup case-insensitive. ## 17.0.0 - Upgrade to Unicode 17.0.0 diff --git a/tests/test_unicodedata2.py b/tests/test_unicodedata2.py index 6695031..42c11e8 100644 --- a/tests/test_unicodedata2.py +++ b/tests/test_unicodedata2.py @@ -440,6 +440,26 @@ def test_lookup_nonexistant(self): ]: self.assertRaises(KeyError, self.db.lookup, nonexistent) + def test_hangul_syllables(self): + self.assertEqual(self.db.lookup("HANGUL SYLLABLE GA"), "\uac00") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE GGWEOSS"), "\uafe8") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE DOLS"), "\ub3d0") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE RYAN"), "\ub7b8") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE MWIK"), "\ubba0") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE BBWAEM"), "\ubf88") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE SSEOL"), "\uc370") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE YI"), "\uc758") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE JJYOSS"), "\ucb40") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE KYEOLS"), "\ucf28") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE PAN"), "\ud310") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE HWEOK"), "\ud6f8") + self.assertEqual(self.db.lookup("HANGUL SYLLABLE HIH"), "\ud7a3") + + self.assertEqual(self.db.lookup("haNGul SYllABle WAe"), '\uc65c') + self.assertEqual(self.db.lookup("HAngUL syLLabLE waE"), '\uc65c') + + self.assertRaises(ValueError, self.db.name, "\ud7a4") + def test_tangut_ideographs(self): self.assertEqual(self.db.lookup("TANGUT IDEOGRAPH-17000"), "\U00017000") self.assertEqual(self.db.lookup("TANGUT IDEOGRAPH-187FF"), "\U000187ff") diff --git a/unicodedata2/unicodedata.c b/unicodedata2/unicodedata.c index 4e69b72..f78e33f 100644 --- a/unicodedata2/unicodedata.c +++ b/unicodedata2/unicodedata.c @@ -1146,6 +1146,18 @@ _cmpname(PyObject *self, int code, const char* name, int namelen) return buffer[namelen] == '\0'; } +/* The generated prefixes are uppercase ASCII. PyPy lacks PyOS_strnicmp. */ +static int +name_startswith(const char *name, const char *prefix) +{ + while (*prefix) { + if (UNICODEDATA2_TOUPPER(*name++) != *prefix++) { + return 0; + } + } + return 1; +} + static void find_syllable(const char *str, int *len, int *pos, int count, int column) { @@ -1156,7 +1168,7 @@ find_syllable(const char *str, int *len, int *pos, int count, int column) len1 = Py_SAFE_DOWNCAST(strlen(s), size_t, int); if (len1 <= *len) continue; - if (strncmp(str, s, len1) == 0) { + if (name_startswith(str, s)) { *len = len1; *pos = i; } @@ -1211,18 +1223,6 @@ parse_hex_code(const char *name, int namelen) return v; } -/* The generated prefixes are uppercase ASCII. PyPy lacks PyOS_strnicmp. */ -static int -name_startswith(const char *name, const char *prefix) -{ - while (*prefix) { - if (UNICODEDATA2_TOUPPER(*name++) != *prefix++) { - return 0; - } - } - return 1; -} - static int _getcode(PyObject* self, const char* name, int namelen, Py_UCS4* code, int with_named_seq) From abea68c1e739760ae3e2ff2439975fe24396245f Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Fri, 18 Sep 2026 12:09:24 +0100 Subject: [PATCH 3/4] Add algorithmic Hangul decomposition Backport CPython 56c4f10d6e and its regression cases for both database views. Update the function checksum for the 11,172 Hangul mappings; generated data is unchanged. --- CHANGELOG.md | 1 + tests/test_unicodedata2.py | 16 ++++++++++++++- unicodedata2/unicodedata.c | 40 ++++++++++++++++++++++++++++---------- 3 files changed, 46 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ec8520e..209f554 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,7 @@ - Require Python 3.9 or newer; drop Python 3.8 support. - Fix historical numeric values in the Unicode 3.2 database. - Make Hangul syllable lookup case-insensitive. + - Return algorithmic Hangul syllable mappings from decomposition(). ## 17.0.0 - Upgrade to Unicode 17.0.0 diff --git a/tests/test_unicodedata2.py b/tests/test_unicodedata2.py index 42c11e8..997ca84 100644 --- a/tests/test_unicodedata2.py +++ b/tests/test_unicodedata2.py @@ -39,7 +39,7 @@ class UnicodeFunctionsTest(UnicodeDatabaseTest): # Update this if the database changes. Make sure to do a full rebuild # (e.g. 'make distclean && make') to get the correct checksum. - expectedchecksum = 'f11a52558bcefd64c833f44f7ccb51faa8ec3310' + expectedchecksum = 'f2f46908e4a8ea1616baf2390f5383c12f610c5f' def test_function_checksum(self): import unicodedata2 @@ -233,6 +233,20 @@ def test_decomposition(self): self.assertEqual(self.db.decomposition('\uFFFE'),'') self.assertEqual(self.db.decomposition('\u00bc'), ' 0031 2044 0034') + # Hangul characters (CPython gh-88091), in both database views. + self.assertEqual(self.db.decomposition('\uAC00'), '1100 1161') + self.assertEqual(self.db.decomposition('\uD4DB'), '1111 1171 11B6') + self.assertEqual(self.db.decomposition('\uC2F8'), '110A 1161') + self.assertEqual(self.db.decomposition('\uD7A3'), '1112 1175 11C2') + self.assertEqual(self.db.ucd_3_2_0.decomposition('\uAC00'), + '1100 1161') + self.assertEqual(self.db.ucd_3_2_0.decomposition('\uD4DB'), + '1111 1171 11B6') + self.assertEqual(self.db.ucd_3_2_0.decomposition('\uC2F8'), + '110A 1161') + self.assertEqual(self.db.ucd_3_2_0.decomposition('\uD7A3'), + '1112 1175 11C2') + self.assertRaises(TypeError, self.db.decomposition) self.assertRaises(TypeError, self.db.decomposition, 'xx') diff --git a/unicodedata2/unicodedata.c b/unicodedata2/unicodedata.c index f78e33f..8e2f5b0 100644 --- a/unicodedata2/unicodedata.c +++ b/unicodedata2/unicodedata.c @@ -392,6 +392,17 @@ unicodedata_UCD_east_asian_width_impl(PyObject *self, int chr) return PyUnicode_FromString(_PyUnicode_EastAsianWidthNames[index]); } +// For Hangul decomposition +#define SBase 0xAC00 +#define LBase 0x1100 +#define VBase 0x1161 +#define TBase 0x11A7 +#define LCount 19 +#define VCount 21 +#define TCount 28 +#define NCount (VCount*TCount) +#define SCount (LCount*NCount) + /*[clinic input] unicodedata.UCD.decomposition @@ -422,6 +433,25 @@ unicodedata_UCD_decomposition_impl(PyObject *self, int chr) return PyUnicode_FromString(""); /* unassigned */ } + // Hangul Decomposition. + // See section 3.12.2, "Hangul Syllable Decomposition" + // https://www.unicode.org/versions/latest/core-spec/chapter-3/#G56669 + if (SBase <= code && code < (SBase + SCount)) { + int SIndex = code - SBase; + int L = LBase + SIndex / NCount; + int V = VBase + (SIndex % NCount) / TCount; + int T = TBase + SIndex % TCount; + if (T != TBase) { + PyOS_snprintf(decomp, sizeof(decomp), + "%04X %04X %04X", L, V, T); + } + else { + PyOS_snprintf(decomp, sizeof(decomp), + "%04X %04X", L, V); + } + return PyUnicode_FromString(decomp); + } + if (code < 0 || code >= 0x110000) index = 0; else { @@ -482,16 +512,6 @@ get_decomp_record(PyObject *self, Py_UCS4 code, int *index, int *prefix, int *co (*index)++; } -#define SBase 0xAC00 -#define LBase 0x1100 -#define VBase 0x1161 -#define TBase 0x11A7 -#define LCount 19 -#define VCount 21 -#define TCount 28 -#define NCount (VCount*TCount) -#define SCount (LCount*NCount) - #define CANONICAL_ORDERING_COUNTING_SORT_THRESHOLD 20 static void From 42c5003190eca26fcbf588d9805da143261207bb Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Fri, 18 Sep 2026 12:09:24 +0100 Subject: [PATCH 4/4] Always return built-in str from normalize Backport CPython c359fcd2f5 and test_normalize_return_type. Convert str subclasses on the empty and already-normalized fast paths without changing the normalization algorithm. --- CHANGELOG.md | 1 + tests/test_unicodedata2.py | 24 ++++++++++++++++++++++++ unicodedata2/unicodedata.c | 15 +++++---------- 3 files changed, 30 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 209f554..3a84c0b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,7 @@ - Fix historical numeric values in the Unicode 3.2 database. - Make Hangul syllable lookup case-insensitive. - Return algorithmic Hangul syllable mappings from decomposition(). + - Always return an exact str from normalize(), including for str subclasses. ## 17.0.0 - Upgrade to Unicode 17.0.0 diff --git a/tests/test_unicodedata2.py b/tests/test_unicodedata2.py index 997ca84..32f1700 100644 --- a/tests/test_unicodedata2.py +++ b/tests/test_unicodedata2.py @@ -642,6 +642,30 @@ def test_linebreak_7643(self): self.assertEqual(len(lines), 1, r"\u%.4x should not be a linebreak" % i) + def test_normalize_return_type(self): + # gh-129569: normalize() return type must always be str + normalize = self.db.normalize + + class MyStr(str): + pass + + normalization_forms = ("NFC", "NFKC", "NFD", "NFKD") + input_strings = ( + # normalized strings + "", + "ascii", + # unnormalized strings + "\u1e0b\u0323", + "\u0071\u0307\u0323", + ) + + for form in normalization_forms: + for input_str in input_strings: + with self.subTest(form=form, input_str=input_str): + self.assertIs(type(normalize(form, input_str)), str) + self.assertIs(type(normalize(form, MyStr(input_str))), str) + + class NormalizationTest(UnicodeDatabaseTest): # Adapted from CPython's Lib/test/test_unicodedata.py (9ab004d41e). # Setup downloads the data; never skip it. unicodedata2 has no is_normalized. diff --git a/unicodedata2/unicodedata.c b/unicodedata2/unicodedata.c index 8e2f5b0..e9af9a4 100644 --- a/unicodedata2/unicodedata.c +++ b/unicodedata2/unicodedata.c @@ -936,35 +936,30 @@ unicodedata_UCD_normalize_impl(PyObject *self, const char *form, if (PyUnicode_GET_LENGTH(input) == 0) { /* Special case empty input strings, since resizing them later would cause internal errors. */ - Py_INCREF(input); - return input; + return PyUnicode_FromObject(input); } if (strcmp(form, "NFC") == 0) { if (is_normalized(self, input, 1, 0)) { - Py_INCREF(input); - return input; + return PyUnicode_FromObject(input); } return nfc_nfkc(self, input, 0); } if (strcmp(form, "NFKC") == 0) { if (is_normalized(self, input, 1, 1)) { - Py_INCREF(input); - return input; + return PyUnicode_FromObject(input); } return nfc_nfkc(self, input, 1); } if (strcmp(form, "NFD") == 0) { if (is_normalized(self, input, 0, 0)) { - Py_INCREF(input); - return input; + return PyUnicode_FromObject(input); } return nfd_nfkd(self, input, 0); } if (strcmp(form, "NFKD") == 0) { if (is_normalized(self, input, 0, 1)) { - Py_INCREF(input); - return input; + return PyUnicode_FromObject(input); } return nfd_nfkd(self, input, 1); }