Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,10 @@
- Backport algorithmic character names from CPython, including Tangut, Jurchen, and Small Seal.
- Backport CPython's faster canonical ordering, also fixing Unicode normalization on PyPy.
- Require Python 3.9 or newer; drop Python 3.8 support.
- Fix historical numeric values in the Unicode 3.2 database.
- Make Hangul syllable lookup case-insensitive.
- Return algorithmic Hangul syllable mappings from decomposition().
- Always return an exact str from normalize(), including for str subclasses.

## 17.0.0
- Upgrade to Unicode 17.0.0
Expand Down
62 changes: 61 additions & 1 deletion tests/test_unicodedata2.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ class UnicodeFunctionsTest(UnicodeDatabaseTest):

# Update this if the database changes. Make sure to do a full rebuild
# (e.g. 'make distclean && make') to get the correct checksum.
expectedchecksum = 'f11a52558bcefd64c833f44f7ccb51faa8ec3310'
expectedchecksum = 'f2f46908e4a8ea1616baf2390f5383c12f610c5f'

def test_function_checksum(self):
import unicodedata2
Expand Down Expand Up @@ -149,12 +149,14 @@ def test_numeric(self):
self.assertEqual(self.db.numeric('\U0001012A', None), 9000)
# Changed in 4.1.0
self.assertEqual(self.db.numeric('\u5793', None), None)
self.assertEqual(self.db.ucd_3_2_0.numeric('\u5793', None), 1e20)
# New in 5.0.0
self.assertEqual(self.db.numeric('\u07c0', None), 0.0)
# New in 5.1.0
self.assertEqual(self.db.numeric('\ua627', None), 7.0)
# Changed in 5.2.0
self.assertEqual(self.db.numeric('\u09f6'), 3/16)
self.assertEqual(self.db.ucd_3_2_0.numeric('\u09f6'), 3.0)
# New in 6.0.0
self.assertEqual(self.db.numeric('\u0b72', None), 0.25)
# New in 12.0.0
Expand Down Expand Up @@ -231,6 +233,20 @@ def test_decomposition(self):
self.assertEqual(self.db.decomposition('\uFFFE'),'')
self.assertEqual(self.db.decomposition('\u00bc'), '<fraction> 0031 2044 0034')

# Hangul characters (CPython gh-88091), in both database views.
self.assertEqual(self.db.decomposition('\uAC00'), '1100 1161')
self.assertEqual(self.db.decomposition('\uD4DB'), '1111 1171 11B6')
self.assertEqual(self.db.decomposition('\uC2F8'), '110A 1161')
self.assertEqual(self.db.decomposition('\uD7A3'), '1112 1175 11C2')
self.assertEqual(self.db.ucd_3_2_0.decomposition('\uAC00'),
'1100 1161')
self.assertEqual(self.db.ucd_3_2_0.decomposition('\uD4DB'),
'1111 1171 11B6')
self.assertEqual(self.db.ucd_3_2_0.decomposition('\uC2F8'),
'110A 1161')
self.assertEqual(self.db.ucd_3_2_0.decomposition('\uD7A3'),
'1112 1175 11C2')

self.assertRaises(TypeError, self.db.decomposition)
self.assertRaises(TypeError, self.db.decomposition, 'xx')

Expand Down Expand Up @@ -438,6 +454,26 @@ def test_lookup_nonexistant(self):
]:
self.assertRaises(KeyError, self.db.lookup, nonexistent)

def test_hangul_syllables(self):
self.assertEqual(self.db.lookup("HANGUL SYLLABLE GA"), "\uac00")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE GGWEOSS"), "\uafe8")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE DOLS"), "\ub3d0")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE RYAN"), "\ub7b8")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE MWIK"), "\ubba0")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE BBWAEM"), "\ubf88")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE SSEOL"), "\uc370")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE YI"), "\uc758")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE JJYOSS"), "\ucb40")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE KYEOLS"), "\ucf28")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE PAN"), "\ud310")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE HWEOK"), "\ud6f8")
self.assertEqual(self.db.lookup("HANGUL SYLLABLE HIH"), "\ud7a3")

self.assertEqual(self.db.lookup("haNGul SYllABle WAe"), '\uc65c')
self.assertEqual(self.db.lookup("HAngUL syLLabLE waE"), '\uc65c')

self.assertRaises(ValueError, self.db.name, "\ud7a4")

def test_tangut_ideographs(self):
self.assertEqual(self.db.lookup("TANGUT IDEOGRAPH-17000"), "\U00017000")
self.assertEqual(self.db.lookup("TANGUT IDEOGRAPH-187FF"), "\U000187ff")
Expand Down Expand Up @@ -606,6 +642,30 @@ def test_linebreak_7643(self):
self.assertEqual(len(lines), 1,
r"\u%.4x should not be a linebreak" % i)

def test_normalize_return_type(self):
# gh-129569: normalize() return type must always be str
normalize = self.db.normalize

class MyStr(str):
pass

normalization_forms = ("NFC", "NFKC", "NFD", "NFKD")
input_strings = (
# normalized strings
"",
"ascii",
# unnormalized strings
"\u1e0b\u0323",
"\u0071\u0307\u0323",
)

for form in normalization_forms:
for input_str in input_strings:
with self.subTest(form=form, input_str=input_str):
self.assertIs(type(normalize(form, input_str)), str)
self.assertIs(type(normalize(form, MyStr(input_str))), str)


class NormalizationTest(UnicodeDatabaseTest):
# Adapted from CPython's Lib/test/test_unicodedata.py (9ab004d41e).
# Setup downloads the data; never skip it. unicodedata2 has no is_normalized.
Expand Down
85 changes: 50 additions & 35 deletions unicodedata2/unicodedata.c
Original file line number Diff line number Diff line change
Expand Up @@ -233,9 +233,9 @@ unicodedata_UCD_numeric_impl(PyObject *self, int chr,
have_old = 1;
rc = -1.0;
}
else if (old->decimal_changed != 0xFF) {
else if (old->numeric_changed != 0.0) {
have_old = 1;
rc = old->decimal_changed;
rc = old->numeric_changed;
}
}

Expand Down Expand Up @@ -392,6 +392,17 @@ unicodedata_UCD_east_asian_width_impl(PyObject *self, int chr)
return PyUnicode_FromString(_PyUnicode_EastAsianWidthNames[index]);
}

// For Hangul decomposition
#define SBase 0xAC00
#define LBase 0x1100
#define VBase 0x1161
#define TBase 0x11A7
#define LCount 19
#define VCount 21
#define TCount 28
#define NCount (VCount*TCount)
#define SCount (LCount*NCount)

/*[clinic input]
unicodedata.UCD.decomposition

Expand Down Expand Up @@ -422,6 +433,25 @@ unicodedata_UCD_decomposition_impl(PyObject *self, int chr)
return PyUnicode_FromString(""); /* unassigned */
}

// Hangul Decomposition.
// See section 3.12.2, "Hangul Syllable Decomposition"
// https://www.unicode.org/versions/latest/core-spec/chapter-3/#G56669
if (SBase <= code && code < (SBase + SCount)) {
int SIndex = code - SBase;
int L = LBase + SIndex / NCount;
int V = VBase + (SIndex % NCount) / TCount;
int T = TBase + SIndex % TCount;
if (T != TBase) {
PyOS_snprintf(decomp, sizeof(decomp),
"%04X %04X %04X", L, V, T);
}
else {
PyOS_snprintf(decomp, sizeof(decomp),
"%04X %04X", L, V);
}
return PyUnicode_FromString(decomp);
}

if (code < 0 || code >= 0x110000)
index = 0;
else {
Expand Down Expand Up @@ -482,16 +512,6 @@ get_decomp_record(PyObject *self, Py_UCS4 code, int *index, int *prefix, int *co
(*index)++;
}

#define SBase 0xAC00
#define LBase 0x1100
#define VBase 0x1161
#define TBase 0x11A7
#define LCount 19
#define VCount 21
#define TCount 28
#define NCount (VCount*TCount)
#define SCount (LCount*NCount)

#define CANONICAL_ORDERING_COUNTING_SORT_THRESHOLD 20

static void
Expand Down Expand Up @@ -916,35 +936,30 @@ unicodedata_UCD_normalize_impl(PyObject *self, const char *form,
if (PyUnicode_GET_LENGTH(input) == 0) {
/* Special case empty input strings, since resizing
them later would cause internal errors. */
Py_INCREF(input);
return input;
return PyUnicode_FromObject(input);
}

if (strcmp(form, "NFC") == 0) {
if (is_normalized(self, input, 1, 0)) {
Py_INCREF(input);
return input;
return PyUnicode_FromObject(input);
}
return nfc_nfkc(self, input, 0);
}
if (strcmp(form, "NFKC") == 0) {
if (is_normalized(self, input, 1, 1)) {
Py_INCREF(input);
return input;
return PyUnicode_FromObject(input);
}
return nfc_nfkc(self, input, 1);
}
if (strcmp(form, "NFD") == 0) {
if (is_normalized(self, input, 0, 0)) {
Py_INCREF(input);
return input;
return PyUnicode_FromObject(input);
}
return nfd_nfkd(self, input, 0);
}
if (strcmp(form, "NFKD") == 0) {
if (is_normalized(self, input, 0, 1)) {
Py_INCREF(input);
return input;
return PyUnicode_FromObject(input);
}
return nfd_nfkd(self, input, 1);
}
Expand Down Expand Up @@ -1146,6 +1161,18 @@ _cmpname(PyObject *self, int code, const char* name, int namelen)
return buffer[namelen] == '\0';
}

/* The generated prefixes are uppercase ASCII. PyPy lacks PyOS_strnicmp. */
static int
name_startswith(const char *name, const char *prefix)
{
while (*prefix) {
if (UNICODEDATA2_TOUPPER(*name++) != *prefix++) {
return 0;
}
}
return 1;
}

static void
find_syllable(const char *str, int *len, int *pos, int count, int column)
{
Expand All @@ -1156,7 +1183,7 @@ find_syllable(const char *str, int *len, int *pos, int count, int column)
len1 = Py_SAFE_DOWNCAST(strlen(s), size_t, int);
if (len1 <= *len)
continue;
if (strncmp(str, s, len1) == 0) {
if (name_startswith(str, s)) {
*len = len1;
*pos = i;
}
Expand Down Expand Up @@ -1211,18 +1238,6 @@ parse_hex_code(const char *name, int namelen)
return v;
}

/* The generated prefixes are uppercase ASCII. PyPy lacks PyOS_strnicmp. */
static int
name_startswith(const char *name, const char *prefix)
{
while (*prefix) {
if (UNICODEDATA2_TOUPPER(*name++) != *prefix++) {
return 0;
}
}
return 1;
}

static int
_getcode(PyObject* self, const char* name, int namelen, Py_UCS4* code,
int with_named_seq)
Expand Down
Loading