From 93b9cf057d5f5547b7bc123fe4a3ee752e66bd94 Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Thu, 17 Sep 2026 12:56:36 +0100 Subject: [PATCH 1/2] Sync unicodedata tests with CPython Port normalization conformance tests from CPython 9ab004d41e into test_unicodedata2.py and remove the stale reference to test_normalization.py. Also add coverage for long combining-mark runs, private-use names, newer character properties, aliases and named sequences, and check numeric consistency across all code points. Download the current database's normalization corpus and the frozen Unicode 3.2 corpus during CI and tox setup. Select the current version from the generated database header, validate the downloaded file headers, and fail on download errors instead of skipping conformance checks. Keep downloaded data out of git and distributions. On unfixed PyPy, both normalization conformance tests and both combining-mark run tests fail. The following canonical-ordering backport makes them pass. Tests for APIs or behavior not yet backported from CPython are left out. --- .github/workflows/ci.yml | 4 + MANIFEST.in | 1 + README.md | 5 + tests/download_test_data.py | 35 +++++ tests/test_unicodedata2.py | 252 ++++++++++++++++++++++++++++++++++-- tox.ini | 1 + 6 files changed, 288 insertions(+), 10 deletions(-) create mode 100644 tests/download_test_data.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2d07fea..e5db580 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -46,6 +46,8 @@ jobs: python-version: "3.x" - name: Install dependencies run: pip install cibuildwheel + - name: Download normalization test data + run: python tests/download_test_data.py - name: Build and Test Wheels run: python -m cibuildwheel --output-dir wheelhouse - uses: actions/upload-artifact@v4 @@ -72,6 +74,8 @@ jobs: platforms: all - name: Install dependencies run: pip install cibuildwheel + - name: Download normalization test data + run: python tests/download_test_data.py - name: Build and Test Wheels run: python -m cibuildwheel --output-dir wheelhouse - uses: actions/upload-artifact@v4 diff --git a/MANIFEST.in b/MANIFEST.in index b2f0076..468d48c 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -4,3 +4,4 @@ include LICENSE include unicodedata2/*.c include unicodedata2/*.h include tests/*.py +prune tests/data diff --git a/README.md b/README.md index d946cca..e45e57c 100644 --- a/README.md +++ b/README.md @@ -23,6 +23,11 @@ We run the tests using `tox`. This can be installed as usual with `pip install t or with `pip install --group dev` (pip 25.1 or newer) to pick it up from `pyproject.toml`. +Tox and CI download the version-matched Unicode normalization data before +running tests. Download failures stop the run. Before running `pytest` directly, +run `python tests/download_test_data.py` once. The downloaded files are not +included in source distributions or wheels. + Without any options, `tox` will run the tests against all of the library's target Python versions. Any missing versions will be skipped. diff --git a/tests/download_test_data.py b/tests/download_test_data.py new file mode 100644 index 0000000..d5c777a --- /dev/null +++ b/tests/download_test_data.py @@ -0,0 +1,35 @@ +"""Download version-matched normalization data before running the tests.""" + +from pathlib import Path +import re +from urllib.request import Request, urlopen + + +def main(): + tests_dir = Path(__file__).resolve().parent + header = tests_dir.parent / 'unicodedata2' / 'unicodedata_db.h' + match = re.search(r'^#define UNIDATA_VERSION "([^"]+)"', + header.read_text(encoding='utf-8'), re.MULTILINE) + if match is None: + raise ValueError('Cannot find UNIDATA_VERSION in %s' % header) + + data_dir = tests_dir / 'data' + data_dir.mkdir(exist_ok=True) + for version in (match.group(1), '3.2.0'): + filename = 'NormalizationTest-%s.txt' % version + if version == '3.2.0': + url = 'https://www.unicode.org/Public/3.2-Update/' + filename + else: + url = ('https://www.unicode.org/Public/%s/ucd/NormalizationTest.txt' + % version) + request = Request(url, headers={'User-Agent': 'unicodedata2'}) + with urlopen(request, timeout=60) as response: + data = response.read() + if data.decode('utf-8').splitlines()[:1] != ['# ' + filename]: + raise ValueError('Unexpected version header in %s' % url) + (data_dir / filename).write_bytes(data) + print('Downloaded %s' % filename) + + +if __name__ == '__main__': + main() diff --git a/tests/test_unicodedata2.py b/tests/test_unicodedata2.py index c2382f9..8080c1b 100644 --- a/tests/test_unicodedata2.py +++ b/tests/test_unicodedata2.py @@ -6,6 +6,8 @@ """ +from functools import partial +from pathlib import Path import sys import unittest import hashlib @@ -14,6 +16,10 @@ encoding = 'utf-8' errors = 'surrogatepass' +# Selected tests adapted from CPython 9ab004d41e: +# Lib/test/test_unicodedata.py and Lib/test/test_ucn.py. +# Property tests cover the current database; normalization tests also cover 3.2.0. + ### Run tests # NOTE: UnicodeMethodsTest upstream tests methods on `str` objects, and @@ -60,6 +66,52 @@ def test_function_checksum(self): result = h.hexdigest() self.assertEqual(result, self.expectedchecksum) + def test_aliases(self): + # Check that the aliases defined in the NameAliases.txt file work. + # This should be updated when new aliases are added or the file + # should be downloaded and parsed instead. See #12753. + aliases = [ + ('LATIN CAPITAL LETTER GHA', 0x01A2), + ('LATIN SMALL LETTER GHA', 0x01A3), + ('KANNADA LETTER LLLA', 0x0CDE), + ('LAO LETTER FO FON', 0x0E9D), + ('LAO LETTER FO FAY', 0x0E9F), + ('LAO LETTER RO', 0x0EA3), + ('LAO LETTER LO', 0x0EA5), + ('TIBETAN MARK BKA- SHOG GI MGO RGYAN', 0x0FD0), + ('YI SYLLABLE ITERATION MARK', 0xA015), + ('PRESENTATION FORM FOR VERTICAL RIGHT WHITE LENTICULAR BRACKET', 0xFE18), + ('BYZANTINE MUSICAL SYMBOL FTHORA SKLIRON CHROMA VASIS', 0x1D0C5) + ] + for alias, codepoint in aliases: + self.assertEqual(self.db.lookup(alias), chr(codepoint)) + name = self.db.name(chr(codepoint)) + self.assertNotEqual(name, alias) + self.assertEqual(self.db.lookup(alias), + self.db.lookup(name)) + with self.assertRaises(KeyError): + self.db.ucd_3_2_0.lookup(alias) + + def test_named_sequences_sample(self): + # Check a few named sequences. See #12753. + sequences = [ + ('LATIN SMALL LETTER R WITH TILDE', '\u0072\u0303'), + ('TAMIL SYLLABLE SAI', '\u0BB8\u0BC8'), + ('TAMIL SYLLABLE MOO', '\u0BAE\u0BCB'), + ('TAMIL SYLLABLE NNOO', '\u0BA3\u0BCB'), + ('TAMIL CONSONANT KSS', '\u0B95\u0BCD\u0BB7\u0BCD'), + ] + for seqname, codepoints in sequences: + self.assertEqual(self.db.lookup(seqname), codepoints) + with self.assertRaises(KeyError): + self.db.ucd_3_2_0.lookup(seqname) + + def test_errors(self): + self.assertRaises(TypeError, self.db.name) + self.assertRaises(TypeError, self.db.name, 'xx') + self.assertRaises(TypeError, self.db.lookup) + self.assertRaises(KeyError, self.db.lookup, 'unknown') + def test_name_inverse_lookup(self): for i in range(sys.maxunicode + 1): char = chr(i) @@ -75,6 +127,13 @@ def test_digit(self): self.assertEqual(self.db.digit('\U00020000', None), None) self.assertEqual(self.db.digit('\U0001D7FD'), 7) + # New in 13.0.0 + self.assertEqual(self.db.digit('\U0001fbf9', None), 9) + # New in 14.0.0 + self.assertEqual(self.db.digit('\U00016ac9', None), 9) + # New in 15.0.0 + self.assertEqual(self.db.digit('\U0001e4f9', None), 9) + self.assertRaises(TypeError, self.db.digit) self.assertRaises(TypeError, self.db.digit, 'xx') self.assertRaises(ValueError, self.db.digit, 'x') @@ -84,9 +143,28 @@ def test_numeric(self): self.assertEqual(self.db.numeric('9'), 9) self.assertEqual(self.db.numeric('\u215b'), 0.125) self.assertEqual(self.db.numeric('\u2468'), 9.0) - self.assertEqual(self.db.numeric('\ua627'), 7.0) self.assertEqual(self.db.numeric('\U00020000', None), None) - self.assertEqual(self.db.numeric('\U0001012A'), 9000) + + # New in 4.1.0 + self.assertEqual(self.db.numeric('\U0001012A', None), 9000) + # Changed in 4.1.0 + self.assertEqual(self.db.numeric('\u5793', None), None) + # New in 5.0.0 + self.assertEqual(self.db.numeric('\u07c0', None), 0.0) + # New in 5.1.0 + self.assertEqual(self.db.numeric('\ua627', None), 7.0) + # Changed in 5.2.0 + self.assertEqual(self.db.numeric('\u09f6'), 3/16) + # New in 6.0.0 + self.assertEqual(self.db.numeric('\u0b72', None), 0.25) + # New in 12.0.0 + self.assertEqual(self.db.numeric('\U0001ed3c', None), 0.5) + # New in 13.0.0 + self.assertEqual(self.db.numeric('\U0001fbf9', None), 9) + # New in 14.0.0 + self.assertEqual(self.db.numeric('\U00016ac9', None), 9) + # New in 15.0.0 + self.assertEqual(self.db.numeric('\U0001e4f9', None), 9) self.assertRaises(TypeError, self.db.numeric) self.assertRaises(TypeError, self.db.numeric, 'xx') @@ -100,6 +178,18 @@ def test_decimal(self): self.assertEqual(self.db.decimal('\U00020000', None), None) self.assertEqual(self.db.decimal('\U0001D7FD'), 7) + # New in 4.1.0 + self.assertEqual(self.db.decimal('\xb2', None), None) + self.assertEqual(self.db.decimal('\u1369', None), None) + # New in 5.0.0 + self.assertEqual(self.db.decimal('\u07c0', None), 0) + # New in 13.0.0 + self.assertEqual(self.db.decimal('\U0001fbf9', None), 9) + # New in 14.0.0 + self.assertEqual(self.db.decimal('\U00016ac9', None), 9) + # New in 15.0.0 + self.assertEqual(self.db.decimal('\U0001e4f9', None), 9) + self.assertRaises(TypeError, self.db.decimal) self.assertRaises(TypeError, self.db.decimal, 'xx') self.assertRaises(ValueError, self.db.decimal, 'x') @@ -109,7 +199,21 @@ def test_category(self): self.assertEqual(self.db.category('a'), 'Ll') self.assertEqual(self.db.category('A'), 'Lu') self.assertEqual(self.db.category('\U00020000'), 'Lo') + + # New in 4.1.0 self.assertEqual(self.db.category('\U0001012A'), 'No') + self.assertEqual(self.db.category('\U000e01ef'), 'Mn') + # New in 5.1.0 + self.assertEqual(self.db.category('\u0374'), 'Lm') + # Changed in 13.0.0 + self.assertEqual(self.db.category('\u0b55'), 'Mn') + self.assertEqual(self.db.category('\U0003134a'), 'Lo') + # Changed in 14.0.0 + self.assertEqual(self.db.category('\u061d'), 'Po') + self.assertEqual(self.db.category('\U0002b738'), 'Lo') + # Changed in 15.0.0 + self.assertEqual(self.db.category('\u0cf3'), 'Mc') + self.assertEqual(self.db.category('\U000323af'), 'Lo') self.assertRaises(TypeError, self.db.category) self.assertRaises(TypeError, self.db.category, 'xx') @@ -136,6 +240,16 @@ def test_mirrored(self): self.assertEqual(self.db.mirrored('\u2201'), 1) self.assertEqual(self.db.mirrored('\U00020000'), 0) + # New in 5.0.0 + self.assertEqual(self.db.mirrored('\u0f3a'), 1) + self.assertEqual(self.db.mirrored('\U0001d7c3'), 1) + # New in 11.0.0 + self.assertEqual(self.db.mirrored('\u29a1'), 0) + # New in 14.0.0 + self.assertEqual(self.db.mirrored('\u2e5c'), 1) + # New in 16.0.0 + self.assertEqual(self.db.mirrored('\u226D'), 1) + self.assertRaises(TypeError, self.db.mirrored) self.assertRaises(TypeError, self.db.mirrored, 'xx') @@ -145,15 +259,61 @@ def test_combining(self): self.assertEqual(self.db.combining('\u20e1'), 230) self.assertEqual(self.db.combining('\U00020000'), 0) + # New in 4.1.0 + self.assertEqual(self.db.combining('\u0350'), 230) + # New in 9.0.0 + self.assertEqual(self.db.combining('\U0001e94a'), 7) + # New in 13.0.0 + self.assertEqual(self.db.combining('\u1abf'), 220) + self.assertEqual(self.db.combining('\U00016ff1'), 6) + # New in 14.0.0 + self.assertEqual(self.db.combining('\u0c3c'), 7) + self.assertEqual(self.db.combining('\U0001e2ae'), 230) + # New in 15.0.0 + self.assertEqual(self.db.combining('\U00010efd'), 220) + # New in 16.0.0 + self.assertEqual(self.db.combining('\u0897'), 230) + # New in 17.0.0 + self.assertEqual(self.db.combining('\u1ACF'), 230) + self.assertRaises(TypeError, self.db.combining) self.assertRaises(TypeError, self.db.combining, 'xx') - def test_normalize(self): - self.assertRaises(TypeError, self.db.normalize) - self.assertRaises(ValueError, self.db.normalize, 'unknown', 'xx') - self.assertEqual(self.db.normalize('NFKC', ''), '') - # The rest can be found in test_normalization.py - # which requires an external file. + def test_no_names_in_pua(self): + puas = [*range(0xe000, 0xf8ff), + *range(0xf0000, 0xfffff), + *range(0x100000, 0x10ffff)] + for i in puas: + char = chr(i) + self.assertRaises(ValueError, self.db.name, char) + + def test_long_combining_mark_run(self): + # gh-149079: avoid quadratic canonical ordering. + payload = "a" + ("\u0300\u0327" * 32) + nfd = "a" + ("\u0327" * 32) + ("\u0300" * 32) + nfc = "\u00e0" + ("\u0327" * 32) + ("\u0300" * 31) + + self.assertEqual(self.db.normalize("NFD", payload), nfd) + self.assertEqual(self.db.normalize("NFKD", payload), nfd) + self.assertEqual(self.db.normalize("NFC", payload), nfc) + self.assertEqual(self.db.normalize("NFKC", payload), nfc) + + def test_combining_mark_run_fast_paths(self): + # gh-149079: cover short runs and already-sorted long runs. + short_payload = "a" + ("\u0300\u0327" * 9) + "\u0300" + short_nfd = "a" + ("\u0327" * 9) + ("\u0300" * 10) + short_nfc = "\u00e0" + ("\u0327" * 9) + ("\u0300" * 9) + long_sorted = "a" + ("\u0327" * 30) + ("\u0300" * 30) + long_sorted_nfc = "\u00e0" + ("\u0327" * 30) + ("\u0300" * 29) + + self.assertEqual(self.db.normalize("NFD", short_payload), short_nfd) + self.assertEqual(self.db.normalize("NFKD", short_payload), short_nfd) + self.assertEqual(self.db.normalize("NFC", short_payload), short_nfc) + self.assertEqual(self.db.normalize("NFKC", short_payload), short_nfc) + self.assertEqual(self.db.normalize("NFD", long_sorted), long_sorted) + self.assertEqual(self.db.normalize("NFKD", long_sorted), long_sorted) + self.assertEqual(self.db.normalize("NFC", long_sorted), long_sorted_nfc) + self.assertEqual(self.db.normalize("NFKC", long_sorted), long_sorted_nfc) def test_pr29(self): # http://www.unicode.org/review/pr-29.html @@ -270,7 +430,7 @@ def test_decimal_numeric_consistent(self): # i.e. if a character has a decimal value, # its numeric value should be the same. count = 0 - for i in range(0x10000): + for i in range(sys.maxunicode + 1): c = chr(i) dec = self.db.decimal(c, -1) if dec != -1: @@ -283,7 +443,7 @@ def test_digit_numeric_consistent(self): # i.e. if a character has a digit value, # its numeric value should be the same. count = 0 - for i in range(0x10000): + for i in range(sys.maxunicode + 1): c = chr(i) dec = self.db.digit(c, -1) if dec != -1: @@ -333,5 +493,77 @@ def test_linebreak_7643(self): self.assertEqual(len(lines), 1, r"\u%.4x should not be a linebreak" % i) +class NormalizationTest(UnicodeDatabaseTest): + # Adapted from CPython's Lib/test/test_unicodedata.py (9ab004d41e). + # Setup downloads the data; never skip it. unicodedata2 has no is_normalized. + @staticmethod + def unistr(data): + data = [int(x, 16) for x in data.split(" ")] + return "".join([chr(x) for x in data]) + + def test_normalization(self): + self.check_normalization(self.db) + + def test_normalization_3_2_0(self): + self.check_normalization(self.db.ucd_3_2_0) + + def check_normalization(self, ucd): + filename = 'NormalizationTest-%s.txt' % ucd.unidata_version + path = Path(__file__).with_name('data') / filename + with path.open(encoding='utf-8') as testdata: + self.assertEqual(testdata.readline().strip(), '# ' + filename) + self.run_normalization_tests(testdata, ucd) + + def run_normalization_tests(self, testdata, ucd): + part = None + part1_data = set() + + NFC = partial(ucd.normalize, "NFC") + NFKC = partial(ucd.normalize, "NFKC") + NFD = partial(ucd.normalize, "NFD") + NFKD = partial(ucd.normalize, "NFKD") + + for line in testdata: + if '#' in line: + line = line.split('#')[0] + line = line.strip() + if not line: + continue + if line.startswith("@Part"): + part = line.split()[0] + continue + c1,c2,c3,c4,c5 = [self.unistr(x) for x in line.split(';')[:-1]] + + # Perform tests + self.assertTrue(c2 == NFC(c1) == NFC(c2) == NFC(c3), line) + self.assertTrue(c4 == NFC(c4) == NFC(c5), line) + self.assertTrue(c3 == NFD(c1) == NFD(c2) == NFD(c3), line) + self.assertTrue(c5 == NFD(c4) == NFD(c5), line) + self.assertTrue(c4 == NFKC(c1) == NFKC(c2) == \ + NFKC(c3) == NFKC(c4) == NFKC(c5), + line) + self.assertTrue(c5 == NFKD(c1) == NFKD(c2) == \ + NFKD(c3) == NFKD(c4) == NFKD(c5), + line) + + # Record part 1 data + if part == "@Part1": + part1_data.add(c1) + + # Perform tests for all other data + for X in map(chr, range(sys.maxunicode + 1)): + if X in part1_data: + continue + self.assertTrue(X == NFC(X) == NFD(X) == NFKC(X) == NFKD(X), ord(X)) + + def test_edge_cases(self): + self.assertRaises(TypeError, self.db.normalize) + self.assertRaises(ValueError, self.db.normalize, 'unknown', 'xx') + self.assertEqual(self.db.normalize('NFKC', ''), '') + + def test_bug_834676(self): + # Check for bug 834676 + self.db.normalize('NFC', '\ud55c\uae00') + if __name__ == "__main__": unittest.main() diff --git a/tox.ini b/tox.ini index 2f06dd9..4be04a2 100644 --- a/tox.ini +++ b/tox.ini @@ -5,4 +5,5 @@ skip_missing_interpreters = true [testenv] deps = pytest changedir = tests +commands_pre = python download_test_data.py commands = pytest {posargs} From 619968e1a36c22a9cecc7358b04294e80ecc6cb9 Mon Sep 17 00:00:00 2001 From: Cosimo Lupo Date: Thu, 17 Sep 2026 12:56:36 +0100 Subject: [PATCH 2/2] Backport CPython's canonical ordering sort Port the normalization sorting changes from CPython 991224b1e8 and 90748760d3 (gh-149079). Already-sorted combining-character runs are skipped; unsorted runs use stable insertion sort below 20 characters and stable counting sort for longer runs, avoiding quadratic canonical ordering. Both paths sort the temporary Py_UCS4 buffer before creating the Python string. This also fixes lost mark reordering on PyPy, where writes to PyUnicode_DATA after materialization were not reflected in the returned string. Keep this backport's self null check and resize a temporary scratch-buffer pointer, so allocation failure preserves the old pointer for cleanup. Otherwise the helpers and normalization function follow CPython 9ab004d41e. All 34 tests pass on CPython 3.9, CPython 3.14 and PyPy 3.11, including the Unicode 17 and 3.2 conformance corpora. Mixed-mark and threshold-boundary probes match CPython's normalization results. --- CHANGELOG.md | 1 + unicodedata2/unicodedata.c | 146 ++++++++++++++++++++++++++++--------- 2 files changed, 113 insertions(+), 34 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 842bfde..f23768f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## Unreleased + - Backport CPython's faster canonical ordering, also fixing Unicode normalization on PyPy. - Require Python 3.9 or newer; drop Python 3.8 support. ## 17.0.0 diff --git a/unicodedata2/unicodedata.c b/unicodedata2/unicodedata.c index 5c22061..17a6761 100644 --- a/unicodedata2/unicodedata.c +++ b/unicodedata2/unicodedata.c @@ -492,19 +492,73 @@ get_decomp_record(PyObject *self, Py_UCS4 code, int *index, int *prefix, int *co #define NCount (VCount*TCount) #define SCount (LCount*NCount) +#define CANONICAL_ORDERING_COUNTING_SORT_THRESHOLD 20 + +static void +canonical_ordering_sort_insertion(Py_UCS4 *data, Py_ssize_t length) +{ + for (Py_ssize_t i = 1; i < length; i++) { + Py_UCS4 code = data[i]; + unsigned char combining = _getrecord_ex(code)->combining; + Py_ssize_t j = i; + + while (j > 0) { + Py_UCS4 previous = data[j - 1]; + if (_getrecord_ex(previous)->combining <= combining) { + break; + } + data[j] = previous; + j--; + } + if (j != i) { + data[j] = code; + } + } +} + +static void +canonical_ordering_sort_counting(Py_UCS4 *data, Py_ssize_t length, + Py_UCS4 *sortbuf) +{ + Py_ssize_t counts[256] = {0}; + Py_ssize_t total = 0; + + for (Py_ssize_t i = 0; i < length; i++) { + Py_UCS4 code = data[i]; + unsigned char combining = _getrecord_ex(code)->combining; + counts[combining]++; + } + + for (size_t i = 0; i < Py_ARRAY_LENGTH(counts); i++) { + Py_ssize_t count = counts[i]; + counts[i] = total; + total += count; + } + + /* Reuse counts[] as the next output slot for each CCC. */ + for (Py_ssize_t i = 0; i < length; i++) { + Py_UCS4 code = data[i]; + unsigned char combining = _getrecord_ex(code)->combining; + sortbuf[counts[combining]++] = code; + } + memcpy(data, sortbuf, length * sizeof(Py_UCS4)); +} + static PyObject* nfd_nfkd(PyObject *self, PyObject *input, int k) { PyObject *result; Py_UCS4 *output; Py_ssize_t i, o, osize; - int kind; - void *data; + int input_kind; + const void *input_data; /* Longest decomposition in Unicode 3.2: U+FDFA */ Py_UCS4 stack[20]; Py_ssize_t space, isize; int index, prefix, count, stackptr; unsigned char prev, cur; + Py_UCS4 *sortbuf = NULL; + Py_ssize_t sortbuflen = 0; stackptr = 0; isize = PyUnicode_GET_LENGTH(input); @@ -524,11 +578,11 @@ nfd_nfkd(PyObject *self, PyObject *input, int k) return NULL; } i = o = 0; - kind = PyUnicode_KIND(input); - data = PyUnicode_DATA(input); + input_kind = PyUnicode_KIND(input); + input_data = PyUnicode_DATA(input); while (i < isize) { - stack[stackptr++] = PyUnicode_READ(kind, data, i++); + stack[stackptr++] = PyUnicode_READ(input_kind, input_data, i++); while(stackptr) { Py_UCS4 code = stack[--stackptr]; /* Hangul Decomposition adds three characters in @@ -545,7 +599,9 @@ nfd_nfkd(PyObject *self, PyObject *input, int k) } output = new_output; } - /* Hangul Decomposition. */ + // Hangul Decomposition. + // See section 3.12.2, "Hangul Syllable Decomposition" + // https://www.unicode.org/versions/latest/core-spec/chapter-3/#G56669 if (SBase <= code && code < (SBase+SCount)) { int SIndex = code - SBase; int L = LBase + SIndex / NCount; @@ -588,40 +644,62 @@ nfd_nfkd(PyObject *self, PyObject *input, int k) } } - result = PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND, - output, o); - PyMem_Free(output); - if (!result) - return NULL; - /* result is guaranteed to be ready, as it is compact. */ - kind = PyUnicode_KIND(result); - data = PyUnicode_DATA(result); - - /* Sort canonically. */ + /* Sort each consecutive combining-character run canonically. */ i = 0; - prev = _getrecord_ex(PyUnicode_READ(kind, data, i))->combining; - for (i++; i < PyUnicode_GET_LENGTH(result); i++) { - cur = _getrecord_ex(PyUnicode_READ(kind, data, i))->combining; - if (prev == 0 || cur == 0 || prev <= cur) { - prev = cur; + while (i < o) { + Py_ssize_t run_length, run_start; + int needs_sort = 0; + + Py_UCS4 ch = output[i]; + prev = _getrecord_ex(ch)->combining; + if (prev == 0) { + i++; continue; } - /* Non-canonical order. Need to switch *i with previous. */ - o = i - 1; - while (1) { - Py_UCS4 tmp = PyUnicode_READ(kind, data, o+1); - PyUnicode_WRITE(kind, data, o+1, - PyUnicode_READ(kind, data, o)); - PyUnicode_WRITE(kind, data, o, tmp); - o--; - if (o < 0) - break; - prev = _getrecord_ex(PyUnicode_READ(kind, data, o))->combining; - if (prev == 0 || prev <= cur) + + run_start = i++; + while (i < o) { + Py_UCS4 ch = output[i]; + cur = _getrecord_ex(ch)->combining; + if (cur == 0) { break; + } + if (prev > cur) { + needs_sort = 1; + } + prev = cur; + i++; + } + if (!needs_sort) { + continue; + } + + run_length = i - run_start; + if (run_length < CANONICAL_ORDERING_COUNTING_SORT_THRESHOLD) { + canonical_ordering_sort_insertion(output + run_start, run_length); + continue; } - prev = _getrecord_ex(PyUnicode_READ(kind, data, i))->combining; + + if (run_length > sortbuflen) { + /* PyMem_Resize overwrites its argument even on failure. */ + Py_UCS4 *new_sortbuf = sortbuf; + PyMem_Resize(new_sortbuf, Py_UCS4, run_length); + if (new_sortbuf == NULL) { + PyErr_NoMemory(); + PyMem_Free(sortbuf); + PyMem_Free(output); + return NULL; + } + sortbuf = new_sortbuf; + sortbuflen = run_length; + } + + canonical_ordering_sort_counting(output + run_start, run_length, + sortbuf); } + PyMem_Free(sortbuf); + result = PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND, output, o); + PyMem_Free(output); return result; }