diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index f6f8089a065059..07c97e13969d4b 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -3679,10 +3679,11 @@ def iconv_encoding_available(name): # Encodings iconv may provide but for which CPython has no built-in codec # (cp1047 is EBCDIC, i.e. not ASCII-compatible). _ICONV_ONLY = ['cp1047', 'cp1133', 'GEORGIAN-PS', 'ARMSCII-8'] -# Encodings that leave a shift state pending, so encoding ends with a flush. -_ICONV_SHIFT_STATE = [('ISO-2022-CN-EXT', 'ABC\u4e2dDEF'), - ('ISO-2022-CN', 'ABC\u4e2dDEF'), - ('ISO-2022-JP', 'ABC\u65e5DEF')] +# Stateful encodings, with a character that leaves a shift state pending. +_ICONV_SHIFT_STATE = [('ISO-2022-CN-EXT', '\u4e2d'), + ('ISO-2022-CN', '\u4e2d'), + ('ISO-2022-JP', '\u65e5'), + ('ISO-2022-KR', '\ud55c')] @unittest.skipUnless(hasattr(codecs, 'iconv_encode'), @@ -3735,6 +3736,35 @@ def test_encode_errors(self): self.assertEqual(codecs.iconv_encode(enc, 'a€b', 'xmlcharrefreplace')[0], b'a€b') + def test_encode_errors_shift_state(self): + # A replacement must not land inside a shifted region. + cases = [('replace', '?'), + ('ignore', ''), + ('backslashreplace', '\\U0001f600'), + ('xmlcharrefreplace', '😀')] + tested = False + for enc, char in _ICONV_SHIFT_STATE: + if not iconv_encoding_available(enc): + continue + try: + data = codecs.iconv_encode(enc, char)[0] + except UnicodeEncodeError: + # This iconv cannot represent the character. + continue + if codecs.iconv_decode(enc, data, 'strict', True)[0] != char: + # iconv substituted the character instead of encoding it. + continue + tested = True + text = char + '\U0001f600' + char + for errors, replacement in cases: + with self.subTest(encoding=enc, errors=errors): + data = codecs.iconv_encode(enc, text, errors)[0] + self.assertEqual( + codecs.iconv_decode(enc, data, 'strict', True)[0], + char + replacement + char, data) + if not tested: + self.skipTest('no stateful iconv encoding is available') + def test_decode_errors(self): enc = self.require('ASCII') bad = b'a\xffb' @@ -3827,12 +3857,12 @@ def test_encode_shift_state_flush(self): # An iconv that cannot represent the character either rejects it or # substitutes for it silently, as macOS does for ISO-2022-CN. tested = False - for enc, text in _ICONV_SHIFT_STATE: + for enc, char in _ICONV_SHIFT_STATE: if not iconv_encoding_available(enc): continue with self.subTest(encoding=enc): try: - data = codecs.iconv_encode(enc, text)[0] + data = codecs.iconv_encode(enc, 'ABC' + char + 'DEF')[0] except UnicodeEncodeError: continue tested = True diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index eec02f662e7905..06934e567dd3d7 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -8368,6 +8368,51 @@ iconv_grow_writer(PyBytesWriter *writer, char **pout, char **poutend) return 0; } +/* Return the output and the conversion to the initial shift state; a stateless + encoding writes nothing. Returns 0 on success, -1 on error. */ +static int +iconv_reset_shift_state(iconv_t cd, PyBytesWriter *writer, + char **pout, char **poutend) +{ + for (;;) { + /* A NULL input buffer means the reset, but its size is passed as a + real pointer: some implementations do not accept a NULL there. */ + size_t inleft = 0; + size_t outleft = (size_t)(*poutend - *pout); + /* Only -1 is a failure; a positive result counts nonreversible + conversions. */ + if (iconv(cd, NULL, &inleft, pout, &outleft) != (size_t)-1) { + return 0; + } + if (errno != E2BIG) { + PyErr_SetFromErrno(PyExc_OSError); + return -1; + } + if (iconv_grow_writer(writer, pout, poutend) < 0) { + return -1; + } + } +} + +/* Test whether one code point is encodable, on a descriptor separate from the + one which writes the output. Substituting it (e.g. writing SI in + ISO-2022-KR) changes the shift state, and dropping such output would + desynchronize the descriptor from what is already written. */ +static int +iconv_encodable(iconv_t cd, const char *cp, size_t unit) +{ + char buf[32]; /* enough for one code point with escape sequences */ + char *inptr = (char *)cp; + size_t inleft = unit; + char *out = buf; + size_t outleft = sizeof(buf); + + iconv(cd, NULL, NULL, NULL, NULL); /* start in the initial state */ + /* Only a zero result means that the code point was encoded. */ + /* See the note above on the void* cast of the iconv() input buffer. */ + return iconv(cd, (void *)&inptr, &inleft, &out, &outleft) == 0; +} + /* * Encode a str to bytes with iconv(). * @@ -8435,6 +8480,8 @@ _PyUnicode_EncodeIconv(const char *encoding, PyObject *unicode, const char *uend = data + (size_t)ulen * unit; int flushing = 0; int careful = 0; /* feed one code point per iconv() call */ + /* tests a code point in the careful mode */ + iconv_t probe_cd = (iconv_t)-1; /* A generous initial estimate for the output size. */ writer = PyBytesWriter_Create(ulen + (ulen >> 1) + 16); @@ -8453,12 +8500,24 @@ _PyUnicode_EncodeIconv(const char *encoding, PyObject *unicode, } char *out_before = out; size_t outleft = (size_t)(outend - out); - /* When the whole string is converted, a final iconv() call with a - NULL input flushes any pending shift sequence (e.g. ISO-2022). */ - /* See the note above on the void* cast of the iconv() input buffer. */ - size_t ret = iconv(cd, flushing ? NULL : (void *)&inptr, &inleft, &out, &outleft); - if (!flushing) { - up = inptr; + size_t ret; + if (careful && !flushing + && !iconv_encodable(probe_cd, up, (size_t)unit)) + { + /* Do not feed it to cd, whose shift state has to match + the written output. */ + ret = (size_t)-1; + errno = EILSEQ; + } + else { + /* When the whole string is converted, a final iconv() call with + a NULL input flushes any pending shift sequence (ISO-2022). */ + /* See the note above on the void* cast of the iconv() input. */ + ret = iconv(cd, flushing ? NULL : (void *)&inptr, &inleft, + &out, &outleft); + if (!flushing) { + up = inptr; + } } if (ret != (size_t)-1) { @@ -8472,6 +8531,11 @@ _PyUnicode_EncodeIconv(const char *encoding, PyObject *unicode, if (ret > 0) { if (!careful) { careful = 1; + probe_cd = iconv_open_or_set_error(encoding, source, + encoding); + if (probe_cd == (iconv_t)-1) { + goto done; + } iconv(cd, NULL, NULL, NULL, NULL); out = PyBytesWriter_GetData(writer); outend = out + PyBytesWriter_GetSize(writer); @@ -8531,28 +8595,27 @@ _PyUnicode_EncodeIconv(const char *encoding, PyObject *unicode, replen = PyBytes_GET_SIZE(repbytes); } - while (outend - out < replen) { - if (iconv_grow_writer(writer, &out, &outend) < 0) { - if (repbytes != NULL) { - Py_DECREF(repbytes); - } - else { - Py_DECREF(rep); - } - goto done; - } + /* The replacement is copied verbatim, so the output must return to the + initial shift state first, or it reads back as encoded data. */ + int failed = replen > 0 + && iconv_reset_shift_state(cd, writer, &out, &outend) < 0; + while (!failed && outend - out < replen) { + failed = iconv_grow_writer(writer, &out, &outend) < 0; + } + if (!failed) { + memcpy(out, repdata, replen); + out += replen; } - memcpy(out, repdata, replen); - out += replen; if (repbytes != NULL) { Py_DECREF(repbytes); } else { Py_DECREF(rep); } + if (failed) { + goto done; + } up = ustart + (size_t)newpos * unit; - /* Reset the shift state after the injected replacement bytes. */ - iconv(cd, NULL, NULL, NULL, NULL); } if (PyBytesWriter_Resize(writer, out - (char *)PyBytesWriter_GetData(writer)) < 0) { @@ -8566,6 +8629,9 @@ _PyUnicode_EncodeIconv(const char *encoding, PyObject *unicode, PyBytesWriter_Discard(writer); } iconv_close(cd); + if (probe_cd != (iconv_t)-1) { + iconv_close(probe_cd); + } PyMem_Free(widened); Py_XDECREF(errorHandler); Py_XDECREF(exc);