bpo-24214: Fixed the UTF-8 and UTF-16 incremental decoders. (GH-14304)

author Serhiy Storchaka <storchaka@gmail.com>

Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)

committer GitHub <noreply@github.com>

Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)
author Serhiy Storchaka <storchaka@gmail.com>
Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)
committer GitHub <noreply@github.com>
Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)
diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py

index d98f178384abf164f80d8e5e9c195f6b9f3438d4..4317dfceb0399e831a35ee75e78b6828be940a91 100644 (file)
--- a/Lib/test/test_codecs.py
+++ b/Lib/test/test_codecs.py
@@ -429,11 +429,18 @@ class ReadTest(MixInCheckStateHandling):
      def test_incremental_surrogatepass(self):
          # Test incremental decoder for surrogatepass handler:
          # see issue #24214
+        # High surrogate
          data = '\uD901'.encode(self.encoding, 'surrogatepass')
          for i in range(1, len(data)):
              dec = codecs.getincrementaldecoder(self.encoding)('surrogatepass')
              self.assertEqual(dec.decode(data[:i]), '')
              self.assertEqual(dec.decode(data[i:], True), '\uD901')
+        # Low surrogate
+        data = '\uDC02'.encode(self.encoding, 'surrogatepass')
+        for i in range(1, len(data)):
+            dec = codecs.getincrementaldecoder(self.encoding)('surrogatepass')
+            self.assertEqual(dec.decode(data[:i]), '')
+            self.assertEqual(dec.decode(data[i:]), '\uDC02')
  
  
  class UTF32Test(ReadTest, unittest.TestCase):
@@ -874,6 +881,23 @@ class UTF8Test(ReadTest, unittest.TestCase):
          with self.assertRaises(UnicodeDecodeError):
              b"abc\xed\xa0z".decode(self.encoding, "surrogatepass")
  
+    def test_incremental_errors(self):
+        # Test that the incremental decoder can fail with final=False.
+        # See issue #24214
+        cases = [b'\x80', b'\xBF', b'\xC0', b'\xC1', b'\xF5', b'\xF6', b'\xFF']
+        for prefix in (b'\xC2', b'\xDF', b'\xE0', b'\xE0\xA0', b'\xEF',
+                       b'\xEF\xBF', b'\xF0', b'\xF0\x90', b'\xF0\x90\x80',
+                       b'\xF4', b'\xF4\x8F', b'\xF4\x8F\xBF'):
+            for suffix in b'\x7F', b'\xC0':
+                cases.append(prefix + suffix)
+        cases.extend((b'\xE0\x80', b'\xE0\x9F', b'\xED\xA0\x80',
+                      b'\xED\xBF\xBF', b'\xF0\x80', b'\xF0\x8F', b'\xF4\x90'))
+
+        for data in cases:
+            with self.subTest(data=data):
+                dec = codecs.getincrementaldecoder(self.encoding)()
+                self.assertRaises(UnicodeDecodeError, dec.decode, data)
+
  
  class UTF7Test(ReadTest, unittest.TestCase):
      encoding = "utf-7"
diff --git a/Misc/NEWS.d/next/Core and Builtins/2019-06-22-12-45-20.bpo-24214.hIiHeD.rst b/Misc/NEWS.d/next/Core and Builtins/2019-06-22-12-45-20.bpo-24214.hIiHeD.rst

new file mode 100644 (file)

index 0000000..2d70ce0
--- /dev/null
+++ b/Misc/NEWS.d/next/Core and Builtins/2019-06-22-12-45-20.bpo-24214.hIiHeD.rst
@@ -0,0 +1,2 @@
+Improved support of the surrogatepass error handler in the UTF-8 and UTF-16
+incremental decoders.
diff --git a/Objects/stringlib/codecs.h b/Objects/stringlib/codecs.h

index 8645bc26cff8cac5075e56645de0b479f9a5c780..d6f2b98f2b30a33f9e69243f720cc8ce3db9d015 100644 (file)
--- a/Objects/stringlib/codecs.h
+++ b/Objects/stringlib/codecs.h
@@ -207,7 +207,7 @@ STRINGLIB(utf8_decode)(const char **inptr, const char *end,
                      goto InvalidContinuation1;
              } else if (ch == 0xF4 && ch2 >= 0x90) {
                  /* invalid sequence
-                   \xF4\x90\x80\80- -- 110000- overflow */
+                   \xF4\x90\x80\x80- -- 110000- overflow */
                  goto InvalidContinuation1;
              }
              if (!IS_CONTINUATION_BYTE(ch3)) {
@@ -573,10 +573,10 @@ STRINGLIB(utf16_decode)(const unsigned char **inptr, const unsigned char *e,
          }
  
          /* UTF-16 code pair: */
-        if (q >= e)
-            goto UnexpectedEnd;
          if (!Py_UNICODE_IS_HIGH_SURROGATE(ch))
              goto IllegalEncoding;
+        if (q >= e)
+            goto UnexpectedEnd;
          ch2 = (q[ihi] << 8) | q[ilo];
          q += 2;
          if (!Py_UNICODE_IS_LOW_SURROGATE(ch2))
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c

index 625be4b5594b153cfe7ebbc7aa59d72c406ec5a2..cb1456ea847acfd813d3c5f37c46e9692789fe43 100644 (file)
--- a/Objects/unicodeobject.c
+++ b/Objects/unicodeobject.c
@@ -4945,11 +4945,15 @@ unicode_decode_utf8(const char *s, Py_ssize_t size,
              endinpos = startinpos + 1;
              break;
          case 2:
-        case 3:
-        case 4:
-            if (s == end || consumed) {
+            if (consumed && (unsigned char)s[0] == 0xED && end - s == 2
+                && (unsigned char)s[1] >= 0xA0 && (unsigned char)s[1] <= 0xBF)
+            {
+                /* Truncated surrogate code in range D800-DFFF */
                  goto End;
              }
+            /* fall through */
+        case 3:
+        case 4:
              errmsg = "invalid continuation byte";
              startinpos = s - starts;
              endinpos = startinpos + ch - 1;
author	Serhiy Storchaka <storchaka@gmail.com>
	Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)
committer	GitHub <noreply@github.com>
	Tue, 25 Jun 2019 08:54:18 +0000 (11:54 +0300)
Lib/test/test_codecs.py		patch \| blob \| history
Misc/NEWS.d/next/Core and Builtins/2019-06-22-12-45-20.bpo-24214.hIiHeD.rst	[new file with mode: 0644]	patch \| blob
Objects/stringlib/codecs.h		patch \| blob \| history
Objects/unicodeobject.c		patch \| blob \| history