Blame SOURCES/00211-pep466-UTF-7-decoder-fix-illegal-unicode.patch

f63228
f63228
# HG changeset patch
f63228
# User Serhiy Storchaka <storchaka@gmail.com>
f63228
# Date 1382204269 -10800
f63228
# Node ID 214c0aac7540947d88a38ff0061734547ef86710
f63228
# Parent  c207ac413457a1b834e4b7dcf1a6836cd6e036e3
f63228
Issue #19279: UTF-7 decoder no more produces illegal unicode strings.
f63228
f63228
diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py
f63228
--- a/Lib/test/test_codecs.py
f63228
+++ b/Lib/test/test_codecs.py
f63228
@@ -611,6 +611,35 @@ class UTF7Test(ReadTest):
f63228
             ]
f63228
         )
f63228
 
f63228
+    def test_errors(self):
f63228
+        tests = [
f63228
+            ('a\xffb', u'a\ufffdb'),
f63228
+            ('a+IK', u'a\ufffd'),
f63228
+            ('a+IK-b', u'a\ufffdb'),
f63228
+            ('a+IK,b', u'a\ufffdb'),
f63228
+            ('a+IKx', u'a\u20ac\ufffd'),
f63228
+            ('a+IKx-b', u'a\u20ac\ufffdb'),
f63228
+            ('a+IKwgr', u'a\u20ac\ufffd'),
f63228
+            ('a+IKwgr-b', u'a\u20ac\ufffdb'),
f63228
+            ('a+IKwgr,', u'a\u20ac\ufffd'),
f63228
+            ('a+IKwgr,-b', u'a\u20ac\ufffd-b'),
f63228
+            ('a+IKwgrB', u'a\u20ac\u20ac\ufffd'),
f63228
+            ('a+IKwgrB-b', u'a\u20ac\u20ac\ufffdb'),
f63228
+            ('a+/,+IKw-b', u'a\ufffd\u20acb'),
f63228
+            ('a+//,+IKw-b', u'a\ufffd\u20acb'),
f63228
+            ('a+///,+IKw-b', u'a\uffff\ufffd\u20acb'),
f63228
+            ('a+////,+IKw-b', u'a\uffff\ufffd\u20acb'),
f63228
+        ]
f63228
+        for raw, expected in tests:
f63228
+            self.assertRaises(UnicodeDecodeError, codecs.utf_7_decode,
f63228
+                              raw, 'strict', True)
f63228
+            self.assertEqual(raw.decode('utf-7', 'replace'), expected)
f63228
+
f63228
+    def test_nonbmp(self):
f63228
+        self.assertEqual(u'\U000104A0'.encode(self.encoding), '+2AHcoA-')
f63228
+        self.assertEqual(u'\ud801\udca0'.encode(self.encoding), '+2AHcoA-')
f63228
+        self.assertEqual('+2AHcoA-'.decode(self.encoding), u'\U000104A0')
f63228
+
f63228
 class UTF16ExTest(unittest.TestCase):
f63228
 
f63228
     def test_errors(self):
f63228
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c
f63228
--- a/Objects/unicodeobject.c
f63228
+++ b/Objects/unicodeobject.c
f63228
@@ -1671,6 +1671,7 @@ PyObject *PyUnicode_DecodeUTF7Stateful(c
f63228
                                        (base64buffer >> (base64bits-16));
f63228
                     base64bits -= 16;
f63228
                     base64buffer &= (1 << base64bits) - 1; /* clear high bits */
f63228
+                    assert(outCh <= 0xffff);
f63228
                     if (surrogate) {
f63228
                         /* expecting a second surrogate */
f63228
                         if (outCh >= 0xDC00 && outCh <= 0xDFFF) {
f63228
@@ -1737,6 +1738,7 @@ PyObject *PyUnicode_DecodeUTF7Stateful(c
f63228
                 inShift = 1;
f63228
                 shiftOutStart = p;
f63228
                 base64bits = 0;
f63228
+                base64buffer = 0;
f63228
             }
f63228
         }
f63228
         else if (DECODE_DIRECT(ch)) { /* character decodes as itself */
f63228