1515:status: Alpha
1616"""
1717
18- from __future__ import unicode_literals
19-
2018import sys
2119
2220# Module version
2624# Documentation strings format
2725__docformat__ = "restructuredtext en"
2826
27+ # Note: this module deliberately does NOT `from __future__ import
28+ # unicode_literals`. UnicodeDecodeError requires its encoding-name,
29+ # object and reason arguments to be the *native* str type; on Python 2
30+ # that means plain bytes-string literals, not unicode ones.
31+
2932# Encoding name: not cesu-8, which uses a different zero-byte
3033NAME = "mutf8"
3134
@@ -170,14 +173,19 @@ def decoder(data):
170173
171174 def next_byte (_it , start , count ):
172175 try :
173- return next (_it )[1 ]
176+ return byte_to_int ( next (_it )[1 ])
174177 except StopIteration :
175178 raise UnicodeDecodeError (
176179 NAME , data , start , start + count , "incomplete byte sequence"
177180 )
178181
179182 it = iter (enumerate (data ))
180- for i , d in it :
183+ for i , raw_d in it :
184+ # Iterating a Python 3 bytes object yields ints, but iterating a
185+ # Python 2 str yields single-character strings: normalize here so
186+ # the bitwise logic below works the same on both, while `data`
187+ # itself stays a real bytes-like object for UnicodeDecodeError.
188+ d = byte_to_int (raw_d )
181189 if d == 0x00 : # 00000000
182190 raise UnicodeDecodeError (
183191 NAME , data , i , i + 1 , "embedded zero-byte not allowed"
@@ -196,6 +204,11 @@ def next_byte(_it, start, count):
196204 for i1 , dm in enumerate (DECODE_MAP [6 ]):
197205 d1 = next_byte (it , i , i1 + 1 )
198206 value = dm .apply (d1 , value , data , i , i1 + 1 )
207+ # The 6 bytes reconstruct the supplementary
208+ # character's 20-bit offset from the surrogate
209+ # pair; add back the base to get the real
210+ # code point (U+10000..U+10FFFF).
211+ value += 0x10000
199212 else : # 1110xxxx
200213 value = d & 0x0F
201214 for i1 , dm in enumerate (DECODE_MAP [3 ]):
@@ -228,7 +241,11 @@ def decode_modified_utf8(data, errors="strict"):
228241 :raises UnicodeDecodeError: sequence is invalid.
229242 """
230243 value , length = "" , 0
231- it = iter (decoder (byte_to_int (d ) for d in data ))
244+ # decoder() normalizes each item internally (via byte_to_int) as it
245+ # iterates, so the original bytes-like `data` can be passed directly;
246+ # it also needs to stay a real bytes-like object here, since
247+ # UnicodeDecodeError requires one for the errors decoder() raises.
248+ it = iter (decoder (data ))
232249 while True :
233250 try :
234251 value += next (it )
@@ -242,7 +259,10 @@ def decode_modified_utf8(data, errors="strict"):
242259 if errors == "ignore" :
243260 pass
244261 elif errors == "replace" :
245- value += "\uFFFD "
262+ # Explicit u-prefix: without `unicode_literals` active in
263+ # this module, a plain literal would not interpret \u as
264+ # an escape sequence on Python 2.
265+ value += u"\uFFFD "
246266 length += 1
247267 return value , length
248268
0 commit comments