Skip to content

Commit 398b30e

Browse files
committed
Fixed issues with modified UTF-8
Signed-off-by: Thomas Calmant <thomas.calmant@gmail.com>
1 parent 2177e2c commit 398b30e

1 file changed

Lines changed: 26 additions & 6 deletions

File tree

‎javaobj/modifiedutf8.py‎

Lines changed: 26 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -15,8 +15,6 @@
1515
:status: Alpha
1616
"""
1717

18-
from __future__ import unicode_literals
19-
2018
import sys
2119

2220
# Module version
@@ -26,6 +24,11 @@
2624
# Documentation strings format
2725
__docformat__ = "restructuredtext en"
2826

27+
# Note: this module deliberately does NOT `from __future__ import
28+
# unicode_literals`. UnicodeDecodeError requires its encoding-name,
29+
# object and reason arguments to be the *native* str type; on Python 2
30+
# that means plain bytes-string literals, not unicode ones.
31+
2932
# Encoding name: not cesu-8, which uses a different zero-byte
3033
NAME = "mutf8"
3134

@@ -170,14 +173,19 @@ def decoder(data):
170173

171174
def next_byte(_it, start, count):
172175
try:
173-
return next(_it)[1]
176+
return byte_to_int(next(_it)[1])
174177
except StopIteration:
175178
raise UnicodeDecodeError(
176179
NAME, data, start, start + count, "incomplete byte sequence"
177180
)
178181

179182
it = iter(enumerate(data))
180-
for i, d in it:
183+
for i, raw_d in it:
184+
# Iterating a Python 3 bytes object yields ints, but iterating a
185+
# Python 2 str yields single-character strings: normalize here so
186+
# the bitwise logic below works the same on both, while `data`
187+
# itself stays a real bytes-like object for UnicodeDecodeError.
188+
d = byte_to_int(raw_d)
181189
if d == 0x00: # 00000000
182190
raise UnicodeDecodeError(
183191
NAME, data, i, i + 1, "embedded zero-byte not allowed"
@@ -196,6 +204,11 @@ def next_byte(_it, start, count):
196204
for i1, dm in enumerate(DECODE_MAP[6]):
197205
d1 = next_byte(it, i, i1 + 1)
198206
value = dm.apply(d1, value, data, i, i1 + 1)
207+
# The 6 bytes reconstruct the supplementary
208+
# character's 20-bit offset from the surrogate
209+
# pair; add back the base to get the real
210+
# code point (U+10000..U+10FFFF).
211+
value += 0x10000
199212
else: # 1110xxxx
200213
value = d & 0x0F
201214
for i1, dm in enumerate(DECODE_MAP[3]):
@@ -228,7 +241,11 @@ def decode_modified_utf8(data, errors="strict"):
228241
:raises UnicodeDecodeError: sequence is invalid.
229242
"""
230243
value, length = "", 0
231-
it = iter(decoder(byte_to_int(d) for d in data))
244+
# decoder() normalizes each item internally (via byte_to_int) as it
245+
# iterates, so the original bytes-like `data` can be passed directly;
246+
# it also needs to stay a real bytes-like object here, since
247+
# UnicodeDecodeError requires one for the errors decoder() raises.
248+
it = iter(decoder(data))
232249
while True:
233250
try:
234251
value += next(it)
@@ -242,7 +259,10 @@ def decode_modified_utf8(data, errors="strict"):
242259
if errors == "ignore":
243260
pass
244261
elif errors == "replace":
245-
value += "\uFFFD"
262+
# Explicit u-prefix: without `unicode_literals` active in
263+
# this module, a plain literal would not interpret \u as
264+
# an escape sequence on Python 2.
265+
value += u"\uFFFD"
246266
length += 1
247267
return value, length
248268

0 commit comments

Comments
 (0)