from typing import List
def utf8_encode(text: str) -> List[int]:
"""Text as UTF-8 bytes.
A Python str can hold a lone surrogate (json.loads("\\"\\\\ud800\\"") makes
one), which str.encode refuses; it becomes U+FFFD here, as TextEncoder
does in the browser, so the three languages agree.
"""
if not isinstance(text, str):
raise TypeError("text must be a string")
out = []
for ch in text:
cp = ord(ch)
if 0xD800 <= cp <= 0xDFFF:
cp = 0xFFFD
if cp < 0x80:
out.append(cp)
elif cp < 0x800:
out.extend((0xC0 | (cp >> 6), 0x80 | (cp & 63)))
elif cp < 0x10000:
out.extend((0xE0 | (cp >> 12), 0x80 | ((cp >> 6) & 63), 0x80 | (cp & 63)))
else:
out.extend((0xF0 | (cp >> 18), 0x80 | ((cp >> 12) & 63), 0x80 | ((cp >> 6) & 63), 0x80 | (cp & 63)))
return out