from typing import List def utf8_encode(text: str) -> List[int]: """Text as UTF-8 bytes. A Python str can hold a lone surrogate (json.loads("\\"\\\\ud800\\"") makes one), which str.encode refuses; it becomes U+FFFD here, as TextEncoder does in the browser, so the three languages agree. """ if not isinstance(text, str): raise TypeError("text must be a string") out = [] for ch in text: cp = ord(ch) if 0xD800 <= cp <= 0xDFFF: cp = 0xFFFD if cp < 0x80: out.append(cp) elif cp < 0x800: out.extend((0xC0 | (cp >> 6), 0x80 | (cp & 63))) elif cp < 0x10000: out.extend((0xE0 | (cp >> 12), 0x80 | ((cp >> 6) & 63), 0x80 | (cp & 63))) else: out.extend((0xF0 | (cp >> 18), 0x80 | ((cp >> 12) & 63), 0x80 | ((cp >> 6) & 63), 0x80 | (cp & 63))) return out