from typing import List #: Skipped, not counted, in the middle of a name: "Ludwig van Beethoven" is "LB". _PARTICLES = frozenset([ "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du", "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn", ]) def _is_whitespace(ch: str) -> bool: """The Unicode White_Space property, spelled out so all three languages agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust do not. """ cp = ord(ch) return ( 0x09 <= cp <= 0x0D or cp == 0x20 or cp == 0x85 or cp == 0xA0 or cp == 0x1680 or 0x2000 <= cp <= 0x200A or cp == 0x2028 or cp == 0x2029 or cp == 0x202F or cp == 0x205F or cp == 0x3000 ) def _is_skippable(ch: str) -> bool: # ASCII punctuation and typographic quotes are skipped at the start of a part. cp = ord(ch) if cp < 0x80: return not ("0" <= ch <= "9" or "A" <= ch <= "Z" or "a" <= ch <= "z") return cp in (0x2018, 0x2019, 0x201C, 0x201D) def _upper_one(ch: str) -> str: # One-to-one mappings only, so "ß" stays "ß" in all three languages. mapped = ch.upper() return mapped if len(mapped) == 1 else ch def _lower_one(ch: str) -> str: mapped = ch.lower() return mapped if len(mapped) == 1 else ch def initials(name: str) -> str: """Initials from a personal name: "Jean-Paul Sartre" is "JPS".""" if not isinstance(name, str): raise TypeError("initials needs a string, received %r" % (name,)) words: List[str] = [] current = "" for ch in name: if _is_whitespace(ch): if current: words.append(current) current = "" else: current += ch if current: words.append(current) out = "" for i, word in enumerate(words): if 0 < i < len(words) - 1 and "".join(_lower_one(c) for c in word) in _PARTICLES: continue # Hyphens and full stops split a word into parts that each give a letter. at_part_start = True for ch in word: if ch == "-" or ch == ".": at_part_start = True elif at_part_start and not _is_skippable(ch): out += _upper_one(ch) at_part_start = False return out