2,366 bytes · the Python implementation · view raw
from typing import List
#: Skipped, not counted, in the middle of a name: "Ludwig van Beethoven" is "LB".
_PARTICLES = frozenset([
"van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du",
"la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn",
])
def _is_whitespace(ch: str) -> bool:
"""The Unicode White_Space property, spelled out so all three languages agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust do not. """
cp = ord(ch)
return (
0x09 <= cp <= 0x0Dor cp == 0x20or cp == 0x85or cp == 0xA0or cp == 0x1680or0x2000 <= cp <= 0x200Aor cp == 0x2028or cp == 0x2029or cp == 0x202For cp == 0x205For cp == 0x3000
)
def _is_skippable(ch: str) -> bool:
# ASCII punctuation and typographic quotes are skipped at the start of a part.
cp = ord(ch)
if cp < 0x80:
returnnot ("0" <= ch <= "9"or"A" <= ch <= "Z"or"a" <= ch <= "z")
return cp in (0x2018, 0x2019, 0x201C, 0x201D)
def _upper_one(ch: str) -> str:
# One-to-one mappings only, so "ß" stays "ß" in all three languages.
mapped = ch.upper()
return mapped if len(mapped) == 1else ch
def _lower_one(ch: str) -> str:
mapped = ch.lower()
return mapped if len(mapped) == 1else ch
def initials(name: str) -> str:
"""Initials from a personal name: "Jean-Paul Sartre" is "JPS"."""ifnot isinstance(name, str):
raise TypeError("initials needs a string, received %r" % (name,))
words: List[str] = []
current = ""for ch in name:
if _is_whitespace(ch):
if current:
words.append(current)
current = ""else:
current += ch
if current:
words.append(current)
out = ""for i, word in enumerate(words):
if0 < i < len(words) - 1and"".join(_lower_one(c) for c in word) in _PARTICLES:
continue# Hyphens and full stops split a word into parts that each give a letter.
at_part_start = Truefor ch in word:
if ch == "-"or ch == ".":
at_part_start = Trueelif at_part_start andnot _is_skippable(ch):
out += _upper_one(ch)
at_part_start = Falsereturn out