Functional Weave
Code in Python

text.initials@1.0.0

impl/python.py

2,366 bytes · the Python implementation · view raw

from typing import List

#: Skipped, not counted, in the middle of a name: "Ludwig van Beethoven" is "LB".
_PARTICLES = frozenset([
    "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du",
    "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn",
])


def _is_whitespace(ch: str) -> bool:
    """The Unicode White_Space property, spelled out so all three languages
    agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust
    do not.
    """
    cp = ord(ch)
    return (
        0x09 <= cp <= 0x0D
        or cp == 0x20
        or cp == 0x85
        or cp == 0xA0
        or cp == 0x1680
        or 0x2000 <= cp <= 0x200A
        or cp == 0x2028
        or cp == 0x2029
        or cp == 0x202F
        or cp == 0x205F
        or cp == 0x3000
    )


def _is_skippable(ch: str) -> bool:
    # ASCII punctuation and typographic quotes are skipped at the start of a part.
    cp = ord(ch)
    if cp < 0x80:
        return not ("0" <= ch <= "9" or "A" <= ch <= "Z" or "a" <= ch <= "z")
    return cp in (0x2018, 0x2019, 0x201C, 0x201D)


def _upper_one(ch: str) -> str:
    # One-to-one mappings only, so "ß" stays "ß" in all three languages.
    mapped = ch.upper()
    return mapped if len(mapped) == 1 else ch


def _lower_one(ch: str) -> str:
    mapped = ch.lower()
    return mapped if len(mapped) == 1 else ch


def initials(name: str) -> str:
    """Initials from a personal name: "Jean-Paul Sartre" is "JPS"."""
    if not isinstance(name, str):
        raise TypeError("initials needs a string, received %r" % (name,))

    words: List[str] = []
    current = ""
    for ch in name:
        if _is_whitespace(ch):
            if current:
                words.append(current)
            current = ""
        else:
            current += ch
    if current:
        words.append(current)

    out = ""
    for i, word in enumerate(words):
        if 0 < i < len(words) - 1 and "".join(_lower_one(c) for c in word) in _PARTICLES:
            continue
        # Hyphens and full stops split a word into parts that each give a letter.
        at_part_start = True
        for ch in word:
            if ch == "-" or ch == ".":
                at_part_start = True
            elif at_part_start and not _is_skippable(ch):
                out += _upper_one(ch)
                at_part_start = False
    return out