Functional Weave
Code in Python

text.normalise-name@1.0.0

impl/python.py

3,607 bytes · the Python implementation · view raw

from typing import List

#: Characters after which a new capital starts: hyphen, both apostrophes, full stop.
_PART_SEPARATORS = ("-", "'", "’", ".")

#: Lower-cased in the middle of a name: "Ludwig van Beethoven".
_PARTICLES = frozenset([
    "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du",
    "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn",
])

#: Names where "mac" is not a Gaelic prefix, so no capital follows it.
_NOT_MAC = frozenset([
    "macari", "macaulay", "macedo", "macey", "machado", "machell", "machen",
    "machin", "macias", "mackay", "mackey", "mackie", "mackin", "macklin",
])


def _is_whitespace(ch: str) -> bool:
    """The Unicode White_Space property, spelled out so all three languages
    agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust
    do not.
    """
    cp = ord(ch)
    return (
        0x09 <= cp <= 0x0D
        or cp == 0x20
        or cp == 0x85
        or cp == 0xA0
        or cp == 0x1680
        or 0x2000 <= cp <= 0x200A
        or cp == 0x2028
        or cp == 0x2029
        or cp == 0x202F
        or cp == 0x205F
        or cp == 0x3000
    )


def _lower_one(ch: str) -> str:
    # Only one-to-one mappings: "İ".lower() is two code points, and taking it
    # would make Python disagree with the length-preserving rule in all three.
    mapped = ch.lower()
    return mapped if len(mapped) == 1 else ch


def _upper_one(ch: str) -> str:
    # "ß".upper() is "SS": left alone, for the same reason.
    mapped = ch.upper()
    return mapped if len(mapped) == 1 else ch


def _lower_all(chars: str) -> str:
    return "".join(_lower_one(ch) for ch in chars)


def _is_mixed_case(word: str) -> bool:
    upper = any(_lower_one(ch) != ch for ch in word)
    lower = any(_upper_one(ch) != ch for ch in word)
    return upper and lower


def _capitalise(text: str) -> str:
    return _upper_one(text[0]) + text[1:] if text else ""


def _case_part(part: str) -> str:
    """Title-case one hyphen/apostrophe-delimited part, with the Mc and Mac rules."""
    lower = _lower_all(part)
    if len(lower) >= 3 and lower.startswith("mc"):
        return "Mc" + _capitalise(lower[2:])
    if len(lower) >= 6 and lower.startswith("mac") and lower not in _NOT_MAC:
        return "Mac" + _capitalise(lower[3:])
    return _capitalise(lower)


def _title_word(word: str) -> str:
    out = ""
    part = ""
    for ch in word:
        if ch in _PART_SEPARATORS:
            out += _case_part(part) + ch
            part = ""
        else:
            part += ch
    return out + _case_part(part)


def normalise_name(value: str) -> str:
    """Tidy a personal name: trim, collapse whitespace, and title-case words
    typed all in one case, with the Mc, Mac, O', hyphen and particle rules.
    """
    if not isinstance(value, str):
        raise TypeError("normaliseName needs a string, received %r" % (value,))

    words: List[str] = []
    current = ""
    for ch in value:
        if _is_whitespace(ch):
            if current:
                words.append(current)
            current = ""
        else:
            current += ch
    if current:
        words.append(current)

    out: List[str] = []
    for i, word in enumerate(words):
        # Someone who typed their own capitals knows better than any rule.
        if _is_mixed_case(word):
            out.append(word)
            continue
        lower = _lower_all(word)
        if 0 < i < len(words) - 1 and lower in _PARTICLES:
            out.append(lower)
            continue
        out.append(_title_word(word))
    return " ".join(out)