3,607 bytes · the Python implementation · view raw
from typing import List
#: Characters after which a new capital starts: hyphen, both apostrophes, full stop.
_PART_SEPARATORS = ("-", "'", "’", ".")
#: Lower-cased in the middle of a name: "Ludwig van Beethoven".
_PARTICLES = frozenset([
"van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du",
"la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn",
])
#: Names where "mac" is not a Gaelic prefix, so no capital follows it.
_NOT_MAC = frozenset([
"macari", "macaulay", "macedo", "macey", "machado", "machell", "machen",
"machin", "macias", "mackay", "mackey", "mackie", "mackin", "macklin",
])
def _is_whitespace(ch: str) -> bool:
"""The Unicode White_Space property, spelled out so all three languages agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust do not. """
cp = ord(ch)
return (
0x09 <= cp <= 0x0Dor cp == 0x20or cp == 0x85or cp == 0xA0or cp == 0x1680or0x2000 <= cp <= 0x200Aor cp == 0x2028or cp == 0x2029or cp == 0x202For cp == 0x205For cp == 0x3000
)
def _lower_one(ch: str) -> str:
# Only one-to-one mappings: "İ".lower() is two code points, and taking it# would make Python disagree with the length-preserving rule in all three.
mapped = ch.lower()
return mapped if len(mapped) == 1else ch
def _upper_one(ch: str) -> str:
# "ß".upper() is "SS": left alone, for the same reason.
mapped = ch.upper()
return mapped if len(mapped) == 1else ch
def _lower_all(chars: str) -> str:
return"".join(_lower_one(ch) for ch in chars)
def _is_mixed_case(word: str) -> bool:
upper = any(_lower_one(ch) != ch for ch in word)
lower = any(_upper_one(ch) != ch for ch in word)
return upper and lower
def _capitalise(text: str) -> str:
return _upper_one(text[0]) + text[1:] if text else""def _case_part(part: str) -> str:
"""Title-case one hyphen/apostrophe-delimited part, with the Mc and Mac rules."""
lower = _lower_all(part)
if len(lower) >= 3and lower.startswith("mc"):
return"Mc" + _capitalise(lower[2:])
if len(lower) >= 6and lower.startswith("mac") and lower notin _NOT_MAC:
return"Mac" + _capitalise(lower[3:])
return _capitalise(lower)
def _title_word(word: str) -> str:
out = ""
part = ""for ch in word:
if ch in _PART_SEPARATORS:
out += _case_part(part) + ch
part = ""else:
part += ch
return out + _case_part(part)
def normalise_name(value: str) -> str:
"""Tidy a personal name: trim, collapse whitespace, and title-case words typed all in one case, with the Mc, Mac, O', hyphen and particle rules. """ifnot isinstance(value, str):
raise TypeError("normaliseName needs a string, received %r" % (value,))
words: List[str] = []
current = ""for ch in value:
if _is_whitespace(ch):
if current:
words.append(current)
current = ""else:
current += ch
if current:
words.append(current)
out: List[str] = []
for i, word in enumerate(words):
# Someone who typed their own capitals knows better than any rule.if _is_mixed_case(word):
out.append(word)
continue
lower = _lower_all(word)
if0 < i < len(words) - 1and lower in _PARTICLES:
out.append(lower)
continue
out.append(_title_word(word))
return" ".join(out)