from typing import List #: Characters after which a new capital starts: hyphen, both apostrophes, full stop. _PART_SEPARATORS = ("-", "'", "’", ".") #: Lower-cased in the middle of a name: "Ludwig van Beethoven". _PARTICLES = frozenset([ "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du", "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn", ]) #: Names where "mac" is not a Gaelic prefix, so no capital follows it. _NOT_MAC = frozenset([ "macari", "macaulay", "macedo", "macey", "machado", "machell", "machen", "machin", "macias", "mackay", "mackey", "mackie", "mackin", "macklin", ]) def _is_whitespace(ch: str) -> bool: """The Unicode White_Space property, spelled out so all three languages agree: str.isspace also accepts U+001C-U+001F, which JavaScript and Rust do not. """ cp = ord(ch) return ( 0x09 <= cp <= 0x0D or cp == 0x20 or cp == 0x85 or cp == 0xA0 or cp == 0x1680 or 0x2000 <= cp <= 0x200A or cp == 0x2028 or cp == 0x2029 or cp == 0x202F or cp == 0x205F or cp == 0x3000 ) def _lower_one(ch: str) -> str: # Only one-to-one mappings: "İ".lower() is two code points, and taking it # would make Python disagree with the length-preserving rule in all three. mapped = ch.lower() return mapped if len(mapped) == 1 else ch def _upper_one(ch: str) -> str: # "ß".upper() is "SS": left alone, for the same reason. mapped = ch.upper() return mapped if len(mapped) == 1 else ch def _lower_all(chars: str) -> str: return "".join(_lower_one(ch) for ch in chars) def _is_mixed_case(word: str) -> bool: upper = any(_lower_one(ch) != ch for ch in word) lower = any(_upper_one(ch) != ch for ch in word) return upper and lower def _capitalise(text: str) -> str: return _upper_one(text[0]) + text[1:] if text else "" def _case_part(part: str) -> str: """Title-case one hyphen/apostrophe-delimited part, with the Mc and Mac rules.""" lower = _lower_all(part) if len(lower) >= 3 and lower.startswith("mc"): return "Mc" + _capitalise(lower[2:]) if len(lower) >= 6 and lower.startswith("mac") and lower not in _NOT_MAC: return "Mac" + _capitalise(lower[3:]) return _capitalise(lower) def _title_word(word: str) -> str: out = "" part = "" for ch in word: if ch in _PART_SEPARATORS: out += _case_part(part) + ch part = "" else: part += ch return out + _case_part(part) def normalise_name(value: str) -> str: """Tidy a personal name: trim, collapse whitespace, and title-case words typed all in one case, with the Mc, Mac, O', hyphen and particle rules. """ if not isinstance(value, str): raise TypeError("normaliseName needs a string, received %r" % (value,)) words: List[str] = [] current = "" for ch in value: if _is_whitespace(ch): if current: words.append(current) current = "" else: current += ch if current: words.append(current) out: List[str] = [] for i, word in enumerate(words): # Someone who typed their own capitals knows better than any rule. if _is_mixed_case(word): out.append(word) continue lower = _lower_all(word) if 0 < i < len(words) - 1 and lower in _PARTICLES: out.append(lower) continue out.append(_title_word(word)) return " ".join(out)