Functional Weave
Code in Rust

text.normalise-name@1.0.0

impl/typescript.ts

3,942 bytes · the TypeScript implementation · view raw

/**
 * The Unicode White_Space property, spelled out so all three languages agree:
 * JavaScript's \s, Python's str.isspace and Rust's char::is_whitespace each
 * draw the line slightly differently.
 */
function isWhitespace(cp: number): boolean {
  return (
    (cp >= 0x09 && cp <= 0x0d) ||
    cp === 0x20 ||
    cp === 0x85 ||
    cp === 0xa0 ||
    cp === 0x1680 ||
    (cp >= 0x2000 && cp <= 0x200a) ||
    cp === 0x2028 ||
    cp === 0x2029 ||
    cp === 0x202f ||
    cp === 0x205f ||
    cp === 0x3000
  );
}

/** Characters after which a new capital starts: hyphen, both apostrophes, full stop. */
const PART_SEPARATORS = new Set(["-", "'", "’", "."]);

/** Lower-cased in the middle of a name: "Ludwig van Beethoven". */
const PARTICLES = new Set([
  "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du",
  "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn",
]);

/** Names where "mac" is not a Gaelic prefix, so no capital follows it. */
const NOT_MAC = new Set([
  "macari", "macaulay", "macedo", "macey", "machado", "machell", "machen",
  "machin", "macias", "mackay", "mackey", "mackie", "mackin", "macklin",
]);

/**
 * Case-map one code point, but only when the mapping is itself one code
 * point. "ß".toUpperCase() is "SS"; leaving such characters alone is what
 * keeps TypeScript, Python and Rust identical.
 */
function lowerOne(ch: string): string {
  const mapped = ch.toLowerCase();
  return Array.from(mapped).length === 1 ? mapped : ch;
}

function upperOne(ch: string): string {
  const mapped = ch.toUpperCase();
  return Array.from(mapped).length === 1 ? mapped : ch;
}

function lowerAll(chars: string[]): string[] {
  return chars.map(lowerOne);
}

/** True when the word has both an upper-case and a lower-case letter. */
function isMixedCase(chars: string[]): boolean {
  let upper = false;
  let lower = false;
  for (const ch of chars) {
    if (lowerOne(ch) !== ch) upper = true;
    if (upperOne(ch) !== ch) lower = true;
  }
  return upper && lower;
}

function capitalise(chars: string[]): string {
  if (chars.length === 0) return "";
  return upperOne(chars[0]) + chars.slice(1).join("");
}

/** Title-case one hyphen/apostrophe-delimited part, with the Mc and Mac rules. */
function casePart(part: string[]): string {
  const lower = lowerAll(part);
  const text = lower.join("");
  if (lower.length >= 3 && text.startsWith("mc")) {
    return "Mc" + capitalise(lower.slice(2));
  }
  if (lower.length >= 6 && text.startsWith("mac") && !NOT_MAC.has(text)) {
    return "Mac" + capitalise(lower.slice(3));
  }
  return capitalise(lower);
}

function titleWord(word: string[]): string {
  let out = "";
  let part: string[] = [];
  for (const ch of word) {
    if (PART_SEPARATORS.has(ch)) {
      out += casePart(part) + ch;
      part = [];
    } else {
      part.push(ch);
    }
  }
  return out + casePart(part);
}

/**
 * Tidy a personal name: trim, collapse whitespace, and title-case words typed
 * all in one case, with the Mc, Mac, O', hyphen and particle rules.
 */
export function normaliseName(value: string): string {
  if (typeof value !== "string") {
    throw new TypeError(`normaliseName needs a string, received ${value}`);
  }

  // Split on whitespace by code point, so a no-break space separates words.
  const words: string[][] = [];
  let current: string[] = [];
  for (const ch of value) {
    if (isWhitespace(ch.codePointAt(0) as number)) {
      if (current.length) words.push(current);
      current = [];
    } else {
      current.push(ch);
    }
  }
  if (current.length) words.push(current);

  return words
    .map((word, i) => {
      // Someone who typed their own capitals knows better than any rule.
      if (isMixedCase(word)) return word.join("");
      const lower = lowerAll(word).join("");
      if (i > 0 && i < words.length - 1 && PARTICLES.has(lower)) return lower;
      return titleWord(word);
    })
    .join(" ");
}