use super::funejson::Value; /// Characters after which a new capital starts: hyphen, both apostrophes, full stop. const PART_SEPARATORS: [char; 4] = ['-', '\'', '\u{2019}', '.']; /// Lower-cased in the middle of a name: "Ludwig van Beethoven". const PARTICLES: [&str; 20] = [ "van", "von", "de", "da", "di", "del", "della", "der", "den", "des", "du", "la", "le", "ter", "ten", "dos", "das", "do", "bin", "ibn", ]; /// Names where "mac" is not a Gaelic prefix, so no capital follows it. const NOT_MAC: [&str; 14] = [ "macari", "macaulay", "macedo", "macey", "machado", "machell", "machen", "machin", "macias", "mackay", "mackey", "mackie", "mackin", "macklin", ]; /// The Unicode White_Space property, spelled out so all three languages agree. fn is_whitespace(ch: char) -> bool { let cp = ch as u32; (0x09..=0x0d).contains(&cp) || cp == 0x20 || cp == 0x85 || cp == 0xa0 || cp == 0x1680 || (0x2000..=0x200a).contains(&cp) || cp == 0x2028 || cp == 0x2029 || cp == 0x202f || cp == 0x205f || cp == 0x3000 } /// Case-map one character, but only when the mapping is itself one character. /// 'ß' upper-cases to "SS"; leaving such characters alone keeps the three /// languages identical. fn lower_one(ch: char) -> char { let mapped: Vec = ch.to_lowercase().collect(); if mapped.len() == 1 { mapped[0] } else { ch } } fn upper_one(ch: char) -> char { let mapped: Vec = ch.to_uppercase().collect(); if mapped.len() == 1 { mapped[0] } else { ch } } fn is_mixed_case(word: &[char]) -> bool { let upper = word.iter().any(|&c| lower_one(c) != c); let lower = word.iter().any(|&c| upper_one(c) != c); upper && lower } fn capitalise(chars: &[char]) -> String { match chars.split_first() { None => String::new(), Some((first, rest)) => { let mut out = String::new(); out.push(upper_one(*first)); out.extend(rest.iter()); out } } } /// Title-case one hyphen/apostrophe-delimited part, with the Mc and Mac rules. fn case_part(part: &[char]) -> String { let lower: Vec = part.iter().map(|&c| lower_one(c)).collect(); let text: String = lower.iter().collect(); if lower.len() >= 3 && text.starts_with("mc") { return format!("Mc{}", capitalise(&lower[2..])); } if lower.len() >= 6 && text.starts_with("mac") && !NOT_MAC.contains(&text.as_str()) { return format!("Mac{}", capitalise(&lower[3..])); } capitalise(&lower) } fn title_word(word: &[char]) -> String { let mut out = String::new(); let mut part: Vec = Vec::new(); for &ch in word { if PART_SEPARATORS.contains(&ch) { out.push_str(&case_part(&part)); out.push(ch); part.clear(); } else { part.push(ch); } } out.push_str(&case_part(&part)); out } /// Tidy a personal name: trim, collapse whitespace, and title-case words typed /// all in one case, with the Mc, Mac, O', hyphen and particle rules. pub fn normalise_name(value: &str) -> String { let mut words: Vec> = Vec::new(); let mut current: Vec = Vec::new(); for ch in value.chars() { if is_whitespace(ch) { if !current.is_empty() { words.push(std::mem::take(&mut current)); } } else { current.push(ch); } } if !current.is_empty() { words.push(current); } let count = words.len(); let out: Vec = words .iter() .enumerate() .map(|(i, word)| { // Someone who typed their own capitals knows better than any rule. if is_mixed_case(word) { return word.iter().collect(); } let lower: String = word.iter().map(|&c| lower_one(c)).collect(); if i > 0 && i + 1 < count && PARTICLES.contains(&lower.as_str()) { return lower; } title_word(word) }) .collect(); out.join(" ") } pub fn fune_vector(args: &[Value]) -> Value { // Refuse what the typed signature cannot hold, with the wording TypeScript // and Python use, rather than let the conversion below quietly change it. if !matches!(args[0], Value::Str(_)) { panic!("normaliseName needs a string, received {:?}", args[0]); } Value::Str(normalise_name(args[0].as_str())) }