use super::funejson::Value; /// The transliteration table: every letter of the Latin-1 supplement /// (U+00C0-U+00FF) plus the seven letters Windows-1252 adds. /// /// Spelled out character by character rather than left to a Unicode /// normalisation crate, because that is the only way three languages with /// three different Unicode stacks can be pinned to the same output by one /// vector - and because a slug should not cost a dependency. /// /// Diacritics are dropped rather than expanded, so U+00E4 is "a" and not "ae": /// that is the convention every URL slug in the wild follows. The exceptions /// are the letters that are not accented vowels at all - sharp s is genuinely /// two letters, and the ligatures and thorn are letters in their own right. fn transliterate(ch: char) -> &'static str { match ch { '\u{00c0}' | '\u{00c1}' | '\u{00c2}' | '\u{00c3}' | '\u{00c4}' | '\u{00c5}' => "a", '\u{00c6}' => "ae", '\u{00c7}' => "c", '\u{00c8}' | '\u{00c9}' | '\u{00ca}' | '\u{00cb}' => "e", '\u{00cc}' | '\u{00cd}' | '\u{00ce}' | '\u{00cf}' => "i", '\u{00d0}' => "d", '\u{00d1}' => "n", '\u{00d2}' | '\u{00d3}' | '\u{00d4}' | '\u{00d5}' | '\u{00d6}' | '\u{00d8}' => "o", '\u{00d9}' | '\u{00da}' | '\u{00db}' | '\u{00dc}' => "u", '\u{00dd}' => "y", '\u{00de}' => "th", '\u{00df}' => "ss", '\u{00e0}' | '\u{00e1}' | '\u{00e2}' | '\u{00e3}' | '\u{00e4}' | '\u{00e5}' => "a", '\u{00e6}' => "ae", '\u{00e7}' => "c", '\u{00e8}' | '\u{00e9}' | '\u{00ea}' | '\u{00eb}' => "e", '\u{00ec}' | '\u{00ed}' | '\u{00ee}' | '\u{00ef}' => "i", '\u{00f0}' => "d", '\u{00f1}' => "n", '\u{00f2}' | '\u{00f3}' | '\u{00f4}' | '\u{00f5}' | '\u{00f6}' | '\u{00f8}' => "o", '\u{00f9}' | '\u{00fa}' | '\u{00fb}' | '\u{00fc}' => "u", '\u{00fd}' => "y", '\u{00fe}' => "th", '\u{00ff}' => "y", '\u{0152}' | '\u{0153}' => "oe", '\u{0160}' | '\u{0161}' => "s", '\u{017d}' | '\u{017e}' => "z", '\u{0178}' => "y", _ => "", } } /// A URL-safe slug: lowercase ASCII letters and digits, single hyphens between /// them, none at either end. /// /// Anything outside the transliteration table is dropped rather than guessed, /// so a title written entirely in another script slugifies to "". That empty /// string is returned, not a panic: the caller knows what its fallback is (an /// id, a date, a hash) and this function does not. pub fn slugify(value: &str) -> String { let mut out = String::with_capacity(value.len()); // A pending separator rather than a trailing hyphen plus a trim: it // collapses runs and drops the leading and trailing ones in one pass. let mut pending_separator = false; // `chars()` yields whole code points, so a character outside the Basic // Multilingual Plane is dropped once rather than as several separators. for ch in value.chars() { if ch.is_ascii_alphanumeric() { if pending_separator { out.push('-'); pending_separator = false; } // Digits are unaffected by ASCII lowercasing, so one branch covers both. out.push(ch.to_ascii_lowercase()); continue; } let piece = transliterate(ch); if piece.is_empty() { // Remember that something was skipped, but only emit the hyphen // once a keeper follows: that is what trims both ends. pending_separator = !out.is_empty(); continue; } if pending_separator { out.push('-'); pending_separator = false; } out.push_str(piece); } out } /// A slug guaranteed to be non-empty, falling back when nothing survives. pub fn slugify_or(value: &str, fallback: &str) -> String { let slug = slugify(value); if slug.is_empty() { fallback.to_string() } else { slug } } pub fn fune_vector(args: &[Value]) -> Value { // Refuse what the typed signature cannot hold, with the wording TypeScript // and Python use, rather than let the conversion below quietly change it. if !matches!(args[0], Value::Str(_)) { panic!("slugify needs a string, received {:?}", args[0]); } Value::Str(slugify(args[0].as_str())) }