3,043 bytes · the Python implementation · view raw
from typing import Dict
# The transliteration table: every letter of the Latin-1 supplement# (U+00C0-U+00FF) plus the seven letters Windows-1252 adds.## Spelled out character by character rather than left to unicodedata, because# that is the only way three languages with three different Unicode stacks can# be pinned to the same output by one vector.## Diacritics are dropped rather than expanded, so "a-umlaut" is "a" and not# "ae": that is the convention every URL slug in the wild follows. The# exceptions are the letters that are not accented vowels at all - "sharp s" is# genuinely two letters, and the ligatures and thorn are letters in their own right.
TRANSLITERATIONS: Dict[str, str] = {
"À": "a", "Á": "a", "Â": "a", "Ã": "a", "Ä": "a",
"Å": "a", "Æ": "ae", "Ç": "c", "È": "e", "É": "e",
"Ê": "e", "Ë": "e", "Ì": "i", "Í": "i", "Î": "i",
"Ï": "i", "Ð": "d", "Ñ": "n", "Ò": "o", "Ó": "o",
"Ô": "o", "Õ": "o", "Ö": "o", "Ø": "o", "Ù": "u",
"Ú": "u", "Û": "u", "Ü": "u", "Ý": "y", "Þ": "th",
"ß": "ss",
"à": "a", "á": "a", "â": "a", "ã": "a", "ä": "a",
"å": "a", "æ": "ae", "ç": "c", "è": "e", "é": "e",
"ê": "e", "ë": "e", "ì": "i", "í": "i", "î": "i",
"ï": "i", "ð": "d", "ñ": "n", "ò": "o", "ó": "o",
"ô": "o", "õ": "o", "ö": "o", "ø": "o", "ù": "u",
"ú": "u", "û": "u", "ü": "u", "ý": "y", "þ": "th",
"ÿ": "y",
"Œ": "oe", "œ": "oe", "Š": "s", "š": "s",
"Ž": "z", "ž": "z", "Ÿ": "y",
}
def slugify(value: str) -> str:
"""A URL-safe slug: lowercase ASCII letters and digits, single hyphens between them, none at either end. Anything outside the transliteration table is dropped rather than guessed, so a title written entirely in another script slugifies to "". That empty string is returned, not raised: the caller knows what its fallback is (an id, a date, a hash) and this function does not. """ifnot isinstance(value, str):
raise TypeError("slugify needs a string, received %r" % (value,))
out = []
length = 0# A pending separator rather than a trailing hyphen plus a strip: it# collapses runs and drops the leading and trailing ones in one pass.
pending_separator = Falsefor ch in value:
code = ord(ch)
if0x61 <= code <= 0x7Aor0x30 <= code <= 0x39:
piece = ch
elif0x41 <= code <= 0x5A:
piece = chr(code + 32)
else:
piece = TRANSLITERATIONS.get(ch, "")
if piece == "":
pending_separator = length > 0continueif pending_separator:
out.append("-")
length += 1
pending_separator = False
out.append(piece)
length += len(piece)
return"".join(out)
def slugify_or(value: str, fallback: str) -> str:
"""A slug guaranteed to be non-empty, falling back when nothing survives."""
slug = slugify(value)
return slug if slug else fallback