supported_languages
def supported_languages()->tuple[__main__.LanguageSupport, ...]:Return the language registry in ISO-code order.
Return the language registry in ISO-code order.
Return a fresh normalizer for a BCP-47 tag, ISO 639-1 code, or language name.
language accepts plain codes ("hi"), BCP-47 tags with a region or script subtag ("hi-IN", "zh-Hans", "es-ES"), and English language names ("hindi"). Only the primary language subtag drives selection: a region/script suffix is accepted but does not currently change which normalizer is returned. Keyword options such as tts_mode=True are validated against the selected normalizer and raise ValueError if it doesn’t accept them, instead of being silently ignored.
A supported language and the normalization level available for it.
tier is one of:
"dedicated": a normalizer written and reviewed for this language."script": no dedicated class yet, but the language shares a script with a dedicated normalizer that is reused as-is (for example Marathi reuses the Devanagari normalizer, without Hindi’s number/TTS tuning)."generic": only conservative, script-safe cleanup is applied; no language-specific rules exist.Conservative, script-safe normalization for languages without bespoke rules.
assert len(supported_languages()) == 101
tiers = {"dedicated": 0, "script": 0, "generic": 0}
for entry in supported_languages():
tiers[entry.tier] += 1
assert tiers == {"dedicated": 15, "script": 4, "generic": 82}
# BCP-47 region/script subtags are accepted; only the primary language subtag
# drives selection.
assert get_normalizer("french")("Vingt et un euros.") == "21 euros"
assert get_normalizer("es-ES")("El total es 1.234,50.") == "el total es 1234.50"
assert get_normalizer("Russian")("Privet, mir!") == "privet mir"
assert get_normalizer("hi-IN")("नमस्ते, दुनिया!") == get_normalizer("hi")("नमस्ते, दुनिया!")
assert get_normalizer("en-IN")("Colour and favour.") == get_normalizer("en")("Colour and favour.")
# ar/zh/ru must use their dedicated normalizers (from international.py), not
# the generic fallback: dedicated Chinese/Arabic normalization differs from
# the generic tier by converting written numerals and removing all
# inter-character whitespace, which the generic normalizer does not do.
assert get_normalizer("zh")("你好,世界!") == "你好世界"
assert get_normalizer("zh")("我有二十一個蘋果。") == "我有21個蘋果"
assert get_normalizer("ar")("أَهْلًا، وَسَهْلًا!") == "اهلا وسهلا"
# Odia ("or") is a dedicated normalizer even though it isn't one of the ~99
# languages Whisper itself documents. "od-IN" and "odia" are accepted as
# aliases for the same "or" code.
assert LANGUAGE_REGISTRY["or"].tier == "dedicated"
assert get_normalizer("od-IN")("2024") == get_normalizer("or")("2024")
assert get_normalizer("odia")("2024") == get_normalizer("or")("2024")
# Script tier: Marathi (mr) and Assamese (as) now support tts_mode via
# indic_numtowords. Nepali (ne) and Sanskrit (sa) remain script-only (not
# covered by indic_numtowords).
assert LANGUAGE_REGISTRY["mr"].tier == "script"
assert isinstance(get_normalizer("mr"), DevanagariNormalizer)
assert get_normalizer("mr")("क: ख") == get_normalizer("hi")("क: ख") # shared Devanagari visarga rule
# Marathi now supports tts_mode via DevanagariNormalizer (indic_numtowords supports 'mr').
assert get_normalizer("mr", tts_mode=True).tts_mode is True
assert LANGUAGE_REGISTRY["as"].tier == "script"
assert isinstance(get_normalizer("as"), BengaliNormalizer)
assert get_normalizer("as").lang == "as"
# tts_mode is validated against the selected normalizer instead of being
# silently ignored or forwarded to a constructor that doesn't accept it.
assert get_normalizer("hi", tts_mode=True).tts_mode is True
assert get_normalizer("bn", tts_mode=True).tts_mode is True
# French now supports tts_mode via _LatinTextNormalizer (num2words converts digits to words).
assert get_normalizer("fr", tts_mode=True).tts_mode is True
try:
get_normalizer("xx")
except ValueError as error:
assert "Unsupported language" in str(error)
else:
raise AssertionError("unknown languages must be rejected")
# Generic-tier smoke test: every generic-tier code must lowercase, strip
# punctuation and symbols (including emoji, which are Unicode Symbol
# category), collapse whitespace, and stay safe on empty/mixed-script input,
# without bespoke per-language fixtures.
generic_codes = [entry.code for entry in supported_languages() if entry.tier == "generic"]
assert len(generic_codes) == 82
for code in generic_codes:
normalizer = get_normalizer(code)
assert normalizer("") == ""
assert normalizer("HELLO, World!!") == "hello world"
assert normalizer("Café 😀 مرحبا") == "café مرحبا"# Per-language regression fixtures — dedicated and script tier
# Marathi (script tier): Devanagari normalisation + correct number words
mr = get_normalizer('mr')
assert mr('क: ख') == get_normalizer('hi')('क: ख') # visarga rule shared
mr_tts = get_normalizer('mr', tts_mode=True)
assert mr_tts.tts_mode is True
result_mr = mr_tts('मला 42 रुपये द्या')
assert '42' not in result_mr, f'MR TTS: digit not converted: {result_mr}'
# Marathi uses 'बेचाळीस', not Hindi 'बयालीस'
assert 'बेचाळीस' in result_mr, f'MR: wrong number word: {result_mr}'
# Assamese (script tier): Bengali normaliser + Assamese char remapping + TTS
as_norm = get_normalizer('as')
assert as_norm.lang == 'as'
as_tts = get_normalizer('as', tts_mode=True)
result_as = as_tts('আমি 10 টকা')
assert '10' not in result_as, f'AS TTS: digit not converted: {result_as}'
# Nepali / Sanskrit: tts_mode unsupported
try:
get_normalizer('ne', tts_mode=True)
raise AssertionError('ne should raise ValueError')
except ValueError as e:
assert 'tts_mode' in str(e) or 'not supported' in str(e)
try:
get_normalizer('sa', tts_mode=True)
raise AssertionError('sa should raise ValueError')
except ValueError as e:
assert 'tts_mode' in str(e) or 'not supported' in str(e)
# Generic tier: smoke coverage for a representative sample
for code, sample, expected in [
('de', 'Guten Tag!', 'guten tag'),
('tr', 'Merhaba dünya!', 'merhaba dünya'),
('ko', '안녕하세요!', '안녕하세요'),
('th', 'สวัสดี!', 'สวัสดี'),
]:
result = get_normalizer(code)(sample)
assert result == expected, f'{code}: got {result!r}, expected {expected!r}'
print('Per-language regression fixtures passed')# Generic-tier TTS regression tests
import importlib
if importlib.util.find_spec('num2words') is not None:
de_tts = get_normalizer('de', tts_mode=True)
result = de_tts('42 Euro')
assert '42' not in result, f'DE TTS: digits not converted: {result}'
nl_tts = get_normalizer('nl', tts_mode=True)
result = nl_tts('42 euro')
assert '42' not in result, f'NL TTS: digits not converted: {result}'
# Verify lang attr is injected
assert de_tts.lang == 'de', 'lang not injected for generic tier'
print('Generic TTS tests passed')
else:
print('num2words not installed; skipping generic TTS tests')Generic TTS tests passed