Multilingual normalizers> A registry for conservative, language-aware text normalization.


source

supported_languages

def supported_languages()->tuple[__main__.LanguageSupport, ...]:

Return the language registry in ISO-code order.


source

get_normalizer

def get_normalizer(
    language:str, **options
)->Callable[[str], str]:

Return a fresh normalizer for a BCP-47 tag, ISO 639-1 code, or language name.

language accepts plain codes ("hi"), BCP-47 tags with a region or script subtag ("hi-IN", "zh-Hans", "es-ES"), and English language names ("hindi"). Only the primary language subtag drives selection: a region/script suffix is accepted but does not currently change which normalizer is returned. Keyword options such as tts_mode=True are validated against the selected normalizer and raise ValueError if it doesn’t accept them, instead of being silently ignored.


source

LanguageSupport

def LanguageSupport(
    code:str, name:str, factory:Callable[..., Callable[[str], str]], capabilities:tuple[str, ...], tier:str
)->None:

A supported language and the normalization level available for it.

tier is one of:

  • "dedicated": a normalizer written and reviewed for this language.
  • "script": no dedicated class yet, but the language shares a script with a dedicated normalizer that is reused as-is (for example Marathi reuses the Devanagari normalizer, without Hindi’s number/TTS tuning).
  • "generic": only conservative, script-safe cleanup is applied; no language-specific rules exist.

source

MultilingualTextNormalizer

def MultilingualTextNormalizer(
    lang:str='', tts_mode:bool=False
):

Conservative, script-safe normalization for languages without bespoke rules.

assert len(supported_languages()) == 101

tiers = {"dedicated": 0, "script": 0, "generic": 0}
for entry in supported_languages():
    tiers[entry.tier] += 1
assert tiers == {"dedicated": 15, "script": 4, "generic": 82}

# BCP-47 region/script subtags are accepted; only the primary language subtag
# drives selection.
assert get_normalizer("french")("Vingt et un euros.") == "21 euros"
assert get_normalizer("es-ES")("El total es 1.234,50.") == "el total es 1234.50"
assert get_normalizer("Russian")("Privet, mir!") == "privet mir"
assert get_normalizer("hi-IN")("नमस्ते, दुनिया!") == get_normalizer("hi")("नमस्ते, दुनिया!")
assert get_normalizer("en-IN")("Colour and favour.") == get_normalizer("en")("Colour and favour.")

# ar/zh/ru must use their dedicated normalizers (from international.py), not
# the generic fallback: dedicated Chinese/Arabic normalization differs from
# the generic tier by converting written numerals and removing all
# inter-character whitespace, which the generic normalizer does not do.
assert get_normalizer("zh")("你好,世界!") == "你好世界"
assert get_normalizer("zh")("我有二十一個蘋果。") == "我有21個蘋果"
assert get_normalizer("ar")("أَهْلًا، وَسَهْلًا!") == "اهلا وسهلا"

# Odia ("or") is a dedicated normalizer even though it isn't one of the ~99
# languages Whisper itself documents. "od-IN" and "odia" are accepted as
# aliases for the same "or" code.
assert LANGUAGE_REGISTRY["or"].tier == "dedicated"
assert get_normalizer("od-IN")("2024") == get_normalizer("or")("2024")
assert get_normalizer("odia")("2024") == get_normalizer("or")("2024")

# Script tier: Marathi (mr) and Assamese (as) now support tts_mode via
# indic_numtowords. Nepali (ne) and Sanskrit (sa) remain script-only (not
# covered by indic_numtowords).
assert LANGUAGE_REGISTRY["mr"].tier == "script"
assert isinstance(get_normalizer("mr"), DevanagariNormalizer)
assert get_normalizer("mr")("क: ख") == get_normalizer("hi")("क: ख")  # shared Devanagari visarga rule
# Marathi now supports tts_mode via DevanagariNormalizer (indic_numtowords supports 'mr').
assert get_normalizer("mr", tts_mode=True).tts_mode is True

assert LANGUAGE_REGISTRY["as"].tier == "script"
assert isinstance(get_normalizer("as"), BengaliNormalizer)
assert get_normalizer("as").lang == "as"

# tts_mode is validated against the selected normalizer instead of being
# silently ignored or forwarded to a constructor that doesn't accept it.
assert get_normalizer("hi", tts_mode=True).tts_mode is True
assert get_normalizer("bn", tts_mode=True).tts_mode is True
# French now supports tts_mode via _LatinTextNormalizer (num2words converts digits to words).
assert get_normalizer("fr", tts_mode=True).tts_mode is True

try:
    get_normalizer("xx")
except ValueError as error:
    assert "Unsupported language" in str(error)
else:
    raise AssertionError("unknown languages must be rejected")

# Generic-tier smoke test: every generic-tier code must lowercase, strip
# punctuation and symbols (including emoji, which are Unicode Symbol
# category), collapse whitespace, and stay safe on empty/mixed-script input,
# without bespoke per-language fixtures.
generic_codes = [entry.code for entry in supported_languages() if entry.tier == "generic"]
assert len(generic_codes) == 82
for code in generic_codes:
    normalizer = get_normalizer(code)
    assert normalizer("") == ""
    assert normalizer("HELLO, World!!") == "hello world"
    assert normalizer("Café 😀 مرحبا") == "café مرحبا"
# Per-language regression fixtures — dedicated and script tier

# Marathi (script tier): Devanagari normalisation + correct number words
mr = get_normalizer('mr')
assert mr('क: ख') == get_normalizer('hi')('क: ख')  # visarga rule shared
mr_tts = get_normalizer('mr', tts_mode=True)
assert mr_tts.tts_mode is True
result_mr = mr_tts('मला 42 रुपये द्या')
assert '42' not in result_mr, f'MR TTS: digit not converted: {result_mr}'
# Marathi uses 'बेचाळीस', not Hindi 'बयालीस'
assert 'बेचाळीस' in result_mr, f'MR: wrong number word: {result_mr}'

# Assamese (script tier): Bengali normaliser + Assamese char remapping + TTS
as_norm = get_normalizer('as')
assert as_norm.lang == 'as'
as_tts = get_normalizer('as', tts_mode=True)
result_as = as_tts('আমি 10 টকা')
assert '10' not in result_as, f'AS TTS: digit not converted: {result_as}'

# Nepali / Sanskrit: tts_mode unsupported
try:
    get_normalizer('ne', tts_mode=True)
    raise AssertionError('ne should raise ValueError')
except ValueError as e:
    assert 'tts_mode' in str(e) or 'not supported' in str(e)

try:
    get_normalizer('sa', tts_mode=True)
    raise AssertionError('sa should raise ValueError')
except ValueError as e:
    assert 'tts_mode' in str(e) or 'not supported' in str(e)

# Generic tier: smoke coverage for a representative sample
for code, sample, expected in [
    ('de', 'Guten Tag!', 'guten tag'),
    ('tr', 'Merhaba dünya!', 'merhaba dünya'),
    ('ko', '안녕하세요!', '안녕하세요'),
    ('th', 'สวัสดี!', 'สวัสดี'),
]:
    result = get_normalizer(code)(sample)
    assert result == expected, f'{code}: got {result!r}, expected {expected!r}'

print('Per-language regression fixtures passed')
# Generic-tier TTS regression tests
import importlib

if importlib.util.find_spec('num2words') is not None:
    de_tts = get_normalizer('de', tts_mode=True)
    result = de_tts('42 Euro')
    assert '42' not in result, f'DE TTS: digits not converted: {result}'

    nl_tts = get_normalizer('nl', tts_mode=True)
    result = nl_tts('42 euro')
    assert '42' not in result, f'NL TTS: digits not converted: {result}'

    # Verify lang attr is injected
    assert de_tts.lang == 'de', 'lang not injected for generic tier'
    print('Generic TTS tests passed')
else:
    print('num2words not installed; skipping generic TTS tests')
Generic TTS tests passed