source

SpanishTextNormalizer

def SpanishTextNormalizer(
    tts_mode:bool=False
):

Normalize Spanish text without removing accents, dieresis, or enye.


source

FrenchTextNormalizer

def FrenchTextNormalizer(
    tts_mode:bool=False
):

Normalize French text without removing accents or ligatures.


source

RussianTextNormalizer

def RussianTextNormalizer(
    tts_mode:bool=False
):

Normalize Russian case, the optional letter ё, and numeric separators.


source

ChineseTextNormalizer

def ChineseTextNormalizer(
    tts_mode:bool=False
):

Normalize Chinese punctuation, whitespace, and common written numerals.


source

ArabicTextNormalizer

def ArabicTextNormalizer(
    tts_mode:bool=False
):

Normalize Arabic orthographic variants, diacritics, digits, and punctuation.

french_normalizer = FrenchTextNormalizer()
assert french_normalizer("J\u2019ai pay\u00e9 vingt et un euros pour l\u2019\u00e9t\u00e9.") == "j ai pay\u00e9 21 euros pour l \u00e9t\u00e9"
assert french_normalizer("Le total est 1\u202f234,50 \u20ac.") == "le total est 1234.50"

spanish_normalizer = SpanishTextNormalizer()
assert spanish_normalizer("\u00bfPag\u00f3 veinti\u00fan euros?") == "pag\u00f3 21 euros"
assert spanish_normalizer("El total es 1.234,50 \u20ac.") == "el total es 1234.50"
assert spanish_normalizer("El ni\u00f1o lleg\u00f3.") == "el ni\u00f1o lleg\u00f3"
arabic_normalizer = ArabicTextNormalizer()
assert arabic_normalizer("أَهْلًا، وَسَهْلًا!") == "اهلا وسهلا"
assert arabic_normalizer("السعر ١٬٢٣٤٫٥٠ ريال.") == "السعر 1234.50 ريال"

chinese_normalizer = ChineseTextNormalizer()
assert chinese_normalizer("我有二十一個蘋果。") == "我有21個蘋果"
assert chinese_normalizer("二〇二四年") == "2024年"
assert chinese_normalizer("你好, 世界!") == "你好世界"

russian_normalizer = RussianTextNormalizer()
assert russian_normalizer("Ёлка — зелёная.") == "елка зеленая"
assert russian_normalizer("Цена: 1 234,50 руб.") == "цена 1234.50 руб"
class TestLatinTextNormalizer(_LatinTextNormalizer):
    language = "fr"


assert TestLatinTextNormalizer()("[noise] L’été—test (aside)\u200b!") == "l été test"
assert _NonLatinTextNormalizer()("[noise] Привет (aside)\u200b, мир!") == "привет мир"

assert arabic_normalizer("مــسؤوليةُ الفتى") == "مسوولية الفتي"
assert arabic_normalizer("شَيْئًا") == "شييا"

assert chinese_normalizer("十") == "10"
assert chinese_normalizer("两百零三万零十") == "2030010"

assert russian_normalizer("Цена 1\u00a0234,50 и 9\u202f876,00.") == "цена 1234.50 и 9876.00"
assert spanish_normalizer("Un niño vive en una casa.") == "un niño vive en una casa"
assert spanish_normalizer("Veinte euros.") == "20 euros"
assert chinese_normalizer("一万亿") == "1000000000000"
assert chinese_normalizer("一万亿零三万") == "1000000030000"
# Per-language regression fixtures

# French: accents preserved, numbers handled, locale decimal comma
fr = FrenchTextNormalizer()
assert fr('Vingt et un euros.') == '21 euros'
assert fr('café et thé') == 'café et thé'
assert fr('1\u202f234,56') == '1234.56'  # narrow no-break space thousands + comma decimal
assert fr('') == ''
assert fr('  ') == ''

# Spanish: accents/enye preserved, locale decimal comma
es = SpanishTextNormalizer()
assert es('El total es 1.234,50.') == 'el total es 1234.50'
assert es('Mañana será otro día.') == 'mañana será otro día'
assert es('') == ''

# Arabic: diacritics stripped, hamza normalised, digits converted
ar = ArabicTextNormalizer()
assert ar('أَهْلًا') == 'اهلا'
assert ar('٤٢') == '42'  # Eastern-Arabic → Western digits
assert ar('') == ''

# Chinese: punctuation stripped, whitespace removed, written numerals → digits
zh = ChineseTextNormalizer()
assert zh('你好,世界!') == '你好世界'
assert zh('我有二十一個蘋果。') == '我有21個蘋果'
assert zh('') == ''

# Russian: ё → е, numeric separators normalised
ru = RussianTextNormalizer()
assert ru('ёж') == 'еж'
assert ru('1\u00a0234,56') == '1234.56'  # NBSP thousands + comma decimal
assert ru('') == ''

print('Per-language regression tests passed')
# TTS mode regression tests (requires whisper_normalizer[tts])
import importlib

if importlib.util.find_spec('num2words') is not None:
    # French TTS: digits (including decimals/thousands) -> words; accents preserved
    fr_tts = FrenchTextNormalizer(tts_mode=True)
    assert 'quarante' in fr_tts('42 euros'), f'FR: 42 not spelled out: {fr_tts("42 euros")}'
    assert 'virgule' in fr_tts('1\u202f234,56 euros'), f'FR: decimal not converted as one number: {fr_tts("1 234,56 euros")}'
    assert 'é' in fr_tts('café 5'), 'FR: accent stripped unexpectedly'

    # Spanish TTS: digits -> words; accents preserved
    es_tts = SpanishTextNormalizer(tts_mode=True)
    assert 'cuarenta' in es_tts('42 euros'), f'ES: 42 not spelled out: {es_tts("42 euros")}'

    # Arabic TTS: Eastern-Arabic digits converted to words
    ar_tts = ArabicTextNormalizer(tts_mode=True)
    result_ar = ar_tts('٤٢ درهم')
    assert '42' not in result_ar, f'AR: digits not converted in TTS mode: {result_ar!r}'

    # Russian TTS: digits -> words; ё normalisation still applies
    ru_tts = RussianTextNormalizer(tts_mode=True)
    result_ru = ru_tts('42 рублей')
    assert '42' not in result_ru, f'RU: digits not converted in TTS mode: {result_ru!r}'

    # Chinese: num2words has no converter for 'zh', so tts_mode=True must raise
    # ValueError at construction time rather than silently leaving digits unconverted.
    try:
        ChineseTextNormalizer(tts_mode=True)
        raise AssertionError('ZH: tts_mode=True should raise ValueError (unsupported by num2words)')
    except ValueError as e:
        assert 'zh' in str(e)

    # Non-TTS mode still converts written Chinese numerals to Arabic digits.
    zh = ChineseTextNormalizer()
    assert '21' in zh('二十一'), 'ZH: Chinese numeral not converted in normal mode'

    print('International TTS tests passed')
else:
    print('num2words not installed; skipping TTS tests')
International TTS tests passed