SpanishTextNormalizer
def SpanishTextNormalizer(
tts_mode:bool=False
):Normalize Spanish text without removing accents, dieresis, or enye.
Normalize Spanish text without removing accents, dieresis, or enye.
Normalize French text without removing accents or ligatures.
Normalize Russian case, the optional letter ё, and numeric separators.
Normalize Chinese punctuation, whitespace, and common written numerals.
Normalize Arabic orthographic variants, diacritics, digits, and punctuation.
french_normalizer = FrenchTextNormalizer()
assert french_normalizer("J\u2019ai pay\u00e9 vingt et un euros pour l\u2019\u00e9t\u00e9.") == "j ai pay\u00e9 21 euros pour l \u00e9t\u00e9"
assert french_normalizer("Le total est 1\u202f234,50 \u20ac.") == "le total est 1234.50"
spanish_normalizer = SpanishTextNormalizer()
assert spanish_normalizer("\u00bfPag\u00f3 veinti\u00fan euros?") == "pag\u00f3 21 euros"
assert spanish_normalizer("El total es 1.234,50 \u20ac.") == "el total es 1234.50"
assert spanish_normalizer("El ni\u00f1o lleg\u00f3.") == "el ni\u00f1o lleg\u00f3"arabic_normalizer = ArabicTextNormalizer()
assert arabic_normalizer("أَهْلًا، وَسَهْلًا!") == "اهلا وسهلا"
assert arabic_normalizer("السعر ١٬٢٣٤٫٥٠ ريال.") == "السعر 1234.50 ريال"
chinese_normalizer = ChineseTextNormalizer()
assert chinese_normalizer("我有二十一個蘋果。") == "我有21個蘋果"
assert chinese_normalizer("二〇二四年") == "2024年"
assert chinese_normalizer("你好, 世界!") == "你好世界"
russian_normalizer = RussianTextNormalizer()
assert russian_normalizer("Ёлка — зелёная.") == "елка зеленая"
assert russian_normalizer("Цена: 1 234,50 руб.") == "цена 1234.50 руб"class TestLatinTextNormalizer(_LatinTextNormalizer):
language = "fr"
assert TestLatinTextNormalizer()("[noise] L’été—test (aside)\u200b!") == "l été test"
assert _NonLatinTextNormalizer()("[noise] Привет (aside)\u200b, мир!") == "привет мир"
assert arabic_normalizer("مــسؤوليةُ الفتى") == "مسوولية الفتي"
assert arabic_normalizer("شَيْئًا") == "شييا"
assert chinese_normalizer("十") == "10"
assert chinese_normalizer("两百零三万零十") == "2030010"
assert russian_normalizer("Цена 1\u00a0234,50 и 9\u202f876,00.") == "цена 1234.50 и 9876.00"# Per-language regression fixtures
# French: accents preserved, numbers handled, locale decimal comma
fr = FrenchTextNormalizer()
assert fr('Vingt et un euros.') == '21 euros'
assert fr('café et thé') == 'café et thé'
assert fr('1\u202f234,56') == '1234.56' # narrow no-break space thousands + comma decimal
assert fr('') == ''
assert fr(' ') == ''
# Spanish: accents/enye preserved, locale decimal comma
es = SpanishTextNormalizer()
assert es('El total es 1.234,50.') == 'el total es 1234.50'
assert es('Mañana será otro día.') == 'mañana será otro día'
assert es('') == ''
# Arabic: diacritics stripped, hamza normalised, digits converted
ar = ArabicTextNormalizer()
assert ar('أَهْلًا') == 'اهلا'
assert ar('٤٢') == '42' # Eastern-Arabic → Western digits
assert ar('') == ''
# Chinese: punctuation stripped, whitespace removed, written numerals → digits
zh = ChineseTextNormalizer()
assert zh('你好,世界!') == '你好世界'
assert zh('我有二十一個蘋果。') == '我有21個蘋果'
assert zh('') == ''
# Russian: ё → е, numeric separators normalised
ru = RussianTextNormalizer()
assert ru('ёж') == 'еж'
assert ru('1\u00a0234,56') == '1234.56' # NBSP thousands + comma decimal
assert ru('') == ''
print('Per-language regression tests passed')# TTS mode regression tests (requires whisper_normalizer[tts])
import importlib
if importlib.util.find_spec('num2words') is not None:
# French TTS: digits (including decimals/thousands) -> words; accents preserved
fr_tts = FrenchTextNormalizer(tts_mode=True)
assert 'quarante' in fr_tts('42 euros'), f'FR: 42 not spelled out: {fr_tts("42 euros")}'
assert 'virgule' in fr_tts('1\u202f234,56 euros'), f'FR: decimal not converted as one number: {fr_tts("1 234,56 euros")}'
assert 'é' in fr_tts('café 5'), 'FR: accent stripped unexpectedly'
# Spanish TTS: digits -> words; accents preserved
es_tts = SpanishTextNormalizer(tts_mode=True)
assert 'cuarenta' in es_tts('42 euros'), f'ES: 42 not spelled out: {es_tts("42 euros")}'
# Arabic TTS: Eastern-Arabic digits converted to words
ar_tts = ArabicTextNormalizer(tts_mode=True)
result_ar = ar_tts('٤٢ درهم')
assert '42' not in result_ar, f'AR: digits not converted in TTS mode: {result_ar!r}'
# Russian TTS: digits -> words; ё normalisation still applies
ru_tts = RussianTextNormalizer(tts_mode=True)
result_ru = ru_tts('42 рублей')
assert '42' not in result_ru, f'RU: digits not converted in TTS mode: {result_ru!r}'
# Chinese: num2words has no converter for 'zh', so tts_mode=True must raise
# ValueError at construction time rather than silently leaving digits unconverted.
try:
ChineseTextNormalizer(tts_mode=True)
raise AssertionError('ZH: tts_mode=True should raise ValueError (unsupported by num2words)')
except ValueError as e:
assert 'zh' in str(e)
# Non-TTS mode still converts written Chinese numerals to Arabic digits.
zh = ChineseTextNormalizer()
assert '21' in zh('二十一'), 'ZH: Chinese numeral not converted in normal mode'
print('International TTS tests passed')
else:
print('num2words not installed; skipping TTS tests')International TTS tests passed