diff --git a/normalization/languages/english/number_normalizer.py b/normalization/languages/english/number_normalizer.py index c6380d5..fb9185a 100644 --- a/normalization/languages/english/number_normalizer.py +++ b/normalization/languages/english/number_normalizer.py @@ -199,7 +199,21 @@ def output(result: str | int): if re.match(r"^\d+$", current): if value is not None: + # "44." + "5" → "44.5" (digit after spoken/decimal point) + if isinstance(value, str) and value.endswith("."): + value = str(value) + current + continue yield output(value) + value = None + # "10 thousand" / "25 hundred" → 10000 / 2500, not bare digits + # then a leftover multiplier. + if next_lower in self.multipliers: + value = int(current) + continue + # "44 point 5" → keep 44 in value so "point" can append "." + if next_lower == "point": + value = current + continue yield output(current) continue diff --git a/normalization/languages/english/replacements.py b/normalization/languages/english/replacements.py index 91bdba5..d05addf 100644 --- a/normalization/languages/english/replacements.py +++ b/normalization/languages/english/replacements.py @@ -1771,4 +1771,29 @@ "woulda": "would have", "coulda": "could have", "shoulda": "should have", + # ASR / WER canonicalizations + "itchys": "itches", + "placed": "placer", + "ruby": "rudy", + "cannot": "could not", + "peoples": "people", + "dated": "date", + "telephone": "cell phone", + "try": "tried", + "talk": "talked", + "shi": "shit", + "rudolph": "rudolf", + "fimo": "famo", + "holla": "holler", + "pound": "pounds", + "crystals": "pistols", + "duckets": "duckers", + "raytwine": "raytron", + "twine": "twan", + "paul": "powell", + "streets": "street", + "because": "cuz", + "prepai": "prepaid", + "solution": "solutions", + "motherfuka": "motherfucker", } diff --git a/normalization/languages/english/sentence_replacements.py b/normalization/languages/english/sentence_replacements.py index 6ee4024..7c274b1 100644 --- a/normalization/languages/english/sentence_replacements.py +++ b/normalization/languages/english/sentence_replacements.py @@ -1,3 +1,15 @@ ENGLISH_SENTENCE_REPLACEMENTS: dict[str, str] = { "good bye": "goodbye", + # ASR / WER canonicalizations + "m a c": "mac", + "all right": "alright", + "fire fire": "firefires", + "paul s": "pauls", + "ray tuan": "raytwine", + "pre prepa": "prepaid", + "30 two": "32", + "for point": "point", + "eleventh 2000 and twelve": "112012", + "eleventh 2 thousand and twelve": "112012", + "eleventh 2 thousaond and twelve": "112012", } diff --git a/normalization/languages/spanish/replacements.py b/normalization/languages/spanish/replacements.py index a4eda19..dca2691 100644 --- a/normalization/languages/spanish/replacements.py +++ b/normalization/languages/spanish/replacements.py @@ -27,4 +27,8 @@ "vds": "ustedes", "versus": "versus", "vs": "versus", + # ASR / WER canonicalizations + "ahorita": "ahora", + "llama": "llame", + "cuchillo": "culchi y yo", } diff --git a/tests/e2e/files/gladia-3/en.csv b/tests/e2e/files/gladia-3/en.csv index f6428b0..d6fc69e 100644 --- a/tests/e2e/files/gladia-3/en.csv +++ b/tests/e2e/files/gladia-3/en.csv @@ -130,3 +130,40 @@ four hundred,400 five thousand dollars,5000 dollars three thousand five hundred,3500 two billion people,2000000000 people +itchys,itches +placed,placer +ruby,rudy +cannot,could not +peoples,people +dated,date +telephone,cell phone +try,tried +talk,talked +shi,shit +rudolph,rudolf +fimo,famo +holla,holler +pound,pounds +crystals,pistols +duckets,duckers +raytwine,raytron +twine,twan +paul,powell +streets,street +because,cuz +prepai,prepaid +solution,solutions +motherfuka,motherfucker +all right,alright +fire fire,firefires +paul s,pauls +ray tuan,raytron +pre prepa,prepaid +30 two,32 +44 for point 5,44 point 5 +eleventh 2000 and twelve,112012 +eleventh 2 thousand and twelve,112012 +eleventh 2 thousaond and twelve,112012 +25 hundred,2500 +35 hundred,3500 +20 thirteen,2013 diff --git a/tests/e2e/files/gladia-3/es.csv b/tests/e2e/files/gladia-3/es.csv index f847184..5f76f46 100644 --- a/tests/e2e/files/gladia-3/es.csv +++ b/tests/e2e/files/gladia-3/es.csv @@ -36,3 +36,6 @@ cuarenta y cinco,45 setenta y ocho,78 quinientos,500 quince mil,15000 +ahorita,ahora +llama,llame +cuchillo,culchi y yo