study-buddy / app /services /transcript_normalization.py
GitHub Actions
deploy b3d187d69e5df10bb2ec396e89f539f853465a06
14184e3
Raw
History Blame Contribute Delete
6.31 kB
"""Deterministic transcript normalization for voice-clone references.
This intentionally runs only on Voice Lab reference transcripts. Chat/STT
dictation should keep the recognizer's normal user-facing formatting.
"""
from __future__ import annotations
from dataclasses import dataclass
import re
@dataclass(frozen=True)
class TranscriptNormalizationResult:
raw_transcript: str
normalized_transcript: str
changed: bool
_SMALL = (
"zero", "one", "two", "three", "four", "five", "six", "seven", "eight",
"nine", "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen",
"sixteen", "seventeen", "eighteen", "nineteen",
)
_TENS = (
"", "", "twenty", "thirty", "forty", "fifty", "sixty", "seventy",
"eighty", "ninety",
)
_SCALES = (
(1_000_000_000_000, "trillion"),
(1_000_000_000, "billion"),
(1_000_000, "million"),
(1_000, "thousand"),
)
_DIGITS = tuple(_SMALL[:10])
_ORDINAL_DENOMINATORS = {
2: ("half", "halves"),
3: ("third", "thirds"),
4: ("fourth", "fourths"),
5: ("fifth", "fifths"),
6: ("sixth", "sixths"),
7: ("seventh", "sevenths"),
8: ("eighth", "eighths"),
9: ("ninth", "ninths"),
10: ("tenth", "tenths"),
11: ("eleventh", "elevenths"),
12: ("twelfth", "twelfths"),
}
def normalize_voice_reference_transcript(text: str) -> TranscriptNormalizationResult:
raw = text or ""
normalized = raw
normalized = _replace_currency(normalized)
normalized = _replace_percent(normalized)
normalized = _replace_fractions(normalized)
normalized = _replace_decimals(normalized)
normalized = _replace_integers(normalized)
normalized = _tidy_spacing(normalized)
return TranscriptNormalizationResult(
raw_transcript=raw,
normalized_transcript=normalized,
changed=normalized != raw,
)
def _replace_currency(text: str) -> str:
def repl(match: re.Match[str]) -> str:
value = match.group("value")
negative, integer_part, decimal_part = _split_numeric(value)
prefix = "minus " if negative else ""
dollars = _int_to_words(integer_part)
cents = _cents_value(decimal_part)
if cents:
return f"{prefix}{dollars} {_plural(integer_part, 'dollar', 'dollars')} and {_int_to_words(cents)} {_plural(cents, 'cent', 'cents')}"
return f"{prefix}{dollars} {_plural(integer_part, 'dollar', 'dollars')}"
return re.sub(r"\$\s*(?P<value>[+-]?\d[\d,]*(?:\.\d+)?)", repl, text)
def _replace_percent(text: str) -> str:
return re.sub(
r"(?<![\w.])(?P<value>[+-]?\d[\d,]*(?:\.\d+)?)\s*%",
lambda match: f"{_number_to_words(match.group('value'))} percent",
text,
)
def _replace_fractions(text: str) -> str:
def repl(match: re.Match[str]) -> str:
numerator = int(match.group("num").replace(",", ""))
denominator = int(match.group("den").replace(",", ""))
singular, plural = _denominator_to_ordinal(denominator)
return f"{_int_to_words(numerator)} {singular if numerator == 1 else plural}"
return re.sub(
r"(?<![\w.])(?P<num>\d[\d,]*)\s*/\s*(?P<den>\d[\d,]*)(?!\w)",
repl,
text,
)
def _replace_decimals(text: str) -> str:
return re.sub(
r"(?<![\w.])(?P<value>[+-]?\d[\d,]*\.\d+)(?!\w)",
lambda match: _decimal_to_words(match.group("value")),
text,
)
def _replace_integers(text: str) -> str:
return re.sub(
r"(?<![\w.])(?P<value>[+-]?\d[\d,]*)(?!\w)",
lambda match: _integer_token_to_words(match.group("value")),
text,
)
def _number_to_words(value: str) -> str:
if "." in value:
return _decimal_to_words(value)
return _integer_token_to_words(value)
def _integer_token_to_words(value: str) -> str:
try:
negative, integer_part, _decimal_part = _split_numeric(value)
words = _int_to_words(integer_part)
return f"minus {words}" if negative else words
except (TypeError, ValueError):
return value
def _decimal_to_words(value: str) -> str:
try:
negative, integer_part, decimal_part = _split_numeric(value)
words = _int_to_words(integer_part)
digits = " ".join(_DIGITS[int(char)] for char in decimal_part)
prefix = "minus " if negative else ""
return f"{prefix}{words} point {digits}"
except (TypeError, ValueError):
return value
def _split_numeric(value: str) -> tuple[bool, int, str]:
cleaned = value.replace(",", "").strip()
negative = cleaned.startswith("-")
cleaned = cleaned.lstrip("+-")
if "." in cleaned:
integer, decimal = cleaned.split(".", 1)
else:
integer, decimal = cleaned, ""
return negative, int(integer or "0"), decimal
def _int_to_words(value: int) -> str:
if value < 0:
return f"minus {_int_to_words(abs(value))}"
if value < 20:
return _SMALL[value]
if value < 100:
tens, remainder = divmod(value, 10)
return _TENS[tens] if remainder == 0 else f"{_TENS[tens]} {_SMALL[remainder]}"
if value < 1_000:
hundreds, remainder = divmod(value, 100)
head = f"{_SMALL[hundreds]} hundred"
return head if remainder == 0 else f"{head} {_int_to_words(remainder)}"
for scale, label in _SCALES:
if value >= scale:
quotient, remainder = divmod(value, scale)
head = f"{_int_to_words(quotient)} {label}"
return head if remainder == 0 else f"{head} {_int_to_words(remainder)}"
raise ValueError(f"number_out_of_range: {value}")
def _cents_value(decimal_part: str) -> int:
if not decimal_part:
return 0
cents = int((decimal_part + "00")[:2])
return max(0, cents)
def _plural(value: int, singular: str, plural: str) -> str:
return singular if value == 1 else plural
def _denominator_to_ordinal(value: int) -> tuple[str, str]:
if value in _ORDINAL_DENOMINATORS:
return _ORDINAL_DENOMINATORS[value]
base = _int_to_words(value)
if base.endswith("y"):
singular = f"{base[:-1]}ieth"
else:
singular = f"{base}th"
return singular, f"{singular}s"
def _tidy_spacing(text: str) -> str:
text = re.sub(r"\s+([,.!?;:])", r"\1", text)
text = re.sub(r"\s{2,}", " ", text)
return text.strip()