mirror of
https://github.com/denizsafak/abogen.git
synced 2026-07-22 07:10:28 +02:00
Add number normalization and enhance UI for voice preview
- Implemented number conversion to words for grouped numbers in text normalization. - Added configuration options for number conversion in ApostropheConfig. - Updated the normalize_apostrophes function to include number normalization. - Enhanced the dashboard UI with new sections for manuscript and narrator defaults. - Improved voice preview functionality with better handling of audio playback and status updates. - Refactored HTML templates for cleaner structure and added new fields for speaker settings. - Updated CSS styles for improved layout and responsiveness. - Added tests to ensure correct spelling of grouped numbers in the normalization process. - Included num2words as a new dependency in pyproject.toml.
This commit is contained in:
@@ -3,6 +3,10 @@ import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Iterable, List, Optional, Sequence, Tuple
|
||||
try: # pragma: no cover - optional dependency guard
|
||||
from num2words import num2words
|
||||
except Exception: # pragma: no cover - graceful degradation
|
||||
num2words = None # type: ignore
|
||||
|
||||
# ---------- Configuration Dataclass ----------
|
||||
|
||||
@@ -24,6 +28,8 @@ class ApostropheConfig:
|
||||
joiner: str = "" # Replacement used when collapsing internal apostrophes
|
||||
lowercase_for_matching: bool = True # Normalize to lower for rule matching (not output)
|
||||
protect_cultural_names: bool = True # Always keep O'Brien, D'Angelo, etc.
|
||||
convert_numbers: bool = True # Convert grouped numbers such as 12,500 to words
|
||||
number_lang: str = "en" # num2words language code
|
||||
|
||||
# ---------- Dictionaries / Patterns ----------
|
||||
|
||||
@@ -80,6 +86,7 @@ CONTRACTIONS_EXACT = {
|
||||
}
|
||||
|
||||
# For ambiguous 'd and 's we handle separately
|
||||
_NUMBER_WITH_GROUP_RE = re.compile(r"(?<![\w\d])(-?\d{1,3}(?:,\d{3})+)(?![\w\d])")
|
||||
AMBIGUOUS_D_BASES = {"i","you","he","she","we","they"}
|
||||
AMBIGUOUS_S_BASES = {"it","that","what","where","who","when","how","there","here"}
|
||||
|
||||
@@ -528,6 +535,7 @@ def normalize_apostrophes(text: str, cfg: ApostropheConfig | None = None) -> Tup
|
||||
cfg = ApostropheConfig()
|
||||
|
||||
text = normalize_unicode_apostrophes(text)
|
||||
text = _normalize_grouped_numbers(text, cfg)
|
||||
tokens = tokenize(text)
|
||||
|
||||
results = []
|
||||
@@ -542,6 +550,41 @@ def normalize_apostrophes(text: str, cfg: ApostropheConfig | None = None) -> Tup
|
||||
normalized_text = _cleanup_spacing(" ".join(filtered))
|
||||
return normalized_text, results
|
||||
|
||||
def _normalize_grouped_numbers(text: str, cfg: ApostropheConfig) -> str:
|
||||
if not text or not cfg.convert_numbers:
|
||||
return text
|
||||
|
||||
def _replace(match: re.Match[str]) -> str:
|
||||
token = match.group(1)
|
||||
cleaned = token.replace(",", "")
|
||||
if not cleaned:
|
||||
return token
|
||||
negative = cleaned.startswith("-")
|
||||
cleaned_digits = cleaned[1:] if negative else cleaned
|
||||
|
||||
if not cleaned_digits.isdigit():
|
||||
return cleaned_digits if not negative else f"-{cleaned_digits}"
|
||||
|
||||
if num2words is None:
|
||||
return ("-" if negative else "") + cleaned_digits
|
||||
|
||||
try:
|
||||
value = int(cleaned)
|
||||
except ValueError:
|
||||
return cleaned
|
||||
|
||||
language = (cfg.number_lang or "en").strip() or "en"
|
||||
try:
|
||||
words = num2words(abs(value), lang=language)
|
||||
except Exception: # pragma: no cover - unsupported locale
|
||||
return str(value)
|
||||
|
||||
if value < 0:
|
||||
words = f"minus {words}"
|
||||
return words
|
||||
|
||||
return _NUMBER_WITH_GROUP_RE.sub(_replace, text)
|
||||
|
||||
# ---------- Optional phoneme hint post-processing ----------
|
||||
|
||||
def apply_phoneme_hints(text: str, iz_marker="‹IZ›") -> str:
|
||||
@@ -567,6 +610,7 @@ def normalize_for_pipeline(text: str, *, config: Optional[ApostropheConfig] = No
|
||||
normalized, _details = normalize_apostrophes(text, cfg)
|
||||
normalized = expand_titles_and_suffixes(normalized)
|
||||
normalized = ensure_terminal_punctuation(normalized)
|
||||
normalized = _normalize_grouped_numbers(normalized, cfg)
|
||||
if cfg.add_phoneme_hints:
|
||||
normalized = apply_phoneme_hints(normalized, iz_marker=cfg.sibilant_iz_marker)
|
||||
return normalized
|
||||
|
||||
Reference in New Issue
Block a user