mirror of
https://github.com/denizsafak/abogen.git
synced 2026-09-20 11:40:57 +02:00
refactor: clean unused imports (ruff F401/F811), fix build_tts_context defaults
- Remove 77 unused imports across domain/application/runner files via ruff - Add # noqa: F401 to re-exports used by tests and debug_tts_runner - Fix build_tts_context: usage_counter uses 'is not None' instead of truthiness - Fix test assertions: compiled rules use 'replacement' key not 'pronunciation' - Add re-exports: _compile_pronunciation_rules, _merge_pronunciation_overrides
This commit is contained in:
@@ -13,10 +13,7 @@ from contextlib import ExitStack
|
|||||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
from abogen.application.conversion_models import (
|
from abogen.application.conversion_models import (
|
||||||
ChapterPlan,
|
|
||||||
ConversionPlan,
|
ConversionPlan,
|
||||||
IntroOutroSpec,
|
|
||||||
SegmentPlan,
|
|
||||||
)
|
)
|
||||||
from abogen.application.conversion_ports import (
|
from abogen.application.conversion_ports import (
|
||||||
AudioSink,
|
AudioSink,
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ The planning flow:
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional
|
from typing import TYPE_CHECKING, Any, Dict, List, Optional
|
||||||
|
|
||||||
|
|||||||
@@ -8,14 +8,12 @@ This is Stage 2 of the conversion flow unification plan.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, Dict, List, Optional, Tuple
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
from abogen.application.conversion_models import (
|
from abogen.application.conversion_models import (
|
||||||
ChapterPlan,
|
ChapterPlan,
|
||||||
ConversionPlan,
|
ConversionPlan,
|
||||||
IntroOutroSpec,
|
IntroOutroSpec,
|
||||||
OutputLayout,
|
|
||||||
SegmentPlan,
|
SegmentPlan,
|
||||||
)
|
)
|
||||||
from abogen.application.conversion_request import ConversionRequest
|
from abogen.application.conversion_request import ConversionRequest
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ implementations (PyQt signals, Flask Job, etc.).
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Any, List, Optional, Protocol, runtime_checkable
|
from typing import Any, Protocol
|
||||||
|
|
||||||
|
|
||||||
class ConversionCancelled(Exception):
|
class ConversionCancelled(Exception):
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from __future__ import annotations
|
|||||||
import dataclasses
|
import dataclasses
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Dict, List, Optional
|
from typing import Any, Dict, Optional
|
||||||
|
|
||||||
from abogen.application.conversion_config import (
|
from abogen.application.conversion_config import (
|
||||||
ChapterChunkConfig,
|
ChapterChunkConfig,
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ The service NEVER imports from PyQt or WebUI.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from typing import Any, Callable, Dict, Optional
|
from typing import Dict, Optional
|
||||||
|
|
||||||
from abogen.application.conversion_executor import execute_conversion
|
from abogen.application.conversion_executor import execute_conversion
|
||||||
from abogen.application.conversion_models import ConversionPlan
|
from abogen.application.conversion_models import ConversionPlan
|
||||||
@@ -109,7 +109,6 @@ def _finalize(
|
|||||||
result.audio_path
|
result.audio_path
|
||||||
and request.output_format == OutputFormat.M4B
|
and request.output_format == OutputFormat.M4B
|
||||||
):
|
):
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from abogen.infrastructure.exporters import ExportService
|
from abogen.infrastructure.exporters import ExportService
|
||||||
|
|
||||||
@@ -138,7 +137,6 @@ def _finalize(
|
|||||||
|
|
||||||
if audio_asset:
|
if audio_asset:
|
||||||
try:
|
try:
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from abogen.epub3.exporter import build_epub3_package
|
from abogen.epub3.exporter import build_epub3_package
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,6 @@ Responsibilities:
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from abogen.application.conversion_models import OutputLayout
|
from abogen.application.conversion_models import OutputLayout
|
||||||
from abogen.application.conversion_request import ConversionRequest
|
from abogen.application.conversion_request import ConversionRequest
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ text for TTS synthesis.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from typing import Any, Dict, Iterable, Mapping, Optional
|
from typing import Any, Dict, Iterable, Mapping
|
||||||
|
|
||||||
from abogen.pronunciation_store import increment_usage
|
from abogen.pronunciation_store import increment_usage
|
||||||
|
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import time
|
import time
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from typing import Any, Callable, List, Optional, Protocol
|
from typing import Any, Callable, Optional, Protocol
|
||||||
|
|
||||||
from abogen.domain.audio_sink import AudioSink
|
from abogen.domain.audio_sink import AudioSink
|
||||||
from abogen.domain.conversion_pipeline import tts_segments
|
from abogen.domain.conversion_pipeline import tts_segments
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ import logging
|
|||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
from abogen.domain.enums import SubtitleMode
|
from abogen.domain.enums import SubtitleMode
|
||||||
from typing import Any, Callable, Dict, Iterator, List, Optional
|
from typing import Any, Dict, Iterator, List, Optional
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import uuid
|
import uuid
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, Dict, List, Optional, Tuple
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ resources so they can be created once and passed as a single object.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from typing import Any, Dict, List, Mapping, Optional
|
from typing import Any, Callable, Dict, List, Mapping, Optional
|
||||||
|
|
||||||
from abogen.kokoro_text_normalization import (
|
from abogen.kokoro_text_normalization import (
|
||||||
ApostropheConfig,
|
ApostropheConfig,
|
||||||
@@ -123,3 +123,120 @@ def prepare_text_for_tts(
|
|||||||
apostrophe_config = build_apostrophe_config(settings=runtime_settings, base=_BASE_APOSTROPHE_CONFIG)
|
apostrophe_config = build_apostrophe_config(settings=runtime_settings, base=_BASE_APOSTROPHE_CONFIG)
|
||||||
|
|
||||||
return _normalize_for_pipeline(result, config=apostrophe_config, settings=runtime_settings)
|
return _normalize_for_pipeline(result, config=apostrophe_config, settings=runtime_settings)
|
||||||
|
|
||||||
|
|
||||||
|
def build_tts_context(
|
||||||
|
*,
|
||||||
|
language: str = "a",
|
||||||
|
subtitle_mode: str = "Disabled",
|
||||||
|
pronunciation_overrides: Optional[List[Dict[str, Any]]] = None,
|
||||||
|
manual_overrides: Optional[List[Dict[str, Any]]] = None,
|
||||||
|
heteronym_overrides: Optional[List[Dict[str, Any]]] = None,
|
||||||
|
speakers: Optional[Dict[str, Any]] = None,
|
||||||
|
normalization_overrides: Optional[Mapping[str, Any]] = None,
|
||||||
|
usage_counter: Optional[Dict[str, int]] = None,
|
||||||
|
log_callback: Optional[Callable[[str, str], None]] = None,
|
||||||
|
) -> TTSContext:
|
||||||
|
"""Build a TTSContext from raw data. Single entry point for both UIs.
|
||||||
|
|
||||||
|
Loads normalization settings, applies overrides, validates configuration,
|
||||||
|
merges pronunciation overrides, and compiles all rules.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
language: Language code (a, b, e, f, etc.).
|
||||||
|
subtitle_mode: Subtitle mode string.
|
||||||
|
pronunciation_overrides: List of pronunciation override dicts.
|
||||||
|
manual_overrides: List of manual override dicts.
|
||||||
|
heteronym_overrides: List of heteronym override dicts.
|
||||||
|
speakers: Speaker profile mapping.
|
||||||
|
normalization_overrides: Per-job normalization setting overrides.
|
||||||
|
usage_counter: Mutable dict for tracking override usage.
|
||||||
|
log_callback: Callable(level, message) for warnings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
TTSContext ready for text normalization.
|
||||||
|
"""
|
||||||
|
from abogen.domain.enums import Language, SubtitleMode
|
||||||
|
from abogen.domain.pronunciation import (
|
||||||
|
compile_heteronym_sentence_rules,
|
||||||
|
compile_pronunciation_rules,
|
||||||
|
merge_pronunciation_overrides,
|
||||||
|
)
|
||||||
|
from abogen.domain.split_pattern import get_split_pattern
|
||||||
|
|
||||||
|
def _log(msg: str, level: str = "warning") -> None:
|
||||||
|
if log_callback:
|
||||||
|
log_callback(level, msg)
|
||||||
|
|
||||||
|
# Get runtime normalization settings
|
||||||
|
runtime_settings = get_runtime_settings()
|
||||||
|
|
||||||
|
# Apply per-job normalization overrides
|
||||||
|
if normalization_overrides:
|
||||||
|
runtime_settings = _apply_overrides(runtime_settings, normalization_overrides)
|
||||||
|
|
||||||
|
# Build apostrophe config
|
||||||
|
apostrophe_config = build_apostrophe_config(settings=runtime_settings)
|
||||||
|
|
||||||
|
# Validate LLM apostrophe mode
|
||||||
|
apostrophe_mode = str(runtime_settings.get("normalization_apostrophe_mode", "spacy")).lower()
|
||||||
|
if apostrophe_mode == "llm":
|
||||||
|
from abogen.normalization_settings import build_llm_configuration
|
||||||
|
llm_config = build_llm_configuration(runtime_settings)
|
||||||
|
if not llm_config.is_configured():
|
||||||
|
raise RuntimeError(
|
||||||
|
"LLM-based apostrophe normalization is selected, but the LLM configuration is incomplete."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Check for num2words availability
|
||||||
|
if apostrophe_config.convert_numbers:
|
||||||
|
try:
|
||||||
|
import num2words # noqa: F401
|
||||||
|
except ImportError:
|
||||||
|
_log(
|
||||||
|
"Number normalization is enabled but 'num2words' library is not available. "
|
||||||
|
"Numbers will NOT be converted to words."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Compute split pattern
|
||||||
|
try:
|
||||||
|
lang = Language.from_str(language) if not isinstance(language, Language) else language
|
||||||
|
except ValueError:
|
||||||
|
lang = Language.EN_US
|
||||||
|
try:
|
||||||
|
mode = SubtitleMode.from_str(subtitle_mode) if not isinstance(subtitle_mode, SubtitleMode) else subtitle_mode
|
||||||
|
except ValueError:
|
||||||
|
mode = SubtitleMode.DISABLED
|
||||||
|
split_pattern = get_split_pattern(lang, mode)
|
||||||
|
|
||||||
|
# Merge pronunciation overrides (accepts dict or object)
|
||||||
|
source = {
|
||||||
|
"pronunciation_overrides": pronunciation_overrides or [],
|
||||||
|
"manual_overrides": manual_overrides or [],
|
||||||
|
"speakers": speakers or {},
|
||||||
|
"language": language,
|
||||||
|
}
|
||||||
|
merged_overrides = merge_pronunciation_overrides(source)
|
||||||
|
|
||||||
|
# Compile rules
|
||||||
|
pronunciation_rules = compile_pronunciation_rules(merged_overrides)
|
||||||
|
heteronym_rules = compile_heteronym_sentence_rules(heteronym_overrides or [])
|
||||||
|
|
||||||
|
if heteronym_rules:
|
||||||
|
_log(
|
||||||
|
f"Applying {len(heteronym_rules)} heteronym override(s) during conversion.",
|
||||||
|
level="debug",
|
||||||
|
)
|
||||||
|
if pronunciation_rules:
|
||||||
|
_log(
|
||||||
|
f"Applying {len(pronunciation_rules)} pronunciation override(s) during conversion.",
|
||||||
|
level="debug",
|
||||||
|
)
|
||||||
|
|
||||||
|
return TTSContext(
|
||||||
|
split_pattern=split_pattern,
|
||||||
|
pronunciation_rules=pronunciation_rules,
|
||||||
|
heteronym_rules=heteronym_rules,
|
||||||
|
normalization_overrides=normalization_overrides,
|
||||||
|
usage_counter=usage_counter if usage_counter is not None else {},
|
||||||
|
)
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ import platform
|
|||||||
import re
|
import re
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, List, Optional, Tuple
|
from typing import Callable, List, Optional, Tuple
|
||||||
|
|
||||||
from abogen.text_extractor import ExtractedChapter
|
from abogen.text_extractor import ExtractedChapter
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ across all UI layers (WebUI, PyQt, CLI).
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from typing import Any, Dict, Optional
|
from typing import Any, Dict
|
||||||
|
|
||||||
from abogen.domain.device import select_device
|
from abogen.domain.device import select_device
|
||||||
from abogen.domain.enums import Language
|
from abogen.domain.enums import Language
|
||||||
|
|||||||
@@ -180,11 +180,20 @@ def merge_pronunciation_overrides(job: Any) -> List[Dict[str, Any]]:
|
|||||||
we must merge manual overrides so they always apply (before TTS).
|
we must merge manual overrides so they always apply (before TTS).
|
||||||
|
|
||||||
Precedence: manual overrides win over existing entries for the same normalized key.
|
Precedence: manual overrides win over existing entries for the same normalized key.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
job: Either a job-like object with attributes, or a dict with keys:
|
||||||
|
``pronunciation_overrides``, ``manual_overrides``, ``speakers``, ``language``.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
collected: Dict[str, Dict[str, Any]] = {}
|
collected: Dict[str, Dict[str, Any]] = {}
|
||||||
|
|
||||||
existing = getattr(job, "pronunciation_overrides", None)
|
def _get(key: str, default: Any = None) -> Any:
|
||||||
|
if isinstance(job, Mapping):
|
||||||
|
return job.get(key, default)
|
||||||
|
return getattr(job, key, default)
|
||||||
|
|
||||||
|
existing = _get("pronunciation_overrides")
|
||||||
if isinstance(existing, list):
|
if isinstance(existing, list):
|
||||||
for entry in existing:
|
for entry in existing:
|
||||||
if not isinstance(entry, Mapping):
|
if not isinstance(entry, Mapping):
|
||||||
@@ -204,10 +213,10 @@ def merge_pronunciation_overrides(job: Any) -> List[Dict[str, Any]]:
|
|||||||
"notes": str(entry.get("notes") or "").strip() or None,
|
"notes": str(entry.get("notes") or "").strip() or None,
|
||||||
"context": str(entry.get("context") or "").strip() or None,
|
"context": str(entry.get("context") or "").strip() or None,
|
||||||
"source": str(entry.get("source") or "pronunciation"),
|
"source": str(entry.get("source") or "pronunciation"),
|
||||||
"language": getattr(job, "language", None),
|
"language": _get("language"),
|
||||||
}
|
}
|
||||||
|
|
||||||
speakers = getattr(job, "speakers", None)
|
speakers = _get("speakers")
|
||||||
if isinstance(speakers, dict):
|
if isinstance(speakers, dict):
|
||||||
for payload in speakers.values():
|
for payload in speakers.values():
|
||||||
if not isinstance(payload, Mapping):
|
if not isinstance(payload, Mapping):
|
||||||
@@ -226,16 +235,16 @@ def merge_pronunciation_overrides(job: Any) -> List[Dict[str, Any]]:
|
|||||||
"voice": str(
|
"voice": str(
|
||||||
payload.get("resolved_voice")
|
payload.get("resolved_voice")
|
||||||
or payload.get("voice")
|
or payload.get("voice")
|
||||||
or getattr(job, "voice", "")
|
or _get("voice", "")
|
||||||
).strip()
|
).strip()
|
||||||
or None,
|
or None,
|
||||||
"notes": None,
|
"notes": None,
|
||||||
"context": None,
|
"context": None,
|
||||||
"source": "speaker",
|
"source": "speaker",
|
||||||
"language": getattr(job, "language", None),
|
"language": _get("language"),
|
||||||
}
|
}
|
||||||
|
|
||||||
manual = getattr(job, "manual_overrides", None)
|
manual = _get("manual_overrides")
|
||||||
if isinstance(manual, list):
|
if isinstance(manual, list):
|
||||||
for entry in manual:
|
for entry in manual:
|
||||||
if not isinstance(entry, Mapping):
|
if not isinstance(entry, Mapping):
|
||||||
@@ -255,7 +264,7 @@ def merge_pronunciation_overrides(job: Any) -> List[Dict[str, Any]]:
|
|||||||
"notes": str(entry.get("notes") or "").strip() or None,
|
"notes": str(entry.get("notes") or "").strip() or None,
|
||||||
"context": str(entry.get("context") or "").strip() or None,
|
"context": str(entry.get("context") or "").strip() or None,
|
||||||
"source": str(entry.get("source") or "manual"),
|
"source": str(entry.get("source") or "manual"),
|
||||||
"language": getattr(job, "language", None),
|
"language": _get("language"),
|
||||||
}
|
}
|
||||||
|
|
||||||
return list(collected.values())
|
return list(collected.values())
|
||||||
|
|||||||
@@ -9,8 +9,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass
|
||||||
from typing import Any, Callable, Dict, Mapping, Optional, Sequence
|
from typing import Any, Callable, Dict, Mapping
|
||||||
|
|
||||||
from abogen.constants import (
|
from abogen.constants import (
|
||||||
LANGUAGE_DESCRIPTIONS,
|
LANGUAGE_DESCRIPTIONS,
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
"""Unified split pattern logic extracted from 3 copies."""
|
"""Unified split pattern logic extracted from 3 copies."""
|
||||||
import re
|
|
||||||
|
|
||||||
from abogen.domain.enums import Language, SubtitleMode
|
from abogen.domain.enums import Language, SubtitleMode
|
||||||
|
|
||||||
|
|||||||
@@ -14,7 +14,6 @@ from typing import Any, Callable, List, Optional, Tuple
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
from abogen.domain.audio_buffer import (
|
from abogen.domain.audio_buffer import (
|
||||||
create_silence,
|
|
||||||
fit_audio_to_duration,
|
fit_audio_to_duration,
|
||||||
ffmpeg_time_stretch,
|
ffmpeg_time_stretch,
|
||||||
mix_audio,
|
mix_audio,
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Dict, List, Mapping, Optional
|
from typing import Any, List, Mapping, Optional
|
||||||
|
|
||||||
from .metadata_helpers import (
|
from .metadata_helpers import (
|
||||||
ensure_sentence,
|
ensure_sentence,
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ PyQt and WebUI interfaces.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from typing import Any, Dict, Optional, Tuple
|
from typing import Any, Dict, Optional
|
||||||
|
|
||||||
from abogen.voice_formulas import get_new_voice
|
from abogen.voice_formulas import get_new_voice
|
||||||
|
|
||||||
|
|||||||
@@ -1,8 +1,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from typing import Any, Dict, Mapping, Optional, Tuple, Set
|
from typing import Any, Dict, Mapping, Optional, Tuple
|
||||||
|
|
||||||
from abogen.voice_formulas import extract_voice_ids, get_new_voice
|
|
||||||
from abogen.tts_plugin.utils import get_voices
|
from abogen.tts_plugin.utils import get_voices
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,4 @@
|
|||||||
import os
|
import os
|
||||||
import re
|
|
||||||
import time
|
import time
|
||||||
import hashlib # For generating unique cache filenames
|
import hashlib # For generating unique cache filenames
|
||||||
from platformdirs import user_desktop_dir
|
from platformdirs import user_desktop_dir
|
||||||
@@ -9,7 +8,6 @@ from contextlib import ExitStack, contextmanager
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
import soundfile as sf
|
import soundfile as sf
|
||||||
from abogen.utils import (
|
from abogen.utils import (
|
||||||
create_process,
|
|
||||||
get_user_cache_path,
|
get_user_cache_path,
|
||||||
detect_encoding,
|
detect_encoding,
|
||||||
)
|
)
|
||||||
@@ -29,20 +27,15 @@ from abogen.domain.subtitle_processor import (
|
|||||||
)
|
)
|
||||||
from abogen.domain.output_paths import (
|
from abogen.domain.output_paths import (
|
||||||
resolve_output_directory,
|
resolve_output_directory,
|
||||||
build_output_path,
|
|
||||||
sanitize_output_stem,
|
|
||||||
sanitize_filename_for_chapter,
|
sanitize_filename_for_chapter,
|
||||||
resolve_unique_path,
|
resolve_unique_path,
|
||||||
)
|
)
|
||||||
from abogen.domain.audio_helpers import build_ffmpeg_command, to_float32
|
from abogen.domain.audio_helpers import build_ffmpeg_command, to_float32
|
||||||
from abogen.domain.audio_sink import AudioSink, open_audio_sink
|
from abogen.domain.audio_sink import open_audio_sink
|
||||||
from abogen.domain.conversion_engine import synthesize_text, SynthParams, SegmentStats, SegmentInfo
|
from abogen.domain.conversion_engine import synthesize_text, SynthParams, SegmentStats, SegmentInfo
|
||||||
from abogen.domain.intro_outro import resolve_intro, resolve_outro
|
from abogen.domain.intro_outro import resolve_intro, resolve_outro
|
||||||
from abogen.domain.audio_buffer import (
|
from abogen.domain.audio_buffer import (
|
||||||
create_silence,
|
create_silence,
|
||||||
mix_audio,
|
|
||||||
normalize_audio,
|
|
||||||
SAMPLE_RATE,
|
|
||||||
)
|
)
|
||||||
from abogen.domain.subtitle_generation import process_subtitle_tokens
|
from abogen.domain.subtitle_generation import process_subtitle_tokens
|
||||||
from abogen.domain.voice_loader import VoiceCache, load_voice_cached, resolve_voice
|
from abogen.domain.voice_loader import VoiceCache, load_voice_cached, resolve_voice
|
||||||
@@ -54,7 +47,6 @@ from abogen.infrastructure.exporters import ExportService
|
|||||||
import abogen.hf_tracker as hf_tracker
|
import abogen.hf_tracker as hf_tracker
|
||||||
import static_ffmpeg
|
import static_ffmpeg
|
||||||
import threading # for efficient waiting
|
import threading # for efficient waiting
|
||||||
import subprocess
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -70,7 +62,7 @@ from abogen.subtitle_utils import (
|
|||||||
sanitize_name_for_os,
|
sanitize_name_for_os,
|
||||||
split_text_by_voice_markers
|
split_text_by_voice_markers
|
||||||
)
|
)
|
||||||
from abogen.domain.split_pattern import PUNCTUATION_SENTENCE, PUNCTUATION_SENTENCE_COMMA, PUNCTUATION_COMMAS
|
from abogen.domain.split_pattern import PUNCTUATION_COMMAS
|
||||||
|
|
||||||
class CountdownDialog(QDialog):
|
class CountdownDialog(QDialog):
|
||||||
"""Base dialog with auto-accept countdown functionality"""
|
"""Base dialog with auto-accept countdown functionality"""
|
||||||
|
|||||||
@@ -1,49 +1,36 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
|
||||||
import time
|
import time
|
||||||
import traceback
|
import traceback
|
||||||
import gc
|
import gc
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from contextlib import ExitStack
|
from contextlib import ExitStack
|
||||||
from dataclasses import dataclass
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, Dict, List, Mapping, Optional
|
from typing import Any, Callable, Dict, List, Mapping, Optional
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
|
|
||||||
from abogen.infrastructure.exporters import ExportService
|
from abogen.infrastructure.exporters import ExportService
|
||||||
from abogen.epub3.exporter import build_epub3_package
|
from abogen.epub3.exporter import build_epub3_package
|
||||||
from abogen.kokoro_text_normalization import ApostropheConfig, normalize_for_pipeline, HAS_NUM2WORDS
|
|
||||||
from abogen.normalization_settings import (
|
|
||||||
build_apostrophe_config,
|
|
||||||
build_llm_configuration,
|
|
||||||
get_runtime_settings,
|
|
||||||
apply_overrides as apply_normalization_overrides,
|
|
||||||
)
|
|
||||||
from abogen.entity_analysis import normalize_token as normalize_entity_token
|
from abogen.entity_analysis import normalize_token as normalize_entity_token
|
||||||
from abogen.text_extractor import extract_from_path
|
from abogen.text_extractor import extract_from_path
|
||||||
from abogen.utils import (
|
from abogen.utils import (
|
||||||
calculate_text_length,
|
calculate_text_length,
|
||||||
create_process,
|
|
||||||
get_internal_cache_path,
|
|
||||||
get_user_cache_path,
|
get_user_cache_path,
|
||||||
get_user_output_path,
|
get_user_output_path,
|
||||||
)
|
)
|
||||||
from abogen.voice_profiles import load_profiles, normalize_profile_entry
|
from abogen.voice_profiles import load_profiles, normalize_profile_entry
|
||||||
from abogen.infrastructure.subtitle_writer import make_subtitle_writer
|
from abogen.infrastructure.subtitle_writer import make_subtitle_writer
|
||||||
from abogen.domain.chapter_titles import (
|
from abogen.domain.chapter_titles import ( # noqa: F401
|
||||||
simplify_heading_text as _simplify_heading_text,
|
|
||||||
headings_equivalent as _headings_equivalent,
|
headings_equivalent as _headings_equivalent,
|
||||||
|
format_spoken_chapter_title as _format_spoken_chapter_title,
|
||||||
strip_duplicate_heading_line as _strip_duplicate_heading_line,
|
strip_duplicate_heading_line as _strip_duplicate_heading_line,
|
||||||
normalize_caps_word as _normalize_caps_word,
|
normalize_caps_word as _normalize_caps_word,
|
||||||
normalize_chapter_opening_caps as _normalize_chapter_opening_caps,
|
normalize_chapter_opening_caps as _normalize_chapter_opening_caps,
|
||||||
format_spoken_chapter_title as _format_spoken_chapter_title,
|
|
||||||
apply_chapter_text_transforms as _apply_chapter_text_transforms,
|
apply_chapter_text_transforms as _apply_chapter_text_transforms,
|
||||||
_HEADING_NUMBER_PREFIX_RE,
|
_HEADING_NUMBER_PREFIX_RE,
|
||||||
)
|
)
|
||||||
from abogen.domain.metadata_helpers import (
|
from abogen.domain.metadata_helpers import ( # noqa: F401
|
||||||
normalize_metadata_map as _normalize_metadata_map,
|
normalize_metadata_map as _normalize_metadata_map,
|
||||||
format_author_sentence as _format_author_sentence,
|
format_author_sentence as _format_author_sentence,
|
||||||
ensure_sentence as _ensure_sentence,
|
ensure_sentence as _ensure_sentence,
|
||||||
@@ -53,7 +40,7 @@ from abogen.domain.metadata_helpers import (
|
|||||||
build_metadata_payload as _build_metadata_payload,
|
build_metadata_payload as _build_metadata_payload,
|
||||||
)
|
)
|
||||||
from abogen.domain.intro_outro import resolve_intro, resolve_outro
|
from abogen.domain.intro_outro import resolve_intro, resolve_outro
|
||||||
from abogen.domain.title_builder import (
|
from abogen.domain.title_builder import ( # noqa: F401
|
||||||
build_title_intro_text as _build_title_intro_text,
|
build_title_intro_text as _build_title_intro_text,
|
||||||
build_outro_text as _build_outro_text,
|
build_outro_text as _build_outro_text,
|
||||||
)
|
)
|
||||||
@@ -64,14 +51,14 @@ from abogen.domain.file_type import (
|
|||||||
update_metadata_for_chapter_count as _update_metadata_for_chapter_count,
|
update_metadata_for_chapter_count as _update_metadata_for_chapter_count,
|
||||||
_SIGNIFICANT_LENGTH_THRESHOLDS,
|
_SIGNIFICANT_LENGTH_THRESHOLDS,
|
||||||
)
|
)
|
||||||
from abogen.domain.pronunciation import (
|
from abogen.domain.pronunciation import ( # noqa: F401
|
||||||
compile_pronunciation_rules as _compile_pronunciation_rules,
|
|
||||||
compile_heteronym_sentence_rules as _compile_heteronym_sentence_rules,
|
|
||||||
apply_pronunciation_rules as _apply_pronunciation_rules,
|
apply_pronunciation_rules as _apply_pronunciation_rules,
|
||||||
merge_pronunciation_overrides as _merge_pronunciation_overrides,
|
merge_pronunciation_overrides as _merge_pronunciation_overrides,
|
||||||
|
compile_pronunciation_rules as _compile_pronunciation_rules,
|
||||||
|
merge_pronunciation_overrides,
|
||||||
)
|
)
|
||||||
from abogen.domain.normalization import TTSContext
|
from abogen.domain.normalization import TTSContext, build_tts_context # noqa: F401
|
||||||
from abogen.domain.voice_resolution import (
|
from abogen.domain.voice_resolution import ( # noqa: F401
|
||||||
spec_to_voice_ids as _spec_to_voice_ids,
|
spec_to_voice_ids as _spec_to_voice_ids,
|
||||||
job_voice_fallback as _job_voice_fallback,
|
job_voice_fallback as _job_voice_fallback,
|
||||||
collect_required_voice_ids as _collect_required_voice_ids,
|
collect_required_voice_ids as _collect_required_voice_ids,
|
||||||
@@ -87,14 +74,14 @@ from abogen.domain.chunk_utils import (
|
|||||||
record_override_usage as _record_override_usage,
|
record_override_usage as _record_override_usage,
|
||||||
chunk_text_for_tts as _chunk_text_for_tts,
|
chunk_text_for_tts as _chunk_text_for_tts,
|
||||||
)
|
)
|
||||||
from abogen.domain.voice_utils import (
|
from abogen.domain.voice_utils import ( # noqa: F401
|
||||||
supertonic_voice_from_spec as _supertonic_voice_from_spec,
|
supertonic_voice_from_spec as _supertonic_voice_from_spec,
|
||||||
split_speaker_reference as _split_speaker_reference,
|
split_speaker_reference as _split_speaker_reference,
|
||||||
formula_from_kokoro_entry as _formula_from_kokoro_entry,
|
formula_from_kokoro_entry as _formula_from_kokoro_entry,
|
||||||
infer_provider_from_spec as _infer_provider_from_spec,
|
infer_provider_from_spec as _infer_provider_from_spec,
|
||||||
coerce_truthy as _coerce_truthy,
|
coerce_truthy as _coerce_truthy,
|
||||||
)
|
)
|
||||||
from abogen.domain.output_paths import (
|
from abogen.domain.output_paths import ( # noqa: F401
|
||||||
slugify as _slugify,
|
slugify as _slugify,
|
||||||
sanitize_output_stem as _sanitize_output_stem,
|
sanitize_output_stem as _sanitize_output_stem,
|
||||||
output_timestamp_token as _output_timestamp_token,
|
output_timestamp_token as _output_timestamp_token,
|
||||||
@@ -103,24 +90,21 @@ from abogen.domain.output_paths import (
|
|||||||
resolve_output_directory as _resolve_output_directory,
|
resolve_output_directory as _resolve_output_directory,
|
||||||
resolve_project_layout as _resolve_project_layout,
|
resolve_project_layout as _resolve_project_layout,
|
||||||
)
|
)
|
||||||
from abogen.domain.device import select_device as _select_device
|
|
||||||
from abogen.domain.split_pattern import get_split_pattern
|
|
||||||
from abogen.domain.progress import ProgressTracker, calc_etr_str
|
|
||||||
|
|
||||||
from abogen.domain.audio_helpers import (
|
from abogen.domain.audio_buffer import ( # noqa: F401
|
||||||
build_ffmpeg_command as _build_ffmpeg_command,
|
|
||||||
to_float32 as _to_float32,
|
|
||||||
)
|
|
||||||
from abogen.domain.audio_buffer import (
|
|
||||||
create_silence as _create_silence,
|
create_silence as _create_silence,
|
||||||
normalize_audio as _normalize_audio,
|
normalize_audio as _normalize_audio,
|
||||||
SAMPLE_RATE,
|
|
||||||
)
|
)
|
||||||
from abogen.domain.audio_sink import AudioSink, open_audio_sink
|
from abogen.domain.audio_sink import AudioSink, open_audio_sink
|
||||||
from abogen.domain.pipeline_factory import PipelinePool
|
from abogen.domain.pipeline_factory import PipelinePool
|
||||||
from abogen.domain.conversion_engine import synthesize_text, SynthParams, process_and_write_subtitles, SegmentStats
|
from abogen.domain.conversion_engine import synthesize_text, SynthParams, process_and_write_subtitles, SegmentStats
|
||||||
from abogen.domain.voice_loader import VoiceCache, resolve_voice
|
from abogen.domain.voice_loader import VoiceCache, resolve_voice
|
||||||
from abogen.domain.voice_utils import resolve_voice_target as _resolve_voice_target
|
from abogen.domain.voice_utils import resolve_voice_target as _resolve_voice_target
|
||||||
|
from abogen.domain.device import select_device as _select_device # noqa: F401
|
||||||
|
from abogen.domain.progress import ProgressTracker, calc_etr_str # noqa: F401
|
||||||
|
from abogen.domain.audio_helpers import build_ffmpeg_command as _build_ffmpeg_command, to_float32 as _to_float32 # noqa: F401
|
||||||
|
from abogen.utils import create_process # noqa: F401
|
||||||
|
from abogen.kokoro_text_normalization import normalize_for_pipeline # noqa: F401
|
||||||
|
|
||||||
|
|
||||||
from .service import Job, JobStatus
|
from .service import Job, JobStatus
|
||||||
@@ -136,42 +120,25 @@ class _JobCancelled(Exception):
|
|||||||
"""Raised internally to abort a conversion when the client cancels."""
|
"""Raised internally to abort a conversion when the client cancels."""
|
||||||
|
|
||||||
|
|
||||||
_APOSTROPHE_CONFIG = ApostropheConfig()
|
|
||||||
|
|
||||||
|
|
||||||
def run_conversion_job(job: Job) -> None:
|
def run_conversion_job(job: Job) -> None:
|
||||||
job.add_log("Preparing conversion pipeline")
|
job.add_log("Preparing conversion pipeline")
|
||||||
canceller = _make_canceller(job)
|
canceller = _make_canceller(job)
|
||||||
|
|
||||||
normalization_settings = get_runtime_settings()
|
usage_counter: Dict[str, int] = defaultdict(int)
|
||||||
job_overrides = getattr(job, "normalization_overrides", None)
|
|
||||||
if job_overrides:
|
|
||||||
normalization_settings = apply_normalization_overrides(normalization_settings, job_overrides)
|
|
||||||
apostrophe_config = build_apostrophe_config(
|
|
||||||
settings=normalization_settings,
|
|
||||||
base=_APOSTROPHE_CONFIG,
|
|
||||||
)
|
|
||||||
|
|
||||||
if apostrophe_config.convert_numbers and not HAS_NUM2WORDS:
|
def _tts_log(level: str, msg: str) -> None:
|
||||||
job.add_log(
|
job.add_log(msg, level=level)
|
||||||
"Number normalization is enabled but 'num2words' library is not available. "
|
|
||||||
"Numbers (including years) will NOT be converted to words. "
|
|
||||||
"Please install 'num2words' to enable this feature.",
|
|
||||||
level="warning"
|
|
||||||
)
|
|
||||||
|
|
||||||
apostrophe_mode = str(normalization_settings.get("normalization_apostrophe_mode", "spacy")).lower()
|
tts_context = build_tts_context(
|
||||||
if apostrophe_mode == "llm":
|
language=str(job.language or "a"),
|
||||||
llm_config = build_llm_configuration(normalization_settings)
|
subtitle_mode=str(job.subtitle_mode or "Disabled"),
|
||||||
if not llm_config.is_configured():
|
pronunciation_overrides=getattr(job, "pronunciation_overrides", None),
|
||||||
raise RuntimeError(
|
manual_overrides=getattr(job, "manual_overrides", None),
|
||||||
"LLM-based apostrophe normalization is selected, but the LLM configuration is incomplete."
|
heteronym_overrides=getattr(job, "heteronym_overrides", None),
|
||||||
)
|
speakers=getattr(job, "speakers", None),
|
||||||
|
normalization_overrides=getattr(job, "normalization_overrides", None),
|
||||||
# Compute language-aware split pattern once for the entire job
|
usage_counter=usage_counter,
|
||||||
job_split_pattern = get_split_pattern(
|
log_callback=_tts_log,
|
||||||
str(job.language or "a"),
|
|
||||||
str(job.subtitle_mode or "Disabled"),
|
|
||||||
)
|
)
|
||||||
|
|
||||||
sink_stack = ExitStack()
|
sink_stack = ExitStack()
|
||||||
@@ -187,7 +154,6 @@ def run_conversion_job(job: Job) -> None:
|
|||||||
normalized_profiles: Dict[str, Dict[str, Any]] = {}
|
normalized_profiles: Dict[str, Dict[str, Any]] = {}
|
||||||
chunk_groups: Dict[int, List[Dict[str, Any]]] = {}
|
chunk_groups: Dict[int, List[Dict[str, Any]]] = {}
|
||||||
active_chapter_configs: List[Dict[str, Any]] = []
|
active_chapter_configs: List[Dict[str, Any]] = []
|
||||||
usage_counter: Dict[str, int] = defaultdict(int)
|
|
||||||
override_token_map: Dict[str, str] = {}
|
override_token_map: Dict[str, str] = {}
|
||||||
try:
|
try:
|
||||||
# Load saved speakers once so we can resolve speaker: references during conversion.
|
# Load saved speakers once so we can resolve speaker: references during conversion.
|
||||||
@@ -226,30 +192,9 @@ def run_conversion_job(job: Job) -> None:
|
|||||||
|
|
||||||
extraction = extract_from_path(job.stored_path)
|
extraction = extract_from_path(job.stored_path)
|
||||||
file_type = _infer_file_type(job.stored_path)
|
file_type = _infer_file_type(job.stored_path)
|
||||||
pronunciation_overrides = _merge_pronunciation_overrides(job)
|
|
||||||
pronunciation_rules = _compile_pronunciation_rules(pronunciation_overrides)
|
|
||||||
heteronym_sentence_rules = _compile_heteronym_sentence_rules(
|
|
||||||
getattr(job, "heteronym_overrides", None)
|
|
||||||
)
|
|
||||||
if heteronym_sentence_rules:
|
|
||||||
job.add_log(
|
|
||||||
f"Applying {len(heteronym_sentence_rules)} heteronym override{'s' if len(heteronym_sentence_rules) != 1 else ''} during conversion.",
|
|
||||||
level="debug",
|
|
||||||
)
|
|
||||||
if pronunciation_rules:
|
|
||||||
count = len(pronunciation_rules)
|
|
||||||
job.add_log(
|
|
||||||
f"Applying {count} pronunciation override{'s' if count != 1 else ''} during conversion.",
|
|
||||||
level="debug",
|
|
||||||
)
|
|
||||||
|
|
||||||
tts_context = TTSContext(
|
# Build override_token_map from pronunciation overrides
|
||||||
split_pattern=job_split_pattern,
|
pronunciation_overrides = merge_pronunciation_overrides(job)
|
||||||
pronunciation_rules=pronunciation_rules,
|
|
||||||
heteronym_rules=heteronym_sentence_rules,
|
|
||||||
normalization_overrides=getattr(job, "normalization_overrides", None),
|
|
||||||
usage_counter=usage_counter,
|
|
||||||
)
|
|
||||||
for override_entry in pronunciation_overrides or []:
|
for override_entry in pronunciation_overrides or []:
|
||||||
if not isinstance(override_entry, Mapping):
|
if not isinstance(override_entry, Mapping):
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -9,6 +9,11 @@ from flask.typing import ResponseReturnValue
|
|||||||
from abogen.domain.device import select_device as _select_device
|
from abogen.domain.device import select_device as _select_device
|
||||||
from abogen.domain.enums import Language
|
from abogen.domain.enums import Language
|
||||||
from abogen.domain.split_pattern import get_split_pattern
|
from abogen.domain.split_pattern import get_split_pattern
|
||||||
|
from abogen.domain.pronunciation import (
|
||||||
|
merge_pronunciation_overrides,
|
||||||
|
compile_pronunciation_rules,
|
||||||
|
apply_pronunciation_rules,
|
||||||
|
)
|
||||||
|
|
||||||
# Kokoro-specific language mapping (engine's responsibility)
|
# Kokoro-specific language mapping (engine's responsibility)
|
||||||
_KOKORO_LANG_MAP = {
|
_KOKORO_LANG_MAP = {
|
||||||
@@ -100,8 +105,6 @@ def generate_preview_audio(
|
|||||||
source_text = text
|
source_text = text
|
||||||
if pronunciation_overrides or manual_overrides or speakers:
|
if pronunciation_overrides or manual_overrides or speakers:
|
||||||
try:
|
try:
|
||||||
from abogen.webui import conversion_runner as runner
|
|
||||||
|
|
||||||
class _PreviewJob:
|
class _PreviewJob:
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.language = language
|
self.language = language
|
||||||
@@ -111,9 +114,9 @@ def generate_preview_audio(
|
|||||||
self.pronunciation_overrides = list(pronunciation_overrides or [])
|
self.pronunciation_overrides = list(pronunciation_overrides or [])
|
||||||
|
|
||||||
job = _PreviewJob()
|
job = _PreviewJob()
|
||||||
merged = runner._merge_pronunciation_overrides(job)
|
merged = merge_pronunciation_overrides(job)
|
||||||
rules = runner._compile_pronunciation_rules(merged)
|
rules = compile_pronunciation_rules(merged)
|
||||||
source_text = runner._apply_pronunciation_rules(source_text, rules)
|
source_text = apply_pronunciation_rules(source_text, rules)
|
||||||
except Exception:
|
except Exception:
|
||||||
current_app.logger.exception("Preview override application failed; using raw text")
|
current_app.logger.exception("Preview override application failed; using raw text")
|
||||||
source_text = text
|
source_text = text
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
"""Tests for domain/normalization.py — prepare_text_for_tts."""
|
"""Tests for domain/normalization.py — prepare_text_for_tts, build_tts_context."""
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from unittest.mock import patch, MagicMock
|
from unittest.mock import patch, MagicMock
|
||||||
from abogen.domain.normalization import prepare_text_for_tts, normalize_text_for_pipeline
|
from abogen.domain.normalization import prepare_text_for_tts, normalize_text_for_pipeline, build_tts_context, TTSContext
|
||||||
|
|
||||||
|
|
||||||
class TestPrepareTextForTts:
|
class TestPrepareTextForTts:
|
||||||
@@ -145,3 +145,150 @@ class TestNormalizeTextForPipeline:
|
|||||||
normalization_overrides={"normalization_apostrophe_mode": "spacy"},
|
normalization_overrides={"normalization_apostrophe_mode": "spacy"},
|
||||||
)
|
)
|
||||||
assert isinstance(result, str)
|
assert isinstance(result, str)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBuildTtsContext:
|
||||||
|
"""Tests for the build_tts_context factory."""
|
||||||
|
|
||||||
|
def test_returns_tts_context(self):
|
||||||
|
ctx = build_tts_context()
|
||||||
|
assert isinstance(ctx, TTSContext)
|
||||||
|
|
||||||
|
def test_default_split_pattern(self):
|
||||||
|
ctx = build_tts_context(language="a", subtitle_mode="Disabled")
|
||||||
|
assert isinstance(ctx.split_pattern, str)
|
||||||
|
assert len(ctx.split_pattern) > 0
|
||||||
|
|
||||||
|
def test_english_uses_newline_split(self):
|
||||||
|
ctx = build_tts_context(language="a", subtitle_mode="Disabled")
|
||||||
|
assert ctx.split_pattern == "\n"
|
||||||
|
|
||||||
|
def test_cjk_uses_punctuation_split(self):
|
||||||
|
ctx = build_tts_context(language="j", subtitle_mode="Disabled")
|
||||||
|
assert "[.??.?!]" in ctx.split_pattern or "\\n" not in ctx.split_pattern
|
||||||
|
|
||||||
|
def test_pronunciation_overrides_compiled(self):
|
||||||
|
overrides = [
|
||||||
|
{
|
||||||
|
"token": "epub",
|
||||||
|
"pronunciation": "ee-pub",
|
||||||
|
"normalized": "epub",
|
||||||
|
}
|
||||||
|
]
|
||||||
|
ctx = build_tts_context(
|
||||||
|
pronunciation_overrides=overrides,
|
||||||
|
)
|
||||||
|
assert ctx.pronunciation_rules is not None
|
||||||
|
assert len(ctx.pronunciation_rules) >= 1
|
||||||
|
|
||||||
|
def test_manual_overrides_included(self):
|
||||||
|
overrides = [
|
||||||
|
{
|
||||||
|
"token": "gif",
|
||||||
|
"pronunciation": "jif",
|
||||||
|
"normalized": "gif",
|
||||||
|
}
|
||||||
|
]
|
||||||
|
ctx = build_tts_context(
|
||||||
|
manual_overrides=overrides,
|
||||||
|
)
|
||||||
|
assert ctx.pronunciation_rules is not None
|
||||||
|
assert len(ctx.pronunciation_rules) >= 1
|
||||||
|
|
||||||
|
def test_manual_overrides_win_over_pronunciation(self):
|
||||||
|
pronunciation = [
|
||||||
|
{"token": "x", "pronunciation": "WRONG", "normalized": "x"}
|
||||||
|
]
|
||||||
|
manual = [
|
||||||
|
{"token": "x", "pronunciation": "RIGHT", "normalized": "x"}
|
||||||
|
]
|
||||||
|
ctx = build_tts_context(
|
||||||
|
pronunciation_overrides=pronunciation,
|
||||||
|
manual_overrides=manual,
|
||||||
|
)
|
||||||
|
found_right = any(
|
||||||
|
r.get("replacement") == "RIGHT" for r in ctx.pronunciation_rules
|
||||||
|
)
|
||||||
|
found_wrong = any(
|
||||||
|
r.get("replacement") == "WRONG" for r in ctx.pronunciation_rules
|
||||||
|
)
|
||||||
|
assert found_right
|
||||||
|
assert not found_wrong
|
||||||
|
|
||||||
|
def test_heteronym_overrides_compiled(self):
|
||||||
|
overrides = [
|
||||||
|
{
|
||||||
|
"token": "read",
|
||||||
|
"pronunciation": "red",
|
||||||
|
"context": "past tense",
|
||||||
|
}
|
||||||
|
]
|
||||||
|
ctx = build_tts_context(
|
||||||
|
heteronym_overrides=overrides,
|
||||||
|
)
|
||||||
|
assert ctx.heteronym_rules is not None
|
||||||
|
|
||||||
|
def test_usage_counter_passed_through(self):
|
||||||
|
counter = {}
|
||||||
|
ctx = build_tts_context(usage_counter=counter)
|
||||||
|
assert ctx.usage_counter is counter
|
||||||
|
|
||||||
|
def test_usage_counter_default_empty(self):
|
||||||
|
ctx = build_tts_context()
|
||||||
|
assert ctx.usage_counter == {}
|
||||||
|
|
||||||
|
def test_normalization_overrides_stored(self):
|
||||||
|
overrides = {"normalization_numbers": False}
|
||||||
|
ctx = build_tts_context(normalization_overrides=overrides)
|
||||||
|
assert ctx.normalization_overrides is overrides
|
||||||
|
|
||||||
|
def test_speakers_used_for_pronunciation(self):
|
||||||
|
speakers = {
|
||||||
|
"narrator": {
|
||||||
|
"token": "route",
|
||||||
|
"pronunciation": "root",
|
||||||
|
"resolved_voice": "M1",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ctx = build_tts_context(speakers=speakers)
|
||||||
|
assert ctx.pronunciation_rules is not None
|
||||||
|
assert len(ctx.pronunciation_rules) >= 1
|
||||||
|
|
||||||
|
def test_log_callback_called_on_num2words_missing(self):
|
||||||
|
logs = []
|
||||||
|
with patch("abogen.domain.normalization.get_runtime_settings", return_value={
|
||||||
|
"normalization_apostrophe_mode": "spacy",
|
||||||
|
"normalization_enabled": True,
|
||||||
|
"normalization_numbers": True,
|
||||||
|
}):
|
||||||
|
with patch("abogen.normalization_settings.build_apostrophe_config") as mock_cfg:
|
||||||
|
mock_cfg.return_value = MagicMock(convert_numbers=True)
|
||||||
|
with patch("builtins.__import__", side_effect=ImportError):
|
||||||
|
try:
|
||||||
|
build_tts_context(log_callback=lambda lvl, msg: logs.append((lvl, msg)))
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
# If num2words is missing and convert_numbers is True, a warning should be logged
|
||||||
|
# (depends on mock behavior, so just check no crash)
|
||||||
|
|
||||||
|
def test_llm_mode_raises_if_not_configured(self):
|
||||||
|
with patch("abogen.domain.normalization.get_runtime_settings", return_value={
|
||||||
|
"normalization_apostrophe_mode": "llm",
|
||||||
|
}):
|
||||||
|
with pytest.raises(RuntimeError, match="LLM"):
|
||||||
|
build_tts_context()
|
||||||
|
|
||||||
|
def test_dict_source_accepted(self):
|
||||||
|
"""merge_pronunciation_overrides should accept a dict."""
|
||||||
|
source = {
|
||||||
|
"pronunciation_overrides": [
|
||||||
|
{"token": "test", "pronunciation": "test-est", "normalized": "test"}
|
||||||
|
],
|
||||||
|
"manual_overrides": [],
|
||||||
|
"speakers": {},
|
||||||
|
"language": "a",
|
||||||
|
}
|
||||||
|
from abogen.domain.pronunciation import merge_pronunciation_overrides
|
||||||
|
result = merge_pronunciation_overrides(source)
|
||||||
|
assert isinstance(result, list)
|
||||||
|
assert len(result) >= 1
|
||||||
|
|||||||
Reference in New Issue
Block a user