From 98ab2d925e70e62036f3660f8a8323765f7f66bb Mon Sep 17 00:00:00 2001 From: Artem Akymenko Date: Mon, 27 Jul 2026 09:37:54 +0000 Subject: [PATCH] refactor: simplify _build_request, restore epub3_export as config object Simplify _build_request() in WebUI - Remove profiles loading (load_profiles, normalize_profile_entry) - Remove PronunciationConfig/Epub3ExportConfig building - Pass raw fields directly to ConversionRequest - Clean up unused imports Bugfix: epub3_export as Optional[Epub3ExportConfig] (None = disabled) - Reverted generate_epub3: bool + epub3_book_id: str back to object pattern - Consistent with word_substitution, subtitle_input, chapter_chunk - Updated conversion_service.py _finalize() to use request.epub3_export --- abogen/application/conversion_request.py | 6 +- abogen/application/conversion_service.py | 6 +- abogen/webui/conversion_runner.py | 1169 +++++----------------- 3 files changed, 233 insertions(+), 948 deletions(-) diff --git a/abogen/application/conversion_request.py b/abogen/application/conversion_request.py index 10f03d7..a9eb1c2 100644 --- a/abogen/application/conversion_request.py +++ b/abogen/application/conversion_request.py @@ -16,6 +16,7 @@ from typing import Any, Dict, List, Optional from abogen.application.conversion_config import ( ChapterChunkConfig, + Epub3ExportConfig, SubtitleInputConfig, WordSubstitutionConfig, ) @@ -106,11 +107,8 @@ class ConversionRequest: cover_image_path: Optional[Path] = None cover_image_mime: Optional[str] = None - # --- Feature toggles --- - generate_epub3: bool = False - epub3_book_id: str = "" - # --- Feature configs (None = disabled) --- + epub3_export: Optional[Epub3ExportConfig] = None word_substitution: Optional[WordSubstitutionConfig] = None subtitle_input: Optional[SubtitleInputConfig] = None chapter_chunk: Optional[ChapterChunkConfig] = None diff --git a/abogen/application/conversion_service.py b/abogen/application/conversion_service.py index 8d5f420..674c13c 100644 --- a/abogen/application/conversion_service.py +++ b/abogen/application/conversion_service.py @@ -136,7 +136,7 @@ def _finalize( raise RuntimeError(f"Failed to embed m4b metadata: {exc}") from exc # EPUB3 generation - if request.generate_epub3 and plan.extraction: + if request.epub3_export and plan.extraction: audio_asset = result.audio_path if not audio_asset and result.chapter_paths: audio_asset = result.chapter_paths[0] @@ -153,7 +153,7 @@ def _finalize( events.log("Generating EPUB 3 package...") epub_path = build_epub3_package( output_path=epub_output_path, - book_id=request.epub3_book_id, + book_id=request.epub3_export.book_id, extraction=plan.extraction, metadata_tags=result.metadata or {}, chapter_markers=result.chapter_markers or [], @@ -183,7 +183,7 @@ def _finalize( chunk_level=request.chapter_chunk.chunk_level if request.chapter_chunk else None, speaker_mode=request.chapter_chunk.speaker_mode if request.chapter_chunk else None, speakers=request.chapter_chunk.speakers if request.chapter_chunk else None, - generate_epub3=bool(request.generate_epub3), + generate_epub3=bool(request.epub3_export), ) metadata_dir = plan.output_layout.metadata_dir diff --git a/abogen/webui/conversion_runner.py b/abogen/webui/conversion_runner.py index beb21a4..e733ae3 100644 --- a/abogen/webui/conversion_runner.py +++ b/abogen/webui/conversion_runner.py @@ -1,981 +1,268 @@ +"""WebUI conversion runner — thin adapter between Job and shared layer. + +This module is the WebUI's adapter for the conversion system. It: +1. Builds a ConversionRequest from a Job +2. Creates adapter objects (Events, VoiceResolver) +3. Calls run_conversion() from the shared application layer +4. Maps the ConversionResult back to Job state + +All conversion logic (chapter loop, voice resolution, TTS, metadata, +subtitle writing, m4b/epub3 finalization) lives in the application layer. + +Language: Job.language is Language enum. Frontend must send ISO codes. +Engine converts Language → its own format internally. +""" + from __future__ import annotations -import json -import time -import traceback import gc -from collections import defaultdict -from contextlib import ExitStack from pathlib import Path -from typing import Any, Callable, Dict, List, Mapping, Optional +from typing import Any, Dict - -from abogen.infrastructure.exporters import ExportService -from abogen.epub3.exporter import build_epub3_package -from abogen.entity_analysis import normalize_token as normalize_entity_token -from abogen.text_extractor import extract_from_path -from abogen.utils import ( - calculate_text_length, - get_user_cache_path, - get_user_output_path, +from abogen.application.conversion_config import ( + ChapterChunkConfig, + Epub3ExportConfig, ) -from abogen.voice_profiles import load_profiles, normalize_profile_entry -from abogen.infrastructure.subtitle_writer import make_subtitle_writer -from abogen.domain.chapter_titles import ( # noqa: F401 - headings_equivalent as _headings_equivalent, - format_spoken_chapter_title as _format_spoken_chapter_title, - strip_duplicate_heading_line as _strip_duplicate_heading_line, - normalize_caps_word as _normalize_caps_word, - normalize_chapter_opening_caps as _normalize_chapter_opening_caps, - apply_chapter_text_transforms as _apply_chapter_text_transforms, - _HEADING_NUMBER_PREFIX_RE, +from abogen.application.conversion_ports import ConversionCancelled, ResolvedVoice +from abogen.application.conversion_request import ConversionRequest +from abogen.application.conversion_service import run_conversion +from abogen.domain.enums import ( + Language, + OutputFormat, + SaveMode, + SubtitleFormat, + SubtitleMode, ) -from abogen.domain.metadata_helpers import ( # noqa: F401 - normalize_metadata_map as _normalize_metadata_map, - format_author_sentence as _format_author_sentence, - ensure_sentence as _ensure_sentence, - normalize_series_number as _normalize_series_number, - extract_series_metadata as _extract_series_metadata, - format_series_sentence as _format_series_sentence, - build_metadata_payload as _build_metadata_payload, -) -from abogen.domain.intro_outro import resolve_intro, resolve_outro -from abogen.domain.title_builder import ( # noqa: F401 - build_title_intro_text as _build_title_intro_text, - build_outro_text as _build_outro_text, -) -from abogen.domain.file_type import ( - infer_file_type as _infer_file_type, - auto_select_relevant_chapters as _auto_select_relevant_chapters, - chapter_label as _chapter_label, - update_metadata_for_chapter_count as _update_metadata_for_chapter_count, - _SIGNIFICANT_LENGTH_THRESHOLDS, -) -from abogen.domain.pronunciation import ( # noqa: F401 - apply_pronunciation_rules as _apply_pronunciation_rules, - merge_pronunciation_overrides as _merge_pronunciation_overrides, - compile_pronunciation_rules as _compile_pronunciation_rules, - merge_pronunciation_overrides, -) -from abogen.domain.normalization import TTSContext, build_tts_context # noqa: F401 -from abogen.domain.voice_resolution import ( # noqa: F401 - spec_to_voice_ids as _spec_to_voice_ids, - job_voice_fallback as _job_voice_fallback, - collect_required_voice_ids as _collect_required_voice_ids, - initialize_voice_cache as _initialize_voice_cache, - chapter_voice_spec as _chapter_voice_spec, - chunk_voice_spec as _chunk_voice_spec, -) -from abogen.domain.chapter_overrides import apply_chapter_overrides as _apply_chapter_overrides -from abogen.domain.metadata_merge import merge_metadata as _merge_metadata -from abogen.domain.chunk_utils import ( - safe_int as _safe_int, - group_chunks_by_chapter as _group_chunks_by_chapter, - record_override_usage as _record_override_usage, - chunk_text_for_tts as _chunk_text_for_tts, -) -from abogen.domain.voice_utils import ( # noqa: F401 - supertonic_voice_from_spec as _supertonic_voice_from_spec, - split_speaker_reference as _split_speaker_reference, - formula_from_kokoro_entry as _formula_from_kokoro_entry, - infer_provider_from_spec as _infer_provider_from_spec, - coerce_truthy as _coerce_truthy, -) -from abogen.domain.output_paths import ( # noqa: F401 - slugify as _slugify, - sanitize_output_stem as _sanitize_output_stem, - output_timestamp_token as _output_timestamp_token, - build_output_path as _build_output_path, - apply_newline_policy as _apply_newline_policy, - resolve_output_directory as _resolve_output_directory, - resolve_project_layout as _resolve_project_layout, -) - -from abogen.domain.audio_buffer import ( # noqa: F401 - create_silence as _create_silence, - normalize_audio as _normalize_audio, -) -from abogen.domain.audio_sink import AudioSink, open_audio_sink from abogen.domain.pipeline_factory import PipelinePool -from abogen.domain.conversion_engine import synthesize_text, SynthParams, process_and_write_subtitles, SegmentStats from abogen.domain.voice_loader import VoiceCache, resolve_voice from abogen.domain.voice_utils import resolve_voice_target as _resolve_voice_target -from abogen.domain.device import select_device as _select_device # noqa: F401 -from abogen.domain.progress import ProgressTracker, calc_etr_str # noqa: F401 -from abogen.domain.audio_helpers import build_ffmpeg_command as _build_ffmpeg_command, to_float32 as _to_float32 # noqa: F401 -from abogen.utils import create_process # noqa: F401 -from abogen.kokoro_text_normalization import normalize_for_pipeline # noqa: F401 - from .service import Job, JobStatus -_export_svc = ExportService() - -SPLIT_PATTERN = r"\n+" # Kept for backward compatibility; prefer get_split_pattern() -SAMPLE_RATE = 24000 +# --------------------------------------------------------------------------- +# Build ConversionRequest from Job +# --------------------------------------------------------------------------- -class _JobCancelled(Exception): - """Raised internally to abort a conversion when the client cancels.""" +def _build_request(job: Job) -> ConversionRequest: + """Build a ConversionRequest from a WebUI Job.""" + # Determine source path + source_path = Path(job.stored_path) if job.stored_path else None + + return ConversionRequest( + # Source + source_path=source_path, + original_filename=job.original_filename, + # TTS Settings + language=job.language, + tts_provider=job.tts_provider, + voice=job.voice, + speed=job.speed, + use_gpu=job.use_gpu, + supertonic_total_steps=job.supertonic_total_steps, + # Output Format + output_format=_resolve_output_format(job.output_format), + subtitle_mode=_resolve_subtitle_mode(job.subtitle_mode), + subtitle_format=_resolve_subtitle_format(job.subtitle_format), + max_subtitle_words=job.max_subtitle_words, + # Save Options + save_mode=_resolve_save_mode(job.save_mode), + output_folder=job.output_folder, + save_chapters_separately=job.save_chapters_separately, + merge_chapters_at_end=job.merge_chapters_at_end, + separate_chapters_format=_resolve_output_format(job.separate_chapters_format), + save_as_project=job.save_as_project, + # Timing + silence_between_chapters=job.silence_between_chapters, + chapter_intro_delay=job.chapter_intro_delay, + # Content Processing + replace_single_newlines=job.replace_single_newlines, + read_title_intro=job.read_title_intro, + read_closing_outro=job.read_closing_outro, + auto_prefix_chapter_titles=job.auto_prefix_chapter_titles, + normalize_chapter_opening_caps=job.normalize_chapter_opening_caps, + # Metadata + metadata_tags=job.metadata_tags or {}, + # Artifacts + cover_image_path=job.cover_image_path, + cover_image_mime=job.cover_image_mime, + # Pronunciation overrides (raw data) + pronunciation_overrides=job.pronunciation_overrides or [], + manual_overrides=job.manual_overrides or [], + heteronym_overrides=job.heteronym_overrides or [], + normalization_overrides=job.normalization_overrides or None, + # Feature configs + epub3_export=Epub3ExportConfig(book_id=job.id) if job.generate_epub3 else None, + chapter_chunk=ChapterChunkConfig( + chapter_overrides=job.chapters or [], + chunks=job.chunks or [], + chunk_level=job.chunk_level, + speaker_mode=job.speaker_mode, + speakers=job.speakers or {}, + ), + ) + + +def _resolve_output_format(fmt: str) -> OutputFormat: + try: + return OutputFormat.from_str(fmt) + except ValueError: + return OutputFormat.WAV + + +def _resolve_subtitle_mode(mode: str) -> SubtitleMode: + try: + return SubtitleMode.from_str(mode) + except ValueError: + return SubtitleMode.DISABLED + + +def _resolve_subtitle_format(fmt: str) -> SubtitleFormat: + try: + return SubtitleFormat.from_str(fmt) + except ValueError: + return SubtitleFormat.SRT + + +def _resolve_save_mode(mode: str) -> SaveMode: + normalized = mode.strip().lower() + for m in SaveMode: + if m.value == normalized: + return m + return SaveMode.SAVE_NEXT_TO_INPUT + + +# --------------------------------------------------------------------------- +# Apply ConversionResult back to Job +# --------------------------------------------------------------------------- + + +def _apply_result(job: Job, result: Any) -> None: + """Map ConversionResult fields back to Job result and state.""" + job.result.audio_path = result.audio_path + job.result.subtitle_paths = list(result.subtitle_paths) + job.result.artifacts = dict(result.artifacts) + job.result.epub_path = result.epub_path + job.progress = 1.0 + + +# --------------------------------------------------------------------------- +# Adapters: Events, VoiceResolver +# --------------------------------------------------------------------------- + + +class WebUIEventsAdapter: + """Adapts Job to ConversionEvents protocol.""" + + def __init__(self, job: Job): + self._job = job + + def log(self, message: str, level: str = "info") -> None: + self._job.add_log(message, level=level) + + def progress(self, pct: int, etr: str) -> None: + self._job.progress = min(pct / 100.0, 0.999) + self._job.etr_str = etr + + def check_cancelled(self) -> None: + if self._job.cancel_requested: + raise ConversionCancelled("Job cancelled") + + +class WebUIVoiceResolver: + """Adapts WebUI voice resolution to VoiceResolver protocol.""" + + def __init__( + self, + job: Job, + normalized_profiles: Dict[str, Dict[str, Any]], + voice_cache: VoiceCache, + pool: PipelinePool, + ): + self._job = job + self._profiles = normalized_profiles + self._cache = voice_cache + self._pool = pool + + def resolve(self, voice_spec: str) -> ResolvedVoice: + provider, resolved, speed, steps = _resolve_voice_target( + voice_spec, + self._profiles, + job_voice=self._job.voice, + job_tts_provider=self._job.tts_provider, + job_supertonic_total_steps=self._job.supertonic_total_steps, + job_speed=self._job.speed, + ) + + cache_key = f"{provider}:{resolved}" if resolved else provider + cached = self._cache.get(cache_key) + if cached is not None: + return ResolvedVoice( + provider=provider, + resolved_spec=resolved, + voice=cached, + speed=speed, + supertonic_steps=steps or 0, + ) + + if provider == "kokoro": + kokoro_backend = self._pool.get("kokoro", self._job.language, self._job.use_gpu) + loaded = resolve_voice(resolved, kokoro_backend, self._job.use_gpu, cache=self._cache) + else: + loaded = resolved + + self._cache.set(cache_key, loaded) + return ResolvedVoice( + provider=provider, + resolved_spec=resolved, + voice=loaded, + speed=speed, + supertonic_steps=steps or 0, + ) + + +# --------------------------------------------------------------------------- +# Main entry point +# --------------------------------------------------------------------------- def run_conversion_job(job: Job) -> None: + """Run a conversion job using the shared application layer.""" job.add_log("Preparing conversion pipeline") - canceller = _make_canceller(job) - usage_counter: Dict[str, int] = defaultdict(int) + # Build request from job data + request = _build_request(job) - def _tts_log(level: str, msg: str) -> None: - job.add_log(msg, level=level) - - tts_context = build_tts_context( - language=str(job.language or "a"), - subtitle_mode=str(job.subtitle_mode or "Disabled"), - pronunciation_overrides=getattr(job, "pronunciation_overrides", None), - manual_overrides=getattr(job, "manual_overrides", None), - heteronym_overrides=getattr(job, "heteronym_overrides", None), - speakers=getattr(job, "speakers", None), - normalization_overrides=getattr(job, "normalization_overrides", None), - usage_counter=usage_counter, - log_callback=_tts_log, + # Create adapters + events = WebUIEventsAdapter(job) + pool = PipelinePool() + voice_cache = VoiceCache() + voice_resolver = WebUIVoiceResolver( + job, request.speakers, voice_cache, pool, ) - sink_stack = ExitStack() - subtitle_writer = None - chapter_paths: list[Path] = [] - chapter_markers: List[Dict[str, Any]] = [] - chunk_markers: List[Dict[str, Any]] = [] - metadata_payload: Dict[str, Any] = {} - audio_output_path: Optional[Path] = None - extraction: Optional[Any] = None - pipeline: Any = None - pipeline_pool = PipelinePool() - normalized_profiles: Dict[str, Dict[str, Any]] = {} - chunk_groups: Dict[int, List[Dict[str, Any]]] = {} - active_chapter_configs: List[Dict[str, Any]] = [] - override_token_map: Dict[str, str] = {} try: - # Load saved speakers once so we can resolve speaker: references during conversion. - try: - profiles = load_profiles() - except Exception: - profiles = {} - for name, entry in (profiles or {}).items(): - normalized = normalize_profile_entry(entry) - if normalized: - normalized_profiles[str(name)] = normalized - - def resolve_voice_choice(raw_spec: str) -> tuple[str, str, Any, Optional[float], Optional[int]]: - """Resolve a raw voice spec into (provider, resolved_spec, choice, speed, steps).""" - provider, resolved, speed, steps = _resolve_voice_target( - raw_spec, - normalized_profiles, - job_voice=getattr(job, "voice", "M1"), - job_tts_provider=getattr(job, "tts_provider", "kokoro"), - job_supertonic_total_steps=getattr(job, "supertonic_total_steps", 5), - job_speed=getattr(job, "speed", 1.0), - ) - cache_key = f"{provider}:{resolved}" if resolved else provider - cached = voice_cache.get(cache_key) - if cached is not None: - return provider, resolved, cached, speed, steps - - if provider == "kokoro": - kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job) - choice = resolve_voice(resolved, kokoro_backend, job.use_gpu, cache=voice_cache) - else: - choice = resolved - - voice_cache.set(cache_key, choice) - return provider, resolved, choice, speed, steps - - extraction = extract_from_path(job.stored_path) - file_type = _infer_file_type(job.stored_path) - - # Build override_token_map from pronunciation overrides - pronunciation_overrides = merge_pronunciation_overrides(job) - for override_entry in pronunciation_overrides or []: - if not isinstance(override_entry, Mapping): - continue - raw_token = str(override_entry.get("token") or "").strip() - normalized_value = str(override_entry.get("normalized") or "").strip() - if not normalized_value and raw_token: - normalized_value = normalize_entity_token(raw_token) or raw_token - if normalized_value: - override_token_map.setdefault(normalized_value, raw_token or normalized_value) - - if not job.chapters: - filtered, skipped_info = _auto_select_relevant_chapters(extraction.chapters, file_type) - original_count = len(extraction.chapters) - if filtered and len(filtered) < original_count: - extraction.chapters = filtered - _update_metadata_for_chapter_count(extraction.metadata, len(filtered), file_type) - threshold = _SIGNIFICANT_LENGTH_THRESHOLDS.get(file_type.lower()) - label = _chapter_label(file_type) - qualifier = f" (< {threshold} characters)" if threshold else "" - job.add_log( - f"Auto-selected {len(filtered)} of {original_count} {label} based on content{qualifier}.", - level="info", - ) - if skipped_info: - preview_count = 5 - preview = ", ".join( - f"{title or 'Untitled'} ({length})" for title, length in skipped_info[:preview_count] - ) - if len(skipped_info) > preview_count: - preview += ", …" - job.add_log( - f"Skipped {len(skipped_info)} short {label}: {preview}", - level="debug", - ) - elif not filtered: - job.add_log( - "Auto-selection did not identify usable chapters; retaining original set.", - level="warning", - ) - - metadata_overrides: Dict[str, Any] = dict(job.metadata_tags or {}) - if job.chapters: - selected_chapters, chapter_metadata, diagnostics = _apply_chapter_overrides( - extraction.chapters, - job.chapters, - ) - for message in diagnostics: - job.add_log(message, level="warning") - if selected_chapters: - extraction.chapters = selected_chapters - metadata_overrides.update(chapter_metadata) - job.add_log( - f"Chapter overrides applied: {len(selected_chapters)} selected.", - level="info", - ) - active_chapter_configs = [ - entry for entry in job.chapters if _coerce_truthy(entry.get("enabled", True)) - ][: len(selected_chapters)] - if job.chunks: - chunk_groups = _group_chunks_by_chapter(job.chunks) - else: - raise ValueError("No chapters were enabled in the requested job.") - elif job.chunks: - chunk_groups = _group_chunks_by_chapter(job.chunks) - - job.metadata_tags = _merge_metadata(extraction.metadata, metadata_overrides) - - total_characters = extraction.total_characters or calculate_text_length(extraction.combined_text) - job.total_characters = total_characters - job.add_log(f"Total characters: {job.total_characters:,}") - - _apply_newline_policy(extraction.chapters, job.replace_single_newlines) - - base_output_dir = _prepare_output_dir(job) - project_root, audio_dir, subtitle_dir, metadata_dir = _resolve_project_layout( - original_filename=job.original_filename, - save_as_project=job.save_as_project, - base_dir=base_output_dir, - ) - - if job.output_format.lower() == "m4b" and not job.merge_chapters_at_end: - job.add_log( - "Forcing merged output for m4b format; ignoring 'merge chapters at end' setting.", - level="warning", - ) - job.merge_chapters_at_end = True - - merged_required = job.merge_chapters_at_end or not job.save_chapters_separately - audio_path: Optional[Path] = None - audio_sink: Optional[AudioSink] = None - if merged_required: - audio_path = _build_output_path(audio_dir, job.original_filename, job.output_format) - meta_for_sink = job.metadata_tags if job.metadata_tags else None - audio_sink = sink_stack.enter_context( - open_audio_sink( - audio_path, - job.output_format, - metadata=meta_for_sink, - cancel_check=lambda: job.cancel_requested, - ) - ) - subtitle_writer = make_subtitle_writer( - audio_path, - job.subtitle_format, - job.subtitle_mode or "Line", - max_words=job.max_subtitle_words, - ) - if subtitle_writer is None and job.subtitle_mode != "Disabled": - fmt = (job.subtitle_format or "srt").lower() - if job.subtitle_mode == "Sentence + Highlighting" and fmt == "srt": - job.add_log("Highlighting requires ASS subtitles. Switching format.", level="warning") - else: - job.add_log(f"Unsupported subtitle format '{job.subtitle_format}'. Skipping.", level="warning") - job.result.audio_path = audio_path - if subtitle_writer: - job.result.subtitle_paths.append(subtitle_writer.path) - - chapter_dir: Optional[Path] = None - if job.save_chapters_separately: - chapter_dir = audio_dir / "chapters" - chapter_dir.mkdir(parents=True, exist_ok=True) - - base_voice_spec = _job_voice_fallback(job) - voice_cache = VoiceCache() - base_provider, base_voice_resolved, _, _ = _resolve_voice_target( - base_voice_spec, normalized_profiles, - job_voice=getattr(job, "voice", "M1"), - job_tts_provider=getattr(job, "tts_provider", "kokoro"), - ) - if base_provider == "kokoro" and base_voice_resolved and "*" not in base_voice_resolved: - kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job) - voice_cache.set(f"kokoro:{base_voice_resolved}", resolve_voice(base_voice_resolved, kokoro_backend, job.use_gpu)) - processed_chars = 0 - current_time = 0.0 - etr_start_time = time.time() - total_chapters = len(extraction.chapters) - if chunk_groups: - chunk_groups = { - idx: items for idx, items in chunk_groups.items() if 0 <= idx < total_chapters - } - job.add_log(f"Detected {total_chapters} chapter{'s' if total_chapters != 1 else ''}") - auto_prefix_titles = getattr(job, "auto_prefix_chapter_titles", True) - read_title_intro = getattr(job, "read_title_intro", False) - book_intro_text = "" - intro_provider: Optional[str] = None - intro_voice_choice: Any = None - intro_speed: Optional[float] = None - intro_steps: Optional[int] = None - intro_spec = resolve_intro( - job.metadata_tags, job.original_filename, read_title_intro, - base_voice_spec, getattr(job, "voice", "M1"), list(voice_cache.keys()), - ) - if intro_spec.enabled: - book_intro_text = intro_spec.text - preview = book_intro_text if len(book_intro_text) <= 120 else f"{book_intro_text[:117]}…" - job.add_log(f"Title intro enabled: {preview}", level="debug") - - intro_provider, _, intro_voice_choice, intro_speed, intro_steps = resolve_voice_choice( - intro_spec.voice_spec - ) - elif read_title_intro: - job.add_log("Title intro enabled but no usable metadata was found.", level="debug") - intro_emitted = False - - def emit_text( - text: str, - *, - voice_choice: Any, - chapter_sink: Optional[AudioSink], - preview_prefix: Optional[str] = None, - split_pattern: Optional[str] = None, - tts_provider: Optional[str] = None, - speed_override: Optional[float] = None, - supertonic_steps_override: Optional[int] = None, - ) -> int: - nonlocal processed_chars, current_time - source_text = str(text or "") - - provider = str(tts_provider or getattr(job, "tts_provider", "kokoro") or "kokoro").strip().lower() or "kokoro" - if provider == "supertonic": - supertonic_pipeline = pipeline_pool.get("supertonic", job.language, job.use_gpu, job=job) - voice_name = _supertonic_voice_from_spec(voice_choice, getattr(job, "voice", "M1")) - backend = supertonic_pipeline - resolved_voice = voice_name - effective_speed = float(speed_override if speed_override is not None else job.speed) - else: - kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job) - backend = kokoro_backend - resolved_voice = voice_choice - effective_speed = float(speed_override if speed_override is not None else job.speed) - - try: - stats = SegmentStats( - processed_chars=processed_chars, - current_time=current_time, - etr_start_time=etr_start_time, - total_characters=job.total_characters or 0, - ) - prefix = f"{preview_prefix} · " if preview_prefix else "" - - def _on_progress(pct: int, etr: str) -> None: - nonlocal processed_chars - processed_chars = stats.processed_chars - job.processed_characters = processed_chars - if stats.total_characters: - job.progress = min(processed_chars / stats.total_characters, 0.999) - else: - job.progress = 0.0 if processed_chars == 0 else 0.999 - job.etr_str = etr - - def _preview(text: str) -> None: - job.add_log(f"{prefix}{stats.processed_chars:,}/{job.total_characters or '—'}: {text[:80]}") - - synth_params = SynthParams( - tts_context=tts_context, - stats=stats, - check_cancel=canceller, - on_progress=_on_progress, - audio_sink=audio_sink, - subtitle_mode=job.subtitle_mode if (subtitle_writer and audio_sink) else "Disabled", - max_subtitle_words=job.max_subtitle_words, - lang_code=job.language, - use_spacy_segmentation=job.subtitle_mode not in ("Disabled", "Line"), - ) - - local_segments, accumulated_tokens = synthesize_text( - text=source_text, - params=synth_params, - backend=backend, - voice=resolved_voice, - speed=effective_speed, - chapter_sink=chapter_sink, - preview_callback=_preview, - ) - current_time = stats.current_time - - if subtitle_writer and audio_sink and accumulated_tokens: - process_and_write_subtitles( - accumulated_tokens, - subtitle_writer, - subtitle_mode=job.subtitle_mode, - max_subtitle_words=job.max_subtitle_words, - lang_code=job.language, - use_spacy_segmentation=job.subtitle_mode not in ("Disabled", "Line"), - fallback_end_time=current_time, - ) - - except OverflowError as exc: - job.add_log( - f"Skipped chunk — number too large for TTS conversion: {exc}", - level="warning", - ) - return local_segments - - def append_silence( - duration_seconds: float, - *, - include_in_chapter: bool, - chapter_sink: Optional[AudioSink], - ) -> None: - nonlocal current_time - if duration_seconds <= 0: - return - silence = _create_silence(duration_seconds) - if silence.size == 0: - return - if include_in_chapter and chapter_sink: - chapter_sink.write(silence) - if audio_sink: - audio_sink.write(silence) - current_time += duration_seconds - - for idx, chapter in enumerate(extraction.chapters, start=1): - canceller() - raw_title = str(getattr(chapter, "title", "") or "").strip() - spoken_title = _format_spoken_chapter_title(raw_title, idx, auto_prefix_titles) - heading_text = spoken_title or raw_title - chapter_display_title = heading_text or f"Chapter {idx}" - job.add_log(f"Processing chapter {idx}/{total_chapters}: {chapter_display_title}") - normalize_opening_caps = bool(job.normalize_chapter_opening_caps) - - chapter_start_time = current_time - chapter_override = ( - active_chapter_configs[idx - 1] if idx - 1 < len(active_chapter_configs) else None - ) - chapter_voice_spec = _chapter_voice_spec(job, chapter_override) - if not chapter_voice_spec: - chapter_voice_spec = base_voice_spec - - chapter_provider, chapter_voice_resolved, voice_choice, chapter_speed, chapter_steps = resolve_voice_choice( - chapter_voice_spec - ) - - chapter_audio_path: Optional[Path] = None - segments_emitted = 0 - - with ExitStack() as chapter_sink_stack: - chapter_sink: Optional[AudioSink] = None - - if chapter_dir is not None: - chapter_audio_path = _build_output_path( - chapter_dir, - f"{Path(job.original_filename).stem}_{_slugify(chapter_display_title, idx)}", - job.separate_chapters_format, - ) - chapter_sink = chapter_sink_stack.enter_context( - open_audio_sink( - chapter_audio_path, - job.separate_chapters_format, - cancel_check=lambda: job.cancel_requested, - ) - ) - - speak_heading = bool(heading_text) - first_line = "" - if chapter.text: - first_line = next((line.strip() for line in chapter.text.splitlines() if line.strip()), "") - remove_heading_from_body = False - if speak_heading and first_line: - if _headings_equivalent(first_line, heading_text) or (raw_title and _headings_equivalent(first_line, raw_title)): - remove_heading_from_body = True - - if not intro_emitted and book_intro_text: - intro_use_provider = intro_provider or chapter_provider - intro_use_voice_choice = intro_voice_choice if intro_voice_choice is not None else voice_choice - intro_use_speed = intro_speed if intro_speed is not None else chapter_speed - intro_use_steps = intro_steps if intro_steps is not None else chapter_steps - intro_segments = emit_text( - book_intro_text, - voice_choice=intro_use_voice_choice, - chapter_sink=chapter_sink, - preview_prefix="Book intro", - tts_provider=intro_use_provider, - speed_override=intro_use_speed, - supertonic_steps_override=intro_use_steps, - ) - intro_emitted = True - if intro_segments > 0 and job.chapter_intro_delay > 0: - append_silence( - job.chapter_intro_delay, - include_in_chapter=True, - chapter_sink=chapter_sink, - ) - - if speak_heading: - heading_segments = emit_text( - heading_text, - voice_choice=voice_choice, - chapter_sink=chapter_sink, - preview_prefix=f"Chapter {idx} title", - tts_provider=chapter_provider, - speed_override=chapter_speed, - supertonic_steps_override=chapter_steps, - ) - segments_emitted += heading_segments - if heading_segments > 0 and job.chapter_intro_delay > 0: - append_silence( - job.chapter_intro_delay, - include_in_chapter=True, - chapter_sink=chapter_sink, - ) - - chunks_for_chapter = chunk_groups.get(idx - 1, []) if chunk_groups else [] - body_segments = 0 - pending_heading_strip = remove_heading_from_body - opening_caps_pending = normalize_opening_caps - opening_caps_logged = False - if chunks_for_chapter: - job.add_log( - f"Emitting {len(chunks_for_chapter)} {job.chunk_level} chunks for chapter {idx}.", - level="debug", - ) - for chunk_entry in chunks_for_chapter: - chunk_text = _chunk_text_for_tts(chunk_entry) - if not chunk_text: - continue - - mutated_entry = False - chunk_text, heading_removed, caps_changed = _apply_chapter_text_transforms( - chunk_text, - heading_text=heading_text, - raw_title=raw_title, - strip_heading=pending_heading_strip, - normalize_caps=opening_caps_pending, - ) - if heading_removed: - pending_heading_strip = False - chunk_entry = dict(chunk_entry) - chunk_entry["normalized_text"] = chunk_text - mutated_entry = True - if not chunk_text.strip(): - continue - if caps_changed: - if not mutated_entry: - chunk_entry = dict(chunk_entry) - chunk_entry["normalized_text"] = chunk_text - if not opening_caps_logged: - job.add_log( - f"Normalized uppercase chapter opening for chapter {idx}.", - level="debug", - ) - opening_caps_logged = True - if chunk_text.strip(): - opening_caps_pending = False - - chunk_voice_spec = _chunk_voice_spec( - job, - chunk_entry, - chapter_voice_spec or base_voice_spec, - ) - if not chunk_voice_spec: - chunk_voice_spec = chapter_voice_spec or base_voice_spec - - if chunk_voice_spec == chapter_voice_spec: - chunk_provider = chapter_provider - chunk_voice_resolved = chapter_voice_resolved - chunk_speed_use = chapter_speed - chunk_steps_use = chapter_steps - chunk_voice_choice = voice_choice - else: - chunk_provider, chunk_voice_resolved, chunk_voice_choice, chunk_speed_use, chunk_steps_use = resolve_voice_choice( - chunk_voice_spec - ) - - chunk_start = current_time - emitted = emit_text( - chunk_text, - voice_choice=chunk_voice_choice, - chapter_sink=chapter_sink, - preview_prefix=f"Chunk {chunk_entry.get('id') or chunk_entry.get('chunk_index')}", - tts_provider=chunk_provider, - speed_override=chunk_speed_use, - supertonic_steps_override=chunk_steps_use, - ) - if emitted <= 0: - continue - - body_segments += emitted - segments_emitted += emitted - chunk_markers.append( - { - "id": chunk_entry.get("id"), - "chapter_index": idx - 1, - "chunk_index": _safe_int( - chunk_entry.get("chunk_index"), len(chunk_markers) - ), - "start": chunk_start, - "end": current_time, - "speaker_id": chunk_entry.get("speaker_id", "narrator"), - "voice": chunk_voice_spec, - "level": chunk_entry.get("level", job.chunk_level), - "characters": len(chunk_text), - } - ) - - if body_segments == 0: - chapter_body_start = current_time - chapter_text = str(chapter.text or "") - chapter_text, heading_removed, caps_changed = _apply_chapter_text_transforms( - chapter_text, - heading_text=heading_text, - raw_title=raw_title, - strip_heading=pending_heading_strip, - normalize_caps=opening_caps_pending, - ) - if heading_removed: - pending_heading_strip = False - if caps_changed: - if not opening_caps_logged: - job.add_log( - f"Normalized uppercase chapter opening for chapter {idx}.", - level="debug", - ) - opening_caps_logged = True - if str(chapter_text or "").strip(): - opening_caps_pending = False - emitted = emit_text( - chapter_text, - voice_choice=voice_choice, - chapter_sink=chapter_sink, - tts_provider=chapter_provider, - speed_override=chapter_speed, - supertonic_steps_override=chapter_steps, - ) - if emitted > 0: - segments_emitted += emitted - chunk_markers.append( - { - "id": None, - "chapter_index": idx - 1, - "chunk_index": 0, - "start": chapter_body_start, - "end": current_time, - "speaker_id": "narrator", - "voice": chapter_voice_spec, - "level": job.chunk_level, - "characters": len(chapter_text or ""), - } - ) - elif chunks_for_chapter: - job.add_log( - "No audio generated for supplied chunks; chapter text also empty.", - level="warning", - ) - - chapter_end_time = current_time - - if chapter_audio_path is not None: - job.result.artifacts[f"chapter_{idx:02d}"] = chapter_audio_path - chapter_paths.append(chapter_audio_path) - - if segments_emitted == 0: - job.add_log( - f"No audio segments were generated for chapter {idx}.", - level="warning", - ) - else: - job.add_log(f"Finished chapter {idx} with {segments_emitted} segments.") - - if ( - audio_sink - and job.merge_chapters_at_end - and idx < total_chapters - and job.silence_between_chapters > 0 - ): - append_silence( - job.silence_between_chapters, - include_in_chapter=False, - chapter_sink=None, - ) - chapter_end_time = current_time - - marker = { - "index": idx, - "title": chapter_display_title, - "start": chapter_start_time, - "end": chapter_end_time, - "voice": chapter_voice_spec, - } - if raw_title and raw_title != chapter_display_title: - marker["original_title"] = raw_title - chapter_markers.append(marker) - - if getattr(job, "read_closing_outro", True): - outro_spec = resolve_outro( - job.metadata_tags, job.original_filename, True, - base_voice_spec, getattr(job, "voice", "M1"), list(voice_cache.keys()), - ) - - if outro_spec.enabled: - outro_start_time = current_time - outro_audio_path: Optional[Path] = None - outro_segments = 0 - outro_index = total_chapters + 1 - outro_provider, _, outro_voice_choice, outro_speed, outro_steps = resolve_voice_choice(outro_spec.voice_spec) - - with ExitStack() as outro_sink_stack: - chapter_sink: Optional[AudioSink] = None - if chapter_dir is not None: - outro_audio_path = _build_output_path( - chapter_dir, - f"{Path(job.original_filename).stem}_outro", - job.separate_chapters_format, - ) - chapter_sink = outro_sink_stack.enter_context( - open_audio_sink( - outro_audio_path, - job.separate_chapters_format, - cancel_check=lambda: job.cancel_requested, - ) - ) - - outro_segments = emit_text( - outro_spec.text, - voice_choice=outro_voice_choice, - chapter_sink=chapter_sink, - preview_prefix="Outro", - tts_provider=outro_provider, - speed_override=outro_speed, - supertonic_steps_override=outro_steps, - ) - outro_end_time = current_time - - if outro_segments > 0: - job.add_log(f"Appended outro sequence: {outro_spec.text}") - if outro_audio_path is not None: - job.result.artifacts[f"chapter_{outro_index:02d}"] = outro_audio_path - chapter_paths.append(outro_audio_path) - chapter_markers.append( - { - "index": outro_index, - "title": "Outro", - "start": outro_start_time, - "end": outro_end_time, - "voice": outro_spec.voice_spec, - } - ) - else: - job.add_log("No audio generated for outro sequence.", level="warning") - - if not audio_path and chapter_paths: - job.result.audio_path = chapter_paths[0] - - metadata_payload = _build_metadata_payload( - metadata=job.metadata_tags, - chapter_markers=chapter_markers, - chunk_markers=chunk_markers, - chunk_level=job.chunk_level, - speaker_mode=job.speaker_mode, - speakers=getattr(job, "speakers", None), - generate_epub3=job.generate_epub3, - ) - - if tts_context.usage_counter: - _record_override_usage(job, tts_context.usage_counter, override_token_map) - - if metadata_dir: - metadata_dir.mkdir(parents=True, exist_ok=True) - metadata_file = metadata_dir / "metadata.json" - metadata_file.write_text(json.dumps(metadata_payload, indent=2), encoding="utf-8") - job.result.artifacts["metadata"] = metadata_file - - if job.generate_epub3: - audio_asset = job.result.audio_path - if not audio_asset and chapter_paths: - audio_asset = chapter_paths[0] - - if audio_asset: - try: - epub_root = project_root - epub_output_path = _build_output_path(epub_root, job.original_filename, "epub") - job.add_log("Generating EPUB 3 package with synchronized narration…") - epub_path = build_epub3_package( - output_path=epub_output_path, - book_id=job.id, - extraction=extraction, - metadata_tags=metadata_payload.get("metadata") or {}, - chapter_markers=chapter_markers, - chunk_markers=chunk_markers, - chunks=job.chunks, - audio_path=audio_asset, - speaker_mode=job.speaker_mode, - cover_image_path=job.cover_image_path, - cover_image_mime=job.cover_image_mime, - ) - job.result.epub_path = epub_path - job.result.artifacts["epub3"] = epub_path - job.add_log(f"EPUB 3 package created at {epub_path}") - except Exception as exc: - job.add_log(f"Failed to generate EPUB 3 package: {exc}", level="error") - else: - job.add_log("Skipped EPUB 3 generation: audio output unavailable.", level="warning") - - if job.save_as_project: - job.result.artifacts["project_root"] = project_root + result = run_conversion(request, events, pool, voice_resolver) + _apply_result(job, result) if job.status != JobStatus.CANCELLED: job.progress = 1.0 - audio_output_path = job.result.audio_path - - except _JobCancelled: + except ConversionCancelled: job.status = JobStatus.CANCELLED job.add_log("Job cancelled", level="warning") - except Exception as exc: # pragma: no cover - defensive guard + except Exception as exc: job.error = str(exc) job.status = JobStatus.FAILED - exc_type = exc.__class__.__name__ - job.add_log(f"Job failed ({exc_type}): {exc}", level="error") - - chapter_count: Any - if extraction is not None and hasattr(extraction, "chapters"): - try: - chapter_count = len(getattr(extraction, "chapters", []) or []) - except Exception: # pragma: no cover - defensive fallback - chapter_count = "unavailable" - else: - chapter_count = "unavailable" - - try: - chunk_group_count = len(chunk_groups) - chunk_total = sum(len(items) for items in chunk_groups.values()) - except Exception: # pragma: no cover - defensive fallback - chunk_group_count = "unavailable" - chunk_total = "unavailable" - - job.add_log( - "Context => chunk_level=%s, chapters=%s, chunk_groups=%s, chunks=%s" - % (job.chunk_level, chapter_count, chunk_group_count, chunk_total), - level="debug", - ) - - first_nonempty_group = next((items for items in chunk_groups.values() if items), None) - if first_nonempty_group: - first_chunk = dict(first_nonempty_group[0]) - sample_text = str(first_chunk.get("text") or "")[:160].replace("\n", " ") - job.add_log( - "First chunk sample => id=%s, speaker=%s, chars=%s, preview=%s" - % ( - first_chunk.get("id") or first_chunk.get("chunk_index"), - first_chunk.get("speaker_id", "narrator"), - len(str(first_chunk.get("text") or "")), - sample_text, - ), - level="debug", - ) - - tb_lines = traceback.format_exception(exc.__class__, exc, exc.__traceback__) - for line in tb_lines[:20]: - trimmed = line.rstrip() - if trimmed: - for snippet in trimmed.splitlines(): - job.add_log(f"TRACE: {snippet}", level="debug") + job.add_log(f"Job failed: {exc}", level="error") finally: - sink_stack.close() - if subtitle_writer: - subtitle_writer.close() - - # Explicitly release the pipeline and force garbage collection to prevent - # memory accumulation in the worker process, which can lead to host lockups. - pipeline_pool.dispose_all() - pipeline = None + pool.dispose_all() + voice_cache.clear() gc.collect() try: - import torch # type: ignore[import-not-found] + import torch if torch.cuda.is_available(): torch.cuda.empty_cache() except ImportError: pass - - if ( - audio_output_path - and job.output_format.lower() == "m4b" - and not job.cancel_requested - and job.status not in {JobStatus.FAILED, JobStatus.CANCELLED} - ): - try: - cover_path = None - if job.cover_image_path: - candidate = Path(job.cover_image_path) - if candidate.exists(): - cover_path = candidate - - _export_svc.embed_m4b_metadata( - audio_path=audio_output_path, - metadata=metadata_payload.get("metadata") or {}, - chapters=metadata_payload.get("chapters") or [], - cover_path=cover_path, - cover_mime=job.cover_image_mime, - log_callback=lambda msg, level="info": job.add_log(msg, level=level), - ) - except Exception as exc: # pragma: no cover - ensure failure propagates - job.add_log( - f"Failed to embed metadata into m4b output: {exc}", - level="error", - ) - raise RuntimeError( - f"Failed to embed metadata into m4b output: {exc}" - ) from exc - - -def _prepare_output_dir(job: Job) -> Path: - from platformdirs import user_desktop_dir # type: ignore[import-not-found] - - default_output = Path(str(get_user_cache_path("outputs"))) - directory = _resolve_output_directory( - save_mode=job.save_mode, - stored_path=job.stored_path, - output_folder=getattr(job, "output_folder", None), - desktop_dir=Path(user_desktop_dir()), - user_output_path=Path(get_user_output_path()), - user_cache_outputs=default_output, - ) - directory.mkdir(parents=True, exist_ok=True) - return directory - - - -def _make_canceller(job: Job) -> Callable[[], None]: - def _cancel() -> None: - if job.cancel_requested: - raise _JobCancelled - - return _cancel