mirror of
https://github.com/denizsafak/abogen.git
synced 2026-09-20 11:40:57 +02:00
- conversion_planner.py: caps normalization in _build_chapters() - conversion_executor.py: heading dedup + state machine via headings_equivalent() - Fixed seg_start_time → chapter_body_start bug in executor - Removed getattr fallback defaults in both adapters - Added 4 tests for caps normalization and heading dedup - Updated ARCHITECTURE_REFACTOR_PLAN.md with deferred WebUI cleanup
982 lines
44 KiB
Python
982 lines
44 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
import time
|
|
import traceback
|
|
import gc
|
|
from collections import defaultdict
|
|
from contextlib import ExitStack
|
|
from pathlib import Path
|
|
from typing import Any, Callable, Dict, List, Mapping, Optional
|
|
|
|
|
|
from abogen.infrastructure.exporters import ExportService
|
|
from abogen.epub3.exporter import build_epub3_package
|
|
from abogen.entity_analysis import normalize_token as normalize_entity_token
|
|
from abogen.text_extractor import extract_from_path
|
|
from abogen.utils import (
|
|
calculate_text_length,
|
|
get_user_cache_path,
|
|
get_user_output_path,
|
|
)
|
|
from abogen.voice_profiles import load_profiles, normalize_profile_entry
|
|
from abogen.infrastructure.subtitle_writer import make_subtitle_writer
|
|
from abogen.domain.chapter_titles import ( # noqa: F401
|
|
headings_equivalent as _headings_equivalent,
|
|
format_spoken_chapter_title as _format_spoken_chapter_title,
|
|
strip_duplicate_heading_line as _strip_duplicate_heading_line,
|
|
normalize_caps_word as _normalize_caps_word,
|
|
normalize_chapter_opening_caps as _normalize_chapter_opening_caps,
|
|
apply_chapter_text_transforms as _apply_chapter_text_transforms,
|
|
_HEADING_NUMBER_PREFIX_RE,
|
|
)
|
|
from abogen.domain.metadata_helpers import ( # noqa: F401
|
|
normalize_metadata_map as _normalize_metadata_map,
|
|
format_author_sentence as _format_author_sentence,
|
|
ensure_sentence as _ensure_sentence,
|
|
normalize_series_number as _normalize_series_number,
|
|
extract_series_metadata as _extract_series_metadata,
|
|
format_series_sentence as _format_series_sentence,
|
|
build_metadata_payload as _build_metadata_payload,
|
|
)
|
|
from abogen.domain.intro_outro import resolve_intro, resolve_outro
|
|
from abogen.domain.title_builder import ( # noqa: F401
|
|
build_title_intro_text as _build_title_intro_text,
|
|
build_outro_text as _build_outro_text,
|
|
)
|
|
from abogen.domain.file_type import (
|
|
infer_file_type as _infer_file_type,
|
|
auto_select_relevant_chapters as _auto_select_relevant_chapters,
|
|
chapter_label as _chapter_label,
|
|
update_metadata_for_chapter_count as _update_metadata_for_chapter_count,
|
|
_SIGNIFICANT_LENGTH_THRESHOLDS,
|
|
)
|
|
from abogen.domain.pronunciation import ( # noqa: F401
|
|
apply_pronunciation_rules as _apply_pronunciation_rules,
|
|
merge_pronunciation_overrides as _merge_pronunciation_overrides,
|
|
compile_pronunciation_rules as _compile_pronunciation_rules,
|
|
merge_pronunciation_overrides,
|
|
)
|
|
from abogen.domain.normalization import TTSContext, build_tts_context # noqa: F401
|
|
from abogen.domain.voice_resolution import ( # noqa: F401
|
|
spec_to_voice_ids as _spec_to_voice_ids,
|
|
job_voice_fallback as _job_voice_fallback,
|
|
collect_required_voice_ids as _collect_required_voice_ids,
|
|
initialize_voice_cache as _initialize_voice_cache,
|
|
chapter_voice_spec as _chapter_voice_spec,
|
|
chunk_voice_spec as _chunk_voice_spec,
|
|
)
|
|
from abogen.domain.chapter_overrides import apply_chapter_overrides as _apply_chapter_overrides
|
|
from abogen.domain.metadata_merge import merge_metadata as _merge_metadata
|
|
from abogen.domain.chunk_utils import (
|
|
safe_int as _safe_int,
|
|
group_chunks_by_chapter as _group_chunks_by_chapter,
|
|
record_override_usage as _record_override_usage,
|
|
chunk_text_for_tts as _chunk_text_for_tts,
|
|
)
|
|
from abogen.domain.voice_utils import ( # noqa: F401
|
|
supertonic_voice_from_spec as _supertonic_voice_from_spec,
|
|
split_speaker_reference as _split_speaker_reference,
|
|
formula_from_kokoro_entry as _formula_from_kokoro_entry,
|
|
infer_provider_from_spec as _infer_provider_from_spec,
|
|
coerce_truthy as _coerce_truthy,
|
|
)
|
|
from abogen.domain.output_paths import ( # noqa: F401
|
|
slugify as _slugify,
|
|
sanitize_output_stem as _sanitize_output_stem,
|
|
output_timestamp_token as _output_timestamp_token,
|
|
build_output_path as _build_output_path,
|
|
apply_newline_policy as _apply_newline_policy,
|
|
resolve_output_directory as _resolve_output_directory,
|
|
resolve_project_layout as _resolve_project_layout,
|
|
)
|
|
|
|
from abogen.domain.audio_buffer import ( # noqa: F401
|
|
create_silence as _create_silence,
|
|
normalize_audio as _normalize_audio,
|
|
)
|
|
from abogen.domain.audio_sink import AudioSink, open_audio_sink
|
|
from abogen.domain.pipeline_factory import PipelinePool
|
|
from abogen.domain.conversion_engine import synthesize_text, SynthParams, process_and_write_subtitles, SegmentStats
|
|
from abogen.domain.voice_loader import VoiceCache, resolve_voice
|
|
from abogen.domain.voice_utils import resolve_voice_target as _resolve_voice_target
|
|
from abogen.domain.device import select_device as _select_device # noqa: F401
|
|
from abogen.domain.progress import ProgressTracker, calc_etr_str # noqa: F401
|
|
from abogen.domain.audio_helpers import build_ffmpeg_command as _build_ffmpeg_command, to_float32 as _to_float32 # noqa: F401
|
|
from abogen.utils import create_process # noqa: F401
|
|
from abogen.kokoro_text_normalization import normalize_for_pipeline # noqa: F401
|
|
|
|
|
|
from .service import Job, JobStatus
|
|
|
|
|
|
_export_svc = ExportService()
|
|
|
|
SPLIT_PATTERN = r"\n+" # Kept for backward compatibility; prefer get_split_pattern()
|
|
SAMPLE_RATE = 24000
|
|
|
|
|
|
class _JobCancelled(Exception):
|
|
"""Raised internally to abort a conversion when the client cancels."""
|
|
|
|
|
|
def run_conversion_job(job: Job) -> None:
|
|
job.add_log("Preparing conversion pipeline")
|
|
canceller = _make_canceller(job)
|
|
|
|
usage_counter: Dict[str, int] = defaultdict(int)
|
|
|
|
def _tts_log(level: str, msg: str) -> None:
|
|
job.add_log(msg, level=level)
|
|
|
|
tts_context = build_tts_context(
|
|
language=str(job.language or "a"),
|
|
subtitle_mode=str(job.subtitle_mode or "Disabled"),
|
|
pronunciation_overrides=getattr(job, "pronunciation_overrides", None),
|
|
manual_overrides=getattr(job, "manual_overrides", None),
|
|
heteronym_overrides=getattr(job, "heteronym_overrides", None),
|
|
speakers=getattr(job, "speakers", None),
|
|
normalization_overrides=getattr(job, "normalization_overrides", None),
|
|
usage_counter=usage_counter,
|
|
log_callback=_tts_log,
|
|
)
|
|
|
|
sink_stack = ExitStack()
|
|
subtitle_writer = None
|
|
chapter_paths: list[Path] = []
|
|
chapter_markers: List[Dict[str, Any]] = []
|
|
chunk_markers: List[Dict[str, Any]] = []
|
|
metadata_payload: Dict[str, Any] = {}
|
|
audio_output_path: Optional[Path] = None
|
|
extraction: Optional[Any] = None
|
|
pipeline: Any = None
|
|
pipeline_pool = PipelinePool()
|
|
normalized_profiles: Dict[str, Dict[str, Any]] = {}
|
|
chunk_groups: Dict[int, List[Dict[str, Any]]] = {}
|
|
active_chapter_configs: List[Dict[str, Any]] = []
|
|
override_token_map: Dict[str, str] = {}
|
|
try:
|
|
# Load saved speakers once so we can resolve speaker: references during conversion.
|
|
try:
|
|
profiles = load_profiles()
|
|
except Exception:
|
|
profiles = {}
|
|
for name, entry in (profiles or {}).items():
|
|
normalized = normalize_profile_entry(entry)
|
|
if normalized:
|
|
normalized_profiles[str(name)] = normalized
|
|
|
|
def resolve_voice_choice(raw_spec: str) -> tuple[str, str, Any, Optional[float], Optional[int]]:
|
|
"""Resolve a raw voice spec into (provider, resolved_spec, choice, speed, steps)."""
|
|
provider, resolved, speed, steps = _resolve_voice_target(
|
|
raw_spec,
|
|
normalized_profiles,
|
|
job_voice=getattr(job, "voice", "M1"),
|
|
job_tts_provider=getattr(job, "tts_provider", "kokoro"),
|
|
job_supertonic_total_steps=getattr(job, "supertonic_total_steps", 5),
|
|
job_speed=getattr(job, "speed", 1.0),
|
|
)
|
|
cache_key = f"{provider}:{resolved}" if resolved else provider
|
|
cached = voice_cache.get(cache_key)
|
|
if cached is not None:
|
|
return provider, resolved, cached, speed, steps
|
|
|
|
if provider == "kokoro":
|
|
kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job)
|
|
choice = resolve_voice(resolved, kokoro_backend, job.use_gpu, cache=voice_cache)
|
|
else:
|
|
choice = resolved
|
|
|
|
voice_cache.set(cache_key, choice)
|
|
return provider, resolved, choice, speed, steps
|
|
|
|
extraction = extract_from_path(job.stored_path)
|
|
file_type = _infer_file_type(job.stored_path)
|
|
|
|
# Build override_token_map from pronunciation overrides
|
|
pronunciation_overrides = merge_pronunciation_overrides(job)
|
|
for override_entry in pronunciation_overrides or []:
|
|
if not isinstance(override_entry, Mapping):
|
|
continue
|
|
raw_token = str(override_entry.get("token") or "").strip()
|
|
normalized_value = str(override_entry.get("normalized") or "").strip()
|
|
if not normalized_value and raw_token:
|
|
normalized_value = normalize_entity_token(raw_token) or raw_token
|
|
if normalized_value:
|
|
override_token_map.setdefault(normalized_value, raw_token or normalized_value)
|
|
|
|
if not job.chapters:
|
|
filtered, skipped_info = _auto_select_relevant_chapters(extraction.chapters, file_type)
|
|
original_count = len(extraction.chapters)
|
|
if filtered and len(filtered) < original_count:
|
|
extraction.chapters = filtered
|
|
_update_metadata_for_chapter_count(extraction.metadata, len(filtered), file_type)
|
|
threshold = _SIGNIFICANT_LENGTH_THRESHOLDS.get(file_type.lower())
|
|
label = _chapter_label(file_type)
|
|
qualifier = f" (< {threshold} characters)" if threshold else ""
|
|
job.add_log(
|
|
f"Auto-selected {len(filtered)} of {original_count} {label} based on content{qualifier}.",
|
|
level="info",
|
|
)
|
|
if skipped_info:
|
|
preview_count = 5
|
|
preview = ", ".join(
|
|
f"{title or 'Untitled'} ({length})" for title, length in skipped_info[:preview_count]
|
|
)
|
|
if len(skipped_info) > preview_count:
|
|
preview += ", …"
|
|
job.add_log(
|
|
f"Skipped {len(skipped_info)} short {label}: {preview}",
|
|
level="debug",
|
|
)
|
|
elif not filtered:
|
|
job.add_log(
|
|
"Auto-selection did not identify usable chapters; retaining original set.",
|
|
level="warning",
|
|
)
|
|
|
|
metadata_overrides: Dict[str, Any] = dict(job.metadata_tags or {})
|
|
if job.chapters:
|
|
selected_chapters, chapter_metadata, diagnostics = _apply_chapter_overrides(
|
|
extraction.chapters,
|
|
job.chapters,
|
|
)
|
|
for message in diagnostics:
|
|
job.add_log(message, level="warning")
|
|
if selected_chapters:
|
|
extraction.chapters = selected_chapters
|
|
metadata_overrides.update(chapter_metadata)
|
|
job.add_log(
|
|
f"Chapter overrides applied: {len(selected_chapters)} selected.",
|
|
level="info",
|
|
)
|
|
active_chapter_configs = [
|
|
entry for entry in job.chapters if _coerce_truthy(entry.get("enabled", True))
|
|
][: len(selected_chapters)]
|
|
if job.chunks:
|
|
chunk_groups = _group_chunks_by_chapter(job.chunks)
|
|
else:
|
|
raise ValueError("No chapters were enabled in the requested job.")
|
|
elif job.chunks:
|
|
chunk_groups = _group_chunks_by_chapter(job.chunks)
|
|
|
|
job.metadata_tags = _merge_metadata(extraction.metadata, metadata_overrides)
|
|
|
|
total_characters = extraction.total_characters or calculate_text_length(extraction.combined_text)
|
|
job.total_characters = total_characters
|
|
job.add_log(f"Total characters: {job.total_characters:,}")
|
|
|
|
_apply_newline_policy(extraction.chapters, job.replace_single_newlines)
|
|
|
|
base_output_dir = _prepare_output_dir(job)
|
|
project_root, audio_dir, subtitle_dir, metadata_dir = _resolve_project_layout(
|
|
original_filename=job.original_filename,
|
|
save_as_project=job.save_as_project,
|
|
base_dir=base_output_dir,
|
|
)
|
|
|
|
if job.output_format.lower() == "m4b" and not job.merge_chapters_at_end:
|
|
job.add_log(
|
|
"Forcing merged output for m4b format; ignoring 'merge chapters at end' setting.",
|
|
level="warning",
|
|
)
|
|
job.merge_chapters_at_end = True
|
|
|
|
merged_required = job.merge_chapters_at_end or not job.save_chapters_separately
|
|
audio_path: Optional[Path] = None
|
|
audio_sink: Optional[AudioSink] = None
|
|
if merged_required:
|
|
audio_path = _build_output_path(audio_dir, job.original_filename, job.output_format)
|
|
meta_for_sink = job.metadata_tags if job.metadata_tags else None
|
|
audio_sink = sink_stack.enter_context(
|
|
open_audio_sink(
|
|
audio_path,
|
|
job.output_format,
|
|
metadata=meta_for_sink,
|
|
cancel_check=lambda: job.cancel_requested,
|
|
)
|
|
)
|
|
subtitle_writer = make_subtitle_writer(
|
|
audio_path,
|
|
job.subtitle_format,
|
|
job.subtitle_mode or "Line",
|
|
max_words=job.max_subtitle_words,
|
|
)
|
|
if subtitle_writer is None and job.subtitle_mode != "Disabled":
|
|
fmt = (job.subtitle_format or "srt").lower()
|
|
if job.subtitle_mode == "Sentence + Highlighting" and fmt == "srt":
|
|
job.add_log("Highlighting requires ASS subtitles. Switching format.", level="warning")
|
|
else:
|
|
job.add_log(f"Unsupported subtitle format '{job.subtitle_format}'. Skipping.", level="warning")
|
|
job.result.audio_path = audio_path
|
|
if subtitle_writer:
|
|
job.result.subtitle_paths.append(subtitle_writer.path)
|
|
|
|
chapter_dir: Optional[Path] = None
|
|
if job.save_chapters_separately:
|
|
chapter_dir = audio_dir / "chapters"
|
|
chapter_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
base_voice_spec = _job_voice_fallback(job)
|
|
voice_cache = VoiceCache()
|
|
base_provider, base_voice_resolved, _, _ = _resolve_voice_target(
|
|
base_voice_spec, normalized_profiles,
|
|
job_voice=getattr(job, "voice", "M1"),
|
|
job_tts_provider=getattr(job, "tts_provider", "kokoro"),
|
|
)
|
|
if base_provider == "kokoro" and base_voice_resolved and "*" not in base_voice_resolved:
|
|
kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job)
|
|
voice_cache.set(f"kokoro:{base_voice_resolved}", resolve_voice(base_voice_resolved, kokoro_backend, job.use_gpu))
|
|
processed_chars = 0
|
|
current_time = 0.0
|
|
etr_start_time = time.time()
|
|
total_chapters = len(extraction.chapters)
|
|
if chunk_groups:
|
|
chunk_groups = {
|
|
idx: items for idx, items in chunk_groups.items() if 0 <= idx < total_chapters
|
|
}
|
|
job.add_log(f"Detected {total_chapters} chapter{'s' if total_chapters != 1 else ''}")
|
|
auto_prefix_titles = getattr(job, "auto_prefix_chapter_titles", True)
|
|
read_title_intro = getattr(job, "read_title_intro", False)
|
|
book_intro_text = ""
|
|
intro_provider: Optional[str] = None
|
|
intro_voice_choice: Any = None
|
|
intro_speed: Optional[float] = None
|
|
intro_steps: Optional[int] = None
|
|
intro_spec = resolve_intro(
|
|
job.metadata_tags, job.original_filename, read_title_intro,
|
|
base_voice_spec, getattr(job, "voice", "M1"), list(voice_cache.keys()),
|
|
)
|
|
if intro_spec.enabled:
|
|
book_intro_text = intro_spec.text
|
|
preview = book_intro_text if len(book_intro_text) <= 120 else f"{book_intro_text[:117]}…"
|
|
job.add_log(f"Title intro enabled: {preview}", level="debug")
|
|
|
|
intro_provider, _, intro_voice_choice, intro_speed, intro_steps = resolve_voice_choice(
|
|
intro_spec.voice_spec
|
|
)
|
|
elif read_title_intro:
|
|
job.add_log("Title intro enabled but no usable metadata was found.", level="debug")
|
|
intro_emitted = False
|
|
|
|
def emit_text(
|
|
text: str,
|
|
*,
|
|
voice_choice: Any,
|
|
chapter_sink: Optional[AudioSink],
|
|
preview_prefix: Optional[str] = None,
|
|
split_pattern: Optional[str] = None,
|
|
tts_provider: Optional[str] = None,
|
|
speed_override: Optional[float] = None,
|
|
supertonic_steps_override: Optional[int] = None,
|
|
) -> int:
|
|
nonlocal processed_chars, current_time
|
|
source_text = str(text or "")
|
|
|
|
provider = str(tts_provider or getattr(job, "tts_provider", "kokoro") or "kokoro").strip().lower() or "kokoro"
|
|
if provider == "supertonic":
|
|
supertonic_pipeline = pipeline_pool.get("supertonic", job.language, job.use_gpu, job=job)
|
|
voice_name = _supertonic_voice_from_spec(voice_choice, getattr(job, "voice", "M1"))
|
|
backend = supertonic_pipeline
|
|
resolved_voice = voice_name
|
|
effective_speed = float(speed_override if speed_override is not None else job.speed)
|
|
else:
|
|
kokoro_backend = pipeline_pool.get("kokoro", job.language, job.use_gpu, job=job)
|
|
backend = kokoro_backend
|
|
resolved_voice = voice_choice
|
|
effective_speed = float(speed_override if speed_override is not None else job.speed)
|
|
|
|
try:
|
|
stats = SegmentStats(
|
|
processed_chars=processed_chars,
|
|
current_time=current_time,
|
|
etr_start_time=etr_start_time,
|
|
total_characters=job.total_characters or 0,
|
|
)
|
|
prefix = f"{preview_prefix} · " if preview_prefix else ""
|
|
|
|
def _on_progress(pct: int, etr: str) -> None:
|
|
nonlocal processed_chars
|
|
processed_chars = stats.processed_chars
|
|
job.processed_characters = processed_chars
|
|
if stats.total_characters:
|
|
job.progress = min(processed_chars / stats.total_characters, 0.999)
|
|
else:
|
|
job.progress = 0.0 if processed_chars == 0 else 0.999
|
|
job.etr_str = etr
|
|
|
|
def _preview(text: str) -> None:
|
|
job.add_log(f"{prefix}{stats.processed_chars:,}/{job.total_characters or '—'}: {text[:80]}")
|
|
|
|
synth_params = SynthParams(
|
|
tts_context=tts_context,
|
|
stats=stats,
|
|
check_cancel=canceller,
|
|
on_progress=_on_progress,
|
|
audio_sink=audio_sink,
|
|
subtitle_mode=job.subtitle_mode if (subtitle_writer and audio_sink) else "Disabled",
|
|
max_subtitle_words=job.max_subtitle_words,
|
|
lang_code=job.language,
|
|
use_spacy_segmentation=job.subtitle_mode not in ("Disabled", "Line"),
|
|
)
|
|
|
|
local_segments, accumulated_tokens = synthesize_text(
|
|
text=source_text,
|
|
params=synth_params,
|
|
backend=backend,
|
|
voice=resolved_voice,
|
|
speed=effective_speed,
|
|
chapter_sink=chapter_sink,
|
|
preview_callback=_preview,
|
|
)
|
|
current_time = stats.current_time
|
|
|
|
if subtitle_writer and audio_sink and accumulated_tokens:
|
|
process_and_write_subtitles(
|
|
accumulated_tokens,
|
|
subtitle_writer,
|
|
subtitle_mode=job.subtitle_mode,
|
|
max_subtitle_words=job.max_subtitle_words,
|
|
lang_code=job.language,
|
|
use_spacy_segmentation=job.subtitle_mode not in ("Disabled", "Line"),
|
|
fallback_end_time=current_time,
|
|
)
|
|
|
|
except OverflowError as exc:
|
|
job.add_log(
|
|
f"Skipped chunk — number too large for TTS conversion: {exc}",
|
|
level="warning",
|
|
)
|
|
return local_segments
|
|
|
|
def append_silence(
|
|
duration_seconds: float,
|
|
*,
|
|
include_in_chapter: bool,
|
|
chapter_sink: Optional[AudioSink],
|
|
) -> None:
|
|
nonlocal current_time
|
|
if duration_seconds <= 0:
|
|
return
|
|
silence = _create_silence(duration_seconds)
|
|
if silence.size == 0:
|
|
return
|
|
if include_in_chapter and chapter_sink:
|
|
chapter_sink.write(silence)
|
|
if audio_sink:
|
|
audio_sink.write(silence)
|
|
current_time += duration_seconds
|
|
|
|
for idx, chapter in enumerate(extraction.chapters, start=1):
|
|
canceller()
|
|
raw_title = str(getattr(chapter, "title", "") or "").strip()
|
|
spoken_title = _format_spoken_chapter_title(raw_title, idx, auto_prefix_titles)
|
|
heading_text = spoken_title or raw_title
|
|
chapter_display_title = heading_text or f"Chapter {idx}"
|
|
job.add_log(f"Processing chapter {idx}/{total_chapters}: {chapter_display_title}")
|
|
normalize_opening_caps = bool(job.normalize_chapter_opening_caps)
|
|
|
|
chapter_start_time = current_time
|
|
chapter_override = (
|
|
active_chapter_configs[idx - 1] if idx - 1 < len(active_chapter_configs) else None
|
|
)
|
|
chapter_voice_spec = _chapter_voice_spec(job, chapter_override)
|
|
if not chapter_voice_spec:
|
|
chapter_voice_spec = base_voice_spec
|
|
|
|
chapter_provider, chapter_voice_resolved, voice_choice, chapter_speed, chapter_steps = resolve_voice_choice(
|
|
chapter_voice_spec
|
|
)
|
|
|
|
chapter_audio_path: Optional[Path] = None
|
|
segments_emitted = 0
|
|
|
|
with ExitStack() as chapter_sink_stack:
|
|
chapter_sink: Optional[AudioSink] = None
|
|
|
|
if chapter_dir is not None:
|
|
chapter_audio_path = _build_output_path(
|
|
chapter_dir,
|
|
f"{Path(job.original_filename).stem}_{_slugify(chapter_display_title, idx)}",
|
|
job.separate_chapters_format,
|
|
)
|
|
chapter_sink = chapter_sink_stack.enter_context(
|
|
open_audio_sink(
|
|
chapter_audio_path,
|
|
job.separate_chapters_format,
|
|
cancel_check=lambda: job.cancel_requested,
|
|
)
|
|
)
|
|
|
|
speak_heading = bool(heading_text)
|
|
first_line = ""
|
|
if chapter.text:
|
|
first_line = next((line.strip() for line in chapter.text.splitlines() if line.strip()), "")
|
|
remove_heading_from_body = False
|
|
if speak_heading and first_line:
|
|
if _headings_equivalent(first_line, heading_text) or (raw_title and _headings_equivalent(first_line, raw_title)):
|
|
remove_heading_from_body = True
|
|
|
|
if not intro_emitted and book_intro_text:
|
|
intro_use_provider = intro_provider or chapter_provider
|
|
intro_use_voice_choice = intro_voice_choice if intro_voice_choice is not None else voice_choice
|
|
intro_use_speed = intro_speed if intro_speed is not None else chapter_speed
|
|
intro_use_steps = intro_steps if intro_steps is not None else chapter_steps
|
|
intro_segments = emit_text(
|
|
book_intro_text,
|
|
voice_choice=intro_use_voice_choice,
|
|
chapter_sink=chapter_sink,
|
|
preview_prefix="Book intro",
|
|
tts_provider=intro_use_provider,
|
|
speed_override=intro_use_speed,
|
|
supertonic_steps_override=intro_use_steps,
|
|
)
|
|
intro_emitted = True
|
|
if intro_segments > 0 and job.chapter_intro_delay > 0:
|
|
append_silence(
|
|
job.chapter_intro_delay,
|
|
include_in_chapter=True,
|
|
chapter_sink=chapter_sink,
|
|
)
|
|
|
|
if speak_heading:
|
|
heading_segments = emit_text(
|
|
heading_text,
|
|
voice_choice=voice_choice,
|
|
chapter_sink=chapter_sink,
|
|
preview_prefix=f"Chapter {idx} title",
|
|
tts_provider=chapter_provider,
|
|
speed_override=chapter_speed,
|
|
supertonic_steps_override=chapter_steps,
|
|
)
|
|
segments_emitted += heading_segments
|
|
if heading_segments > 0 and job.chapter_intro_delay > 0:
|
|
append_silence(
|
|
job.chapter_intro_delay,
|
|
include_in_chapter=True,
|
|
chapter_sink=chapter_sink,
|
|
)
|
|
|
|
chunks_for_chapter = chunk_groups.get(idx - 1, []) if chunk_groups else []
|
|
body_segments = 0
|
|
pending_heading_strip = remove_heading_from_body
|
|
opening_caps_pending = normalize_opening_caps
|
|
opening_caps_logged = False
|
|
if chunks_for_chapter:
|
|
job.add_log(
|
|
f"Emitting {len(chunks_for_chapter)} {job.chunk_level} chunks for chapter {idx}.",
|
|
level="debug",
|
|
)
|
|
for chunk_entry in chunks_for_chapter:
|
|
chunk_text = _chunk_text_for_tts(chunk_entry)
|
|
if not chunk_text:
|
|
continue
|
|
|
|
mutated_entry = False
|
|
chunk_text, heading_removed, caps_changed = _apply_chapter_text_transforms(
|
|
chunk_text,
|
|
heading_text=heading_text,
|
|
raw_title=raw_title,
|
|
strip_heading=pending_heading_strip,
|
|
normalize_caps=opening_caps_pending,
|
|
)
|
|
if heading_removed:
|
|
pending_heading_strip = False
|
|
chunk_entry = dict(chunk_entry)
|
|
chunk_entry["normalized_text"] = chunk_text
|
|
mutated_entry = True
|
|
if not chunk_text.strip():
|
|
continue
|
|
if caps_changed:
|
|
if not mutated_entry:
|
|
chunk_entry = dict(chunk_entry)
|
|
chunk_entry["normalized_text"] = chunk_text
|
|
if not opening_caps_logged:
|
|
job.add_log(
|
|
f"Normalized uppercase chapter opening for chapter {idx}.",
|
|
level="debug",
|
|
)
|
|
opening_caps_logged = True
|
|
if chunk_text.strip():
|
|
opening_caps_pending = False
|
|
|
|
chunk_voice_spec = _chunk_voice_spec(
|
|
job,
|
|
chunk_entry,
|
|
chapter_voice_spec or base_voice_spec,
|
|
)
|
|
if not chunk_voice_spec:
|
|
chunk_voice_spec = chapter_voice_spec or base_voice_spec
|
|
|
|
if chunk_voice_spec == chapter_voice_spec:
|
|
chunk_provider = chapter_provider
|
|
chunk_voice_resolved = chapter_voice_resolved
|
|
chunk_speed_use = chapter_speed
|
|
chunk_steps_use = chapter_steps
|
|
chunk_voice_choice = voice_choice
|
|
else:
|
|
chunk_provider, chunk_voice_resolved, chunk_voice_choice, chunk_speed_use, chunk_steps_use = resolve_voice_choice(
|
|
chunk_voice_spec
|
|
)
|
|
|
|
chunk_start = current_time
|
|
emitted = emit_text(
|
|
chunk_text,
|
|
voice_choice=chunk_voice_choice,
|
|
chapter_sink=chapter_sink,
|
|
preview_prefix=f"Chunk {chunk_entry.get('id') or chunk_entry.get('chunk_index')}",
|
|
tts_provider=chunk_provider,
|
|
speed_override=chunk_speed_use,
|
|
supertonic_steps_override=chunk_steps_use,
|
|
)
|
|
if emitted <= 0:
|
|
continue
|
|
|
|
body_segments += emitted
|
|
segments_emitted += emitted
|
|
chunk_markers.append(
|
|
{
|
|
"id": chunk_entry.get("id"),
|
|
"chapter_index": idx - 1,
|
|
"chunk_index": _safe_int(
|
|
chunk_entry.get("chunk_index"), len(chunk_markers)
|
|
),
|
|
"start": chunk_start,
|
|
"end": current_time,
|
|
"speaker_id": chunk_entry.get("speaker_id", "narrator"),
|
|
"voice": chunk_voice_spec,
|
|
"level": chunk_entry.get("level", job.chunk_level),
|
|
"characters": len(chunk_text),
|
|
}
|
|
)
|
|
|
|
if body_segments == 0:
|
|
chapter_body_start = current_time
|
|
chapter_text = str(chapter.text or "")
|
|
chapter_text, heading_removed, caps_changed = _apply_chapter_text_transforms(
|
|
chapter_text,
|
|
heading_text=heading_text,
|
|
raw_title=raw_title,
|
|
strip_heading=pending_heading_strip,
|
|
normalize_caps=opening_caps_pending,
|
|
)
|
|
if heading_removed:
|
|
pending_heading_strip = False
|
|
if caps_changed:
|
|
if not opening_caps_logged:
|
|
job.add_log(
|
|
f"Normalized uppercase chapter opening for chapter {idx}.",
|
|
level="debug",
|
|
)
|
|
opening_caps_logged = True
|
|
if str(chapter_text or "").strip():
|
|
opening_caps_pending = False
|
|
emitted = emit_text(
|
|
chapter_text,
|
|
voice_choice=voice_choice,
|
|
chapter_sink=chapter_sink,
|
|
tts_provider=chapter_provider,
|
|
speed_override=chapter_speed,
|
|
supertonic_steps_override=chapter_steps,
|
|
)
|
|
if emitted > 0:
|
|
segments_emitted += emitted
|
|
chunk_markers.append(
|
|
{
|
|
"id": None,
|
|
"chapter_index": idx - 1,
|
|
"chunk_index": 0,
|
|
"start": chapter_body_start,
|
|
"end": current_time,
|
|
"speaker_id": "narrator",
|
|
"voice": chapter_voice_spec,
|
|
"level": job.chunk_level,
|
|
"characters": len(chapter_text or ""),
|
|
}
|
|
)
|
|
elif chunks_for_chapter:
|
|
job.add_log(
|
|
"No audio generated for supplied chunks; chapter text also empty.",
|
|
level="warning",
|
|
)
|
|
|
|
chapter_end_time = current_time
|
|
|
|
if chapter_audio_path is not None:
|
|
job.result.artifacts[f"chapter_{idx:02d}"] = chapter_audio_path
|
|
chapter_paths.append(chapter_audio_path)
|
|
|
|
if segments_emitted == 0:
|
|
job.add_log(
|
|
f"No audio segments were generated for chapter {idx}.",
|
|
level="warning",
|
|
)
|
|
else:
|
|
job.add_log(f"Finished chapter {idx} with {segments_emitted} segments.")
|
|
|
|
if (
|
|
audio_sink
|
|
and job.merge_chapters_at_end
|
|
and idx < total_chapters
|
|
and job.silence_between_chapters > 0
|
|
):
|
|
append_silence(
|
|
job.silence_between_chapters,
|
|
include_in_chapter=False,
|
|
chapter_sink=None,
|
|
)
|
|
chapter_end_time = current_time
|
|
|
|
marker = {
|
|
"index": idx,
|
|
"title": chapter_display_title,
|
|
"start": chapter_start_time,
|
|
"end": chapter_end_time,
|
|
"voice": chapter_voice_spec,
|
|
}
|
|
if raw_title and raw_title != chapter_display_title:
|
|
marker["original_title"] = raw_title
|
|
chapter_markers.append(marker)
|
|
|
|
if getattr(job, "read_closing_outro", True):
|
|
outro_spec = resolve_outro(
|
|
job.metadata_tags, job.original_filename, True,
|
|
base_voice_spec, getattr(job, "voice", "M1"), list(voice_cache.keys()),
|
|
)
|
|
|
|
if outro_spec.enabled:
|
|
outro_start_time = current_time
|
|
outro_audio_path: Optional[Path] = None
|
|
outro_segments = 0
|
|
outro_index = total_chapters + 1
|
|
outro_provider, _, outro_voice_choice, outro_speed, outro_steps = resolve_voice_choice(outro_spec.voice_spec)
|
|
|
|
with ExitStack() as outro_sink_stack:
|
|
chapter_sink: Optional[AudioSink] = None
|
|
if chapter_dir is not None:
|
|
outro_audio_path = _build_output_path(
|
|
chapter_dir,
|
|
f"{Path(job.original_filename).stem}_outro",
|
|
job.separate_chapters_format,
|
|
)
|
|
chapter_sink = outro_sink_stack.enter_context(
|
|
open_audio_sink(
|
|
outro_audio_path,
|
|
job.separate_chapters_format,
|
|
cancel_check=lambda: job.cancel_requested,
|
|
)
|
|
)
|
|
|
|
outro_segments = emit_text(
|
|
outro_spec.text,
|
|
voice_choice=outro_voice_choice,
|
|
chapter_sink=chapter_sink,
|
|
preview_prefix="Outro",
|
|
tts_provider=outro_provider,
|
|
speed_override=outro_speed,
|
|
supertonic_steps_override=outro_steps,
|
|
)
|
|
outro_end_time = current_time
|
|
|
|
if outro_segments > 0:
|
|
job.add_log(f"Appended outro sequence: {outro_spec.text}")
|
|
if outro_audio_path is not None:
|
|
job.result.artifacts[f"chapter_{outro_index:02d}"] = outro_audio_path
|
|
chapter_paths.append(outro_audio_path)
|
|
chapter_markers.append(
|
|
{
|
|
"index": outro_index,
|
|
"title": "Outro",
|
|
"start": outro_start_time,
|
|
"end": outro_end_time,
|
|
"voice": outro_spec.voice_spec,
|
|
}
|
|
)
|
|
else:
|
|
job.add_log("No audio generated for outro sequence.", level="warning")
|
|
|
|
if not audio_path and chapter_paths:
|
|
job.result.audio_path = chapter_paths[0]
|
|
|
|
metadata_payload = _build_metadata_payload(
|
|
metadata=job.metadata_tags,
|
|
chapter_markers=chapter_markers,
|
|
chunk_markers=chunk_markers,
|
|
chunk_level=job.chunk_level,
|
|
speaker_mode=job.speaker_mode,
|
|
speakers=getattr(job, "speakers", None),
|
|
generate_epub3=job.generate_epub3,
|
|
)
|
|
|
|
if tts_context.usage_counter:
|
|
_record_override_usage(job, tts_context.usage_counter, override_token_map)
|
|
|
|
if metadata_dir:
|
|
metadata_dir.mkdir(parents=True, exist_ok=True)
|
|
metadata_file = metadata_dir / "metadata.json"
|
|
metadata_file.write_text(json.dumps(metadata_payload, indent=2), encoding="utf-8")
|
|
job.result.artifacts["metadata"] = metadata_file
|
|
|
|
if job.generate_epub3:
|
|
audio_asset = job.result.audio_path
|
|
if not audio_asset and chapter_paths:
|
|
audio_asset = chapter_paths[0]
|
|
|
|
if audio_asset:
|
|
try:
|
|
epub_root = project_root
|
|
epub_output_path = _build_output_path(epub_root, job.original_filename, "epub")
|
|
job.add_log("Generating EPUB 3 package with synchronized narration…")
|
|
epub_path = build_epub3_package(
|
|
output_path=epub_output_path,
|
|
book_id=job.id,
|
|
extraction=extraction,
|
|
metadata_tags=metadata_payload.get("metadata") or {},
|
|
chapter_markers=chapter_markers,
|
|
chunk_markers=chunk_markers,
|
|
chunks=job.chunks,
|
|
audio_path=audio_asset,
|
|
speaker_mode=job.speaker_mode,
|
|
cover_image_path=job.cover_image_path,
|
|
cover_image_mime=job.cover_image_mime,
|
|
)
|
|
job.result.epub_path = epub_path
|
|
job.result.artifacts["epub3"] = epub_path
|
|
job.add_log(f"EPUB 3 package created at {epub_path}")
|
|
except Exception as exc:
|
|
job.add_log(f"Failed to generate EPUB 3 package: {exc}", level="error")
|
|
else:
|
|
job.add_log("Skipped EPUB 3 generation: audio output unavailable.", level="warning")
|
|
|
|
if job.save_as_project:
|
|
job.result.artifacts["project_root"] = project_root
|
|
|
|
if job.status != JobStatus.CANCELLED:
|
|
job.progress = 1.0
|
|
|
|
audio_output_path = job.result.audio_path
|
|
|
|
except _JobCancelled:
|
|
job.status = JobStatus.CANCELLED
|
|
job.add_log("Job cancelled", level="warning")
|
|
except Exception as exc: # pragma: no cover - defensive guard
|
|
job.error = str(exc)
|
|
job.status = JobStatus.FAILED
|
|
exc_type = exc.__class__.__name__
|
|
job.add_log(f"Job failed ({exc_type}): {exc}", level="error")
|
|
|
|
chapter_count: Any
|
|
if extraction is not None and hasattr(extraction, "chapters"):
|
|
try:
|
|
chapter_count = len(getattr(extraction, "chapters", []) or [])
|
|
except Exception: # pragma: no cover - defensive fallback
|
|
chapter_count = "unavailable"
|
|
else:
|
|
chapter_count = "unavailable"
|
|
|
|
try:
|
|
chunk_group_count = len(chunk_groups)
|
|
chunk_total = sum(len(items) for items in chunk_groups.values())
|
|
except Exception: # pragma: no cover - defensive fallback
|
|
chunk_group_count = "unavailable"
|
|
chunk_total = "unavailable"
|
|
|
|
job.add_log(
|
|
"Context => chunk_level=%s, chapters=%s, chunk_groups=%s, chunks=%s"
|
|
% (job.chunk_level, chapter_count, chunk_group_count, chunk_total),
|
|
level="debug",
|
|
)
|
|
|
|
first_nonempty_group = next((items for items in chunk_groups.values() if items), None)
|
|
if first_nonempty_group:
|
|
first_chunk = dict(first_nonempty_group[0])
|
|
sample_text = str(first_chunk.get("text") or "")[:160].replace("\n", " ")
|
|
job.add_log(
|
|
"First chunk sample => id=%s, speaker=%s, chars=%s, preview=%s"
|
|
% (
|
|
first_chunk.get("id") or first_chunk.get("chunk_index"),
|
|
first_chunk.get("speaker_id", "narrator"),
|
|
len(str(first_chunk.get("text") or "")),
|
|
sample_text,
|
|
),
|
|
level="debug",
|
|
)
|
|
|
|
tb_lines = traceback.format_exception(exc.__class__, exc, exc.__traceback__)
|
|
for line in tb_lines[:20]:
|
|
trimmed = line.rstrip()
|
|
if trimmed:
|
|
for snippet in trimmed.splitlines():
|
|
job.add_log(f"TRACE: {snippet}", level="debug")
|
|
finally:
|
|
sink_stack.close()
|
|
if subtitle_writer:
|
|
subtitle_writer.close()
|
|
|
|
# Explicitly release the pipeline and force garbage collection to prevent
|
|
# memory accumulation in the worker process, which can lead to host lockups.
|
|
pipeline_pool.dispose_all()
|
|
pipeline = None
|
|
gc.collect()
|
|
try:
|
|
import torch # type: ignore[import-not-found]
|
|
if torch.cuda.is_available():
|
|
torch.cuda.empty_cache()
|
|
except ImportError:
|
|
pass
|
|
|
|
if (
|
|
audio_output_path
|
|
and job.output_format.lower() == "m4b"
|
|
and not job.cancel_requested
|
|
and job.status not in {JobStatus.FAILED, JobStatus.CANCELLED}
|
|
):
|
|
try:
|
|
cover_path = None
|
|
if job.cover_image_path:
|
|
candidate = Path(job.cover_image_path)
|
|
if candidate.exists():
|
|
cover_path = candidate
|
|
|
|
_export_svc.embed_m4b_metadata(
|
|
audio_path=audio_output_path,
|
|
metadata=metadata_payload.get("metadata") or {},
|
|
chapters=metadata_payload.get("chapters") or [],
|
|
cover_path=cover_path,
|
|
cover_mime=job.cover_image_mime,
|
|
log_callback=lambda msg, level="info": job.add_log(msg, level=level),
|
|
)
|
|
except Exception as exc: # pragma: no cover - ensure failure propagates
|
|
job.add_log(
|
|
f"Failed to embed metadata into m4b output: {exc}",
|
|
level="error",
|
|
)
|
|
raise RuntimeError(
|
|
f"Failed to embed metadata into m4b output: {exc}"
|
|
) from exc
|
|
|
|
|
|
def _prepare_output_dir(job: Job) -> Path:
|
|
from platformdirs import user_desktop_dir # type: ignore[import-not-found]
|
|
|
|
default_output = Path(str(get_user_cache_path("outputs")))
|
|
directory = _resolve_output_directory(
|
|
save_mode=job.save_mode,
|
|
stored_path=job.stored_path,
|
|
output_folder=getattr(job, "output_folder", None),
|
|
desktop_dir=Path(user_desktop_dir()),
|
|
user_output_path=Path(get_user_output_path()),
|
|
user_cache_outputs=default_output,
|
|
)
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory
|
|
|
|
|
|
|
|
def _make_canceller(job: Job) -> Callable[[], None]:
|
|
def _cancel() -> None:
|
|
if job.cancel_requested:
|
|
raise _JobCancelled
|
|
|
|
return _cancel
|