from __future__ import annotations
import html
import re
import shutil
import uuid
from dataclasses import dataclass
from datetime import datetime, timezone
from pathlib import Path
from tempfile import TemporaryDirectory
from typing import Any, Dict, Iterable, List, Optional, Pattern, Sequence, Tuple
import zipfile
from abogen.text_extractor import ExtractedChapter, ExtractionResult
from abogen.domain.metadata_helpers import normalize_metadata_map
@dataclass(slots=True)
class ChunkOverlay:
id: str
text: str
original_text: Optional[str]
start: Optional[float]
end: Optional[float]
speaker_id: str
voice: Optional[Dict[str, str]]
level: Optional[str] = None
group_id: Optional[str] = None
@dataclass(slots=True)
class ChapterDocument:
index: int # zero-based
title: str
xhtml_name: str
smil_name: str
chunks: List[ChunkOverlay]
start: Optional[float]
end: Optional[float]
class EPUB3PackageBuilder:
"""Constructs an EPUB 3 package with media overlays."""
def __init__(
self,
*,
output_path: Path,
book_id: str,
extraction: ExtractionResult,
metadata_tags: Dict[str, Any],
chapter_markers: Sequence[Dict[str, Any]],
chunk_markers: Sequence[Dict[str, Any]],
chunks: Iterable[Dict[str, Any]],
audio_path: Path,
speaker_mode: str = "single",
cover_image_path: Optional[Path] = None,
cover_image_mime: Optional[str] = None,
) -> None:
self.output_path = output_path
self.book_id = book_id or str(uuid.uuid4())
self.extraction = extraction
self.metadata_tags = normalize_metadata_map(metadata_tags)
self.chapter_markers = list(chapter_markers or [])
self.chunk_markers = list(chunk_markers or [])
self.chunks = list(chunks or [])
self.audio_path = audio_path
self.speaker_mode = speaker_mode or "single"
self.cover_image_path = cover_image_path if cover_image_path and cover_image_path.exists() else None
self.cover_image_mime = cover_image_mime
self._combined_metadata = _combine_metadata(extraction.metadata, self.metadata_tags)
self._title = self._combined_metadata.get("title") or self._fallback_title()
self._authors = _split_authors(self._combined_metadata)
self._language = self._determine_language()
self._publisher = self._combined_metadata.get("publisher") or ""
self._description = self._combined_metadata.get("comment")
self._duration = _calculate_total_duration(self.chunk_markers, self.chapter_markers)
self._modified = _utc_now_iso()
def build(self) -> Path:
if not self.audio_path or not self.audio_path.exists():
raise FileNotFoundError(f"Audio asset missing: {self.audio_path}")
chapter_documents = self._build_chapter_documents()
with TemporaryDirectory() as tmp_dir:
root = Path(tmp_dir)
oebps = root / "OEBPS"
text_dir = oebps / "text"
smil_dir = oebps / "smil"
audio_dir = oebps / "audio"
image_dir = oebps / "images"
stylesheet_dir = oebps / "styles"
for directory in (oebps, text_dir, smil_dir, audio_dir, stylesheet_dir):
directory.mkdir(parents=True, exist_ok=True)
if self.cover_image_path:
image_dir.mkdir(parents=True, exist_ok=True)
_write_mimetype(root)
_write_container_xml(root)
audio_filename = self.audio_path.name
embedded_audio = audio_dir / audio_filename
shutil.copy2(self.audio_path, embedded_audio)
if self.cover_image_path:
shutil.copy2(self.cover_image_path, image_dir / self.cover_image_path.name)
stylesheet_path = stylesheet_dir / "style.css"
stylesheet_path.write_text(_DEFAULT_STYLESHEET, encoding="utf-8")
for chapter in chapter_documents:
chapter_path = text_dir / chapter.xhtml_name
chapter_path.write_text(
self._render_chapter_xhtml(chapter),
encoding="utf-8",
)
smil_path = smil_dir / chapter.smil_name
smil_path.write_text(
self._render_chapter_smil(chapter, f"audio/{audio_filename}"),
encoding="utf-8",
)
nav_path = oebps / "nav.xhtml"
nav_path.write_text(self._render_nav(chapter_documents), encoding="utf-8")
opf_path = oebps / "content.opf"
opf_path.write_text(
self._render_opf(
chapter_documents,
audio_filename,
has_cover=self.cover_image_path is not None,
stylesheet_path=stylesheet_path.relative_to(oebps),
),
encoding="utf-8",
)
self.output_path.parent.mkdir(parents=True, exist_ok=True)
with zipfile.ZipFile(self.output_path, "w", compression=zipfile.ZIP_DEFLATED) as archive:
# Ensure mimetype is the first entry and stored without compression
mimetype_path = root / "mimetype"
info = zipfile.ZipInfo("mimetype")
info.compress_type = zipfile.ZIP_STORED
archive.writestr(info, mimetype_path.read_bytes())
for file_path in sorted(root.rglob("*")):
if file_path == mimetype_path or file_path.is_dir():
continue
archive.write(file_path, file_path.relative_to(root))
return self.output_path
# ------------------------------------------------------------------
def _build_chapter_documents(self) -> List[ChapterDocument]:
chunk_lookup = _build_chunk_lookup(self.chunks)
markers_by_chapter = _group_markers_by_chapter(self.chunk_markers)
chapter_meta = {int(entry.get("index", idx + 1)) - 1: dict(entry) for idx, entry in enumerate(self.chapter_markers)}
documents: List[ChapterDocument] = []
for chapter_index, chapter in enumerate(self.extraction.chapters):
markers = markers_by_chapter.get(chapter_index, [])
if not markers and chunk_lookup.by_chapter.get(chapter_index):
markers = [
{
"id": item.get("id"),
"chapter_index": chapter_index,
"chunk_index": item.get("chunk_index"),
"start": None,
"end": None,
"speaker_id": item.get("speaker_id", "narrator"),
"voice": item.get("voice"),
}
for item in chunk_lookup.by_chapter.get(chapter_index, [])
]
if not markers:
markers = [
{
"id": f"chap{chapter_index:04d}_auto0000",
"chapter_index": chapter_index,
"chunk_index": 0,
"start": chapter_meta.get(chapter_index, {}).get("start"),
"end": chapter_meta.get(chapter_index, {}).get("end"),
"speaker_id": "narrator",
"voice": None,
}
]
overlays = self._build_overlays_for_chapter(
chapter_index,
markers,
chunk_lookup,
)
xhtml_name = f"chapter_{chapter_index + 1:04d}.xhtml"
smil_name = f"chapter_{chapter_index + 1:04d}.smil"
chapter_start = chapter_meta.get(chapter_index, {}).get("start")
chapter_end = chapter_meta.get(chapter_index, {}).get("end")
documents.append(
ChapterDocument(
index=chapter_index,
title=chapter.title or f"Chapter {chapter_index + 1}",
xhtml_name=xhtml_name,
smil_name=smil_name,
chunks=overlays,
start=chapter_start,
end=chapter_end,
)
)
return documents
def _build_overlays_for_chapter(
self,
chapter_index: int,
markers: Sequence[Dict[str, Any]],
chunk_lookup: "ChunkLookup",
) -> List[ChunkOverlay]:
overlays: List[ChunkOverlay] = []
used_ids: set[str] = set()
chapter_chunks = list(chunk_lookup.by_chapter.get(chapter_index, []))
chapter_chunks.sort(key=lambda entry: _safe_int(entry.get("chunk_index")))
for position, marker in enumerate(markers):
chunk_id = marker.get("id")
chunk_entry = None
if chunk_id and chunk_id in chunk_lookup.by_id:
chunk_entry = chunk_lookup.by_id[chunk_id]
else:
candidate_index = _safe_int(marker.get("chunk_index"))
chunk_entry = _find_chunk_by_index(chapter_chunks, candidate_index)
if chunk_entry is None and chapter_chunks and position < len(chapter_chunks):
chunk_entry = chapter_chunks[position]
level = None
if chunk_entry is None:
text = self.extraction.chapters[chapter_index].text
speaker_id = str(marker.get("speaker_id") or "narrator")
voice = marker.get("voice")
else:
display_text = chunk_entry.get("display_text")
text = str(chunk_entry.get("text") or "")
speaker_id = str(chunk_entry.get("speaker_id") or marker.get("speaker_id") or "narrator")
voice = chunk_entry.get("voice") or chunk_entry.get("resolved_voice") or marker.get("voice")
level = chunk_entry.get("level") or None
if chunk_entry is None:
level = None
normalized_id = _normalize_chunk_id(chunk_id) if chunk_id else None
if not normalized_id:
normalized_id = f"chap{chapter_index:04d}_chunk{position:04d}"
while normalized_id in used_ids:
normalized_id = f"{normalized_id}_dup"
used_ids.add(normalized_id)
raw_group_key = chunk_entry.get("id") if chunk_entry else chunk_id
group_id = _derive_group_id(raw_group_key, level)
normalized_group_id = _normalize_chunk_id(group_id) if group_id else None
original_text = None
if chunk_entry is not None:
original_text = chunk_entry.get("original_text") or chunk_entry.get("display_text")
overlays.append(
ChunkOverlay(
id=normalized_id,
text=text or self.extraction.chapters[chapter_index].text,
original_text=str(original_text) if original_text is not None else None,
start=_safe_float(marker.get("start")),
end=_safe_float(marker.get("end")),
speaker_id=speaker_id,
voice=voice if isinstance(voice, dict) else None,
level=str(level) if level else None,
group_id=normalized_group_id,
)
)
chapter_text = ""
if 0 <= chapter_index < len(self.extraction.chapters):
chapter_entry = self.extraction.chapters[chapter_index]
chapter_text = getattr(chapter_entry, "text", "") or ""
_restore_original_chunk_text(chapter_text, overlays)
return overlays
def _render_chapter_xhtml(self, chapter: ChapterDocument) -> str:
language = html.escape(self._language or "en")
title = html.escape(chapter.title)
grouped_chunks = _group_chunks_for_render(chapter.chunks)
chunk_html = "\n".join(
_render_chunk_group_html(group_id, items) for group_id, items in grouped_chunks
)
if not chunk_html:
chunk_html = "
"
original_block = ""
if chapter.chunks:
original_text = "".join((chunk.original_text if chunk.original_text is not None else (chunk.text or "")) for chunk in chapter.chunks)
if original_text:
safe_original = html.escape(original_text)
original_block = (
"