diff --git a/abogen/domain/metadata_extraction.py b/abogen/domain/metadata_extraction.py index 1383b09..a3929af 100644 --- a/abogen/domain/metadata_extraction.py +++ b/abogen/domain/metadata_extraction.py @@ -1,16 +1,20 @@ """Metadata extraction and processing utilities. -This module provides functions for extracting metadata from text content -and generating ffmpeg metadata arguments. +This module provides functions for extracting metadata from text content, +formatting metadata tags for TTS embedding, and generating ffmpeg metadata arguments. """ from __future__ import annotations import datetime +import logging import os import re +import uuid from pathlib import Path -from typing import Dict, List, Optional, Tuple +from typing import Any, Dict, List, Optional, Tuple + +logger = logging.getLogger(__name__) def extract_metadata_from_text(text: str) -> Dict[str, Optional[str]]: @@ -189,3 +193,290 @@ def read_text_for_metadata( return f.read() except Exception: return "" + + +def format_metadata_tags( + metadata: Dict[str, Any], + filename: str, + chapter_count: int, + file_type: str, + cover_bytes: Optional[bytes] = None, + cache_dir: Optional[str] = None, +) -> str: + """Format metadata tags for insertion into TTS text. + + Builds <> tags that are later parsed by + extract_metadata_from_text() and fed to ffmpeg. + + Args: + metadata: Dict with keys like 'title', 'authors' (list), + 'publication_year', 'description', 'cover_image' (bytes). + filename: Fallback filename (without extension) for title/album. + chapter_count: Number of chapters/pages. + file_type: 'epub', 'pdf', or 'markdown'. + cover_bytes: Optional cover image bytes to save to cache. + cache_dir: Directory for cover cache (uses default if None). + + Returns: + Newline-joined string of <> tags. + """ + title = metadata.get("title") or filename + authors = metadata.get("authors") or ["Unknown"] + authors_text = ", ".join(authors) if isinstance(authors, list) else str(authors) + year = metadata.get("publication_year") or str(datetime.datetime.now().year) + + chapter_label = "Chapters" if file_type in ("epub", "markdown") else "Pages" + chapter_text = f"{chapter_count} {chapter_label}" + + tags = [ + f"<>", + f"<>", + f"<>", + f"<>", + f"<>", + f"<>", + f"<>", + ] + + cover_path = _save_cover_to_cache(cover_bytes, cache_dir) + if cover_path: + tags.append(f"<>") + + return "\n".join(tags) + + +def _save_cover_to_cache( + cover_bytes: Optional[bytes], + cache_dir: Optional[str] = None, +) -> Optional[str]: + """Save cover image bytes to cache directory. + + Args: + cover_bytes: Raw image bytes (e.g. JPEG/PNG). + cache_dir: Directory to save to. If None, returns None. + + Returns: + Normalized path to saved cover file, or None on failure. + """ + if not cover_bytes: + return None + if cache_dir is None: + return None + + try: + cover_path = os.path.join(cache_dir, f"cover_{uuid.uuid4()}.jpg") + cover_path = os.path.normpath(cover_path) + with open(cover_path, "wb") as f: + f.write(cover_bytes) + return cover_path + except Exception as e: + logger.warning("Failed to save cover image: %s", e) + return None + + +def extract_book_metadata_epub(book: Any) -> Dict[str, Any]: + """Extract metadata from an opened ebooklib EPUB book. + + Args: + book: An opened ebooklib EPUB book object. + + Returns: + Dict with keys: title, authors, description, publisher, + publication_year, cover_image (bytes or None). + """ + import ebooklib + + metadata: Dict[str, Any] = { + "title": None, + "authors": [], + "description": None, + "cover_image": None, + "publisher": None, + "publication_year": None, + } + + try: + title_items = book.get_metadata("DC", "title") + if title_items and len(title_items) > 0: + metadata["title"] = title_items[0][0] + except Exception as e: + logger.warning("Error extracting title metadata: %s", e) + + try: + author_items = book.get_metadata("DC", "creator") + if author_items: + metadata["authors"] = [ + author[0] for author in author_items if len(author) > 0 + ] + except Exception as e: + logger.warning("Error extracting author metadata: %s", e) + + try: + desc_items = book.get_metadata("DC", "description") + if desc_items and len(desc_items) > 0: + metadata["description"] = desc_items[0][0] + except Exception as e: + logger.warning("Error extracting description metadata: %s", e) + + try: + publisher_items = book.get_metadata("DC", "publisher") + if publisher_items and len(publisher_items) > 0: + metadata["publisher"] = publisher_items[0][0] + except Exception as e: + logger.warning("Error extracting publisher metadata: %s", e) + + try: + date_items = book.get_metadata("DC", "date") + if date_items and len(date_items) > 0: + date_str = date_items[0][0] + year_match = re.search(r"\b(19|20)\d{2}\b", date_str) + if year_match: + metadata["publication_year"] = year_match.group(0) + else: + metadata["publication_year"] = date_str + except Exception as e: + logger.warning("Error extracting publication date metadata: %s", e) + + for item in book.get_items_of_type(ebooklib.ITEM_COVER): + metadata["cover_image"] = item.get_content() + break + + if not metadata["cover_image"]: + for item in book.get_items_of_type(ebooklib.ITEM_IMAGE): + if "cover" in item.get_name().lower(): + metadata["cover_image"] = item.get_content() + break + + return metadata + + +def extract_book_metadata_pdf(pdf_doc: Any) -> Dict[str, Any]: + """Extract metadata from an opened PyMuPDF document. + + Args: + pdf_doc: An opened fitz.Document object. + + Returns: + Dict with keys: title, authors, description, publisher, + publication_year, cover_image (bytes or None). + """ + metadata: Dict[str, Any] = { + "title": None, + "authors": [], + "description": None, + "cover_image": None, + "publisher": None, + "publication_year": None, + } + + pdf_info = pdf_doc.metadata + if pdf_info: + metadata["title"] = pdf_info.get("title", None) + author = pdf_info.get("author", None) + if author: + metadata["authors"] = [author] + metadata["description"] = pdf_info.get("subject", None) + keywords = pdf_info.get("keywords", None) + if keywords: + if metadata["description"]: + metadata["description"] += f"\n\nKeywords: {keywords}" + else: + metadata["description"] = f"Keywords: {keywords}" + metadata["publisher"] = pdf_info.get("creator", None) + + if "creationDate" in pdf_info: + date_str = pdf_info["creationDate"] + year_match = re.search(r"D:(\d{4})", date_str) + if year_match: + metadata["publication_year"] = year_match.group(1) + elif "modDate" in pdf_info: + date_str = pdf_info["modDate"] + year_match = re.search(r"D:(\d{4})", date_str) + if year_match: + metadata["publication_year"] = year_match.group(1) + + if len(pdf_doc) > 0: + try: + import fitz + pix = pdf_doc[0].get_pixmap(matrix=fitz.Matrix(2, 2)) + metadata["cover_image"] = pix.tobytes("png") + except Exception: + pass + + return metadata + + +def extract_book_metadata_markdown( + markdown_text: str, + markdown_toc: Optional[List[Dict[str, Any]]] = None, +) -> Dict[str, Any]: + """Extract metadata from markdown frontmatter and first heading. + + Args: + markdown_text: Raw markdown text content. + markdown_toc: Optional table of contents list (each item has + 'level' and 'name' keys). + + Returns: + Dict with keys: title, authors, description, publication_year. + cover_image is always None for markdown. + """ + metadata: Dict[str, Any] = { + "title": None, + "authors": [], + "description": None, + "cover_image": None, + "publisher": None, + "publication_year": None, + } + + if not markdown_text: + return metadata + + frontmatter_match = re.match( + r"^---\s*\n(.*?)\n---\s*\n", markdown_text, re.DOTALL + ) + if frontmatter_match: + try: + frontmatter = frontmatter_match.group(1) + title_match = re.search( + r"^title:\s*(.+)$", frontmatter, re.MULTILINE | re.IGNORECASE + ) + if title_match: + metadata["title"] = title_match.group(1).strip().strip("\"'") + + author_match = re.search( + r"^author:\s*(.+)$", frontmatter, re.MULTILINE | re.IGNORECASE + ) + if author_match: + metadata["authors"] = [ + author_match.group(1).strip().strip("\"'") + ] + + desc_match = re.search( + r"^description:\s*(.+)$", frontmatter, re.MULTILINE | re.IGNORECASE + ) + if desc_match: + metadata["description"] = ( + desc_match.group(1).strip().strip("\"'") + ) + + date_match = re.search( + r"^date:\s*(.+)$", frontmatter, re.MULTILINE | re.IGNORECASE + ) + if date_match: + date_str = date_match.group(1).strip().strip("\"'") + year_match = re.search(r"\b(19|20)\d{2}\b", date_str) + if year_match: + metadata["publication_year"] = year_match.group(0) + except Exception as e: + logger.warning("Error parsing markdown frontmatter: %s", e) + + if not metadata["title"] and markdown_toc: + first_h1 = next( + (h for h in markdown_toc if h.get("level") == 1), None + ) + if first_h1: + metadata["title"] = first_h1.get("name") + + return metadata diff --git a/abogen/pyqt/book_handler.py b/abogen/pyqt/book_handler.py index fc1d806..f2ea8c7 100644 --- a/abogen/pyqt/book_handler.py +++ b/abogen/pyqt/book_handler.py @@ -29,6 +29,12 @@ from abogen.utils import ( get_resource_path, ) from abogen.book_parser import get_book_parser +from abogen.domain.metadata_extraction import ( + extract_book_metadata_epub, + extract_book_metadata_pdf, + extract_book_metadata_markdown, + format_metadata_tags, +) from abogen.subtitle_utils import ( clean_text, @@ -948,169 +954,14 @@ class HandlerDialog(QDialog): self.previewEdit.setHtml(html_content) def _extract_book_metadata(self): - metadata = { - "title": None, - "authors": [], - "description": None, - "cover_image": None, - "publisher": None, - "publication_year": None, - } - if self.parser.file_type == "epub": - try: - title_items = self.book.get_metadata("DC", "title") - if title_items and len(title_items) > 0: - metadata["title"] = title_items[0][0] - except Exception as e: - logging.warning(f"Error extracting title metadata: {e}") - - try: - author_items = self.book.get_metadata("DC", "creator") - if author_items: - metadata["authors"] = [ - author[0] for author in author_items if len(author) > 0 - ] - except Exception as e: - logging.warning(f"Error extracting author metadata: {e}") - - try: - desc_items = self.book.get_metadata("DC", "description") - if desc_items and len(desc_items) > 0: - metadata["description"] = desc_items[0][0] - except Exception as e: - logging.warning(f"Error extracting description metadata: {e}") - - try: - publisher_items = self.book.get_metadata("DC", "publisher") - if publisher_items and len(publisher_items) > 0: - metadata["publisher"] = publisher_items[0][0] - except Exception as e: - logging.warning(f"Error extracting publisher metadata: {e}") - - # Try to extract publication year - try: - date_items = self.book.get_metadata("DC", "date") - if date_items and len(date_items) > 0: - date_str = date_items[0][0] - # Try to extract just the year from the date string - year_match = re.search(r"\b(19|20)\d{2}\b", date_str) - if year_match: - metadata["publication_year"] = year_match.group(0) - else: - metadata["publication_year"] = date_str - except Exception as e: - logging.warning(f"Error extracting publication date metadata: {e}") - - for item in self.book.get_items_of_type(ebooklib.ITEM_COVER): - metadata["cover_image"] = item.get_content() - break - - if not metadata["cover_image"]: - for item in self.book.get_items_of_type(ebooklib.ITEM_IMAGE): - if "cover" in item.get_name().lower(): - metadata["cover_image"] = item.get_content() - break + return extract_book_metadata_epub(self.book) elif self.parser.file_type == "markdown": - # Extract metadata from markdown frontmatter or first heading - if self.markdown_text: - # Try to extract YAML frontmatter - frontmatter_match = re.match( - r"^---\s*\n(.*?)\n---\s*\n", self.markdown_text, re.DOTALL - ) - if frontmatter_match: - try: - frontmatter = frontmatter_match.group(1) - # Simple YAML-like parsing for common fields - title_match = re.search( - r"^title:\s*(.+)$", - frontmatter, - re.MULTILINE | re.IGNORECASE, - ) - if title_match: - metadata["title"] = ( - title_match.group(1).strip().strip("\"'") - ) - - author_match = re.search( - r"^author:\s*(.+)$", - frontmatter, - re.MULTILINE | re.IGNORECASE, - ) - if author_match: - metadata["authors"] = [ - author_match.group(1).strip().strip("\"'") - ] - - desc_match = re.search( - r"^description:\s*(.+)$", - frontmatter, - re.MULTILINE | re.IGNORECASE, - ) - if desc_match: - metadata["description"] = ( - desc_match.group(1).strip().strip("\"'") - ) - - date_match = re.search( - r"^date:\s*(.+)$", frontmatter, re.MULTILINE | re.IGNORECASE - ) - if date_match: - date_str = date_match.group(1).strip().strip("\"'") - year_match = re.search(r"\b(19|20)\d{2}\b", date_str) - if year_match: - metadata["publication_year"] = year_match.group(0) - except Exception as e: - logging.warning(f"Error parsing markdown frontmatter: {e}") - - # Fallback: use first H1 header as title if no frontmatter title - if not metadata["title"] and self.markdown_toc: - # Find the first level 1 header - first_h1 = next( - (h for h in self.markdown_toc if h["level"] == 1), None - ) - if first_h1: - metadata["title"] = first_h1["name"] + return extract_book_metadata_markdown( + self.markdown_text, self.markdown_toc + ) else: - pdf_info = self.pdf_doc.metadata - if pdf_info: - metadata["title"] = pdf_info.get("title", None) - - author = pdf_info.get("author", None) - if author: - metadata["authors"] = [author] - - metadata["description"] = pdf_info.get("subject", None) - - keywords = pdf_info.get("keywords", None) - if keywords: - if metadata["description"]: - metadata["description"] += f"\n\nKeywords: {keywords}" - else: - metadata["description"] = f"Keywords: {keywords}" - - metadata["publisher"] = pdf_info.get("creator", None) - - # Try to extract publication date from PDF metadata - if "creationDate" in pdf_info: - date_str = pdf_info["creationDate"] - year_match = re.search(r"D:(\d{4})", date_str) - if year_match: - metadata["publication_year"] = year_match.group(1) - elif "modDate" in pdf_info: - date_str = pdf_info["modDate"] - year_match = re.search(r"D:(\d{4})", date_str) - if year_match: - metadata["publication_year"] = year_match.group(1) - - if len(self.pdf_doc) > 0: - try: - pix = self.pdf_doc[0].get_pixmap(matrix=fitz.Matrix(2, 2)) - metadata["cover_image"] = pix.tobytes("png") - except Exception: - pass - - return metadata + return extract_book_metadata_pdf(self.pdf_doc) def get_selected_text(self): # If a background loader thread is running, wait for it to finish to @@ -1136,59 +987,21 @@ class HandlerDialog(QDialog): def _format_metadata_tags(self): """Format metadata tags for insertion at the beginning of the text""" - import datetime from abogen.utils import get_user_cache_path - metadata = self.book_metadata filename = os.path.splitext(os.path.basename(self.book_path))[0] - current_year = str(datetime.datetime.now().year) + chapter_count = len(self.checked_chapters) + cache_dir = get_user_cache_path() - # Get values with fallbacks - title = metadata.get("title") or filename - authors = metadata.get("authors") or ["Unknown"] - authors_text = ", ".join(authors) - album_artist = authors_text or "Unknown" - year = ( - metadata.get("publication_year") or current_year - ) # Use publication year if available - - # Count chapters/pages - total_chapters = len(self.checked_chapters) - chapter_text = ( - f"{total_chapters} {'Chapters' if self.parser.file_type == 'epub' else 'Pages'}" + return format_metadata_tags( + self.book_metadata, + filename, + chapter_count, + self.parser.file_type, + cover_bytes=self.book_metadata.get("cover_image"), + cache_dir=cache_dir, ) - # Handle cover image - cover_tag = "" - if metadata.get("cover_image"): - try: - import uuid - - cache_dir = get_user_cache_path() - cover_path = os.path.join(cache_dir, f"cover_{uuid.uuid4()}.jpg") - cover_path = os.path.normpath(cover_path) - with open(cover_path, "wb") as f: - f.write(metadata["cover_image"]) - cover_tag = f"<>" - except Exception as e: - logging.warning(f"Failed to save cover image: {e}") - - # Format metadata tags - metadata_tags = [ - f"<>", - f"<>", - f"<>", - f"<>", - f"<>", - f"<>", - f"<>", - ] - - if cover_tag: - metadata_tags.append(cover_tag) - - return "\n".join(metadata_tags) - def _get_markdown_selected_text(self): """Get selected text from markdown chapters""" all_checked_identifiers = set() diff --git a/tests/test_domain_metadata_book.py b/tests/test_domain_metadata_book.py new file mode 100644 index 0000000..686847a --- /dev/null +++ b/tests/test_domain_metadata_book.py @@ -0,0 +1,114 @@ +"""Tests for domain/metadata_extraction.py — format_metadata_tags, extract_book_metadata_*.""" + +import os + +import pytest + +from abogen.domain.metadata_extraction import ( + extract_book_metadata_markdown, + format_metadata_tags, + _save_cover_to_cache, +) + + +class TestFormatMetadataTags: + def test_basic_epub(self): + metadata = { + "title": "My Book", + "authors": ["Author One", "Author Two"], + "publication_year": "2023", + } + result = format_metadata_tags(metadata, "fallback", 10, "epub") + assert "<>" in result + assert "<>" in result + assert "<>" in result + assert "<>" in result + assert "<>" in result + + def test_pdf_uses_pages(self): + metadata = {"title": "PDF Doc", "authors": ["Writer"]} + result = format_metadata_tags(metadata, "doc", 50, "pdf") + assert "50 Pages" in result + + def test_markdown_uses_chapters(self): + metadata = {"title": "MD Doc"} + result = format_metadata_tags(metadata, "doc", 3, "markdown") + assert "3 Chapters" in result + + def test_fallback_title(self): + metadata = {} + result = format_metadata_tags(metadata, "fallback_name", 1, "epub") + assert "<>" in result + + def test_unknown_authors(self): + metadata = {"authors": []} + result = format_metadata_tags(metadata, "file", 1, "epub") + assert "<>" in result + + def test_cover_bytes_saved(self, tmp_path): + metadata = {"title": "With Cover"} + cover = b"\x89PNG\r\n\x1a\n" + b"\x00" * 100 # Fake image bytes + result = format_metadata_tags( + metadata, "file", 1, "epub", + cover_bytes=cover, cache_dir=str(tmp_path), + ) + assert "<>" in result + + +class TestSaveCoverToCache: + def test_saves_file(self, tmp_path): + data = b"\x89PNG" + b"\x00" * 50 + result = _save_cover_to_cache(data, str(tmp_path)) + assert result is not None + assert os.path.exists(result) + assert open(result, "rb").read() == data + + def test_none_bytes(self, tmp_path): + assert _save_cover_to_cache(None, str(tmp_path)) is None + + def test_none_cache_dir(self): + assert _save_cover_to_cache(b"data", None) is None + + def test_returns_normalized_path(self, tmp_path): + result = _save_cover_to_cache(b"data", str(tmp_path)) + assert result == os.path.normpath(result) + + +class TestExtractBookMetadataMarkdown: + def test_frontmatter(self): + text = "---\ntitle: Test Title\nauthor: Test Author\ndate: 2024\n---\n\nBody" + result = extract_book_metadata_markdown(text) + assert result["title"] == "Test Title" + assert result["authors"] == ["Test Author"] + assert result["publication_year"] == "2024" + + def test_fallback_to_h1(self, ): + text = "# My Heading\n\nSome content" + toc = [{"level": 1, "name": "My Heading"}] + result = extract_book_metadata_markdown(text, toc) + assert result["title"] == "My Heading" + + def test_empty_text(self): + result = extract_book_metadata_markdown("") + assert result["title"] is None + assert result["authors"] == [] + + def test_frontmatter_with_quotes(self): + text = '---\ntitle: "Quoted Title"\nauthor: \'Quoted Author\'\n---\n\nBody' + result = extract_book_metadata_markdown(text) + assert result["title"] == "Quoted Title" + assert result["authors"] == ["Quoted Author"]