From e650d17b07cceb660ef8686fd1f122d50fa3a05c Mon Sep 17 00:00:00 2001 From: historia Date: Tue, 18 Aug 2026 02:17:39 -0400 Subject: refactor(converter): deduplicate epub extraction --- converter/extractors.py | 92 +++++-------------------------------------------- 1 file changed, 9 insertions(+), 83 deletions(-) (limited to 'converter') diff --git a/converter/extractors.py b/converter/extractors.py index 9139626..cd270a1 100644 --- a/converter/extractors.py +++ b/converter/extractors.py @@ -43,12 +43,11 @@ def extract_sections(file_path: Path) -> List[Section]: one chapter at a time. TXT and PDF files have no chapter structure and always yield a single section. """ - extension = file_path.suffix.lower() - if extension == ".epub": + if file_path.suffix.lower() == ".epub": chapters = _extract_epub_chapters(file_path) - if len(chapters) > 1: - return chapters - return [Section(file_path.stem, extract_epub(file_path))] + if not chapters: + raise RuntimeError("All EPUB extraction methods failed") + return chapters return [Section(file_path.stem, extract_text(file_path))] @@ -199,49 +198,11 @@ def clean_html(html_content: str) -> str: def extract_epub(file_path: Path) -> str: - """Extract text from EPUB, trying several methods in order.""" - methods = [ - _extract_epub_ebooklib, - _extract_epub_zipfile, - _extract_epub_manual, - ] - - for method in methods: - try: - text = method(file_path) - if text and text.strip(): - logger.info("EPUB extraction successful (%s): %d characters", method.__name__, len(text)) - return text - except Exception as exc: - logger.warning("EPUB method %s failed: %s", method.__name__, exc) - - raise RuntimeError("All EPUB extraction methods failed") - - -def _extract_epub_ebooklib(file_path: Path) -> str: - """Extract using ebooklib, following the spine (reading) order.""" - import ebooklib - from ebooklib import epub - - book = epub.read_epub(str(file_path)) - text_parts = [] - - for entry in book.spine: - item_id = entry[0] if isinstance(entry, (tuple, list)) else entry - try: - item = book.get_item_with_id(item_id) - if item and item.get_type() == ebooklib.ITEM_DOCUMENT: - content = item.get_body_content() - if content: - if isinstance(content, bytes): - content = content.decode("utf-8", errors="ignore") - cleaned = clean_html(str(content)) - if cleaned.strip(): - text_parts.append(cleaned) - except Exception as exc: - logger.debug("Skipping EPUB spine item %r: %s", item_id, exc) - - return "\n\n".join(text_parts) + """Extract the book's text from EPUB, trying several methods in order.""" + chapters = _extract_epub_chapters(file_path) + if not chapters: + raise RuntimeError("All EPUB extraction methods failed") + return "\n\n".join(section.text for section in chapters) def _natural_key(name: str): @@ -250,41 +211,6 @@ def _natural_key(name: str): for part in re.split(r"(\d+)", name)] -def _extract_epub_zipfile(file_path: Path) -> str: - """Extract by parsing HTML members of the EPUB zip directly.""" - text_parts = [] - with zipfile.ZipFile(file_path, "r") as epub_zip: - for file_name in sorted(epub_zip.namelist(), key=_natural_key): - if file_name.lower().endswith((".html", ".xhtml", ".htm")): - try: - content = epub_zip.read(file_name).decode("utf-8", errors="ignore") - cleaned = clean_html(content) - if cleaned.strip(): - text_parts.append(cleaned) - except Exception as exc: - logger.debug("Skipping EPUB member %r: %s", file_name, exc) - return "\n\n".join(text_parts) - - -def _extract_epub_manual(file_path: Path) -> str: - """Last-resort extraction from any markup-looking EPUB member.""" - skipped_extensions = (".jpg", ".jpeg", ".png", ".gif", ".css", ".js") - text_parts = [] - with zipfile.ZipFile(file_path, "r") as epub_zip: - for file_name in sorted(epub_zip.namelist(), key=_natural_key): - if file_name.lower().endswith(skipped_extensions): - continue - try: - content = epub_zip.read(file_name).decode("utf-8", errors="ignore") - if "<" in content and len(content.strip()) > 100: - cleaned = clean_html(content) - if cleaned: - text_parts.append(cleaned) - except Exception as exc: - logger.debug("Skipping EPUB member %r: %s", file_name, exc) - return "\n\n".join(text_parts) - - def _extract_txt(file_path: Path) -> str: """Extract from TXT, handling BOMs and common encodings (latin-1 is the catch-all). -- cgit v1.2.3