aboutsummaryrefslogtreecommitdiff
path: root/converter
diff options
context:
space:
mode:
Diffstat (limited to 'converter')
-rw-r--r--converter/extractors.py92
1 files changed, 9 insertions, 83 deletions
diff --git a/converter/extractors.py b/converter/extractors.py
index 9139626..cd270a1 100644
--- a/converter/extractors.py
+++ b/converter/extractors.py
@@ -43,12 +43,11 @@ def extract_sections(file_path: Path) -> List[Section]:
one chapter at a time. TXT and PDF files have no chapter structure and
always yield a single section.
"""
- extension = file_path.suffix.lower()
- if extension == ".epub":
+ if file_path.suffix.lower() == ".epub":
chapters = _extract_epub_chapters(file_path)
- if len(chapters) > 1:
- return chapters
- return [Section(file_path.stem, extract_epub(file_path))]
+ if not chapters:
+ raise RuntimeError("All EPUB extraction methods failed")
+ return chapters
return [Section(file_path.stem, extract_text(file_path))]
@@ -199,49 +198,11 @@ def clean_html(html_content: str) -> str:
def extract_epub(file_path: Path) -> str:
- """Extract text from EPUB, trying several methods in order."""
- methods = [
- _extract_epub_ebooklib,
- _extract_epub_zipfile,
- _extract_epub_manual,
- ]
-
- for method in methods:
- try:
- text = method(file_path)
- if text and text.strip():
- logger.info("EPUB extraction successful (%s): %d characters", method.__name__, len(text))
- return text
- except Exception as exc:
- logger.warning("EPUB method %s failed: %s", method.__name__, exc)
-
- raise RuntimeError("All EPUB extraction methods failed")
-
-
-def _extract_epub_ebooklib(file_path: Path) -> str:
- """Extract using ebooklib, following the spine (reading) order."""
- import ebooklib
- from ebooklib import epub
-
- book = epub.read_epub(str(file_path))
- text_parts = []
-
- for entry in book.spine:
- item_id = entry[0] if isinstance(entry, (tuple, list)) else entry
- try:
- item = book.get_item_with_id(item_id)
- if item and item.get_type() == ebooklib.ITEM_DOCUMENT:
- content = item.get_body_content()
- if content:
- if isinstance(content, bytes):
- content = content.decode("utf-8", errors="ignore")
- cleaned = clean_html(str(content))
- if cleaned.strip():
- text_parts.append(cleaned)
- except Exception as exc:
- logger.debug("Skipping EPUB spine item %r: %s", item_id, exc)
-
- return "\n\n".join(text_parts)
+ """Extract the book's text from EPUB, trying several methods in order."""
+ chapters = _extract_epub_chapters(file_path)
+ if not chapters:
+ raise RuntimeError("All EPUB extraction methods failed")
+ return "\n\n".join(section.text for section in chapters)
def _natural_key(name: str):
@@ -250,41 +211,6 @@ def _natural_key(name: str):
for part in re.split(r"(\d+)", name)]
-def _extract_epub_zipfile(file_path: Path) -> str:
- """Extract by parsing HTML members of the EPUB zip directly."""
- text_parts = []
- with zipfile.ZipFile(file_path, "r") as epub_zip:
- for file_name in sorted(epub_zip.namelist(), key=_natural_key):
- if file_name.lower().endswith((".html", ".xhtml", ".htm")):
- try:
- content = epub_zip.read(file_name).decode("utf-8", errors="ignore")
- cleaned = clean_html(content)
- if cleaned.strip():
- text_parts.append(cleaned)
- except Exception as exc:
- logger.debug("Skipping EPUB member %r: %s", file_name, exc)
- return "\n\n".join(text_parts)
-
-
-def _extract_epub_manual(file_path: Path) -> str:
- """Last-resort extraction from any markup-looking EPUB member."""
- skipped_extensions = (".jpg", ".jpeg", ".png", ".gif", ".css", ".js")
- text_parts = []
- with zipfile.ZipFile(file_path, "r") as epub_zip:
- for file_name in sorted(epub_zip.namelist(), key=_natural_key):
- if file_name.lower().endswith(skipped_extensions):
- continue
- try:
- content = epub_zip.read(file_name).decode("utf-8", errors="ignore")
- if "<" in content and len(content.strip()) > 100:
- cleaned = clean_html(content)
- if cleaned:
- text_parts.append(cleaned)
- except Exception as exc:
- logger.debug("Skipping EPUB member %r: %s", file_name, exc)
- return "\n\n".join(text_parts)
-
-
def _extract_txt(file_path: Path) -> str:
"""Extract from TXT, handling BOMs and common encodings (latin-1 is the catch-all).