aboutsummaryrefslogtreecommitdiff
path: root/converter/extractors.py
diff options
context:
space:
mode:
Diffstat (limited to 'converter/extractors.py')
-rw-r--r--converter/extractors.py185
1 files changed, 185 insertions, 0 deletions
diff --git a/converter/extractors.py b/converter/extractors.py
new file mode 100644
index 0000000..e49b48a
--- /dev/null
+++ b/converter/extractors.py
@@ -0,0 +1,185 @@
+"""Text extraction from book files (TXT, PDF, EPUB) and text/HTML cleaning."""
+
+import logging
+import re
+import zipfile
+from html import unescape
+from pathlib import Path
+
+try:
+ from bs4 import BeautifulSoup
+ BS4_AVAILABLE = True
+except ImportError:
+ BS4_AVAILABLE = False
+
+logger = logging.getLogger(__name__)
+
+
+def extract_text(file_path: Path) -> str:
+ """Extract text from a book file based on its extension."""
+ extension = file_path.suffix.lower()
+ if extension == ".txt":
+ return _extract_txt(file_path)
+ if extension == ".pdf":
+ return _extract_pdf(file_path)
+ if extension == ".epub":
+ return extract_epub(file_path)
+ raise ValueError(f"Unsupported file format: {extension}")
+
+
+def clean_text(text: str) -> str:
+ """Normalize whitespace and strip standalone page numbers.
+
+ Page numbers are removed only when they appear as a short number alone on
+ its own line (before whitespace collapsing), so inline numbers like
+ "42 years", "1,000" or "3.5" are preserved.
+ """
+ if not text:
+ return ""
+ # Standalone page numbers (digits alone on a line) must go BEFORE the
+ # newline-collapsing step below.
+ text = re.sub(r"(?m)^\s*\d{1,4}\s*$", " ", text)
+ text = re.sub(r"\s+", " ", text)
+ return text.strip()
+
+
+def clean_html(html_content: str) -> str:
+ """Strip markup, scripts and styles from HTML content."""
+ if not html_content:
+ return ""
+
+ if BS4_AVAILABLE:
+ try:
+ soup = BeautifulSoup(html_content, "html.parser")
+ for tag in soup(["script", "style"]):
+ tag.decompose()
+ text = soup.get_text()
+ lines = (line.strip() for line in text.splitlines())
+ chunks = (phrase.strip() for line in lines for phrase in line.split(" "))
+ return " ".join(chunk for chunk in chunks if chunk)
+ except Exception as exc:
+ logger.debug("BeautifulSoup cleaning failed, falling back to regex: %s", exc)
+
+ # Fallback regex cleaning
+ html_content = re.sub(r"<style[^>]*>.*?</style>", "", html_content, flags=re.DOTALL | re.IGNORECASE)
+ html_content = re.sub(r"<script[^>]*>.*?</script>", "", html_content, flags=re.DOTALL | re.IGNORECASE)
+ html_content = re.sub(r"<[^>]+>", " ", html_content)
+ html_content = unescape(html_content)
+ html_content = re.sub(r"\s+", " ", html_content)
+ return html_content.strip()
+
+
+def extract_epub(file_path: Path) -> str:
+ """Extract text from EPUB, trying several methods in order."""
+ methods = [
+ _extract_epub_ebooklib,
+ _extract_epub_zipfile,
+ _extract_epub_manual,
+ ]
+
+ for method in methods:
+ try:
+ text = method(file_path)
+ if text and text.strip():
+ logger.info("EPUB extraction successful (%s): %d characters", method.__name__, len(text))
+ return text
+ except Exception as exc:
+ logger.warning("EPUB method %s failed: %s", method.__name__, exc)
+
+ raise RuntimeError("All EPUB extraction methods failed")
+
+
+def _extract_epub_ebooklib(file_path: Path) -> str:
+ """Extract using ebooklib, following the spine (reading) order."""
+ import ebooklib
+ from ebooklib import epub
+
+ book = epub.read_epub(str(file_path))
+ text_parts = []
+
+ for entry in book.spine:
+ item_id = entry[0] if isinstance(entry, (tuple, list)) else entry
+ try:
+ item = book.get_item_with_id(item_id)
+ if item and item.get_type() == ebooklib.ITEM_DOCUMENT:
+ content = item.get_body_content()
+ if content:
+ if isinstance(content, bytes):
+ content = content.decode("utf-8", errors="ignore")
+ cleaned = clean_html(str(content))
+ if cleaned.strip():
+ text_parts.append(cleaned)
+ except Exception as exc:
+ logger.debug("Skipping EPUB spine item %r: %s", item_id, exc)
+
+ return "\n\n".join(text_parts)
+
+
+def _extract_epub_zipfile(file_path: Path) -> str:
+ """Extract by parsing HTML members of the EPUB zip directly."""
+ text_parts = []
+ with zipfile.ZipFile(file_path, "r") as epub_zip:
+ for file_name in sorted(epub_zip.namelist()):
+ if file_name.lower().endswith((".html", ".xhtml", ".htm")):
+ try:
+ content = epub_zip.read(file_name).decode("utf-8", errors="ignore")
+ cleaned = clean_html(content)
+ if cleaned.strip():
+ text_parts.append(cleaned)
+ except Exception as exc:
+ logger.debug("Skipping EPUB member %r: %s", file_name, exc)
+ return "\n\n".join(text_parts)
+
+
+def _extract_epub_manual(file_path: Path) -> str:
+ """Last-resort extraction from any markup-looking EPUB member."""
+ skipped_extensions = (".jpg", ".jpeg", ".png", ".gif", ".css", ".js")
+ text_parts = []
+ with zipfile.ZipFile(file_path, "r") as epub_zip:
+ for file_name in sorted(epub_zip.namelist()):
+ if file_name.lower().endswith(skipped_extensions):
+ continue
+ try:
+ content = epub_zip.read(file_name).decode("utf-8", errors="ignore")
+ if "<" in content and len(content.strip()) > 100:
+ cleaned = clean_html(content)
+ if cleaned:
+ text_parts.append(cleaned)
+ except Exception as exc:
+ logger.debug("Skipping EPUB member %r: %s", file_name, exc)
+ return "\n\n".join(text_parts)
+
+
+def _extract_txt(file_path: Path) -> str:
+ """Extract from TXT, trying common encodings (latin-1 is the catch-all)."""
+ for encoding in ("utf-8", "utf-16", "cp1252", "latin-1"):
+ try:
+ with open(file_path, "r", encoding=encoding) as f:
+ return clean_text(f.read())
+ except UnicodeError:
+ continue
+ raise ValueError(f"Could not decode text file: {file_path}")
+
+
+def _extract_pdf(file_path: Path) -> str:
+ """Extract from PDF."""
+ from pypdf import PdfReader
+
+ text = ""
+ with open(file_path, "rb") as file:
+ pdf_reader = PdfReader(file)
+ total_pages = len(pdf_reader.pages)
+ logger.info("PDF has %d pages", total_pages)
+
+ for page_num, page in enumerate(pdf_reader.pages, 1):
+ try:
+ page_text = page.extract_text() or ""
+ if page_text.strip():
+ text += f"\n\n{page_text}"
+ if page_num % 10 == 0:
+ logger.debug("Extracted %d/%d pages", page_num, total_pages)
+ except Exception as exc:
+ logger.warning("Failed to extract page %d: %s", page_num, exc)
+
+ logger.info("Extracted text from %d pages, %d characters total", total_pages, len(text))
+ return clean_text(text)