From a30bd534151f757dad38763ae479fcd465d2ba0d Mon Sep 17 00:00:00 2001 From: historia Date: Mon, 17 Aug 2026 18:32:14 -0400 Subject: refactor(converter): extract logic into package --- converter/extractors.py | 185 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 185 insertions(+) create mode 100644 converter/extractors.py (limited to 'converter/extractors.py') diff --git a/converter/extractors.py b/converter/extractors.py new file mode 100644 index 0000000..e49b48a --- /dev/null +++ b/converter/extractors.py @@ -0,0 +1,185 @@ +"""Text extraction from book files (TXT, PDF, EPUB) and text/HTML cleaning.""" + +import logging +import re +import zipfile +from html import unescape +from pathlib import Path + +try: + from bs4 import BeautifulSoup + BS4_AVAILABLE = True +except ImportError: + BS4_AVAILABLE = False + +logger = logging.getLogger(__name__) + + +def extract_text(file_path: Path) -> str: + """Extract text from a book file based on its extension.""" + extension = file_path.suffix.lower() + if extension == ".txt": + return _extract_txt(file_path) + if extension == ".pdf": + return _extract_pdf(file_path) + if extension == ".epub": + return extract_epub(file_path) + raise ValueError(f"Unsupported file format: {extension}") + + +def clean_text(text: str) -> str: + """Normalize whitespace and strip standalone page numbers. + + Page numbers are removed only when they appear as a short number alone on + its own line (before whitespace collapsing), so inline numbers like + "42 years", "1,000" or "3.5" are preserved. + """ + if not text: + return "" + # Standalone page numbers (digits alone on a line) must go BEFORE the + # newline-collapsing step below. + text = re.sub(r"(?m)^\s*\d{1,4}\s*$", " ", text) + text = re.sub(r"\s+", " ", text) + return text.strip() + + +def clean_html(html_content: str) -> str: + """Strip markup, scripts and styles from HTML content.""" + if not html_content: + return "" + + if BS4_AVAILABLE: + try: + soup = BeautifulSoup(html_content, "html.parser") + for tag in soup(["script", "style"]): + tag.decompose() + text = soup.get_text() + lines = (line.strip() for line in text.splitlines()) + chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) + return " ".join(chunk for chunk in chunks if chunk) + except Exception as exc: + logger.debug("BeautifulSoup cleaning failed, falling back to regex: %s", exc) + + # Fallback regex cleaning + html_content = re.sub(r"]*>.*?", "", html_content, flags=re.DOTALL | re.IGNORECASE) + html_content = re.sub(r"]*>.*?", "", html_content, flags=re.DOTALL | re.IGNORECASE) + html_content = re.sub(r"<[^>]+>", " ", html_content) + html_content = unescape(html_content) + html_content = re.sub(r"\s+", " ", html_content) + return html_content.strip() + + +def extract_epub(file_path: Path) -> str: + """Extract text from EPUB, trying several methods in order.""" + methods = [ + _extract_epub_ebooklib, + _extract_epub_zipfile, + _extract_epub_manual, + ] + + for method in methods: + try: + text = method(file_path) + if text and text.strip(): + logger.info("EPUB extraction successful (%s): %d characters", method.__name__, len(text)) + return text + except Exception as exc: + logger.warning("EPUB method %s failed: %s", method.__name__, exc) + + raise RuntimeError("All EPUB extraction methods failed") + + +def _extract_epub_ebooklib(file_path: Path) -> str: + """Extract using ebooklib, following the spine (reading) order.""" + import ebooklib + from ebooklib import epub + + book = epub.read_epub(str(file_path)) + text_parts = [] + + for entry in book.spine: + item_id = entry[0] if isinstance(entry, (tuple, list)) else entry + try: + item = book.get_item_with_id(item_id) + if item and item.get_type() == ebooklib.ITEM_DOCUMENT: + content = item.get_body_content() + if content: + if isinstance(content, bytes): + content = content.decode("utf-8", errors="ignore") + cleaned = clean_html(str(content)) + if cleaned.strip(): + text_parts.append(cleaned) + except Exception as exc: + logger.debug("Skipping EPUB spine item %r: %s", item_id, exc) + + return "\n\n".join(text_parts) + + +def _extract_epub_zipfile(file_path: Path) -> str: + """Extract by parsing HTML members of the EPUB zip directly.""" + text_parts = [] + with zipfile.ZipFile(file_path, "r") as epub_zip: + for file_name in sorted(epub_zip.namelist()): + if file_name.lower().endswith((".html", ".xhtml", ".htm")): + try: + content = epub_zip.read(file_name).decode("utf-8", errors="ignore") + cleaned = clean_html(content) + if cleaned.strip(): + text_parts.append(cleaned) + except Exception as exc: + logger.debug("Skipping EPUB member %r: %s", file_name, exc) + return "\n\n".join(text_parts) + + +def _extract_epub_manual(file_path: Path) -> str: + """Last-resort extraction from any markup-looking EPUB member.""" + skipped_extensions = (".jpg", ".jpeg", ".png", ".gif", ".css", ".js") + text_parts = [] + with zipfile.ZipFile(file_path, "r") as epub_zip: + for file_name in sorted(epub_zip.namelist()): + if file_name.lower().endswith(skipped_extensions): + continue + try: + content = epub_zip.read(file_name).decode("utf-8", errors="ignore") + if "<" in content and len(content.strip()) > 100: + cleaned = clean_html(content) + if cleaned: + text_parts.append(cleaned) + except Exception as exc: + logger.debug("Skipping EPUB member %r: %s", file_name, exc) + return "\n\n".join(text_parts) + + +def _extract_txt(file_path: Path) -> str: + """Extract from TXT, trying common encodings (latin-1 is the catch-all).""" + for encoding in ("utf-8", "utf-16", "cp1252", "latin-1"): + try: + with open(file_path, "r", encoding=encoding) as f: + return clean_text(f.read()) + except UnicodeError: + continue + raise ValueError(f"Could not decode text file: {file_path}") + + +def _extract_pdf(file_path: Path) -> str: + """Extract from PDF.""" + from pypdf import PdfReader + + text = "" + with open(file_path, "rb") as file: + pdf_reader = PdfReader(file) + total_pages = len(pdf_reader.pages) + logger.info("PDF has %d pages", total_pages) + + for page_num, page in enumerate(pdf_reader.pages, 1): + try: + page_text = page.extract_text() or "" + if page_text.strip(): + text += f"\n\n{page_text}" + if page_num % 10 == 0: + logger.debug("Extracted %d/%d pages", page_num, total_pages) + except Exception as exc: + logger.warning("Failed to extract page %d: %s", page_num, exc) + + logger.info("Extracted text from %d pages, %d characters total", total_pages, len(text)) + return clean_text(text) -- cgit v1.2.3