aboutsummaryrefslogtreecommitdiff
path: root/tests/test_extractors.py
diff options
context:
space:
mode:
Diffstat (limited to 'tests/test_extractors.py')
-rw-r--r--tests/test_extractors.py94
1 files changed, 94 insertions, 0 deletions
diff --git a/tests/test_extractors.py b/tests/test_extractors.py
new file mode 100644
index 0000000..c497267
--- /dev/null
+++ b/tests/test_extractors.py
@@ -0,0 +1,94 @@
+"""Tests for file text extraction."""
+
+import tempfile
+import unittest
+from pathlib import Path
+
+from converter.extractors import extract_text
+
+
+class TxtExtractionTests(unittest.TestCase):
+ def _extract(self, data: bytes) -> str:
+ with tempfile.TemporaryDirectory() as tmp:
+ path = Path(tmp) / "book.txt"
+ path.write_bytes(data)
+ return extract_text(path)
+
+ def test_utf8(self):
+ self.assertEqual(self._extract("héllo wörld".encode("utf-8")), "héllo wörld")
+
+ def test_utf16_with_bom(self):
+ self.assertEqual(self._extract("héllo".encode("utf-16")), "héllo")
+
+ def test_cp1252(self):
+ self.assertEqual(self._extract("“quotes”".encode("cp1252")), "“quotes”")
+
+ def test_latin1_fallback(self):
+ # 0x81 is undefined in cp1252, forcing the latin-1 catch-all
+ self.assertEqual(self._extract(b"caf\x81"), "caf\x81")
+
+ def test_unsupported_format(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ path = Path(tmp) / "book.xyz"
+ path.write_bytes(b"data")
+ with self.assertRaises(ValueError):
+ extract_text(path)
+
+
+def _build_test_epub(path: Path) -> None:
+ from ebooklib import epub
+
+ book = epub.EpubBook()
+ book.set_identifier("test-id")
+ book.set_title("Test Book")
+ book.set_language("en")
+
+ chapter1 = epub.EpubHtml(title="One", file_name="chap1.xhtml", lang="en")
+ chapter1.content = "<html><body><p>First chapter text.</p></body></html>"
+ chapter2 = epub.EpubHtml(title="Two", file_name="chap2.xhtml", lang="en")
+ chapter2.content = "<html><body><p>Second chapter text.</p></body></html>"
+
+ book.add_item(chapter1)
+ book.add_item(chapter2)
+ book.toc = (chapter1, chapter2)
+ book.spine = ["nav", chapter1, chapter2]
+ book.add_item(epub.EpubNcx())
+ book.add_item(epub.EpubNav())
+
+ epub.write_epub(str(path), book)
+
+
+class EpubExtractionTests(unittest.TestCase):
+ def setUp(self):
+ try:
+ import ebooklib # noqa: F401
+ except ImportError:
+ self.skipTest("ebooklib not installed")
+
+ def test_ebooklib_extraction(self):
+ # Regression test: the ebooklib path used to silently return "" due to
+ # isinstance(item, ebooklib.ITEM_DOCUMENT) (an int, not a class).
+ from converter.extractors import _extract_epub_ebooklib
+
+ with tempfile.TemporaryDirectory() as tmp:
+ path = Path(tmp) / "book.epub"
+ _build_test_epub(path)
+ text = _extract_epub_ebooklib(path)
+
+ self.assertIn("First chapter text.", text)
+ self.assertIn("Second chapter text.", text)
+
+ def test_epub_extraction_follows_spine_order(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ path = Path(tmp) / "book.epub"
+ _build_test_epub(path)
+ text = extract_text(path)
+
+ self.assertIn("First chapter text.", text)
+ self.assertIn("Second chapter text.", text)
+ self.assertLess(text.index("First chapter text."),
+ text.index("Second chapter text."))
+
+
+if __name__ == "__main__":
+ unittest.main()