aboutsummaryrefslogtreecommitdiff
path: root/tests/test_extractors.py
blob: 09b688e05f639f0ba5e1da869b7f37ec0d4d2274 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
"""Tests for file text extraction."""

import tempfile
import unittest
from pathlib import Path

from converter.extractors import extract_text


class TxtExtractionTests(unittest.TestCase):
    def _extract(self, data: bytes) -> str:
        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.txt"
            path.write_bytes(data)
            return extract_text(path)

    def test_utf8(self):
        self.assertEqual(self._extract("héllo wörld".encode("utf-8")), "héllo wörld")

    def test_utf16_with_bom(self):
        self.assertEqual(self._extract("héllo".encode("utf-16")), "héllo")

    def test_cp1252(self):
        self.assertEqual(self._extract("“quotes”".encode("cp1252")), "“quotes”")

    def test_latin1_fallback(self):
        # 0x81 is undefined in cp1252, forcing the latin-1 catch-all
        self.assertEqual(self._extract(b"caf\x81"), "caf\x81")

    def test_unsupported_format(self):
        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.xyz"
            path.write_bytes(b"data")
            with self.assertRaises(ValueError):
                extract_text(path)


def _build_test_epub(path: Path) -> None:
    from ebooklib import epub

    book = epub.EpubBook()
    book.set_identifier("test-id")
    book.set_title("Test Book")
    book.set_language("en")

    chapter1 = epub.EpubHtml(title="One", file_name="chap1.xhtml", lang="en")
    chapter1.content = "<html><body><p>First chapter text.</p></body></html>"
    chapter2 = epub.EpubHtml(title="Two", file_name="chap2.xhtml", lang="en")
    chapter2.content = "<html><body><p>Second chapter text.</p></body></html>"

    book.add_item(chapter1)
    book.add_item(chapter2)
    book.toc = (chapter1, chapter2)
    book.spine = ["nav", chapter1, chapter2]
    book.add_item(epub.EpubNcx())
    book.add_item(epub.EpubNav())

    epub.write_epub(str(path), book)


class EpubExtractionTests(unittest.TestCase):
    def setUp(self):
        try:
            import ebooklib  # noqa: F401
        except ImportError:
            self.skipTest("ebooklib not installed")

    def test_ebooklib_extraction(self):
        # Regression test: the ebooklib path used to silently return "" due to
        # isinstance(item, ebooklib.ITEM_DOCUMENT) (an int, not a class).
        from converter.extractors import _extract_epub_ebooklib

        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.epub"
            _build_test_epub(path)
            text = _extract_epub_ebooklib(path)

        self.assertIn("First chapter text.", text)
        self.assertIn("Second chapter text.", text)

    def test_epub_extraction_follows_spine_order(self):
        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.epub"
            _build_test_epub(path)
            text = extract_text(path)

        self.assertIn("First chapter text.", text)
        self.assertIn("Second chapter text.", text)
        self.assertLess(text.index("First chapter text."),
                        text.index("Second chapter text."))


class ExtractSectionsTests(unittest.TestCase):
    def setUp(self):
        try:
            import ebooklib  # noqa: F401
        except ImportError:
            self.skipTest("ebooklib not installed")

    def test_epub_sections_split_on_chapters(self):
        from converter.extractors import extract_sections

        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.epub"
            _build_test_epub(path)
            sections = extract_sections(path)

        self.assertEqual(len(sections), 2)
        self.assertEqual(sections[0].title, "One")
        self.assertEqual(sections[1].title, "Two")
        self.assertIn("First chapter text.", sections[0].text)
        self.assertIn("Second chapter text.", sections[1].text)

    def test_txt_is_single_section(self):
        from converter.extractors import extract_sections

        with tempfile.TemporaryDirectory() as tmp:
            path = Path(tmp) / "book.txt"
            path.write_text("Hello world.", encoding="utf-8")
            sections = extract_sections(path)

        self.assertEqual(len(sections), 1)
        self.assertEqual(sections[0].title, "book")
        self.assertEqual(sections[0].text, "Hello world.")


if __name__ == "__main__":
    unittest.main()