Skip to content

Parsers

EpubParser

Source code in tts_studio/parsers.py
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
class EpubParser:
    @staticmethod
    def _extract_paragraphs(soup):
        """
        Layered paragraph extraction:
        1. <p> tags
        2. block-level containers
        3. newline split fallback

        Uses an empty separator so that words split across inline tags
        (e.g. ``well-<span>known</span>``) stay joined as ``well-known``
        instead of becoming two words with a spurious TTS break.
        """

        # 1️⃣ Standard <p> tags
        paragraphs = []
        for p in soup.find_all("p"):
            text = re.sub(r"\s+", " ", p.get_text("", strip=False)).strip()
            if text:
                paragraphs.append(text)
        if paragraphs:
            return paragraphs

        # 2️⃣ Block-level fallback
        block_tags = soup.find_all(["div", "section", "article", "li"])
        paragraphs = [
            re.sub(r"\s+", " ", tag.get_text(" ", strip=True)).strip()
            for tag in block_tags
        ]
        paragraphs = [p for p in paragraphs if p]
        if paragraphs:
            return paragraphs

        # 3️⃣ <br> / newline fallback
        raw_text = soup.get_text("\n", strip=True)
        return [p.strip() for p in raw_text.split("\n") if p.strip()]

    @staticmethod
    def _extract_title(soup, item):
        """
        More flexible title detection.
        """
        title_tag = soup.find(["h1", "h2", "h3", "title"])
        if title_tag and title_tag.get_text(strip=True):
            return title_tag.get_text(strip=True)

        return os.path.basename(item.get_name())

    @staticmethod
    def extract_chapters(epub_file):
        """
        Extract chapters from an EPUB file with layered fallback parsing.

        Keeps original behavior but improves robustness across
        non-standard EPUB structures.
        """

        book = epub.read_epub(epub_file)
        chapters = []

        # ✅ Use spine order (correct reading order)
        spine_items = []
        for idref, _ in book.spine:
            item = book.get_item_with_id(idref)
            if item and item.get_type() == ITEM_DOCUMENT:
                spine_items.append(item)

        for idx, item in enumerate(spine_items):
            soup = BeautifulSoup(item.get_body_content(), "html.parser")

            title = EpubParser._extract_title(soup, item)
            paragraphs = EpubParser._extract_paragraphs(soup)

            full_text = "\n\n".join(paragraphs).strip()

            if not full_text:
                continue

            sentences = []
            for para in paragraphs:
                sentences.extend(split_sentences(para))

            chapters.append(
                {
                    "title": title,
                    "content": full_text,
                    "sentences": sentences,
                    "order": idx + 1,
                }
            )

        # 🔎 Heuristic fallback:
        # If only one large chapter detected, try splitting by headings
        if len(chapters) == 1:
            item = spine_items[0] if spine_items else None
            if item:
                soup = BeautifulSoup(item.get_body_content(), "html.parser")
                headings = soup.find_all(["h1", "h2", "h3"])

                if len(headings) > 1:
                    split_chapters = []
                    for idx, header in enumerate(headings):
                        content = []
                        for sib in header.find_next_siblings():
                            if sib.name in ["h1", "h2", "h3"]:
                                break
                            text = sib.get_text(" ", strip=True)
                            if text:
                                content.append(text)

                        text = "\n\n".join(content).strip()
                        if text:
                            split_chapters.append(
                                {
                                    "title": header.get_text(strip=True),
                                    "content": text,
                                    "sentences": split_sentences(text),
                                    "order": idx + 1,
                                }
                            )

                    if split_chapters:
                        chapters = split_chapters

        if not chapters:
            all_text_chunks = []
            for item in book.get_items_of_type(ITEM_DOCUMENT):
                soup = BeautifulSoup(item.get_body_content(), "html.parser")
                text = soup.get_text(" ", strip=True)
                if text:
                    all_text_chunks.append(text)

            if all_text_chunks:
                combined = "\n\n".join(all_text_chunks)
                sentences = split_sentences(combined)
                book_title = book.get_metadata("DC", "title")
                title = book_title[0][0] if book_title else os.path.basename(epub_file)
                chapters.append(
                    {
                        "title": title,
                        "content": combined,
                        "sentences": sentences,
                        "order": 1,
                    }
                )

        return chapters

extract_chapters(epub_file) staticmethod

Extract chapters from an EPUB file with layered fallback parsing.

Keeps original behavior but improves robustness across non-standard EPUB structures.

Source code in tts_studio/parsers.py
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
@staticmethod
def extract_chapters(epub_file):
    """
    Extract chapters from an EPUB file with layered fallback parsing.

    Keeps original behavior but improves robustness across
    non-standard EPUB structures.
    """

    book = epub.read_epub(epub_file)
    chapters = []

    # ✅ Use spine order (correct reading order)
    spine_items = []
    for idref, _ in book.spine:
        item = book.get_item_with_id(idref)
        if item and item.get_type() == ITEM_DOCUMENT:
            spine_items.append(item)

    for idx, item in enumerate(spine_items):
        soup = BeautifulSoup(item.get_body_content(), "html.parser")

        title = EpubParser._extract_title(soup, item)
        paragraphs = EpubParser._extract_paragraphs(soup)

        full_text = "\n\n".join(paragraphs).strip()

        if not full_text:
            continue

        sentences = []
        for para in paragraphs:
            sentences.extend(split_sentences(para))

        chapters.append(
            {
                "title": title,
                "content": full_text,
                "sentences": sentences,
                "order": idx + 1,
            }
        )

    # 🔎 Heuristic fallback:
    # If only one large chapter detected, try splitting by headings
    if len(chapters) == 1:
        item = spine_items[0] if spine_items else None
        if item:
            soup = BeautifulSoup(item.get_body_content(), "html.parser")
            headings = soup.find_all(["h1", "h2", "h3"])

            if len(headings) > 1:
                split_chapters = []
                for idx, header in enumerate(headings):
                    content = []
                    for sib in header.find_next_siblings():
                        if sib.name in ["h1", "h2", "h3"]:
                            break
                        text = sib.get_text(" ", strip=True)
                        if text:
                            content.append(text)

                    text = "\n\n".join(content).strip()
                    if text:
                        split_chapters.append(
                            {
                                "title": header.get_text(strip=True),
                                "content": text,
                                "sentences": split_sentences(text),
                                "order": idx + 1,
                            }
                        )

                if split_chapters:
                    chapters = split_chapters

    if not chapters:
        all_text_chunks = []
        for item in book.get_items_of_type(ITEM_DOCUMENT):
            soup = BeautifulSoup(item.get_body_content(), "html.parser")
            text = soup.get_text(" ", strip=True)
            if text:
                all_text_chunks.append(text)

        if all_text_chunks:
            combined = "\n\n".join(all_text_chunks)
            sentences = split_sentences(combined)
            book_title = book.get_metadata("DC", "title")
            title = book_title[0][0] if book_title else os.path.basename(epub_file)
            chapters.append(
                {
                    "title": title,
                    "content": combined,
                    "sentences": sentences,
                    "order": 1,
                }
            )

    return chapters

PdfParser

Source code in tts_studio/parsers.py
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
class PdfParser:
    def __init__(self, pdf_path):
        self.pdf_path = pdf_path

    def _extract_markdown(self, doc):
        """Layout-aware markdown extraction, page by page behind a single
        progress bar. Some PDFs contain malformed tables that crash
        pymupdf's table code (``Rect(t.bbox)`` on a table with no cells);
        those pages fall back to plain text instead of failing the book."""
        from tqdm import tqdm

        parts = []
        fallback_pages = 0
        with tqdm(
            total=doc.page_count, desc="  PDF extract", ncols=80, leave=False
        ) as pbar:
            for page in doc:
                try:
                    parts.append(
                        pymupdf4llm.to_markdown(
                            doc,
                            pages=[page.number],
                            write_images=False,
                            show_progress=False,
                        )
                    )
                except Exception:
                    parts.append(page.get_text("text"))
                    fallback_pages += 1
                pbar.update(1)
        if fallback_pages:
            tqdm.write(
                f"  ⚠  {fallback_pages} page(s) extracted as plain text "
                "(layout analysis failed on them)"
            )
        return "\n\n".join(parts)

    def get_chapters(self):
        """
        Extract layout-aware text using pymupdf4llm + layout activation.
        """
        # Open document - layout analysis is now active
        doc = fitz.open(self.pdf_path)

        # Use layout-aware markdown extraction (handles columns, reading order)
        md_text = self._extract_markdown(doc)
        doc.close()

        # Split by markdown headers
        sections = re.split(r"(?m)^#{1,6}\s+", md_text)

        chapters = []
        for i, section in enumerate(sections):
            section = section.strip()
            if len(section) < 50:  # Skip tiny fragments
                continue

            lines = section.split("\n", 1)
            title = lines[0].strip("# ").strip() if lines else f"Section {i+1}"
            content = lines[1].strip() if len(lines) > 1 else section

            # Paragraphs and sentences (strip markdown so it isn't read aloud)
            paragraphs = [
                normalize_text(p, strip_markdown=True)
                for p in re.split(r"\n{2,}", content)
                if p.strip()
            ]
            paragraphs = [p for p in paragraphs if p]
            sentences = []
            for para in paragraphs:
                sentences.extend(split_sentences(para))

            chapters.append(
                {
                    "title": title[:100],
                    "content": "\n\n".join(paragraphs),
                    "paragraphs": paragraphs,
                    "sentences": sentences,
                    "order": i + 1,
                }
            )

        # fallback: if markdown-based section splitting yielded nothing,
        # fall back to raw page text.
        if not chapters:
            doc = fitz.open(self.pdf_path)
            full_text = []
            for page in doc:
                txt = page.get_text("text")
                if txt:
                    full_text.append(txt)
            doc.close()
            combined = "\n\n".join(full_text).strip()
            if combined:
                paras = [
                    normalize_text(p)
                    for p in re.split(r"\n{2,}", combined)
                    if p.strip()
                ]
                paras = [p for p in paras if p]
                sentences = []
                for para in paras:
                    sentences.extend(split_sentences(para))
                chapters.append(
                    {
                        "title": os.path.basename(self.pdf_path),
                        "content": combined,
                        "paragraphs": paras,
                        "sentences": sentences,
                        "order": 1,
                    }
                )

        return chapters

get_chapters()

Extract layout-aware text using pymupdf4llm + layout activation.

Source code in tts_studio/parsers.py
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
def get_chapters(self):
    """
    Extract layout-aware text using pymupdf4llm + layout activation.
    """
    # Open document - layout analysis is now active
    doc = fitz.open(self.pdf_path)

    # Use layout-aware markdown extraction (handles columns, reading order)
    md_text = self._extract_markdown(doc)
    doc.close()

    # Split by markdown headers
    sections = re.split(r"(?m)^#{1,6}\s+", md_text)

    chapters = []
    for i, section in enumerate(sections):
        section = section.strip()
        if len(section) < 50:  # Skip tiny fragments
            continue

        lines = section.split("\n", 1)
        title = lines[0].strip("# ").strip() if lines else f"Section {i+1}"
        content = lines[1].strip() if len(lines) > 1 else section

        # Paragraphs and sentences (strip markdown so it isn't read aloud)
        paragraphs = [
            normalize_text(p, strip_markdown=True)
            for p in re.split(r"\n{2,}", content)
            if p.strip()
        ]
        paragraphs = [p for p in paragraphs if p]
        sentences = []
        for para in paragraphs:
            sentences.extend(split_sentences(para))

        chapters.append(
            {
                "title": title[:100],
                "content": "\n\n".join(paragraphs),
                "paragraphs": paragraphs,
                "sentences": sentences,
                "order": i + 1,
            }
        )

    # fallback: if markdown-based section splitting yielded nothing,
    # fall back to raw page text.
    if not chapters:
        doc = fitz.open(self.pdf_path)
        full_text = []
        for page in doc:
            txt = page.get_text("text")
            if txt:
                full_text.append(txt)
        doc.close()
        combined = "\n\n".join(full_text).strip()
        if combined:
            paras = [
                normalize_text(p)
                for p in re.split(r"\n{2,}", combined)
                if p.strip()
            ]
            paras = [p for p in paras if p]
            sentences = []
            for para in paras:
                sentences.extend(split_sentences(para))
            chapters.append(
                {
                    "title": os.path.basename(self.pdf_path),
                    "content": combined,
                    "paragraphs": paras,
                    "sentences": sentences,
                    "order": 1,
                }
            )

    return chapters