13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159 | class EpubParser:
@staticmethod
def _extract_paragraphs(soup):
"""
Layered paragraph extraction:
1. <p> tags
2. block-level containers
3. newline split fallback
Uses an empty separator so that words split across inline tags
(e.g. ``well-<span>known</span>``) stay joined as ``well-known``
instead of becoming two words with a spurious TTS break.
"""
# 1️⃣ Standard <p> tags
paragraphs = []
for p in soup.find_all("p"):
text = re.sub(r"\s+", " ", p.get_text("", strip=False)).strip()
if text:
paragraphs.append(text)
if paragraphs:
return paragraphs
# 2️⃣ Block-level fallback
block_tags = soup.find_all(["div", "section", "article", "li"])
paragraphs = [
re.sub(r"\s+", " ", tag.get_text(" ", strip=True)).strip()
for tag in block_tags
]
paragraphs = [p for p in paragraphs if p]
if paragraphs:
return paragraphs
# 3️⃣ <br> / newline fallback
raw_text = soup.get_text("\n", strip=True)
return [p.strip() for p in raw_text.split("\n") if p.strip()]
@staticmethod
def _extract_title(soup, item):
"""
More flexible title detection.
"""
title_tag = soup.find(["h1", "h2", "h3", "title"])
if title_tag and title_tag.get_text(strip=True):
return title_tag.get_text(strip=True)
return os.path.basename(item.get_name())
@staticmethod
def extract_chapters(epub_file):
"""
Extract chapters from an EPUB file with layered fallback parsing.
Keeps original behavior but improves robustness across
non-standard EPUB structures.
"""
book = epub.read_epub(epub_file)
chapters = []
# ✅ Use spine order (correct reading order)
spine_items = []
for idref, _ in book.spine:
item = book.get_item_with_id(idref)
if item and item.get_type() == ITEM_DOCUMENT:
spine_items.append(item)
for idx, item in enumerate(spine_items):
soup = BeautifulSoup(item.get_body_content(), "html.parser")
title = EpubParser._extract_title(soup, item)
paragraphs = EpubParser._extract_paragraphs(soup)
full_text = "\n\n".join(paragraphs).strip()
if not full_text:
continue
sentences = []
for para in paragraphs:
sentences.extend(split_sentences(para))
chapters.append(
{
"title": title,
"content": full_text,
"sentences": sentences,
"order": idx + 1,
}
)
# 🔎 Heuristic fallback:
# If only one large chapter detected, try splitting by headings
if len(chapters) == 1:
item = spine_items[0] if spine_items else None
if item:
soup = BeautifulSoup(item.get_body_content(), "html.parser")
headings = soup.find_all(["h1", "h2", "h3"])
if len(headings) > 1:
split_chapters = []
for idx, header in enumerate(headings):
content = []
for sib in header.find_next_siblings():
if sib.name in ["h1", "h2", "h3"]:
break
text = sib.get_text(" ", strip=True)
if text:
content.append(text)
text = "\n\n".join(content).strip()
if text:
split_chapters.append(
{
"title": header.get_text(strip=True),
"content": text,
"sentences": split_sentences(text),
"order": idx + 1,
}
)
if split_chapters:
chapters = split_chapters
if not chapters:
all_text_chunks = []
for item in book.get_items_of_type(ITEM_DOCUMENT):
soup = BeautifulSoup(item.get_body_content(), "html.parser")
text = soup.get_text(" ", strip=True)
if text:
all_text_chunks.append(text)
if all_text_chunks:
combined = "\n\n".join(all_text_chunks)
sentences = split_sentences(combined)
book_title = book.get_metadata("DC", "title")
title = book_title[0][0] if book_title else os.path.basename(epub_file)
chapters.append(
{
"title": title,
"content": combined,
"sentences": sentences,
"order": 1,
}
)
return chapters
|