Files
epub-tts/src/epub_tts/preprocess.py
T
2026-08-10 16:37:59 -03:00

363 lines
13 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
"""Preprocess epub books into txt files.
"""
# stdlib modules
import os
import sys
import re
import zipfile
# pip installed packages
import numpy as np
import warnings
from tqdm import tqdm
from bs4 import BeautifulSoup
import ebooklib
from ebooklib import epub
import soundfile as sf
from lxml import etree
from PIL import Image
import nltk
from nltk.tokenize import sent_tokenize
# Local imports
from .tts_generic import GenericTTSBackend
from . import logger, PathLike
namespaces = {
"calibre":"http://calibre.kovidgoyal.net/2009/metadata",
"dc":"http://purl.org/dc/elements/1.1/",
"dcterms":"http://purl.org/dc/terms/",
"opf":"http://www.idpf.org/2007/opf",
"u":"urn:oasis:names:tc:opendocument:xmlns:container",
"xsi":"http://www.w3.org/2001/XMLSchema-instance",
}
warnings.filterwarnings("ignore", module="ebooklib.epub")
def ensure_punkt():
try:
nltk.data.find("tokenizers/punkt")
except LookupError:
nltk.download("punkt")
try:
nltk.data.find("tokenizers/punkt_tab")
except LookupError:
nltk.download("punkt_tab")
def chap2text_epub(chap, item_id=None, toc=None):
"""
Extract chapter title and paragraphs from an EPUB chapter.
Args:
chap: The chapter content (HTML).
item_id: The ID of the item in the EPUB spine (for fallback naming).
toc: The EPUB's table of contents (for fallback title extraction).
Returns:
tuple: (chapter_title_text, paragraphs)
"""
blacklist = [
"[document]",
"noscript",
"header",
"html",
"meta",
"head",
"input",
"script",
]
paragraphs = []
soup = BeautifulSoup(chap, "html.parser")
# Step 1: Try to find chapter title in heading tags (<h1>, <h2>, <h3>)
heading_tags = ['h1', 'h2', 'h3']
chapter_title_text = None
for tag in heading_tags:
heading = soup.find(tag)
if heading and heading.text.strip():
chapter_title_text = heading.text.strip()
print(f"Found title in <{tag}>: '{chapter_title_text}'")
break
# Step 2: If no heading found, try elements with common class names
if not chapter_title_text:
common_classes = ['chapter', 'chapter-title', 'title', 'heading']
for class_name in common_classes:
element = soup.find(class_=class_name)
if element and element.text.strip():
chapter_title_text = element.text.strip()
print(f"Found title in class '{class_name}': '{chapter_title_text}'")
break
# Step 3: Fallback to TOC if provided
if not chapter_title_text and toc and item_id:
for toc_item in toc:
if toc_item.href.split('#')[0] == item_id:
chapter_title_text = toc_item.title
print(f"Found title in TOC for item '{item_id}': '{chapter_title_text}'")
break
# Step 4: Fallback to item ID or generic name
if not chapter_title_text:
chapter_title_text = item_id.replace('.xhtml', '').replace('_', ' ').title() if item_id else None
print(f"No title found, using fallback: '{chapter_title_text}'")
# Remove footnotes (links with only numbers)
for a in soup.findAll("a", href=True):
if not any(char.isalpha() for char in a.text):
a.extract()
# Remove superscript numbers (e.g., footnote markers)
for sup in soup.findAll("sup"):
if sup.text.isdigit():
sup.extract()
# Extract paragraphs
chapter_paragraphs = soup.find_all("p")
if not chapter_paragraphs:
print(f"No <p> tags found in '{chapter_title_text or item_id}'. Trying <div>.")
chapter_paragraphs = soup.find_all("div")
for p in chapter_paragraphs:
paragraph_text = "".join(p.strings).strip()
if paragraph_text:
paragraphs.append(paragraph_text)
return chapter_title_text, paragraphs
def get_epub_cover(epub_path):
try:
with zipfile.ZipFile(epub_path) as z:
t = etree.fromstring(z.read("META-INF/container.xml"))
rootfile_path = t.xpath("/u:container/u:rootfiles/u:rootfile",
namespaces=namespaces)[0].get("full-path")
t = etree.fromstring(z.read(rootfile_path))
cover_meta = t.xpath("//opf:metadata/opf:meta[@name='cover']",
namespaces=namespaces)
if not cover_meta:
print("No cover image found.")
return None
cover_id = cover_meta[0].get("content")
cover_item = t.xpath("//opf:manifest/opf:item[@id='" + cover_id + "']",
namespaces=namespaces)
if not cover_item:
print("No cover image found.")
return None
cover_href = cover_item[0].get("href")
cover_path = os.path.join(os.path.dirname(rootfile_path), cover_href)
if os.name == 'nt' and '\\' in cover_path:
cover_path = cover_path.replace("\\", "/")
return z.open(cover_path)
except FileNotFoundError:
print(f"Could not get cover image of {epub_path}")
def export(book, sourcefile):
book_contents = []
cover_image = get_epub_cover(sourcefile)
image_path = None
if cover_image is not None:
image = Image.open(cover_image)
image_filename = sourcefile.replace(".epub", ".png")
image_path = os.path.join(image_filename)
image.save(image_path)
print(f"Cover image saved to {image_path}")
# Get the table of contents
toc = book.get_toc() if hasattr(book, 'get_toc') else []
spine_ids = [spine_tuple[0] for spine_tuple in book.spine if spine_tuple[1] == 'yes']
items = {item.get_id(): item for item in book.get_items() if item.get_type() == ebooklib.ITEM_DOCUMENT}
for id in spine_ids:
item = items.get(id)
if item is None:
continue
# Pass item_id and toc to chap2text_epub
chapter_title, chapter_paragraphs = chap2text_epub(item.get_content(), item_id=id, toc=toc)
book_contents.append({"title": chapter_title, "paragraphs": chapter_paragraphs})
outfile = sourcefile.replace(".epub", ".txt")
check_for_file(outfile)
print(f"Exporting {sourcefile} to {outfile}")
author = book.get_metadata("DC", "creator")[0][0]
booktitle = book.get_metadata("DC", "title")[0][0]
with open(outfile, "w", encoding='utf-8') as file:
file.write(f"Title: {booktitle}\n")
file.write(f"Author: {author}\n\n")
file.write(f"# Title\n")
file.write(f"{booktitle}, by {author}\n\n")
for i, chapter in enumerate(book_contents, start=1):
if not chapter["paragraphs"] or chapter["paragraphs"] == ['']:
continue
else:
# Use chapter title if available, otherwise fallback to "Part {i}"
title = chapter["title"] if chapter["title"] else f"Part {i}"
file.write(f"# {title}\n\n")
for paragraph in chapter["paragraphs"]:
clean = re.sub(r'[\s\n]+', ' ', paragraph)
clean = re.sub(r'[“”]', '"', clean) # Curly double quotes to standard double quotes
clean = re.sub(r'[‘’]', "'", clean) # Curly single quotes to standard single quotes
clean = re.sub(r'--', ', ', clean)
file.write(f"{clean}\n\n")
return book_contents
def get_book(sourcefile):
book_contents = []
book_title = sourcefile
book_author = "Unknown"
chapter_titles = []
with open(sourcefile, "r", encoding="utf-8") as file:
current_chapter = {"title": "blank", "paragraphs": []}
initialized_first_chapter = False
lines_skipped = 0
for line in file:
if lines_skipped < 2 and (line.startswith("Title") or line.startswith("Author")):
lines_skipped += 1
if line.startswith('Title: '):
book_title = line.replace('Title: ', '').strip()
elif line.startswith('Author: '):
book_author = line.replace('Author: ', '').strip()
continue
line = line.strip()
if line.startswith("#"):
if current_chapter["paragraphs"] or not initialized_first_chapter:
if initialized_first_chapter:
book_contents.append(current_chapter)
current_chapter = {"title": None, "paragraphs": []}
initialized_first_chapter = True
chapter_title = line[1:].strip()
if any(c.isalnum() for c in chapter_title):
current_chapter["title"] = chapter_title
chapter_titles.append(current_chapter["title"])
else:
current_chapter["title"] = "blank"
chapter_titles.append("blank")
elif line:
if not initialized_first_chapter:
chapter_titles.append("blank")
initialized_first_chapter = True
if any(char.isalnum() for char in line):
sentences = sent_tokenize(line)
cleaned_sentences = [s for s in sentences if any(char.isalnum() for char in s)]
line = ' '.join(cleaned_sentences)
current_chapter["paragraphs"].append(line)
# Append the last chapter if it contains any paragraphs.
if current_chapter["paragraphs"]:
book_contents.append(current_chapter)
return book_contents, book_title, book_author, chapter_titles
def sort_key(s):
# extract number from the string
return int(re.findall(r'\d+', s)[0])
def check_for_file(filename):
if os.path.isfile(filename):
print(f"The file '{filename}' already exists.")
overwrite = input("Do you want to overwrite the file? (y/n): ")
if overwrite.lower() != 'y':
print("Exiting without overwriting the file.")
sys.exit()
else:
os.remove(filename)
def append_silence(tempfile, duration=1200):
audio = AudioSegment.from_file(tempfile)
# Create a silence segment
silence = AudioSegment.silent(duration)
# Append the silence segment to the audio
combined = audio + silence
# Save the combined audio back to file
combined.export(tempfile, format="flac")
def break_long_sentence(sentence, max_length=200):
# Split sentence based on commas
comma_segments = sentence.split(',')
segments = []
current_segment = ""
for segment in comma_segments:
# Check if adding the next segment exceeds max_length
temp_segment = current_segment + ("," if current_segment else "") + segment
if len(temp_segment) > max_length:
# Add the current segment to the list and reset it
if current_segment:
segments.append(current_segment)
# Start a new segment with the current part
current_segment = segment.strip()
else:
# Continue building the current segment
current_segment = temp_segment.strip()
# Don't forget to add the last segment if it exists
if current_segment:
segments.append(current_segment)
return segments
def process_large_text(line):
# Tokenize the text into sentences
sentences = sent_tokenize(line)
# Initialize a list to store processed sentences
results = []
i = 0
while i < len(sentences):
sentence = sentences[i]
word_count = len(sentence.split())
# Combine with the next sentence if this one has fewer than 8 words
if word_count < 8 and i + 1 < len(sentences):
# Combine the current and next sentence
sentence = sentence + ' ' + sentences[i + 1]
i += 1 # Skip the next sentence since it's already combined
if len(sentence) > 500:
# Break the long sentences into smaller parts using commas
results.extend(break_long_sentence(sentence, max_length=350))
else:
results.append(sentence)
i += 1 # Move to the next sentence
# Before returning, combine last elements if they are too short
if results and len(results[-1].split()) < 8:
if len(results) > 1:
# Combine the last two sentences if they are both short
results[-2] += ' ' + results[-1]
results.pop()
return results
def conditional_sentence_case(sent):
# Split the sentence into words
words = sent.split()
length = len(words)
# Iterate through words to check for three consecutive uppercase words
for i in range(length - 2):
if words[i].isupper() and words[i+1].isupper() and words[i+2].isupper():
# Convert the entire sentence to lowercase and capitalize the first letter
sent = ' '.join(words).lower().capitalize()
break # No need to continue checking once a match is found
return sent
def preprocess_book(book_path):
ensure_punkt()
book = epub.read_epub(book_path)
export(book, book_path)