363 lines
13 KiB
Python
363 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
|
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
|
"""Preprocess epub books into txt files.
|
|
"""
|
|
|
|
# stdlib modules
|
|
import os
|
|
import sys
|
|
import re
|
|
import zipfile
|
|
|
|
# pip installed packages
|
|
import numpy as np
|
|
import warnings
|
|
from tqdm import tqdm
|
|
from bs4 import BeautifulSoup
|
|
import ebooklib
|
|
from ebooklib import epub
|
|
import soundfile as sf
|
|
from lxml import etree
|
|
|
|
from PIL import Image
|
|
import nltk
|
|
from nltk.tokenize import sent_tokenize
|
|
|
|
|
|
# Local imports
|
|
from .tts_generic import GenericTTSBackend
|
|
from . import logger, PathLike
|
|
|
|
|
|
|
|
namespaces = {
|
|
"calibre":"http://calibre.kovidgoyal.net/2009/metadata",
|
|
"dc":"http://purl.org/dc/elements/1.1/",
|
|
"dcterms":"http://purl.org/dc/terms/",
|
|
"opf":"http://www.idpf.org/2007/opf",
|
|
"u":"urn:oasis:names:tc:opendocument:xmlns:container",
|
|
"xsi":"http://www.w3.org/2001/XMLSchema-instance",
|
|
}
|
|
|
|
warnings.filterwarnings("ignore", module="ebooklib.epub")
|
|
|
|
def ensure_punkt():
|
|
try:
|
|
nltk.data.find("tokenizers/punkt")
|
|
except LookupError:
|
|
nltk.download("punkt")
|
|
try:
|
|
nltk.data.find("tokenizers/punkt_tab")
|
|
except LookupError:
|
|
nltk.download("punkt_tab")
|
|
|
|
|
|
def chap2text_epub(chap, item_id=None, toc=None):
|
|
"""
|
|
Extract chapter title and paragraphs from an EPUB chapter.
|
|
|
|
Args:
|
|
chap: The chapter content (HTML).
|
|
item_id: The ID of the item in the EPUB spine (for fallback naming).
|
|
toc: The EPUB's table of contents (for fallback title extraction).
|
|
|
|
Returns:
|
|
tuple: (chapter_title_text, paragraphs)
|
|
"""
|
|
blacklist = [
|
|
"[document]",
|
|
"noscript",
|
|
"header",
|
|
"html",
|
|
"meta",
|
|
"head",
|
|
"input",
|
|
"script",
|
|
]
|
|
paragraphs = []
|
|
soup = BeautifulSoup(chap, "html.parser")
|
|
|
|
# Step 1: Try to find chapter title in heading tags (<h1>, <h2>, <h3>)
|
|
heading_tags = ['h1', 'h2', 'h3']
|
|
chapter_title_text = None
|
|
for tag in heading_tags:
|
|
heading = soup.find(tag)
|
|
if heading and heading.text.strip():
|
|
chapter_title_text = heading.text.strip()
|
|
print(f"Found title in <{tag}>: '{chapter_title_text}'")
|
|
break
|
|
|
|
# Step 2: If no heading found, try elements with common class names
|
|
if not chapter_title_text:
|
|
common_classes = ['chapter', 'chapter-title', 'title', 'heading']
|
|
for class_name in common_classes:
|
|
element = soup.find(class_=class_name)
|
|
if element and element.text.strip():
|
|
chapter_title_text = element.text.strip()
|
|
print(f"Found title in class '{class_name}': '{chapter_title_text}'")
|
|
break
|
|
|
|
# Step 3: Fallback to TOC if provided
|
|
if not chapter_title_text and toc and item_id:
|
|
for toc_item in toc:
|
|
if toc_item.href.split('#')[0] == item_id:
|
|
chapter_title_text = toc_item.title
|
|
print(f"Found title in TOC for item '{item_id}': '{chapter_title_text}'")
|
|
break
|
|
|
|
# Step 4: Fallback to item ID or generic name
|
|
if not chapter_title_text:
|
|
chapter_title_text = item_id.replace('.xhtml', '').replace('_', ' ').title() if item_id else None
|
|
print(f"No title found, using fallback: '{chapter_title_text}'")
|
|
|
|
# Remove footnotes (links with only numbers)
|
|
for a in soup.findAll("a", href=True):
|
|
if not any(char.isalpha() for char in a.text):
|
|
a.extract()
|
|
|
|
# Remove superscript numbers (e.g., footnote markers)
|
|
for sup in soup.findAll("sup"):
|
|
if sup.text.isdigit():
|
|
sup.extract()
|
|
|
|
# Extract paragraphs
|
|
chapter_paragraphs = soup.find_all("p")
|
|
if not chapter_paragraphs:
|
|
print(f"No <p> tags found in '{chapter_title_text or item_id}'. Trying <div>.")
|
|
chapter_paragraphs = soup.find_all("div")
|
|
|
|
for p in chapter_paragraphs:
|
|
paragraph_text = "".join(p.strings).strip()
|
|
if paragraph_text:
|
|
paragraphs.append(paragraph_text)
|
|
|
|
return chapter_title_text, paragraphs
|
|
|
|
def get_epub_cover(epub_path):
|
|
try:
|
|
with zipfile.ZipFile(epub_path) as z:
|
|
t = etree.fromstring(z.read("META-INF/container.xml"))
|
|
rootfile_path = t.xpath("/u:container/u:rootfiles/u:rootfile",
|
|
namespaces=namespaces)[0].get("full-path")
|
|
|
|
t = etree.fromstring(z.read(rootfile_path))
|
|
cover_meta = t.xpath("//opf:metadata/opf:meta[@name='cover']",
|
|
namespaces=namespaces)
|
|
if not cover_meta:
|
|
print("No cover image found.")
|
|
return None
|
|
cover_id = cover_meta[0].get("content")
|
|
|
|
cover_item = t.xpath("//opf:manifest/opf:item[@id='" + cover_id + "']",
|
|
namespaces=namespaces)
|
|
if not cover_item:
|
|
print("No cover image found.")
|
|
return None
|
|
cover_href = cover_item[0].get("href")
|
|
cover_path = os.path.join(os.path.dirname(rootfile_path), cover_href)
|
|
if os.name == 'nt' and '\\' in cover_path:
|
|
cover_path = cover_path.replace("\\", "/")
|
|
return z.open(cover_path)
|
|
except FileNotFoundError:
|
|
print(f"Could not get cover image of {epub_path}")
|
|
|
|
def export(book, sourcefile):
|
|
book_contents = []
|
|
cover_image = get_epub_cover(sourcefile)
|
|
image_path = None
|
|
|
|
if cover_image is not None:
|
|
image = Image.open(cover_image)
|
|
image_filename = sourcefile.replace(".epub", ".png")
|
|
image_path = os.path.join(image_filename)
|
|
image.save(image_path)
|
|
print(f"Cover image saved to {image_path}")
|
|
|
|
# Get the table of contents
|
|
toc = book.get_toc() if hasattr(book, 'get_toc') else []
|
|
|
|
spine_ids = [spine_tuple[0] for spine_tuple in book.spine if spine_tuple[1] == 'yes']
|
|
items = {item.get_id(): item for item in book.get_items() if item.get_type() == ebooklib.ITEM_DOCUMENT}
|
|
|
|
for id in spine_ids:
|
|
item = items.get(id)
|
|
if item is None:
|
|
continue
|
|
# Pass item_id and toc to chap2text_epub
|
|
chapter_title, chapter_paragraphs = chap2text_epub(item.get_content(), item_id=id, toc=toc)
|
|
book_contents.append({"title": chapter_title, "paragraphs": chapter_paragraphs})
|
|
|
|
outfile = sourcefile.replace(".epub", ".txt")
|
|
check_for_file(outfile)
|
|
print(f"Exporting {sourcefile} to {outfile}")
|
|
author = book.get_metadata("DC", "creator")[0][0]
|
|
booktitle = book.get_metadata("DC", "title")[0][0]
|
|
|
|
with open(outfile, "w", encoding='utf-8') as file:
|
|
file.write(f"Title: {booktitle}\n")
|
|
file.write(f"Author: {author}\n\n")
|
|
file.write(f"# Title\n")
|
|
file.write(f"{booktitle}, by {author}\n\n")
|
|
for i, chapter in enumerate(book_contents, start=1):
|
|
if not chapter["paragraphs"] or chapter["paragraphs"] == ['']:
|
|
continue
|
|
else:
|
|
# Use chapter title if available, otherwise fallback to "Part {i}"
|
|
title = chapter["title"] if chapter["title"] else f"Part {i}"
|
|
file.write(f"# {title}\n\n")
|
|
for paragraph in chapter["paragraphs"]:
|
|
clean = re.sub(r'[\s\n]+', ' ', paragraph)
|
|
clean = re.sub(r'[“”]', '"', clean) # Curly double quotes to standard double quotes
|
|
clean = re.sub(r'[‘’]', "'", clean) # Curly single quotes to standard single quotes
|
|
clean = re.sub(r'--', ', ', clean)
|
|
file.write(f"{clean}\n\n")
|
|
|
|
return book_contents
|
|
|
|
def get_book(sourcefile):
|
|
book_contents = []
|
|
book_title = sourcefile
|
|
book_author = "Unknown"
|
|
chapter_titles = []
|
|
|
|
with open(sourcefile, "r", encoding="utf-8") as file:
|
|
current_chapter = {"title": "blank", "paragraphs": []}
|
|
initialized_first_chapter = False
|
|
lines_skipped = 0
|
|
for line in file:
|
|
|
|
if lines_skipped < 2 and (line.startswith("Title") or line.startswith("Author")):
|
|
lines_skipped += 1
|
|
if line.startswith('Title: '):
|
|
book_title = line.replace('Title: ', '').strip()
|
|
elif line.startswith('Author: '):
|
|
book_author = line.replace('Author: ', '').strip()
|
|
continue
|
|
|
|
line = line.strip()
|
|
if line.startswith("#"):
|
|
if current_chapter["paragraphs"] or not initialized_first_chapter:
|
|
if initialized_first_chapter:
|
|
book_contents.append(current_chapter)
|
|
current_chapter = {"title": None, "paragraphs": []}
|
|
initialized_first_chapter = True
|
|
chapter_title = line[1:].strip()
|
|
if any(c.isalnum() for c in chapter_title):
|
|
current_chapter["title"] = chapter_title
|
|
chapter_titles.append(current_chapter["title"])
|
|
else:
|
|
current_chapter["title"] = "blank"
|
|
chapter_titles.append("blank")
|
|
elif line:
|
|
if not initialized_first_chapter:
|
|
chapter_titles.append("blank")
|
|
initialized_first_chapter = True
|
|
if any(char.isalnum() for char in line):
|
|
sentences = sent_tokenize(line)
|
|
cleaned_sentences = [s for s in sentences if any(char.isalnum() for char in s)]
|
|
line = ' '.join(cleaned_sentences)
|
|
current_chapter["paragraphs"].append(line)
|
|
|
|
# Append the last chapter if it contains any paragraphs.
|
|
if current_chapter["paragraphs"]:
|
|
book_contents.append(current_chapter)
|
|
|
|
return book_contents, book_title, book_author, chapter_titles
|
|
|
|
def sort_key(s):
|
|
# extract number from the string
|
|
return int(re.findall(r'\d+', s)[0])
|
|
|
|
def check_for_file(filename):
|
|
if os.path.isfile(filename):
|
|
print(f"The file '{filename}' already exists.")
|
|
overwrite = input("Do you want to overwrite the file? (y/n): ")
|
|
if overwrite.lower() != 'y':
|
|
print("Exiting without overwriting the file.")
|
|
sys.exit()
|
|
else:
|
|
os.remove(filename)
|
|
|
|
def append_silence(tempfile, duration=1200):
|
|
audio = AudioSegment.from_file(tempfile)
|
|
# Create a silence segment
|
|
silence = AudioSegment.silent(duration)
|
|
# Append the silence segment to the audio
|
|
combined = audio + silence
|
|
# Save the combined audio back to file
|
|
combined.export(tempfile, format="flac")
|
|
|
|
def break_long_sentence(sentence, max_length=200):
|
|
# Split sentence based on commas
|
|
comma_segments = sentence.split(',')
|
|
segments = []
|
|
current_segment = ""
|
|
for segment in comma_segments:
|
|
# Check if adding the next segment exceeds max_length
|
|
temp_segment = current_segment + ("," if current_segment else "") + segment
|
|
if len(temp_segment) > max_length:
|
|
# Add the current segment to the list and reset it
|
|
if current_segment:
|
|
segments.append(current_segment)
|
|
# Start a new segment with the current part
|
|
current_segment = segment.strip()
|
|
else:
|
|
# Continue building the current segment
|
|
current_segment = temp_segment.strip()
|
|
# Don't forget to add the last segment if it exists
|
|
if current_segment:
|
|
segments.append(current_segment)
|
|
return segments
|
|
|
|
def process_large_text(line):
|
|
# Tokenize the text into sentences
|
|
sentences = sent_tokenize(line)
|
|
# Initialize a list to store processed sentences
|
|
results = []
|
|
|
|
i = 0
|
|
while i < len(sentences):
|
|
sentence = sentences[i]
|
|
word_count = len(sentence.split())
|
|
|
|
# Combine with the next sentence if this one has fewer than 8 words
|
|
if word_count < 8 and i + 1 < len(sentences):
|
|
# Combine the current and next sentence
|
|
sentence = sentence + ' ' + sentences[i + 1]
|
|
i += 1 # Skip the next sentence since it's already combined
|
|
|
|
if len(sentence) > 500:
|
|
# Break the long sentences into smaller parts using commas
|
|
results.extend(break_long_sentence(sentence, max_length=350))
|
|
else:
|
|
results.append(sentence)
|
|
|
|
i += 1 # Move to the next sentence
|
|
|
|
# Before returning, combine last elements if they are too short
|
|
if results and len(results[-1].split()) < 8:
|
|
if len(results) > 1:
|
|
# Combine the last two sentences if they are both short
|
|
results[-2] += ' ' + results[-1]
|
|
results.pop()
|
|
|
|
return results
|
|
|
|
def conditional_sentence_case(sent):
|
|
# Split the sentence into words
|
|
words = sent.split()
|
|
length = len(words)
|
|
# Iterate through words to check for three consecutive uppercase words
|
|
for i in range(length - 2):
|
|
if words[i].isupper() and words[i+1].isupper() and words[i+2].isupper():
|
|
# Convert the entire sentence to lowercase and capitalize the first letter
|
|
sent = ' '.join(words).lower().capitalize()
|
|
break # No need to continue checking once a match is found
|
|
return sent
|
|
|
|
def preprocess_book(book_path):
|
|
ensure_punkt()
|
|
book = epub.read_epub(book_path)
|
|
export(book, book_path) |