tts-kokoro engine underway
This commit is contained in:
@@ -0,0 +1,363 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
||||
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
||||
"""Preprocess epub books into txt files.
|
||||
"""
|
||||
|
||||
# stdlib modules
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import zipfile
|
||||
|
||||
# pip installed packages
|
||||
import numpy as np
|
||||
import warnings
|
||||
from tqdm import tqdm
|
||||
from bs4 import BeautifulSoup
|
||||
import ebooklib
|
||||
from ebooklib import epub
|
||||
import soundfile as sf
|
||||
from lxml import etree
|
||||
|
||||
from PIL import Image
|
||||
import nltk
|
||||
from nltk.tokenize import sent_tokenize
|
||||
|
||||
|
||||
# Local imports
|
||||
from .tts_generic import GenericTTSBackend
|
||||
from . import log, PathLike
|
||||
|
||||
|
||||
|
||||
namespaces = {
|
||||
"calibre":"http://calibre.kovidgoyal.net/2009/metadata",
|
||||
"dc":"http://purl.org/dc/elements/1.1/",
|
||||
"dcterms":"http://purl.org/dc/terms/",
|
||||
"opf":"http://www.idpf.org/2007/opf",
|
||||
"u":"urn:oasis:names:tc:opendocument:xmlns:container",
|
||||
"xsi":"http://www.w3.org/2001/XMLSchema-instance",
|
||||
}
|
||||
|
||||
warnings.filterwarnings("ignore", module="ebooklib.epub")
|
||||
|
||||
def ensure_punkt():
|
||||
try:
|
||||
nltk.data.find("tokenizers/punkt")
|
||||
except LookupError:
|
||||
nltk.download("punkt")
|
||||
try:
|
||||
nltk.data.find("tokenizers/punkt_tab")
|
||||
except LookupError:
|
||||
nltk.download("punkt_tab")
|
||||
|
||||
|
||||
def chap2text_epub(chap, item_id=None, toc=None):
|
||||
"""
|
||||
Extract chapter title and paragraphs from an EPUB chapter.
|
||||
|
||||
Args:
|
||||
chap: The chapter content (HTML).
|
||||
item_id: The ID of the item in the EPUB spine (for fallback naming).
|
||||
toc: The EPUB's table of contents (for fallback title extraction).
|
||||
|
||||
Returns:
|
||||
tuple: (chapter_title_text, paragraphs)
|
||||
"""
|
||||
blacklist = [
|
||||
"[document]",
|
||||
"noscript",
|
||||
"header",
|
||||
"html",
|
||||
"meta",
|
||||
"head",
|
||||
"input",
|
||||
"script",
|
||||
]
|
||||
paragraphs = []
|
||||
soup = BeautifulSoup(chap, "html.parser")
|
||||
|
||||
# Step 1: Try to find chapter title in heading tags (<h1>, <h2>, <h3>)
|
||||
heading_tags = ['h1', 'h2', 'h3']
|
||||
chapter_title_text = None
|
||||
for tag in heading_tags:
|
||||
heading = soup.find(tag)
|
||||
if heading and heading.text.strip():
|
||||
chapter_title_text = heading.text.strip()
|
||||
print(f"Found title in <{tag}>: '{chapter_title_text}'")
|
||||
break
|
||||
|
||||
# Step 2: If no heading found, try elements with common class names
|
||||
if not chapter_title_text:
|
||||
common_classes = ['chapter', 'chapter-title', 'title', 'heading']
|
||||
for class_name in common_classes:
|
||||
element = soup.find(class_=class_name)
|
||||
if element and element.text.strip():
|
||||
chapter_title_text = element.text.strip()
|
||||
print(f"Found title in class '{class_name}': '{chapter_title_text}'")
|
||||
break
|
||||
|
||||
# Step 3: Fallback to TOC if provided
|
||||
if not chapter_title_text and toc and item_id:
|
||||
for toc_item in toc:
|
||||
if toc_item.href.split('#')[0] == item_id:
|
||||
chapter_title_text = toc_item.title
|
||||
print(f"Found title in TOC for item '{item_id}': '{chapter_title_text}'")
|
||||
break
|
||||
|
||||
# Step 4: Fallback to item ID or generic name
|
||||
if not chapter_title_text:
|
||||
chapter_title_text = item_id.replace('.xhtml', '').replace('_', ' ').title() if item_id else None
|
||||
print(f"No title found, using fallback: '{chapter_title_text}'")
|
||||
|
||||
# Remove footnotes (links with only numbers)
|
||||
for a in soup.findAll("a", href=True):
|
||||
if not any(char.isalpha() for char in a.text):
|
||||
a.extract()
|
||||
|
||||
# Remove superscript numbers (e.g., footnote markers)
|
||||
for sup in soup.findAll("sup"):
|
||||
if sup.text.isdigit():
|
||||
sup.extract()
|
||||
|
||||
# Extract paragraphs
|
||||
chapter_paragraphs = soup.find_all("p")
|
||||
if not chapter_paragraphs:
|
||||
print(f"No <p> tags found in '{chapter_title_text or item_id}'. Trying <div>.")
|
||||
chapter_paragraphs = soup.find_all("div")
|
||||
|
||||
for p in chapter_paragraphs:
|
||||
paragraph_text = "".join(p.strings).strip()
|
||||
if paragraph_text:
|
||||
paragraphs.append(paragraph_text)
|
||||
|
||||
return chapter_title_text, paragraphs
|
||||
|
||||
def get_epub_cover(epub_path):
|
||||
try:
|
||||
with zipfile.ZipFile(epub_path) as z:
|
||||
t = etree.fromstring(z.read("META-INF/container.xml"))
|
||||
rootfile_path = t.xpath("/u:container/u:rootfiles/u:rootfile",
|
||||
namespaces=namespaces)[0].get("full-path")
|
||||
|
||||
t = etree.fromstring(z.read(rootfile_path))
|
||||
cover_meta = t.xpath("//opf:metadata/opf:meta[@name='cover']",
|
||||
namespaces=namespaces)
|
||||
if not cover_meta:
|
||||
print("No cover image found.")
|
||||
return None
|
||||
cover_id = cover_meta[0].get("content")
|
||||
|
||||
cover_item = t.xpath("//opf:manifest/opf:item[@id='" + cover_id + "']",
|
||||
namespaces=namespaces)
|
||||
if not cover_item:
|
||||
print("No cover image found.")
|
||||
return None
|
||||
cover_href = cover_item[0].get("href")
|
||||
cover_path = os.path.join(os.path.dirname(rootfile_path), cover_href)
|
||||
if os.name == 'nt' and '\\' in cover_path:
|
||||
cover_path = cover_path.replace("\\", "/")
|
||||
return z.open(cover_path)
|
||||
except FileNotFoundError:
|
||||
print(f"Could not get cover image of {epub_path}")
|
||||
|
||||
def export(book, sourcefile):
|
||||
book_contents = []
|
||||
cover_image = get_epub_cover(sourcefile)
|
||||
image_path = None
|
||||
|
||||
if cover_image is not None:
|
||||
image = Image.open(cover_image)
|
||||
image_filename = sourcefile.replace(".epub", ".png")
|
||||
image_path = os.path.join(image_filename)
|
||||
image.save(image_path)
|
||||
print(f"Cover image saved to {image_path}")
|
||||
|
||||
# Get the table of contents
|
||||
toc = book.get_toc() if hasattr(book, 'get_toc') else []
|
||||
|
||||
spine_ids = [spine_tuple[0] for spine_tuple in book.spine if spine_tuple[1] == 'yes']
|
||||
items = {item.get_id(): item for item in book.get_items() if item.get_type() == ebooklib.ITEM_DOCUMENT}
|
||||
|
||||
for id in spine_ids:
|
||||
item = items.get(id)
|
||||
if item is None:
|
||||
continue
|
||||
# Pass item_id and toc to chap2text_epub
|
||||
chapter_title, chapter_paragraphs = chap2text_epub(item.get_content(), item_id=id, toc=toc)
|
||||
book_contents.append({"title": chapter_title, "paragraphs": chapter_paragraphs})
|
||||
|
||||
outfile = sourcefile.replace(".epub", ".txt")
|
||||
check_for_file(outfile)
|
||||
print(f"Exporting {sourcefile} to {outfile}")
|
||||
author = book.get_metadata("DC", "creator")[0][0]
|
||||
booktitle = book.get_metadata("DC", "title")[0][0]
|
||||
|
||||
with open(outfile, "w", encoding='utf-8') as file:
|
||||
file.write(f"Title: {booktitle}\n")
|
||||
file.write(f"Author: {author}\n\n")
|
||||
file.write(f"# Title\n")
|
||||
file.write(f"{booktitle}, by {author}\n\n")
|
||||
for i, chapter in enumerate(book_contents, start=1):
|
||||
if not chapter["paragraphs"] or chapter["paragraphs"] == ['']:
|
||||
continue
|
||||
else:
|
||||
# Use chapter title if available, otherwise fallback to "Part {i}"
|
||||
title = chapter["title"] if chapter["title"] else f"Part {i}"
|
||||
file.write(f"# {title}\n\n")
|
||||
for paragraph in chapter["paragraphs"]:
|
||||
clean = re.sub(r'[\s\n]+', ' ', paragraph)
|
||||
clean = re.sub(r'[“”]', '"', clean) # Curly double quotes to standard double quotes
|
||||
clean = re.sub(r'[‘’]', "'", clean) # Curly single quotes to standard single quotes
|
||||
clean = re.sub(r'--', ', ', clean)
|
||||
file.write(f"{clean}\n\n")
|
||||
|
||||
return book_contents
|
||||
|
||||
def get_book(sourcefile):
|
||||
book_contents = []
|
||||
book_title = sourcefile
|
||||
book_author = "Unknown"
|
||||
chapter_titles = []
|
||||
|
||||
with open(sourcefile, "r", encoding="utf-8") as file:
|
||||
current_chapter = {"title": "blank", "paragraphs": []}
|
||||
initialized_first_chapter = False
|
||||
lines_skipped = 0
|
||||
for line in file:
|
||||
|
||||
if lines_skipped < 2 and (line.startswith("Title") or line.startswith("Author")):
|
||||
lines_skipped += 1
|
||||
if line.startswith('Title: '):
|
||||
book_title = line.replace('Title: ', '').strip()
|
||||
elif line.startswith('Author: '):
|
||||
book_author = line.replace('Author: ', '').strip()
|
||||
continue
|
||||
|
||||
line = line.strip()
|
||||
if line.startswith("#"):
|
||||
if current_chapter["paragraphs"] or not initialized_first_chapter:
|
||||
if initialized_first_chapter:
|
||||
book_contents.append(current_chapter)
|
||||
current_chapter = {"title": None, "paragraphs": []}
|
||||
initialized_first_chapter = True
|
||||
chapter_title = line[1:].strip()
|
||||
if any(c.isalnum() for c in chapter_title):
|
||||
current_chapter["title"] = chapter_title
|
||||
chapter_titles.append(current_chapter["title"])
|
||||
else:
|
||||
current_chapter["title"] = "blank"
|
||||
chapter_titles.append("blank")
|
||||
elif line:
|
||||
if not initialized_first_chapter:
|
||||
chapter_titles.append("blank")
|
||||
initialized_first_chapter = True
|
||||
if any(char.isalnum() for char in line):
|
||||
sentences = sent_tokenize(line)
|
||||
cleaned_sentences = [s for s in sentences if any(char.isalnum() for char in s)]
|
||||
line = ' '.join(cleaned_sentences)
|
||||
current_chapter["paragraphs"].append(line)
|
||||
|
||||
# Append the last chapter if it contains any paragraphs.
|
||||
if current_chapter["paragraphs"]:
|
||||
book_contents.append(current_chapter)
|
||||
|
||||
return book_contents, book_title, book_author, chapter_titles
|
||||
|
||||
def sort_key(s):
|
||||
# extract number from the string
|
||||
return int(re.findall(r'\d+', s)[0])
|
||||
|
||||
def check_for_file(filename):
|
||||
if os.path.isfile(filename):
|
||||
print(f"The file '{filename}' already exists.")
|
||||
overwrite = input("Do you want to overwrite the file? (y/n): ")
|
||||
if overwrite.lower() != 'y':
|
||||
print("Exiting without overwriting the file.")
|
||||
sys.exit()
|
||||
else:
|
||||
os.remove(filename)
|
||||
|
||||
def append_silence(tempfile, duration=1200):
|
||||
audio = AudioSegment.from_file(tempfile)
|
||||
# Create a silence segment
|
||||
silence = AudioSegment.silent(duration)
|
||||
# Append the silence segment to the audio
|
||||
combined = audio + silence
|
||||
# Save the combined audio back to file
|
||||
combined.export(tempfile, format="flac")
|
||||
|
||||
def break_long_sentence(sentence, max_length=200):
|
||||
# Split sentence based on commas
|
||||
comma_segments = sentence.split(',')
|
||||
segments = []
|
||||
current_segment = ""
|
||||
for segment in comma_segments:
|
||||
# Check if adding the next segment exceeds max_length
|
||||
temp_segment = current_segment + ("," if current_segment else "") + segment
|
||||
if len(temp_segment) > max_length:
|
||||
# Add the current segment to the list and reset it
|
||||
if current_segment:
|
||||
segments.append(current_segment)
|
||||
# Start a new segment with the current part
|
||||
current_segment = segment.strip()
|
||||
else:
|
||||
# Continue building the current segment
|
||||
current_segment = temp_segment.strip()
|
||||
# Don't forget to add the last segment if it exists
|
||||
if current_segment:
|
||||
segments.append(current_segment)
|
||||
return segments
|
||||
|
||||
def process_large_text(line):
|
||||
# Tokenize the text into sentences
|
||||
sentences = sent_tokenize(line)
|
||||
# Initialize a list to store processed sentences
|
||||
results = []
|
||||
|
||||
i = 0
|
||||
while i < len(sentences):
|
||||
sentence = sentences[i]
|
||||
word_count = len(sentence.split())
|
||||
|
||||
# Combine with the next sentence if this one has fewer than 8 words
|
||||
if word_count < 8 and i + 1 < len(sentences):
|
||||
# Combine the current and next sentence
|
||||
sentence = sentence + ' ' + sentences[i + 1]
|
||||
i += 1 # Skip the next sentence since it's already combined
|
||||
|
||||
if len(sentence) > 500:
|
||||
# Break the long sentences into smaller parts using commas
|
||||
results.extend(break_long_sentence(sentence, max_length=350))
|
||||
else:
|
||||
results.append(sentence)
|
||||
|
||||
i += 1 # Move to the next sentence
|
||||
|
||||
# Before returning, combine last elements if they are too short
|
||||
if results and len(results[-1].split()) < 8:
|
||||
if len(results) > 1:
|
||||
# Combine the last two sentences if they are both short
|
||||
results[-2] += ' ' + results[-1]
|
||||
results.pop()
|
||||
|
||||
return results
|
||||
|
||||
def conditional_sentence_case(sent):
|
||||
# Split the sentence into words
|
||||
words = sent.split()
|
||||
length = len(words)
|
||||
# Iterate through words to check for three consecutive uppercase words
|
||||
for i in range(length - 2):
|
||||
if words[i].isupper() and words[i+1].isupper() and words[i+2].isupper():
|
||||
# Convert the entire sentence to lowercase and capitalize the first letter
|
||||
sent = ' '.join(words).lower().capitalize()
|
||||
break # No need to continue checking once a match is found
|
||||
return sent
|
||||
|
||||
def preprocess_book(book_path):
|
||||
ensure_punkt()
|
||||
book = epub.read_epub(book_path)
|
||||
export(book, book_path)
|
||||
Reference in New Issue
Block a user