diff --git a/pyproject.toml b/pyproject.toml
index f8c2055..36298c7 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -28,6 +28,18 @@ classifiers = [
dependencies = [
"load-dotenv>=0.1.0",
+ "beautifulsoup4",
+ "ebooklib",
+ "kokoro>=0.9.4",
+ "lxml",
+ "mutagen",
+ "nltk",
+ "numpy",
+ "pillow",
+ "pydub",
+ "soundfile",
+ "tqdm",
+ "audioop-lts; python_version >= '3.13'",
]
[project.optional-dependencies]
diff --git a/src/epub_tts/__init__.py b/src/epub_tts/__init__.py
index 6afc9fe..42e1206 100644
--- a/src/epub_tts/__init__.py
+++ b/src/epub_tts/__init__.py
@@ -18,7 +18,7 @@ from pathlib import Path
import os
import pprint
-def log (message):
+def log (message, level=None):
pp = pprint.PrettyPrinter(
indent=4,
width=os.get_terminal_size().columns,
diff --git a/src/epub_tts/__main__.py b/src/epub_tts/__main__.py
index 8fd912c..9543fda 100644
--- a/src/epub_tts/__main__.py
+++ b/src/epub_tts/__main__.py
@@ -23,18 +23,26 @@ optional arguments:
-b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}
Backend to use for TTS
"""
+# stdlib
import os
import sys
from pathlib import Path
from argparse import ArgumentParser
+# Pip packages
from dotenv import load_dotenv
-from . import tts_aedocw, tts_generic
-from . import log, PathLike
+# Local imports
+from . import tts_aedocw, tts_generic, tts_kokoro
+from . import log, PathLike
+from .utils import check_env
+
+# Setup env
load_dotenv() # load environment variables from .env file if present
-backend = {
+# Globals
+WHICH = ["ffmpeg"] # Neded in $PATH
+BACKENDS = {
# keys are the backend names, values are the corresponding TTS classes
"default": tts_generic.GenericTTSBackend,
"edge": tts_aedocw.TTSEdge,
@@ -45,7 +53,8 @@ backend = {
"epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge,
"epub2tts-chatterbox": tts_aedocw.AedocwChatterbox,
"epub2tts-kokoro": tts_aedocw.AedocwKokoro,
- "generic-epub2tts": tts_aedocw.Epub2TTS
+ "generic-epub2tts": tts_aedocw.Epub2TTS,
+ "kokoro": tts_kokoro.KokoroBackend,
}
@@ -57,42 +66,73 @@ def main(args=None):
action="version",
version=f"%(prog)s {__import__('epub_tts').__version__}",
)
- p.add_argument("-i", "--input", help="Input EPUB file", required=True)
- p.add_argument("-o", "--output", help="Output audio file", required=True)
- p.add_argument("-l", "--language", help="Language to use for TTS (see your backend's documentation for available voices)")
- p.add_argument("-v", "--voice", help="Voice to use for TTS (see your backend's documentation for available voices)")
- p.add_argument("--speed", type=float, default=1.0, help="Playback speed multiplier (default: 1.0) (not all backends support this)")
- p.add_argument("--short-pause", type=int, default=None, help="Short pause duration in milliseconds between phrases or sentences (not all backends support this)")
- p.add_argument("--long-pause", type=int, default=None, help="Long pause duration in milliseconds between sections or paragraphs (not all backends support this)")
- p.add_argument("-c", "--cover", help="Path to cover image to embed or use for output metadata")
- p.add_argument("-r", "--replace", action="append", nargs=2, help="Replace text in the intermediate output. Specify pairs of old_text new_text. Can be used multiple times.")
- p.add_argument(
- "-b",
- "--backend",
- help="Backend to use for TTS",
- default="default",
- choices=backend.keys(),
- )
+ # Input options and processing
+ p.add_argument("-i", "--input",
+ nargs="+", required=True,
+ help="Input EPUB file")
+ p.add_argument("-r", "--replace", action="append", nargs=2,
+ help="Replace text in the intermediate output. "
+ "Specify pairs of old_text new_text. "
+ "Can be used multiple times.")
+ p.add_argument("--check-env",action="store_true",
+ help="Check the runtime environment and exit")
+ p.add_argument("-b", "--backend", default="default",
+ choices=BACKENDS.keys(),
+ help="Backend to use for TTS")
+
+ # Output options
+ p.add_argument("-o", "--output",
+ help="Output audio file", required=True)
+ p.add_argument("-c", "--cover",
+ help="Path to cover image to embed or use "
+ "for output metadata")
+
+ # Speech options
+ p.add_argument("-l", "--language",
+ help="Language to use for TTS (see your "
+ "backend's documentation for available voices)")
+ p.add_argument("-v", "--voice",
+ help="Voice to use for TTS (see your backend's "
+ "documentation for available voices)")
+ p.add_argument("--speed", type=float, default=1.0,
+ help="Playback speed multiplier (default: 1.0) "
+ "(not all backends support this)")
+ p.add_argument("--short-pause", type=int, default=None,
+ help="Short pause duration in milliseconds between "
+ "phrases or sentences (not all backends support this)")
+ p.add_argument("--long-pause", type=int, default=None,
+ help="Long pause duration in milliseconds between "
+ "sections or paragraphs (not all backends support this)")
+ p.add_argument("--notitles", action="store_true",
+ help="Do not read chapter titles")
+
+ # Parse
args = p.parse_args(args or sys.argv[1:])
setattr(args, 'replace_map',
{old: new for old, new in args.replace} if args.replace else None)
log(args)
- # log(
- # f"Converting {args.input} to {args.output} "
- # f"using {args.backend} backend "
- # "and args: " + ", ".join(
- # f"{k}={v}" for k, v in vars(args).items()
- # if k not in ("input", "output", "backend") and v is not None)
- # )
-
- #log(args.replace_map)
-
- backend[args.backend](**vars(args)).run(
- args.input, args.output,
- **vars(args),
- )
+ # Execute actions
+ if args.check_env:
+ check_env()
+
+ engine = BACKENDS[args.backend](**vars(args))
+ if len(args.input) > 1 and not output_dest.is_dir():
+ log(f"Output must be a directory when multiple input files are provided", level="error")
+ sys.exit(1)
+ for file in args.input:
+ log(file)
+ if not Path(file).exists():
+ log(f"Input file {file} does not exist", level="error")
+ sys.exit(1)
+ output_dest = Path(args.output)
+ engine.run(
+ file, output_dest,
+ **vars(args),
+ )
+
if __name__ == "__main__":
sys.exit(main())
+
diff --git a/src/epub_tts/postprocess.py b/src/epub_tts/postprocess.py
new file mode 100644
index 0000000..9c8cb64
--- /dev/null
+++ b/src/epub_tts/postprocess.py
@@ -0,0 +1,111 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
+# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
+"""Postprocess audio files into an audiobook.
+"""
+import subprocess
+
+
+
+# stdlib modules
+import os
+import sys
+
+# pip installed packages
+import numpy as np
+import soundfile
+import torch
+from tqdm import tqdm
+from kokoro import KPipeline
+from ebooklib import epub
+import soundfile as sf
+from mutagen import mp4
+from pydub import AudioSegment
+from mutagen import mp4
+# Local imports
+from .tts_generic import GenericTTSBackend
+from . import log, PathLike
+from .preprocess import preprocess_book
+
+def generate_metadata(files, author, title, chapter_titles):
+ chap = 0
+ start_time = 0
+ with open("FFMETADATAFILE", "w") as file:
+ file.write(";FFMETADATA1\n")
+ file.write(f"ARTIST={author}\n")
+ file.write(f"ALBUM={title}\n")
+ file.write(f"TITLE={title}\n")
+ file.write("DESCRIPTION=Made with https://github.com/aedocw/epub2tts-kokoro\n")
+ for file_name in files:
+ duration = get_duration(file_name)
+ file.write("[CHAPTER]\n")
+ file.write("TIMEBASE=1/1000\n")
+ file.write(f"START={start_time}\n")
+ file.write(f"END={start_time + duration}\n")
+ file.write(f"title={chapter_titles[chap]}\n")
+ chap += 1
+ start_time += duration
+
+def get_duration(file_path):
+ audio = AudioSegment.from_file(file_path)
+ duration_milliseconds = len(audio)
+ return duration_milliseconds
+
+def make_m4b(files, sourcefile, speaker):
+ filelist = "filelist.txt"
+ basefile = sourcefile.replace(".txt", "")
+ outputm4a = f"{basefile} ({speaker}).m4a"
+ outputm4b = f"{basefile} ({speaker}).m4b"
+ with open(filelist, "w") as f:
+ for filename in files:
+ filename = filename.replace("'", "'\\''")
+ f.write(f"file '{filename}'\n")
+ ffmpeg_command = [
+ "ffmpeg",
+ "-f",
+ "concat",
+ "-safe",
+ "0",
+ "-i",
+ filelist,
+ "-codec:a",
+ "flac",
+ "-f",
+ "mp4",
+ "-strict",
+ "-2",
+ outputm4a,
+ ]
+ subprocess.run(ffmpeg_command)
+ ffmpeg_command = [
+ "ffmpeg",
+ "-i",
+ outputm4a,
+ "-i",
+ "FFMETADATAFILE",
+ "-map_metadata",
+ "1",
+ "-codec",
+ "aac",
+ outputm4b,
+ ]
+ subprocess.run(ffmpeg_command)
+ os.remove(filelist)
+ os.remove("FFMETADATAFILE")
+ os.remove(outputm4a)
+ for f in files:
+ os.remove(f)
+ return outputm4b
+
+def add_cover(cover_img, filename):
+ try:
+ if os.path.isfile(cover_img):
+ m4b = mp4.MP4(filename)
+ cover_image = open(cover_img, "rb").read()
+ m4b["covr"] = [mp4.MP4Cover(cover_image)]
+ m4b.save()
+ else:
+ print(f"Cover image {cover_img} not found")
+ except:
+ print(f"Cover image {cover_img} not found")
diff --git a/src/epub_tts/preprocess.py b/src/epub_tts/preprocess.py
new file mode 100644
index 0000000..43a3e72
--- /dev/null
+++ b/src/epub_tts/preprocess.py
@@ -0,0 +1,363 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
+# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
+"""Preprocess epub books into txt files.
+"""
+
+# stdlib modules
+import os
+import sys
+import re
+import zipfile
+
+# pip installed packages
+import numpy as np
+import warnings
+from tqdm import tqdm
+from bs4 import BeautifulSoup
+import ebooklib
+from ebooklib import epub
+import soundfile as sf
+from lxml import etree
+
+from PIL import Image
+import nltk
+from nltk.tokenize import sent_tokenize
+
+
+# Local imports
+from .tts_generic import GenericTTSBackend
+from . import log, PathLike
+
+
+
+namespaces = {
+ "calibre":"http://calibre.kovidgoyal.net/2009/metadata",
+ "dc":"http://purl.org/dc/elements/1.1/",
+ "dcterms":"http://purl.org/dc/terms/",
+ "opf":"http://www.idpf.org/2007/opf",
+ "u":"urn:oasis:names:tc:opendocument:xmlns:container",
+ "xsi":"http://www.w3.org/2001/XMLSchema-instance",
+}
+
+warnings.filterwarnings("ignore", module="ebooklib.epub")
+
+def ensure_punkt():
+ try:
+ nltk.data.find("tokenizers/punkt")
+ except LookupError:
+ nltk.download("punkt")
+ try:
+ nltk.data.find("tokenizers/punkt_tab")
+ except LookupError:
+ nltk.download("punkt_tab")
+
+
+def chap2text_epub(chap, item_id=None, toc=None):
+ """
+ Extract chapter title and paragraphs from an EPUB chapter.
+
+ Args:
+ chap: The chapter content (HTML).
+ item_id: The ID of the item in the EPUB spine (for fallback naming).
+ toc: The EPUB's table of contents (for fallback title extraction).
+
+ Returns:
+ tuple: (chapter_title_text, paragraphs)
+ """
+ blacklist = [
+ "[document]",
+ "noscript",
+ "header",
+ "html",
+ "meta",
+ "head",
+ "input",
+ "script",
+ ]
+ paragraphs = []
+ soup = BeautifulSoup(chap, "html.parser")
+
+ # Step 1: Try to find chapter title in heading tags (
, , )
+ heading_tags = ['h1', 'h2', 'h3']
+ chapter_title_text = None
+ for tag in heading_tags:
+ heading = soup.find(tag)
+ if heading and heading.text.strip():
+ chapter_title_text = heading.text.strip()
+ print(f"Found title in <{tag}>: '{chapter_title_text}'")
+ break
+
+ # Step 2: If no heading found, try elements with common class names
+ if not chapter_title_text:
+ common_classes = ['chapter', 'chapter-title', 'title', 'heading']
+ for class_name in common_classes:
+ element = soup.find(class_=class_name)
+ if element and element.text.strip():
+ chapter_title_text = element.text.strip()
+ print(f"Found title in class '{class_name}': '{chapter_title_text}'")
+ break
+
+ # Step 3: Fallback to TOC if provided
+ if not chapter_title_text and toc and item_id:
+ for toc_item in toc:
+ if toc_item.href.split('#')[0] == item_id:
+ chapter_title_text = toc_item.title
+ print(f"Found title in TOC for item '{item_id}': '{chapter_title_text}'")
+ break
+
+ # Step 4: Fallback to item ID or generic name
+ if not chapter_title_text:
+ chapter_title_text = item_id.replace('.xhtml', '').replace('_', ' ').title() if item_id else None
+ print(f"No title found, using fallback: '{chapter_title_text}'")
+
+ # Remove footnotes (links with only numbers)
+ for a in soup.findAll("a", href=True):
+ if not any(char.isalpha() for char in a.text):
+ a.extract()
+
+ # Remove superscript numbers (e.g., footnote markers)
+ for sup in soup.findAll("sup"):
+ if sup.text.isdigit():
+ sup.extract()
+
+ # Extract paragraphs
+ chapter_paragraphs = soup.find_all("p")
+ if not chapter_paragraphs:
+ print(f"No
tags found in '{chapter_title_text or item_id}'. Trying
.")
+ chapter_paragraphs = soup.find_all("div")
+
+ for p in chapter_paragraphs:
+ paragraph_text = "".join(p.strings).strip()
+ if paragraph_text:
+ paragraphs.append(paragraph_text)
+
+ return chapter_title_text, paragraphs
+
+def get_epub_cover(epub_path):
+ try:
+ with zipfile.ZipFile(epub_path) as z:
+ t = etree.fromstring(z.read("META-INF/container.xml"))
+ rootfile_path = t.xpath("/u:container/u:rootfiles/u:rootfile",
+ namespaces=namespaces)[0].get("full-path")
+
+ t = etree.fromstring(z.read(rootfile_path))
+ cover_meta = t.xpath("//opf:metadata/opf:meta[@name='cover']",
+ namespaces=namespaces)
+ if not cover_meta:
+ print("No cover image found.")
+ return None
+ cover_id = cover_meta[0].get("content")
+
+ cover_item = t.xpath("//opf:manifest/opf:item[@id='" + cover_id + "']",
+ namespaces=namespaces)
+ if not cover_item:
+ print("No cover image found.")
+ return None
+ cover_href = cover_item[0].get("href")
+ cover_path = os.path.join(os.path.dirname(rootfile_path), cover_href)
+ if os.name == 'nt' and '\\' in cover_path:
+ cover_path = cover_path.replace("\\", "/")
+ return z.open(cover_path)
+ except FileNotFoundError:
+ print(f"Could not get cover image of {epub_path}")
+
+def export(book, sourcefile):
+ book_contents = []
+ cover_image = get_epub_cover(sourcefile)
+ image_path = None
+
+ if cover_image is not None:
+ image = Image.open(cover_image)
+ image_filename = sourcefile.replace(".epub", ".png")
+ image_path = os.path.join(image_filename)
+ image.save(image_path)
+ print(f"Cover image saved to {image_path}")
+
+ # Get the table of contents
+ toc = book.get_toc() if hasattr(book, 'get_toc') else []
+
+ spine_ids = [spine_tuple[0] for spine_tuple in book.spine if spine_tuple[1] == 'yes']
+ items = {item.get_id(): item for item in book.get_items() if item.get_type() == ebooklib.ITEM_DOCUMENT}
+
+ for id in spine_ids:
+ item = items.get(id)
+ if item is None:
+ continue
+ # Pass item_id and toc to chap2text_epub
+ chapter_title, chapter_paragraphs = chap2text_epub(item.get_content(), item_id=id, toc=toc)
+ book_contents.append({"title": chapter_title, "paragraphs": chapter_paragraphs})
+
+ outfile = sourcefile.replace(".epub", ".txt")
+ check_for_file(outfile)
+ print(f"Exporting {sourcefile} to {outfile}")
+ author = book.get_metadata("DC", "creator")[0][0]
+ booktitle = book.get_metadata("DC", "title")[0][0]
+
+ with open(outfile, "w", encoding='utf-8') as file:
+ file.write(f"Title: {booktitle}\n")
+ file.write(f"Author: {author}\n\n")
+ file.write(f"# Title\n")
+ file.write(f"{booktitle}, by {author}\n\n")
+ for i, chapter in enumerate(book_contents, start=1):
+ if not chapter["paragraphs"] or chapter["paragraphs"] == ['']:
+ continue
+ else:
+ # Use chapter title if available, otherwise fallback to "Part {i}"
+ title = chapter["title"] if chapter["title"] else f"Part {i}"
+ file.write(f"# {title}\n\n")
+ for paragraph in chapter["paragraphs"]:
+ clean = re.sub(r'[\s\n]+', ' ', paragraph)
+ clean = re.sub(r'[“”]', '"', clean) # Curly double quotes to standard double quotes
+ clean = re.sub(r'[‘’]', "'", clean) # Curly single quotes to standard single quotes
+ clean = re.sub(r'--', ', ', clean)
+ file.write(f"{clean}\n\n")
+
+ return book_contents
+
+def get_book(sourcefile):
+ book_contents = []
+ book_title = sourcefile
+ book_author = "Unknown"
+ chapter_titles = []
+
+ with open(sourcefile, "r", encoding="utf-8") as file:
+ current_chapter = {"title": "blank", "paragraphs": []}
+ initialized_first_chapter = False
+ lines_skipped = 0
+ for line in file:
+
+ if lines_skipped < 2 and (line.startswith("Title") or line.startswith("Author")):
+ lines_skipped += 1
+ if line.startswith('Title: '):
+ book_title = line.replace('Title: ', '').strip()
+ elif line.startswith('Author: '):
+ book_author = line.replace('Author: ', '').strip()
+ continue
+
+ line = line.strip()
+ if line.startswith("#"):
+ if current_chapter["paragraphs"] or not initialized_first_chapter:
+ if initialized_first_chapter:
+ book_contents.append(current_chapter)
+ current_chapter = {"title": None, "paragraphs": []}
+ initialized_first_chapter = True
+ chapter_title = line[1:].strip()
+ if any(c.isalnum() for c in chapter_title):
+ current_chapter["title"] = chapter_title
+ chapter_titles.append(current_chapter["title"])
+ else:
+ current_chapter["title"] = "blank"
+ chapter_titles.append("blank")
+ elif line:
+ if not initialized_first_chapter:
+ chapter_titles.append("blank")
+ initialized_first_chapter = True
+ if any(char.isalnum() for char in line):
+ sentences = sent_tokenize(line)
+ cleaned_sentences = [s for s in sentences if any(char.isalnum() for char in s)]
+ line = ' '.join(cleaned_sentences)
+ current_chapter["paragraphs"].append(line)
+
+ # Append the last chapter if it contains any paragraphs.
+ if current_chapter["paragraphs"]:
+ book_contents.append(current_chapter)
+
+ return book_contents, book_title, book_author, chapter_titles
+
+def sort_key(s):
+ # extract number from the string
+ return int(re.findall(r'\d+', s)[0])
+
+def check_for_file(filename):
+ if os.path.isfile(filename):
+ print(f"The file '{filename}' already exists.")
+ overwrite = input("Do you want to overwrite the file? (y/n): ")
+ if overwrite.lower() != 'y':
+ print("Exiting without overwriting the file.")
+ sys.exit()
+ else:
+ os.remove(filename)
+
+def append_silence(tempfile, duration=1200):
+ audio = AudioSegment.from_file(tempfile)
+ # Create a silence segment
+ silence = AudioSegment.silent(duration)
+ # Append the silence segment to the audio
+ combined = audio + silence
+ # Save the combined audio back to file
+ combined.export(tempfile, format="flac")
+
+def break_long_sentence(sentence, max_length=200):
+ # Split sentence based on commas
+ comma_segments = sentence.split(',')
+ segments = []
+ current_segment = ""
+ for segment in comma_segments:
+ # Check if adding the next segment exceeds max_length
+ temp_segment = current_segment + ("," if current_segment else "") + segment
+ if len(temp_segment) > max_length:
+ # Add the current segment to the list and reset it
+ if current_segment:
+ segments.append(current_segment)
+ # Start a new segment with the current part
+ current_segment = segment.strip()
+ else:
+ # Continue building the current segment
+ current_segment = temp_segment.strip()
+ # Don't forget to add the last segment if it exists
+ if current_segment:
+ segments.append(current_segment)
+ return segments
+
+def process_large_text(line):
+ # Tokenize the text into sentences
+ sentences = sent_tokenize(line)
+ # Initialize a list to store processed sentences
+ results = []
+
+ i = 0
+ while i < len(sentences):
+ sentence = sentences[i]
+ word_count = len(sentence.split())
+
+ # Combine with the next sentence if this one has fewer than 8 words
+ if word_count < 8 and i + 1 < len(sentences):
+ # Combine the current and next sentence
+ sentence = sentence + ' ' + sentences[i + 1]
+ i += 1 # Skip the next sentence since it's already combined
+
+ if len(sentence) > 500:
+ # Break the long sentences into smaller parts using commas
+ results.extend(break_long_sentence(sentence, max_length=350))
+ else:
+ results.append(sentence)
+
+ i += 1 # Move to the next sentence
+
+ # Before returning, combine last elements if they are too short
+ if results and len(results[-1].split()) < 8:
+ if len(results) > 1:
+ # Combine the last two sentences if they are both short
+ results[-2] += ' ' + results[-1]
+ results.pop()
+
+ return results
+
+def conditional_sentence_case(sent):
+ # Split the sentence into words
+ words = sent.split()
+ length = len(words)
+ # Iterate through words to check for three consecutive uppercase words
+ for i in range(length - 2):
+ if words[i].isupper() and words[i+1].isupper() and words[i+2].isupper():
+ # Convert the entire sentence to lowercase and capitalize the first letter
+ sent = ' '.join(words).lower().capitalize()
+ break # No need to continue checking once a match is found
+ return sent
+
+def preprocess_book(book_path):
+ ensure_punkt()
+ book = epub.read_epub(book_path)
+ export(book, book_path)
\ No newline at end of file
diff --git a/src/epub_tts/sample_download.py b/src/epub_tts/sample_download.py
index 2641a2b..04c32d5 100644
--- a/src/epub_tts/sample_download.py
+++ b/src/epub_tts/sample_download.py
@@ -13,7 +13,7 @@ import zipfile
from pathlib import Path
from typing import Optional
-from epub_tts.vendor_utils import get_repo_path
+from epub_tts.utils import get_repo_path
def get_project_root() -> Path:
diff --git a/src/epub_tts/tts_generic.py b/src/epub_tts/tts_generic.py
index ee73a63..9fd5f16 100644
--- a/src/epub_tts/tts_generic.py
+++ b/src/epub_tts/tts_generic.py
@@ -34,7 +34,7 @@ import subprocess
from pathlib import Path
from typing import Optional, Callable
-from .vendor_utils import build_run_command, ensure_venv
+from .utils import build_run_command, ensure_venv
from . import log, PathLike
class GenericTTSBackend:
diff --git a/src/epub_tts/tts_kokoro.py b/src/epub_tts/tts_kokoro.py
new file mode 100644
index 0000000..a8e4a40
--- /dev/null
+++ b/src/epub_tts/tts_kokoro.py
@@ -0,0 +1,205 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
+# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
+"""Backend for the kokoro engine.
+No vendored repository needed, kokoro is a pure Python package that can be installed via pip.
+
+"""
+
+# stdlib modules
+import subprocess
+import os
+import sys
+# Automatically enable MPS fallback on Apple Silicon macOS
+if sys.platform == 'darwin':
+ os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
+
+# pip installed packages
+import numpy as np
+import soundfile
+import torch
+from tqdm import tqdm
+from kokoro import KPipeline
+from ebooklib import epub
+import soundfile as sf
+from mutagen import mp4
+from pydub import AudioSegment
+from mutagen import mp4
+# Local imports
+from .tts_generic import GenericTTSBackend
+from . import log, PathLike
+from .preprocess import preprocess_book
+
+class KokoroBackend(GenericTTSBackend):
+ """Generic kokoro backend wrapper.
+
+ Parameters:
+ repo: repository name (one of the keys in _CMD_MAP)
+ backend_cmd: explicit command/executable to use (overrides repo mapping)
+ language: optional language code
+ voice: optional voice name
+
+ The GenericTTSBackend stores the default command name in self.backend_cmd,
+ but the vendored adapters may still run the package in their own venv if
+ the console script is not present.
+ """
+
+ INTERMEDIATE_TXT = True
+ INTERMEDIATE_CALL = GenericTTSBackend._replace_map
+ DEFAULT_SPEAKER = "am_liam" # ["af_heart", "am_michael", "am_liam"]
+
+ def run(self,
+ input_source: PathLike,
+ output_dest: PathLike,
+ **kwargs
+ ) -> subprocess.CompletedProcess:
+
+ command = self._build_command(input_source, output_dest)
+ log(command)
+
+ # # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first
+ # if not input_source.endswith(".txt"):
+ # completed = subprocess.run(
+ # self._build_command(input_source, output_dest),
+ # cwd=str(self.CWD),
+ # env=self.ENV,
+ # check=True)
+
+ # if self.INTERMEDIATE_TXT:
+ # txt_file = self._normalize_path(input_source).with_suffix(".txt")
+ # if txt_file.exists():
+ # if (self.INTERMEDIATE_CALL is not None and
+ # kwargs.get('replace_map', None) is not None):
+ # self.INTERMEDIATE_CALL(
+ # txt_file,
+ # txt_file.with_stem(txt_file.stem + "_replaced"), kwargs.get('replace_map', {}))
+ # txt_file = txt_file.with_stem(txt_file.stem + "_replaced")
+ # completed = subprocess.run(
+ # self._build_command(txt_file, output_dest),
+ # cwd=str(self.CWD),
+ # env=self.ENV,
+ # check=True
+ # )
+
+
+ # If we get an epub, export that to txt file
+ if input_source.endswith(".epub"):
+ book = preprocess_book(input_source)
+
+ # Check for GPU
+ if torch.cuda.is_available():
+ print('Nvidia GPU available. Setting as default device.')
+ torch.set_default_device('cuda')
+ elif torch.xpu.is_available():
+ print('Intel XPU (GPU) available. Setting as default device.')
+ torch.set_default_device('xpu')
+ elif torch.backends.mps.is_available():
+ print('Apple MPS GPU available. Setting as default device.')
+ torch.set_default_device('mps')
+ elif torch.backends.rocm.is_available():
+ print('AMD ROCm GPU available. Setting as default device.')
+ torch.set_default_device('rocm')
+ elif torch.is_vulkan_available():
+ print('Vulkan GPU available. Setting as default device.')
+ torch.set_default_device('vulkan')
+ else:
+ print('No GPU available. Using CPU.')
+ torch.set_default_device('cpu')
+
+
+
+ book_contents, book_title, book_author, chapter_titles = get_book(args.sourcefile)
+ files = read_book(book_contents, args.speaker, args.paragraphpause, args.speed, args.notitles)
+ generate_metadata(files, book_author, book_title, chapter_titles)
+ m4bfilename = make_m4b(files, args.sourcefile, args.speaker)
+ add_cover(args.cover, m4bfilename)
+
+
+
+
+ def kokoro_read(paragraph, speaker, filename, pipeline, speed):
+ audio_segments = []
+ sentences = process_large_text(paragraph)
+ for sent in sentences:
+ sent = conditional_sentence_case(sent.strip())
+ for gs, ps, audio in pipeline(sent, voice=speaker, speed=speed, split_pattern=r'\n\n\n'):
+ audio_segments.append(audio)
+
+ final_audio = np.concatenate(audio_segments)
+ soundfile.write(filename, final_audio, 24000)
+
+ def read_book(book_contents, speaker, paragraphpause, speed, notitles):
+ current_device_name = torch.get_default_device() if torch.get_default_device() else 'cpu'
+ current_device = torch.device(current_device_name)
+ print(f"Attempting to use device: {current_device}")
+
+ pipeline = KPipeline(lang_code=speaker[0])
+
+ # Explicitly move the model to the current default device (e.g., 'xpu')
+ if hasattr(pipeline, 'model') and pipeline.model is not None:
+ try:
+ pipeline.model.to(current_device)
+ print(f"Kokoro model explicitly moved to {current_device}")
+ except Exception as e:
+ print(f"Error moving Kokoro model to {current_device}: {e}")
+ else:
+ print("Warning: KPipeline does not have a 'model' attribute or model is None.")
+
+ segments = []
+ for i, chapter in enumerate(book_contents, start=1):
+ files = []
+ partname = f"part{i}.flac"
+ print(f"\n\n")
+
+ if os.path.isfile(partname):
+ print(f"{partname} exists, skipping to next chapter")
+ segments.append(partname)
+ else:
+ print(f"Chapter: {chapter['title']}\n")
+ print(f"Section name: \"{chapter['title']}\"")
+ if chapter["title"] == "":
+ chapter["title"] = "blank"
+ if chapter["title"] != "Title" and notitles != True:
+ title_temp = "title.flac"
+ if not os.path.isfile(title_temp):
+ kokoro_read(chapter['title'] + ".", speaker, "title_temp.wav", pipeline, speed)
+ append_silence("title_temp.wav", paragraphpause)
+ # Convert to flac
+ audio = AudioSegment.from_file("title_temp.wav")
+ audio.export(title_temp, format="flac")
+ os.remove("title_temp.wav")
+ files.append(title_temp)
+
+ for pindex, paragraph in enumerate(
+ tqdm(chapter["paragraphs"], desc=f"Generating audio files: ",unit='pg')
+ ):
+ ptemp = f"pgraphs{pindex}.flac"
+ if os.path.isfile(ptemp):
+ print(f"{ptemp} exists, skipping to next paragraph")
+ else:
+ #sentences = sent_tokenize(paragraph)
+ filenames = ["sntnc1.wav"]
+ kokoro_read(paragraph, speaker, "sntnc1.wav", pipeline, speed)
+ append_silence("sntnc1.wav", paragraphpause)
+ # combine sentences in paragraph
+ sorted_files = sorted(filenames, key=sort_key)
+ if os.path.exists("sntnc0.wav"):
+ sorted_files.insert(0, "sntnc0.wav")
+ combined = AudioSegment.empty()
+ for file in sorted_files:
+ combined += AudioSegment.from_file(file)
+ combined.export(ptemp, format="flac")
+ for file in sorted_files:
+ os.remove(file)
+ files.append(ptemp)
+ # combine paragraphs into chapter
+ append_silence(files[-1], 2000)
+ combined = AudioSegment.empty()
+ for file in files:
+ combined += AudioSegment.from_file(file)
+ combined.export(partname, format="flac")
+ for file in files:
+ os.remove(file)
+ segments.append(partname)
+ return segments
diff --git a/src/epub_tts/vendor_utils.py b/src/epub_tts/utils.py
similarity index 82%
rename from src/epub_tts/vendor_utils.py
rename to src/epub_tts/utils.py
index 4b9c318..75158a2 100644
--- a/src/epub_tts/vendor_utils.py
+++ b/src/epub_tts/utils.py
@@ -6,6 +6,7 @@
"""Utilities for running vendored EPUB->TTS repositories in isolated venvs."""
import subprocess
+import shutil
import sys
import venv
from pathlib import Path
@@ -71,3 +72,21 @@ def build_run_command(
# else str(venv_path / "bin" / backend_cmd)
# )
+
+
+def check_env(self, which=None):
+ if not which:
+ return None
+ if isinstance(which, str):
+ if " " in which:
+ which = which.split()
+ which = [which]
+ for bin in which:
+ p = shutil.which(bin)
+ if not p:
+ log(f"{bin} is either NOT installed or "
+ "NOT in your system PATH.")
+ return False
+ else:
+ log(f"{bin} found at: {p}")
+ return True
\ No newline at end of file