tts-kokoro engine underway

This commit is contained in:
2026-08-07 19:00:45 -03:00
parent a1f564a051
commit 2777b34026
9 changed files with 787 additions and 37 deletions
+205
View File
@@ -0,0 +1,205 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
"""Backend for the kokoro engine.
No vendored repository needed, kokoro is a pure Python package that can be installed via pip.
"""
# stdlib modules
import subprocess
import os
import sys
# Automatically enable MPS fallback on Apple Silicon macOS
if sys.platform == 'darwin':
os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
# pip installed packages
import numpy as np
import soundfile
import torch
from tqdm import tqdm
from kokoro import KPipeline
from ebooklib import epub
import soundfile as sf
from mutagen import mp4
from pydub import AudioSegment
from mutagen import mp4
# Local imports
from .tts_generic import GenericTTSBackend
from . import log, PathLike
from .preprocess import preprocess_book
class KokoroBackend(GenericTTSBackend):
"""Generic kokoro backend wrapper.
Parameters:
repo: repository name (one of the keys in _CMD_MAP)
backend_cmd: explicit command/executable to use (overrides repo mapping)
language: optional language code
voice: optional voice name
The GenericTTSBackend stores the default command name in self.backend_cmd,
but the vendored adapters may still run the package in their own venv if
the console script is not present.
"""
INTERMEDIATE_TXT = True
INTERMEDIATE_CALL = GenericTTSBackend._replace_map
DEFAULT_SPEAKER = "am_liam" # ["af_heart", "am_michael", "am_liam"]
def run(self,
input_source: PathLike,
output_dest: PathLike,
**kwargs
) -> subprocess.CompletedProcess:
command = self._build_command(input_source, output_dest)
log(command)
# # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first
# if not input_source.endswith(".txt"):
# completed = subprocess.run(
# self._build_command(input_source, output_dest),
# cwd=str(self.CWD),
# env=self.ENV,
# check=True)
# if self.INTERMEDIATE_TXT:
# txt_file = self._normalize_path(input_source).with_suffix(".txt")
# if txt_file.exists():
# if (self.INTERMEDIATE_CALL is not None and
# kwargs.get('replace_map', None) is not None):
# self.INTERMEDIATE_CALL(
# txt_file,
# txt_file.with_stem(txt_file.stem + "_replaced"), kwargs.get('replace_map', {}))
# txt_file = txt_file.with_stem(txt_file.stem + "_replaced")
# completed = subprocess.run(
# self._build_command(txt_file, output_dest),
# cwd=str(self.CWD),
# env=self.ENV,
# check=True
# )
# If we get an epub, export that to txt file
if input_source.endswith(".epub"):
book = preprocess_book(input_source)
# Check for GPU
if torch.cuda.is_available():
print('Nvidia GPU available. Setting as default device.')
torch.set_default_device('cuda')
elif torch.xpu.is_available():
print('Intel XPU (GPU) available. Setting as default device.')
torch.set_default_device('xpu')
elif torch.backends.mps.is_available():
print('Apple MPS GPU available. Setting as default device.')
torch.set_default_device('mps')
elif torch.backends.rocm.is_available():
print('AMD ROCm GPU available. Setting as default device.')
torch.set_default_device('rocm')
elif torch.is_vulkan_available():
print('Vulkan GPU available. Setting as default device.')
torch.set_default_device('vulkan')
else:
print('No GPU available. Using CPU.')
torch.set_default_device('cpu')
book_contents, book_title, book_author, chapter_titles = get_book(args.sourcefile)
files = read_book(book_contents, args.speaker, args.paragraphpause, args.speed, args.notitles)
generate_metadata(files, book_author, book_title, chapter_titles)
m4bfilename = make_m4b(files, args.sourcefile, args.speaker)
add_cover(args.cover, m4bfilename)
def kokoro_read(paragraph, speaker, filename, pipeline, speed):
audio_segments = []
sentences = process_large_text(paragraph)
for sent in sentences:
sent = conditional_sentence_case(sent.strip())
for gs, ps, audio in pipeline(sent, voice=speaker, speed=speed, split_pattern=r'\n\n\n'):
audio_segments.append(audio)
final_audio = np.concatenate(audio_segments)
soundfile.write(filename, final_audio, 24000)
def read_book(book_contents, speaker, paragraphpause, speed, notitles):
current_device_name = torch.get_default_device() if torch.get_default_device() else 'cpu'
current_device = torch.device(current_device_name)
print(f"Attempting to use device: {current_device}")
pipeline = KPipeline(lang_code=speaker[0])
# Explicitly move the model to the current default device (e.g., 'xpu')
if hasattr(pipeline, 'model') and pipeline.model is not None:
try:
pipeline.model.to(current_device)
print(f"Kokoro model explicitly moved to {current_device}")
except Exception as e:
print(f"Error moving Kokoro model to {current_device}: {e}")
else:
print("Warning: KPipeline does not have a 'model' attribute or model is None.")
segments = []
for i, chapter in enumerate(book_contents, start=1):
files = []
partname = f"part{i}.flac"
print(f"\n\n")
if os.path.isfile(partname):
print(f"{partname} exists, skipping to next chapter")
segments.append(partname)
else:
print(f"Chapter: {chapter['title']}\n")
print(f"Section name: \"{chapter['title']}\"")
if chapter["title"] == "":
chapter["title"] = "blank"
if chapter["title"] != "Title" and notitles != True:
title_temp = "title.flac"
if not os.path.isfile(title_temp):
kokoro_read(chapter['title'] + ".", speaker, "title_temp.wav", pipeline, speed)
append_silence("title_temp.wav", paragraphpause)
# Convert to flac
audio = AudioSegment.from_file("title_temp.wav")
audio.export(title_temp, format="flac")
os.remove("title_temp.wav")
files.append(title_temp)
for pindex, paragraph in enumerate(
tqdm(chapter["paragraphs"], desc=f"Generating audio files: ",unit='pg')
):
ptemp = f"pgraphs{pindex}.flac"
if os.path.isfile(ptemp):
print(f"{ptemp} exists, skipping to next paragraph")
else:
#sentences = sent_tokenize(paragraph)
filenames = ["sntnc1.wav"]
kokoro_read(paragraph, speaker, "sntnc1.wav", pipeline, speed)
append_silence("sntnc1.wav", paragraphpause)
# combine sentences in paragraph
sorted_files = sorted(filenames, key=sort_key)
if os.path.exists("sntnc0.wav"):
sorted_files.insert(0, "sntnc0.wav")
combined = AudioSegment.empty()
for file in sorted_files:
combined += AudioSegment.from_file(file)
combined.export(ptemp, format="flac")
for file in sorted_files:
os.remove(file)
files.append(ptemp)
# combine paragraphs into chapter
append_silence(files[-1], 2000)
combined = AudioSegment.empty()
for file in files:
combined += AudioSegment.from_file(file)
combined.export(partname, format="flac")
for file in files:
os.remove(file)
segments.append(partname)
return segments