tts-kokoro engine underway
This commit is contained in:
@@ -0,0 +1,205 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
||||
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
||||
"""Backend for the kokoro engine.
|
||||
No vendored repository needed, kokoro is a pure Python package that can be installed via pip.
|
||||
|
||||
"""
|
||||
|
||||
# stdlib modules
|
||||
import subprocess
|
||||
import os
|
||||
import sys
|
||||
# Automatically enable MPS fallback on Apple Silicon macOS
|
||||
if sys.platform == 'darwin':
|
||||
os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
|
||||
|
||||
# pip installed packages
|
||||
import numpy as np
|
||||
import soundfile
|
||||
import torch
|
||||
from tqdm import tqdm
|
||||
from kokoro import KPipeline
|
||||
from ebooklib import epub
|
||||
import soundfile as sf
|
||||
from mutagen import mp4
|
||||
from pydub import AudioSegment
|
||||
from mutagen import mp4
|
||||
# Local imports
|
||||
from .tts_generic import GenericTTSBackend
|
||||
from . import log, PathLike
|
||||
from .preprocess import preprocess_book
|
||||
|
||||
class KokoroBackend(GenericTTSBackend):
|
||||
"""Generic kokoro backend wrapper.
|
||||
|
||||
Parameters:
|
||||
repo: repository name (one of the keys in _CMD_MAP)
|
||||
backend_cmd: explicit command/executable to use (overrides repo mapping)
|
||||
language: optional language code
|
||||
voice: optional voice name
|
||||
|
||||
The GenericTTSBackend stores the default command name in self.backend_cmd,
|
||||
but the vendored adapters may still run the package in their own venv if
|
||||
the console script is not present.
|
||||
"""
|
||||
|
||||
INTERMEDIATE_TXT = True
|
||||
INTERMEDIATE_CALL = GenericTTSBackend._replace_map
|
||||
DEFAULT_SPEAKER = "am_liam" # ["af_heart", "am_michael", "am_liam"]
|
||||
|
||||
def run(self,
|
||||
input_source: PathLike,
|
||||
output_dest: PathLike,
|
||||
**kwargs
|
||||
) -> subprocess.CompletedProcess:
|
||||
|
||||
command = self._build_command(input_source, output_dest)
|
||||
log(command)
|
||||
|
||||
# # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first
|
||||
# if not input_source.endswith(".txt"):
|
||||
# completed = subprocess.run(
|
||||
# self._build_command(input_source, output_dest),
|
||||
# cwd=str(self.CWD),
|
||||
# env=self.ENV,
|
||||
# check=True)
|
||||
|
||||
# if self.INTERMEDIATE_TXT:
|
||||
# txt_file = self._normalize_path(input_source).with_suffix(".txt")
|
||||
# if txt_file.exists():
|
||||
# if (self.INTERMEDIATE_CALL is not None and
|
||||
# kwargs.get('replace_map', None) is not None):
|
||||
# self.INTERMEDIATE_CALL(
|
||||
# txt_file,
|
||||
# txt_file.with_stem(txt_file.stem + "_replaced"), kwargs.get('replace_map', {}))
|
||||
# txt_file = txt_file.with_stem(txt_file.stem + "_replaced")
|
||||
# completed = subprocess.run(
|
||||
# self._build_command(txt_file, output_dest),
|
||||
# cwd=str(self.CWD),
|
||||
# env=self.ENV,
|
||||
# check=True
|
||||
# )
|
||||
|
||||
|
||||
# If we get an epub, export that to txt file
|
||||
if input_source.endswith(".epub"):
|
||||
book = preprocess_book(input_source)
|
||||
|
||||
# Check for GPU
|
||||
if torch.cuda.is_available():
|
||||
print('Nvidia GPU available. Setting as default device.')
|
||||
torch.set_default_device('cuda')
|
||||
elif torch.xpu.is_available():
|
||||
print('Intel XPU (GPU) available. Setting as default device.')
|
||||
torch.set_default_device('xpu')
|
||||
elif torch.backends.mps.is_available():
|
||||
print('Apple MPS GPU available. Setting as default device.')
|
||||
torch.set_default_device('mps')
|
||||
elif torch.backends.rocm.is_available():
|
||||
print('AMD ROCm GPU available. Setting as default device.')
|
||||
torch.set_default_device('rocm')
|
||||
elif torch.is_vulkan_available():
|
||||
print('Vulkan GPU available. Setting as default device.')
|
||||
torch.set_default_device('vulkan')
|
||||
else:
|
||||
print('No GPU available. Using CPU.')
|
||||
torch.set_default_device('cpu')
|
||||
|
||||
|
||||
|
||||
book_contents, book_title, book_author, chapter_titles = get_book(args.sourcefile)
|
||||
files = read_book(book_contents, args.speaker, args.paragraphpause, args.speed, args.notitles)
|
||||
generate_metadata(files, book_author, book_title, chapter_titles)
|
||||
m4bfilename = make_m4b(files, args.sourcefile, args.speaker)
|
||||
add_cover(args.cover, m4bfilename)
|
||||
|
||||
|
||||
|
||||
|
||||
def kokoro_read(paragraph, speaker, filename, pipeline, speed):
|
||||
audio_segments = []
|
||||
sentences = process_large_text(paragraph)
|
||||
for sent in sentences:
|
||||
sent = conditional_sentence_case(sent.strip())
|
||||
for gs, ps, audio in pipeline(sent, voice=speaker, speed=speed, split_pattern=r'\n\n\n'):
|
||||
audio_segments.append(audio)
|
||||
|
||||
final_audio = np.concatenate(audio_segments)
|
||||
soundfile.write(filename, final_audio, 24000)
|
||||
|
||||
def read_book(book_contents, speaker, paragraphpause, speed, notitles):
|
||||
current_device_name = torch.get_default_device() if torch.get_default_device() else 'cpu'
|
||||
current_device = torch.device(current_device_name)
|
||||
print(f"Attempting to use device: {current_device}")
|
||||
|
||||
pipeline = KPipeline(lang_code=speaker[0])
|
||||
|
||||
# Explicitly move the model to the current default device (e.g., 'xpu')
|
||||
if hasattr(pipeline, 'model') and pipeline.model is not None:
|
||||
try:
|
||||
pipeline.model.to(current_device)
|
||||
print(f"Kokoro model explicitly moved to {current_device}")
|
||||
except Exception as e:
|
||||
print(f"Error moving Kokoro model to {current_device}: {e}")
|
||||
else:
|
||||
print("Warning: KPipeline does not have a 'model' attribute or model is None.")
|
||||
|
||||
segments = []
|
||||
for i, chapter in enumerate(book_contents, start=1):
|
||||
files = []
|
||||
partname = f"part{i}.flac"
|
||||
print(f"\n\n")
|
||||
|
||||
if os.path.isfile(partname):
|
||||
print(f"{partname} exists, skipping to next chapter")
|
||||
segments.append(partname)
|
||||
else:
|
||||
print(f"Chapter: {chapter['title']}\n")
|
||||
print(f"Section name: \"{chapter['title']}\"")
|
||||
if chapter["title"] == "":
|
||||
chapter["title"] = "blank"
|
||||
if chapter["title"] != "Title" and notitles != True:
|
||||
title_temp = "title.flac"
|
||||
if not os.path.isfile(title_temp):
|
||||
kokoro_read(chapter['title'] + ".", speaker, "title_temp.wav", pipeline, speed)
|
||||
append_silence("title_temp.wav", paragraphpause)
|
||||
# Convert to flac
|
||||
audio = AudioSegment.from_file("title_temp.wav")
|
||||
audio.export(title_temp, format="flac")
|
||||
os.remove("title_temp.wav")
|
||||
files.append(title_temp)
|
||||
|
||||
for pindex, paragraph in enumerate(
|
||||
tqdm(chapter["paragraphs"], desc=f"Generating audio files: ",unit='pg')
|
||||
):
|
||||
ptemp = f"pgraphs{pindex}.flac"
|
||||
if os.path.isfile(ptemp):
|
||||
print(f"{ptemp} exists, skipping to next paragraph")
|
||||
else:
|
||||
#sentences = sent_tokenize(paragraph)
|
||||
filenames = ["sntnc1.wav"]
|
||||
kokoro_read(paragraph, speaker, "sntnc1.wav", pipeline, speed)
|
||||
append_silence("sntnc1.wav", paragraphpause)
|
||||
# combine sentences in paragraph
|
||||
sorted_files = sorted(filenames, key=sort_key)
|
||||
if os.path.exists("sntnc0.wav"):
|
||||
sorted_files.insert(0, "sntnc0.wav")
|
||||
combined = AudioSegment.empty()
|
||||
for file in sorted_files:
|
||||
combined += AudioSegment.from_file(file)
|
||||
combined.export(ptemp, format="flac")
|
||||
for file in sorted_files:
|
||||
os.remove(file)
|
||||
files.append(ptemp)
|
||||
# combine paragraphs into chapter
|
||||
append_silence(files[-1], 2000)
|
||||
combined = AudioSegment.empty()
|
||||
for file in files:
|
||||
combined += AudioSegment.from_file(file)
|
||||
combined.export(partname, format="flac")
|
||||
for file in files:
|
||||
os.remove(file)
|
||||
segments.append(partname)
|
||||
return segments
|
||||
Reference in New Issue
Block a user