206 lines
8.6 KiB
Python
206 lines
8.6 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
|
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
|
"""Backend for the kokoro engine.
|
|
No vendored repository needed, kokoro is a pure Python package that can be installed via pip.
|
|
|
|
"""
|
|
|
|
# stdlib modules
|
|
import subprocess
|
|
import os
|
|
import sys
|
|
# Automatically enable MPS fallback on Apple Silicon macOS
|
|
if sys.platform == 'darwin':
|
|
os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
|
|
|
|
# pip installed packages
|
|
import numpy as np
|
|
import soundfile
|
|
import torch
|
|
from tqdm import tqdm
|
|
from kokoro import KPipeline
|
|
from ebooklib import epub
|
|
import soundfile as sf
|
|
from mutagen import mp4
|
|
from pydub import AudioSegment
|
|
from mutagen import mp4
|
|
# Local imports
|
|
from .tts_generic import GenericTTSBackend
|
|
from . import log, PathLike
|
|
from .preprocess import preprocess_book
|
|
|
|
class KokoroBackend(GenericTTSBackend):
|
|
"""Generic kokoro backend wrapper.
|
|
|
|
Parameters:
|
|
repo: repository name (one of the keys in _CMD_MAP)
|
|
backend_cmd: explicit command/executable to use (overrides repo mapping)
|
|
language: optional language code
|
|
voice: optional voice name
|
|
|
|
The GenericTTSBackend stores the default command name in self.backend_cmd,
|
|
but the vendored adapters may still run the package in their own venv if
|
|
the console script is not present.
|
|
"""
|
|
|
|
INTERMEDIATE_TXT = True
|
|
INTERMEDIATE_CALL = GenericTTSBackend._replace_map
|
|
DEFAULT_SPEAKER = "am_liam" # ["af_heart", "am_michael", "am_liam"]
|
|
|
|
def run(self,
|
|
input_source: PathLike,
|
|
output_dest: PathLike,
|
|
**kwargs
|
|
) -> subprocess.CompletedProcess:
|
|
|
|
command = self._build_command(input_source, output_dest)
|
|
log(command)
|
|
|
|
# # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first
|
|
# if not input_source.endswith(".txt"):
|
|
# completed = subprocess.run(
|
|
# self._build_command(input_source, output_dest),
|
|
# cwd=str(self.CWD),
|
|
# env=self.ENV,
|
|
# check=True)
|
|
|
|
# if self.INTERMEDIATE_TXT:
|
|
# txt_file = self._normalize_path(input_source).with_suffix(".txt")
|
|
# if txt_file.exists():
|
|
# if (self.INTERMEDIATE_CALL is not None and
|
|
# kwargs.get('replace_map', None) is not None):
|
|
# self.INTERMEDIATE_CALL(
|
|
# txt_file,
|
|
# txt_file.with_stem(txt_file.stem + "_replaced"), kwargs.get('replace_map', {}))
|
|
# txt_file = txt_file.with_stem(txt_file.stem + "_replaced")
|
|
# completed = subprocess.run(
|
|
# self._build_command(txt_file, output_dest),
|
|
# cwd=str(self.CWD),
|
|
# env=self.ENV,
|
|
# check=True
|
|
# )
|
|
|
|
|
|
# If we get an epub, export that to txt file
|
|
if input_source.endswith(".epub"):
|
|
book = preprocess_book(input_source)
|
|
|
|
# Check for GPU
|
|
if torch.cuda.is_available():
|
|
print('Nvidia GPU available. Setting as default device.')
|
|
torch.set_default_device('cuda')
|
|
elif torch.xpu.is_available():
|
|
print('Intel XPU (GPU) available. Setting as default device.')
|
|
torch.set_default_device('xpu')
|
|
elif torch.backends.mps.is_available():
|
|
print('Apple MPS GPU available. Setting as default device.')
|
|
torch.set_default_device('mps')
|
|
elif torch.backends.rocm.is_available():
|
|
print('AMD ROCm GPU available. Setting as default device.')
|
|
torch.set_default_device('rocm')
|
|
elif torch.is_vulkan_available():
|
|
print('Vulkan GPU available. Setting as default device.')
|
|
torch.set_default_device('vulkan')
|
|
else:
|
|
print('No GPU available. Using CPU.')
|
|
torch.set_default_device('cpu')
|
|
|
|
|
|
|
|
book_contents, book_title, book_author, chapter_titles = get_book(args.sourcefile)
|
|
files = read_book(book_contents, args.speaker, args.paragraphpause, args.speed, args.notitles)
|
|
generate_metadata(files, book_author, book_title, chapter_titles)
|
|
m4bfilename = make_m4b(files, args.sourcefile, args.speaker)
|
|
add_cover(args.cover, m4bfilename)
|
|
|
|
|
|
|
|
|
|
def kokoro_read(paragraph, speaker, filename, pipeline, speed):
|
|
audio_segments = []
|
|
sentences = process_large_text(paragraph)
|
|
for sent in sentences:
|
|
sent = conditional_sentence_case(sent.strip())
|
|
for gs, ps, audio in pipeline(sent, voice=speaker, speed=speed, split_pattern=r'\n\n\n'):
|
|
audio_segments.append(audio)
|
|
|
|
final_audio = np.concatenate(audio_segments)
|
|
soundfile.write(filename, final_audio, 24000)
|
|
|
|
def read_book(book_contents, speaker, paragraphpause, speed, notitles):
|
|
current_device_name = torch.get_default_device() if torch.get_default_device() else 'cpu'
|
|
current_device = torch.device(current_device_name)
|
|
print(f"Attempting to use device: {current_device}")
|
|
|
|
pipeline = KPipeline(lang_code=speaker[0])
|
|
|
|
# Explicitly move the model to the current default device (e.g., 'xpu')
|
|
if hasattr(pipeline, 'model') and pipeline.model is not None:
|
|
try:
|
|
pipeline.model.to(current_device)
|
|
print(f"Kokoro model explicitly moved to {current_device}")
|
|
except Exception as e:
|
|
print(f"Error moving Kokoro model to {current_device}: {e}")
|
|
else:
|
|
print("Warning: KPipeline does not have a 'model' attribute or model is None.")
|
|
|
|
segments = []
|
|
for i, chapter in enumerate(book_contents, start=1):
|
|
files = []
|
|
partname = f"part{i}.flac"
|
|
print(f"\n\n")
|
|
|
|
if os.path.isfile(partname):
|
|
print(f"{partname} exists, skipping to next chapter")
|
|
segments.append(partname)
|
|
else:
|
|
print(f"Chapter: {chapter['title']}\n")
|
|
print(f"Section name: \"{chapter['title']}\"")
|
|
if chapter["title"] == "":
|
|
chapter["title"] = "blank"
|
|
if chapter["title"] != "Title" and notitles != True:
|
|
title_temp = "title.flac"
|
|
if not os.path.isfile(title_temp):
|
|
kokoro_read(chapter['title'] + ".", speaker, "title_temp.wav", pipeline, speed)
|
|
append_silence("title_temp.wav", paragraphpause)
|
|
# Convert to flac
|
|
audio = AudioSegment.from_file("title_temp.wav")
|
|
audio.export(title_temp, format="flac")
|
|
os.remove("title_temp.wav")
|
|
files.append(title_temp)
|
|
|
|
for pindex, paragraph in enumerate(
|
|
tqdm(chapter["paragraphs"], desc=f"Generating audio files: ",unit='pg')
|
|
):
|
|
ptemp = f"pgraphs{pindex}.flac"
|
|
if os.path.isfile(ptemp):
|
|
print(f"{ptemp} exists, skipping to next paragraph")
|
|
else:
|
|
#sentences = sent_tokenize(paragraph)
|
|
filenames = ["sntnc1.wav"]
|
|
kokoro_read(paragraph, speaker, "sntnc1.wav", pipeline, speed)
|
|
append_silence("sntnc1.wav", paragraphpause)
|
|
# combine sentences in paragraph
|
|
sorted_files = sorted(filenames, key=sort_key)
|
|
if os.path.exists("sntnc0.wav"):
|
|
sorted_files.insert(0, "sntnc0.wav")
|
|
combined = AudioSegment.empty()
|
|
for file in sorted_files:
|
|
combined += AudioSegment.from_file(file)
|
|
combined.export(ptemp, format="flac")
|
|
for file in sorted_files:
|
|
os.remove(file)
|
|
files.append(ptemp)
|
|
# combine paragraphs into chapter
|
|
append_silence(files[-1], 2000)
|
|
combined = AudioSegment.empty()
|
|
for file in files:
|
|
combined += AudioSegment.from_file(file)
|
|
combined.export(partname, format="flac")
|
|
for file in files:
|
|
os.remove(file)
|
|
segments.append(partname)
|
|
return segments
|