#!/usr/bin/env python3 # -*- coding: utf-8 -*- # Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa # See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) """Backend for the kokoro engine. No vendored repository needed, kokoro is a pure Python package that can be installed via pip. """ # stdlib modules import subprocess import os import sys # Automatically enable MPS fallback on Apple Silicon macOS if sys.platform == 'darwin': os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1' # pip installed packages import numpy as np import soundfile import torch from tqdm import tqdm from kokoro import KPipeline from ebooklib import epub import soundfile as sf from mutagen import mp4 from pydub import AudioSegment from mutagen import mp4 # Local imports from .tts_generic import GenericTTSBackend from . import log, PathLike from .preprocess import preprocess_book class KokoroBackend(GenericTTSBackend): """Generic kokoro backend wrapper. Parameters: repo: repository name (one of the keys in _CMD_MAP) backend_cmd: explicit command/executable to use (overrides repo mapping) language: optional language code voice: optional voice name The GenericTTSBackend stores the default command name in self.backend_cmd, but the vendored adapters may still run the package in their own venv if the console script is not present. """ INTERMEDIATE_TXT = True INTERMEDIATE_CALL = GenericTTSBackend._replace_map DEFAULT_SPEAKER = "am_liam" # ["af_heart", "am_michael", "am_liam"] def run(self, input_source: PathLike, output_dest: PathLike, **kwargs ) -> subprocess.CompletedProcess: command = self._build_command(input_source, output_dest) log(command) # # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first # if not input_source.endswith(".txt"): # completed = subprocess.run( # self._build_command(input_source, output_dest), # cwd=str(self.CWD), # env=self.ENV, # check=True) # if self.INTERMEDIATE_TXT: # txt_file = self._normalize_path(input_source).with_suffix(".txt") # if txt_file.exists(): # if (self.INTERMEDIATE_CALL is not None and # kwargs.get('replace_map', None) is not None): # self.INTERMEDIATE_CALL( # txt_file, # txt_file.with_stem(txt_file.stem + "_replaced"), kwargs.get('replace_map', {})) # txt_file = txt_file.with_stem(txt_file.stem + "_replaced") # completed = subprocess.run( # self._build_command(txt_file, output_dest), # cwd=str(self.CWD), # env=self.ENV, # check=True # ) # If we get an epub, export that to txt file if input_source.endswith(".epub"): book = preprocess_book(input_source) # Check for GPU if torch.cuda.is_available(): print('Nvidia GPU available. Setting as default device.') torch.set_default_device('cuda') elif torch.xpu.is_available(): print('Intel XPU (GPU) available. Setting as default device.') torch.set_default_device('xpu') elif torch.backends.mps.is_available(): print('Apple MPS GPU available. Setting as default device.') torch.set_default_device('mps') elif torch.backends.rocm.is_available(): print('AMD ROCm GPU available. Setting as default device.') torch.set_default_device('rocm') elif torch.is_vulkan_available(): print('Vulkan GPU available. Setting as default device.') torch.set_default_device('vulkan') else: print('No GPU available. Using CPU.') torch.set_default_device('cpu') book_contents, book_title, book_author, chapter_titles = get_book(args.sourcefile) files = read_book(book_contents, args.speaker, args.paragraphpause, args.speed, args.notitles) generate_metadata(files, book_author, book_title, chapter_titles) m4bfilename = make_m4b(files, args.sourcefile, args.speaker) add_cover(args.cover, m4bfilename) def kokoro_read(paragraph, speaker, filename, pipeline, speed): audio_segments = [] sentences = process_large_text(paragraph) for sent in sentences: sent = conditional_sentence_case(sent.strip()) for gs, ps, audio in pipeline(sent, voice=speaker, speed=speed, split_pattern=r'\n\n\n'): audio_segments.append(audio) final_audio = np.concatenate(audio_segments) soundfile.write(filename, final_audio, 24000) def read_book(book_contents, speaker, paragraphpause, speed, notitles): current_device_name = torch.get_default_device() if torch.get_default_device() else 'cpu' current_device = torch.device(current_device_name) print(f"Attempting to use device: {current_device}") pipeline = KPipeline(lang_code=speaker[0]) # Explicitly move the model to the current default device (e.g., 'xpu') if hasattr(pipeline, 'model') and pipeline.model is not None: try: pipeline.model.to(current_device) print(f"Kokoro model explicitly moved to {current_device}") except Exception as e: print(f"Error moving Kokoro model to {current_device}: {e}") else: print("Warning: KPipeline does not have a 'model' attribute or model is None.") segments = [] for i, chapter in enumerate(book_contents, start=1): files = [] partname = f"part{i}.flac" print(f"\n\n") if os.path.isfile(partname): print(f"{partname} exists, skipping to next chapter") segments.append(partname) else: print(f"Chapter: {chapter['title']}\n") print(f"Section name: \"{chapter['title']}\"") if chapter["title"] == "": chapter["title"] = "blank" if chapter["title"] != "Title" and notitles != True: title_temp = "title.flac" if not os.path.isfile(title_temp): kokoro_read(chapter['title'] + ".", speaker, "title_temp.wav", pipeline, speed) append_silence("title_temp.wav", paragraphpause) # Convert to flac audio = AudioSegment.from_file("title_temp.wav") audio.export(title_temp, format="flac") os.remove("title_temp.wav") files.append(title_temp) for pindex, paragraph in enumerate( tqdm(chapter["paragraphs"], desc=f"Generating audio files: ",unit='pg') ): ptemp = f"pgraphs{pindex}.flac" if os.path.isfile(ptemp): print(f"{ptemp} exists, skipping to next paragraph") else: #sentences = sent_tokenize(paragraph) filenames = ["sntnc1.wav"] kokoro_read(paragraph, speaker, "sntnc1.wav", pipeline, speed) append_silence("sntnc1.wav", paragraphpause) # combine sentences in paragraph sorted_files = sorted(filenames, key=sort_key) if os.path.exists("sntnc0.wav"): sorted_files.insert(0, "sntnc0.wav") combined = AudioSegment.empty() for file in sorted_files: combined += AudioSegment.from_file(file) combined.export(ptemp, format="flac") for file in sorted_files: os.remove(file) files.append(ptemp) # combine paragraphs into chapter append_silence(files[-1], 2000) combined = AudioSegment.empty() for file in files: combined += AudioSegment.from_file(file) combined.export(partname, format="flac") for file in files: os.remove(file) segments.append(partname) return segments