tts-kokoro engine underway

This commit is contained in:
2026-08-07 19:00:45 -03:00
parent a1f564a051
commit 2777b34026
9 changed files with 787 additions and 37 deletions
+74 -34
View File
@@ -23,18 +23,26 @@ optional arguments:
-b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}
Backend to use for TTS
"""
# stdlib
import os
import sys
from pathlib import Path
from argparse import ArgumentParser
# Pip packages
from dotenv import load_dotenv
from . import tts_aedocw, tts_generic
from . import log, PathLike
# Local imports
from . import tts_aedocw, tts_generic, tts_kokoro
from . import log, PathLike
from .utils import check_env
# Setup env
load_dotenv() # load environment variables from .env file if present
backend = {
# Globals
WHICH = ["ffmpeg"] # Neded in $PATH
BACKENDS = {
# keys are the backend names, values are the corresponding TTS classes
"default": tts_generic.GenericTTSBackend,
"edge": tts_aedocw.TTSEdge,
@@ -45,7 +53,8 @@ backend = {
"epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge,
"epub2tts-chatterbox": tts_aedocw.AedocwChatterbox,
"epub2tts-kokoro": tts_aedocw.AedocwKokoro,
"generic-epub2tts": tts_aedocw.Epub2TTS
"generic-epub2tts": tts_aedocw.Epub2TTS,
"kokoro": tts_kokoro.KokoroBackend,
}
@@ -57,42 +66,73 @@ def main(args=None):
action="version",
version=f"%(prog)s {__import__('epub_tts').__version__}",
)
p.add_argument("-i", "--input", help="Input EPUB file", required=True)
p.add_argument("-o", "--output", help="Output audio file", required=True)
p.add_argument("-l", "--language", help="Language to use for TTS (see your backend's documentation for available voices)")
p.add_argument("-v", "--voice", help="Voice to use for TTS (see your backend's documentation for available voices)")
p.add_argument("--speed", type=float, default=1.0, help="Playback speed multiplier (default: 1.0) (not all backends support this)")
p.add_argument("--short-pause", type=int, default=None, help="Short pause duration in milliseconds between phrases or sentences (not all backends support this)")
p.add_argument("--long-pause", type=int, default=None, help="Long pause duration in milliseconds between sections or paragraphs (not all backends support this)")
p.add_argument("-c", "--cover", help="Path to cover image to embed or use for output metadata")
p.add_argument("-r", "--replace", action="append", nargs=2, help="Replace text in the intermediate output. Specify pairs of old_text new_text. Can be used multiple times.")
p.add_argument(
"-b",
"--backend",
help="Backend to use for TTS",
default="default",
choices=backend.keys(),
)
# Input options and processing
p.add_argument("-i", "--input",
nargs="+", required=True,
help="Input EPUB file")
p.add_argument("-r", "--replace", action="append", nargs=2,
help="Replace text in the intermediate output. "
"Specify pairs of old_text new_text. "
"Can be used multiple times.")
p.add_argument("--check-env",action="store_true",
help="Check the runtime environment and exit")
p.add_argument("-b", "--backend", default="default",
choices=BACKENDS.keys(),
help="Backend to use for TTS")
# Output options
p.add_argument("-o", "--output",
help="Output audio file", required=True)
p.add_argument("-c", "--cover",
help="Path to cover image to embed or use "
"for output metadata")
# Speech options
p.add_argument("-l", "--language",
help="Language to use for TTS (see your "
"backend's documentation for available voices)")
p.add_argument("-v", "--voice",
help="Voice to use for TTS (see your backend's "
"documentation for available voices)")
p.add_argument("--speed", type=float, default=1.0,
help="Playback speed multiplier (default: 1.0) "
"(not all backends support this)")
p.add_argument("--short-pause", type=int, default=None,
help="Short pause duration in milliseconds between "
"phrases or sentences (not all backends support this)")
p.add_argument("--long-pause", type=int, default=None,
help="Long pause duration in milliseconds between "
"sections or paragraphs (not all backends support this)")
p.add_argument("--notitles", action="store_true",
help="Do not read chapter titles")
# Parse
args = p.parse_args(args or sys.argv[1:])
setattr(args, 'replace_map',
{old: new for old, new in args.replace} if args.replace else None)
log(args)
# log(
# f"Converting {args.input} to {args.output} "
# f"using {args.backend} backend "
# "and args: " + ", ".join(
# f"{k}={v}" for k, v in vars(args).items()
# if k not in ("input", "output", "backend") and v is not None)
# )
#log(args.replace_map)
backend[args.backend](**vars(args)).run(
args.input, args.output,
**vars(args),
)
# Execute actions
if args.check_env:
check_env()
engine = BACKENDS[args.backend](**vars(args))
if len(args.input) > 1 and not output_dest.is_dir():
log(f"Output must be a directory when multiple input files are provided", level="error")
sys.exit(1)
for file in args.input:
log(file)
if not Path(file).exists():
log(f"Input file {file} does not exist", level="error")
sys.exit(1)
output_dest = Path(args.output)
engine.run(
file, output_dest,
**vars(args),
)
if __name__ == "__main__":
sys.exit(main())