Compare commits
8
Commits
067319f7ef
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2b26381a84 | ||
|
|
2d34e175bd | ||
|
|
b45321a19f | ||
|
|
ea16f87fc9 | ||
|
|
2f9b95e095 | ||
|
|
1266aef52a | ||
|
|
da8e4ee602 | ||
|
|
8a0f66ebc6 |
+11
-5
@@ -27,19 +27,25 @@ classifiers = [
|
||||
]
|
||||
|
||||
dependencies = [
|
||||
"load-dotenv>=0.1.0",
|
||||
"audioop-lts; python_version >= '3.13'",
|
||||
"beautifulsoup4",
|
||||
"colorlog",
|
||||
"dotenv",
|
||||
"ebooklib",
|
||||
"kokoro>=0.9.4",
|
||||
"load-dotenv>=0.1.0",
|
||||
"lxml",
|
||||
"mutagen",
|
||||
"nltk",
|
||||
"numpy",
|
||||
"pillow",
|
||||
"numpy", # --index-url https://download.pytorch.org/whl/cu132
|
||||
"pillow", # --index-url https://download.pytorch.org/whl/cu132
|
||||
"pyyaml",
|
||||
"pydub",
|
||||
"soundfile",
|
||||
"tqdm",
|
||||
"audioop-lts; python_version >= '3.13'",
|
||||
"torch", # --index-url https://download.pytorch.org/whl/cu132
|
||||
"torchaudio", # --index-url https://download.pytorch.org/whl/cu132
|
||||
"torchcodec", # PyTorch 2.9+ # --index-url https://download.pytorch.org/whl/cu132
|
||||
"tqdm", # --index-url https://download.pytorch.org/whl/cu132
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
|
||||
@@ -14,7 +14,6 @@ except PackageNotFoundError:
|
||||
pass
|
||||
|
||||
|
||||
|
||||
from .logger import logger, loglevel_map, get_logger
|
||||
from .utils import DataDict, PathLike, path
|
||||
from .app import App
|
||||
#from .logger import logger, loglevel_map, get_logger
|
||||
#from .utils import DataDict, PathLike, path
|
||||
#from .app import App
|
||||
|
||||
@@ -4,7 +4,116 @@
|
||||
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
||||
|
||||
import sys
|
||||
<<<<<<< HEAD
|
||||
from pathlib import Path
|
||||
from argparse import ArgumentParser
|
||||
|
||||
# Pip packages
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# Local imports
|
||||
from . import tts_aedocw, tts_generic
|
||||
#from . import tts_kokoro
|
||||
from . import log, PathLike
|
||||
from .utils import check_env
|
||||
|
||||
# Setup env
|
||||
load_dotenv() # load environment variables from .env file if present
|
||||
|
||||
# Globals
|
||||
WHICH = ["ffmpeg"] # Neded in $PATH
|
||||
BACKENDS = {
|
||||
# keys are the backend names, values are the corresponding TTS classes
|
||||
"default": tts_generic.GenericTTSBackend,
|
||||
"edge": tts_aedocw.TTSEdge,
|
||||
# aedocw-backed implementations — these expect the corresponding
|
||||
# package "console scripts"/entrypoints to be available in the
|
||||
# environment (or an explicit backend_cmd to be provided).
|
||||
"epub2tts": tts_aedocw.AedocwEpub2TTS,
|
||||
"epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge,
|
||||
"epub2tts-chatterbox": tts_aedocw.AedocwChatterbox,
|
||||
"epub2tts-kokoro": tts_aedocw.AedocwKokoro,
|
||||
"generic-epub2tts": tts_aedocw.Epub2TTS,
|
||||
#"kokoro": tts_kokoro.KokoroBackend,
|
||||
}
|
||||
|
||||
|
||||
def main(args=None):
|
||||
p = ArgumentParser(description="Convert EPUB to audio using TTS")
|
||||
p.add_argument(
|
||||
"--version",
|
||||
help="Show version and exit",
|
||||
action="version",
|
||||
version=f"%(prog)s {__import__('epub_tts').__version__}",
|
||||
)
|
||||
# Input options and processing
|
||||
p.add_argument("-i", "--input",
|
||||
nargs="+", required=True,
|
||||
help="Input EPUB file")
|
||||
p.add_argument("-r", "--replace", action="append", nargs=2,
|
||||
help="Replace text in the intermediate output. "
|
||||
"Specify pairs of old_text new_text. "
|
||||
"Can be used multiple times.")
|
||||
p.add_argument("--check-env",action="store_true",
|
||||
help="Check the runtime environment and exit")
|
||||
p.add_argument("-b", "--backend", default="default",
|
||||
choices=BACKENDS.keys(),
|
||||
help="Backend to use for TTS")
|
||||
|
||||
# Output options
|
||||
p.add_argument("-o", "--output",
|
||||
help="Output audio file", required=True)
|
||||
p.add_argument("-c", "--cover",
|
||||
help="Path to cover image to embed or use "
|
||||
"for output metadata")
|
||||
|
||||
# Speech options
|
||||
p.add_argument("-l", "--language",
|
||||
help="Language to use for TTS (see your "
|
||||
"backend's documentation for available voices)")
|
||||
p.add_argument("-v", "--voice",
|
||||
help="Voice to use for TTS (see your backend's "
|
||||
"documentation for available voices)")
|
||||
p.add_argument("--speed", type=float, default=1.0,
|
||||
help="Playback speed multiplier (default: 1.0) "
|
||||
"(not all backends support this)")
|
||||
p.add_argument("--short-pause", type=int, default=None,
|
||||
help="Short pause duration in milliseconds between "
|
||||
"phrases or sentences (not all backends support this)")
|
||||
p.add_argument("--long-pause", type=int, default=None,
|
||||
help="Long pause duration in milliseconds between "
|
||||
"sections or paragraphs (not all backends support this)")
|
||||
p.add_argument("--notitles", action="store_true",
|
||||
help="Do not read chapter titles")
|
||||
|
||||
# Parse
|
||||
args = p.parse_args(args or sys.argv[1:])
|
||||
setattr(args, 'replace_map',
|
||||
{old: new for old, new in args.replace} if args.replace else None)
|
||||
log(args)
|
||||
|
||||
# Execute actions
|
||||
if args.check_env:
|
||||
check_env()
|
||||
|
||||
engine = BACKENDS[args.backend](**vars(args))
|
||||
if len(args.input) > 1 and not output_dest.is_dir():
|
||||
log(f"Output must be a directory when multiple input files are provided", level="error")
|
||||
sys.exit(1)
|
||||
for file in args.input:
|
||||
log(file)
|
||||
if not Path(file).exists():
|
||||
log(f"Input file {file} does not exist", level="error")
|
||||
sys.exit(1)
|
||||
output_dest = Path(args.output)
|
||||
engine.run(
|
||||
file, output_dest,
|
||||
**vars(args),
|
||||
)
|
||||
|
||||
=======
|
||||
from .cli import cli
|
||||
>>>>>>> dev
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+173
-61
@@ -12,6 +12,18 @@ import os
|
||||
import re
|
||||
import sys
|
||||
import pprint
|
||||
import locale
|
||||
import yaml
|
||||
from typing import Any, Optional, Callable, Required
|
||||
|
||||
# pip packages
|
||||
from dotenv import load_dotenv
|
||||
import colorlog
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
loglevel_map = logging.getLevelNamesMapping()
|
||||
|
||||
|
||||
# Globals
|
||||
format_str = "[%(name)s][%(filename)s:%(lineno)04d][%(relativeCreated)s]::%(levelname).4s: (%(funcName)s) %(message)s"
|
||||
@@ -41,20 +53,78 @@ format_str = "[%(name)s][%(filename)s:%(lineno)04d][%(relativeCreated)s]::%(leve
|
||||
# %(processName)s Process name (if available)
|
||||
# %(message)s The result of record.getMessage(), computed just as
|
||||
# the record is emitted
|
||||
formatter = logging.Formatter(format_str)
|
||||
logging.basicConfig(
|
||||
# 50=CRITICAL/FATAL, 40=ERROR, 30=WARN/WARNING, 20=INFO, 10=DEBUG, 0=NOTSET
|
||||
# loglevel_map = logging.getLevelNamesMapping()
|
||||
level=logging.INFO,
|
||||
format=format_str,
|
||||
)
|
||||
# formatter = logging.Formatter(format_str)
|
||||
# logging.basicConfig(
|
||||
# # 50=CRITICAL/FATAL, 40=ERROR, 30=WARN/WARNING, 20=INFO, 10=DEBUG, 0=NOTSET
|
||||
# # loglevel_map = logging.getLevelNamesMapping()
|
||||
# level=logging.INFO,
|
||||
# format=format_str,
|
||||
# )
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
loglevel_map = logging.getLevelNamesMapping()
|
||||
# True regex pattern:
|
||||
#locale.nl_langinfo(locale.YESEXPR)
|
||||
RE_TRUE = r"^[yY]|^[sS]|^[tT]|^[oO][nN]|1"
|
||||
# False regex pattern: (use with caution!)
|
||||
# locale.nl_langinfo(locale.NOEXPR)
|
||||
RE_FALSE = r"^[nN]|^[fF]|^[oO][fF]+|0"
|
||||
# None:
|
||||
# matches any combination of word boundaries and whitespaces.
|
||||
# \b cannot be used in a character range. See
|
||||
RE_NONE = r"^\b*\s*\b*\s*$"
|
||||
|
||||
|
||||
# Logging utils
|
||||
def user_bool(user_str: str, empty_is_none: bool = False, from_yaml=False) -> None|bool:
|
||||
"""Use caution in this call, only if you are absolutely sure that user_str should be bool or none/empty"""
|
||||
if from_yaml:
|
||||
return yaml.YAMLObject().from_yaml(user_str)
|
||||
if re.match(RE_TRUE, user_str):
|
||||
return True
|
||||
if re.match(RE_FALSE, user_str):
|
||||
return False
|
||||
if re.match(RE_NONE, user_str):
|
||||
if empty_is_none:
|
||||
return None
|
||||
else:
|
||||
return False
|
||||
raise ValueError(user_str)
|
||||
|
||||
|
||||
def get_env(env_name: str,
|
||||
cast_type: Optional[Callable] = str,
|
||||
default: Optional[str] = None,
|
||||
) -> Any | str:
|
||||
if not isinstance(cast_type, Callable):
|
||||
raise ValueError(cast_type)
|
||||
return cast_type(os.environ.get(env_name, default))
|
||||
|
||||
|
||||
def get_logger(name):
|
||||
return logging.getLogger(name)
|
||||
|
||||
def create_logger(name = __name__, loglevel: int = logging.INFO):
|
||||
load_dotenv()
|
||||
logging.captureWarnings(get_env("CAPTURE_WARNINGS", user_bool, "True"))
|
||||
|
||||
handler = colorlog.StreamHandler()
|
||||
handler.setFormatter(colorlog.ColoredFormatter(
|
||||
#"%(log_color)s%(levelname)-8s%(reset)s %(blue)s%(message)s",
|
||||
f"%(log_color)s{format_str}%(reset)s",
|
||||
log_colors={
|
||||
'DEBUG': 'cyan',
|
||||
'INFO': 'green',
|
||||
'WARNING': 'orange',
|
||||
'ERROR': 'red',
|
||||
'CRITICAL': 'yellow',
|
||||
}))
|
||||
|
||||
logger = colorlog.getLogger(name)
|
||||
logger.addHandler(handler)
|
||||
logger.setLevel(loglevel)
|
||||
|
||||
return logger
|
||||
|
||||
|
||||
# Simple logger by printing
|
||||
def print(message):
|
||||
pp = pprint.PrettyPrinter(
|
||||
indent=4,
|
||||
@@ -74,7 +144,7 @@ def get_level(
|
||||
verbose: bool = False,
|
||||
debug: bool = False,
|
||||
quiet: bool = False,
|
||||
default: int = logging.INFO,
|
||||
loglevel: int = logging.INFO,
|
||||
) -> int:
|
||||
# Find lowest priority loglevel
|
||||
if "--debug" in cmdline:
|
||||
@@ -85,65 +155,107 @@ def get_level(
|
||||
quiet = True
|
||||
|
||||
if debug:
|
||||
msg = ["Returning loglevel 'DEBUG'"]
|
||||
|
||||
if verbose or quiet:
|
||||
msg.append(", ignoring other loglevel flags ")
|
||||
|
||||
if verbose and quiet:
|
||||
msg.append(", ignoring other loglevel flags "
|
||||
"(--verbose and --quiet) which are also set")
|
||||
elif verbose:
|
||||
msg.append("(--verbose), which is also set")
|
||||
else:
|
||||
msg.append("(--quiet), which is also set")
|
||||
msg.append(".")
|
||||
if not quiet:
|
||||
logger.info("".join(msg))
|
||||
return logging.DEBUG
|
||||
|
||||
for i, a in enumerate(cmdline):
|
||||
if a == "--loglevel":
|
||||
loglevel = loglevel_map.get(cmdline[i+1], None)
|
||||
elif a.startswith("--loglevel="):
|
||||
loglevel = loglevel_map.get(a.replace("--loglevel=", ""), None)
|
||||
if verbose:
|
||||
if quiet:
|
||||
return logging.INFO
|
||||
#msg.append(", ignoring '--quiet' loglevel flag which was also set")
|
||||
logger.info("Returning loglevel 'INFO' ('--verbose' flag was set).")
|
||||
return logging.INFO
|
||||
|
||||
# lower = more priority
|
||||
if loglevel_map[loglevel] > loglevel_map["INFO"]:
|
||||
config.logger.debug("self.loglevel='%s'(%d) > INFO", config.loglevel, loglevel_map[config.loglevel])
|
||||
config.loglevel = "INFO"
|
||||
if quiet:
|
||||
return loglevel
|
||||
|
||||
def get_logger(
|
||||
verbose:bool = False,
|
||||
debug:bool = False,
|
||||
quiet: bool = False,
|
||||
loglevel: str|int = logging.INFO,
|
||||
logfile: str|Path = None,
|
||||
):
|
||||
"""Initialize the package logger from CLI flags and defaults.
|
||||
# msg = [f"Returning default loglevel '{loglevel_map[loglevel]}'."]
|
||||
# logger.info("".join(msg))
|
||||
# return loglevel
|
||||
# logger.info("Returning loglevel 'DEBUG' and ignoring other loglevel flags ("
|
||||
# f"{"--verbose" if "--verbose" in cmdline} is also set"
|
||||
# f"{"--quiet" if "--quiet" in cmdline}"
|
||||
# ")")
|
||||
# return logging.DEBUG
|
||||
# if "--verbose" in cmdline:
|
||||
# verbose = True
|
||||
# logger.info("Returning loglevel 'INFO'")
|
||||
# return logging.INFO
|
||||
# if "--quiet" in cmdline:
|
||||
# logger.info("Returning loglevel 'ERROR'")
|
||||
# return logging.ERROR
|
||||
# return logging.DEBUG
|
||||
|
||||
Resolve the effective logging level from the class args (or sys.argv).
|
||||
--debug (or debug_level=True) takes priority, then --verbose (or verbose=True)
|
||||
is compared to --loglevel (or loglevel=) and the lower one is proritized.
|
||||
"""
|
||||
# for i, a in enumerate(cmdline):
|
||||
# if a == "--loglevel":
|
||||
# loglevel = loglevel_map.get(cmdline[i+1], None)
|
||||
# elif a.startswith("--loglevel="):
|
||||
# loglevel = loglevel_map.get(a.replace("--loglevel=", ""), None)
|
||||
|
||||
# # lower = more priority
|
||||
# if loglevel_map[loglevel] > loglevel_map["INFO"]:
|
||||
# logger.debug("self.loglevel='%s'(%d) > INFO", loglevel, loglevel_map[loglevel])
|
||||
# loglevel = "INFO"
|
||||
|
||||
# def get_logger(
|
||||
# verbose:bool = False,
|
||||
# debug:bool = False,
|
||||
# quiet: bool = False,
|
||||
# loglevel: str|int = logging.INFO,
|
||||
# logfile: str|Path = None,
|
||||
# ):
|
||||
# """Initialize the package logger from CLI flags and defaults.
|
||||
|
||||
# Resolve the effective logging level from the class args (or sys.argv).
|
||||
# --debug (or debug_level=True) takes priority, then --verbose (or verbose=True)
|
||||
# is compared to --loglevel (or loglevel=) and the lower one is proritized.
|
||||
# """
|
||||
|
||||
|
||||
|
||||
config.logger.info("Setting loglevel to %s",config.loglevel)
|
||||
config.logger.setLevel(loglevel_map[config.loglevel])
|
||||
# config.logger.info("Setting loglevel to %s",config.loglevel)
|
||||
# config.logger.setLevel(loglevel_map[config.loglevel])
|
||||
|
||||
# Get logfile
|
||||
for i, a in enumerate(config._args_list):
|
||||
if a == "--logfile":
|
||||
config.logfile = config._args_list[i+1]
|
||||
elif a.startswith("--logfile="):
|
||||
config.logfile = a.replace("--logfile=", "")
|
||||
# # Get logfile
|
||||
# for i, a in enumerate(config._args_list):
|
||||
# if a == "--logfile":
|
||||
# config.logfile = config._args_list[i+1]
|
||||
# elif a.startswith("--logfile="):
|
||||
# config.logfile = a.replace("--logfile=", "")
|
||||
|
||||
if "logfile" in config and config.logfile:
|
||||
if logfile:
|
||||
config.logger.warning("Overriding --logfile='%s' from setup_logger(logfile='%s')", config.logfile, logfile)
|
||||
config.logfile = logfile
|
||||
fh = logging.FileHandler(logfile)
|
||||
fh.setLevel(config.loglevel)
|
||||
fh.setFormatter(logger.formatter)
|
||||
config.logger.addHandler(fh)
|
||||
# if "logfile" in config and config.logfile:
|
||||
# if logfile:
|
||||
# config.logger.warning("Overriding --logfile='%s' from setup_logger(logfile='%s')", config.logfile, logfile)
|
||||
# config.logfile = logfile
|
||||
# fh = logging.FileHandler(logfile)
|
||||
# fh.setLevel(config.loglevel)
|
||||
# fh.setFormatter(logger.formatter)
|
||||
# config.logger.addHandler(fh)
|
||||
|
||||
# if self.debug_level:
|
||||
# logger.info("Loglevel: DEBUG")
|
||||
# if self.loglevel:
|
||||
# logger.warning("Ignoring '--loglevel' flag because DEBUG/'--debug' was also set.")
|
||||
# if "--verbose" in self._args_list:
|
||||
# logger.warning("Ignoring '--verbose' flag because '--debug' was also set.")
|
||||
# self.verbose = False
|
||||
# if self.loglevel and self.verbose:
|
||||
# self.warning("Conflict: trying to --loglevel=%s (%d) and --verbose.", self.loglevel, loglevel_map[self.loglevel])
|
||||
# if loglevel_map[self.loglevel] < logging.INFO:
|
||||
# self.logger.info("Setting loglevel to %s (%d), which is more verbose than INFO.", self.loglevel, loglevel_map[self.loglevel])
|
||||
# # if self.debug_level:
|
||||
# # logger.info("Loglevel: DEBUG")
|
||||
# # if self.loglevel:
|
||||
# # logger.warning("Ignoring '--loglevel' flag because DEBUG/'--debug' was also set.")
|
||||
# # if "--verbose" in self._args_list:
|
||||
# # logger.warning("Ignoring '--verbose' flag because '--debug' was also set.")
|
||||
# # self.verbose = False
|
||||
# # if self.loglevel and self.verbose:
|
||||
# # self.warning("Conflict: trying to --loglevel=%s (%d) and --verbose.", self.loglevel, loglevel_map[self.loglevel])
|
||||
# # if loglevel_map[self.loglevel] < logging.INFO:
|
||||
# # self.logger.info("Setting loglevel to %s (%d), which is more verbose than INFO.", self.loglevel, loglevel_map[self.loglevel])
|
||||
|
||||
|
||||
return config.logger
|
||||
# return config.logger
|
||||
@@ -0,0 +1,93 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8; tab-width: 4; -*-
|
||||
# vim: set fileencoding=utf-8 tabstop=4 shiftwidth=4 expandtab:
|
||||
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
||||
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
||||
|
||||
from pathlib import Path
|
||||
import os
|
||||
import sys
|
||||
import logging
|
||||
from datetime import datetime as dt
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from .logger import logger, create_logger
|
||||
|
||||
logger = create_logger(__name__)
|
||||
from spantrack import cli, run_config
|
||||
|
||||
load_dotenv()
|
||||
os.environ["NLTK_DATA"] = 'C:\\Users\\renat\\AppData\\Roaming\\nltk_data'
|
||||
os.environ["HF_TOKEN"] = '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'
|
||||
|
||||
# Globals
|
||||
time_format: str = os.environ.get("TIMEFMT", r"%Y-%m-%dT%H:%M:%S.%f")
|
||||
caught_exceptions: tuple[Exception] = (KeyboardInterrupt, BaseException)
|
||||
|
||||
|
||||
def main():
|
||||
logger.info("Start main function.")
|
||||
root = Path('D:\\DATA\\EBOOKS\\_SEM_DRM\\')
|
||||
globs = [
|
||||
#"Night School*.epub",
|
||||
#"Midnight Line*.epub",
|
||||
#"Past Tense*.epub",
|
||||
#"Blue Moon*.epub",
|
||||
"The Sentinel*.epub",
|
||||
"Better Off Dead*.epub",
|
||||
"No Plan B*.epub",
|
||||
]
|
||||
|
||||
cmd = []
|
||||
if "--dry-run" in sys.argv:
|
||||
cmd.extend("--dry-run")
|
||||
cmd.extend(["--progress",
|
||||
"--device", run_config.Device["CUDA"], # spantrack.run_config.Device, default=Device.AUTO,
|
||||
#"--lexicon", # Path
|
||||
#"--audio-format", # AudioFormat, default=AudioFormat.MP3, spantrack.run_config._AUDIO_MEDIA_TYPES
|
||||
#"--lang", # Language, default=Language.EN_US,
|
||||
"--voice","am_liam",
|
||||
])
|
||||
result_output = {"pending":[],"success":[],"error":[]}
|
||||
logger.info("Start main loop: ")
|
||||
logger.info("try (for glob in globs: %s", ", ".join([f"'{str(g)}'" for g in globs]))
|
||||
logger.info("except: %s", ", ".join([f"{e.__name__}" for e in caught_exceptions]) )
|
||||
logger.info("finally: sys.exit( len(errors) + len(pending) )")
|
||||
try:
|
||||
for glob_i,glob in enumerate(globs):
|
||||
file_list = [f for f in root.glob(glob)]
|
||||
if len(file_list) != 1:
|
||||
logger.error(f"Glob pattern '{glob}' found {"more" if len(file_list)>1 else "less"} than 1 file: %s", file_list)
|
||||
result_output["error"].append(glob)
|
||||
continue
|
||||
result_output["pending"].append(str(file_list[0]))
|
||||
logger.info(
|
||||
"Files to parse:\n%s",
|
||||
",\n".join([f"[{i+1}] '{str(f)}'" for i,f in enumerate(result_output["pending"])]))
|
||||
for file in result_output["pending"]:
|
||||
start_dt = dt.now()
|
||||
logger.info(f"[{start_dt.strftime(time_format)}] Narrating '{file}'")
|
||||
cli.main([file] + cmd)
|
||||
finish_dt = dt.now()
|
||||
logger.info(f"[{finish_dt .strftime(time_format)}] Success! '{file}'")
|
||||
elapsed_time = finish_dt-start_dt
|
||||
elapsed_total_seconds = elapsed_time.total_seconds()
|
||||
elapsed_hours= elapsed_total_seconds // 3600
|
||||
elapsed_minutes = (elapsed_total_seconds % 3600) // 60
|
||||
elapsed_seconds = elapsed_total_seconds % 60
|
||||
elapsed_msg = "Total elapsed time: %dh %dmin %ds"
|
||||
logger.info(elapsed_msg, elapsed_hours, elapsed_minutes, elapsed_seconds)
|
||||
result_output["success"].append(file)
|
||||
result_output["pending"].remove(file)
|
||||
except caught_exceptions as e:
|
||||
logger.critical("Caught exception: %s", str(e))
|
||||
finally:
|
||||
logger.info("Exiting with results:\nDONE:%s\nLIST:%s\nERR:%s",
|
||||
",\n".join([f"\t[{i+1}] '{str(f)}'" for i,f in enumerate(result_output["success"])]),
|
||||
",\n".join([f"\t[{i+1}] '{str(f)}'" for i,f in enumerate(result_output["pending"])]),
|
||||
",\n".join([f"\t[{i+1}] '{str(f)}'" for i,f in enumerate(result_output["error"])]),
|
||||
)
|
||||
sys.exit(
|
||||
len(result_output["pending"]) +
|
||||
len(result_output["error"])
|
||||
)
|
||||
@@ -9,12 +9,14 @@ No vendored repository needed, kokoro is a pure Python package that can be insta
|
||||
|
||||
# stdlib modules
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
import os
|
||||
import sys
|
||||
# Automatically enable MPS fallback on Apple Silicon macOS
|
||||
if sys.platform == 'darwin':
|
||||
os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
|
||||
|
||||
|
||||
# pip installed packages
|
||||
import numpy as np
|
||||
import soundfile
|
||||
@@ -203,3 +205,109 @@ class KokoroBackend(GenericTTSBackend):
|
||||
os.remove(file)
|
||||
segments.append(partname)
|
||||
return segments
|
||||
|
||||
# ***
|
||||
|
||||
# READ
|
||||
def read_from_txt_to_wav(text_file:PathLike,
|
||||
speaker:str="af_heart") -> Path:
|
||||
if not isinstance(text_file,Path):
|
||||
text_file = Path(text_file).resolve()
|
||||
else:
|
||||
text_file = text_file.resolve()
|
||||
|
||||
# Check if text file exists
|
||||
if not text_file.exists():
|
||||
print(f"Error: Text file '{text_file}' not found.")
|
||||
return False
|
||||
|
||||
# Read the text from the file
|
||||
with open(text_file, 'r', encoding='utf-8') as f:
|
||||
text_contents = f.read()
|
||||
|
||||
# Generate output filename (replace .txt extension with .wav)
|
||||
output_file = text_file.with_suffix('.wav')
|
||||
|
||||
# Check for CUDA GPU
|
||||
if torch.cuda.is_available():
|
||||
print('CUDA GPU available')
|
||||
torch.set_default_device('cuda')
|
||||
|
||||
print(f"Generating audio for speaker '{speaker}' from '{text_file}'...")
|
||||
|
||||
# Create pipeline with language code (first character of speaker name)
|
||||
pipeline = KPipeline(lang_code=speaker[0])
|
||||
|
||||
# Generate audio segments
|
||||
audio_segments = []
|
||||
for gs, ps, audio in pipeline(text_contents, voice=speaker, speed=1, split_pattern=r'\n\n\n'):
|
||||
audio_segments.append(audio)
|
||||
|
||||
# Concatenate all audio segments
|
||||
final_audio = np.concatenate(audio_segments)
|
||||
|
||||
# Write to wav file
|
||||
soundfile.write(output_file, final_audio, 24000)
|
||||
|
||||
print(f"Audio saved to '{output_file}'")
|
||||
return output_file
|
||||
|
||||
|
||||
|
||||
def get_speakers(self) -> list[str]:
|
||||
"""Return list of available speakers.
|
||||
See https://huggingface.co/hexgrad/Kokoro-82M/blob/main/VOICES.md"""
|
||||
speakers = [
|
||||
# 🇺🇸 American English: 11F 9M
|
||||
# Overall Grade 'A'
|
||||
"af_heart",
|
||||
# Overall Grade 'C+'
|
||||
"af_aoede","af_kore","af_sarah",
|
||||
"am_fenrir","am_michael","am_puck",
|
||||
# Other
|
||||
"af_alloy", "af_bella", "af_jessica", "af_nicole", "af_nova", "af_river", "af_sky", "am_adam", "am_echo", "am_eric", "am_liam", "am_onyx", "am_santa", "bf_alice",
|
||||
# 🇬🇧 British English: 4F 4M
|
||||
"bf_emma", "bf_isabella", "bf_lily", "bm_daniel", "bm_fable", "bm_george", "bm_lewis",
|
||||
# 🇧🇷 Brazilian Portuguese: 1F 2M
|
||||
"pf_dora",
|
||||
"pm_alex",
|
||||
"pm_santa"]
|
||||
# Old list:
|
||||
# ["af_heart", "af_joy", "af_sad", "af_angry", "af_fear", "af_surprise"]
|
||||
return speakers
|
||||
|
||||
def gen_speaker_samples(
|
||||
self,
|
||||
samples: list =None,
|
||||
output_path: PathLike=None,
|
||||
) -> list[str]:
|
||||
"""Generate sample speakers audio in output_path."""
|
||||
result = self._build_command("gen_samples.py", *samples, output_path=output_path)
|
||||
return [str(self._normalize_path(s)) for s in samples]
|
||||
|
||||
def gen_speaker_samples():
|
||||
|
||||
|
||||
if torch.cuda.is_available():
|
||||
print('CUDA GPU available')
|
||||
torch.set_default_device('cuda')
|
||||
|
||||
for speaker in speakers:
|
||||
file = speaker + "_sample.wav"
|
||||
if os.path.exists(file):
|
||||
print(f"Sample for {speaker} already exists.")
|
||||
continue
|
||||
else:
|
||||
print(f"Creating {speaker}")
|
||||
pipeline = KPipeline(lang_code=speaker[0])
|
||||
sentence = f"Hello, this voice is {speaker[3:]}. The quick brown fox jumped over the lazy dog. The fish twisted and turned on the bent hook. Press the pants and sew a button on the vest. The swan dive was far short of perfect."
|
||||
audio_segments = []
|
||||
for gs, ps, audio in pipeline(
|
||||
sentence,
|
||||
repo_id='hexgrad/Kokoro-82M',
|
||||
voice=speaker,
|
||||
speed=1,
|
||||
split_pattern=r'\n\n\n'):
|
||||
audio_segments.append(audio)
|
||||
final_audio = np.concatenate(audio_segments)
|
||||
soundfile.write(file, final_audio, 24000)
|
||||
|
||||
@@ -13,7 +13,8 @@ from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
from collections.abc import MutableMapping
|
||||
|
||||
from . import logger, PathLike
|
||||
from .logger import logger
|
||||
from . import PathLike
|
||||
|
||||
|
||||
def path(fpath: str|Path = None):
|
||||
|
||||
Vendored
+1
-1
Submodule src/vendor/epub2tts-edge updated: 6fb7a0f125...a25ead5312
Vendored
+1
-1
Submodule src/vendor/epub2tts-kokoro updated: dd27e5721a...9f25e02fe7
Reference in New Issue
Block a user