4 Commits
Author SHA1 Message Date
renatoxsr 2f9b95e095 . 2026-08-12 08:58:40 -03:00
renatoxsr 1266aef52a Merge branch 'dev' 2026-08-12 08:56:27 -03:00
renatoxsr da8e4ee602 .. 2026-08-12 08:50:39 -03:00
renatoxsr 8a0f66ebc6 spantrack 2026-08-12 08:47:49 -03:00
6 changed files with 249 additions and 7 deletions
+8 -5
View File
@@ -27,19 +27,22 @@ classifiers = [
] ]
dependencies = [ dependencies = [
"load-dotenv>=0.1.0", "audioop-lts; python_version >= '3.13'",
"beautifulsoup4", "beautifulsoup4",
"ebooklib", "ebooklib",
"kokoro>=0.9.4", "kokoro>=0.9.4",
"load-dotenv>=0.1.0",
"lxml", "lxml",
"mutagen", "mutagen",
"nltk", "nltk",
"numpy", "numpy", # --index-url https://download.pytorch.org/whl/cu132
"pillow", "pillow", # --index-url https://download.pytorch.org/whl/cu132
"pydub", "pydub",
"soundfile", "soundfile",
"tqdm", "torch", # --index-url https://download.pytorch.org/whl/cu132
"audioop-lts; python_version >= '3.13'", "torchaudio", # --index-url https://download.pytorch.org/whl/cu132
"torchcodec", # PyTorch 2.9+ # --index-url https://download.pytorch.org/whl/cu132
"tqdm", # --index-url https://download.pytorch.org/whl/cu132
] ]
[project.optional-dependencies] [project.optional-dependencies]
+109
View File
@@ -4,7 +4,116 @@
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) # See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
import sys import sys
<<<<<<< HEAD
from pathlib import Path
from argparse import ArgumentParser
# Pip packages
from dotenv import load_dotenv
# Local imports
from . import tts_aedocw, tts_generic
#from . import tts_kokoro
from . import log, PathLike
from .utils import check_env
# Setup env
load_dotenv() # load environment variables from .env file if present
# Globals
WHICH = ["ffmpeg"] # Neded in $PATH
BACKENDS = {
# keys are the backend names, values are the corresponding TTS classes
"default": tts_generic.GenericTTSBackend,
"edge": tts_aedocw.TTSEdge,
# aedocw-backed implementations — these expect the corresponding
# package "console scripts"/entrypoints to be available in the
# environment (or an explicit backend_cmd to be provided).
"epub2tts": tts_aedocw.AedocwEpub2TTS,
"epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge,
"epub2tts-chatterbox": tts_aedocw.AedocwChatterbox,
"epub2tts-kokoro": tts_aedocw.AedocwKokoro,
"generic-epub2tts": tts_aedocw.Epub2TTS,
#"kokoro": tts_kokoro.KokoroBackend,
}
def main(args=None):
p = ArgumentParser(description="Convert EPUB to audio using TTS")
p.add_argument(
"--version",
help="Show version and exit",
action="version",
version=f"%(prog)s {__import__('epub_tts').__version__}",
)
# Input options and processing
p.add_argument("-i", "--input",
nargs="+", required=True,
help="Input EPUB file")
p.add_argument("-r", "--replace", action="append", nargs=2,
help="Replace text in the intermediate output. "
"Specify pairs of old_text new_text. "
"Can be used multiple times.")
p.add_argument("--check-env",action="store_true",
help="Check the runtime environment and exit")
p.add_argument("-b", "--backend", default="default",
choices=BACKENDS.keys(),
help="Backend to use for TTS")
# Output options
p.add_argument("-o", "--output",
help="Output audio file", required=True)
p.add_argument("-c", "--cover",
help="Path to cover image to embed or use "
"for output metadata")
# Speech options
p.add_argument("-l", "--language",
help="Language to use for TTS (see your "
"backend's documentation for available voices)")
p.add_argument("-v", "--voice",
help="Voice to use for TTS (see your backend's "
"documentation for available voices)")
p.add_argument("--speed", type=float, default=1.0,
help="Playback speed multiplier (default: 1.0) "
"(not all backends support this)")
p.add_argument("--short-pause", type=int, default=None,
help="Short pause duration in milliseconds between "
"phrases or sentences (not all backends support this)")
p.add_argument("--long-pause", type=int, default=None,
help="Long pause duration in milliseconds between "
"sections or paragraphs (not all backends support this)")
p.add_argument("--notitles", action="store_true",
help="Do not read chapter titles")
# Parse
args = p.parse_args(args or sys.argv[1:])
setattr(args, 'replace_map',
{old: new for old, new in args.replace} if args.replace else None)
log(args)
# Execute actions
if args.check_env:
check_env()
engine = BACKENDS[args.backend](**vars(args))
if len(args.input) > 1 and not output_dest.is_dir():
log(f"Output must be a directory when multiple input files are provided", level="error")
sys.exit(1)
for file in args.input:
log(file)
if not Path(file).exists():
log(f"Input file {file} does not exist", level="error")
sys.exit(1)
output_dest = Path(args.output)
engine.run(
file, output_dest,
**vars(args),
)
=======
from .cli import cli from .cli import cli
>>>>>>> dev
if __name__ == "__main__": if __name__ == "__main__":
+22
View File
@@ -0,0 +1,22 @@
import os
import logging
logger = logging.get_logger()
os.environ["NLTK_DATA"] = 'C:\\Users\\renat\\AppData\\Roaming\\nltk_data'
os.environ["HF_TOKEN"] = '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'
from spantrack import cli
root = 'D:\\DATA\\EBOOKS\\_SEM_DRM\\'
globs = [
"Night School*.epub",
"The Midnight*.epub",
"Past Tense*.epub",
"Blue Moon*.epub",
"The Sentinel",
"Better Off Dead",
"No Plan B",
]
for glob in globs:
f = root.glob(glob)
if len(f) > 1:
logger.error("Glob pattern found more than 1 file: %s", f)
continue
cli.main([f.resolve(), "--voice", "am_liam"])
+108
View File
@@ -9,12 +9,14 @@ No vendored repository needed, kokoro is a pure Python package that can be insta
# stdlib modules # stdlib modules
import subprocess import subprocess
from pathlib import Path
import os import os
import sys import sys
# Automatically enable MPS fallback on Apple Silicon macOS # Automatically enable MPS fallback on Apple Silicon macOS
if sys.platform == 'darwin': if sys.platform == 'darwin':
os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1' os.environ['PYTORCH_ENABLE_MPS_FALLBACK'] = '1'
# pip installed packages # pip installed packages
import numpy as np import numpy as np
import soundfile import soundfile
@@ -203,3 +205,109 @@ class KokoroBackend(GenericTTSBackend):
os.remove(file) os.remove(file)
segments.append(partname) segments.append(partname)
return segments return segments
# ***
# READ
def read_from_txt_to_wav(text_file:PathLike,
speaker:str="af_heart") -> Path:
if not isinstance(text_file,Path):
text_file = Path(text_file).resolve()
else:
text_file = text_file.resolve()
# Check if text file exists
if not text_file.exists():
print(f"Error: Text file '{text_file}' not found.")
return False
# Read the text from the file
with open(text_file, 'r', encoding='utf-8') as f:
text_contents = f.read()
# Generate output filename (replace .txt extension with .wav)
output_file = text_file.with_suffix('.wav')
# Check for CUDA GPU
if torch.cuda.is_available():
print('CUDA GPU available')
torch.set_default_device('cuda')
print(f"Generating audio for speaker '{speaker}' from '{text_file}'...")
# Create pipeline with language code (first character of speaker name)
pipeline = KPipeline(lang_code=speaker[0])
# Generate audio segments
audio_segments = []
for gs, ps, audio in pipeline(text_contents, voice=speaker, speed=1, split_pattern=r'\n\n\n'):
audio_segments.append(audio)
# Concatenate all audio segments
final_audio = np.concatenate(audio_segments)
# Write to wav file
soundfile.write(output_file, final_audio, 24000)
print(f"Audio saved to '{output_file}'")
return output_file
def get_speakers(self) -> list[str]:
"""Return list of available speakers.
See https://huggingface.co/hexgrad/Kokoro-82M/blob/main/VOICES.md"""
speakers = [
# 🇺🇸 American English: 11F 9M
# Overall Grade 'A'
"af_heart",
# Overall Grade 'C+'
"af_aoede","af_kore","af_sarah",
"am_fenrir","am_michael","am_puck",
# Other
"af_alloy", "af_bella", "af_jessica", "af_nicole", "af_nova", "af_river", "af_sky", "am_adam", "am_echo", "am_eric", "am_liam", "am_onyx", "am_santa", "bf_alice",
# 🇬🇧 British English: 4F 4M
"bf_emma", "bf_isabella", "bf_lily", "bm_daniel", "bm_fable", "bm_george", "bm_lewis",
# 🇧🇷 Brazilian Portuguese: 1F 2M
"pf_dora",
"pm_alex",
"pm_santa"]
# Old list:
# ["af_heart", "af_joy", "af_sad", "af_angry", "af_fear", "af_surprise"]
return speakers
def gen_speaker_samples(
self,
samples: list =None,
output_path: PathLike=None,
) -> list[str]:
"""Generate sample speakers audio in output_path."""
result = self._build_command("gen_samples.py", *samples, output_path=output_path)
return [str(self._normalize_path(s)) for s in samples]
def gen_speaker_samples():
if torch.cuda.is_available():
print('CUDA GPU available')
torch.set_default_device('cuda')
for speaker in speakers:
file = speaker + "_sample.wav"
if os.path.exists(file):
print(f"Sample for {speaker} already exists.")
continue
else:
print(f"Creating {speaker}")
pipeline = KPipeline(lang_code=speaker[0])
sentence = f"Hello, this voice is {speaker[3:]}. The quick brown fox jumped over the lazy dog. The fish twisted and turned on the bent hook. Press the pants and sew a button on the vest. The swan dive was far short of perfect."
audio_segments = []
for gs, ps, audio in pipeline(
sentence,
repo_id='hexgrad/Kokoro-82M',
voice=speaker,
speed=1,
split_pattern=r'\n\n\n'):
audio_segments.append(audio)
final_audio = np.concatenate(audio_segments)
soundfile.write(file, final_audio, 24000)