Unified aedocw backends

This commit is contained in:
2026-08-07 00:03:19 -03:00
parent da8940bdd4
commit b1c7a42172
12 changed files with 169 additions and 397 deletions
+4 -1
View File
@@ -137,6 +137,8 @@ ENV/
env.bak/ env.bak/
venv.bak/ venv.bak/
.tmp-samples/ .tmp-samples/
.tmp*
.venv*
# Spyder project settings # Spyder project settings
.spyderproject .spyderproject
@@ -223,7 +225,8 @@ $RECYCLE.BIN/
.LSOverride .LSOverride
# Icon must end with two \r # Icon must end with two \r
Icon␍ Icon
# Thumbnails # Thumbnails
._* ._*
+1
View File
@@ -0,0 +1 @@
3.11
+29
View File
@@ -2,6 +2,35 @@
Python project to gather several projects that convert EPUB files to audiobooks (M4B files, mostly), with different TTS backends. Python project to gather several projects that convert EPUB files to audiobooks (M4B files, mostly), with different TTS backends.
# TL;DR
## Setup
```sh
git clone https://git.silveirarosa.com/renatoxsr/epub-tts
cd .\epub-tts\
pyenv local 3.11
python3 -m venv .venv
git submodule update --init --recursive
```
## Install on Windows
```pwsh
.\.venv\Scripts\python.exe -m pip install -e '.[dev]'
.\.venv\Scripts\download-epub-minimal.exe
.\.venv\Scripts\download-moby-dick.exe
.\.venv\Scripts\download-moby-dick-media-overlays.exe
.\.venv\Scripts\epub-tts.exe -i .\.tmp-samples\minimal.epub -o .\.tmp-m4b\minimal.m4b -b epub2tts-edge
```
## Install on POSIX
```sh
.\.venv\bin\python -m pip install -e '.[dev]'
.\.venv\bin\download-epub-minimal
.\.venv\bin\download-moby-dick
.\.venv\bin\download-moby-dick-media-overlays
.\.venv\bin\epub-tts -i ./.tmp-samples/minimal.epub -o ./.tmp-m4b/minimal.m4b -b epub2tts-edge
```
# Requirements # Requirements
Recent Python (>=3.11), pip (>=22.3), some sort of virtula environment recommended (such as the builtin venv package), and a Rust compatible platform (avoid embedded systems which do not have pre-built binaries for the compiled requirements, e.g., Android's Termux try to build everything from sources and fails miserably when doing so). Recent Python (>=3.11), pip (>=22.3), some sort of virtula environment recommended (such as the builtin venv package), and a Rust compatible platform (avoid embedded systems which do not have pre-built binaries for the compiled requirements, e.g., Android's Termux try to build everything from sources and fails miserably when doing so).
+19 -6
View File
@@ -23,7 +23,9 @@ optional arguments:
-b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro} -b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}
Backend to use for TTS Backend to use for TTS
""" """
import os
import sys import sys
from pathlib import Path
from argparse import ArgumentParser from argparse import ArgumentParser
from . import tts_aedocw, tts_edge, tts_generic from . import tts_aedocw, tts_edge, tts_generic
@@ -39,6 +41,7 @@ backend = {
"epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge, "epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge,
"epub2tts-chatterbox": tts_aedocw.AedocwChatterbox, "epub2tts-chatterbox": tts_aedocw.AedocwChatterbox,
"epub2tts-kokoro": tts_aedocw.AedocwKokoro, "epub2tts-kokoro": tts_aedocw.AedocwKokoro,
"generic-epub2tts": tts_aedocw.Epub2TTS
} }
@@ -63,18 +66,28 @@ def main(args=None):
) )
args = p.parse_args(args or sys.argv[1:]) args = p.parse_args(args or sys.argv[1:])
print( print(
f"Converting {args.input} to {args.output} using voice " f"Converting {args.input} to {args.output} "
f"{args.voice}, language {args.language}, " f"using {args.backend} backend "
f"backend {args.backend}, ..." "and args: " + ", ".join(
f"{k}={v}" for k, v in vars(args).items()
if k not in ("input", "output", "backend") and v is not None)
) )
input_path = Path(args.input).resolve()
if not input_path.exists():
raise FileNotFoundError(f"Input file {input_path} does not exist.")
output_path = Path(args.output).resolve()
if not output_path.parent.exists():
os.makedirs(output_path.parent, exist_ok=True)
backend_cls = backend[args.backend] backend_cls = backend[args.backend]
backend_instance = backend_cls( backend_instance = backend_cls(
backend_cmd=None, backend_cmd=None,
language=args.language,
voice=args.voice,
) )
backend_instance.run(args.input, args.output) backend_instance.run(
input_path, output_path,
**vars(args),
)
if __name__ == "__main__": if __name__ == "__main__":
+25 -94
View File
@@ -13,12 +13,6 @@ The vendored packages are run in isolated virtual environments when the
corresponding console script is not available in the current environment. corresponding console script is not available in the current environment.
""" """
from typing import Optional
from . import tts_epub2tts as epub2tts_adapter
from . import tts_epub2tts_chatterbox as epub2tts_chatterbox_adapter
from . import tts_epub2tts_edge as epub2tts_edge_adapter
from . import tts_epub2tts_kokoro as epub2tts_kokoro_adapter
from .tts_generic import GenericTTSBackend from .tts_generic import GenericTTSBackend
@@ -35,110 +29,45 @@ class AedocwBackend(GenericTTSBackend):
but the vendored adapters may still run the package in their own venv if but the vendored adapters may still run the package in their own venv if
the console script is not present. the console script is not present.
""" """
FLAGS = {
_CMD_MAP = { "speaker": "--speaker",
"epub2tts": "epub2tts", "voice": "--speaker",
"epub2tts-edge": "epub2tts-edge", "language": "--language",
"epub2tts-chatterbox": "epub2tts-chatterbox", }
"epub2tts-kokoro": "epub2tts-kokoro", INTERMEDIATE_TXT = True
}
_RUNNER_MAP = {
"epub2tts": epub2tts_adapter.run,
"epub2tts-edge": epub2tts_edge_adapter.run,
"epub2tts-chatterbox": epub2tts_chatterbox_adapter.run,
"epub2tts-kokoro": epub2tts_kokoro_adapter.run,
}
def __init__(
self,
repo: str = "epub2tts",
backend_cmd: Optional[str] = None,
language: Optional[str] = None,
voice: Optional[str] = None,
):
if backend_cmd is None:
backend_cmd = self._CMD_MAP.get(repo, "epub2tts")
super().__init__(backend_cmd=backend_cmd, language=language, voice=voice)
self.repo = repo
def run(self, input_source, output_dest):
runner = self._RUNNER_MAP.get(self.repo)
if runner is None:
raise ValueError(f"Unknown repository backend: {self.repo}")
return runner(
input_source,
output_dest,
language=self.language,
voice=self.voice,
backend_cmd=self.backend_cmd,
)
class AedocwEpub2TTS(AedocwBackend): class AedocwEpub2TTS(AedocwBackend):
"""Backend configured for github.com/aedocw/epub2tts.""" """Backend configured for github.com/aedocw/epub2tts."""
pass
def __init__(
self,
backend_cmd: Optional[str] = None,
language: Optional[str] = None,
voice: Optional[str] = None,
):
super().__init__(
repo="epub2tts", backend_cmd=backend_cmd, language=language, voice=voice
)
class AedocwEpub2TTSEdge(AedocwBackend): class AedocwEpub2TTSEdge(AedocwBackend):
"""Backend configured for github.com/aedocw/epub2tts-edge.""" """Backend configured for github.com/aedocw/epub2tts-edge."""
DEFAULT_SPEAKER = "en-US-AndrewNeural"
def __init__(
self, class TTSEdge(AedocwEpub2TTSEdge):
backend_cmd: Optional[str] = None, """Backward-compatible edge backend wrapper."""
language: Optional[str] = None, pass
voice: Optional[str] = None,
):
super().__init__(
repo="epub2tts-edge",
backend_cmd=backend_cmd,
language=language,
voice=voice,
)
class AedocwChatterbox(AedocwBackend): class AedocwChatterbox(AedocwBackend):
"""Backend configured for github.com/aedocw/epub2tts-chatterbox.""" """Backend configured for github.com/aedocw/epub2tts-chatterbox."""
def __init__( DEFAULT_SAMPLE = "none"
self,
backend_cmd: Optional[str] = None,
language: Optional[str] = None,
voice: Optional[str] = None,
):
super().__init__(
repo="epub2tts-chatterbox",
backend_cmd=backend_cmd,
language=language,
voice=voice,
)
class AedocwKokoro(AedocwBackend): class AedocwKokoro(AedocwBackend):
"""Backend configured for github.com/aedocw/epub2tts-kokoro.""" """Backend configured for github.com/aedocw/epub2tts-kokoro."""
DEFAULT_SPEAKER = "af_heart"
def __init__(
self,
backend_cmd: Optional[str] = None, class Epub2TTS(AedocwBackend):
language: Optional[str] = None, """Backward-compatible chatterbox backend wrapper."""
voice: Optional[str] = None,
): pass
super().__init__(
repo="epub2tts-kokoro",
backend_cmd=backend_cmd,
language=language,
voice=voice,
)
__all__ = [ __all__ = [
@@ -147,4 +76,6 @@ __all__ = [
"AedocwEpub2TTSEdge", "AedocwEpub2TTSEdge",
"AedocwChatterbox", "AedocwChatterbox",
"AedocwKokoro", "AedocwKokoro",
"TTSEdge",
"Epub2TTS",
] ]
-14
View File
@@ -1,14 +0,0 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
"""Compatibility shim for the edge backend in epub-tts."""
from .tts_aedocw import AedocwEpub2TTSEdge
class TTSEdge(AedocwEpub2TTSEdge):
"""Backward-compatible edge backend wrapper."""
pass
-75
View File
@@ -1,75 +0,0 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
"""Adapter for the vendored github.com/aedocw/epub2tts repository."""
import subprocess
from pathlib import Path
from typing import Optional
from .vendor_utils import build_run_command
SUPPORTED_AUDIO_FORMATS = {
".m4b": "m4b",
".wav": "wav",
".flac": "flac",
}
def _normalize_path(path_input: Path | str) -> Path:
return Path(path_input)
def _build_command(
input_source: Path,
output_dest: Path,
language: Optional[str] = None,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> list[str]:
command = build_run_command("epub2tts", "epub2tts", backend_cmd=backend_cmd)
command.append(str(input_source))
if language:
command.extend(["--language", language])
if voice:
command.extend(["--speaker", voice])
output_format = SUPPORTED_AUDIO_FORMATS.get(output_dest.suffix.lower(), "m4b")
if output_format != "m4b":
command.extend(["--audioformat", output_format])
return command
def run(
input_source: Path | str,
output_dest: Path | str,
language: Optional[str] = None,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> subprocess.CompletedProcess:
input_path = _normalize_path(input_source)
output_path = _normalize_path(output_dest)
output_path.parent.mkdir(parents=True, exist_ok=True)
command = _build_command(
input_path,
output_path,
language=language,
voice=voice,
backend_cmd=backend_cmd,
)
completed = subprocess.run(command, cwd=str(input_path.parent), check=True)
generated_path = input_path.with_suffix(
SUPPORTED_AUDIO_FORMATS.get(output_path.suffix.lower(), "m4b")
)
if generated_path != output_path:
generated_path.replace(output_path)
return completed
-62
View File
@@ -1,62 +0,0 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
"""Adapter for the vendored github.com/aedocw/epub2tts-chatterbox repository."""
import subprocess
from pathlib import Path
from typing import Optional
from .vendor_utils import build_run_command
DEFAULT_SAMPLE = "none"
def _normalize_path(path_input: Path | str) -> Path:
return Path(path_input)
def _build_command(
input_source: Path,
sample: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> list[str]:
command = build_run_command(
"epub2tts-chatterbox",
"epub2tts_chatterbox",
backend_cmd=backend_cmd,
)
command.append(str(input_source))
if sample and sample != DEFAULT_SAMPLE:
command.extend(["--sample", sample])
return command
def run(
input_source: Path | str,
output_dest: Path | str,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> subprocess.CompletedProcess:
input_path = _normalize_path(input_source)
output_path = _normalize_path(output_dest)
output_path.parent.mkdir(parents=True, exist_ok=True)
sample_text = DEFAULT_SAMPLE
cwd = input_path.parent
if voice:
voice_path = Path(voice)
sample_text = voice_path.name
if voice_path.parent != Path(""):
cwd = str(voice_path.parent)
command = _build_command(input_path, sample=sample_text, backend_cmd=backend_cmd)
completed = subprocess.run(command, cwd=cwd, check=True)
generated_path = Path(cwd) / f"{input_path.stem} ({sample_text}).m4b"
if generated_path != output_path:
generated_path.replace(output_path)
return completed
-53
View File
@@ -1,53 +0,0 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
"""Adapter for the vendored github.com/aedocw/epub2tts-edge repository."""
import subprocess
from pathlib import Path
from typing import Optional
from .vendor_utils import build_run_command
DEFAULT_SPEAKER = "en-US-AndrewNeural"
def _normalize_path(path_input: Path | str) -> Path:
return Path(path_input)
def _build_command(
input_source: Path,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> list[str]:
command = build_run_command(
"epub2tts-edge", "epub2tts_edge", backend_cmd=backend_cmd
)
command.append(str(input_source))
if voice:
command.extend(["--speaker", voice])
return command
def run(
input_source: Path | str,
output_dest: Path | str,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> subprocess.CompletedProcess:
input_path = _normalize_path(input_source)
output_path = _normalize_path(output_dest)
output_path.parent.mkdir(parents=True, exist_ok=True)
speaker = voice or DEFAULT_SPEAKER
command = _build_command(input_path, voice=speaker, backend_cmd=backend_cmd)
completed = subprocess.run(command, cwd=str(input_path.parent), check=True)
generated_path = input_path.parent / f"{input_path.stem} ({speaker}).m4b"
if generated_path != output_path:
generated_path.replace(output_path)
return completed
-53
View File
@@ -1,53 +0,0 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
"""Adapter for the vendored github.com/aedocw/epub2tts-kokoro repository."""
import subprocess
from pathlib import Path
from typing import Optional
from .vendor_utils import build_run_command
DEFAULT_SPEAKER = "af_heart"
def _normalize_path(path_input: Path | str) -> Path:
return Path(path_input)
def _build_command(
input_source: Path,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> list[str]:
command = build_run_command(
"epub2tts-kokoro", "epub2tts_kokoro", backend_cmd=backend_cmd
)
command.append(str(input_source))
if voice:
command.extend(["--speaker", voice])
return command
def run(
input_source: Path | str,
output_dest: Path | str,
voice: Optional[str] = None,
backend_cmd: Optional[str] = None,
) -> subprocess.CompletedProcess:
input_path = _normalize_path(input_source)
output_path = _normalize_path(output_dest)
output_path.parent.mkdir(parents=True, exist_ok=True)
speaker = voice or DEFAULT_SPEAKER
command = _build_command(input_path, voice=speaker, backend_cmd=backend_cmd)
completed = subprocess.run(command, cwd=str(input_path.parent), check=True)
generated_path = input_path.parent / f"{input_path.stem} ({speaker}).m4b"
if generated_path != output_path:
generated_path.replace(output_path)
return completed
+84 -38
View File
@@ -8,16 +8,23 @@
This module provides a generic TTS backend that can be used to convert text to speech using various TTS engines. It defines a `GenericTTSBackend` class that can be initialized with a specific backend command, language, and voice. The class provides methods to normalize paths, determine input and output flags based on the type of input/output (file or directory), build the command for the TTS engine, and run the command. This module provides a generic TTS backend that can be used to convert text to speech using various TTS engines. It defines a `GenericTTSBackend` class that can be initialized with a specific backend command, language, and voice. The class provides methods to normalize paths, determine input and output flags based on the type of input/output (file or directory), build the command for the TTS engine, and run the command.
Classes: Classes:
GenericTTSBackend: GenericTTSBackend:
Parameters: Parameters:
backend_cmd (str): Command or executable for the backend. backend_cmd (str): Command or executable for the backend.
language (Optional[str]): Language code for TTS output. language (Optional[str]): Language code for TTS output.
voice (Optional[str]): Voice name for TTS output. voice (Optional[str]): Voice name for TTS output.
Methods: Methods:
_normalize_path(value: PathLike) -> Path: Normalize a string or Path to Path. _normalize_path(value: PathLike) -> Path: Normalize a string or Path to Path.
_input_flag(input_path: Path) -> str: Return backend flag for input file or directory.
_output_flag(output_path: Path) -> str: Return backend flag for output file or directory. gen_input_flag(input_path: Path) -> str: Return backend flag for input file or directory.
gen_output_flag(output_path: Path) -> str: Return backend flag for output file or directory.
_build_command(input_source: PathLike, output_dest: PathLike) -> list[str]: Build command list for subprocess. _build_command(input_source: PathLike, output_dest: PathLike) -> list[str]: Build command list for subprocess.
run(input_source: PathLike, output_dest: PathLike) -> subprocess.CompletedProcess: Execute backend command. run(input_source: PathLike, output_dest: PathLike) -> subprocess.CompletedProcess: Execute backend command.
""" """
@@ -28,52 +35,91 @@ from typing import Optional, Union
PathLike = Union[str, Path] PathLike = Union[str, Path]
from .vendor_utils import build_run_command
class GenericTTSBackend: class GenericTTSBackend:
def __init__( """Generic TTS backend for handling text-to-speech conversion.
self,
backend_cmd: str, Parameters:
language: Optional[str] = None, backend_cmd (str): Command or executable for the backend.
voice: Optional[str] = None, language (Optional[str]): Language code for TTS output.
): voice (Optional[str]): Voice name for TTS output.
self.backend_cmd = backend_cmd """
self.language = language
self.voice = voice FLAGS = {}
INTERMEDIATE_TXT = False
DEFAULT_SPEAKER = None
DEFAULT_SAMPLE = None
SUPPORTED_AUDIO_FORMATS = {
# subclasses may override this
".m4b": "m4b",
".wav": "wav",
".flac": "flac",
".mp3": "mp3",
".ogg": "ogg",
}
_INPUT_FLAG_MAP = lambda x: [
"--input-dir", x] if x else ["--input-file", x]
_OUTPUT_FLAG_MAP = lambda x: [
"--output-dir", x] if x else ["--output-file", x]
def __init__(self,**kwargs):
for k,v in kwargs.items():
setattr(self, k, v)
def _normalize_path(self, value: PathLike) -> Path: def _normalize_path(self, value: PathLike) -> Path:
if isinstance(value, Path): if isinstance(value, Path):
return value return value.resolve()
return Path(value) else:
return Path(value).resolve()
def _input_flag(self, input_path: Path) -> str:
return "--input-dir" if input_path.is_dir() else "--input-file"
def _output_flag(self, output_path: Path) -> str: def gen_input_flag(self, input_path: Path) -> str:
return "--output-dir" if output_path.is_dir() else "--output-file" return self._INPUT_FLAG_MAP(self._normalize_path(input_path).is_dir())
def _build_command(
self, input_source: PathLike, output_dest: PathLike
) -> list[str]:
input_path = self._normalize_path(input_source)
output_path = self._normalize_path(output_dest)
command = [ def gen_output_flag(self, output_path: Path) -> str:
self.backend_cmd, return self._OUTPUT_FLAG_MAP(self._normalize_path(output_path).is_dir())
self._input_flag(input_path),
str(input_path),
self._output_flag(output_path),
str(output_path),
]
if self.language:
command.extend(["--language", self.language])
if self.voice:
command.extend(["--voice", self.voice])
def _build_command(self, input_source: PathLike, output_dest: PathLike) -> list[str]:
command = [self.backend_cmd]
command.extend(self.gen_input_flag(input_source))
command.extend(self.gen_output_flag(output_dest))
for flag_name, var_name in self.FLAGS.items():
if getattr(self, flag_name, None) is not None:
command.extend([
f"--{flag_name.replace('_', '-')}",
getattr(self, var_name)])
return command return command
def run(self,
input_source: PathLike,
output_dest: PathLike,
**kwargs
) -> subprocess.CompletedProcess:
def run( completed = subprocess.run(
self, input_source: PathLike, output_dest: PathLike self._build_command(input_source, output_dest),
) -> subprocess.CompletedProcess: cwd=str(self._normalize_path(input_source).parent),
command = self._build_command(input_source, output_dest) check=True)
return subprocess.run(command, check=True)
if self.INTERMEDIATE_TXT:
txt_file = self._normalize_path(input_source).with_suffix(".txt")
if txt_file.exists():
completed = subprocess.run(
self._build_command(txt_file, output_dest),
cwd=str(self._normalize_path(input_source).parent),
check=True
)
return completed
+7 -1
View File
@@ -64,4 +64,10 @@ def build_run_command(
if backend_cmd is None or backend_cmd == repo_name: if backend_cmd is None or backend_cmd == repo_name:
python_executable = ensure_venv(repo_name) python_executable = ensure_venv(repo_name)
return [str(python_executable), "-m", module_name] return [str(python_executable), "-m", module_name]
return [backend_cmd] else:
ensure_venv(repo_name)
venv_path = get_venv_path(repo_name)
return [str( venv_path / "Scripts" / f"{backend_cmd}.exe")
if sys.platform == "win32"
else str(venv_path / "bin" / backend_cmd)]