tts-kokoro engine underway
This commit is contained in:
@@ -0,0 +1,111 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa
|
||||
# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html)
|
||||
"""Postprocess audio files into an audiobook.
|
||||
"""
|
||||
import subprocess
|
||||
|
||||
|
||||
|
||||
# stdlib modules
|
||||
import os
|
||||
import sys
|
||||
|
||||
# pip installed packages
|
||||
import numpy as np
|
||||
import soundfile
|
||||
import torch
|
||||
from tqdm import tqdm
|
||||
from kokoro import KPipeline
|
||||
from ebooklib import epub
|
||||
import soundfile as sf
|
||||
from mutagen import mp4
|
||||
from pydub import AudioSegment
|
||||
from mutagen import mp4
|
||||
# Local imports
|
||||
from .tts_generic import GenericTTSBackend
|
||||
from . import log, PathLike
|
||||
from .preprocess import preprocess_book
|
||||
|
||||
def generate_metadata(files, author, title, chapter_titles):
|
||||
chap = 0
|
||||
start_time = 0
|
||||
with open("FFMETADATAFILE", "w") as file:
|
||||
file.write(";FFMETADATA1\n")
|
||||
file.write(f"ARTIST={author}\n")
|
||||
file.write(f"ALBUM={title}\n")
|
||||
file.write(f"TITLE={title}\n")
|
||||
file.write("DESCRIPTION=Made with https://github.com/aedocw/epub2tts-kokoro\n")
|
||||
for file_name in files:
|
||||
duration = get_duration(file_name)
|
||||
file.write("[CHAPTER]\n")
|
||||
file.write("TIMEBASE=1/1000\n")
|
||||
file.write(f"START={start_time}\n")
|
||||
file.write(f"END={start_time + duration}\n")
|
||||
file.write(f"title={chapter_titles[chap]}\n")
|
||||
chap += 1
|
||||
start_time += duration
|
||||
|
||||
def get_duration(file_path):
|
||||
audio = AudioSegment.from_file(file_path)
|
||||
duration_milliseconds = len(audio)
|
||||
return duration_milliseconds
|
||||
|
||||
def make_m4b(files, sourcefile, speaker):
|
||||
filelist = "filelist.txt"
|
||||
basefile = sourcefile.replace(".txt", "")
|
||||
outputm4a = f"{basefile} ({speaker}).m4a"
|
||||
outputm4b = f"{basefile} ({speaker}).m4b"
|
||||
with open(filelist, "w") as f:
|
||||
for filename in files:
|
||||
filename = filename.replace("'", "'\\''")
|
||||
f.write(f"file '{filename}'\n")
|
||||
ffmpeg_command = [
|
||||
"ffmpeg",
|
||||
"-f",
|
||||
"concat",
|
||||
"-safe",
|
||||
"0",
|
||||
"-i",
|
||||
filelist,
|
||||
"-codec:a",
|
||||
"flac",
|
||||
"-f",
|
||||
"mp4",
|
||||
"-strict",
|
||||
"-2",
|
||||
outputm4a,
|
||||
]
|
||||
subprocess.run(ffmpeg_command)
|
||||
ffmpeg_command = [
|
||||
"ffmpeg",
|
||||
"-i",
|
||||
outputm4a,
|
||||
"-i",
|
||||
"FFMETADATAFILE",
|
||||
"-map_metadata",
|
||||
"1",
|
||||
"-codec",
|
||||
"aac",
|
||||
outputm4b,
|
||||
]
|
||||
subprocess.run(ffmpeg_command)
|
||||
os.remove(filelist)
|
||||
os.remove("FFMETADATAFILE")
|
||||
os.remove(outputm4a)
|
||||
for f in files:
|
||||
os.remove(f)
|
||||
return outputm4b
|
||||
|
||||
def add_cover(cover_img, filename):
|
||||
try:
|
||||
if os.path.isfile(cover_img):
|
||||
m4b = mp4.MP4(filename)
|
||||
cover_image = open(cover_img, "rb").read()
|
||||
m4b["covr"] = [mp4.MP4Cover(cover_image)]
|
||||
m4b.save()
|
||||
else:
|
||||
print(f"Cover image {cover_img} not found")
|
||||
except:
|
||||
print(f"Cover image {cover_img} not found")
|
||||
Reference in New Issue
Block a user