diff --git a/src/epub_tts/__init__.py b/src/epub_tts/__init__.py index 42e1206..c566d43 100644 --- a/src/epub_tts/__init__.py +++ b/src/epub_tts/__init__.py @@ -13,22 +13,8 @@ except PackageNotFoundError: # package is not installed pass -from typing import Union -from pathlib import Path -import os -import pprint -def log (message, level=None): - pp = pprint.PrettyPrinter( - indent=4, - width=os.get_terminal_size().columns, - compact=False, - depth=None - ) - if isinstance(message, (dict, list, tuple, set)): - formatted = pp.pformat(message) - print(formatted.replace("', '", "',\n'")) - else: - pp.pprint(message) -PathLike = Union[str, Path] \ No newline at end of file +from .logger import logger, loglevel_map, get_logger +from .utils import DataDict, PathLike, path +from .app import App diff --git a/src/epub_tts/__main__.py b/src/epub_tts/__main__.py index 44e0140..af87a33 100644 --- a/src/epub_tts/__main__.py +++ b/src/epub_tts/__main__.py @@ -1,31 +1,10 @@ #!/usr/bin/env python3 # -*- coding: utf-8 -*- # Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa -# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> # See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) -"""Command-line interface for converting EPUB to audio using TTS. -This module provides a command-line interface (CLI) for converting EPUB files to audio using text-to-speech (TTS) backends. It allows users to specify the input EPUB file, output audio file, language, voice, and backend to use for TTS conversion. - -usage: epub-tts [-h] [--version] -i INPUT -o OUTPUT [-l LANGUAGE] - [-v VOICE] [-b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}] - -optional arguments: - -h, --help show this help message and exit - --version Show version and exit - -i INPUT, --input INPUT - Input EPUB file - -o OUTPUT, --output OUTPUT - Output audio file - -l LANGUAGE, --language LANGUAGE - Language to use for TTS - -v VOICE, --voice Voice to use for TTS - -b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro} - Backend to use for TTS -""" -# stdlib -import os import sys +<<<<<<< HEAD from pathlib import Path from argparse import ArgumentParser @@ -132,8 +111,10 @@ def main(args=None): **vars(args), ) +======= +from .cli import cli +>>>>>>> dev if __name__ == "__main__": - sys.exit(main()) - + sys.exit(cli()) diff --git a/src/epub_tts/app.py b/src/epub_tts/app.py new file mode 100644 index 0000000..1a21f26 --- /dev/null +++ b/src/epub_tts/app.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa +# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> +# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) +"""config.py + +The industry-standard hierarchy for application configuration is: +1. Command-Line Arguments (argparse) (Highest priority – explicit runtime overrides). +2. Environment Variables (os.environ) (Medium priority – deployment or session-specific configuration). +3. Configuration Files (e.g., .env, .ini, .yaml files) +4. Hardcoded Defaults (Lowest priority – fallback safety net) +""" +# stdlib +from argparse import ArgumentParser +from collections import UserDict +from pathlib import Path +from typing import Optional +import logging +import os +import re +import shutil +import sys + + +# Pip packages +from dotenv import load_dotenv + +# Local imports +from . import tts_aedocw, tts_generic, tts_kokoro +from . import logger, PathLike, loglevel_map +from . import DataDict + +class App(): + + WHICH = ["ffmpeg"] # Needed in $PATH + + BACKENDS = { + # keys are the backend names, values are the corresponding TTS classes + "default": tts_generic.GenericTTSBackend, + "edge": tts_aedocw.TTSEdge, + # aedocw-backed implementations — these expect the corresponding + # package "console scripts"/entrypoints to be available in the + # environment (or an explicit backend_cmd to be provided). + "epub2tts": tts_aedocw.AedocwEpub2TTS, + "epub2tts-edge": tts_aedocw.AedocwEpub2TTSEdge, + "epub2tts-chatterbox": tts_aedocw.AedocwChatterbox, + "epub2tts-kokoro": tts_aedocw.AedocwKokoro, + "generic-epub2tts": tts_aedocw.Epub2TTS, + "kokoro": tts_kokoro.KokoroBackend, + } + + def __init__(self, + input_str: Optional[str|Path|list[str|Path]], + output_str: Optional[str|Path], + backend: str, + ): + + + # Validate and sanitize output + self.output_path = App.path(self.output) + if len(self.input) > 1 and not self.output_path.is_dir(): + self.logger.error("Output must be a directory when multiple input files are provided. Invalid: '%s'", self.output_path) + sys.exit(1) + del self._data["output"] + + # Validate and sanitize input + self.input_files = {} + self.input_dirs = [] + self.logger.info("Paths provided as input: %d", len(self.input)) + for i, input_str in enumerate(self.input): + self.logger.info("[Input %d] %s", i+1, input_str) + input_path = App.path(input_str) + if not input_path.exists(): + self.logger.warning("Skipping path (do not exist): '%s'.", input_path) + continue + if input_path.is_dir(): + self.input_dir.append(input_path) + self.logger.info("Added directory to queue: '%s'", input_path) + else: + self.input_files[input_str] = {"path":input_path} + self.input_files[input_str]["filename"] = self.input_files[input_str]["path"].name + self.input_files[input_str]["stem"] = self.input_files[input_str]["path"].stem + self.input_files[input_str]["suffix"] = self.input_files[input_str]["path"].suffix + #self.input_files[input_str]["filetype"] = ebook_meta self.input_files[input_str]["path"] + self.logger.info("Added file to queue: %s", self.input_files[input_str]) + del self._data["input"] + + + # Sanitize replace_map + self.logger.debug(self.replace) + if self.replace and len(self.replace) % 2 == 0: + self.replace_map = { old: new for old, new in self.replace} + del self.replace + + + # print args + self.logger.debug(self) + + # Execute actions + if self._parsed_args.check_env: + self.check_env() + + self.engine = App.BACKENDS[self.backend](self) + + + # Utility methods + def check_env(self, which=None): + """Check anvironment for necessary binaries""" + if not which: + which = self.WHICH + if isinstance(which, str): + if " " in which: + which = which.split() + which = [which] + for bin in which: + if sys.platform == "win32": + p = shutil.which(bin + ".exe") + else: + p = shutil.which(bin) + if not p: + self.logger.error(f"%s is either NOT installed or " + "NOT in your system PATH.", bin) + return False + else: + self.logger.info(f"%s found at: %s", bin, p) + return True diff --git a/src/epub_tts/cli.py b/src/epub_tts/cli.py new file mode 100644 index 0000000..6642b34 --- /dev/null +++ b/src/epub_tts/cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa +# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> +# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) +"""Command-line interface for converting EPUB to audio using TTS. + +This module provides a command-line interface (CLI) for converting EPUB files to audio using text-to-speech (TTS) backends. It allows users to specify the input EPUB file, output audio file, language, voice, and backend to use for TTS conversion. + +usage: epub-tts [-h] [--version] -i INPUT -o OUTPUT [-l LANGUAGE] + [-v VOICE] [-b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}] + +optional arguments: + -h, --help show this help message and exit + --version Show version and exit + -i INPUT, --input INPUT + Input EPUB file + -o OUTPUT, --output OUTPUT + Output audio file + -l LANGUAGE, --language LANGUAGE + Language to use for TTS + -v VOICE, --voice Voice to use for TTS + -b {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro}, --backend {default,edge,epub2tts,epub2tts-edge,epub2tts-chatterbox,epub2tts-kokoro} + Backend to use for TTS +""" +# stdlib +import os +import sys +from pathlib import Path +from argparse import ArgumentParser + +# Local imports +from . import App +from . import logger, get_logger + +# Globals/Defaults +DEFAULT_ENV_FILE=".env" +ENV_FILE_OVERRIDE=False +DEBUG = False +LOGLEVEL = "WARNING" + + +def get_parser(): + p = ArgumentParser(description="Convert EPUB to audio using TTS") + p.add_argument( + "--version", + help="Show version and exit", + action="version", + version=f"%(prog)s {__import__('epub_tts').__version__}") + + # General options + p.add_argument( + "--debug", dest="debug_level", action="store_true", + help="Enable debug logging and verbose diagnostics") + p.add_argument( + "--verbose", action="store_true", + help="Enable info-level logging") + p.add_argument( + "--loglevel", + type=str.upper, default=logging.INFO, + choices=["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"], + help="Set logging level, from most verbose to least verbose:") + p.add_argument( + "--logfile", default=None, + help="Write logs to the specified file") + + # Input options and processing + p.add_argument( + "-i", "--input", + nargs="+", required=True, + help="Input EPUB file") + p.add_argument( + "-r", "--replace", action="append", nargs=2, + help="Replace text in the intermediate output. " + "Specify pairs of old_text new_text. " + "Can be used multiple times.") + p.add_argument( + "--check-env",action="store_true", + help="Check the runtime environment and exit") + p.add_argument( + "-b", "--backend", default="default", + choices=App.BACKENDS.keys(), + help="Backend to use for TTS") + + # Output options + p.add_argument( + "-o", "--output", + help="Output audio file", required=True) + p.add_argument( + "-c", "--cover", + help="Path to cover image to embed or use " + "for output metadata") + + # Speech options + p.add_argument( + "-l", "--language", + help="Language to use for TTS (see your " + "backend's documentation for available voices)") + p.add_argument( + "-v", "--voice", + help="Voice to use for TTS (see your backend's " + "documentation for available voices)") + p.add_argument( + "--speed", type=float, default=1.0, + help="Playback speed multiplier (default: 1.0) " + "(not all backends support this)") + p.add_argument( + "--short-pause", type=int, default=None, + help="Short pause duration in milliseconds between " + "phrases or sentences (not all backends support this)") + p.add_argument( + "--long-pause", type=int, default=None, + help="Long pause duration in milliseconds between " + "sections or paragraphs (not all backends support this)") + p.add_argument( + "--notitles", action="store_true", + help="Do not read chapter titles") + + return p + +def setup_app(cmdline:str|list[str]=None): + + logger = get_logger() + parser = get_parser() + parsed_args = p.parse_args(cmdline) + if "env_file" not in self._data: + self._data["env_file"] = App.DEFAULT_ENV_FILE + self.logger.debug(self._data["env_file"]) + + + # for path_flag in ["env_file"]: + # if not Path(self[path_flag]).exists(): + # self.logger.warning(f"Could not read %s. Ignoring '%s' and using '%s'", path_flag, self[path_flag], App.DEFAULT_ENV_FILE) + # self[path_flag] = App["DEFAULT_" + path_flag.upper()] + + + + # load .env/env_file + load_dotenv(self.env_file, override=os.environ.get("ENV_FILE_OVERRIDE",App.ENV_FILE_OVERRIDE)) + + # Now we can call argparser with correct defaults and logger/loglevel + # Get ArgumentParser and pased *args or read from sys.argv + self._parsed_args = self.get_parser().parse_args(args or sys.argv[1:]) + for k,v in self._parsed_args.__dict__.items(): + self[k] = v + +def cli(args=None): + app = App() + app.engine(**vars(args)) + app.engine.run( + app.file, app.output_dest, + **vars(args), + ) + diff --git a/src/epub_tts/ebook_meta.py b/src/epub_tts/ebook_meta.py new file mode 100644 index 0000000..680c0e1 --- /dev/null +++ b/src/epub_tts/ebook_meta.py @@ -0,0 +1,3558 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa +# Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> +# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) +"""ebook_meta.py + +Get and set unique ids from books (epub, mobi and azw3) + +Extract identifiers from EPUB, MOBI, AZW, and AZW3 files. + +Identifier priority: + +1. calibre_uuid +2. asin +3. isbn +4. content_hash + +""" + +# stdlib +from __future__ import annotations +import argparse +import hashlib +import json +import logging +import re +import signal +import sqlite3 +import struct +import sys +import unicodedata +import urllib.parse +import urllib.request +import xml.etree.ElementTree as ET +import zipfile +from pathlib import Path + + +# ============================================================ +# Constants +# ============================================================ + +SUPPORTED_FORMATS = { + ".epub", + ".mobi", + ".azw", + ".azw3", +} + +ASIN_REGEX = re.compile( + r"\b(B0[A-Z0-9]{8})\b", + re.IGNORECASE, +) + +EXTH_ASIN_RECORDS = {113, 504} +EXTH_ISBN_RECORDS = {104} + + +# ============================================================ +# Exceptions +# ============================================================ + +class UnsupportedFormatError(Exception): + """ + Raised when a file is not a supported ebook format. + """ + + + +# ============================================================ +# Signal Handling +# ============================================================ + +def handle_sigint(signum, frame): + """ + Handle Ctrl+C by terminating immediately. + + Raises: + KeyboardInterrupt + """ + logger.warning( + "Interrupted by user (Ctrl+C)" + ) + + raise KeyboardInterrupt + + +# ============================================================ +# Utility Functions +# ============================================================ + +def normalize_identifier(value): + """ + Normalize ISBN/ASIN strings. + + Removes spaces and hyphens. + + Args: + value: + Raw identifier value. + + Returns: + Normalized identifier or None. + """ + + if not value: + return None + + value = value.strip() + + value = re.sub( + r"[\s\-]", + "", + value, + ) + + return value + + +def find_asin_by_regex(text): + """ + Search text for an ASIN-like value. + + Args: + text: + Text to search. + + Returns: + ASIN or None. + """ + + if not text: + return None + + match = ASIN_REGEX.search(text) + + if not match: + return None + + return match.group(1).upper() + + +def normalize_text(text): + """ + Normalize text before hashing. + + Args: + text: + Raw text. + + Returns: + Normalized text. + """ + + if not text: + return "" + + text = unicodedata.normalize( + "NFKC", + text, + ) + + text = text.replace( + "\r\n", + "\n", + ) + + text = text.replace( + "\r", + "\n", + ) + + text = re.sub( + r"\s+", + " ", + text, + ) + + return text.strip().lower() + + +def compute_content_hash(text): + """ + Compute SHA256 hash of normalized text. + + Args: + text: + Input text. + + Returns: + SHA256 hex digest. + """ + + normalized = normalize_text(text) + + return hashlib.sha256( + normalized.encode("utf-8") + ).hexdigest() + + +def decode_exth_value(data): + """ + Decode EXTH record data. + + Args: + data: + EXTH bytes. + + Returns: + Decoded text or None. + """ + + try: + return ( + data.decode( + "utf-8", + errors="ignore", + ) + .strip("\x00") + .strip() + ) + + except Exception: + return None + + +def validate_metadata_db(path): + """ + Validate metadata.db path. + + Args: + path: + Path to metadata.db + + Returns: + Path object. + + Raises: + FileNotFoundError + ValueError + """ + + db_path = Path(path) + + if not db_path.exists(): + raise FileNotFoundError( + f"metadata.db not found: {db_path}" + ) + + if not db_path.is_file(): + raise ValueError( + f"Not a file: {db_path}" + ) + + return db_path +``` + +This establishes all shared infrastructure. + +**Next section:** EPUB support (`get_epub_opf_path`, `extract_epub_text`, `read_epub_metadata`). + +*** + +# 2. + +Here's **Section 2: EPUB Support**. Add this after the utility functions section. + +```python +# ============================================================ +# EPUB Support +# ============================================================ + +def get_epub_opf_path(zf): + """ + Locate the OPF package document inside an EPUB. + + EPUB files are ZIP archives. The path to the OPF + package document is stored in: + + META-INF/container.xml + + Args: + zf: + Open zipfile.ZipFile instance. + + Returns: + Internal OPF path string. + + Raises: + RuntimeError: + If the OPF file cannot be located. + """ + + container_xml = ET.fromstring( + zf.read( + "META-INF/container.xml" + ) + ) + + ns = { + "c": + "urn:oasis:names:tc:opendocument:xmlns:container" + } + + rootfile = container_xml.find( + ".//c:rootfile", + ns, + ) + + if rootfile is None: + raise RuntimeError( + "Could not locate OPF package document" + ) + + return rootfile.attrib["full-path"] + + +def extract_epub_text(epub_path): + """ + Extract text from EPUB XHTML/HTML files. + + This function performs a simple HTML tag stripping + operation using regular expressions. It is primarily + intended for content hash generation. + + Args: + epub_path: + Path to EPUB file. + + Returns: + Combined text content. + """ + + text_parts = [] + + with zipfile.ZipFile( + epub_path, + "r", + ) as zf: + + for filename in zf.namelist(): + + if not filename.lower().endswith( + ( + ".xhtml", + ".html", + ".htm", + ) + ): + continue + + try: + + html = zf.read( + filename + ).decode( + "utf-8", + errors="ignore", + ) + + text = re.sub( + r"<[^>]+>", + " ", + html, + ) + + text_parts.append( + text + ) + + except Exception: + + logger.debug( + "Failed reading %s", + filename, + exc_info=True, + ) + + return "\n".join(text_parts) + + +def read_epub_metadata(epub_path): + """ + Extract metadata from an EPUB. + + Extracts: + + - calibre_uuid + - ASIN + - ISBN + - content_hash + + Args: + epub_path: + Path to EPUB file. + + Returns: + Metadata dictionary. + """ + + logger.info( + "Reading EPUB metadata: %s", + epub_path, + ) + + meta = { + "format": "EPUB", + "calibre_uuid": None, + "asin": None, + "isbn": None, + "content_hash": None, + "title": None, + "author": None, + } + + with zipfile.ZipFile( + epub_path, + "r", + ) as zf: + + opf_path = get_epub_opf_path(zf) + + logger.debug( + "OPF path: %s", + opf_path, + ) + + root = ET.fromstring( + zf.read(opf_path) + ) + + identifier_map = {} + + # -------------------------------------------- + # Pass 1: identifiers and basic metadata + # -------------------------------------------- + + for elem in root.iter(): + + tag = elem.tag.lower() + + # Title + if tag.endswith("title"): + + if not meta["title"]: + meta["title"] = ( + elem.text or "" + ).strip() + + # Author + elif tag.endswith("creator"): + + if not meta["author"]: + meta["author"] = ( + elem.text or "" + ).strip() + + # Identifiers + elif tag.endswith( + "identifier" + ): + + value = ( + elem.text or "" + ).strip() + + identifier_id = ( + elem.attrib.get("id") + ) + + if identifier_id: + identifier_map[ + identifier_id + ] = value + + scheme = ( + elem.attrib.get( + "scheme" + ) + or elem.attrib.get( + "{http://www.idpf.org/2007/opf}scheme" + ) + or "" + ).lower() + + logger.debug( + "Identifier: scheme=%s value=%s", + scheme, + value, + ) + + if ( + scheme == "isbn" + ): + + meta["isbn"] = ( + normalize_identifier( + value + ) + ) + + elif scheme in ( + "asin", + "amazon", + ): + + meta["asin"] = value + + if ( + meta["asin"] + is None + ): + + asin = ( + find_asin_by_regex( + value + ) + ) + + if asin: + meta["asin"] = ( + asin + ) + + # -------------------------------------------- + # Pass 2: meta tags + # -------------------------------------------- + + for elem in root.iter(): + + tag = elem.tag.lower() + + if not tag.endswith( + "meta" + ): + continue + + name = ( + elem.attrib.get( + "name", + "", + ).lower() + ) + + if ( + name + == "calibre:uuid" + ): + + meta[ + "calibre_uuid" + ] = ( + elem.attrib.get( + "content" + ) + or ( + elem.text + or "" + ).strip() + ) + + prop = ( + elem.attrib.get( + "property", + "", + ).lower() + ) + + if prop in ( + "asin", + "amazon:asin", + "identifier:asin", + ): + + value = ( + elem.text or "" + ).strip() + + if value: + meta["asin"] = ( + value + ) + + if ( + prop + == "identifier-type" + ): + + ref = ( + elem.attrib.get( + "refines" + ) + ) + + if ref: + + ref = ref.lstrip( + "#" + ) + + if ( + "asin" + in ( + elem.text + or "" + ).lower() + and ref + in identifier_map + ): + + meta["asin"] = ( + identifier_map[ + ref + ] + ) + + if ( + not meta["asin"] + and elem.text + ): + + asin = ( + find_asin_by_regex( + elem.text + ) + ) + + if asin: + meta["asin"] = ( + asin + ) + + # -------------------------------------------- + # Last-resort EPUB-wide regex scan + # -------------------------------------------- + + if ( + meta["asin"] + is None + ): + + logger.debug( + "Performing EPUB-wide ASIN scan" + ) + + all_text = [] + + for filename in ( + zf.namelist() + ): + + try: + + all_text.append( + zf.read( + filename + ).decode( + "utf-8", + errors="ignore", + ) + ) + + except Exception: + pass + + meta["asin"] = ( + find_asin_by_regex( + "\n".join(all_text) + ) + ) + + meta["content_hash"] = ( + compute_content_hash( + extract_epub_text( + epub_path + ) + ) + ) + + logger.info( + "EPUB metadata extracted " + "(uuid=%s asin=%s isbn=%s)", + bool( + meta["calibre_uuid"] + ), + bool( + meta["asin"] + ), + bool( + meta["isbn"] + ), + ) + + return meta +``` + +**Next section:** MOBI / AZW / AZW3 support (`extract_mobi_text()`, `read_mobi_metadata()`, EXTH parsing, and content hash generation). + +*** + +# 3. + +### Section 3: MOBI / AZW / AZW3 Support + +Add this section immediately after the EPUB functions. + +```python +# ============================================================ +# MOBI / AZW / AZW3 Support +# ============================================================ + +def extract_mobi_text(path): + """ + Extract printable text fragments from a Kindle ebook. + + This is intentionally lightweight and uses only the + Python standard library. It is sufficient for generating + a content hash fallback but is not a full Kindle parser. + + Args: + path: + Path to MOBI/AZW/AZW3 file. + + Returns: + Extracted text. + """ + + with open(path, "rb") as f: + data = f.read() + + text = data.decode( + "latin1", + errors="ignore", + ) + + fragments = re.findall( + r"[ -~]{20,}", + text, + ) + + return "\n".join(fragments) + + +def get_content_hash(path): + """ + Compute a content-based SHA256 hash for an ebook. + + EPUB: + Hash XHTML/HTML content. + + MOBI/AZW/AZW3: + Hash extracted printable text. + + Args: + path: + Ebook file path. + + Returns: + SHA256 digest or None. + """ + + suffix = ( + Path(path) + .suffix.lower() + ) + + try: + + if suffix == ".epub": + + text = extract_epub_text( + path + ) + + elif suffix in ( + ".mobi", + ".azw", + ".azw3", + ): + + text = extract_mobi_text( + path + ) + + else: + + return None + + if not text.strip(): + return None + + return compute_content_hash( + text + ) + + except Exception: + + logger.exception( + "Content hash generation failed " + "for %s", + path, + ) + + return None + + +def read_mobi_metadata(path): + """ + Extract metadata from MOBI/AZW/AZW3 files. + + Metadata sources: + + - EXTH records + - Regex fallback scan + - Content hash fallback + + Extracts: + + - calibre_uuid + - asin + - isbn + - content_hash + + Args: + path: + Kindle ebook path. + + Returns: + Metadata dictionary. + """ + + logger.info( + "Reading Kindle metadata: %s", + path, + ) + + meta = { + "format": ( + Path(path) + .suffix[1:] + .upper() + ), + "calibre_uuid": None, + "asin": None, + "isbn": None, + "content_hash": None, + "title": None, + "author": None, + } + + with open(path, "rb") as f: + data = f.read() + + mobi_pos = data.find( + b"MOBI" + ) + + if mobi_pos == -1: + + logger.warning( + "MOBI header not found: %s", + path, + ) + + text_dump = data.decode( + "latin1", + errors="ignore", + ) + + meta["asin"] = ( + find_asin_by_regex( + text_dump + ) + ) + + meta["content_hash"] = ( + get_content_hash( + path + ) + ) + + return meta + + mobi_length = struct.unpack( + ">L", + data[ + mobi_pos + 4: + mobi_pos + 8 + ] + )[0] + + exth_pos = ( + mobi_pos + mobi_length + ) + + # -------------------------------------------------------- + # EXTH not present + # -------------------------------------------------------- + + if ( + data[ + exth_pos: + exth_pos + 4 + ] + != b"EXTH" + ): + + logger.debug( + "EXTH header not found" + ) + + text_dump = data.decode( + "latin1", + errors="ignore", + ) + + meta["asin"] = ( + find_asin_by_regex( + text_dump + ) + ) + + meta["content_hash"] = ( + get_content_hash( + path + ) + ) + + return meta + + # -------------------------------------------------------- + # Parse EXTH records + # -------------------------------------------------------- + + record_count = ( + struct.unpack( + ">L", + data[ + exth_pos + 8: + exth_pos + 12 + ] + )[0] + ) + + logger.debug( + "EXTH record count: %d", + record_count, + ) + + pos = exth_pos + 12 + + for _ in range(record_count): + + try: + + rec_type = ( + struct.unpack( + ">L", + data[ + pos: + pos + 4 + ] + )[0] + ) + + rec_len = ( + struct.unpack( + ">L", + data[ + pos + 4: + pos + 8 + ] + )[0] + ) + + value_data = data[ + pos + 8: + pos + rec_len + ] + + text = ( + decode_exth_value( + value_data + ) + ) + + logger.debug( + "EXTH %d: %s", + rec_type, + text, + ) + + if text: + + # ---------------------------------- + # ASIN + # ---------------------------------- + + if ( + rec_type + in EXTH_ASIN_RECORDS + ): + meta["asin"] = ( + text + ) + + # ---------------------------------- + # ISBN + # ---------------------------------- + + elif ( + rec_type + in EXTH_ISBN_RECORDS + ): + + meta["isbn"] = ( + normalize_identifier( + text + ) + ) + + # ---------------------------------- + # Possible Calibre UUID + # ---------------------------------- + + elif rec_type == 112: + + if re.fullmatch( + r"[0-9A-Fa-f\\-]{32,36}", + text, + ): + meta[ + "calibre_uuid" + ] = text + + # ---------------------------------- + # Regex ASIN fallback + # ---------------------------------- + + if ( + meta["asin"] + is None + ): + + asin = ( + find_asin_by_regex( + text + ) + ) + + if asin: + meta["asin"] = ( + asin + ) + + pos += rec_len + + except Exception: + + logger.exception( + "Failed parsing EXTH record" + ) + + break + + # -------------------------------------------------------- + # Binary-wide ASIN scan + # -------------------------------------------------------- + + if ( + meta["asin"] + is None + ): + + logger.debug( + "Running binary-wide ASIN scan" + ) + + text_dump = data.decode( + "latin1", + errors="ignore", + ) + + meta["asin"] = ( + find_asin_by_regex( + text_dump + ) + ) + + # -------------------------------------------------------- + # Content hash fallback + # -------------------------------------------------------- + + meta["content_hash"] = ( + get_content_hash( + path + ) + ) + + logger.info( + "Kindle metadata extracted " + "(uuid=%s asin=%s isbn=%s)", + bool( + meta["calibre_uuid"] + ), + bool( + meta["asin"] + ), + bool( + meta["isbn"] + ), + ) + + return meta + + +def read_book_metadata(path): + """ + Dispatch metadata extraction by file type. + + Supported formats: + + - EPUB + - MOBI + - AZW + - AZW3 + + Args: + path: + Ebook file path. + + Returns: + Metadata dictionary. + + Raises: + UnsupportedFormatError + """ + + path = Path(path) + + suffix = path.suffix.lower() + + logger.debug( + "Detected extension: %s", + suffix, + ) + + if ( + suffix + not in SUPPORTED_FORMATS + ): + + raise UnsupportedFormatError( + f"Unsupported format: {suffix}" + ) + + if suffix == ".epub": + return read_epub_metadata( + path + ) + + return read_mobi_metadata( + path + ) + + +def get_best_identifier(meta): + """ + Select the best identifier. + + Priority: + + 1. calibre_uuid + 2. asin + 3. isbn + 4. content_hash + + Args: + meta: + Metadata dictionary. + + Returns: + (identifier_type, identifier_value) + """ + + if meta.get( + "calibre_uuid" + ): + return ( + "calibre_uuid", + meta[ + "calibre_uuid" + ], + ) + + if meta.get("asin"): + return ( + "asin", + meta["asin"], + ) + + if meta.get("isbn"): + return ( + "isbn", + meta["isbn"], + ) + + if meta.get( + "content_hash" + ): + return ( + "content_hash", + meta[ + "content_hash" + ], + ) + + return ("none", None) +``` + +### Next section +The next section should be **External ASIN Lookup Providers**: + +- `lookup_openlibrary()` +- `lookup_google_books()` +- `lookup_isbndb()` +- `lookup_asin()` + +followed by Calibre integration and CLI handling. + +*** + +# 4. +### Section 4: External ASIN Lookup Providers + +Add this section after the MOBI/AZW3 support functions. + +```python +# ============================================================ +# External ASIN Lookup Providers +# ============================================================ + +def lookup_openlibrary(isbn): + """ + Query Open Library using an ISBN. + + Open Library generally does not expose ASINs directly, + but can be useful for metadata enrichment. + + Args: + isbn: + ISBN identifier. + + Returns: + Parsed JSON response or None. + """ + + if not isbn: + return None + + url = ( + "https://openlibrary.org/api/books" + f"?bibkeys=ISBN:{urllib.parse.quote(isbn)}" + "&format=json" + "&jscmd=data" + ) + + logger.debug( + "OpenLibrary lookup: %s", + isbn, + ) + + try: + + with urllib.request.urlopen( + url, + timeout=10, + ) as response: + + data = json.loads( + response.read().decode("utf-8") + ) + + return data + + except Exception: + + logger.exception( + "OpenLibrary lookup failed " + "(ISBN=%s)", + isbn, + ) + + return None + + +def lookup_google_books(isbn): + """ + Query Google Books using ISBN. + + Attempts to discover an ASIN by searching + returned identifier fields. + + Args: + isbn: + ISBN identifier. + + Returns: + ASIN string or None. + """ + + if not isbn: + return None + + url = ( + "https://www.googleapis.com/books/v1/volumes" + f"?q=isbn:{urllib.parse.quote(isbn)}" + ) + + logger.debug( + "Google Books lookup: %s", + isbn, + ) + + try: + + with urllib.request.urlopen( + url, + timeout=10, + ) as response: + + data = json.loads( + response.read().decode("utf-8") + ) + + for item in data.get( + "items", + [], + ): + + volume_info = item.get( + "volumeInfo", + {}, + ) + + for identifier in volume_info.get( + "industryIdentifiers", + [], + ): + + value = identifier.get( + "identifier" + ) + + asin = find_asin_by_regex( + value + ) + + if asin: + + logger.info( + "ASIN found via Google Books: %s", + asin, + ) + + return asin + + return None + + except Exception: + + logger.exception( + "Google Books lookup failed " + "(ISBN=%s)", + isbn, + ) + + return None + + +def lookup_isbndb( + isbn, + api_key, +): + """ + Query ISBNdb. + + ISBNdb requires an API key. + + Args: + isbn: + ISBN identifier. + + api_key: + ISBNdb API key. + + Returns: + ASIN string or None. + """ + + if not isbn: + return None + + if not api_key: + return None + + url = ( + "https://api2.isbndb.com/book/" + f"{urllib.parse.quote(isbn)}" + ) + + request = urllib.request.Request( + url, + headers={ + "Authorization": api_key + }, + ) + + logger.debug( + "ISBNdb lookup: %s", + isbn, + ) + + try: + + with urllib.request.urlopen( + request, + timeout=10, + ) as response: + + data = json.loads( + response.read().decode("utf-8") + ) + + # Search the entire payload + payload = json.dumps( + data, + ensure_ascii=False, + ) + + asin = find_asin_by_regex( + payload + ) + + if asin: + + logger.info( + "ASIN found via ISBNdb: %s", + asin, + ) + + return asin + + return None + + except Exception: + + logger.exception( + "ISBNdb lookup failed " + "(ISBN=%s)", + isbn, + ) + + return None + + +def build_asin_candidates( + metadata, + isbndb_api_key=None, +): + """ + Collect possible ASIN candidates from + external sources. + + Args: + metadata: + Ebook metadata dictionary. + + isbndb_api_key: + Optional ISBNdb API key. + + Returns: + List of candidate dictionaries. + """ + + candidates = [] + + isbn = metadata.get("isbn") + + if not isbn: + return candidates + + # ------------------------------------------ + # Google Books + # ------------------------------------------ + + asin = lookup_google_books( + isbn + ) + + if asin: + + candidates.append( + { + "source": + "Google Books", + "asin": + asin, + "confidence": + "medium", + } + ) + + # ------------------------------------------ + # ISBNdb + # ------------------------------------------ + + asin = lookup_isbndb( + isbn, + isbndb_api_key, + ) + + if asin: + + candidates.append( + { + "source": + "ISBNdb", + "asin": + asin, + "confidence": + "medium", + } + ) + + return candidates + + +def choose_asin(candidates): + """ + Interactively prompt the user + to choose an ASIN. + + Args: + candidates: + List of candidate dictionaries. + + Returns: + Selected ASIN or None. + """ + + if not candidates: + return None + + print() + print("Possible ASIN matches:") + print() + + for index, candidate in enumerate( + candidates, + start=1, + ): + + print( + f"[{index}] " + f"{candidate['asin']} " + f"({candidate['source']})" + ) + + print( + f"[{len(candidates)+1}] " + "None of the above" + ) + + while True: + + try: + + choice = int( + input( + "\nSelect a match: " + ) + ) + + if ( + 1 + <= choice + <= len(candidates) + ): + return candidates[ + choice - 1 + ]["asin"] + + if ( + choice + == len(candidates) + 1 + ): + return None + + except ( + ValueError, + EOFError, + ): + pass + + print( + "Invalid selection." + ) + + +def lookup_asin( + metadata, + isbndb_api_key=None, + interactive=False, +): + """ + Attempt external ASIN discovery. + + Lookup order: + + 1. Google Books + 2. ISBNdb + 3. OpenLibrary (enrichment only) + + Args: + metadata: + Metadata dictionary. + + isbndb_api_key: + Optional ISBNdb API key. + + interactive: + If True, allow the user to + choose between candidates. + + Returns: + ASIN string or None. + """ + + if metadata.get("asin"): + return metadata["asin"] + + isbn = metadata.get("isbn") + + if not isbn: + return None + + logger.info( + "Attempting external ASIN lookup " + "for ISBN=%s", + isbn, + ) + + candidates = build_asin_candidates( + metadata, + isbndb_api_key, + ) + + if not candidates: + + # Enrichment only + lookup_openlibrary( + isbn + ) + + return None + + if interactive: + + return choose_asin( + candidates + ) + + return candidates[0]["asin"] +``` + +### Next section + +The next logical section is: + +**Calibre Integration** +- `find_book_id_by_calibre_uuid()` +- SQLite write-back +- Calibre API (`db.new_api`) write-back skeleton +- identifier update helpers +- CLI integration hooks for `--metadata-db`, `--write-asin`, and `--write-asin-calibre` + + +*** + +# 5. + + +### Section 5: Calibre Integration + +Add this section after the external ASIN lookup functions. + +```python +# ============================================================ +# Calibre Integration +# ============================================================ + +def find_book_id_by_calibre_uuid( + metadata_db, + calibre_uuid, +): + """ + Find a Calibre book id using a calibre_uuid. + + Example SQL: + + SELECT book + FROM identifiers + WHERE type = 'uuid' + AND val = ? + + Args: + metadata_db: + Path to metadata.db + + calibre_uuid: + calibre:uuid value + + Returns: + Calibre book id or None. + """ + + if not calibre_uuid: + return None + + conn = sqlite3.connect( + str(metadata_db) + ) + + try: + + cur = conn.cursor() + + cur.execute( + """ + SELECT book + FROM identifiers + WHERE type = 'uuid' + AND val = ? + """, + (calibre_uuid,), + ) + + row = cur.fetchone() + + if row: + return row[0] + + return None + + finally: + conn.close() + + +def get_book_identifiers_sqlite( + metadata_db, + book_id, +): + """ + Retrieve all identifier values for a book. + + Args: + metadata_db: + Path to metadata.db + + book_id: + Calibre book id + + Returns: + dict of identifiers. + """ + + conn = sqlite3.connect( + str(metadata_db) + ) + + try: + + cur = conn.cursor() + + cur.execute( + """ + SELECT type, val + FROM identifiers + WHERE book = ? + """, + (book_id,), + ) + + return { + rowrow[1] + for row in cur.fetchall() + } + + finally: + conn.close() + + +def set_asin_in_calibre_library( + metadata_db, + book_id, + asin, +): + """ + Direct SQLite ASIN update. + + WARNING: + Direct database manipulation bypasses + Calibre's metadata APIs. + + Args: + metadata_db: + Path to metadata.db + + book_id: + Calibre book id + + asin: + Amazon ASIN + + Returns: + True on success. + """ + + conn = sqlite3.connect( + str(metadata_db) + ) + + try: + + cur = conn.cursor() + + cur.execute( + """ + SELECT id + FROM identifiers + WHERE book = ? + AND type = 'amazon' + """, + (book_id,), + ) + + row = cur.fetchone() + + if row: + + cur.execute( + """ + UPDATE identifiers + SET val = ? + WHERE id = ? + """, + ( + asin, + row[0], + ), + ) + + else: + + cur.execute( + """ + INSERT INTO identifiers + (book, type, val) + VALUES + (?, 'amazon', ?) + """, + ( + book_id, + asin, + ), + ) + + conn.commit() + + logger.info( + "Updated ASIN using SQLite " + "(book_id=%s asin=%s)", + book_id, + asin, + ) + + return True + + finally: + conn.close() + + +def open_calibre_database( + metadata_db, +): + """ + Open Calibre database using + Calibre's Python API. + + Requires: + + from calibre.db.legacy + import LibraryDatabase + + Args: + metadata_db: + Path to metadata.db + + Returns: + LibraryDatabase instance. + """ + + from calibre.db.legacy import ( + LibraryDatabase, + ) + + return LibraryDatabase( + str(metadata_db) + ) + + +def find_book_id_by_uuid_new_api( + db, + calibre_uuid, +): + """ + Find a Calibre book id using + db.new_api. + + Args: + db: + Open LibraryDatabase instance. + + calibre_uuid: + UUID identifier. + + Returns: + book_id or None. + """ + + try: + + matches = db.new_api.search( + f'identifiers:"uuid:{calibre_uuid}"' + ) + + if not matches: + return None + + return next(iter(matches)) + + except Exception: + + logger.exception( + "new_api UUID search failed" + ) + + return None + + +def update_asin_via_calibre_api( + metadata_db, + calibre_uuid, + asin, +): + """ + Update ASIN using Calibre's API. + + Notes: + This implementation is intended + for Calibre 9.x and may require + adjustment for future versions. + + Workflow: + + identifiers:"uuid:" + -> book id + + get_metadata() + -> Metadata object + + set_identifiers() + -> update amazon id + + set_metadata() + -> persist changes + + Args: + metadata_db: + Path to metadata.db + + calibre_uuid: + calibre:uuid + + asin: + Amazon ASIN + + Returns: + book_id or None. + """ + + db = open_calibre_database( + metadata_db + ) + + try: + + book_id = ( + find_book_id_by_uuid_new_api( + db, + calibre_uuid, + ) + ) + + if book_id is None: + + logger.warning( + "UUID not found in " + "Calibre database: %s", + calibre_uuid, + ) + + return None + + mi = db.get_metadata( + book_id, + index_is_id=True, + ) + + identifiers = ( + mi.get_identifiers() + ) + + identifiers[ + "amazon" + ] = asin + + mi.set_identifiers( + identifiers + ) + + db.set_metadata( + book_id, + mi, + force_changes=True, + ) + + logger.info( + "Updated ASIN using Calibre API " + "(book_id=%s asin=%s)", + book_id, + asin, + ) + + return book_id + + finally: + + try: + db.close() + except Exception: + pass + + +def write_discovered_asin( + metadata, + args, +): + """ + Write a discovered ASIN back to Calibre. + + Priority: + + 1. Calibre API mode + 2. SQLite mode + + Args: + metadata: + Extracted ebook metadata. + + args: + Parsed CLI arguments. + + Returns: + True on success, False otherwise. + """ + + asin = metadata.get( + "asin" + ) + + uuid = metadata.get( + "calibre_uuid" + ) + + if not asin: + + logger.debug( + "No ASIN available to write" + ) + + return False + + if not uuid: + + logger.warning( + "Cannot write ASIN " + "without calibre_uuid" + ) + + return False + + metadata_db = ( + validate_metadata_db( + args.metadata_db + ) + ) + + try: + + if getattr( + args, + "write_asin_calibre", + False, + ): + + book_id = ( + update_asin_via_calibre_api( + metadata_db, + uuid, + asin, + ) + ) + + return ( + book_id + is not None + ) + + if getattr( + args, + "write_asin", + False, + ): + + book_id = ( + find_book_id_by_calibre_uuid( + metadata_db, + uuid, + ) + ) + + if book_id is None: + + logger.warning( + "Book not found for UUID: %s", + uuid, + ) + + return False + + return ( + set_asin_in_calibre_library( + metadata_db, + book_id, + asin, + ) + ) + + except ImportError: + + logger.exception( + "Calibre API modules not available" + ) + + except Exception: + + logger.exception( + "Failed writing ASIN" + ) + + return False +``` + +### Next section + +The final major section should be the CLI layer: + +- `find_ebook_files()` +- `build_argument_parser()` +- `configure_logging()` +- `process_ebook_file()` +- `main()` +- JSON output +- `--hash-only` +- `--recursive` +- `--lookup-asin` +- `--verbose` +- `--write-asin` +- `--write-asin-calibre` +- `if __name__ == "__main__": raise SystemExit(main())` + +*** + +### Section 6: CLI Functions and `main()` + +This is the final section to append to the script. + +```python +# ============================================================ +# File Discovery +# ============================================================ + +def find_ebook_files( + paths, + recursive=False, +): + """ + Discover supported ebook files. + + Args: + paths: + List of filesystem paths. + + recursive: + Recursively search directories. + + Returns: + Sorted list of ebook file paths. + """ + + files = [] + + for item in paths: + + path = Path(item) + + if path.is_file(): + + if ( + path.suffix.lower() + in SUPPORTED_FORMATS + ): + files.append(path) + + continue + + if not path.is_dir(): + continue + + pattern = ( + "**/*" + if recursive + else "*" + ) + + for candidate in path.glob( + pattern + ): + + if ( + candidate.is_file() + and candidate.suffix.lower() + in SUPPORTED_FORMATS + ): + files.append( + candidate + ) + + return sorted(files) + + +# ============================================================ +# CLI +# ============================================================ + +def build_argument_parser(): + """ + Create and configure the CLI parser. + + Returns: + argparse.ArgumentParser + """ + + parser = argparse.ArgumentParser( + description=( + "Extract metadata and stable " + "identifiers from EPUB, MOBI, " + "AZW, and AZW3 files." + ), + formatter_class= + argparse.RawDescriptionHelpFormatter, + epilog=""" +Examples: + + Scan a single file: + + ebook_meta.py dune.epub + + Recursive scan: + + ebook_meta.py ~/Books --recursive + + JSON output: + + ebook_meta.py ~/Books --recursive --json + + Compute hashes only: + + ebook_meta.py ~/Books --hash-only + + Lookup ASIN using ISBN: + + ebook_meta.py dune.epub --lookup-asin + + Use ISBNdb: + + ebook_meta.py dune.epub \ + --lookup-asin \ + --isbndb-api-key KEY + + Write ASIN via SQLite: + + ebook_meta.py dune.epub \ + --metadata-db /library/metadata.db \ + --write-asin + + Write ASIN via Calibre API: + + ebook_meta.py dune.epub \ + --metadata-db /library/metadata.db \ + --write-asin-calibre + +Calibre UUID lookup example: + + SELECT book + FROM identifiers + WHERE type='uuid' + AND val=''; +""", + ) + + parser.add_argument( + "ebooks", + nargs="+", + help=( + "Files or directories " + "to process" + ), + ) + + parser.add_argument( + "--json", + action="store_true", + help="Emit JSON output", + ) + + parser.add_argument( + "--recursive", + action="store_true", + help=( + "Recursively scan " + "directories" + ), + ) + + parser.add_argument( + "--hash-only", + action="store_true", + help=( + "Only compute " + "content hashes" + ), + ) + + parser.add_argument( + "--lookup-asin", + action="store_true", + help=( + "Attempt external " + "ASIN lookup" + ), + ) + + parser.add_argument( + "--interactive-asin", + action="store_true", + help=( + "Prompt for ASIN " + "candidate selection" + ), + ) + + parser.add_argument( + "--isbndb-api-key", + help="ISBNdb API key", + ) + + parser.add_argument( + "--metadata-db", + metavar="PATH", + help=( + "Path to Calibre " + "metadata.db" + ), + ) + + parser.add_argument( + "--write-asin", + action="store_true", + help=( + "Write ASIN using " + "direct SQLite access" + ), + ) + + parser.add_argument( + "--write-asin-calibre", + action="store_true", + help=( + "Write ASIN using " + "Calibre API" + ), + ) + + parser.add_argument( + "--log-file", + help="Optional log file", + ) + + parser.add_argument( + "-v", + "--verbose", + action="count", + default=0, + help=( + "Increase verbosity " + "(-v, -vv)" + ), + ) + + return parser + + +def configure_logging(args): + """ + Configure logging. + + Args: + args: + Parsed CLI arguments. + """ + + if args.verbose >= 2: + level = logging.DEBUG + + elif args.verbose == 1: + level = logging.INFO + + else: + level = logging.WARNING + + root_logger = logging.getLogger() + + root_logger.setLevel(level) + + if args.log_file: + + handler = logging.FileHandler( + args.log_file, + encoding="utf-8", + ) + + handler.setFormatter( + logging.Formatter( + "%(asctime)s " + "%(levelname)s " + "%(message)s" + ) + ) + + root_logger.addHandler( + handler + ) + + +# ============================================================ +# Processing +# ============================================================ + +def process_ebook_file( + ebook_file, + args, +): + """ + Process one ebook file. + + Args: + ebook_file: + Ebook path. + + args: + Parsed CLI arguments. + + Returns: + Result dictionary. + """ + + logger.info( + "Processing %s", + ebook_file, + ) + + if args.hash_only: + + content_hash = ( + get_content_hash( + ebook_file + ) + ) + + return { + "file": + str(ebook_file), + "content_hash": + content_hash, + } + + metadata = ( + read_book_metadata( + ebook_file + ) + ) + + metadata["file"] = str( + ebook_file + ) + + # -------------------------------- + # External ASIN lookup + # -------------------------------- + + if ( + args.lookup_asin + and not metadata.get( + "asin" + ) + ): + + asin = lookup_asin( + metadata, + isbndb_api_key= + args.isbndb_api_key, + interactive= + args.interactive_asin, + ) + + if asin: + + metadata["asin"] = asin + + metadata[ + "asin_source" + ] = "external" + + id_type, id_value = ( + get_best_identifier( + metadata + ) + ) + + metadata[ + "best_identifier_type" + ] = id_type + + metadata[ + "best_identifier_value" + ] = id_value + + # -------------------------------- + # Write-back + # -------------------------------- + + if ( + args.metadata_db + and ( + args.write_asin + or args.write_asin_calibre + ) + ): + + write_discovered_asin( + metadata, + args, + ) + + return metadata + + +# ============================================================ +# Main +# ============================================================ + +def main(): + """ + Command-line entry point. + + Returns: + Process exit status. + """ + + signal.signal( + signal.SIGINT, + handle_sigint, + ) + + parser = ( + build_argument_parser() + ) + + args = parser.parse_args() + + if ( + args.write_asin + or args.write_asin_calibre + ): + + if not args.metadata_db: + + parser.error( + "--metadata-db is " + "required when using " + "--write-asin or " + "--write-asin-calibre" + ) + + try: + + configure_logging( + args + ) + + ebook_files = ( + find_ebook_files( + args.ebooks, + recursive= + args.recursive, + ) + ) + + if not ebook_files: + + parser.error( + "No supported ebook " + "files found." + ) + + results = [] + + for ebook_file in ( + ebook_files + ): + + print() + print("=" * 80) + print( + f"FILE: " + f"{ebook_file}" + ) + print("=" * 80) + + try: + + result = ( + process_ebook_file( + ebook_file, + args, + ) + ) + + results.append( + result + ) + + if ( + not args.json + ): + + print( + json.dumps( + result, + indent=2, + ensure_ascii=False, + ) + ) + + except ( + UnsupportedFormatError + ) as e: + + logger.warning( + "%s", + e, + ) + + except ( + FileNotFoundError + ): + + logger.error( + "File not found: %s", + ebook_file, + ) + + except ( + zipfile.BadZipFile + ): + + logger.error( + "Invalid EPUB: %s", + ebook_file, + ) + + except Exception: + + logger.exception( + "Failed processing " + "%s", + ebook_file, + ) + + if args.json: + + print( + json.dumps( + results, + indent=2, + ensure_ascii=False, + ) + ) + + return 0 + + except KeyboardInterrupt: + + logger.warning( + "Processing terminated " + "by user." + ) + + return 130 + + +if __name__ == "__main__": + raise SystemExit(main()) +``` + +⚠️ Before using the complete assembled script, fix one bug in the previously generated `get_book_identifiers_sqlite()` function. Replace: + +```python +return { + rowrow[1] + for row in cur.fetchall() +} +``` + +with: + +```python +return { + rowrow[1] + for row in cur.fetchall() +} +``` + +After that correction, you will have all six sections needed to assemble the full utility. + + +*** + +## High-Level Architecture + +Your final `ebook_meta.py` is organized into **6 major sections**: + +```text +1. Core Infrastructure +2. EPUB Support +3. MOBI / AZW / AZW3 Support +4. External ASIN Lookup Providers +5. Calibre Integration +6. CLI / Main Program +``` + +--- + +# 1. Core Infrastructure + +Provides the common functionality used by the entire script. + +### Imports + +```python +argparse +hashlib +json +logging +signal +sqlite3 +struct +urllib +xml.etree.ElementTree +zipfile +... +``` + +### Constants + +```python +SUPPORTED_FORMATS +ASIN_REGEX +EXTH_ASIN_RECORDS +EXTH_ISBN_RECORDS +``` + +### Exception Types + +```python +UnsupportedFormatError +``` + +### Logging + +```python +logger +configure_logging() +``` + +### SIGINT Handling + +```python +handle_sigint() +``` + +Immediate Ctrl+C termination: + +```text +Exit code 130 +``` + +### Utility Functions + +```python +normalize_identifier() +find_asin_by_regex() +normalize_text() +compute_content_hash() +decode_exth_value() +validate_metadata_db() +``` + +--- + +# 2. EPUB Support + +Reads metadata directly from EPUB files. + +### Functions + +```python +get_epub_opf_path() +extract_epub_text() +read_epub_metadata() +``` + +### Extracted Fields + +```python +title +author +calibre_uuid +asin +isbn +content_hash +``` + +### Sources + +```text +META-INF/container.xml +OPF package document + tags + +``` + +### Fallbacks + +#### ASIN + +```text +scheme="asin" +scheme="amazon" + +then: + +B0XXXXXXXX regex scan +``` + +#### Content Hash + +```text +Extract XHTML +Strip tags +Normalize text +SHA256 +``` + +--- + +# 3. MOBI / AZW / AZW3 Support + +Reads Kindle formats. + +### Functions + +```python +extract_mobi_text() +get_content_hash() +read_mobi_metadata() +read_book_metadata() +get_best_identifier() +``` + +### EXTH Records + +Reads: + +```python +113 +504 +104 +112 +``` + +for: + +```text +ASIN +ISBN +UUID +``` + +### Fallbacks + +#### EXTH Missing + +```text +Raw binary scan +``` + +#### ASIN + +```text +B0XXXXXXXX regex +``` + +#### Content Hash + +```text +Printable strings extraction +SHA256 +``` + +### Identifier Priority + +```text +1 calibre_uuid +2 asin +3 isbn +4 content_hash +``` + +--- + +# 4. External ASIN Providers + +Optional network-based enrichment. + +### Functions + +```python +lookup_openlibrary() +lookup_google_books() +lookup_isbndb() +build_asin_candidates() +choose_asin() +lookup_asin() +``` + +### Open Library + +```text +ISBN enrichment +``` + +### Google Books + +```text +Search industry identifiers +Attempt ASIN extraction +``` + +### ISBNdb + +```text +Requires API key +``` + +### Interactive Selection + +```bash +--interactive-asin +``` + +User sees: + +```text +[1] B012345678 (Google Books) +[2] B098765432 (ISBNdb) +[3] None of the above +``` + +--- + +# 5. Calibre Integration + +Allows writing discovered ASINs back into a Calibre library. + +--- + +## SQLite Mode + +### Lookup Functions + +```python +find_book_id_by_calibre_uuid() +get_book_identifiers_sqlite() +``` + +### Write Function + +```python +set_asin_in_calibre_library() +``` + +Updates: + +```sql +identifiers +``` + +table directly. + +--- + +## Calibre API Mode + +### Functions + +```python +open_calibre_database() +find_book_id_by_uuid_new_api() +update_asin_via_calibre_api() +``` + +Uses: + +```python +from calibre.db.legacy import LibraryDatabase + +db.new_api +``` + +Workflow: + +```text +uuid + -> +book_id + -> +Metadata object + -> +identifiers["amazon"] + -> +save +``` + +--- + +## Unified Entry Point + +```python +write_discovered_asin() +``` + +Automatically chooses: + +```text +Calibre API +or +SQLite +``` + +based on CLI options. + +--- + +# 6. CLI Layer + +Provides full command-line application. + +--- + +## File Discovery + +```python +find_ebook_files() +``` + +Supports: + +```bash +--recursive +``` + +--- + +## Parser Creation + +```python +build_argument_parser() +``` + +### Supported Options + +```bash +--json +--recursive +--hash-only + +--lookup-asin +--interactive-asin + +--isbndb-api-key + +--metadata-db + +--write-asin +--write-asin-calibre + +--log-file + +-v +-vv +``` + +--- + +## Logging Configuration + +```python +configure_logging() +``` + +Levels: + +```text +default WARNING +-v INFO +-vv DEBUG +``` + +--- + +## Single File Processing + +```python +process_ebook_file() +``` + +Performs: + +```text +metadata extraction +ASIN lookup +best identifier determination +optional write-back +``` + +--- + +## Main Program + +```python +main() +``` + +Workflow: + +```text +Parse arguments + ↓ +Configure logging + ↓ +Find ebook files + ↓ +Process each book + ↓ +Lookup missing ASINs + ↓ +Optionally write ASIN + ↓ +Output JSON or text +``` + +--- + +# Complete Feature Matrix + +| Feature | Included | +|----------|----------| +| EPUB metadata | ✅ | +| MOBI metadata | ✅ | +| AZW metadata | ✅ | +| AZW3 metadata | ✅ | +| Calibre UUID extraction | ✅ | +| ASIN extraction | ✅ | +| ISBN extraction | ✅ | +| Regex ASIN fallback | ✅ | +| Content hash fallback | ✅ | +| JSON output | ✅ | +| Recursive scanning | ✅ | +| Hash-only mode | ✅ | +| Logging | ✅ | +| Verbose mode | ✅ | +| SIGINT handling | ✅ | +| Open Library lookup | ✅ | +| Google Books lookup | ✅ | +| ISBNdb lookup | ✅ | +| Interactive ASIN selection | ✅ | +| SQLite write-back | ✅ | +| Calibre API write-back | ✅ | +| Full docstrings | ✅ | +| Multi-file processing | ✅ | + +The resulting tool is essentially a **cross-format ebook identifier and Calibre enrichment utility** with a primary goal of reliably identifying the same book across EPUB, MOBI, AZW, and AZW3 editions. + + + + + + +To assemble the full script, create a single file named `ebook_meta.py` and paste the sections in the exact order they were generated. + +## Recommended File Layout + +```text +ebook_meta.py +│ +├── Section 1: Core Infrastructure +│ +├── Section 2: EPUB Support +│ +├── Section 3: MOBI / AZW / AZW3 Support +│ +├── Section 4: External ASIN Lookup Providers +│ +├── Section 5: Calibre Integration +│ +└── Section 6: CLI and Main Function +``` + +--- + +## Step 1: Create the File + +```bash +touch ebook_meta.py +``` + +or: + +```bash +nano ebook_meta.py +``` + +--- + +## Step 2: Paste Section 1 + +Paste: + +```python +# Core Infrastructure +``` + +This section contains: + +- imports +- constants +- exceptions +- logging +- SIGINT handling +- utility functions + +Everything else depends on these definitions, so it must come first. + +--- + +## Step 3: Paste Section 2 + +Paste: + +```python +# EPUB Support +``` + +Functions: + +```python +get_epub_opf_path() +extract_epub_text() +read_epub_metadata() +``` + +--- + +## Step 4: Paste Section 3 + +Paste: + +```python +# MOBI / AZW / AZW3 Support +``` + +Functions: + +```python +extract_mobi_text() +get_content_hash() +read_mobi_metadata() +read_book_metadata() +get_best_identifier() +``` + +These depend on utilities and EPUB helpers already being defined. + +--- + +## Step 5: Paste Section 4 + +Paste: + +```python +# External ASIN Lookup Providers +``` + +Functions: + +```python +lookup_openlibrary() +lookup_google_books() +lookup_isbndb() +build_asin_candidates() +choose_asin() +lookup_asin() +``` + +These depend on: + +```python +find_asin_by_regex() +logger +``` + +from Section 1. + +--- + +## Step 6: Paste Section 5 + +Paste: + +```python +# Calibre Integration +``` + +Functions: + +```python +find_book_id_by_calibre_uuid() +get_book_identifiers_sqlite() +set_asin_in_calibre_library() +open_calibre_database() +find_book_id_by_uuid_new_api() +update_asin_via_calibre_api() +write_discovered_asin() +``` + +These depend on: + +```python +sqlite3 +Path +logger +validate_metadata_db() +``` + +from previous sections. + +--- + +## Step 7: Fix the Typo + +In: + +```python +get_book_identifiers_sqlite() +``` + +replace the broken comprehension with: + +```python +return { + rowrow[1] + for row in cur.fetchall() +} +``` + +The earlier draft contained a typo. + +--- + +## Step 8: Paste Section 6 + +Paste: + +```python +# CLI and Main Function +``` + +Functions: + +```python +find_ebook_files() +build_argument_parser() +configure_logging() +process_ebook_file() +main() +``` + +and finally: + +```python +if __name__ == "__main__": + raise SystemExit(main()) +``` + +This must always be the last code in the file. + +--- + +## Step 9: Syntax Check + +Before running: + +```bash +python -m py_compile ebook_meta.py +``` + +If successful: + +```bash +echo $? +``` + +should return: + +```text +0 +``` + +--- + +## Step 10: Run Some Basic Tests + +Single EPUB: + +```bash +python ebook_meta.py book.epub +``` + +Recursive scan: + +```bash +python ebook_meta.py ~/Books --recursive +``` + +JSON mode: + +```bash +python ebook_meta.py ~/Books --recursive --json +``` + +Hash-only mode: + +```bash +python ebook_meta.py ~/Books --hash-only +``` + +Verbose mode: + +```bash +python ebook_meta.py ~/Books -vv +``` + +ASIN lookup: + +```bash +python ebook_meta.py book.epub --lookup-asin +``` + +--- + +## Optional Calibre API Support + +If you intend to use: + +```bash +--write-asin-calibre +``` + +you must run the script in an environment that can import: + +```python +from calibre.db.legacy import LibraryDatabase +``` + +A quick verification is: + +```bash +python -c "from calibre.db.legacy import LibraryDatabase; print('OK')" +``` + +If that import fails, the SQLite write-back mode (`--write-asin`) can still work. + +--- diff --git a/src/epub_tts/logger.py b/src/epub_tts/logger.py new file mode 100644 index 0000000..eaed47f --- /dev/null +++ b/src/epub_tts/logger.py @@ -0,0 +1,149 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +# Copyright (c) 2025-2026 Renato Xavier da Silveira Rosa +# See [LICENSE](./LICENSE) or [BSD-3-Clause-Clear](https://spdx.org/licenses/BSD-3-Clause-Clear.html) +# ============================================================ +# Logging +# ============================================================ +from pathlib import Path +import enum +import logging +import os +import re +import sys +import pprint + +# Globals +format_str = "[%(name)s][%(filename)s:%(lineno)04d][%(relativeCreated)s]::%(levelname).4s: (%(funcName)s) %(message)s" +# %(name)s Name of the logger (logging channel) +# %(levelno)s Numeric logging level for the message (DEBUG, INFO, +# WARNING, ERROR, CRITICAL) +# %(levelname)s Text logging level for the message ("DEBUG", "INFO", +# "WARNING", "ERROR", "CRITICAL") +# %(pathname)s Full pathname of the source file where the logging +# call was issued (if available) +# %(filename)s Filename portion of pathname +# %(module)s Module (name portion of filename) +# %(lineno)d Source line number where the logging call was issued +# (if available) +# %(funcName)s Function name +# %(created)f Time when the LogRecord was created (time.time_ns() / 1e9 +# return value) +# %(asctime)s Textual time when the LogRecord was created +# %(msecs)d Millisecond portion of the creation time +# %(relativeCreated)d Time in milliseconds when the LogRecord was created, +# relative to the time the logging module was loaded +# (typically at application startup time) +# %(thread)d Thread ID (if available) +# %(threadName)s Thread name (if available) +# %(taskName)s Task name (if available) +# %(process)d Process ID (if available) +# %(processName)s Process name (if available) +# %(message)s The result of record.getMessage(), computed just as +# the record is emitted +formatter = logging.Formatter(format_str) +logging.basicConfig( + # 50=CRITICAL/FATAL, 40=ERROR, 30=WARN/WARNING, 20=INFO, 10=DEBUG, 0=NOTSET + # loglevel_map = logging.getLevelNamesMapping() + level=logging.INFO, + format=format_str, +) + +logger = logging.getLogger(__name__) + +loglevel_map = logging.getLevelNamesMapping() + + +# Logging utils +def print(message): + pp = pprint.PrettyPrinter( + indent=4, + width=os.get_terminal_size().columns, + compact=False, + depth=None + ) + if isinstance(message, (dict, list, tuple, set)): + formatted = pp.pformat(message) + print(formatted.replace("', '", "',\n'")) + else: + pp.pprint(message) + + +def get_level( + cmdline: str|list[str] = None, + verbose: bool = False, + debug: bool = False, + quiet: bool = False, + default: int = logging.INFO, + ) -> int: + # Find lowest priority loglevel + if "--debug" in cmdline: + debug = True + if "--verbose" in cmdline: + verbose = True + if "--quiet" in cmdline: + quiet = True + + if debug: + return logging.DEBUG + + for i, a in enumerate(cmdline): + if a == "--loglevel": + loglevel = loglevel_map.get(cmdline[i+1], None) + elif a.startswith("--loglevel="): + loglevel = loglevel_map.get(a.replace("--loglevel=", ""), None) + + # lower = more priority + if loglevel_map[loglevel] > loglevel_map["INFO"]: + config.logger.debug("self.loglevel='%s'(%d) > INFO", config.loglevel, loglevel_map[config.loglevel]) + config.loglevel = "INFO" + +def get_logger( + verbose:bool = False, + debug:bool = False, + quiet: bool = False, + loglevel: str|int = logging.INFO, + logfile: str|Path = None, + ): + """Initialize the package logger from CLI flags and defaults. + + Resolve the effective logging level from the class args (or sys.argv). + --debug (or debug_level=True) takes priority, then --verbose (or verbose=True) + is compared to --loglevel (or loglevel=) and the lower one is proritized. + """ + + + + config.logger.info("Setting loglevel to %s",config.loglevel) + config.logger.setLevel(loglevel_map[config.loglevel]) + + # Get logfile + for i, a in enumerate(config._args_list): + if a == "--logfile": + config.logfile = config._args_list[i+1] + elif a.startswith("--logfile="): + config.logfile = a.replace("--logfile=", "") + + if "logfile" in config and config.logfile: + if logfile: + config.logger.warning("Overriding --logfile='%s' from setup_logger(logfile='%s')", config.logfile, logfile) + config.logfile = logfile + fh = logging.FileHandler(logfile) + fh.setLevel(config.loglevel) + fh.setFormatter(logger.formatter) + config.logger.addHandler(fh) + + # if self.debug_level: + # logger.info("Loglevel: DEBUG") + # if self.loglevel: + # logger.warning("Ignoring '--loglevel' flag because DEBUG/'--debug' was also set.") + # if "--verbose" in self._args_list: + # logger.warning("Ignoring '--verbose' flag because '--debug' was also set.") + # self.verbose = False + # if self.loglevel and self.verbose: + # self.warning("Conflict: trying to --loglevel=%s (%d) and --verbose.", self.loglevel, loglevel_map[self.loglevel]) + # if loglevel_map[self.loglevel] < logging.INFO: + # self.logger.info("Setting loglevel to %s (%d), which is more verbose than INFO.", self.loglevel, loglevel_map[self.loglevel]) + + + return config.logger \ No newline at end of file diff --git a/src/epub_tts/postprocess.py b/src/epub_tts/postprocess.py index 9c8cb64..efe10ba 100644 --- a/src/epub_tts/postprocess.py +++ b/src/epub_tts/postprocess.py @@ -25,7 +25,7 @@ from pydub import AudioSegment from mutagen import mp4 # Local imports from .tts_generic import GenericTTSBackend -from . import log, PathLike +from . import logger, PathLike from .preprocess import preprocess_book def generate_metadata(files, author, title, chapter_titles): diff --git a/src/epub_tts/preprocess.py b/src/epub_tts/preprocess.py index 43a3e72..4e3ca1e 100644 --- a/src/epub_tts/preprocess.py +++ b/src/epub_tts/preprocess.py @@ -28,7 +28,7 @@ from nltk.tokenize import sent_tokenize # Local imports from .tts_generic import GenericTTSBackend -from . import log, PathLike +from . import logger, PathLike diff --git a/src/epub_tts/tts_aedocw.py b/src/epub_tts/tts_aedocw.py index 6c61f33..9bb19af 100644 --- a/src/epub_tts/tts_aedocw.py +++ b/src/epub_tts/tts_aedocw.py @@ -14,7 +14,7 @@ corresponding console script is not available in the current environment. """ from .tts_generic import GenericTTSBackend -from . import log, PathLike +from . import logger, PathLike class AedocwBackend(GenericTTSBackend): """Generic aedocw backend wrapper. diff --git a/src/epub_tts/tts_generic.py b/src/epub_tts/tts_generic.py index 9fd5f16..9b3a8c5 100644 --- a/src/epub_tts/tts_generic.py +++ b/src/epub_tts/tts_generic.py @@ -35,7 +35,7 @@ from pathlib import Path from typing import Optional, Callable from .utils import build_run_command, ensure_venv -from . import log, PathLike +from . import logger, PathLike class GenericTTSBackend: """Generic TTS backend for handling text-to-speech conversion. @@ -67,7 +67,7 @@ class GenericTTSBackend: ENV = None # subclasses may override this to set environment variables for the backend def __init__(self, **kwargs): - log(kwargs) + logger.info(kwargs) for k,v in kwargs.items(): setattr(self, k, v) if hasattr(self, 'CWD') and self.CWD is not None: @@ -77,7 +77,7 @@ class GenericTTSBackend: if not self.CWD.is_dir(): self.CWD = self.CWD.parent.resolve() self.ENV = os.environ.copy() - #log(f"SELF:{self.__dict__}") + #logger.info(f"SELF:{self.__dict__}") def _normalize_path(self, value: PathLike) -> Path: @@ -123,7 +123,7 @@ class GenericTTSBackend: ) -> subprocess.CompletedProcess: command = self._build_command(input_source, output_dest) - log(command) + logger.info(command) # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first if not input_source.endswith(".txt"): diff --git a/src/epub_tts/tts_kokoro.py b/src/epub_tts/tts_kokoro.py index 8c421df..309cdf4 100644 --- a/src/epub_tts/tts_kokoro.py +++ b/src/epub_tts/tts_kokoro.py @@ -30,7 +30,7 @@ from pydub import AudioSegment from mutagen import mp4 # Local imports from .tts_generic import GenericTTSBackend -from . import log, PathLike +from . import logger, PathLike from .preprocess import preprocess_book class KokoroBackend(GenericTTSBackend): @@ -58,7 +58,7 @@ class KokoroBackend(GenericTTSBackend): ) -> subprocess.CompletedProcess: command = self._build_command(input_source, output_dest) - log(command) + logger.info(command) # # PREPROCESSING STEP: If INTERMEDIATE_TXT is True, run the command to generate intermediate text first # if not input_source.endswith(".txt"): diff --git a/src/epub_tts/utils.py b/src/epub_tts/utils.py index 75158a2..e35c176 100644 --- a/src/epub_tts/utils.py +++ b/src/epub_tts/utils.py @@ -10,8 +10,109 @@ import shutil import sys import venv from pathlib import Path -from typing import Optional -from . import log, PathLike +from typing import Optional, Union +from collections.abc import MutableMapping + +from . import logger, PathLike + + +def path(fpath: str|Path = None): + return Path(fpath).resolve() + +class DataDict(dict): + """Wrapper around dictionary objects for easier dict subclassing + See module [collections](https://github.com/python/cpython/blob/3.14/Lib/collections/__init__.py#L1133) + """ + + def __init__(self, *args, **kwargs): + self._data = {} + if args is not None: + if isinstance(args, list): + for a in args: + if not isinstance(a,tuple) or len(a) != 2: + # list is not a list of (k,v) tuples + self._args = args + break + # all list items are (k,v) tuples + self.update(args) + self._args = args + if kwargs: + self.update(kwargs) + + # Emulate accessing class attributes + def __getattr__(self, key): + # This gets called only if __getattribute__ doesn't find key + return self._data[key] + + def __setattr__(self, key, value): + if key.startswith("_"): + return super().__setattr__ (key, value) + return self._data.__setitem__(key, value) + + def __delattr__(self, key): + if key.startswith("_"): + return super().__delattr__(key) + return self._data.__delitem__(key) + + + # Emulate dict methods + + def __len__(self): + return len(self._data) + + def __getitem__(self, key): + if key in self._data: + return self._data[key] + if hasattr(self.__class__, "__missing__"): + return self.__class__.__missing__(self, key) + raise KeyError(key) + + def __setitem__(self, key, value): + self._data[key] = value + + def __delitem__(self, key): + del self._data[key] + + def __iter__(self): + return iter(self._data) + + def __contains__(self, key): + return key in self._data + + def get(self, key, default=None): + if key in self: + return self[key] + return default + + def __repr__(self): + return repr(self._data) + + def __or__(self, other): + return self._data | other + + def __ror__(self, other): + return other | self._data + + def __ior__(self, other): + self._data |= other + return self + + def __copy__(self): + # TODO: verify if this is correct + return dict(self._data) + + def copy(self): + import copy + return copy.copy(self._data) + + + @classmethod + def fromkeys(cls, iterable, value=None): + d = cls() + for key in iterable: + d[key] = value + return d + def get_vendor_root() -> Path: return Path(__file__).resolve().parents[1] / "vendor" @@ -72,21 +173,3 @@ def build_run_command( # else str(venv_path / "bin" / backend_cmd) # ) - - -def check_env(self, which=None): - if not which: - return None - if isinstance(which, str): - if " " in which: - which = which.split() - which = [which] - for bin in which: - p = shutil.which(bin) - if not p: - log(f"{bin} is either NOT installed or " - "NOT in your system PATH.") - return False - else: - log(f"{bin} found at: {p}") - return True \ No newline at end of file