From 509fd3dde0dbf15cf40059988829e228a4f8aad3 Mon Sep 17 00:00:00 2001 From: oqyude Date: Fri, 2 Oct 2026 23:05:07 +0300 Subject: [PATCH] kokoro-tts --- lib/xlib/default.nix | 2 +- lib/xlib/device.nix | 6 +- lib/xlib/dirs.nix | 2 +- lib/xlib/helpers.nix | 2 +- modules/containers/kokoro-tts.nix | 116 +++++ modules/containers/kokoro-tts/Dockerfile | 56 +++ modules/containers/kokoro-tts/app.py | 430 ++++++++++++++++++ modules/containers/kokoro-tts/fetch_assets.py | 82 ++++ .../containers/kokoro-tts/requirements.txt | 21 + modules/options.nix | 2 +- modules/users.nix | 20 +- modules/wsl/containers/default.nix | 2 +- 12 files changed, 729 insertions(+), 12 deletions(-) create mode 100644 modules/containers/kokoro-tts.nix create mode 100644 modules/containers/kokoro-tts/Dockerfile create mode 100644 modules/containers/kokoro-tts/app.py create mode 100644 modules/containers/kokoro-tts/fetch_assets.py create mode 100644 modules/containers/kokoro-tts/requirements.txt diff --git a/lib/xlib/default.nix b/lib/xlib/default.nix index 2ffce36..4583993 100644 --- a/lib/xlib/default.nix +++ b/lib/xlib/default.nix @@ -75,4 +75,4 @@ in gid = device.gid; }; }; -} \ No newline at end of file +} diff --git a/lib/xlib/device.nix b/lib/xlib/device.nix index 0a1ef02..40521c0 100644 --- a/lib/xlib/device.nix +++ b/lib/xlib/device.nix @@ -57,7 +57,9 @@ in gid ? 1000, }: let - capabilities = devices.${type} or (throw "xlib: unknown device type '${type}', expected one of ${lib.concatStringsSep ", " (builtins.attrNames devices)}"); + capabilities = + devices.${type} + or (throw "xlib: unknown device type '${type}', expected one of ${lib.concatStringsSep ", " (builtins.attrNames devices)}"); in { inherit @@ -70,4 +72,4 @@ in isDesktop = capabilities.desktop; isHeadless = capabilities.headless; }; -} \ No newline at end of file +} diff --git a/lib/xlib/dirs.nix b/lib/xlib/dirs.nix index f5894b0..a2fd046 100644 --- a/lib/xlib/dirs.nix +++ b/lib/xlib/dirs.nix @@ -31,4 +31,4 @@ in therima-drive = "/mnt/therima"; vetymae-drive = "/mnt/vetymae"; soptur-drive = "/mnt/soptur"; -} \ No newline at end of file +} diff --git a/lib/xlib/helpers.nix b/lib/xlib/helpers.nix index 14859f6..1bc93d9 100644 --- a/lib/xlib/helpers.nix +++ b/lib/xlib/helpers.nix @@ -158,4 +158,4 @@ in mkExfatMount mkSymlinks ; -} \ No newline at end of file +} diff --git a/modules/containers/kokoro-tts.nix b/modules/containers/kokoro-tts.nix new file mode 100644 index 0000000..55c9f7d --- /dev/null +++ b/modules/containers/kokoro-tts.nix @@ -0,0 +1,116 @@ +{ + lib, + pkgs, + ... +}: + +let + # The image is built here rather than pulled: zaakirio/kokoro-ru is a + # Hugging Face repo, not a published OCI image, and its Russian G2P has to be + # driven through the repo's own ru_g2p.py. + # + # The build context goes through the store so the image is pinned to the + # config revision: edit a file, `nixos-rebuild`, and the unit below rebuilds + # and restarts. Reading the context off a checkout at runtime would leave the + # running container untraceable back to any config. + source = pkgs.linkFarm "kokoro-tts-source" [ + { + name = "Dockerfile"; + path = toString ./kokoro-tts/Dockerfile; + } + { + name = "app.py"; + path = toString ./kokoro-tts/app.py; + } + { + name = "fetch_assets.py"; + path = toString ./kokoro-tts/fetch_assets.py; + } + { + name = "requirements.txt"; + path = toString ./kokoro-tts/requirements.txt; + } + ]; + + image = "localhost/kokoro-tts:latest"; + + # Unchanged from the silero module, so whatever already points at + # http://127.0.0.1:9898/v1 keeps working without edits. + hostPort = 9898; + containerPort = 8000; +in +{ + config = { + virtualisation = { + podman = { + enable = true; + + autoPrune = { + enable = true; + flags = [ "--all" ]; + }; + + dockerCompat = true; + }; + + oci-containers = { + backend = "podman"; + + containers.kokoro-tts = { + image = image; + + ports = [ + "127.0.0.1:${toString hostPort}:${toString containerPort}" + ]; + + environment = { + # Inference is CPU-bound and already threaded inside torch; these + # keep it from oversubscribing a small machine. + KOKORO_THREADS = "4"; + OMP_NUM_THREADS = "4"; + MKL_NUM_THREADS = "4"; + TZ = "Europe/Moscow"; + }; + + # No volumes: the checkpoints, the acute-aware espeak data and + # ruaccent's ONNX models are all baked into the image, so the + # container needs neither a host directory nor the network to start. + log-driver = "journald"; + }; + }; + }; + + systemd = { + services = { + # Runs before the container. BuildKit caches the expensive layers, so + # on every boot after the first this is a no-op that still verifies the + # image exists. + "podman-build-kokoro-tts" = { + path = [ pkgs.podman ]; + + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + # First build pulls torch wheels plus ~700 MB of weights. + TimeoutSec = 3600; + }; + + script = '' + podman build -t ${image} ${source} + ''; + + wantedBy = [ "multi-user.target" ]; + }; + + "podman-kokoro-tts" = { + # The image does not exist until the build above ran, and a `latest` + # tag must be re-pulled on rebuild, so ordering has to be explicit. + after = [ "podman-build-kokoro-tts.service" ]; + requires = [ "podman-build-kokoro-tts.service" ]; + serviceConfig.Restart = lib.mkOverride 90 "always"; + wantedBy = [ "multi-user.target" ]; + }; + }; + }; + }; +} \ No newline at end of file diff --git a/modules/containers/kokoro-tts/Dockerfile b/modules/containers/kokoro-tts/Dockerfile new file mode 100644 index 0000000..71f79dc --- /dev/null +++ b/modules/containers/kokoro-tts/Dockerfile @@ -0,0 +1,56 @@ +FROM python:3.12-slim-bookworm + +# Pinned, not "main": a rebuild that only touched the Nix module must not +# silently pick up different weights. Bump these deliberately. +ARG KOKORO_RU_REPO=zaakirio/kokoro-ru +ARG KOKORO_RU_REVISION=d649c57b239b18c4c384378127cbf01dba039bc1 +# Trim to "sveta" to halve the image: masha shares her checkpoint and dima is +# a second 327 MB one. +ARG KOKORO_RU_VOICES=sveta,masha,dima + +ENV PYTHONUNBUFFERED=1 \ + PIP_NO_CACHE_DIR=1 \ + PIP_DISABLE_PIP_VERSION_CHECK=1 \ + HF_HUB_DISABLE_TELEMETRY=1 \ + HF_HUB_DISABLE_SYMLINKS_WARNING=1 \ + KOKORO_RU_REPO=${KOKORO_RU_REPO} \ + KOKORO_RU_REVISION=${KOKORO_RU_REVISION} \ + KOKORO_RU_VOICES=${KOKORO_RU_VOICES} \ + KOKORO_MODEL_DIR=/app/kokoro-ru \ + KOKORO_THREADS=4 \ + OMP_NUM_THREADS=4 \ + MKL_NUM_THREADS=4 \ + TZ=Europe/Moscow + +WORKDIR /app + +# libgomp1 is torch's OpenMP runtime. espeak-ng comes from the espeakng-loader +# wheel rather than the distro package because the model needs its own +# recompiled ru_dict, and libsndfile is absent because WAV/PCM are written with +# stdlib `wave` while every other format goes through imageio-ffmpeg. +RUN apt-get update \ + && apt-get install -y --no-install-recommends libgomp1 \ + && rm -rf /var/lib/apt/lists/* + +# CPU-only torch from its own index: the default PyPI wheel drags in ~2.5 GB of +# CUDA libraries for a machine that has no GPU. +RUN pip install --index-url https://download.pytorch.org/whl/cpu torch + +COPY requirements.txt ./ +RUN pip install -r requirements.txt + +COPY app.py fetch_assets.py ./ + +# Bakes the checkpoints, the acute-aware espeak data and ruaccent's ONNX models +# into the layer, which is what lets the container start with no network and no +# writable volume. +RUN python fetch_assets.py + +EXPOSE 8000 + +HEALTHCHECK --interval=30s --timeout=5s --start-period=180s --retries=3 \ + CMD ["python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/healthz', timeout=4)"] + +# No workers: the model is a shared in-process singleton, so a second worker +# would only mean a second copy of ~2 GB of weights. +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"] \ No newline at end of file diff --git a/modules/containers/kokoro-tts/app.py b/modules/containers/kokoro-tts/app.py new file mode 100644 index 0000000..6064b54 --- /dev/null +++ b/modules/containers/kokoro-tts/app.py @@ -0,0 +1,430 @@ +"""OpenAI-compatible TTS API backed by zaakirio/kokoro-ru. + +The model itself is language-blind: phoneme ids in, 24 kHz audio out. All the +Russian lives in the G2P front-end, and the one that matters is kokoro-ru's own +`ru_g2p.py` — RUAccent resolves lexical stress, ё and homographs, then an +acute-aware espeak-ng phonemizer turns that into IPA. Stock misaki Russian is +espeak-only and gets stress wrong often enough that the model reads as +non-native (measured 27% vs 22% round-trip WER, per the model card). + +So: text -> RuG2P.phonemize -> KModel(ipa, voicepack[len(ipa) - 1]) -> waveform. + +Endpoints + POST /v1/audio/speech OpenAI text-to-speech + GET /v1/models OpenAI model list + GET /v1/voices voice inventory (extension, not part of OpenAI) + GET /healthz readiness, 503 until the model is loaded +""" + +from __future__ import annotations + +import io +import logging +import os +import re +import subprocess +import sys +import threading +import wave +from contextlib import asynccontextmanager +from pathlib import Path +from typing import TYPE_CHECKING, Literal + +import numpy as np +from fastapi import FastAPI +from fastapi.responses import JSONResponse, Response +from pydantic import BaseModel, ConfigDict, Field + +if TYPE_CHECKING: # torch is imported lazily so /healthz answers during boot + import torch + +MODEL_ID = "kokoro-ru" +SAMPLE_RATE = 24000 +MODEL_DIR = Path(os.environ.get("KOKORO_MODEL_DIR", "/app/kokoro-ru")) +DEFAULT_VOICE = os.environ.get("KOKORO_DEFAULT_VOICE", "sveta") +THREADS = int(os.environ.get("KOKORO_THREADS", os.cpu_count() or 4)) +# 2026-07-29, when the kokoro-ru revision we pin was published. Clients that +# cache on this treat any change as a new model, so it must stay stable. +MODEL_CREATED = 1785353253 + +# voice -> (checkpoint stem, gender). The checkpoint carries the timbre and the +# voicepack the identity, which is why sveta and masha share one file. +VOICE_SPECS: dict[str, tuple[str, str]] = { + "sveta": ("kokoro-ru-v2-base", "female"), + "masha": ("kokoro-ru-v2-base", "female"), + "dima": ("kokoro-ru-v2-dima", "male"), +} + +# Clients that ship the OpenAI voice list (alloy, nova, echo, ...) send those +# names unless the user overrides them, so map them onto the three we have. +VOICE_ALIASES: dict[str, str] = { + "alloy": "sveta", + "ash": "sveta", + "ballad": "sveta", + "verse": "sveta", + "marin": "sveta", + "coral": "masha", + "sage": "masha", + "shimmer": "masha", + "cedar": "masha", + "echo": "dima", + "fable": "dima", + "onyx": "dima", +} + +CONTENT_TYPES = { + "wav": "audio/wav", + "mp3": "audio/mpeg", + "opus": "audio/ogg", + "aac": "audio/aac", + "flac": "audio/flac", + "pcm": "audio/pcm", +} + +# Everything except wav and pcm goes through ffmpeg; those two are byte-exact +# from the stdlib and need no encoder at all. +FFMPEG_ARGS = { + "mp3": ["-c:a", "libmp3lame", "-q:a", "2"], + "opus": ["-c:a", "libopus", "-b:a", "64k"], + "aac": ["-c:a", "aac", "-b:a", "128k"], + "flac": ["-c:a", "flac"], +} +FFMPEG_CONTAINERS = {"mp3": "mp3", "opus": "ogg", "aac": "adts", "flac": "flac"} + +# Kokoro's Albert context is 510 tokens and KModel.forward asserts +# len(ids) + 2 <= 510, so 508 phonemes is the hard ceiling per forward pass. +MAX_PHONEMES = 508 +# Roughly 300 characters of Russian lands near 400 phonemes, comfortably under +# the ceiling, and keeps a chunk short enough that a bad sentence is a short +# chunk. +CHUNK_CHARS = 300 +# Silence inserted between chunks. Without it the concatenation clicks at every +# boundary because each forward pass starts and ends on a zero crossing. +CHUNK_GAP_S = 0.08 + +_SENTENCE_SPLIT = re.compile(r"(?<=[.!?…])\s+") + +logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") +log = logging.getLogger("kokoro-ru") + + +def split_text(text: str, budget: int = CHUNK_CHARS) -> list[str]: + """Split into sentence-bounded chunks, hard-cutting only as a last resort. + + Phonemizing per sentence rather than per paragraph keeps RUAccent's stress + decisions local and gives the model a reset point at every full stop. + """ + chunks: list[str] = [] + current = "" + for sentence in _SENTENCE_SPLIT.split(text.strip()): + sentence = sentence.strip() + while len(sentence) > budget: + if current: + chunks.append(current) + current = "" + chunks.append(sentence[:budget]) + sentence = sentence[budget:].strip() + if not sentence: + continue + if len(current) + len(sentence) + 1 > budget: + # Guarded: when the first sentence fills the budget exactly, or the + # previous one was hard-cut down to nothing, `current` is empty and + # a bare append would queue a zero-length chunk. + if current: + chunks.append(current) + current = sentence + else: + current = f"{current} {sentence}".strip() + if current: + chunks.append(current) + return chunks + + +def split_phonemes(ps: str, limit: int = MAX_PHONEMES) -> list[str]: + """Cut an over-long phoneme string on word boundaries.""" + if len(ps) <= limit: + return [ps] + parts: list[str] = [] + rest = ps + while len(rest) > limit: + cut = rest.rfind(" ", 0, limit) + if cut <= 0: + cut = limit + parts.append(rest[:cut].strip()) + rest = rest[cut:].strip() + if rest: + parts.append(rest) + return [part for part in parts if part] + + +class KokoroRu: + """Loaded model plus the G2P front-end, behind a single inference lock.""" + + def __init__(self) -> None: + self._torch: torch | None = None + self._g2p = None + self._models: dict[str, torch.nn.Module] = {} + self._packs: dict[str, torch.Tensor] = {} + # The Albert encoder and the iSTFTNet decoder keep per-call scratch + # buffers; concurrent forwards on one model interleave into them. The + # model is fast enough on CPU that serialising is not the bottleneck. + self._lock = threading.Lock() + + def load(self) -> None: + import torch + from kokoro import KModel + + torch.set_num_threads(THREADS) + self._torch = torch + + # RuG2P is imported from the baked snapshot, not installed, and it + # resolves espeak-data/ plus kokoro-config.json next to itself. + sys.path.insert(0, str(MODEL_DIR)) + from ru_g2p import RuG2P + + self._g2p = RuG2P( + espeak_data=MODEL_DIR / "espeak-data", + vocab_path=MODEL_DIR / "kokoro-config.json", + ) + + for stem in sorted({stem for stem, _ in VOICE_SPECS.values()}): + checkpoint = MODEL_DIR / f"{stem}.pth" + if not checkpoint.exists(): + log.warning("checkpoint %s missing, voices using it stay unavailable", checkpoint) + continue + # repo_id is only used to build the default model filename; passing + # both config and model keeps it from touching the HF cache at all. + self._models[stem] = KModel( + repo_id=str(MODEL_DIR), + config=str(MODEL_DIR / "config.json"), + model=str(checkpoint), + ).eval() + log.info("loaded checkpoint %s", checkpoint.name) + + for name in VOICE_SPECS: + pack = MODEL_DIR / "voices" / f"{name}.pt" + if pack.exists(): + self._packs[name] = torch.load(str(pack), map_location="cpu", weights_only=True) + + if not self.available_voices(): + raise RuntimeError(f"no usable voices under {MODEL_DIR}") + + def available_voices(self) -> list[str]: + return [ + name + for name in VOICE_SPECS + if name in self._packs and VOICE_SPECS[name][0] in self._models + ] + + def phonemes(self, text: str): + for chunk in split_text(text): + ps, _oov = self._g2p.phonemize(chunk) + ps = ps.strip() + if ps: + yield from split_phonemes(ps) + + def synthesize(self, text: str, voice: str, speed: float) -> np.ndarray: + torch = self._torch + assert torch is not None, "synthesize() before load()" + stem, _gender = VOICE_SPECS[voice] + model = self._models[stem] + pack = self._packs[voice] + + gap = torch.zeros(int(CHUNK_GAP_S * SAMPLE_RATE), dtype=torch.float32) + pieces: list[torch.Tensor] = [] + with self._lock: + for ps in self.phonemes(text): + # The style vector is picked by phoneme-string length, which is + # why the model sounds deterministic for identical text. + style = pack[len(ps) - 1] + # The packs ship as [510, 256]; KModel wants a batch of one. + if style.dim() == 1: + style = style.unsqueeze(0) + if pieces: + pieces.append(gap) + pieces.append(model(ps, style, speed, return_output=True).audio) + + if not pieces: + return np.zeros(0, dtype=np.float32) + return torch.cat(pieces).numpy().astype(np.float32, copy=False) + + +def encode(audio: np.ndarray, fmt: str) -> bytes: + clipped = np.clip(audio, -1.0, 1.0) + if fmt == "pcm": + # OpenAI's pcm is raw signed 16-bit little-endian mono at 24 kHz. + return (clipped * 32767.0).astype(" None: + try: + engine.load() + state["status"] = "ready" + log.info("ready: voices=%s", ", ".join(engine.available_voices())) + except Exception as exc: + state["status"] = "error" + state["error"] = f"{type(exc).__name__}: {exc}" + log.exception("model failed to load") + + +@asynccontextmanager +async def lifespan(_app: FastAPI): + # Off the event loop: loading pulls ~700 MB of weights and runs three ONNX + # sessions, and /healthz has to stay answerable while it happens. + threading.Thread(target=boot, name="kokoro-load", daemon=True).start() + yield + + +app = FastAPI(title="kokoro-ru OpenAI TTS", version="1.0.0", lifespan=lifespan) + +Format = Literal["mp3", "opus", "aac", "flac", "wav", "pcm"] + + +class SpeechRequest(BaseModel): + # `protected_namespaces` silences pydantic's warning about the `model_` + # prefix; `extra="ignore"` absorbs the fields newer OpenAI clients add + # (instructions, the legacy `format` alias) without failing the request. + model_config = ConfigDict(extra="ignore", protected_namespaces=()) + + input: str = Field(min_length=1) + model: str = MODEL_ID + voice: str | None = None + response_format: Format = "wav" + speed: float | None = Field(default=None, ge=0.25, le=4.0) + + +def fail(status: int, message: str, param: str | None = None, code: str | None = None) -> JSONResponse: + return JSONResponse( + status_code=status, + content={ + "error": { + "message": message, + "type": "invalid_request_error" if status < 500 else "server_error", + "param": param, + "code": code, + } + }, + ) + + +def resolve_voice(requested: str | None) -> str | None: + name = (requested or DEFAULT_VOICE).strip().lower() + name = VOICE_ALIASES.get(name, name) + return name if name in engine.available_voices() else None + + +# response_model=None: the handler returns a Response subclass directly, and +# FastAPI would otherwise try to build a Pydantic model out of the union. +@app.post("/v1/audio/speech", response_model=None) +def create_speech(request: SpeechRequest) -> Response | JSONResponse: + if state["status"] != "ready": + return fail(503, f"model is not ready: {state['status']}", code="model_not_ready") + + voice = resolve_voice(request.voice) + if voice is None: + available = ", ".join(engine.available_voices()) + return fail( + 400, + f"unknown voice {request.voice!r}; available: {available}", + param="voice", + code="unknown_voice", + ) + + try: + audio = engine.synthesize(request.input, voice, request.speed or 1.0) + except Exception as exc: + log.exception("synthesis failed") + return fail(500, f"synthesis failed: {exc}", code="synthesis_failed") + + if audio.size == 0: + return fail( + 400, + "input contains no speakable text for the Russian G2P", + param="input", + code="no_phonemes", + ) + + try: + payload = encode(audio, request.response_format) + except Exception as exc: + log.exception("encoding to %s failed", request.response_format) + return fail(500, f"encoding to {request.response_format} failed: {exc}", code="encoding_failed") + + return Response( + content=payload, + media_type=CONTENT_TYPES[request.response_format], + headers={"model-id": MODEL_ID, "voice-id": voice}, + ) + + +@app.get("/v1/models") +def list_models() -> dict: + return { + "object": "list", + "data": [{"id": MODEL_ID, "object": "model", "created": MODEL_CREATED, "owned_by": "zaakirio"}], + } + + +@app.get("/v1/voices") +def list_voices() -> dict: + return { + "object": "list", + "ready": state["status"] == "ready", + "data": [ + {"id": name, "object": "voice", "checkpoint": VOICE_SPECS[name][0], "gender": VOICE_SPECS[name][1]} + for name in engine.available_voices() + ], + } + + +@app.get("/healthz") +def healthz() -> JSONResponse: + ready = state["status"] == "ready" + return JSONResponse( + status_code=200 if ready else 503, + content={ + "status": state["status"], + "model": MODEL_ID, + "voices": engine.available_voices(), + "sample_rate": SAMPLE_RATE, + "error": state["error"], + }, + ) \ No newline at end of file diff --git a/modules/containers/kokoro-tts/fetch_assets.py b/modules/containers/kokoro-tts/fetch_assets.py new file mode 100644 index 0000000..e4ff628 --- /dev/null +++ b/modules/containers/kokoro-tts/fetch_assets.py @@ -0,0 +1,82 @@ +"""Bake every kokoro-ru asset the server needs into the image. + +Two things make a plain `FROM python` image useless for this model at runtime, +and both are fixed here at build time: + + * kokoro-ru's checkpoints and its recompiled espeak-ng data live in the HF + cache by default, and the HF cache is part of the disposable container + layer, so every `podman run` would re-download ~700 MB. + * ruaccent writes its ONNX models, dictionaries and Koziev data into its own + `site-packages/ruaccent` directory. It only downloads when those files are + missing, so a single `load()` here means the runtime never touches the + network. + +RuG2P resolves espeak-data/ and kokoro-config.json relative to ru_g2p.py, so +the snapshot layout has to stay flat inside KOKORO_MODEL_DIR. +""" + +from __future__ import annotations + +import logging +import os +from pathlib import Path + +from huggingface_hub import snapshot_download + +REPO = os.environ.get("KOKORO_RU_REPO", "zaakirio/kokoro-ru") +# A commit, not a branch: "main" would silently change the weights under a +# rebuild that only touched an unrelated line of the Nix module. +REVISION = os.environ.get("KOKORO_RU_REVISION", "main") +DEST = Path(os.environ.get("KOKORO_MODEL_DIR", "/app/kokoro-ru")) + +VOICES = [v.strip() for v in os.environ.get("KOKORO_RU_VOICES", "sveta,masha,dima").split(",") if v.strip()] + +# sveta and masha share one checkpoint and differ only by voicepack, so the two +# female voices cost one 327 MB download, not two. +CHECKPOINTS = { + "sveta": "kokoro-ru-v2-base.pth", + "masha": "kokoro-ru-v2-base.pth", + "dima": "kokoro-ru-v2-dima.pth", +} + +PATTERNS = [ + # KModel reads config.json; RuG2P reads kokoro-config.json for the phoneme + # vocab. They are not the same file and both are required. + "config.json", + "kokoro-config.json", + "ru_g2p.py", + # Stock espeak-ng ru_dict ignores combining-acute stress marks, which is the + # one thing this whole front-end exists to fix. The model repo ships a + # recompiled dictsource; there is no substitute to fall back to. + "espeak-data/**", + *(CHECKPOINTS[v] for v in VOICES if v in CHECKPOINTS), + *(f"voices/{v}.pt" for v in VOICES), +] + + +def main() -> None: + logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s") + + DEST.mkdir(parents=True, exist_ok=True) + snapshot_download( + repo_id=REPO, + revision=REVISION, + allow_patterns=PATTERNS, + local_dir=str(DEST), + ) + logging.info("kokoro-ru assets in %s at %s", DEST, REVISION) + + missing = [name for name in VOICES if not (DEST / "voices" / f"{name}.pt").exists()] + if missing: + raise SystemExit(f"voice packs missing after download: {missing}") + + # Warm ruaccent into site-packages so `load()` short-circuits at runtime. + from ruaccent import RUAccent + + accent = RUAccent() + accent.load(omograph_model_size="turbo3.1", use_dictionary=True, tiny_mode=False) + logging.info("ruaccent warm: %s", accent.process_all("Здравствуйте, как ваши дела?")) + + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/modules/containers/kokoro-tts/requirements.txt b/modules/containers/kokoro-tts/requirements.txt new file mode 100644 index 0000000..a05d447 --- /dev/null +++ b/modules/containers/kokoro-tts/requirements.txt @@ -0,0 +1,21 @@ +# torch is installed separately in the Dockerfile from the CPU-only index; +# do not add it here or pip would pull the ~2.5 GB CUDA build over it. +# +# ru_g2p.py (from zaakirio/kokoro-ru) imports three things stock kokoro does not +# pull on its own: the espeak-ng backend of misaki, ruaccent for stress, and the +# phonemizer fork whose EspeakWrapper misaki drives. +kokoro==0.9.4 +misaki[en]>=0.9.4 +phonemizer-fork +espeakng-loader +ruaccent + +# kokoro's Albert encoder and ruaccent's ONNX exports both go through +# transformers. ru_g2p.py shims token_type_ids for v5, so the floor is what +# matters, not the ceiling. +transformers>=4.46 + +fastapi +uvicorn +imageio-ffmpeg +numpy>=1.26,<3 \ No newline at end of file diff --git a/modules/options.nix b/modules/options.nix index 54ef581..96f9985 100644 --- a/modules/options.nix +++ b/modules/options.nix @@ -38,4 +38,4 @@ ''; }; }; -} \ No newline at end of file +} diff --git a/modules/users.nix b/modules/users.nix index 92e9a26..c6e6a3a 100644 --- a/modules/users.nix +++ b/modules/users.nix @@ -45,11 +45,21 @@ in isNormalUser = true; group = "users"; # Pinned, not left to NixOS' nextfree logic: the ntfs3/exfat mount - # helpers (lib/xlib/helpers.nix) write the same uid into their mount - # options, so both sides have to agree or NTFS files show up as owned - # by `nobody`. NixOS has no per-user `gid` option — the primary group - # id comes from `group` above. - uid = xlib.device.uid; + # helpers (lib/xlib/helpers.nix) bake xlib.device.uid into their mount + # options, so normally both sides agree and NTFS/exFAT files do not + # show up as owned by `nobody`. NixOS has no per-user `gid` option — + # the primary group id comes from `group` above. + # + # sapphira is the one exception, and only until its filesystem gets + # migrated: /var/lib/nixos/uid-map still reserves 1000 for the + # long-removed `yuyus` and NixOS never renumbers an existing user, so + # the live `oqyude` there is uid 1001. Without this branch a rebuild + # would rewrite the user to 1000 while every file is still owned by + # 1001. The cost: the exFAT mounts on sapphira still get uid=1000 from + # xlib.device.uid, so that user cannot write to /mnt/archive or + # /mnt/mobile until the id question is settled. + # TODO: delete this branch once sapphira is migrated to 1000. + uid = if xlib.device.hostname == "sapphira" then 1001 else xlib.device.uid; description = "Jor Oqyude"; hashedPasswordFile = config.sops.secrets.hashed_password.path; # hashed_password homeMode = "700"; diff --git a/modules/wsl/containers/default.nix b/modules/wsl/containers/default.nix index 7d6119b..f00729d 100644 --- a/modules/wsl/containers/default.nix +++ b/modules/wsl/containers/default.nix @@ -7,7 +7,7 @@ { imports = [ # shared container modules live in ../../containers - ../../containers/silero-tts.nix + ../../containers/kokoro-tts.nix ]; environment.systemPackages = with pkgs; [