Compare commits

..
2 Commits
Author SHA1 Message Date
oqyude 509fd3dde0 kokoro-tts 2026-10-03 01:03:50 +03:00
oqyude 958247b22c soft coding 2026-10-02 23:05:00 +03:00
20 changed files with 781 additions and 31 deletions
+2 -2
View File
@@ -27,10 +27,10 @@ let
# (default is bashInteractive)
user.shell = "${pkgs.zsh}/bin/zsh";
# SSH user (matches `User oqyude` in the client's ~/.ssh/config).
# SSH user (matches the `User` entries in the client's ~/.ssh/config).
# Default is "nix-on-droid"; home stays at the read-only
# /data/data/com.termux.nix/files/home either way.
user.userName = "oqyude";
user.userName = xlib.device.username;
# Minimal termux settings (nix-on-droid options only:
# environment.*, nix.*, time.*, user.*, system.*, android-integration.*)
+1 -1
View File
@@ -14,7 +14,7 @@ let
"${config.home.homeDirectory}/Games/PrismLaunchers/${config.home.username}" =
".local/share/PrismLauncher";
"${xlib.dirs.lamet-drive}/Users/oqyude/Music" = "Music";
"${xlib.dirs.lamet-drive}/Users/${xlib.device.username}/Music" = "Music";
};
mkLinks = lib.mapAttrs' (sourcePath: targetPath: {
name = targetPath;
+8 -8
View File
@@ -218,7 +218,7 @@
enable = true;
settings = {
user = {
name = "oqyude";
name = xlib.device.username;
email = "oqyude@gmail.com";
};
pull = {
@@ -250,31 +250,31 @@
};
sapphira = {
HostName = "192.168.1.20";
User = "oqyude";
User = xlib.device.username;
};
sapphira-tailscale = {
HostName = "100.64.0.0";
User = "oqyude";
User = xlib.device.username;
};
otreca-old = {
HostName = "217.60.3.12";
User = "oqyude";
User = xlib.device.username;
};
otreca = {
HostName = "109.248.161.5";
User = "oqyude";
User = xlib.device.username;
};
otreca-tailscale = {
HostName = "100.64.1.0";
User = "oqyude";
User = xlib.device.username;
};
rydiwo = {
HostName = "192.168.1.102";
User = "oqyude";
User = xlib.device.username;
};
epral = {
HostName = "192.168.1.101";
User = "oqyude";
User = xlib.device.username;
Port = 8022;
};
};
+13 -1
View File
@@ -40,6 +40,8 @@ in
hostname,
type,
username ? "oqyude",
uid ? 1000,
gid ? 1000,
}:
let
device = mkDevice {
@@ -47,6 +49,8 @@ in
hostname
type
username
uid
gid
;
};
in
@@ -56,11 +60,19 @@ in
hostname
type
username
uid
gid
;
};
isDesktop = device.isDesktop;
isHeadless = device.isHeadless;
dirs = mkDirs username;
inherit helpers;
# Bind the host's ids into the mount helpers, so ntfs3/exfat options
# carry the same uid/gid the primary user actually has.
helpers = import ./helpers.nix {
inherit lib;
uid = device.uid;
gid = device.gid;
};
};
}
+11 -1
View File
@@ -49,15 +49,25 @@ in
hostname,
type,
username ? "oqyude",
# The primary user is pinned to 1000 rather than left to NixOS'
# nextfree logic: the mount helpers below write uid=/gid= into
# ntfs3/exfat options, and an NTFS/exFAT volume mounted with a
# different id shows every file as owned by `nobody`.
uid ? 1000,
gid ? 1000,
}:
let
capabilities = devices.${type} or (throw "xlib: unknown device type '${type}', expected one of ${lib.concatStringsSep ", " (builtins.attrNames devices)}");
capabilities =
devices.${type}
or (throw "xlib: unknown device type '${type}', expected one of ${lib.concatStringsSep ", " (builtins.attrNames devices)}");
in
{
inherit
hostname
type
username
uid
gid
;
isDesktop = capabilities.desktop;
isHeadless = capabilities.headless;
+9 -4
View File
@@ -1,5 +1,10 @@
{
lib,
# The primary user's ids, bound from xlib.device by mkXlib. ntfs3/exfat
# volumes carry POSIX ids, so a mount using anything other than the real
# uid/gid shows every file as owned by `nobody`.
uid,
gid,
...
}:
# Pure helper functions for module definitions.
@@ -105,8 +110,8 @@ let
fsType = "ntfs3";
options = [
"defaults"
"uid=1000"
"gid=1000"
"uid=${toString uid}"
"gid=${toString gid}"
"fmask=${mask}"
"dmask=${mask}"
"nofail"
@@ -128,8 +133,8 @@ let
fsType = "exfat";
options = [
"nofail"
"uid=1000"
"gid=1000"
"uid=${toString uid}"
"gid=${toString gid}"
];
};
};
+116
View File
@@ -0,0 +1,116 @@
{
lib,
pkgs,
...
}:
let
# The image is built here rather than pulled: zaakirio/kokoro-ru is a
# Hugging Face repo, not a published OCI image, and its Russian G2P has to be
# driven through the repo's own ru_g2p.py.
#
# The build context goes through the store so the image is pinned to the
# config revision: edit a file, `nixos-rebuild`, and the unit below rebuilds
# and restarts. Reading the context off a checkout at runtime would leave the
# running container untraceable back to any config.
source = pkgs.linkFarm "kokoro-tts-source" [
{
name = "Dockerfile";
path = toString ./kokoro-tts/Dockerfile;
}
{
name = "app.py";
path = toString ./kokoro-tts/app.py;
}
{
name = "fetch_assets.py";
path = toString ./kokoro-tts/fetch_assets.py;
}
{
name = "requirements.txt";
path = toString ./kokoro-tts/requirements.txt;
}
];
image = "localhost/kokoro-tts:latest";
# Unchanged from the silero module, so whatever already points at
# http://127.0.0.1:9898/v1 keeps working without edits.
hostPort = 9898;
containerPort = 8000;
in
{
config = {
virtualisation = {
podman = {
enable = true;
autoPrune = {
enable = true;
flags = [ "--all" ];
};
dockerCompat = true;
};
oci-containers = {
backend = "podman";
containers.kokoro-tts = {
image = image;
ports = [
"127.0.0.1:${toString hostPort}:${toString containerPort}"
];
environment = {
# Inference is CPU-bound and already threaded inside torch; these
# keep it from oversubscribing a small machine.
KOKORO_THREADS = "4";
OMP_NUM_THREADS = "4";
MKL_NUM_THREADS = "4";
TZ = "Europe/Moscow";
};
# No volumes: the checkpoints, the acute-aware espeak data and
# ruaccent's ONNX models are all baked into the image, so the
# container needs neither a host directory nor the network to start.
log-driver = "journald";
};
};
};
systemd = {
services = {
# Runs before the container. BuildKit caches the expensive layers, so
# on every boot after the first this is a no-op that still verifies the
# image exists.
"podman-build-kokoro-tts" = {
path = [ pkgs.podman ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
# First build pulls torch wheels plus ~700 MB of weights.
TimeoutSec = 3600;
};
script = ''
podman build -t ${image} ${source}
'';
wantedBy = [ "multi-user.target" ];
};
"podman-kokoro-tts" = {
# The image does not exist until the build above ran, and a `latest`
# tag must be re-pulled on rebuild, so ordering has to be explicit.
after = [ "podman-build-kokoro-tts.service" ];
requires = [ "podman-build-kokoro-tts.service" ];
serviceConfig.Restart = lib.mkOverride 90 "always";
wantedBy = [ "multi-user.target" ];
};
};
};
};
}
+56
View File
@@ -0,0 +1,56 @@
FROM python:3.12-slim-bookworm
# Pinned, not "main": a rebuild that only touched the Nix module must not
# silently pick up different weights. Bump these deliberately.
ARG KOKORO_RU_REPO=zaakirio/kokoro-ru
ARG KOKORO_RU_REVISION=d649c57b239b18c4c384378127cbf01dba039bc1
# Trim to "sveta" to halve the image: masha shares her checkpoint and dima is
# a second 327 MB one.
ARG KOKORO_RU_VOICES=sveta,masha,dima
ENV PYTHONUNBUFFERED=1 \
PIP_NO_CACHE_DIR=1 \
PIP_DISABLE_PIP_VERSION_CHECK=1 \
HF_HUB_DISABLE_TELEMETRY=1 \
HF_HUB_DISABLE_SYMLINKS_WARNING=1 \
KOKORO_RU_REPO=${KOKORO_RU_REPO} \
KOKORO_RU_REVISION=${KOKORO_RU_REVISION} \
KOKORO_RU_VOICES=${KOKORO_RU_VOICES} \
KOKORO_MODEL_DIR=/app/kokoro-ru \
KOKORO_THREADS=4 \
OMP_NUM_THREADS=4 \
MKL_NUM_THREADS=4 \
TZ=Europe/Moscow
WORKDIR /app
# libgomp1 is torch's OpenMP runtime. espeak-ng comes from the espeakng-loader
# wheel rather than the distro package because the model needs its own
# recompiled ru_dict, and libsndfile is absent because WAV/PCM are written with
# stdlib `wave` while every other format goes through imageio-ffmpeg.
RUN apt-get update \
&& apt-get install -y --no-install-recommends libgomp1 \
&& rm -rf /var/lib/apt/lists/*
# CPU-only torch from its own index: the default PyPI wheel drags in ~2.5 GB of
# CUDA libraries for a machine that has no GPU.
RUN pip install --index-url https://download.pytorch.org/whl/cpu torch
COPY requirements.txt ./
RUN pip install -r requirements.txt
COPY app.py fetch_assets.py ./
# Bakes the checkpoints, the acute-aware espeak data and ruaccent's ONNX models
# into the layer, which is what lets the container start with no network and no
# writable volume.
RUN python fetch_assets.py
EXPOSE 8000
HEALTHCHECK --interval=30s --timeout=5s --start-period=180s --retries=3 \
CMD ["python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/healthz', timeout=4)"]
# No workers: the model is a shared in-process singleton, so a second worker
# would only mean a second copy of ~2 GB of weights.
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]
+430
View File
@@ -0,0 +1,430 @@
"""OpenAI-compatible TTS API backed by zaakirio/kokoro-ru.
The model itself is language-blind: phoneme ids in, 24 kHz audio out. All the
Russian lives in the G2P front-end, and the one that matters is kokoro-ru's own
`ru_g2p.py` — RUAccent resolves lexical stress, ё and homographs, then an
acute-aware espeak-ng phonemizer turns that into IPA. Stock misaki Russian is
espeak-only and gets stress wrong often enough that the model reads as
non-native (measured 27% vs 22% round-trip WER, per the model card).
So: text -> RuG2P.phonemize -> KModel(ipa, voicepack[len(ipa) - 1]) -> waveform.
Endpoints
POST /v1/audio/speech OpenAI text-to-speech
GET /v1/models OpenAI model list
GET /v1/voices voice inventory (extension, not part of OpenAI)
GET /healthz readiness, 503 until the model is loaded
"""
from __future__ import annotations
import io
import logging
import os
import re
import subprocess
import sys
import threading
import wave
from contextlib import asynccontextmanager
from pathlib import Path
from typing import TYPE_CHECKING, Literal
import numpy as np
from fastapi import FastAPI
from fastapi.responses import JSONResponse, Response
from pydantic import BaseModel, ConfigDict, Field
if TYPE_CHECKING: # torch is imported lazily so /healthz answers during boot
import torch
MODEL_ID = "kokoro-ru"
SAMPLE_RATE = 24000
MODEL_DIR = Path(os.environ.get("KOKORO_MODEL_DIR", "/app/kokoro-ru"))
DEFAULT_VOICE = os.environ.get("KOKORO_DEFAULT_VOICE", "sveta")
THREADS = int(os.environ.get("KOKORO_THREADS", os.cpu_count() or 4))
# 2026-07-29, when the kokoro-ru revision we pin was published. Clients that
# cache on this treat any change as a new model, so it must stay stable.
MODEL_CREATED = 1785353253
# voice -> (checkpoint stem, gender). The checkpoint carries the timbre and the
# voicepack the identity, which is why sveta and masha share one file.
VOICE_SPECS: dict[str, tuple[str, str]] = {
"sveta": ("kokoro-ru-v2-base", "female"),
"masha": ("kokoro-ru-v2-base", "female"),
"dima": ("kokoro-ru-v2-dima", "male"),
}
# Clients that ship the OpenAI voice list (alloy, nova, echo, ...) send those
# names unless the user overrides them, so map them onto the three we have.
VOICE_ALIASES: dict[str, str] = {
"alloy": "sveta",
"ash": "sveta",
"ballad": "sveta",
"verse": "sveta",
"marin": "sveta",
"coral": "masha",
"sage": "masha",
"shimmer": "masha",
"cedar": "masha",
"echo": "dima",
"fable": "dima",
"onyx": "dima",
}
CONTENT_TYPES = {
"wav": "audio/wav",
"mp3": "audio/mpeg",
"opus": "audio/ogg",
"aac": "audio/aac",
"flac": "audio/flac",
"pcm": "audio/pcm",
}
# Everything except wav and pcm goes through ffmpeg; those two are byte-exact
# from the stdlib and need no encoder at all.
FFMPEG_ARGS = {
"mp3": ["-c:a", "libmp3lame", "-q:a", "2"],
"opus": ["-c:a", "libopus", "-b:a", "64k"],
"aac": ["-c:a", "aac", "-b:a", "128k"],
"flac": ["-c:a", "flac"],
}
FFMPEG_CONTAINERS = {"mp3": "mp3", "opus": "ogg", "aac": "adts", "flac": "flac"}
# Kokoro's Albert context is 510 tokens and KModel.forward asserts
# len(ids) + 2 <= 510, so 508 phonemes is the hard ceiling per forward pass.
MAX_PHONEMES = 508
# Roughly 300 characters of Russian lands near 400 phonemes, comfortably under
# the ceiling, and keeps a chunk short enough that a bad sentence is a short
# chunk.
CHUNK_CHARS = 300
# Silence inserted between chunks. Without it the concatenation clicks at every
# boundary because each forward pass starts and ends on a zero crossing.
CHUNK_GAP_S = 0.08
_SENTENCE_SPLIT = re.compile(r"(?<=[.!?…])\s+")
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
log = logging.getLogger("kokoro-ru")
def split_text(text: str, budget: int = CHUNK_CHARS) -> list[str]:
"""Split into sentence-bounded chunks, hard-cutting only as a last resort.
Phonemizing per sentence rather than per paragraph keeps RUAccent's stress
decisions local and gives the model a reset point at every full stop.
"""
chunks: list[str] = []
current = ""
for sentence in _SENTENCE_SPLIT.split(text.strip()):
sentence = sentence.strip()
while len(sentence) > budget:
if current:
chunks.append(current)
current = ""
chunks.append(sentence[:budget])
sentence = sentence[budget:].strip()
if not sentence:
continue
if len(current) + len(sentence) + 1 > budget:
# Guarded: when the first sentence fills the budget exactly, or the
# previous one was hard-cut down to nothing, `current` is empty and
# a bare append would queue a zero-length chunk.
if current:
chunks.append(current)
current = sentence
else:
current = f"{current} {sentence}".strip()
if current:
chunks.append(current)
return chunks
def split_phonemes(ps: str, limit: int = MAX_PHONEMES) -> list[str]:
"""Cut an over-long phoneme string on word boundaries."""
if len(ps) <= limit:
return [ps]
parts: list[str] = []
rest = ps
while len(rest) > limit:
cut = rest.rfind(" ", 0, limit)
if cut <= 0:
cut = limit
parts.append(rest[:cut].strip())
rest = rest[cut:].strip()
if rest:
parts.append(rest)
return [part for part in parts if part]
class KokoroRu:
"""Loaded model plus the G2P front-end, behind a single inference lock."""
def __init__(self) -> None:
self._torch: torch | None = None
self._g2p = None
self._models: dict[str, torch.nn.Module] = {}
self._packs: dict[str, torch.Tensor] = {}
# The Albert encoder and the iSTFTNet decoder keep per-call scratch
# buffers; concurrent forwards on one model interleave into them. The
# model is fast enough on CPU that serialising is not the bottleneck.
self._lock = threading.Lock()
def load(self) -> None:
import torch
from kokoro import KModel
torch.set_num_threads(THREADS)
self._torch = torch
# RuG2P is imported from the baked snapshot, not installed, and it
# resolves espeak-data/ plus kokoro-config.json next to itself.
sys.path.insert(0, str(MODEL_DIR))
from ru_g2p import RuG2P
self._g2p = RuG2P(
espeak_data=MODEL_DIR / "espeak-data",
vocab_path=MODEL_DIR / "kokoro-config.json",
)
for stem in sorted({stem for stem, _ in VOICE_SPECS.values()}):
checkpoint = MODEL_DIR / f"{stem}.pth"
if not checkpoint.exists():
log.warning("checkpoint %s missing, voices using it stay unavailable", checkpoint)
continue
# repo_id is only used to build the default model filename; passing
# both config and model keeps it from touching the HF cache at all.
self._models[stem] = KModel(
repo_id=str(MODEL_DIR),
config=str(MODEL_DIR / "config.json"),
model=str(checkpoint),
).eval()
log.info("loaded checkpoint %s", checkpoint.name)
for name in VOICE_SPECS:
pack = MODEL_DIR / "voices" / f"{name}.pt"
if pack.exists():
self._packs[name] = torch.load(str(pack), map_location="cpu", weights_only=True)
if not self.available_voices():
raise RuntimeError(f"no usable voices under {MODEL_DIR}")
def available_voices(self) -> list[str]:
return [
name
for name in VOICE_SPECS
if name in self._packs and VOICE_SPECS[name][0] in self._models
]
def phonemes(self, text: str):
for chunk in split_text(text):
ps, _oov = self._g2p.phonemize(chunk)
ps = ps.strip()
if ps:
yield from split_phonemes(ps)
def synthesize(self, text: str, voice: str, speed: float) -> np.ndarray:
torch = self._torch
assert torch is not None, "synthesize() before load()"
stem, _gender = VOICE_SPECS[voice]
model = self._models[stem]
pack = self._packs[voice]
gap = torch.zeros(int(CHUNK_GAP_S * SAMPLE_RATE), dtype=torch.float32)
pieces: list[torch.Tensor] = []
with self._lock:
for ps in self.phonemes(text):
# The style vector is picked by phoneme-string length, which is
# why the model sounds deterministic for identical text.
style = pack[len(ps) - 1]
# The packs ship as [510, 256]; KModel wants a batch of one.
if style.dim() == 1:
style = style.unsqueeze(0)
if pieces:
pieces.append(gap)
pieces.append(model(ps, style, speed, return_output=True).audio)
if not pieces:
return np.zeros(0, dtype=np.float32)
return torch.cat(pieces).numpy().astype(np.float32, copy=False)
def encode(audio: np.ndarray, fmt: str) -> bytes:
clipped = np.clip(audio, -1.0, 1.0)
if fmt == "pcm":
# OpenAI's pcm is raw signed 16-bit little-endian mono at 24 kHz.
return (clipped * 32767.0).astype("<i2").tobytes()
buffer = io.BytesIO()
with wave.open(buffer, "wb") as out:
out.setnchannels(1)
out.setsampwidth(2)
out.setframerate(SAMPLE_RATE)
out.writeframes((clipped * 32767.0).astype("<i2").tobytes())
wav = buffer.getvalue()
if fmt == "wav":
return wav
import imageio_ffmpeg
command = [
imageio_ffmpeg.get_ffmpeg_exe(),
"-hide_banner",
"-loglevel",
"error",
"-i",
"pipe:0",
"-ar",
str(SAMPLE_RATE),
"-ac",
"1",
*FFMPEG_ARGS[fmt],
"-f",
FFMPEG_CONTAINERS[fmt],
"pipe:1",
]
done = subprocess.run(command, input=wav, capture_output=True, check=False)
if done.returncode != 0:
raise RuntimeError(done.stderr.decode("utf-8", "replace").strip()[-400:])
return done.stdout
engine = KokoroRu()
state: dict[str, str | None] = {"status": "loading", "error": None}
def boot() -> None:
try:
engine.load()
state["status"] = "ready"
log.info("ready: voices=%s", ", ".join(engine.available_voices()))
except Exception as exc:
state["status"] = "error"
state["error"] = f"{type(exc).__name__}: {exc}"
log.exception("model failed to load")
@asynccontextmanager
async def lifespan(_app: FastAPI):
# Off the event loop: loading pulls ~700 MB of weights and runs three ONNX
# sessions, and /healthz has to stay answerable while it happens.
threading.Thread(target=boot, name="kokoro-load", daemon=True).start()
yield
app = FastAPI(title="kokoro-ru OpenAI TTS", version="1.0.0", lifespan=lifespan)
Format = Literal["mp3", "opus", "aac", "flac", "wav", "pcm"]
class SpeechRequest(BaseModel):
# `protected_namespaces` silences pydantic's warning about the `model_`
# prefix; `extra="ignore"` absorbs the fields newer OpenAI clients add
# (instructions, the legacy `format` alias) without failing the request.
model_config = ConfigDict(extra="ignore", protected_namespaces=())
input: str = Field(min_length=1)
model: str = MODEL_ID
voice: str | None = None
response_format: Format = "wav"
speed: float | None = Field(default=None, ge=0.25, le=4.0)
def fail(status: int, message: str, param: str | None = None, code: str | None = None) -> JSONResponse:
return JSONResponse(
status_code=status,
content={
"error": {
"message": message,
"type": "invalid_request_error" if status < 500 else "server_error",
"param": param,
"code": code,
}
},
)
def resolve_voice(requested: str | None) -> str | None:
name = (requested or DEFAULT_VOICE).strip().lower()
name = VOICE_ALIASES.get(name, name)
return name if name in engine.available_voices() else None
# response_model=None: the handler returns a Response subclass directly, and
# FastAPI would otherwise try to build a Pydantic model out of the union.
@app.post("/v1/audio/speech", response_model=None)
def create_speech(request: SpeechRequest) -> Response | JSONResponse:
if state["status"] != "ready":
return fail(503, f"model is not ready: {state['status']}", code="model_not_ready")
voice = resolve_voice(request.voice)
if voice is None:
available = ", ".join(engine.available_voices())
return fail(
400,
f"unknown voice {request.voice!r}; available: {available}",
param="voice",
code="unknown_voice",
)
try:
audio = engine.synthesize(request.input, voice, request.speed or 1.0)
except Exception as exc:
log.exception("synthesis failed")
return fail(500, f"synthesis failed: {exc}", code="synthesis_failed")
if audio.size == 0:
return fail(
400,
"input contains no speakable text for the Russian G2P",
param="input",
code="no_phonemes",
)
try:
payload = encode(audio, request.response_format)
except Exception as exc:
log.exception("encoding to %s failed", request.response_format)
return fail(500, f"encoding to {request.response_format} failed: {exc}", code="encoding_failed")
return Response(
content=payload,
media_type=CONTENT_TYPES[request.response_format],
headers={"model-id": MODEL_ID, "voice-id": voice},
)
@app.get("/v1/models")
def list_models() -> dict:
return {
"object": "list",
"data": [{"id": MODEL_ID, "object": "model", "created": MODEL_CREATED, "owned_by": "zaakirio"}],
}
@app.get("/v1/voices")
def list_voices() -> dict:
return {
"object": "list",
"ready": state["status"] == "ready",
"data": [
{"id": name, "object": "voice", "checkpoint": VOICE_SPECS[name][0], "gender": VOICE_SPECS[name][1]}
for name in engine.available_voices()
],
}
@app.get("/healthz")
def healthz() -> JSONResponse:
ready = state["status"] == "ready"
return JSONResponse(
status_code=200 if ready else 503,
content={
"status": state["status"],
"model": MODEL_ID,
"voices": engine.available_voices(),
"sample_rate": SAMPLE_RATE,
"error": state["error"],
},
)
@@ -0,0 +1,82 @@
"""Bake every kokoro-ru asset the server needs into the image.
Two things make a plain `FROM python` image useless for this model at runtime,
and both are fixed here at build time:
* kokoro-ru's checkpoints and its recompiled espeak-ng data live in the HF
cache by default, and the HF cache is part of the disposable container
layer, so every `podman run` would re-download ~700 MB.
* ruaccent writes its ONNX models, dictionaries and Koziev data into its own
`site-packages/ruaccent` directory. It only downloads when those files are
missing, so a single `load()` here means the runtime never touches the
network.
RuG2P resolves espeak-data/ and kokoro-config.json relative to ru_g2p.py, so
the snapshot layout has to stay flat inside KOKORO_MODEL_DIR.
"""
from __future__ import annotations
import logging
import os
from pathlib import Path
from huggingface_hub import snapshot_download
REPO = os.environ.get("KOKORO_RU_REPO", "zaakirio/kokoro-ru")
# A commit, not a branch: "main" would silently change the weights under a
# rebuild that only touched an unrelated line of the Nix module.
REVISION = os.environ.get("KOKORO_RU_REVISION", "main")
DEST = Path(os.environ.get("KOKORO_MODEL_DIR", "/app/kokoro-ru"))
VOICES = [v.strip() for v in os.environ.get("KOKORO_RU_VOICES", "sveta,masha,dima").split(",") if v.strip()]
# sveta and masha share one checkpoint and differ only by voicepack, so the two
# female voices cost one 327 MB download, not two.
CHECKPOINTS = {
"sveta": "kokoro-ru-v2-base.pth",
"masha": "kokoro-ru-v2-base.pth",
"dima": "kokoro-ru-v2-dima.pth",
}
PATTERNS = [
# KModel reads config.json; RuG2P reads kokoro-config.json for the phoneme
# vocab. They are not the same file and both are required.
"config.json",
"kokoro-config.json",
"ru_g2p.py",
# Stock espeak-ng ru_dict ignores combining-acute stress marks, which is the
# one thing this whole front-end exists to fix. The model repo ships a
# recompiled dictsource; there is no substitute to fall back to.
"espeak-data/**",
*(CHECKPOINTS[v] for v in VOICES if v in CHECKPOINTS),
*(f"voices/{v}.pt" for v in VOICES),
]
def main() -> None:
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s")
DEST.mkdir(parents=True, exist_ok=True)
snapshot_download(
repo_id=REPO,
revision=REVISION,
allow_patterns=PATTERNS,
local_dir=str(DEST),
)
logging.info("kokoro-ru assets in %s at %s", DEST, REVISION)
missing = [name for name in VOICES if not (DEST / "voices" / f"{name}.pt").exists()]
if missing:
raise SystemExit(f"voice packs missing after download: {missing}")
# Warm ruaccent into site-packages so `load()` short-circuits at runtime.
from ruaccent import RUAccent
accent = RUAccent()
accent.load(omograph_model_size="turbo3.1", use_dictionary=True, tiny_mode=False)
logging.info("ruaccent warm: %s", accent.process_all("Здравствуйте, как ваши дела?"))
if __name__ == "__main__":
main()
@@ -0,0 +1,21 @@
# torch is installed separately in the Dockerfile from the CPU-only index;
# do not add it here or pip would pull the ~2.5 GB CUDA build over it.
#
# ru_g2p.py (from zaakirio/kokoro-ru) imports three things stock kokoro does not
# pull on its own: the espeak-ng backend of misaki, ruaccent for stress, and the
# phonemizer fork whose EspeakWrapper misaki drives.
kokoro==0.9.4
misaki[en]>=0.9.4
phonemizer-fork
espeakng-loader
ruaccent
# kokoro's Albert encoder and ruaccent's ONNX exports both go through
# transformers. ru_g2p.py shims token_type_ids for v5, so the floor is what
# matters, not the ceiling.
transformers>=4.46
fastapi
uvicorn
imageio-ffmpeg
numpy>=1.26,<3
+2 -1
View File
@@ -2,6 +2,7 @@
config,
pkgs,
inputs,
xlib,
...
}:
{
@@ -185,7 +186,7 @@
enable = true;
config = {
user = {
name = "oqyude";
name = xlib.device.username;
email = "oqyude@gmail.com";
};
pull = {
+4 -3
View File
@@ -1,6 +1,7 @@
{
config,
pkgs,
xlib,
...
}:
{
@@ -20,7 +21,7 @@
};
shellInit = ''
beet-p() {
local base="/home/oqyude/.config/beets/My"
local base="${xlib.dirs.user-home}/.config/beets/My"
local rel
rel=$(realpath --relative-to="$base" "$PWD")
beet mod "path:$rel" playlist="$*"
@@ -29,7 +30,7 @@
beet im ./ -S $*
}
beet-path() {
realpath --relative-to="/home/oqyude/.config/beets/My" "$1"
realpath --relative-to="${xlib.dirs.user-home}/.config/beets/My" "$1"
}
'';
shellAliases = {
@@ -45,7 +46,7 @@
gc = "git add . && git commit -m 'dev: автокоммит $(date +'%Y-%m-%d %H:%M:%S')'";
y = "yazi";
nix-shellp = "nix-shell --run $SHELL -p";
beet-path-library = "realpath --relative-to='/home/oqyude/.config/beets/My' .";
beet-path-library = "realpath --relative-to='${xlib.dirs.user-home}/.config/beets/My' .";
z-proxy = "export ALL_PROXY=socks5://localhost:10808";
zh-proxy = "export HTTPS_PROXY=http://localhost:10808 && export HTTP_PROXY=http://localhost:10808";
+2 -2
View File
@@ -11,13 +11,13 @@
description = "Prebuild NixOS closure";
serviceConfig = {
CPUQuota = "20%";
User = "oqyude";
User = xlib.device.username;
Group = "users";
Nice = 10;
Type = "oneshot";
WorkingDirectory = "/tmp";
Environment = [
"HOME=/home/oqyude"
"HOME=${xlib.dirs.user-home}"
];
ExecStart = ''
${pkgs.nix}/bin/nix build --no-link /etc/nixos#nixosConfigurations.${config.networking.hostName}.config.system.build.toplevel
+1 -1
View File
@@ -64,7 +64,7 @@
dbtype = "pgsql";
dbuser = "nextcloud";
dbname = "nextcloud";
adminuser = "oqyude";
adminuser = xlib.device.username;
adminpassFile = config.sops.secrets.nextcloud-adminpass.path;
};
settings = {
+16
View File
@@ -44,6 +44,22 @@ in
name = "${user}";
isNormalUser = true;
group = "users";
# Pinned, not left to NixOS' nextfree logic: the ntfs3/exfat mount
# helpers (lib/xlib/helpers.nix) bake xlib.device.uid into their mount
# options, so normally both sides agree and NTFS/exFAT files do not
# show up as owned by `nobody`. NixOS has no per-user `gid` option —
# the primary group id comes from `group` above.
#
# sapphira is the one exception, and only until its filesystem gets
# migrated: /var/lib/nixos/uid-map still reserves 1000 for the
# long-removed `yuyus` and NixOS never renumbers an existing user, so
# the live `oqyude` there is uid 1001. Without this branch a rebuild
# would rewrite the user to 1000 while every file is still owned by
# 1001. The cost: the exFAT mounts on sapphira still get uid=1000 from
# xlib.device.uid, so that user cannot write to /mnt/archive or
# /mnt/mobile until the id question is settled.
# TODO: delete this branch once sapphira is migrated to 1000.
uid = if xlib.device.hostname == "sapphira" then 1001 else xlib.device.uid;
description = "Jor Oqyude";
hashedPasswordFile = config.sops.secrets.hashed_password.path; # hashed_password
homeMode = "700";
+1 -1
View File
@@ -7,7 +7,7 @@
}:
let
serviceName = "rsync-services-sync";
serverAddress = "oqyude@100.64.0.0";
serverAddress = "${xlib.device.username}@100.64.0.0";
serverDir = "${xlib.dirs.services-nodes-folder}/${xlib.device.hostname}";
nodeDir = "${xlib.dirs.services-mnt-folder}";
in
+1 -1
View File
@@ -7,7 +7,7 @@
{
imports = [
# shared container modules live in ../../containers
# ../../containers/3x-ui.nix
../../containers/kokoro-tts.nix
];
environment.systemPackages = with pkgs; [