mirror of
https://github.com/oqyude/nixos.git
synced 2026-10-11 14:27:26 +03:00
CRITICAL BUG #2 in storage guard, caught by live test v8 on sapphira (2026-10-10). RequiresMountsFor=/home/ooyude/External in the unit file told systemd that the service depends on the mount. When the mount was gone and the service was started, systemd's dependency resolver AUTOMATICALLY REMOUNTED the filesystem to satisfy the dependency — THEN checked ConditionPathIsMountPoint. The condition saw the just-remounted filesystem and evaluated to true. Service started on empty external storage. Guard completely bypassed. This is a subtle interaction: - ConditionPathIsMountPoint is a TEST (true/false evaluation) - RequiresMountsFor is a DEPENDENCY (systemd must make it true) A guard should be a TEST, not a dependency that makes the test trivially pass. Remove the dependency. The condition alone is sufficient for both boot-time and runtime checks: - Boot: mount unit starts via local-fs.target, condition is true - Runtime: if mount disappears, condition becomes false on next start attempt. Without RequiresMountsFor, systemd doesn't auto-recover, so the guard fires. Ordering should be expressed via After= in the consumer's systemd.services block (not in the shared helper), e.g.: systemd.services.postgresql.after = [ "home-ooyude-External.mount" ]; Live test v8 sequence (before this fix): 1. umount /home/ooyude/External → OK, gone from /proc/mounts 2. findmnt /home/ooyude/External → exit=1, not in table 3. systemctl start postgresql → STARTED (guard bypassed) 4. journal: no ConditionPath error (service started successfully) This is the second guard bug found by live testing in this session. The first one was the '!' prefix inversion. Both were invisible to nix eval, both only visible at runtime. Lesson reinforced: guards MUST be live-tested, not just statically evaluated. Recovery: v8 test reverted state via emergency_recovery trap (remount + restart all services). System healthy at 10/10.
191 lines
5.1 KiB
Nix
191 lines
5.1 KiB
Nix
{
|
|
lib,
|
|
# The primary user's ids, bound from xlib.device by mkXlib. ntfs3/exfat
|
|
# volumes carry POSIX ids, so a mount using anything other than the real
|
|
# uid/gid shows every file as owned by `nobody`.
|
|
uid,
|
|
gid,
|
|
...
|
|
}:
|
|
# Pure helper functions for module definitions.
|
|
# Injected into every module via `xlib.helpers` (see default.nix).
|
|
#
|
|
# Defined in a `let` because they reference each other (mkTmpDirs uses
|
|
# mkTmpfile, mkServiceStorage uses mkTmpDirs + mkSystemdBind).
|
|
let
|
|
# tmpfiles rule: "type dir mode user group -"
|
|
mkTmpfile =
|
|
type: dir: mode: user: group:
|
|
"${type} ${dir} ${mode} ${user} ${group} -";
|
|
|
|
# several tmpfiles types for the same dir, e.g. ["d" "z"] or ["d" "Z"]
|
|
mkTmpDirs =
|
|
{
|
|
dir,
|
|
mode,
|
|
user,
|
|
group,
|
|
types ? [
|
|
"d"
|
|
"z"
|
|
],
|
|
}:
|
|
map (type: mkTmpfile type dir mode user group) types;
|
|
|
|
# fileSystems bind mount
|
|
mkBindMount =
|
|
{
|
|
what,
|
|
where,
|
|
}:
|
|
{
|
|
"${where}" = {
|
|
device = what;
|
|
fsType = "none";
|
|
options = [
|
|
"bind"
|
|
"nofail"
|
|
];
|
|
};
|
|
};
|
|
|
|
# systemd.mounts bind mount (automount variant)
|
|
mkSystemdBind =
|
|
{
|
|
what,
|
|
where,
|
|
}:
|
|
{
|
|
enable = true;
|
|
options = "bind,x-systemd.automount,nofail";
|
|
requires = [ "local-fs.target" ];
|
|
type = "none";
|
|
wantedBy = [ "multi-user.target" ];
|
|
inherit what where;
|
|
};
|
|
|
|
# Full "service storage" block: services-mnt source dir + /var/lib target,
|
|
# tmpfiles d/z + automount bind. Used as:
|
|
# storage = xlib.helpers.mkServiceStorage { name = "x"; user = "x"; group = "x"; };
|
|
# systemd = storage.systemd;
|
|
mkServiceStorage =
|
|
{
|
|
name,
|
|
user,
|
|
group,
|
|
mode ? "0755",
|
|
target ? "/var/lib/${name}",
|
|
base ? "/mnt/services",
|
|
}:
|
|
let
|
|
sourceDir = "${base}/${name}";
|
|
in
|
|
{
|
|
inherit sourceDir target;
|
|
systemd = {
|
|
tmpfiles.rules = mkTmpDirs {
|
|
dir = sourceDir;
|
|
inherit mode user group;
|
|
};
|
|
mounts = [
|
|
(mkSystemdBind {
|
|
what = sourceDir;
|
|
where = target;
|
|
})
|
|
];
|
|
};
|
|
};
|
|
|
|
# ntfs3 mount, e.g. fileSystems = mkNtfsMount { path = ...; uuid = ...; }
|
|
mkNtfsMount =
|
|
{
|
|
path,
|
|
uuid,
|
|
mask ? "0007",
|
|
enable ? null,
|
|
}:
|
|
{
|
|
"${path}" = {
|
|
device = "/dev/disk/by-uuid/${uuid}";
|
|
fsType = "ntfs3";
|
|
options = [
|
|
"defaults"
|
|
"uid=${toString uid}"
|
|
"gid=${toString gid}"
|
|
"fmask=${mask}"
|
|
"dmask=${mask}"
|
|
"nofail"
|
|
];
|
|
}
|
|
// lib.optionalAttrs (enable != null) { inherit enable; };
|
|
};
|
|
|
|
# exfat mount, e.g. fileSystems = mkExfatMount { path = ...; uuid = ...; }
|
|
mkExfatMount =
|
|
{
|
|
path,
|
|
uuid ? null,
|
|
label ? null,
|
|
}:
|
|
{
|
|
"${path}" = {
|
|
device = if uuid != null then "/dev/disk/by-uuid/${uuid}" else "/dev/disk/by-label/${label}";
|
|
fsType = "exfat";
|
|
options = [
|
|
"nofail"
|
|
"uid=${toString uid}"
|
|
"gid=${toString gid}"
|
|
];
|
|
};
|
|
};
|
|
|
|
# home-manager out-of-store symlinks: path = source (target name = attr name)
|
|
mkSymlinks =
|
|
config: paths:
|
|
lib.mapAttrs' (sourcePath: targetPath: {
|
|
name = targetPath;
|
|
value.source = config.lib.file.mkOutOfStoreSymlink "${sourcePath}";
|
|
}) paths;
|
|
in
|
|
{
|
|
inherit
|
|
mkTmpfile
|
|
mkTmpDirs
|
|
mkBindMount
|
|
mkSystemdBind
|
|
mkServiceStorage
|
|
mkNtfsMount
|
|
mkExfatMount
|
|
mkSymlinks
|
|
;
|
|
|
|
# Storage guard. Returns a systemd serviceConfig fragment that prevents
|
|
# a service from starting when the external storage filesystem
|
|
# (`xlib.dirs.server-home` = `/home/$user/External`) is not actually
|
|
# mounted. Without this guard, services whose `stateDir` / `dataDir` /
|
|
# bind mount source is a subdir of `/mnt/services` would happily start
|
|
# on an empty bind mount and create a fresh empty database — silent
|
|
# data loss. See T4 (B1) in `.agent/tasks/manifest.json` and R1.2 in
|
|
# `.agent/rules/project-rules.md`.
|
|
#
|
|
# Why this anchor: bind mounts under `/mnt/services` are inside the
|
|
# same filesystem as External, so `ConditionPathIsMountPoint` on those
|
|
# paths always reports "yes" (st_dev matches) — useless. We anchor on
|
|
# `server-home` (the real mount) instead.
|
|
#
|
|
# Usage in a service module:
|
|
# systemd.services.<name>.serviceConfig = xlib.helpers.mkStorageGuard xlib;
|
|
# or merge with an existing serviceConfig:
|
|
# serviceConfig = xlib.helpers.mkStorageGuard xlib // { ...other fields... };
|
|
mkStorageGuard = xlib: {
|
|
# Guard via ConditionPathIsMountPoint ONLY. We intentionally do
|
|
# NOT add RequiresMountsFor: that creates a dependency that causes
|
|
# systemd to auto-remount the filesystem when starting the unit,
|
|
# which defeats the guard (tested and confirmed: v8 test on
|
|
# sapphira, 2026-10-10). Ordering should be expressed via After=
|
|
# on the .mount unit in the consumer's systemd.services block,
|
|
# not via a guard-creating dependency here.
|
|
ConditionPathIsMountPoint = [ xlib.dirs.server-home ];
|
|
};
|
|
}
|