feat(storage): path primitives for multiple container roots

Phase 0 of docs/roadmap/storage-locations.md — the resolver layer. No
behaviour change yet: with no registry present, every function here
returns exactly what the old single-root code did, which is what makes
the ~200-site sweep that follows safe to land incrementally.

  primaryRoot / webuiDir     the install-time root, and the one tree that
                             never moves
  storageRoots               every registered root, primary first
  storageRootAvailable       marker present == drive mounted
  pathIsContainerData        replaces the `== "$containers_dir"*` idiom
                             that picks manager vs container-user elevation
  appDir / appDirSlash       THE resolver, memoised
  storageLocationPath/Name   name <-> path, via the root-owned registry
  storageIndexGet/Set        app -> location cache

appDir resolves discovery-first: whichever root actually holds
<slug>/<slug>.config wins, so a hand-move or half-finished migration
self-heals rather than corrupting.

The index exists because of a bug the unit test caught immediately.
Discovery cannot see an unmounted disk, so an installed app on an
unplugged drive looked identical to a brand-new app — and the fallback
handed back the PRIMARY root. Docker would then have created the bind
mounts there and booted the app empty on the wrong disk, which is the
precise failure the availability design exists to prevent. The index is
manager-owned (deliberately not on the removable disk: it must be
readable exactly when that disk is absent), consulted only when the scan
comes up empty, and rewritten by every successful scan so the disk stays
authoritative whenever it is actually present.

Availability is gated in appDir alone rather than at each caller: every
site reaches it by construction. It returns non-zero AND prints an
unusable sentinel path, so the many callers that will never check $?
still fail loudly on something harmless.

scripts/dev/lp-storage-test covers all of it against a throwaway tree.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
librelad 2026-08-24 04:01:19 +01:00
parent 48c78270a9
commit db04f8f000
2 changed files with 418 additions and 0 deletions

67
scripts/dev/lp-storage-test Executable file
View File

@ -0,0 +1,67 @@
#!/bin/bash
# Self-contained unit test for the storage-location primitives in
# scripts/source/paths.sh. Runs against a throwaway tree in $TMPDIR — never
# touches a real install. No arguments; exits non-zero on the first failure.
#
# scripts/dev/lp-storage-test
#
# The case worth keeping honest is the last one: an app whose drive is not
# mounted must resolve to the sentinel, NOT to the primary root. Silently
# falling back is how an app gets rebuilt empty on the wrong disk.
REPO="$(cd "$(dirname "$0")/../.." && pwd)"
BASE="$(mktemp -d "${TMPDIR:-/tmp}/lp-storage-test-XXXXXX")"
trap 'rm -rf "$BASE"' EXIT
export LP_SYSTEM_DIR="$BASE/sys"
export LP_CONTAINERS_DIR="$BASE/primary"
export LP_BACKUPS_DIR="$BASE/bk"
export LP_STORAGE_REGISTRY="$BASE/storage.roots"
mkdir -p "$BASE/primary" "$BASE/disk2" "$BASE/sys/configs"
source "$REPO/scripts/source/paths.sh"
fail=0
chk(){ if [[ "$2" == "$3" ]]; then echo " ok $1"; else echo " FAIL $1: got '$2' want '$3'"; fail=1; fi; }
echo "--- no registry (behaves exactly like the old single-root code) ---"
chk "primaryRoot" "$(primaryRoot)" "$BASE/primary"
chk "webuiDir" "$(webuiDir)" "$BASE/primary/libreportal"
chk "roots" "$(storageRoots)" "$BASE/primary"
chk "appDir new" "$(appDir bookstack)" "$BASE/primary/bookstack"
pathIsContainerData "$BASE/primary/bookstack/x" && echo " ok pathIsContainerData inside" || { echo " FAIL inside"; fail=1; }
pathIsContainerData "$BASE/sys/configs" && { echo " FAIL outside matched"; fail=1; } || echo " ok pathIsContainerData outside"
echo "--- discovery: app deployed on the primary root ---"
mkdir -p "$BASE/primary/bookstack" && : > "$BASE/primary/bookstack/bookstack.config"
storageCacheReset
chk "appDir found" "$(appDir bookstack)" "$BASE/primary/bookstack"
echo "--- second root, mounted (marker present) ---"
printf '2\t%s\t0:0\tuuid-2\n' "$BASE/disk2" > "$BASE/storage.roots"
: > "$BASE/disk2/.libreportal-storage"
mkdir -p "$BASE/disk2/nextcloud" && : > "$BASE/disk2/nextcloud/nextcloud.config"
storageCacheReset
chk "roots both" "$(storageRoots | tr '\n' ' ')" "$BASE/primary $BASE/disk2 "
chk "appDir disk2" "$(appDir nextcloud)" "$BASE/disk2/nextcloud"
chk "appDir primary still" "$(appDir bookstack)" "$BASE/primary/bookstack"
pathIsContainerData "$BASE/disk2/nextcloud" && echo " ok pathIsContainerData disk2" || { echo " FAIL disk2"; fail=1; }
chk "locName disk2" "$(storageLocationName "$BASE/disk2")" "2"
chk "locPath by id" "$(storageLocationPath 2)" "$BASE/disk2"
CFG_STORAGE_LOC_2_NAME="bigdisk"
chk "locPath by name" "$(storageLocationPath bigdisk)" "$BASE/disk2"
chk "locName friendly" "$(storageLocationName "$BASE/disk2")" "bigdisk"
echo "--- intent for an app not yet deployed ---"
CFG_JELLYFIN_STORAGE="bigdisk"
storageCacheReset
chk "appDir intent" "$(appDir jellyfin)" "$BASE/disk2/jellyfin"
echo "--- drive unmounted: marker gone ---"
rm -f "$BASE/disk2/.libreportal-storage"
storageCacheReset
out=$(appDir nextcloud); rc=$?
chk "appDir sentinel" "$out" "/nonexistent-libreportal-unavailable/nextcloud"
chk "appDir rc" "$rc" "1"
appStorageAvailable nextcloud && { echo " FAIL available"; fail=1; } || echo " ok appStorageAvailable false"
appStorageAvailable bookstack && echo " ok primary still available" || { echo " FAIL primary"; fail=1; }
storageRootAvailable "$BASE/disk2" && { echo " FAIL root available"; fail=1; } || echo " ok storageRootAvailable false"
echo; [[ $fail -eq 0 ]] && echo "ALL PASS" || echo "FAILURES"; exit $fail

View File

@ -64,3 +64,354 @@ backup_dir="$LP_BACKUPS_DIR"
# --- Control-plane manager user (configurable; baked into helpers at install) - # --- Control-plane manager user (configurable; baked into helpers at install) -
# The systemd unit + CLI wrapper export LP_MANAGER_USER; else default libreportal. # The systemd unit + CLI wrapper export LP_MANAGER_USER; else default libreportal.
sudo_user_name="${LP_MANAGER_USER:-libreportal}" sudo_user_name="${LP_MANAGER_USER:-libreportal}"
# =============================================================================
# Storage locations — the containers root is a LIST, not a single path.
# =============================================================================
# See docs/roadmap/storage-locations.md. Three ideas, in dependency order:
#
# 1. The set of roots that may hold app data comes from a ROOT-OWNED registry
# ($lp_storage_registry). The manager can read it and never write it — the
# same trust boundary that makes the baked __CONTAINERS_DIR__ safe in the
# /usr/local/lib/libreportal/ helpers. Adding a root goes through
# `libreportal-storage add`, which validates and is the only writer.
#
# 2. An app's directory is DISCOVERED, not declared: whichever root actually
# holds <slug>/<slug>.config wins. CFG_<SLUG>_STORAGE records intent and is
# consulted only when the app isn't on disk yet. Disagreement resolves to
# the disk, which is what makes a hand-move or a half-finished migration
# self-heal instead of corrupting.
#
# 3. A root is AVAILABLE only while its marker file is present. The marker
# lives on the drive, so an unmounted disk has no marker and the root is
# unavailable — appDir fails rather than handing back a path that docker
# would happily populate on the bare mountpoint.
#
# Everything here must stay cheap: pathIsContainerData is on the hot path of
# every file helper, so the roots list and the slug->dir map are both memoised
# in globals and loaded without a subshell.
# Root-owned registry: "<id>\t<path>\t<dev>\t<fs_uuid>" per line. Absent on an
# install that predates storage locations — then the primary root is the only
# root and every function here behaves exactly as the old single-root code did.
lp_storage_registry="${LP_STORAGE_REGISTRY:-/usr/local/lib/libreportal/storage.roots}"
# Written root-owned into a location's top level at registration. Presence is
# the mount test (see idea 3 above).
lp_storage_marker=".libreportal-storage"
# Returned by appDir when a location exists but its drive is not mounted. A path
# that cannot exist, so a caller that ignores the non-zero status still fails
# loudly on something harmless instead of writing to a bare mountpoint.
lp_storage_unavailable="/nonexistent-libreportal-unavailable"
# Memo state. Declared here so the associative array exists before first use and
# so a re-source of paths.sh starts from a clean map rather than a stale one.
declare -A LP_APP_DIR_CACHE=()
LP_APP_DIR_SCANNED=0
declare -a LP_STORAGE_ROOTS=()
LP_STORAGE_ROOTS_LOADED=0
# The install-time containers root. Always present, never removable, and the
# home of anything that must not move (the WebUI's own tree).
primaryRoot()
{
printf '%s' "${LP_CONTAINERS_DIR%/}"
}
# LibrePortal's own container dir. Structurally pinned to the primary root —
# too much of the control plane reaches into it by literal path for it to move.
webuiDir()
{
printf '%s' "${LP_CONTAINERS_DIR%/}/libreportal"
}
# Populate LP_STORAGE_ROOTS (primary first). Sets the global directly — callers
# must NOT wrap this in a subshell or the memo is lost.
_lpStorageRootsLoad()
{
[[ "${LP_STORAGE_ROOTS_LOADED:-0}" == "1" ]] && return 0
LP_STORAGE_ROOTS=("${LP_CONTAINERS_DIR%/}")
local _id _path _rest
if [[ -r "$lp_storage_registry" ]]; then
while IFS=$'\t' read -r _id _path _rest || [[ -n "$_id" ]]; do
[[ -z "$_path" || "$_id" == \#* ]] && continue
_path="${_path%/}"
[[ "$_path" == "${LP_CONTAINERS_DIR%/}" ]] && continue
LP_STORAGE_ROOTS+=("$_path")
done < "$lp_storage_registry"
fi
LP_STORAGE_ROOTS_LOADED=1
return 0
}
# Drop every memo. Call after registering/removing a location or moving an app —
# otherwise a long-lived process (the task processor) keeps serving stale paths.
storageCacheReset()
{
LP_STORAGE_ROOTS=()
LP_STORAGE_ROOTS_LOADED=0
LP_APP_DIR_CACHE=()
LP_APP_DIR_SCANNED=0
return 0
}
# Every registered container-data root, primary first, one per line, no trailing
# slash. Includes roots whose drive is currently absent — callers that care ask
# storageRootAvailable.
storageRoots()
{
_lpStorageRootsLoad
local r
for r in "${LP_STORAGE_ROOTS[@]}"; do
printf '%s\n' "$r"
done
}
# Is this root usable right now? The primary root is definitionally available
# (if it is gone, so is the install). A registered root needs its marker, which
# is only readable when the drive is actually mounted.
storageRootAvailable()
{
local root="${1%/}"
[[ -z "$root" ]] && return 1
[[ "$root" == "${LP_CONTAINERS_DIR%/}" ]] && return 0
[[ -e "$root/$lp_storage_marker" ]]
}
# Is $1 inside ANY container-data root? Replaces the
# `[[ "$p" == "$containers_dir"* ]]` idiom that decides whether a file op runs
# as the container user or the manager. A miss here is a wrong-owner file that
# fails much later, so it deliberately matches the root itself as well as
# anything beneath it, with or without a trailing slash.
pathIsContainerData()
{
local p="$1"
[[ -z "$p" ]] && return 1
_lpStorageRootsLoad
local root
for root in "${LP_STORAGE_ROOTS[@]}"; do
[[ -z "$root" ]] && continue
[[ "$p" == "$root" || "$p" == "$root/"* ]] && return 0
done
return 1
}
# Resolve a location NAME to its root path. Names live in the manager-owned
# per-location config (CFG_STORAGE_LOC_<id>_NAME); paths live in the root-owned
# registry. "default" (and an unset value) is the primary root.
storageLocationPath()
{
local want="$1"
[[ -z "$want" || "$want" == "default" ]] && { primaryRoot; return 0; }
local _id _path _rest name_var
if [[ -r "$lp_storage_registry" ]]; then
while IFS=$'\t' read -r _id _path _rest || [[ -n "$_id" ]]; do
[[ -z "$_path" || "$_id" == \#* ]] && continue
# Match on the id itself, or on the friendly name from its config.
name_var="CFG_STORAGE_LOC_${_id}_NAME"
if [[ "$_id" == "$want" || "${!name_var:-}" == "$want" ]]; then
printf '%s' "${_path%/}"
return 0
fi
done < "$lp_storage_registry"
fi
return 1
}
# Reverse: root path -> location name (falls back to the id, then "default").
storageLocationName()
{
local want="${1%/}"
[[ -z "$want" || "$want" == "${LP_CONTAINERS_DIR%/}" ]] && { printf 'default'; return 0; }
local _id _path _rest name_var
if [[ -r "$lp_storage_registry" ]]; then
while IFS=$'\t' read -r _id _path _rest || [[ -n "$_id" ]]; do
[[ -z "$_path" || "$_id" == \#* ]] && continue
if [[ "${_path%/}" == "$want" ]]; then
name_var="CFG_STORAGE_LOC_${_id}_NAME"
printf '%s' "${!name_var:-$_id}"
return 0
fi
done < "$lp_storage_registry"
fi
return 1
}
# --- app -> location index ---------------------------------------------------
# A manager-owned CACHE of which root each app was last seen on. Discovery still
# wins whenever the disk is present; this exists for the one case discovery
# cannot answer:
#
# an app on a drive that is not mounted is invisible to the scan, and without
# the index appDir would fall back to the primary root — handing back a path
# docker would populate on the wrong disk, booting the app empty. That is the
# exact failure the whole availability design exists to prevent, so "not found
# by the scan" must not silently mean "belongs on the primary root".
#
# Lives in the manager-owned configs tree (never on a removable disk — it has to
# be readable precisely when that disk is gone). Self-heals: every successful
# scan rewrites the entries it observed.
storageIndexFile()
{
printf '%s' "${configs_dir%/}/storage/app_locations"
}
storageIndexGet()
{
local slug="$1" f s r
f=$(storageIndexFile)
[[ -r "$f" ]] || return 1
while IFS=$'\t' read -r s r || [[ -n "$s" ]]; do
[[ "$s" == "$slug" ]] || continue
[[ -z "$r" ]] && return 1
printf '%s' "${r%/}"
return 0
done < "$f"
return 1
}
# Record (or clear, with an empty root) an app's location. Idempotent.
storageIndexSet()
{
local slug="$1" root="${2%/}" f tmp s r
[[ -z "$slug" ]] && return 1
f=$(storageIndexFile)
local cur=""
cur=$(storageIndexGet "$slug" 2>/dev/null) || cur=""
[[ "$cur" == "$root" ]] && return 0
local op=""
declare -F runInstallOp >/dev/null 2>&1 && op="runInstallOp"
$op mkdir -p "${f%/*}" 2>/dev/null
tmp=$(mktemp 2>/dev/null) || return 1
if [[ -r "$f" ]]; then
while IFS=$'\t' read -r s r || [[ -n "$s" ]]; do
[[ -z "$s" || "$s" == "$slug" ]] && continue
printf '%s\t%s\n' "$s" "$r" >> "$tmp"
done < "$f"
fi
[[ -n "$root" ]] && printf '%s\t%s\n' "$slug" "$root" >> "$tmp"
if declare -F runInstallWrite >/dev/null 2>&1; then
runInstallWrite "$f" < "$tmp"
else
cat "$tmp" > "$f" 2>/dev/null
fi
rm -f "$tmp"
return 0
}
storageIndexRemove()
{
storageIndexSet "$1" ""
}
# Where the app's config SAYS it should live. Only consulted when the app isn't
# on disk yet — the disk always wins for a deployed app.
_appDirIntended()
{
local slug="$1"
local key="CFG_${slug^^}_STORAGE"
local want="${!key:-}"
local path
if [[ -n "$want" ]] && path=$(storageLocationPath "$want"); then
printf '%s' "$path"
return 0
fi
primaryRoot
}
# Build the slug -> directory map by scanning every AVAILABLE root for
# <slug>/<slug>.config. Primary root first, so it wins a duplicate — a stray
# copy on a second disk can never hijack an app that is live on the primary.
#
# runFileOp because under rootless the container tree is owned by the docker
# install user and is not list-readable by the manager; without it this silently
# finds nothing and every app resolves to the fallback.
_appDirScan()
{
[[ "${LP_APP_DIR_SCANNED:-0}" == "1" ]] && return 0
LP_APP_DIR_SCANNED=1
_lpStorageRootsLoad
local scan_op=""
declare -F runFileOp >/dev/null 2>&1 && scan_op="runFileOp"
local root cfg slug dir
for root in "${LP_STORAGE_ROOTS[@]}"; do
[[ -z "$root" ]] && continue
storageRootAvailable "$root" || continue
while IFS= read -r cfg; do
[[ -z "$cfg" ]] && continue
dir="${cfg%/*}"
slug="${dir##*/}"
# Only <dir>/<dir>.config counts — an app dir holds other *.config
# payload files and those must not register as apps.
[[ "${cfg##*/}" == "$slug.config" ]] || continue
[[ -n "${LP_APP_DIR_CACHE[$slug]:-}" ]] && continue
LP_APP_DIR_CACHE["$slug"]="$root/$slug"
# Self-heal: what we can see is authoritative, so correct the index
# for a hand-move or a half-finished migration.
storageIndexSet "$slug" "$root" 2>/dev/null
done < <($scan_op find "$root" -mindepth 2 -maxdepth 2 -type f -name '*.config' 2>/dev/null)
done
return 0
}
# THE resolver. Prints the app's directory (no trailing slash).
#
# Returns non-zero — and prints an unusable sentinel path — when the app's
# location exists but its drive is not mounted. This is the single central
# availability gate: every caller reaches it by construction, so a missing disk
# fails here instead of needing a guard at ~200 call sites.
appDir()
{
local slug="$1"
if [[ -z "$slug" ]]; then
printf '%s' "$lp_storage_unavailable"
return 1
fi
_appDirScan
local dir="${LP_APP_DIR_CACHE[$slug]:-}"
# Not on any AVAILABLE root. Before assuming it is new, ask the index —
# an installed app whose drive is unplugged looks identical to a new one,
# and guessing "primary root" there is the corrupting answer.
if [[ -z "$dir" ]]; then
local known=""
if known=$(storageIndexGet "$slug" 2>/dev/null) && [[ -n "$known" ]]; then
dir="$known/$slug"
else
dir="$(_appDirIntended "$slug")/$slug"
fi
fi
if ! storageRootAvailable "${dir%/*}"; then
printf '%s' "$lp_storage_unavailable/$slug"
return 1
fi
printf '%s' "$dir"
return 0
}
# Trailing-slash form, for the many call sites that built "$containers_dir$app/"
appDirSlash()
{
local d
d=$(appDir "$1") || { printf '%s/' "$d"; return 1; }
printf '%s/' "$d"
}
# True when the app's storage is present and usable. Cheap gate for callers that
# want to skip rather than fail (boot reconcile, status generators).
appStorageAvailable()
{
appDir "$1" >/dev/null 2>&1
}