resticRestoreSnapshot forgives the un-mappable-uid lchown failures so a restore is not aborted by them, and reported: "expected, they are already owned correctly". That is true only for LibrePortal's own files, whose owner is the backup user restic already runs as. It is false for container-owned data. _resticUsernsPrefix maps the subuid range and root, but unshare takes one range per option so the backup user's own GID is never mapped — and app data is written as <container-uid>:<backup-user>. Every such chown fails with EINVAL and the file falls back to <backup-user>:<backup-user>. Verified directly: 231543:231543 applies, 231543:1002 does not. Observed on a 13-app restore: grafana's grafana.db is recorded as 231543:1002 and landed as 1002:1002, so grafana (running as 231543) could not write it at mode 0640 and died with "attempt to write a readonly database" — under a restore that reported success. 1626 of that run's 2086 failed chowns were grafana's. This commit does not fix the mapping — that is the backup engine's ownership handling rather than the first-run restore path, and the candidate fixes (newuidmap multi-range maps, or restoring as root via a path-validated helper) want a decision first. See docs/roadmap/first-run-restore.md §3.5. What it fixes is the reporting: count the files and say plainly that container-owned data was not reinstated and the app may fail to write. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
176 lines
7.5 KiB
Bash
176 lines
7.5 KiB
Bash
#!/bin/bash
|
|
|
|
# Build the `unshare` prefix that lets a NON-ROOT restic recreate the container
|
|
# uids a snapshot recorded.
|
|
#
|
|
# Why this is needed: backups run as the docker install user (runBackupOp — the
|
|
# backup engine never gets root). A non-root restic cannot chown a restored file
|
|
# to anyone else, so every file came back owned by that user. For LibrePortal's
|
|
# own files that is correct; for the ones a CONTAINER owns it is fatal. Under
|
|
# rootless, a container process running as uid N appears on the host as
|
|
# subuid_start + N - 1 (prometheus' nobody -> 296605, postgres -> 231141), and an
|
|
# app whose data dir is no longer owned by its own uid does not start:
|
|
# prometheus dies on "open data/queries.active: permission denied", and postgres
|
|
# refuses outright unless its data dir is 0700 and its own. Restores therefore
|
|
# handed back apps that could not boot.
|
|
#
|
|
# The fix needs no new privilege. The docker install user already owns a subuid
|
|
# range (that is what makes rootless work), so it may enter a user namespace in
|
|
# which it is root and those subuids are mappable. Mapping them to THEMSELVES
|
|
# means an id recorded in the snapshot is written back as the same host id.
|
|
#
|
|
# Files recorded as the docker install user's own uid are the one gap: that uid
|
|
# is outside the subuid range and is already consumed by the inner-root mapping,
|
|
# so restic's lchown for them fails with EINVAL. It is harmless — restic runs as
|
|
# inner root, which IS that user on the host, so those files already land with
|
|
# exactly the right owner. resticRestoreErrorsAreBenign below is what keeps that
|
|
# from being reported as a failed restore.
|
|
_resticUsernsPrefix()
|
|
{
|
|
local usr="${docker_install_user:-dockerinstall}"
|
|
command -v unshare >/dev/null 2>&1 || return 0
|
|
|
|
local uline gline ustart ucount gstart gcount
|
|
uline=$(grep "^${usr}:" /etc/subuid 2>/dev/null | head -1)
|
|
gline=$(grep "^${usr}:" /etc/subgid 2>/dev/null | head -1)
|
|
# No subuid range (rooted mode, or a hand-rolled account) — nothing to map,
|
|
# so leave the call exactly as it was rather than guess.
|
|
[[ -n "$uline" && -n "$gline" ]] || return 0
|
|
|
|
ustart="${uline#*:}"; ustart="${ustart%%:*}"; ucount="${uline##*:}"
|
|
gstart="${gline#*:}"; gstart="${gstart%%:*}"; gcount="${gline##*:}"
|
|
[[ "$ustart" =~ ^[0-9]+$ && "$ucount" =~ ^[0-9]+$ ]] || return 0
|
|
[[ "$gstart" =~ ^[0-9]+$ && "$gcount" =~ ^[0-9]+$ ]] || return 0
|
|
|
|
printf '%s\n' unshare --map-root-user \
|
|
"--map-users=${ustart}:${ustart}:${ucount}" \
|
|
"--map-groups=${gstart}:${gstart}:${gcount}"
|
|
}
|
|
|
|
# True when every error restic reported is the expected "cannot map the backup
|
|
# user's own uid" one described above. Anything else — a missing pack, a full
|
|
# disk, a permission problem on the target — must still fail the restore.
|
|
resticRestoreErrorsAreBenign()
|
|
{
|
|
local out="$1"
|
|
local bad
|
|
# Every line restic prints for a failed ownership set, minus the benign form.
|
|
bad=$(printf '%s\n' "$out" | grep -E "^ignoring error for " \
|
|
| grep -vE "lchown .*: (invalid argument|operation not permitted)$")
|
|
[[ -z "$bad" ]]
|
|
}
|
|
|
|
resticRestoreSnapshot()
|
|
{
|
|
local idx="$1"
|
|
local snapshot_id="$2"
|
|
local target_dir="$3"
|
|
local include_path="$4"
|
|
|
|
if [[ -z "$snapshot_id" || -z "$target_dir" ]]; then
|
|
isError "resticRestoreSnapshot requires snapshot_id and target_dir"
|
|
return 1
|
|
fi
|
|
|
|
resticEnvExport "$idx" || return 1
|
|
|
|
runFileOp mkdir -p "$target_dir"
|
|
|
|
local args=(restore "$snapshot_id" --target "$target_dir")
|
|
[[ -n "$include_path" ]] && args+=(--include "$include_path")
|
|
|
|
isNotice "Restoring ${snapshot_id:0:8} from $(resticLocationName "$idx") → $target_dir"
|
|
|
|
local ns_prefix=()
|
|
mapfile -t ns_prefix < <(_resticUsernsPrefix)
|
|
|
|
# Output is captured (not streamed) so the benign-error check below can read
|
|
# it; it is echoed straight back afterwards, so the operator sees the same
|
|
# restic report as before.
|
|
local out rc
|
|
out=$(runBackupOp "${ns_prefix[@]}" restic "${args[@]}" 2>&1)
|
|
rc=$?
|
|
printf '%s\n' "$out"
|
|
|
|
# restic exits non-zero for un-mappable-uid lchowns even though the file
|
|
# CONTENTS landed. Forgive only that case — but do not pretend it is
|
|
# nothing, which is what this used to do.
|
|
#
|
|
# It said "expected, they are already owned correctly". That holds for
|
|
# LibrePortal's own files (owner = the backup user, which is who restic runs
|
|
# as anyway). It does NOT hold for container-owned data, and the namespace
|
|
# this runs in cannot currently map those: unshare takes a single range per
|
|
# option, so _resticUsernsPrefix maps the subuid range and root but not the
|
|
# backup user's own GID — and LibrePortal writes app data as
|
|
# <container-uid>:<backup-user>. Every such chown fails and the file falls
|
|
# back to <backup-user>:<backup-user>.
|
|
#
|
|
# Observed: grafana's grafana.db is recorded in the snapshot as 231543:1002
|
|
# and restored as 1002:1002. At mode 0640 the grafana process — running as
|
|
# 231543 — then cannot write it, and the app dies with "attempt to write a
|
|
# readonly database". The restore reported success.
|
|
#
|
|
# So: still do not fail the restore (the data is there and some apps are
|
|
# rehydrated by other means), but say plainly what was not reinstated.
|
|
if [[ $rc -ne 0 && ${#ns_prefix[@]} -gt 0 ]] && resticRestoreErrorsAreBenign "$out"; then
|
|
local _lch
|
|
_lch=$(printf '%s\n' "$out" | grep -cE "^ignoring error for .*lchown ")
|
|
isNotice "Restore could not reinstate ownership on ${_lch} file(s); they now belong to ${docker_install_user:-the backup user}."
|
|
isNotice "LibrePortal's own files are correct that way. Container-owned data is NOT — an app may fail to write (e.g. a read-only database). Check the app after it starts."
|
|
rc=0
|
|
fi
|
|
|
|
resticEnvUnset
|
|
return $rc
|
|
}
|
|
|
|
resticRestoreAppLatest()
|
|
{
|
|
local idx="$1"
|
|
local app_name="$2"
|
|
local target_dir="$3"
|
|
local host="${4:-$CFG_INSTALL_NAME}"
|
|
|
|
local snapshot_id
|
|
snapshot_id=$(resticSnapshotLatestId "$idx" "$app_name" "$host")
|
|
|
|
if [[ -z "$snapshot_id" ]]; then
|
|
isError "No snapshot found in $(resticLocationName "$idx") for app=$app_name host=$host"
|
|
return 1
|
|
fi
|
|
|
|
# Prefer the path the SNAPSHOT records over this host's layout: they differ
|
|
# whenever the snapshot came from a host with a different --containers-dir,
|
|
# or from a different storage location, and an include filter that matches
|
|
# nothing restores nothing without saying so.
|
|
local include_path=""
|
|
if declare -f storageSnapshotSourcePath >/dev/null 2>&1; then
|
|
include_path=$(storageSnapshotSourcePath "$idx" "$snapshot_id" "$app_name" 2>/dev/null) || include_path=""
|
|
fi
|
|
[[ -z "$include_path" ]] && include_path="$(appDir "$app_name")"
|
|
resticRestoreSnapshot "$idx" "$snapshot_id" "$target_dir" "$include_path"
|
|
}
|
|
|
|
resticRestoreSystemLatest()
|
|
{
|
|
local idx="$1"
|
|
local target_dir="$2"
|
|
local host="${3:-$CFG_INSTALL_NAME}"
|
|
|
|
resticEnvExport "$idx" || return 1
|
|
local snapshot_id
|
|
snapshot_id=$(runBackupOp restic snapshots \
|
|
--tag "system=config" --host "$host" \
|
|
--latest 1 --json --no-lock 2>/dev/null | \
|
|
grep -o '"short_id":"[^"]*"' | head -1 | cut -d'"' -f4)
|
|
resticEnvUnset
|
|
|
|
if [[ -z "$snapshot_id" ]]; then
|
|
isError "No system-config snapshot found in $(resticLocationName "$idx") for host=$host"
|
|
return 1
|
|
fi
|
|
|
|
# Whole-snapshot restore (the snapshot is just the config tree) into staging.
|
|
resticRestoreSnapshot "$idx" "$snapshot_id" "$target_dir"
|
|
}
|