#!/bin/bash # Build the `unshare` prefix that lets a NON-ROOT restic recreate the container # uids a snapshot recorded. # # Why this is needed: backups run as the docker install user (runBackupOp — the # backup engine never gets root). A non-root restic cannot chown a restored file # to anyone else, so every file came back owned by that user. For LibrePortal's # own files that is correct; for the ones a CONTAINER owns it is fatal. Under # rootless, a container process running as uid N appears on the host as # subuid_start + N - 1 (prometheus' nobody -> 296605, postgres -> 231141), and an # app whose data dir is no longer owned by its own uid does not start: # prometheus dies on "open data/queries.active: permission denied", and postgres # refuses outright unless its data dir is 0700 and its own. Restores therefore # handed back apps that could not boot. # # The fix needs no new privilege. The docker install user already owns a subuid # range (that is what makes rootless work), so it may enter a user namespace in # which it is root and those subuids are mappable. Mapping them to THEMSELVES # means an id recorded in the snapshot is written back as the same host id. # # Files recorded as the docker install user's own uid are the one gap: that uid # is outside the subuid range and is already consumed by the inner-root mapping, # so restic's lchown for them fails with EINVAL. It is harmless — restic runs as # inner root, which IS that user on the host, so those files already land with # exactly the right owner. resticRestoreErrorsAreBenign below is what keeps that # from being reported as a failed restore. # True when every error restic reported is the expected "cannot map the backup # user's own uid" one described above. Anything else — a missing pack, a full # disk, a permission problem on the target — must still fail the restore. resticRestoreErrorsAreBenign() { local out="$1" local bad # Every line restic prints for a failed ownership set, minus the benign form. bad=$(printf '%s\n' "$out" | grep -E "^ignoring error for " \ | grep -vE "lchown .*: (invalid argument|operation not permitted)$") [[ -z "$bad" ]] } resticRestoreSnapshot() { local idx="$1" local snapshot_id="$2" local target_dir="$3" local include_path="$4" if [[ -z "$snapshot_id" || -z "$target_dir" ]]; then isError "resticRestoreSnapshot requires snapshot_id and target_dir" return 1 fi resticEnvExport "$idx" || return 1 runFileOp mkdir -p "$target_dir" local args=(restore "$snapshot_id" --target "$target_dir") [[ -n "$include_path" ]] && args+=(--include "$include_path") isNotice "Restoring ${snapshot_id:0:8} from $(resticLocationName "$idx") → $target_dir" local ns_prefix=() mapfile -t ns_prefix < <(backupUsernsPrefix) # Output is captured (not streamed) so the benign-error check below can read # it; it is echoed straight back afterwards, so the operator sees the same # restic report as before. local out rc out=$(runBackupOp "${ns_prefix[@]}" restic "${args[@]}" 2>&1) rc=$? printf '%s\n' "$out" # restic exits non-zero for un-mappable-uid lchowns even though the file # CONTENTS landed. Forgive only that case. # # With the namespace mapping fixed (see restic-userns-exec), the only id # that is still unmappable is the restoring user's own — its slot is spent # on inner root — and a file stored as : lands owned by the # caller regardless, because that is who inner root is on the outside. So # these really are correct, which is what the message used to claim before # the mapping worked and container-owned data was quietly losing its owner. # # Still report the count: if this number is large the mapping has stopped # working again, and the symptom is an app that cannot write its own data. if [[ $rc -ne 0 && ${#ns_prefix[@]} -gt 0 ]] && resticRestoreErrorsAreBenign "$out"; then local _lch _lch=$(printf '%s\n' "$out" | grep -cE "^ignoring error for .*lchown ") isNotice "Ownership warnings on ${_lch} file(s) owned by ${docker_install_user:-the backup user} — expected; they are restored correctly." rc=0 fi resticEnvUnset return $rc } resticRestoreAppLatest() { local idx="$1" local app_name="$2" local target_dir="$3" local host="${4:-$CFG_INSTALL_NAME}" local snapshot_id snapshot_id=$(resticSnapshotLatestId "$idx" "$app_name" "$host") if [[ -z "$snapshot_id" ]]; then isError "No snapshot found in $(resticLocationName "$idx") for app=$app_name host=$host" return 1 fi # Prefer the path the SNAPSHOT records over this host's layout: they differ # whenever the snapshot came from a host with a different --containers-dir, # or from a different storage location, and an include filter that matches # nothing restores nothing without saying so. local include_path="" if declare -f storageSnapshotSourcePath >/dev/null 2>&1; then include_path=$(storageSnapshotSourcePath "$idx" "$snapshot_id" "$app_name" 2>/dev/null) || include_path="" fi [[ -z "$include_path" ]] && include_path="$(appDir "$app_name")" resticRestoreSnapshot "$idx" "$snapshot_id" "$target_dir" "$include_path" } resticRestoreSystemLatest() { local idx="$1" local target_dir="$2" local host="${3:-$CFG_INSTALL_NAME}" resticEnvExport "$idx" || return 1 local snapshot_id snapshot_id=$(runBackupOp restic snapshots \ --tag "system=config" --host "$host" \ --latest 1 --json --no-lock 2>/dev/null | \ grep -o '"short_id":"[^"]*"' | head -1 | cut -d'"' -f4) resticEnvUnset if [[ -z "$snapshot_id" ]]; then isError "No system-config snapshot found in $(resticLocationName "$idx") for host=$host" return 1 fi # Whole-snapshot restore (the snapshot is just the config tree) into staging. resticRestoreSnapshot "$idx" "$snapshot_id" "$target_dir" }