All checks were successful
CI / Lint, typecheck, test (push) Successful in 3m45s
CD / Build and push images (push) Successful in 3m49s
CI / Build container images (push) Has been skipped
CD / Deploy to Test (push) Successful in 9s
CD / Smoke tests against Test (push) Successful in 1m18s
CD / Promote to Int (push) Successful in 11s
CI / Auth e2e pack (push) Successful in 5m35s
CI / Import/export fidelity gate (push) Successful in 47s
Off-host backups for every self-hoster, configured entirely in the admin UI — supersedes the host-specific mirror plan behind #84. shared: - webdav.ts (new package entry like token-crypto): minimal WebDAV client with basic auth — PROPFIND (tolerant multistatus parser), MKCOL, PUT (streamed), GET, DELETE; Nextcloud DAV path derived from the plain server URL, explicit DAV bases pass through - backup-status.ts: additive remote-upload status in status.json, the restore-status.json contract (running/succeeded/failed + staleness bound), the backup_command/backup_maintenance NOTIFY channels, and the one-bundle-per-set naming (dorfteich-backup-<id>.tar.gz) - backup-set.ts moved here from apps/backup (api lists local sets) backup sidecar: - reads the backup.* instance settings directly from the database (admin changes apply next run; local retention row overrides the env) and the app password from the secret store - after each successful set: bundle dump + files archive + manifest into ONE self-contained tar.gz, upload via WebDAV per schedule (off/daily/weekly; manual runs always upload), prune remote bundles — never the newest — and record the outcome in status.json; upload failures alert via a new backupUploadFailed mail (de+en) - command listener on backup_command (run / restore) with a serial queue against the nightly timer - restore orchestrator: restore-status.json → maintenance NOTIFY → grace → (remote: download + manifest-verify bundle) → terminate other DB connections → shared perform-restore path (same code as restore.sh) → final status + maintenance exit api: - MaintenanceGuard (global, registered before the setup gate): 503 maintenance_mode while restore-status says running; health endpoints and the new public GET /backup/restore-status stay exempt; a stale running state (crashed sidecar) unblocks after 30 min - MaintenanceStateService watches the file and restarts the api after a successful restore (fresh caches, migrate-on-start for older dumps); main.ts refuses to touch the database while a restore runs — a container restarting mid-restore must not race pg_restore with migrate deploy - worker sweeps (conversion, mail outbox, scheduler) catch transient database failures instead of dying on an unhandled rejection — the restore's connection termination crashed the api in verification - backup admin endpoints under /admin/system/backup: settings (live connection test before save, password write-only into the secret store), nextcloud/test, sets (local via the ro backups mount + remote via WebDAV), run + restore (type-to-confirm backstop, source validation) — commands travel as NOTIFY payloads; audit actions backup.settings_changed/run_triggered/restore_requested - readyz: new warning-level backup_remote check while a target is configured (26 h daily / 170 h weekly bound) collab: - maintenance listener: on enter, persist + close every live session and refuse new connections until exit (failsafe timeout 30 min) — no in-memory document may write pre-restore content back afterwards web: - Admin → System backup section: status card with remote facts and a "Back up now" button, the Nextcloud settings form with test button, and the restore picker (local + remote sets, type-to-confirm) - global maintenance screen: any 503 maintenance_mode flips the SPA to a status page polling the exempt endpoint, reloading when the instance returns Verified end-to-end against a live stack (fresh DB, native api + sidecar, fake WebDAV server): configure → test → manual backup → bundle upload → readyz/sets/status surfaces → remote restore with maintenance gate, marker rollback and api restart; suites: shared 21, backup 9, collab 11, api 58 files green, lint + i18n:check + typecheck clean. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EwZ4jR4KFAPvpjWevfUGX1
92 lines
3.1 KiB
TypeScript
92 lines
3.1 KiB
TypeScript
/**
|
|
* A restore set (ADR 0015) is one nightly `pg_dump` plus the matching
|
|
* uploads/plugins archive, tied together by a shared backup id derived from
|
|
* the run's UTC start time. This module owns the naming scheme and the pure
|
|
* prune decision; the sidecar's runner applies it to
|
|
* the filesystem, and the api reads it to list local sets for the in-app
|
|
* restore (issue #103).
|
|
*/
|
|
|
|
export interface BackupSet {
|
|
id: string;
|
|
/** File names (not paths) present in the backups directory. */
|
|
files: string[];
|
|
/** Complete = both the dump and the volume archive exist. */
|
|
complete: boolean;
|
|
}
|
|
|
|
const ID_PATTERN = /^(\d{4})(\d{2})(\d{2})-(\d{2})(\d{2})(\d{2})$/;
|
|
const SET_FILE_PATTERN = /^(?:db-|files-)(\d{8}-\d{6})\.(?:dump|tar\.gz)$/;
|
|
|
|
/** Backup id for a run starting now: UTC timestamp, filesystem-safe. */
|
|
export function newBackupId(now: Date): string {
|
|
const pad = (value: number): string => String(value).padStart(2, '0');
|
|
return (
|
|
`${now.getUTCFullYear()}${pad(now.getUTCMonth() + 1)}${pad(now.getUTCDate())}` +
|
|
`-${pad(now.getUTCHours())}${pad(now.getUTCMinutes())}${pad(now.getUTCSeconds())}`
|
|
);
|
|
}
|
|
|
|
/** The UTC time encoded in a backup id, or null for a malformed id. */
|
|
export function backupIdTime(id: string): Date | null {
|
|
const match = ID_PATTERN.exec(id);
|
|
if (!match) return null;
|
|
const [, year, month, day, hour, minute, second] = match;
|
|
return new Date(
|
|
Date.UTC(
|
|
Number(year),
|
|
Number(month) - 1,
|
|
Number(day),
|
|
Number(hour),
|
|
Number(minute),
|
|
Number(second),
|
|
),
|
|
);
|
|
}
|
|
|
|
export function dumpFileName(id: string): string {
|
|
return `db-${id}.dump`;
|
|
}
|
|
|
|
export function archiveFileName(id: string): string {
|
|
return `files-${id}.tar.gz`;
|
|
}
|
|
|
|
/**
|
|
* Groups the backup directory's file names into sets, oldest first. Files
|
|
* that do not belong to the naming scheme (status.json, `.partial` staging
|
|
* files of a running or crashed run) are ignored — prune never touches them.
|
|
*/
|
|
export function listSets(fileNames: string[]): BackupSet[] {
|
|
const byId = new Map<string, string[]>();
|
|
for (const name of fileNames) {
|
|
const match = SET_FILE_PATTERN.exec(name);
|
|
if (!match || !backupIdTime(match[1]!)) continue;
|
|
const files = byId.get(match[1]!) ?? [];
|
|
files.push(name);
|
|
byId.set(match[1]!, files);
|
|
}
|
|
return [...byId.entries()]
|
|
.sort(([a], [b]) => a.localeCompare(b))
|
|
.map(([id, files]) => ({
|
|
id,
|
|
files: files.sort(),
|
|
complete: files.includes(dumpFileName(id)) && files.includes(archiveFileName(id)),
|
|
}));
|
|
}
|
|
|
|
/**
|
|
* The sets prune may delete: older than the retention cutoff — but never
|
|
* the newest complete set, even when it is expired. A stalled instance must
|
|
* always keep one restorable set (issue #83 acceptance criteria).
|
|
*/
|
|
export function expiredSets(sets: BackupSet[], now: Date, retentionDays: number): BackupSet[] {
|
|
const cutoff = now.getTime() - retentionDays * 24 * 60 * 60 * 1000;
|
|
const newestComplete = [...sets].reverse().find((set) => set.complete);
|
|
return sets.filter((set) => {
|
|
if (set === newestComplete) return false;
|
|
const time = backupIdTime(set.id);
|
|
return time !== null && time.getTime() < cutoff;
|
|
});
|
|
}
|