All checks were successful
CI / Lint, typecheck, test (push) Successful in 3m45s
CD / Build and push images (push) Successful in 3m49s
CI / Build container images (push) Has been skipped
CD / Deploy to Test (push) Successful in 9s
CD / Smoke tests against Test (push) Successful in 1m18s
CD / Promote to Int (push) Successful in 11s
CI / Auth e2e pack (push) Successful in 5m35s
CI / Import/export fidelity gate (push) Successful in 47s
Off-host backups for every self-hoster, configured entirely in the admin UI — supersedes the host-specific mirror plan behind #84. shared: - webdav.ts (new package entry like token-crypto): minimal WebDAV client with basic auth — PROPFIND (tolerant multistatus parser), MKCOL, PUT (streamed), GET, DELETE; Nextcloud DAV path derived from the plain server URL, explicit DAV bases pass through - backup-status.ts: additive remote-upload status in status.json, the restore-status.json contract (running/succeeded/failed + staleness bound), the backup_command/backup_maintenance NOTIFY channels, and the one-bundle-per-set naming (dorfteich-backup-<id>.tar.gz) - backup-set.ts moved here from apps/backup (api lists local sets) backup sidecar: - reads the backup.* instance settings directly from the database (admin changes apply next run; local retention row overrides the env) and the app password from the secret store - after each successful set: bundle dump + files archive + manifest into ONE self-contained tar.gz, upload via WebDAV per schedule (off/daily/weekly; manual runs always upload), prune remote bundles — never the newest — and record the outcome in status.json; upload failures alert via a new backupUploadFailed mail (de+en) - command listener on backup_command (run / restore) with a serial queue against the nightly timer - restore orchestrator: restore-status.json → maintenance NOTIFY → grace → (remote: download + manifest-verify bundle) → terminate other DB connections → shared perform-restore path (same code as restore.sh) → final status + maintenance exit api: - MaintenanceGuard (global, registered before the setup gate): 503 maintenance_mode while restore-status says running; health endpoints and the new public GET /backup/restore-status stay exempt; a stale running state (crashed sidecar) unblocks after 30 min - MaintenanceStateService watches the file and restarts the api after a successful restore (fresh caches, migrate-on-start for older dumps); main.ts refuses to touch the database while a restore runs — a container restarting mid-restore must not race pg_restore with migrate deploy - worker sweeps (conversion, mail outbox, scheduler) catch transient database failures instead of dying on an unhandled rejection — the restore's connection termination crashed the api in verification - backup admin endpoints under /admin/system/backup: settings (live connection test before save, password write-only into the secret store), nextcloud/test, sets (local via the ro backups mount + remote via WebDAV), run + restore (type-to-confirm backstop, source validation) — commands travel as NOTIFY payloads; audit actions backup.settings_changed/run_triggered/restore_requested - readyz: new warning-level backup_remote check while a target is configured (26 h daily / 170 h weekly bound) collab: - maintenance listener: on enter, persist + close every live session and refuse new connections until exit (failsafe timeout 30 min) — no in-memory document may write pre-restore content back afterwards web: - Admin → System backup section: status card with remote facts and a "Back up now" button, the Nextcloud settings form with test button, and the restore picker (local + remote sets, type-to-confirm) - global maintenance screen: any 503 maintenance_mode flips the SPA to a status page polling the exempt endpoint, reloading when the instance returns Verified end-to-end against a live stack (fresh DB, native api + sidecar, fake WebDAV server): configure → test → manual backup → bundle upload → readyz/sets/status surfaces → remote restore with maintenance gate, marker rollback and api restart; suites: shared 21, backup 9, collab 11, api 58 files green, lint + i18n:check + typecheck clean. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EwZ4jR4KFAPvpjWevfUGX1
271 lines
8.4 KiB
TypeScript
271 lines
8.4 KiB
TypeScript
import { mkdtempSync, rmSync, writeFileSync } from 'node:fs';
|
|
import { tmpdir } from 'node:os';
|
|
import { join } from 'node:path';
|
|
|
|
import { BACKUP_STATUS_FILE, type BackupStatus } from '@dorfteich/shared';
|
|
import { afterEach, beforeEach, describe, expect, it } from 'vitest';
|
|
|
|
import { backupFreshnessCheck, backupRemoteFreshnessCheck } from './backup-freshness';
|
|
import { ReadinessService } from './readiness.service';
|
|
|
|
import type { BackupTargetService } from '../backup/backup-target.service';
|
|
import type { AppConfig } from '../config/app-config.service';
|
|
import type { PrismaService } from '../prisma/prisma.service';
|
|
import type { InstanceSettingsService } from '../settings/instance-settings.service';
|
|
|
|
const NOW = new Date('2026-07-12T12:00:00Z');
|
|
const HOUR = 3_600_000;
|
|
|
|
let dir: string;
|
|
|
|
function writeStatus(overrides: Partial<BackupStatus>): void {
|
|
const finishedAt = new Date(NOW.getTime() - 3 * HOUR).toISOString();
|
|
const status: BackupStatus = {
|
|
schemaVersion: 1,
|
|
updatedAt: finishedAt,
|
|
retentionDays: 7,
|
|
lastRun: {
|
|
backupId: '20260712-090000',
|
|
startedAt: finishedAt,
|
|
finishedAt,
|
|
durationMs: 1000,
|
|
outcome: 'succeeded',
|
|
sizes: { dumpBytes: 1, archiveBytes: 1 },
|
|
},
|
|
lastSuccess: {
|
|
backupId: '20260712-090000',
|
|
finishedAt,
|
|
sizes: { dumpBytes: 1, archiveBytes: 1 },
|
|
},
|
|
...overrides,
|
|
};
|
|
writeFileSync(join(dir, BACKUP_STATUS_FILE), JSON.stringify(status));
|
|
}
|
|
|
|
beforeEach(() => {
|
|
dir = mkdtempSync(join(tmpdir(), 'dorfteich-freshness-'));
|
|
});
|
|
|
|
afterEach(() => {
|
|
rmSync(dir, { recursive: true, force: true });
|
|
});
|
|
|
|
describe('backupFreshnessCheck', () => {
|
|
it('warns when no status was ever recorded', () => {
|
|
const check = backupFreshnessCheck(join(dir, 'nowhere'), NOW);
|
|
expect(check).toMatchObject({ name: 'backup', status: 'warn' });
|
|
expect(check.detail).toContain('no backup status');
|
|
});
|
|
|
|
it('warns on an unreadable status file', () => {
|
|
writeFileSync(join(dir, BACKUP_STATUS_FILE), '{torn');
|
|
expect(backupFreshnessCheck(dir, NOW).status).toBe('warn');
|
|
});
|
|
|
|
it('warns when no backup ever succeeded, with the last error', () => {
|
|
writeStatus({
|
|
lastSuccess: null,
|
|
lastRun: {
|
|
backupId: '20260712-090000',
|
|
startedAt: NOW.toISOString(),
|
|
finishedAt: NOW.toISOString(),
|
|
durationMs: 5,
|
|
outcome: 'failed',
|
|
error: 'pg_dump failed: connection refused',
|
|
},
|
|
});
|
|
const check = backupFreshnessCheck(dir, NOW);
|
|
expect(check.status).toBe('warn');
|
|
expect(check.detail).toContain('connection refused');
|
|
});
|
|
|
|
it('warns when the last success is older than 26 h', () => {
|
|
const old = new Date(NOW.getTime() - 30 * HOUR).toISOString();
|
|
writeStatus({
|
|
lastSuccess: {
|
|
backupId: '20260711-060000',
|
|
finishedAt: old,
|
|
sizes: { dumpBytes: 1, archiveBytes: 1 },
|
|
},
|
|
});
|
|
const check = backupFreshnessCheck(dir, NOW);
|
|
expect(check.status).toBe('warn');
|
|
expect(check.detail).toContain('30 h old (max 26 h)');
|
|
});
|
|
|
|
it('is ok on a fresh success', () => {
|
|
writeStatus({});
|
|
expect(backupFreshnessCheck(dir, NOW)).toEqual({ name: 'backup', status: 'ok' });
|
|
});
|
|
|
|
it('stays ok but reports a failed run newer than the fresh success', () => {
|
|
writeStatus({
|
|
lastRun: {
|
|
backupId: '20260712-110000',
|
|
startedAt: NOW.toISOString(),
|
|
finishedAt: NOW.toISOString(),
|
|
durationMs: 5,
|
|
outcome: 'failed',
|
|
error: 'disk full',
|
|
},
|
|
});
|
|
const check = backupFreshnessCheck(dir, NOW);
|
|
expect(check.status).toBe('ok');
|
|
expect(check.detail).toContain('disk full');
|
|
});
|
|
});
|
|
|
|
describe('ReadinessService report (issue #85 degraded semantics)', () => {
|
|
// Fakes: the database probe is the only prisma call the report makes.
|
|
const prismaUp = { $queryRaw: async () => [{ pending: 0n }] } as unknown as PrismaService;
|
|
const prismaDown = {
|
|
$queryRaw: async () => {
|
|
throw new Error('connection refused');
|
|
},
|
|
} as unknown as PrismaService;
|
|
|
|
const noTarget = { resolveTarget: async () => null } as unknown as BackupTargetService;
|
|
const configuredTarget = {
|
|
resolveTarget: async () => ({
|
|
baseUrl: 'https://cloud.example.com',
|
|
username: 'u',
|
|
password: 'p',
|
|
folder: 'f',
|
|
}),
|
|
} as unknown as BackupTargetService;
|
|
const dailySettings = { get: async () => 'daily' } as unknown as InstanceSettingsService;
|
|
|
|
function configWith(backupsDir: string): AppConfig {
|
|
return {
|
|
env: {
|
|
// Unreachable sidecars: converter/renderer report warn — irrelevant
|
|
// here, the assertions pin the database/migrations/backup checks.
|
|
PANDOC_URL: 'http://127.0.0.1:59998',
|
|
GOTENBERG_URL: 'http://127.0.0.1:59998',
|
|
BACKUPS_DIR: backupsDir,
|
|
},
|
|
} as AppConfig;
|
|
}
|
|
|
|
it('reports degraded (not unready) on a stale backup with healthy hard checks', async () => {
|
|
const old = new Date(Date.now() - 40 * HOUR).toISOString();
|
|
writeStatus({
|
|
lastSuccess: {
|
|
backupId: '20260710-030000',
|
|
finishedAt: old,
|
|
sizes: { dumpBytes: 1, archiveBytes: 1 },
|
|
},
|
|
});
|
|
const report = await new ReadinessService(
|
|
prismaUp,
|
|
configWith(dir),
|
|
noTarget,
|
|
dailySettings,
|
|
).report();
|
|
|
|
expect(report.status).toBe('degraded');
|
|
const byName = Object.fromEntries(report.checks.map((c) => [c.name, c.status]));
|
|
expect(byName).toMatchObject({ database: 'ok', migrations: 'ok', backup: 'warn' });
|
|
});
|
|
|
|
it('reports unready only on hard failures', async () => {
|
|
const report = await new ReadinessService(
|
|
prismaDown,
|
|
configWith(dir),
|
|
noTarget,
|
|
dailySettings,
|
|
).report();
|
|
expect(report.status).toBe('unready');
|
|
expect(report.checks.find((c) => c.name === 'database')?.status).toBe('failed');
|
|
});
|
|
|
|
it('adds the off-host check only while a target is configured', async () => {
|
|
writeStatus({});
|
|
const withoutTarget = await new ReadinessService(
|
|
prismaUp,
|
|
configWith(dir),
|
|
noTarget,
|
|
dailySettings,
|
|
).report();
|
|
expect(withoutTarget.checks.some((c) => c.name === 'backup_remote')).toBe(false);
|
|
|
|
const withTarget = await new ReadinessService(
|
|
prismaUp,
|
|
configWith(dir),
|
|
configuredTarget,
|
|
dailySettings,
|
|
).report();
|
|
const remote = withTarget.checks.find((c) => c.name === 'backup_remote');
|
|
// status.json has no remote section yet → the copy is missing → warn.
|
|
expect(remote).toMatchObject({ status: 'warn' });
|
|
expect(withTarget.status).toBe('degraded');
|
|
});
|
|
});
|
|
|
|
describe('backupRemoteFreshnessCheck (issue #103)', () => {
|
|
const fresh = new Date(NOW.getTime() - 3 * HOUR).toISOString();
|
|
|
|
it('is ok (with detail) when the schedule is manual-only', () => {
|
|
expect(backupRemoteFreshnessCheck(dir, NOW, 'off')).toMatchObject({
|
|
name: 'backup_remote',
|
|
status: 'ok',
|
|
});
|
|
});
|
|
|
|
it('warns when no upload ever succeeded, with the last error', () => {
|
|
writeStatus({
|
|
remote: {
|
|
lastUpload: {
|
|
backupId: '20260712-090000',
|
|
finishedAt: fresh,
|
|
outcome: 'failed',
|
|
error: 'authentication failed',
|
|
},
|
|
lastSuccessfulUpload: null,
|
|
},
|
|
});
|
|
const check = backupRemoteFreshnessCheck(dir, NOW, 'daily');
|
|
expect(check.status).toBe('warn');
|
|
expect(check.detail).toContain('authentication failed');
|
|
});
|
|
|
|
it('applies the weekly bound to weekly schedules', () => {
|
|
const sixDaysOld = new Date(NOW.getTime() - 6 * 24 * HOUR).toISOString();
|
|
writeStatus({
|
|
remote: {
|
|
lastUpload: {
|
|
backupId: '20260706-090000',
|
|
finishedAt: sixDaysOld,
|
|
outcome: 'succeeded',
|
|
sizeBytes: 5,
|
|
},
|
|
lastSuccessfulUpload: {
|
|
backupId: '20260706-090000',
|
|
finishedAt: sixDaysOld,
|
|
sizeBytes: 5,
|
|
},
|
|
},
|
|
});
|
|
expect(backupRemoteFreshnessCheck(dir, NOW, 'weekly').status).toBe('ok');
|
|
expect(backupRemoteFreshnessCheck(dir, NOW, 'daily').status).toBe('warn');
|
|
});
|
|
|
|
it('is ok on a fresh upload', () => {
|
|
writeStatus({
|
|
remote: {
|
|
lastUpload: {
|
|
backupId: '20260712-090000',
|
|
finishedAt: fresh,
|
|
outcome: 'succeeded',
|
|
sizeBytes: 5,
|
|
},
|
|
lastSuccessfulUpload: { backupId: '20260712-090000', finishedAt: fresh, sizeBytes: 5 },
|
|
},
|
|
});
|
|
expect(backupRemoteFreshnessCheck(dir, NOW, 'daily')).toEqual({
|
|
name: 'backup_remote',
|
|
status: 'ok',
|
|
});
|
|
});
|
|
});
|