Add maintenance mode to silence alerts while working on a server or integration

Rebooting Proxmox or patching a server triggered failure/offline alerts
you then had to dismiss. A maintenance window silences alerts about one
server, integration, or DNS provider for a chosen time. New Maintenance
page (start with a duration and optional reason, end early, see what's
silenced and what isn't) and a banner in the app shell so every signed-in
user can see what is currently silenced. Starting/ending is operator-only
and audit-logged; starting one on a target that already has a window
restarts its clock instead of stacking.

Silenced for the target: server offline/disk alerts, Proxmox/Synology
storage and health alerts, Proxmox backup alerts, and "integration down"
alerts. Not silenced: expiry and update reminders, DNS change notices.

The design goal is that this cannot hide a real outage:
- Every window has a required end (5 min to 7 days); there is no
  open-ended option, so a forgotten window expires by itself.
- A silenced problem is deliberately NOT recorded as "known". If it is
  still present when the window ends it alerts then, as new. A problem
  that was already alerted before the window stays known, so it isn't
  repeated, and is reported cleared only after the window ends.
- Failure alerts keep counting failures during a window without marking
  themselves alerted, so an outage that outlasts the window alerts on the
  very next failed call.

Known limitation, stated on the page: integration-failure alerts are
tracked per service TYPE (all "proxmox"), not per configured instance, so
a window on one Proxmox integration also silences a failure on a second
Proxmox integration while it's open. Fixing that means threading the
integration id through every adapter and the diagnostic log, which is a
much larger change than this feature.

Also moved the API-error-message helper out of Secrets.tsx into a shared
util now that two pages use it. New table maintenance_windows (migration
0008).

Verified with 44 checks: the condition-key-to-subject mapping (including
server:3 vs server:33), the diff rules with silenced subjects (new problem
not recorded, alerts when the window ends; already-known one carried and
not repeated; clears only after the window), window expiry and
integration/DNS-provider source matching, the failure tracker end to end
against a webhook (silent during a window while an unrelated service still
alerts; outage that outlasts the window alerts on the next failure and
only once; fail-and-recover fully inside a window sends nothing), a full
health pass against a real window, and the real router with a stubbed
session (role rules, duration bounds including the missing-duration case,
extend-not-stack, 404s, deleted targets hidden, audit entries). Real dev
database mtime untouched.

Not done: I haven't clicked through the new page or banner in a browser
(they sit behind the Authentik login); it builds and the API behind it is
tested.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
bobbanandClaude Sonnet 5 committed 2026-09-26 02:23:06 +02:00
1 parent 1688de3ea2
commit aae4f0d74f
18 files changed
+1854 -26

No files matched your search

+17 -5
View File
@@ -6,6 +6,7 @@ import { createProxmoxAdapter, type ProxmoxNodeStats } from "../integrations/pro
import { createSynologyAdapter, type SynologyStorageInfo } from "../integrations/synology/adapter.js";
import { notifyHealthIssues, notifyHealthRecovered } from "./notify.js";
import { getInternalFlag, getSettings, setInternalFlag } from "./settingsStore.js";
import { activeSubjects, subjectOfConditionKey } from "./maintenance.js";
const STATE_FLAG = "healthActiveConditions";
/** After a restart the agents haven't had a chance to report yet (they run every 15 min), so server-derived conditions are held rather than judged. */
@@ -216,15 +217,26 @@ export function diffConditions(
previous: ActiveState,
current: HealthCondition[],
held: Set<string> = new Set(),
/** Subjects ("server:3", "integration:1") under an active maintenance window. */
silenced: Set<string> = new Set(),
): { added: HealthCondition[]; resolved: { key: string; message: string }[]; next: ActiveState } {
const next: ActiveState = {};
for (const c of current) next[c.key] = { message: c.message, source: c.source };
const isSilenced = (key: string) => {
const subject = subjectOfConditionKey(key);
return subject !== null && silenced.has(subject);
};
// A silenced problem is not recorded as "known": if it's still there when the window ends it must
// alert then, as new. (One that was already alerted before the window stays known, so it isn't repeated.)
const visible = current.filter((c) => !isSilenced(c.key));
const added = current.filter((c) => !(c.key in previous));
const next: ActiveState = {};
for (const c of visible) next[c.key] = { message: c.message, source: c.source };
const added = visible.filter((c) => !(c.key in previous));
const resolved: { key: string; message: string }[] = [];
for (const [key, entry] of Object.entries(previous)) {
if (key in next) continue;
if (isHeld(entry.source, held)) {
// Held (source unreadable) or silenced: leave it exactly as it was — neither cleared nor re-alerted.
if (isHeld(entry.source, held) || isSilenced(key)) {
next[key] = entry;
} else {
resolved.push({ key, message: entry.message });
@@ -296,7 +308,7 @@ export async function runHealthCheck(now: number = Date.now()): Promise<{ added:
if (inGrace) held.add("server");
const current = evaluateHealth(snapshot, healthChecks, now, { skipServers: inGrace });
const { added, resolved, next } = diffConditions(await loadState(), current, held);
const { added, resolved, next } = diffConditions(await loadState(), current, held, await activeSubjects(new Date(now)));
await setInternalFlag(STATE_FLAG, JSON.stringify(next));
// State is tracked even when the alert toggle is off (the notify functions check it),
@@ -1,4 +1,5 @@
import { getSettings } from "./settingsStore.js";
import { isSourceInMaintenance } from "./maintenance.js";
import { notifyIntegrationDown, notifyIntegrationRecovered } from "./notify.js";
interface SourceHealth {
@@ -31,8 +32,12 @@ export async function trackIntegrationHealth(source: string, ok: boolean): Promi
state.consecutiveFailures += 1;
const { notifications } = await getSettings();
if (notifications.integrationFailureAlerts && !state.alerted && state.consecutiveFailures >= notifications.integrationFailureThreshold) {
state.alerted = true;
await notifyIntegrationDown(source, state.consecutiveFailures);
// During maintenance the failures keep being counted but `alerted` stays false, so if the service is
// still failing once the window ends, the very next failed call alerts — a real outage isn't swallowed.
if (!(await isSourceInMaintenance(source))) {
state.alerted = true;
await notifyIntegrationDown(source, state.consecutiveFailures);
}
}
health.set(source, state);
}
+71
View File
@@ -0,0 +1,71 @@
import { eq, gt } from "drizzle-orm";
import { db } from "../db/client.js";
import { dnsProviders, integrations, maintenanceWindows, servers } from "../db/schema.js";
export type ActiveWindow = typeof maintenanceWindows.$inferSelect;
/** Windows whose end is still in the future. ISO strings compare correctly as text, so this is a plain string comparison. */
export async function listActiveWindows(now: Date = new Date()): Promise<ActiveWindow[]> {
return db.select().from(maintenanceWindows).where(gt(maintenanceWindows.endsAt, now.toISOString()));
}
/**
* The identity a health condition or alert belongs to, as "server:3" /
* "integration:1" / "dns_provider:2", so it can be matched against active windows.
* Derived from the condition key (which already encodes it) rather than stored,
* so conditions persisted by an earlier version still map correctly.
*/
export function subjectOfConditionKey(key: string): string | null {
let m = /^(?:offline|disk):server:(\d+)(?::|$)/.exec(key);
if (m) return `server:${m[1]}`;
m = /^disk:proxmox:(\d+):/.exec(key);
if (m) return `integration:${m[1]}`;
m = /^(?:synology-volume|synology-disk|disk:synology):(\d+):/.exec(key);
if (m) return `integration:${m[1]}`;
return null;
}
export async function activeSubjects(now: Date = new Date()): Promise<Set<string>> {
return new Set((await listActiveWindows(now)).map((w) => `${w.targetType}:${w.targetId}`));
}
export async function isInMaintenance(targetType: ActiveWindow["targetType"], targetId: number, now: Date = new Date()): Promise<boolean> {
return (await activeSubjects(now)).has(`${targetType}:${targetId}`);
}
/**
* Integration-failure alerts are tracked per service TYPE ("proxmox",
* "cloudflare", ...), not per configured instance, so a window on any
* integration or DNS provider of that type silences that type's failure alerts.
* (With two integrations of one type, a real failure on the un-windowed one is
* silenced too while the window is open — a known consequence of that
* granularity, stated on the Maintenance page.)
*/
export async function isSourceInMaintenance(source: string, now: Date = new Date()): Promise<boolean> {
const windows = await listActiveWindows(now);
if (windows.length === 0) return false;
for (const w of windows) {
if (w.targetType === "integration") {
const [row] = await db.select({ type: integrations.type }).from(integrations).where(eq(integrations.id, w.targetId)).limit(1);
if (row?.type === source) return true;
} else if (w.targetType === "dns_provider") {
const [row] = await db.select({ type: dnsProviders.providerType }).from(dnsProviders).where(eq(dnsProviders.id, w.targetId)).limit(1);
if (row?.type === source) return true;
}
}
return false;
}
/** Display name for a window's target, or null if the target no longer exists. */
export async function describeTarget(targetType: ActiveWindow["targetType"], targetId: number): Promise<{ name: string; kind: string } | null> {
if (targetType === "server") {
const [r] = await db.select({ name: servers.name }).from(servers).where(eq(servers.id, targetId)).limit(1);
return r ? { name: r.name, kind: "Server" } : null;
}
if (targetType === "integration") {
const [r] = await db.select({ name: integrations.name, type: integrations.type }).from(integrations).where(eq(integrations.id, targetId)).limit(1);
return r ? { name: r.name, kind: `Integration (${r.type})` } : null;
}
const [r] = await db.select({ name: dnsProviders.name, type: dnsProviders.providerType }).from(dnsProviders).where(eq(dnsProviders.id, targetId)).limit(1);
return r ? { name: r.name, kind: `DNS provider (${r.type})` } : null;
}
@@ -6,6 +6,7 @@ import { loadIntegrationConfig } from "../integrations/loadIntegration.js";
import { createProxmoxAdapter, guestsWithoutBackupCoverage, type ProxmoxBackupTask } from "../integrations/proxmox/adapter.js";
import { notifyProxmoxBackupFailure, notifyProxmoxUncoveredGuests } from "./notify.js";
import { getSettings, getInternalFlag, setInternalFlag } from "./settingsStore.js";
import { isInMaintenance } from "./maintenance.js";
const LAST_RUN_FLAG = "proxmoxBackupCheckLastRunDate";
@@ -19,6 +20,8 @@ async function checkProxmoxBackups(): Promise<void> {
const uncovered: { integrationName: string; guestName: string; vmid: number; node: string }[] = [];
for (const row of rows) {
// A host being worked on can't run or report backups; this check is daily, so tomorrow's pass covers it.
if (await isInMaintenance("integration", row.id)) continue;
try {
const loaded = await loadIntegrationConfig(row.id);
if (!loaded) continue;