Add maintenance mode to silence alerts while working on a server or integration

Rebooting Proxmox or patching a server triggered failure/offline alerts
you then had to dismiss. A maintenance window silences alerts about one
server, integration, or DNS provider for a chosen time. New Maintenance
page (start with a duration and optional reason, end early, see what's
silenced and what isn't) and a banner in the app shell so every signed-in
user can see what is currently silenced. Starting/ending is operator-only
and audit-logged; starting one on a target that already has a window
restarts its clock instead of stacking.

Silenced for the target: server offline/disk alerts, Proxmox/Synology
storage and health alerts, Proxmox backup alerts, and "integration down"
alerts. Not silenced: expiry and update reminders, DNS change notices.

The design goal is that this cannot hide a real outage:
- Every window has a required end (5 min to 7 days); there is no
  open-ended option, so a forgotten window expires by itself.
- A silenced problem is deliberately NOT recorded as "known". If it is
  still present when the window ends it alerts then, as new. A problem
  that was already alerted before the window stays known, so it isn't
  repeated, and is reported cleared only after the window ends.
- Failure alerts keep counting failures during a window without marking
  themselves alerted, so an outage that outlasts the window alerts on the
  very next failed call.

Known limitation, stated on the page: integration-failure alerts are
tracked per service TYPE (all "proxmox"), not per configured instance, so
a window on one Proxmox integration also silences a failure on a second
Proxmox integration while it's open. Fixing that means threading the
integration id through every adapter and the diagnostic log, which is a
much larger change than this feature.

Also moved the API-error-message helper out of Secrets.tsx into a shared
util now that two pages use it. New table maintenance_windows (migration
0008).

Verified with 44 checks: the condition-key-to-subject mapping (including
server:3 vs server:33), the diff rules with silenced subjects (new problem
not recorded, alerts when the window ends; already-known one carried and
not repeated; clears only after the window), window expiry and
integration/DNS-provider source matching, the failure tracker end to end
against a webhook (silent during a window while an unrelated service still
alerts; outage that outlasts the window alerts on the next failure and
only once; fail-and-recover fully inside a window sends nothing), a full
health pass against a real window, and the real router with a stubbed
session (role rules, duration bounds including the missing-duration case,
extend-not-stack, 404s, deleted targets hidden, audit entries). Real dev
database mtime untouched.

Not done: I haven't clicked through the new page or banner in a browser
(they sit behind the Authentik login); it builds and the API behind it is
tested.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
bobbanandClaude Sonnet 5 committed 2026-09-26 02:23:06 +02:00
1 parent 1688de3ea2
commit aae4f0d74f
18 files changed
+1854 -26

No files matched your search

+101
View File
@@ -0,0 +1,101 @@
import { Router } from "express";
import { eq, lt } from "drizzle-orm";
import { z } from "zod";
import { db } from "../db/client.js";
import { dnsProviders, integrations, maintenanceTargetTypes, maintenanceWindows, servers } from "../db/schema.js";
import { requireAuth, requireRole } from "../auth/middleware.js";
import { recordAudit } from "../services/audit.js";
import { describeTarget, listActiveWindows } from "../services/maintenance.js";
import { asyncHandler } from "../utils/asyncHandler.js";
export const maintenanceRouter = Router();
// Anyone signed in can see what's currently silenced (so nobody is surprised by missing alerts);
// starting and ending a window is an operator action, like the other things that change how the homelab behaves.
maintenanceRouter.use(requireAuth);
maintenanceRouter.get("/", asyncHandler(async (_req, res) => {
const windows = [];
for (const w of await listActiveWindows()) {
const target = await describeTarget(w.targetType, w.targetId);
if (!target) continue; // target was deleted — nothing left to silence
windows.push({ ...w, targetName: target.name, targetKind: target.kind });
}
windows.sort((a, b) => a.endsAt.localeCompare(b.endsAt));
const [serverRows, integrationRows, providerRows] = await Promise.all([
db.select({ id: servers.id, name: servers.name }).from(servers).orderBy(servers.name),
db.select({ id: integrations.id, name: integrations.name, type: integrations.type }).from(integrations).orderBy(integrations.name),
db.select({ id: dnsProviders.id, name: dnsProviders.name, providerType: dnsProviders.providerType }).from(dnsProviders).orderBy(dnsProviders.name),
]);
res.json({ windows, targets: { servers: serverRows, integrations: integrationRows, dnsProviders: providerRows } });
}));
const startSchema = z.object({
targetType: z.enum(maintenanceTargetTypes),
targetId: z.number().int().positive(),
// Required and capped: an open-ended window is the way this feature could quietly hide a real outage.
minutes: z.number().int().min(5).max(7 * 24 * 60),
reason: z.string().max(200).optional(),
});
maintenanceRouter.post("/", requireRole("operator"), asyncHandler(async (req, res) => {
const parsed = startSchema.safeParse(req.body);
if (!parsed.success) {
return res.status(400).json({ error: "invalid_body", details: parsed.error.flatten() });
}
const { targetType, targetId, minutes, reason } = parsed.data;
const target = await describeTarget(targetType, targetId);
if (!target) {
return res.status(404).json({ error: "not_found", message: "That target doesn't exist." });
}
const now = new Date();
const endsAt = new Date(now.getTime() + minutes * 60_000).toISOString();
const actor = req.currentUser!;
const createdBy = actor.name ?? actor.email ?? actor.oidcSub;
// Starting maintenance on something already in maintenance restarts its clock rather than stacking windows.
const existing = (await listActiveWindows(now)).find((w) => w.targetType === targetType && w.targetId === targetId);
let window;
if (existing) {
[window] = await db.update(maintenanceWindows).set({ endsAt, reason: reason ?? null, createdBy }).where(eq(maintenanceWindows.id, existing.id)).returning();
} else {
// Long-expired rows are just clutter; clear them out whenever a new one is added.
await db.delete(maintenanceWindows).where(lt(maintenanceWindows.endsAt, new Date(now.getTime() - 24 * 3600_000).toISOString()));
[window] = await db.insert(maintenanceWindows).values({ targetType, targetId, reason: reason ?? null, startedAt: now.toISOString(), endsAt, createdBy }).returning();
}
await recordAudit({
actor,
category: "maintenance",
action: existing ? "extend" : "start",
targetType,
targetId,
detail: { name: target.name, minutes, reason: reason ?? null },
});
res.status(201).json({ window: { ...window, targetName: target.name, targetKind: target.kind } });
}));
maintenanceRouter.delete("/:id", requireRole("operator"), asyncHandler(async (req, res) => {
const id = Number(req.params.id);
const [existing] = await db.select().from(maintenanceWindows).where(eq(maintenanceWindows.id, id)).limit(1);
if (!existing) {
return res.status(404).json({ error: "not_found" });
}
// Ending moves the end to now, so it stops applying immediately; the row is pruned later like any expired one.
await db.update(maintenanceWindows).set({ endsAt: new Date().toISOString() }).where(eq(maintenanceWindows.id, id));
const target = await describeTarget(existing.targetType, existing.targetId);
await recordAudit({
actor: req.currentUser!,
category: "maintenance",
action: "end",
targetType: existing.targetType,
targetId: existing.targetId,
detail: { name: target?.name ?? null },
});
res.status(204).end();
}));