Add maintenance mode to silence alerts while working on a server or integration
Rebooting Proxmox or patching a server triggered failure/offline alerts you then had to dismiss. A maintenance window silences alerts about one server, integration, or DNS provider for a chosen time. New Maintenance page (start with a duration and optional reason, end early, see what's silenced and what isn't) and a banner in the app shell so every signed-in user can see what is currently silenced. Starting/ending is operator-only and audit-logged; starting one on a target that already has a window restarts its clock instead of stacking. Silenced for the target: server offline/disk alerts, Proxmox/Synology storage and health alerts, Proxmox backup alerts, and "integration down" alerts. Not silenced: expiry and update reminders, DNS change notices. The design goal is that this cannot hide a real outage: - Every window has a required end (5 min to 7 days); there is no open-ended option, so a forgotten window expires by itself. - A silenced problem is deliberately NOT recorded as "known". If it is still present when the window ends it alerts then, as new. A problem that was already alerted before the window stays known, so it isn't repeated, and is reported cleared only after the window ends. - Failure alerts keep counting failures during a window without marking themselves alerted, so an outage that outlasts the window alerts on the very next failed call. Known limitation, stated on the page: integration-failure alerts are tracked per service TYPE (all "proxmox"), not per configured instance, so a window on one Proxmox integration also silences a failure on a second Proxmox integration while it's open. Fixing that means threading the integration id through every adapter and the diagnostic log, which is a much larger change than this feature. Also moved the API-error-message helper out of Secrets.tsx into a shared util now that two pages use it. New table maintenance_windows (migration 0008). Verified with 44 checks: the condition-key-to-subject mapping (including server:3 vs server:33), the diff rules with silenced subjects (new problem not recorded, alerts when the window ends; already-known one carried and not repeated; clears only after the window), window expiry and integration/DNS-provider source matching, the failure tracker end to end against a webhook (silent during a window while an unrelated service still alerts; outage that outlasts the window alerts on the next failure and only once; fail-and-recover fully inside a window sends nothing), a full health pass against a real window, and the real router with a stubbed session (role rules, duration bounds including the missing-duration case, extend-not-stack, 404s, deleted targets hidden, audit entries). Real dev database mtime untouched. Not done: I haven't clicked through the new page or banner in a browser (they sit behind the Authentik login); it builds and the API behind it is tested. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
1688de3ea2
commit
aae4f0d74f
18 files changed
+1854
-26
No files matched your search
@@ -0,0 +1,9 @@
|
||||
CREATE TABLE `maintenance_windows` (
|
||||
`id` integer PRIMARY KEY AUTOINCREMENT NOT NULL,
|
||||
`target_type` text NOT NULL,
|
||||
`target_id` integer NOT NULL,
|
||||
`reason` text,
|
||||
`started_at` text NOT NULL,
|
||||
`ends_at` text NOT NULL,
|
||||
`created_by` text
|
||||
);
|
||||
File diff suppressed because it is too large.
Load diff
@@ -57,6 +57,13 @@
|
||||
"when": 1790372043652,
|
||||
"tag": "0007_secret_madripoor",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 8,
|
||||
"version": "6",
|
||||
"when": 1790381917814,
|
||||
"tag": "0008_thin_boom_boom",
|
||||
"breakpoints": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -59,6 +59,21 @@ export const notificationQueue = sqliteTable("notification_queue", {
|
||||
.default(sql`(current_timestamp)`),
|
||||
});
|
||||
|
||||
// ─── Maintenance windows — alerts about a target are silenced until endsAt ──
|
||||
|
||||
export const maintenanceTargetTypes = ["server", "integration", "dns_provider"] as const;
|
||||
export type MaintenanceTargetType = (typeof maintenanceTargetTypes)[number];
|
||||
|
||||
export const maintenanceWindows = sqliteTable("maintenance_windows", {
|
||||
id: integer("id").primaryKey({ autoIncrement: true }),
|
||||
targetType: text("target_type").$type<MaintenanceTargetType>().notNull(),
|
||||
targetId: integer("target_id").notNull(), // polymorphic — no FK; a deleted target's window is simply ignored
|
||||
reason: text("reason"),
|
||||
startedAt: text("started_at").notNull(), // ISO
|
||||
endsAt: text("ends_at").notNull(), // ISO — required: a forgotten open-ended window would silence real problems forever
|
||||
createdBy: text("created_by"),
|
||||
});
|
||||
|
||||
// ─── Settings (key/value) ───────────────────────────────────────────────────
|
||||
|
||||
export const settings = sqliteTable("settings", {
|
||||
|
||||
@@ -23,6 +23,7 @@ import { integrationsRouter } from "./routes/integrations.js";
|
||||
import { settingsRouter } from "./routes/settings.js";
|
||||
import { searchRouter } from "./routes/search.js";
|
||||
import { sessionsRouter } from "./routes/sessions.js";
|
||||
import { maintenanceRouter } from "./routes/maintenance.js";
|
||||
import { initSecretExpiryScheduler } from "./services/secretExpiryScheduler.js";
|
||||
import { initTailscaleKeyExpiryScheduler } from "./services/tailscaleKeyExpiryScheduler.js";
|
||||
import { initLogRetentionScheduler } from "./services/logRetentionScheduler.js";
|
||||
@@ -90,6 +91,7 @@ app.use("/api/integrations", integrationsRouter);
|
||||
app.use("/api/settings", settingsRouter);
|
||||
app.use("/api/search", searchRouter);
|
||||
app.use("/api/sessions", sessionsRouter);
|
||||
app.use("/api/maintenance", maintenanceRouter);
|
||||
|
||||
if (existsSync(webDist)) {
|
||||
app.use(express.static(webDist));
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
import { Router } from "express";
|
||||
import { eq, lt } from "drizzle-orm";
|
||||
import { z } from "zod";
|
||||
import { db } from "../db/client.js";
|
||||
import { dnsProviders, integrations, maintenanceTargetTypes, maintenanceWindows, servers } from "../db/schema.js";
|
||||
import { requireAuth, requireRole } from "../auth/middleware.js";
|
||||
import { recordAudit } from "../services/audit.js";
|
||||
import { describeTarget, listActiveWindows } from "../services/maintenance.js";
|
||||
import { asyncHandler } from "../utils/asyncHandler.js";
|
||||
|
||||
export const maintenanceRouter = Router();
|
||||
|
||||
// Anyone signed in can see what's currently silenced (so nobody is surprised by missing alerts);
|
||||
// starting and ending a window is an operator action, like the other things that change how the homelab behaves.
|
||||
maintenanceRouter.use(requireAuth);
|
||||
|
||||
maintenanceRouter.get("/", asyncHandler(async (_req, res) => {
|
||||
const windows = [];
|
||||
for (const w of await listActiveWindows()) {
|
||||
const target = await describeTarget(w.targetType, w.targetId);
|
||||
if (!target) continue; // target was deleted — nothing left to silence
|
||||
windows.push({ ...w, targetName: target.name, targetKind: target.kind });
|
||||
}
|
||||
windows.sort((a, b) => a.endsAt.localeCompare(b.endsAt));
|
||||
|
||||
const [serverRows, integrationRows, providerRows] = await Promise.all([
|
||||
db.select({ id: servers.id, name: servers.name }).from(servers).orderBy(servers.name),
|
||||
db.select({ id: integrations.id, name: integrations.name, type: integrations.type }).from(integrations).orderBy(integrations.name),
|
||||
db.select({ id: dnsProviders.id, name: dnsProviders.name, providerType: dnsProviders.providerType }).from(dnsProviders).orderBy(dnsProviders.name),
|
||||
]);
|
||||
res.json({ windows, targets: { servers: serverRows, integrations: integrationRows, dnsProviders: providerRows } });
|
||||
}));
|
||||
|
||||
const startSchema = z.object({
|
||||
targetType: z.enum(maintenanceTargetTypes),
|
||||
targetId: z.number().int().positive(),
|
||||
// Required and capped: an open-ended window is the way this feature could quietly hide a real outage.
|
||||
minutes: z.number().int().min(5).max(7 * 24 * 60),
|
||||
reason: z.string().max(200).optional(),
|
||||
});
|
||||
|
||||
maintenanceRouter.post("/", requireRole("operator"), asyncHandler(async (req, res) => {
|
||||
const parsed = startSchema.safeParse(req.body);
|
||||
if (!parsed.success) {
|
||||
return res.status(400).json({ error: "invalid_body", details: parsed.error.flatten() });
|
||||
}
|
||||
const { targetType, targetId, minutes, reason } = parsed.data;
|
||||
|
||||
const target = await describeTarget(targetType, targetId);
|
||||
if (!target) {
|
||||
return res.status(404).json({ error: "not_found", message: "That target doesn't exist." });
|
||||
}
|
||||
|
||||
const now = new Date();
|
||||
const endsAt = new Date(now.getTime() + minutes * 60_000).toISOString();
|
||||
const actor = req.currentUser!;
|
||||
const createdBy = actor.name ?? actor.email ?? actor.oidcSub;
|
||||
|
||||
// Starting maintenance on something already in maintenance restarts its clock rather than stacking windows.
|
||||
const existing = (await listActiveWindows(now)).find((w) => w.targetType === targetType && w.targetId === targetId);
|
||||
let window;
|
||||
if (existing) {
|
||||
[window] = await db.update(maintenanceWindows).set({ endsAt, reason: reason ?? null, createdBy }).where(eq(maintenanceWindows.id, existing.id)).returning();
|
||||
} else {
|
||||
// Long-expired rows are just clutter; clear them out whenever a new one is added.
|
||||
await db.delete(maintenanceWindows).where(lt(maintenanceWindows.endsAt, new Date(now.getTime() - 24 * 3600_000).toISOString()));
|
||||
[window] = await db.insert(maintenanceWindows).values({ targetType, targetId, reason: reason ?? null, startedAt: now.toISOString(), endsAt, createdBy }).returning();
|
||||
}
|
||||
|
||||
await recordAudit({
|
||||
actor,
|
||||
category: "maintenance",
|
||||
action: existing ? "extend" : "start",
|
||||
targetType,
|
||||
targetId,
|
||||
detail: { name: target.name, minutes, reason: reason ?? null },
|
||||
});
|
||||
|
||||
res.status(201).json({ window: { ...window, targetName: target.name, targetKind: target.kind } });
|
||||
}));
|
||||
|
||||
maintenanceRouter.delete("/:id", requireRole("operator"), asyncHandler(async (req, res) => {
|
||||
const id = Number(req.params.id);
|
||||
const [existing] = await db.select().from(maintenanceWindows).where(eq(maintenanceWindows.id, id)).limit(1);
|
||||
if (!existing) {
|
||||
return res.status(404).json({ error: "not_found" });
|
||||
}
|
||||
// Ending moves the end to now, so it stops applying immediately; the row is pruned later like any expired one.
|
||||
await db.update(maintenanceWindows).set({ endsAt: new Date().toISOString() }).where(eq(maintenanceWindows.id, id));
|
||||
|
||||
const target = await describeTarget(existing.targetType, existing.targetId);
|
||||
await recordAudit({
|
||||
actor: req.currentUser!,
|
||||
category: "maintenance",
|
||||
action: "end",
|
||||
targetType: existing.targetType,
|
||||
targetId: existing.targetId,
|
||||
detail: { name: target?.name ?? null },
|
||||
});
|
||||
res.status(204).end();
|
||||
}));
|
||||
@@ -6,6 +6,7 @@ import { createProxmoxAdapter, type ProxmoxNodeStats } from "../integrations/pro
|
||||
import { createSynologyAdapter, type SynologyStorageInfo } from "../integrations/synology/adapter.js";
|
||||
import { notifyHealthIssues, notifyHealthRecovered } from "./notify.js";
|
||||
import { getInternalFlag, getSettings, setInternalFlag } from "./settingsStore.js";
|
||||
import { activeSubjects, subjectOfConditionKey } from "./maintenance.js";
|
||||
|
||||
const STATE_FLAG = "healthActiveConditions";
|
||||
/** After a restart the agents haven't had a chance to report yet (they run every 15 min), so server-derived conditions are held rather than judged. */
|
||||
@@ -216,15 +217,26 @@ export function diffConditions(
|
||||
previous: ActiveState,
|
||||
current: HealthCondition[],
|
||||
held: Set<string> = new Set(),
|
||||
/** Subjects ("server:3", "integration:1") under an active maintenance window. */
|
||||
silenced: Set<string> = new Set(),
|
||||
): { added: HealthCondition[]; resolved: { key: string; message: string }[]; next: ActiveState } {
|
||||
const next: ActiveState = {};
|
||||
for (const c of current) next[c.key] = { message: c.message, source: c.source };
|
||||
const isSilenced = (key: string) => {
|
||||
const subject = subjectOfConditionKey(key);
|
||||
return subject !== null && silenced.has(subject);
|
||||
};
|
||||
// A silenced problem is not recorded as "known": if it's still there when the window ends it must
|
||||
// alert then, as new. (One that was already alerted before the window stays known, so it isn't repeated.)
|
||||
const visible = current.filter((c) => !isSilenced(c.key));
|
||||
|
||||
const added = current.filter((c) => !(c.key in previous));
|
||||
const next: ActiveState = {};
|
||||
for (const c of visible) next[c.key] = { message: c.message, source: c.source };
|
||||
|
||||
const added = visible.filter((c) => !(c.key in previous));
|
||||
const resolved: { key: string; message: string }[] = [];
|
||||
for (const [key, entry] of Object.entries(previous)) {
|
||||
if (key in next) continue;
|
||||
if (isHeld(entry.source, held)) {
|
||||
// Held (source unreadable) or silenced: leave it exactly as it was — neither cleared nor re-alerted.
|
||||
if (isHeld(entry.source, held) || isSilenced(key)) {
|
||||
next[key] = entry;
|
||||
} else {
|
||||
resolved.push({ key, message: entry.message });
|
||||
@@ -296,7 +308,7 @@ export async function runHealthCheck(now: number = Date.now()): Promise<{ added:
|
||||
if (inGrace) held.add("server");
|
||||
|
||||
const current = evaluateHealth(snapshot, healthChecks, now, { skipServers: inGrace });
|
||||
const { added, resolved, next } = diffConditions(await loadState(), current, held);
|
||||
const { added, resolved, next } = diffConditions(await loadState(), current, held, await activeSubjects(new Date(now)));
|
||||
await setInternalFlag(STATE_FLAG, JSON.stringify(next));
|
||||
|
||||
// State is tracked even when the alert toggle is off (the notify functions check it),
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { getSettings } from "./settingsStore.js";
|
||||
import { isSourceInMaintenance } from "./maintenance.js";
|
||||
import { notifyIntegrationDown, notifyIntegrationRecovered } from "./notify.js";
|
||||
|
||||
interface SourceHealth {
|
||||
@@ -31,8 +32,12 @@ export async function trackIntegrationHealth(source: string, ok: boolean): Promi
|
||||
state.consecutiveFailures += 1;
|
||||
const { notifications } = await getSettings();
|
||||
if (notifications.integrationFailureAlerts && !state.alerted && state.consecutiveFailures >= notifications.integrationFailureThreshold) {
|
||||
state.alerted = true;
|
||||
await notifyIntegrationDown(source, state.consecutiveFailures);
|
||||
// During maintenance the failures keep being counted but `alerted` stays false, so if the service is
|
||||
// still failing once the window ends, the very next failed call alerts — a real outage isn't swallowed.
|
||||
if (!(await isSourceInMaintenance(source))) {
|
||||
state.alerted = true;
|
||||
await notifyIntegrationDown(source, state.consecutiveFailures);
|
||||
}
|
||||
}
|
||||
health.set(source, state);
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
import { eq, gt } from "drizzle-orm";
|
||||
import { db } from "../db/client.js";
|
||||
import { dnsProviders, integrations, maintenanceWindows, servers } from "../db/schema.js";
|
||||
|
||||
export type ActiveWindow = typeof maintenanceWindows.$inferSelect;
|
||||
|
||||
/** Windows whose end is still in the future. ISO strings compare correctly as text, so this is a plain string comparison. */
|
||||
export async function listActiveWindows(now: Date = new Date()): Promise<ActiveWindow[]> {
|
||||
return db.select().from(maintenanceWindows).where(gt(maintenanceWindows.endsAt, now.toISOString()));
|
||||
}
|
||||
|
||||
/**
|
||||
* The identity a health condition or alert belongs to, as "server:3" /
|
||||
* "integration:1" / "dns_provider:2", so it can be matched against active windows.
|
||||
* Derived from the condition key (which already encodes it) rather than stored,
|
||||
* so conditions persisted by an earlier version still map correctly.
|
||||
*/
|
||||
export function subjectOfConditionKey(key: string): string | null {
|
||||
let m = /^(?:offline|disk):server:(\d+)(?::|$)/.exec(key);
|
||||
if (m) return `server:${m[1]}`;
|
||||
m = /^disk:proxmox:(\d+):/.exec(key);
|
||||
if (m) return `integration:${m[1]}`;
|
||||
m = /^(?:synology-volume|synology-disk|disk:synology):(\d+):/.exec(key);
|
||||
if (m) return `integration:${m[1]}`;
|
||||
return null;
|
||||
}
|
||||
|
||||
export async function activeSubjects(now: Date = new Date()): Promise<Set<string>> {
|
||||
return new Set((await listActiveWindows(now)).map((w) => `${w.targetType}:${w.targetId}`));
|
||||
}
|
||||
|
||||
export async function isInMaintenance(targetType: ActiveWindow["targetType"], targetId: number, now: Date = new Date()): Promise<boolean> {
|
||||
return (await activeSubjects(now)).has(`${targetType}:${targetId}`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Integration-failure alerts are tracked per service TYPE ("proxmox",
|
||||
* "cloudflare", ...), not per configured instance, so a window on any
|
||||
* integration or DNS provider of that type silences that type's failure alerts.
|
||||
* (With two integrations of one type, a real failure on the un-windowed one is
|
||||
* silenced too while the window is open — a known consequence of that
|
||||
* granularity, stated on the Maintenance page.)
|
||||
*/
|
||||
export async function isSourceInMaintenance(source: string, now: Date = new Date()): Promise<boolean> {
|
||||
const windows = await listActiveWindows(now);
|
||||
if (windows.length === 0) return false;
|
||||
for (const w of windows) {
|
||||
if (w.targetType === "integration") {
|
||||
const [row] = await db.select({ type: integrations.type }).from(integrations).where(eq(integrations.id, w.targetId)).limit(1);
|
||||
if (row?.type === source) return true;
|
||||
} else if (w.targetType === "dns_provider") {
|
||||
const [row] = await db.select({ type: dnsProviders.providerType }).from(dnsProviders).where(eq(dnsProviders.id, w.targetId)).limit(1);
|
||||
if (row?.type === source) return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Display name for a window's target, or null if the target no longer exists. */
|
||||
export async function describeTarget(targetType: ActiveWindow["targetType"], targetId: number): Promise<{ name: string; kind: string } | null> {
|
||||
if (targetType === "server") {
|
||||
const [r] = await db.select({ name: servers.name }).from(servers).where(eq(servers.id, targetId)).limit(1);
|
||||
return r ? { name: r.name, kind: "Server" } : null;
|
||||
}
|
||||
if (targetType === "integration") {
|
||||
const [r] = await db.select({ name: integrations.name, type: integrations.type }).from(integrations).where(eq(integrations.id, targetId)).limit(1);
|
||||
return r ? { name: r.name, kind: `Integration (${r.type})` } : null;
|
||||
}
|
||||
const [r] = await db.select({ name: dnsProviders.name, type: dnsProviders.providerType }).from(dnsProviders).where(eq(dnsProviders.id, targetId)).limit(1);
|
||||
return r ? { name: r.name, kind: `DNS provider (${r.type})` } : null;
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import { loadIntegrationConfig } from "../integrations/loadIntegration.js";
|
||||
import { createProxmoxAdapter, guestsWithoutBackupCoverage, type ProxmoxBackupTask } from "../integrations/proxmox/adapter.js";
|
||||
import { notifyProxmoxBackupFailure, notifyProxmoxUncoveredGuests } from "./notify.js";
|
||||
import { getSettings, getInternalFlag, setInternalFlag } from "./settingsStore.js";
|
||||
import { isInMaintenance } from "./maintenance.js";
|
||||
|
||||
const LAST_RUN_FLAG = "proxmoxBackupCheckLastRunDate";
|
||||
|
||||
@@ -19,6 +20,8 @@ async function checkProxmoxBackups(): Promise<void> {
|
||||
const uncovered: { integrationName: string; guestName: string; vmid: number; node: string }[] = [];
|
||||
|
||||
for (const row of rows) {
|
||||
// A host being worked on can't run or report backups; this check is daily, so tomorrow's pass covers it.
|
||||
if (await isInMaintenance("integration", row.id)) continue;
|
||||
try {
|
||||
const loaded = await loadIntegrationConfig(row.id);
|
||||
if (!loaded) continue;
|
||||
|
||||
Reference in new issue
Block a user