Add an Alerts page under Operations listing everything that's wrong now
One list of the current problems across servers and integrations, instead of waiting for a notification or visiting each page: servers that stopped reporting, full or nearly full disks and volumes (critical from 95%), Synology volume/disk problems, failed or uncovered Proxmox backups, failed Proxmox Backup Server verifications, container image updates, expired or expiring secrets/domains/Tailscale keys, failed Semaphore and Gitea runs, Uptime Kuma monitors that are down, overdue osTicket tickets, and integrations whose calls keep failing. Visible to every role, with severity and kind filters, search, sorting, CSV export and "Check now". It runs the same checks that send the notifications rather than a second copy of them: the detection in the health, automation, Proxmox backup, PBS, Docker update and Tailscale key checks is pulled out into shared collectors that both the schedulers and the page call, so the two can't disagree about what counts as a problem. Notification behaviour is unchanged, including the scheduled backup checks skipping integrations under a maintenance window. Unlike the notifications the page ignores the on/off toggles, and keeps problems under a maintenance window, marked silenced and counted apart. It reads live, so a result is reused for a minute (and Refresh can't re-run everything more than once every ten seconds), and every source has a 20 s limit so one hung integration can't hang the page. Anything it couldn't read is called out at the top instead of looking like all clear, and server checks pause for the same 20 minutes after a restart as the notifications do, with a note saying so. Also gives the newer integrations (PBS, osTicket, Uptime Kuma, phpIPAM) proper names in "integration down" notifications instead of their ids. Verified through the real routes against a scratch database with fake backends (offline and full-disk servers, secrets and domains, a silenced server, a fake PBS with failed verification, a hanging integration, a refused one, a failing-calls streak, caching, the restart grace period, auth), and by rendering the real page against that data in a browser: filters, search, silenced toggle, sorting, Check now, dark mode. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
ad1fb5338f
commit
447f33fff6
19 files changed
+883
-28
No files matched your search
@@ -0,0 +1,387 @@
|
||||
/**
|
||||
* Everything that's wrong right now, in one list — the Alerts page. Nothing here decides what counts as a problem: it asks
|
||||
* the same checks that send the notifications (health, backups, updates, expiry, automation, ...) and turns what they find
|
||||
* into a flat list, so the page and the notifications can't disagree. Unlike the notifications it ignores the per-event
|
||||
* on/off toggles (the page is for looking at, not for being interrupted by) and keeps problems that are under a
|
||||
* maintenance window, flagged as silenced rather than dropped.
|
||||
*
|
||||
* It reads live (a handful of API calls per integration), so a result is kept for a short while rather than re-run for
|
||||
* every viewer, and every source has a time limit so one hung integration can't hang the page. A source that can't be
|
||||
* read is reported as such — silence from it must never look like "all clear".
|
||||
*/
|
||||
import { eq } from "drizzle-orm";
|
||||
import { db } from "../db/client.js";
|
||||
import { integrations, secrets } from "../db/schema.js";
|
||||
import { loadIntegrationConfig } from "../integrations/loadIntegration.js";
|
||||
import { createUptimeKumaAdapter } from "../integrations/uptimekuma/adapter.js";
|
||||
import { createOsTicketAdapter } from "../integrations/osticket/adapter.js";
|
||||
import { activeSubjects, isSourceInMaintenance, subjectOfConditionKey } from "./maintenance.js";
|
||||
import { collectSnapshot, evaluateHealth, startupGraceRemainingMs } from "./healthMonitor.js";
|
||||
import { collectAutomation, evaluateAutomation } from "./automationMonitor.js";
|
||||
import { collectDockerUpdates } from "./dockerUpdateScheduler.js";
|
||||
import { collectProxmoxBackupProblems } from "./proxmoxBackupScheduler.js";
|
||||
import { collectPbsProblems } from "./pbsVerificationScheduler.js";
|
||||
import { collectTailscaleKeyExpiries } from "./tailscaleKeyExpiryScheduler.js";
|
||||
import { collectDomainAlerts } from "./domainMonitor.js";
|
||||
import { computeSecretStatus } from "./secretStatus.js";
|
||||
import { getFailingSources } from "./integrationHealthMonitor.js";
|
||||
import { getSettings } from "./settingsStore.js";
|
||||
import { sourceLabel } from "./notify.js";
|
||||
import type { Alert, AlertCategory, AlertSeverity, SourceFailure } from "./alertTypes.js";
|
||||
|
||||
export interface AlertsReport {
|
||||
alerts: Alert[];
|
||||
/** Active problems by severity — those under a maintenance window are counted apart, not in these. */
|
||||
counts: { critical: number; warning: number; info: number; silenced: number };
|
||||
/** Things that couldn't be checked, and which checks that leaves blind. */
|
||||
couldntCheck: { name: string; error: string; affects: string[] }[];
|
||||
/** Context worth knowing about how complete the picture is. */
|
||||
notes: string[];
|
||||
generatedAt: string;
|
||||
}
|
||||
|
||||
/** Where each kind of source lives in the app, and what to call it. */
|
||||
const SOURCES: Record<string, { label: string; link: string }> = {
|
||||
server: { label: "Server", link: "/servers" },
|
||||
proxmox: { label: "Proxmox", link: "/proxmox" },
|
||||
synology: { label: "Synology", link: "/synology" },
|
||||
semaphore: { label: "Semaphore", link: "/semaphore" },
|
||||
gitea: { label: "Gitea", link: "/gitea" },
|
||||
dockhand: { label: "Docker", link: "/docker" },
|
||||
tailscale: { label: "Tailscale", link: "/tailscale" },
|
||||
pbs: { label: "Proxmox Backup", link: "/pbs" },
|
||||
uptimekuma: { label: "Uptime Kuma", link: "/uptime-kuma" },
|
||||
osticket: { label: "osTicket", link: "/osticket" },
|
||||
secrets: { label: "Secrets", link: "/secrets" },
|
||||
domains: { label: "Domains", link: "/domains" },
|
||||
};
|
||||
|
||||
const SOURCE_TIMEOUT_MS = 20_000;
|
||||
/** How long a result is reused. */
|
||||
const CACHE_MS = 60_000;
|
||||
/** Pressing Refresh over and over shouldn't hammer every integration — a result younger than this is reused even then. */
|
||||
const MIN_REFRESH_MS = 10_000;
|
||||
|
||||
const SEVERITY_ORDER: Record<AlertSeverity, number> = { critical: 0, warning: 1, info: 2 };
|
||||
|
||||
/** Which kind of source, and which one, a health/automation condition key belongs to. */
|
||||
function originOfConditionKey(key: string): { type: string; id: number | null; category: AlertCategory } {
|
||||
let m = /^(?:offline|disk):server:(\d+)/.exec(key);
|
||||
if (m) return { type: "server", id: Number(m[1]), category: key.startsWith("offline") ? "offline" : "disk" };
|
||||
m = /^disk:proxmox:(\d+):/.exec(key);
|
||||
if (m) return { type: "proxmox", id: Number(m[1]), category: "disk" };
|
||||
m = /^(?:synology-volume|synology-disk|disk:synology):(\d+):/.exec(key);
|
||||
if (m) return { type: "synology", id: Number(m[1]), category: "disk" };
|
||||
m = /^automation:(semaphore|gitea):(\d+):/.exec(key);
|
||||
if (m) return { type: m[1], id: Number(m[2]), category: "automation" };
|
||||
return { type: "server", id: null, category: "disk" };
|
||||
}
|
||||
|
||||
function withTimeout<T>(work: Promise<T>): Promise<T> {
|
||||
let timer: ReturnType<typeof setTimeout>;
|
||||
const limit = new Promise<never>((_, reject) => {
|
||||
timer = setTimeout(() => reject(new Error(`timed out after ${SOURCE_TIMEOUT_MS / 1000}s`)), SOURCE_TIMEOUT_MS);
|
||||
});
|
||||
return Promise.race([work, limit]).finally(() => clearTimeout(timer));
|
||||
}
|
||||
|
||||
const errorText = (err: unknown) => (err instanceof Error ? err.message : String(err));
|
||||
const plural = (n: number, one: string, many = `${one}s`) => `${n} ${n === 1 ? one : many}`;
|
||||
|
||||
export async function collectAlerts(): Promise<AlertsReport> {
|
||||
const alerts: Alert[] = [];
|
||||
const notes: string[] = [];
|
||||
const blind = new Map<string, { error: string; affects: Set<string> }>();
|
||||
const subjects = await activeSubjects();
|
||||
const intRows = await db.select({ id: integrations.id, name: integrations.name, type: integrations.type, enabled: integrations.enabled }).from(integrations);
|
||||
const intById = new Map(intRows.map((r) => [r.id, r]));
|
||||
|
||||
function push(a: { category: AlertCategory; severity: AlertSeverity; type: string; message: string; key: string; link?: string | null; silenced?: boolean }) {
|
||||
const meta = SOURCES[a.type];
|
||||
alerts.push({
|
||||
id: `${a.category}:${a.key}`,
|
||||
severity: a.severity,
|
||||
category: a.category,
|
||||
source: meta?.label ?? sourceLabel(a.type),
|
||||
message: a.message,
|
||||
link: a.link === undefined ? (meta?.link ?? null) : a.link,
|
||||
silenced: !!a.silenced,
|
||||
});
|
||||
}
|
||||
|
||||
function cannotCheck(name: string, error: string, check: string) {
|
||||
const entry = blind.get(name) ?? { error, affects: new Set<string>() };
|
||||
entry.affects.add(check);
|
||||
blind.set(name, entry);
|
||||
}
|
||||
const cannotCheckIntegration = (f: SourceFailure, check: string) => {
|
||||
const row = intById.get(f.integrationId);
|
||||
cannotCheck(`${row ? (SOURCES[row.type]?.label ?? row.type) : "Integration"} “${f.integrationName}”`, f.message, check);
|
||||
};
|
||||
const silencedIntegration = (id: number) => subjects.has(`integration:${id}`);
|
||||
|
||||
// One entry per check. A check that blows up, or hangs, only costs its own section of the page.
|
||||
async function check(name: string, work: () => Promise<void>) {
|
||||
try {
|
||||
await withTimeout(work());
|
||||
} catch (err) {
|
||||
cannotCheck(name, errorText(err), name);
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all([
|
||||
check("Server and storage health", async () => {
|
||||
const { healthChecks } = await getSettings();
|
||||
const { snapshot, held } = await collectSnapshot();
|
||||
const now = Date.now();
|
||||
const grace = startupGraceRemainingMs(now);
|
||||
if (grace > 0) {
|
||||
notes.push(
|
||||
`Server-offline and server-disk checks are paused for another ${Math.ceil(grace / 60_000)} min after the app restarted, so agents get a chance to report before any server is judged.`,
|
||||
);
|
||||
}
|
||||
for (const c of evaluateHealth(snapshot, healthChecks, now, { skipServers: grace > 0 })) {
|
||||
const origin = originOfConditionKey(c.key);
|
||||
const subject = subjectOfConditionKey(c.key);
|
||||
push({
|
||||
category: origin.category,
|
||||
severity: c.severity ?? "warning",
|
||||
type: origin.type,
|
||||
message: c.message,
|
||||
key: c.key,
|
||||
link: origin.type === "server" && origin.id !== null ? `/servers/${origin.id}` : undefined,
|
||||
silenced: subject !== null && subjects.has(subject),
|
||||
});
|
||||
}
|
||||
// Integrations the check couldn't read this time. (A whole-integration entry covers its nodes, so skip those.)
|
||||
for (const h of held) {
|
||||
const m = /^(proxmox|synology):(\d+)(?::(.+))?$/.exec(h);
|
||||
if (!m || (m[3] && held.has(`${m[1]}:${m[2]}`))) continue;
|
||||
const row = intById.get(Number(m[2]));
|
||||
cannotCheck(`${SOURCES[m[1]].label} “${row?.name ?? `#${m[2]}`}”${m[3] ? ` (node ${m[3]})` : ""}`, "couldn't be read — see the Diagnostic Log", "Server and storage health");
|
||||
}
|
||||
}),
|
||||
|
||||
check("Automation runs", async () => {
|
||||
const { items, held } = await collectAutomation();
|
||||
for (const c of evaluateAutomation(items).conditions) {
|
||||
const origin = originOfConditionKey(c.key);
|
||||
const subject = subjectOfConditionKey(c.key);
|
||||
push({ category: "automation", severity: "warning", type: origin.type, message: c.message, key: c.key, silenced: subject !== null && subjects.has(subject) });
|
||||
}
|
||||
for (const h of held) {
|
||||
const m = /^(semaphore|gitea):(\d+)$/.exec(h);
|
||||
if (!m) continue;
|
||||
const row = intById.get(Number(m[2]));
|
||||
cannotCheck(`${SOURCES[m[1]].label} “${row?.name ?? `#${m[2]}`}”`, "couldn't be read — see the Diagnostic Log", "Automation runs");
|
||||
}
|
||||
}),
|
||||
|
||||
check("Proxmox backups", async () => {
|
||||
const { failures, uncovered, sourceFailures } = await collectProxmoxBackupProblems({ skipSilenced: false });
|
||||
for (const f of failures) {
|
||||
push({
|
||||
category: "backup",
|
||||
severity: "critical",
|
||||
type: "proxmox",
|
||||
message: `Latest backup on ${f.node}${f.guestId ? ` (guest ${f.guestId})` : ""} didn't succeed [${f.integrationName}]: ${f.status}`,
|
||||
key: `proxmox-backup:${f.integrationId}:${f.node}`,
|
||||
silenced: f.silenced,
|
||||
});
|
||||
}
|
||||
for (const u of uncovered) {
|
||||
push({
|
||||
category: "backup",
|
||||
severity: "warning",
|
||||
type: "proxmox",
|
||||
message: `${u.guestName} (#${u.vmid}) on ${u.node} isn't covered by any backup job [${u.integrationName}]`,
|
||||
key: `proxmox-uncovered:${u.integrationId}:${u.vmid}`,
|
||||
silenced: u.silenced,
|
||||
});
|
||||
}
|
||||
sourceFailures.forEach((f) => cannotCheckIntegration(f, "Proxmox backups"));
|
||||
}),
|
||||
|
||||
check("Backup verification", async () => {
|
||||
const { problems, sourceFailures } = await collectPbsProblems({ skipSilenced: false });
|
||||
for (const p of problems) {
|
||||
push({
|
||||
category: "backup",
|
||||
severity: p.error ? "warning" : "critical",
|
||||
type: "pbs",
|
||||
message: p.error
|
||||
? `Datastore "${p.datastore}" couldn't be read [${p.integrationName}]: ${p.error}`
|
||||
: `Datastore "${p.datastore}" has ${plural(p.failedCount, "snapshot")} that failed verification [${p.integrationName}]`,
|
||||
key: `pbs:${p.integrationId}:${p.datastore}`,
|
||||
silenced: p.silenced,
|
||||
});
|
||||
}
|
||||
sourceFailures.forEach((f) => cannotCheckIntegration(f, "Backup verification"));
|
||||
}),
|
||||
|
||||
check("Image updates", async () => {
|
||||
const { items, failures } = await collectDockerUpdates();
|
||||
for (const u of items) {
|
||||
push({
|
||||
category: "updates",
|
||||
severity: "info",
|
||||
type: "dockhand",
|
||||
message: `${u.containerName} [${u.environmentName}, ${u.integrationName}] has an image update available${u.newerVersion ? ` → ${u.newerVersion}` : ""}`,
|
||||
key: `docker:${u.integrationId}:${u.environmentName}:${u.containerName}`,
|
||||
silenced: silencedIntegration(u.integrationId),
|
||||
});
|
||||
}
|
||||
failures.forEach((f) => cannotCheckIntegration(f, "Image updates"));
|
||||
}),
|
||||
|
||||
check("Tailscale keys", async () => {
|
||||
const { items, failures } = await collectTailscaleKeyExpiries();
|
||||
for (const k of items) {
|
||||
push({
|
||||
category: "expiry",
|
||||
severity: k.daysLeft < 0 ? "critical" : "warning",
|
||||
type: "tailscale",
|
||||
message: k.daysLeft < 0 ? `Key for ${k.deviceLabel} [${k.integrationName}] has expired` : `Key for ${k.deviceLabel} [${k.integrationName}] expires in ${plural(k.daysLeft, "day")}`,
|
||||
key: `tailscale-key:${k.integrationId}:${k.deviceLabel}`,
|
||||
silenced: silencedIntegration(k.integrationId),
|
||||
});
|
||||
}
|
||||
failures.forEach((f) => cannotCheckIntegration(f, "Tailscale keys"));
|
||||
}),
|
||||
|
||||
check("Uptime Kuma", async () => {
|
||||
for (const row of intRows.filter((r) => r.type === "uptimekuma" && r.enabled)) {
|
||||
try {
|
||||
const loaded = await loadIntegrationConfig(row.id);
|
||||
if (!loaded) continue;
|
||||
for (const m of await createUptimeKumaAdapter(loaded.config as any).listMonitors()) {
|
||||
if (m.status !== "down") continue;
|
||||
push({
|
||||
category: "monitoring",
|
||||
severity: "critical",
|
||||
type: "uptimekuma",
|
||||
message: `Monitor "${m.name}" is down${m.target ? ` (${m.target}${m.port ? `:${m.port}` : ""})` : ""} [${row.name}]`,
|
||||
key: `kuma:${row.id}:${m.id}`,
|
||||
silenced: silencedIntegration(row.id),
|
||||
});
|
||||
}
|
||||
} catch (err) {
|
||||
cannotCheckIntegration({ integrationId: row.id, integrationName: row.name, message: errorText(err) }, "Uptime Kuma");
|
||||
}
|
||||
}
|
||||
}),
|
||||
|
||||
check("osTicket", async () => {
|
||||
for (const row of intRows.filter((r) => r.type === "osticket" && r.enabled)) {
|
||||
try {
|
||||
const loaded = await loadIntegrationConfig(row.id);
|
||||
if (!loaded) continue;
|
||||
const overdue = (await createOsTicketAdapter(loaded.config as any).listOpenTickets()).filter((t) => t.isOverdue).length;
|
||||
if (overdue > 0) {
|
||||
push({
|
||||
category: "tickets",
|
||||
severity: "warning",
|
||||
type: "osticket",
|
||||
message: `${plural(overdue, "open ticket")} ${overdue === 1 ? "is" : "are"} overdue [${row.name}]`,
|
||||
key: `osticket:${row.id}`,
|
||||
silenced: silencedIntegration(row.id),
|
||||
});
|
||||
}
|
||||
} catch (err) {
|
||||
cannotCheckIntegration({ integrationId: row.id, integrationName: row.name, message: errorText(err) }, "osTicket");
|
||||
}
|
||||
}
|
||||
}),
|
||||
|
||||
check("Secrets", async () => {
|
||||
for (const s of await db.select().from(secrets)) {
|
||||
const status = computeSecretStatus(s.expiryDate, s.warnDays);
|
||||
if (status.status === "expired") {
|
||||
push({ category: "expiry", severity: "critical", type: "secrets", message: `Secret "${s.name}" expired ${plural(Math.abs(status.daysLeft), "day")} ago (${s.expiryDate})`, key: `secret:${s.id}` });
|
||||
} else if (status.status === "expiring") {
|
||||
push({ category: "expiry", severity: "warning", type: "secrets", message: `Secret "${s.name}" expires in ${plural(status.daysLeft, "day")} (${s.expiryDate})`, key: `secret:${s.id}` });
|
||||
}
|
||||
if (s.checkHost && s.lastCheckError) {
|
||||
push({
|
||||
category: "expiry",
|
||||
severity: "warning",
|
||||
type: "secrets",
|
||||
message: `Couldn't read the live certificate for "${s.name}" (${s.checkHost}:${s.checkPort ?? 443}) — the expiry shown may be stale: ${s.lastCheckError}`,
|
||||
key: `secret-check:${s.id}`,
|
||||
});
|
||||
}
|
||||
}
|
||||
}),
|
||||
|
||||
check("Domains", async () => {
|
||||
const { expiring, staleChecks } = await collectDomainAlerts();
|
||||
for (const d of expiring) {
|
||||
push({
|
||||
category: "expiry",
|
||||
severity: d.status === "expired" ? "critical" : "warning",
|
||||
type: "domains",
|
||||
message: d.status === "expired" ? `Domain ${d.name} expired on ${d.expiresAt}` : `Domain ${d.name} expires in ${plural(d.daysLeft, "day")} (${d.expiresAt})`,
|
||||
key: `domain:${d.name}`,
|
||||
});
|
||||
}
|
||||
for (const s of staleChecks) {
|
||||
push({ category: "expiry", severity: "info", type: "domains", message: `Couldn't refresh the registration for ${s.name} — the expiry shown may be stale: ${s.error}`, key: `domain-stale:${s.name}` });
|
||||
}
|
||||
}),
|
||||
|
||||
check("Integration failures", async () => {
|
||||
const { notifications } = await getSettings();
|
||||
for (const f of getFailingSources()) {
|
||||
push({
|
||||
category: "integration",
|
||||
severity: f.alerted ? "critical" : "warning",
|
||||
type: f.source,
|
||||
message: `${sourceLabel(f.source)} has failed its last ${plural(f.consecutiveFailures, "call")} in a row${f.alerted ? "" : ` (a notification goes out after ${notifications.integrationFailureThreshold})`} — see the Diagnostic Log`,
|
||||
key: `failing:${f.source}`,
|
||||
link: null,
|
||||
silenced: await isSourceInMaintenance(f.source),
|
||||
});
|
||||
}
|
||||
}),
|
||||
]);
|
||||
|
||||
alerts.sort(
|
||||
(a, b) =>
|
||||
Number(a.silenced) - Number(b.silenced) ||
|
||||
SEVERITY_ORDER[a.severity] - SEVERITY_ORDER[b.severity] ||
|
||||
a.source.localeCompare(b.source) ||
|
||||
a.message.localeCompare(b.message),
|
||||
);
|
||||
|
||||
const active = alerts.filter((a) => !a.silenced);
|
||||
return {
|
||||
alerts,
|
||||
counts: {
|
||||
critical: active.filter((a) => a.severity === "critical").length,
|
||||
warning: active.filter((a) => a.severity === "warning").length,
|
||||
info: active.filter((a) => a.severity === "info").length,
|
||||
silenced: alerts.length - active.length,
|
||||
},
|
||||
couldntCheck: [...blind.entries()].map(([name, v]) => ({ name, error: v.error, affects: [...v.affects] })),
|
||||
notes,
|
||||
generatedAt: new Date().toISOString(),
|
||||
};
|
||||
}
|
||||
|
||||
let cache: { at: number; report: AlertsReport } | null = null;
|
||||
let inFlight: Promise<AlertsReport> | null = null;
|
||||
|
||||
/** The current alerts, reusing a recent result unless `force` asks for a fresh one (and even then not more than once every few seconds). */
|
||||
export async function getAlerts(force: boolean): Promise<AlertsReport & { cached: boolean }> {
|
||||
const age = cache ? Date.now() - cache.at : Infinity;
|
||||
if (cache && age < (force ? MIN_REFRESH_MS : CACHE_MS)) return { ...cache.report, cached: true };
|
||||
inFlight ??= collectAlerts()
|
||||
.then((report) => {
|
||||
cache = { at: Date.now(), report };
|
||||
return report;
|
||||
})
|
||||
.finally(() => {
|
||||
inFlight = null;
|
||||
});
|
||||
return { ...(await inFlight), cached: false };
|
||||
}
|
||||
Reference in new issue
Block a user