/** * Everything that's wrong right now, in one list — the Alerts page. Nothing here decides what counts as a problem: it asks * the same checks that send the notifications (health, backups, updates, expiry, automation, ...) and turns what they find * into a flat list, so the page and the notifications can't disagree. Unlike the notifications it ignores the per-event * on/off toggles (the page is for looking at, not for being interrupted by) and keeps problems that are under a * maintenance window, flagged as silenced rather than dropped. * * It reads live (a handful of API calls per integration), so a result is kept for a short while rather than re-run for * every viewer, and every source has a time limit so one hung integration can't hang the page. A source that can't be * read is reported as such — silence from it must never look like "all clear". */ import { eq } from "drizzle-orm"; import { db } from "../db/client.js"; import { integrations, secrets } from "../db/schema.js"; import { loadIntegrationConfig } from "../integrations/loadIntegration.js"; import { createUptimeKumaAdapter } from "../integrations/uptimekuma/adapter.js"; import { createOsTicketAdapter } from "../integrations/osticket/adapter.js"; import { activeSubjects, isSourceInMaintenance, subjectOfConditionKey } from "./maintenance.js"; import { collectSnapshot, evaluateHealth, startupGraceRemainingMs } from "./healthMonitor.js"; import { collectAutomation, evaluateAutomation } from "./automationMonitor.js"; import { collectDockerUpdates } from "./dockerUpdateScheduler.js"; import { collectProxmoxBackupProblems } from "./proxmoxBackupScheduler.js"; import { collectPbsProblems } from "./pbsVerificationScheduler.js"; import { collectTailscaleKeyExpiries } from "./tailscaleKeyExpiryScheduler.js"; import { collectDomainAlerts } from "./domainMonitor.js"; import { computeSecretStatus } from "./secretStatus.js"; import { getFailingSources } from "./integrationHealthMonitor.js"; import { getSettings } from "./settingsStore.js"; import { sourceLabel } from "./notify.js"; import type { Alert, AlertCategory, AlertSeverity, SourceFailure } from "./alertTypes.js"; export interface AlertsReport { alerts: Alert[]; /** Active problems by severity — those under a maintenance window are counted apart, not in these. */ counts: { critical: number; warning: number; info: number; silenced: number }; /** Things that couldn't be checked, and which checks that leaves blind. */ couldntCheck: { name: string; error: string; affects: string[] }[]; /** Context worth knowing about how complete the picture is. */ notes: string[]; generatedAt: string; } /** Where each kind of source lives in the app, and what to call it. */ const SOURCES: Record = { server: { label: "Server", link: "/servers" }, proxmox: { label: "Proxmox", link: "/proxmox" }, synology: { label: "Synology", link: "/synology" }, semaphore: { label: "Semaphore", link: "/semaphore" }, gitea: { label: "Gitea", link: "/gitea" }, dockhand: { label: "Docker", link: "/docker" }, tailscale: { label: "Tailscale", link: "/tailscale" }, pbs: { label: "Proxmox Backup", link: "/pbs" }, uptimekuma: { label: "Uptime Kuma", link: "/uptime-kuma" }, osticket: { label: "osTicket", link: "/osticket" }, secrets: { label: "Secrets", link: "/secrets" }, domains: { label: "Domains", link: "/domains" }, }; const SOURCE_TIMEOUT_MS = 20_000; /** How long a result is reused. */ const CACHE_MS = 60_000; /** Pressing Refresh over and over shouldn't hammer every integration — a result younger than this is reused even then. */ const MIN_REFRESH_MS = 10_000; const SEVERITY_ORDER: Record = { critical: 0, warning: 1, info: 2 }; /** Which kind of source, and which one, a health/automation condition key belongs to. */ function originOfConditionKey(key: string): { type: string; id: number | null; category: AlertCategory } { let m = /^(?:offline|disk):server:(\d+)/.exec(key); if (m) return { type: "server", id: Number(m[1]), category: key.startsWith("offline") ? "offline" : "disk" }; m = /^disk:proxmox:(\d+):/.exec(key); if (m) return { type: "proxmox", id: Number(m[1]), category: "disk" }; m = /^(?:synology-volume|synology-disk|disk:synology):(\d+):/.exec(key); if (m) return { type: "synology", id: Number(m[1]), category: "disk" }; m = /^automation:(semaphore|gitea):(\d+):/.exec(key); if (m) return { type: m[1], id: Number(m[2]), category: "automation" }; return { type: "server", id: null, category: "disk" }; } function withTimeout(work: Promise): Promise { let timer: ReturnType; const limit = new Promise((_, reject) => { timer = setTimeout(() => reject(new Error(`timed out after ${SOURCE_TIMEOUT_MS / 1000}s`)), SOURCE_TIMEOUT_MS); }); return Promise.race([work, limit]).finally(() => clearTimeout(timer)); } const errorText = (err: unknown) => (err instanceof Error ? err.message : String(err)); const plural = (n: number, one: string, many = `${one}s`) => `${n} ${n === 1 ? one : many}`; export async function collectAlerts(): Promise { const alerts: Alert[] = []; const notes: string[] = []; const blind = new Map }>(); const subjects = await activeSubjects(); const intRows = await db.select({ id: integrations.id, name: integrations.name, type: integrations.type, enabled: integrations.enabled }).from(integrations); const intById = new Map(intRows.map((r) => [r.id, r])); function push(a: { category: AlertCategory; severity: AlertSeverity; type: string; message: string; key: string; link?: string | null; silenced?: boolean }) { const meta = SOURCES[a.type]; alerts.push({ id: `${a.category}:${a.key}`, severity: a.severity, category: a.category, source: meta?.label ?? sourceLabel(a.type), message: a.message, link: a.link === undefined ? (meta?.link ?? null) : a.link, silenced: !!a.silenced, }); } function cannotCheck(name: string, error: string, check: string) { const entry = blind.get(name) ?? { error, affects: new Set() }; entry.affects.add(check); blind.set(name, entry); } const cannotCheckIntegration = (f: SourceFailure, check: string) => { const row = intById.get(f.integrationId); cannotCheck(`${row ? (SOURCES[row.type]?.label ?? row.type) : "Integration"} “${f.integrationName}”`, f.message, check); }; const silencedIntegration = (id: number) => subjects.has(`integration:${id}`); // One entry per check. A check that blows up, or hangs, only costs its own section of the page. async function check(name: string, work: () => Promise) { try { await withTimeout(work()); } catch (err) { cannotCheck(name, errorText(err), name); } } await Promise.all([ check("Server and storage health", async () => { const { healthChecks } = await getSettings(); const { snapshot, held } = await collectSnapshot(); const now = Date.now(); const grace = startupGraceRemainingMs(now); if (grace > 0) { notes.push( `Server-offline and server-disk checks are paused for another ${Math.ceil(grace / 60_000)} min after the app restarted, so agents get a chance to report before any server is judged.`, ); } for (const c of evaluateHealth(snapshot, healthChecks, now, { skipServers: grace > 0 })) { const origin = originOfConditionKey(c.key); const subject = subjectOfConditionKey(c.key); push({ category: origin.category, severity: c.severity ?? "warning", type: origin.type, message: c.message, key: c.key, link: origin.type === "server" && origin.id !== null ? `/servers/${origin.id}` : undefined, silenced: subject !== null && subjects.has(subject), }); } // Integrations the check couldn't read this time. (A whole-integration entry covers its nodes, so skip those.) for (const h of held) { const m = /^(proxmox|synology):(\d+)(?::(.+))?$/.exec(h); if (!m || (m[3] && held.has(`${m[1]}:${m[2]}`))) continue; const row = intById.get(Number(m[2])); cannotCheck(`${SOURCES[m[1]].label} “${row?.name ?? `#${m[2]}`}”${m[3] ? ` (node ${m[3]})` : ""}`, "couldn't be read — see the Diagnostic Log", "Server and storage health"); } }), check("Automation runs", async () => { const { items, held } = await collectAutomation(); for (const c of evaluateAutomation(items).conditions) { const origin = originOfConditionKey(c.key); const subject = subjectOfConditionKey(c.key); push({ category: "automation", severity: "warning", type: origin.type, message: c.message, key: c.key, silenced: subject !== null && subjects.has(subject) }); } for (const h of held) { const m = /^(semaphore|gitea):(\d+)$/.exec(h); if (!m) continue; const row = intById.get(Number(m[2])); cannotCheck(`${SOURCES[m[1]].label} “${row?.name ?? `#${m[2]}`}”`, "couldn't be read — see the Diagnostic Log", "Automation runs"); } }), check("Proxmox backups", async () => { const { failures, uncovered, sourceFailures } = await collectProxmoxBackupProblems({ skipSilenced: false }); for (const f of failures) { push({ category: "backup", severity: "critical", type: "proxmox", message: `Latest backup on ${f.node}${f.guestId ? ` (guest ${f.guestId})` : ""} didn't succeed [${f.integrationName}]: ${f.status}`, key: `proxmox-backup:${f.integrationId}:${f.node}`, silenced: f.silenced, }); } for (const u of uncovered) { push({ category: "backup", severity: "warning", type: "proxmox", message: `${u.guestName} (#${u.vmid}) on ${u.node} isn't covered by any backup job [${u.integrationName}]`, key: `proxmox-uncovered:${u.integrationId}:${u.vmid}`, silenced: u.silenced, }); } sourceFailures.forEach((f) => cannotCheckIntegration(f, "Proxmox backups")); }), check("Backup verification", async () => { const { problems, sourceFailures } = await collectPbsProblems({ skipSilenced: false }); for (const p of problems) { push({ category: "backup", severity: p.error ? "warning" : "critical", type: "pbs", message: p.error ? `Datastore "${p.datastore}" couldn't be read [${p.integrationName}]: ${p.error}` : `Datastore "${p.datastore}" has ${plural(p.failedCount, "snapshot")} that failed verification [${p.integrationName}]`, key: `pbs:${p.integrationId}:${p.datastore}`, silenced: p.silenced, }); } sourceFailures.forEach((f) => cannotCheckIntegration(f, "Backup verification")); }), check("Image updates", async () => { const { items, failures } = await collectDockerUpdates(); for (const u of items) { push({ category: "updates", severity: "info", type: "dockhand", message: `${u.containerName} [${u.environmentName}, ${u.integrationName}] has an image update available${u.newerVersion ? ` → ${u.newerVersion}` : ""}`, key: `docker:${u.integrationId}:${u.environmentName}:${u.containerName}`, silenced: silencedIntegration(u.integrationId), }); } failures.forEach((f) => cannotCheckIntegration(f, "Image updates")); }), check("Tailscale keys", async () => { const { items, failures } = await collectTailscaleKeyExpiries(); for (const k of items) { push({ category: "expiry", severity: k.daysLeft < 0 ? "critical" : "warning", type: "tailscale", message: k.daysLeft < 0 ? `Key for ${k.deviceLabel} [${k.integrationName}] has expired` : `Key for ${k.deviceLabel} [${k.integrationName}] expires in ${plural(k.daysLeft, "day")}`, key: `tailscale-key:${k.integrationId}:${k.deviceLabel}`, silenced: silencedIntegration(k.integrationId), }); } failures.forEach((f) => cannotCheckIntegration(f, "Tailscale keys")); }), check("Uptime Kuma", async () => { for (const row of intRows.filter((r) => r.type === "uptimekuma" && r.enabled)) { try { const loaded = await loadIntegrationConfig(row.id); if (!loaded) continue; for (const m of await createUptimeKumaAdapter(loaded.config as any).listMonitors()) { if (m.status !== "down") continue; push({ category: "monitoring", severity: "critical", type: "uptimekuma", message: `Monitor "${m.name}" is down${m.target ? ` (${m.target}${m.port ? `:${m.port}` : ""})` : ""} [${row.name}]`, key: `kuma:${row.id}:${m.id}`, silenced: silencedIntegration(row.id), }); } } catch (err) { cannotCheckIntegration({ integrationId: row.id, integrationName: row.name, message: errorText(err) }, "Uptime Kuma"); } } }), check("osTicket", async () => { for (const row of intRows.filter((r) => r.type === "osticket" && r.enabled)) { try { const loaded = await loadIntegrationConfig(row.id); if (!loaded) continue; const overdue = (await createOsTicketAdapter(loaded.config as any).listOpenTickets()).filter((t) => t.isOverdue).length; if (overdue > 0) { push({ category: "tickets", severity: "warning", type: "osticket", message: `${plural(overdue, "open ticket")} ${overdue === 1 ? "is" : "are"} overdue [${row.name}]`, key: `osticket:${row.id}`, silenced: silencedIntegration(row.id), }); } } catch (err) { cannotCheckIntegration({ integrationId: row.id, integrationName: row.name, message: errorText(err) }, "osTicket"); } } }), check("Secrets", async () => { for (const s of await db.select().from(secrets)) { const status = computeSecretStatus(s.expiryDate, s.warnDays); if (status.status === "expired") { push({ category: "expiry", severity: "critical", type: "secrets", message: `Secret "${s.name}" expired ${plural(Math.abs(status.daysLeft), "day")} ago (${s.expiryDate})`, key: `secret:${s.id}` }); } else if (status.status === "expiring") { push({ category: "expiry", severity: "warning", type: "secrets", message: `Secret "${s.name}" expires in ${plural(status.daysLeft, "day")} (${s.expiryDate})`, key: `secret:${s.id}` }); } if (s.checkHost && s.lastCheckError) { push({ category: "expiry", severity: "warning", type: "secrets", message: `Couldn't read the live certificate for "${s.name}" (${s.checkHost}:${s.checkPort ?? 443}) — the expiry shown may be stale: ${s.lastCheckError}`, key: `secret-check:${s.id}`, }); } } }), check("Domains", async () => { const { expiring, staleChecks } = await collectDomainAlerts(); for (const d of expiring) { push({ category: "expiry", severity: d.status === "expired" ? "critical" : "warning", type: "domains", message: d.status === "expired" ? `Domain ${d.name} expired on ${d.expiresAt}` : `Domain ${d.name} expires in ${plural(d.daysLeft, "day")} (${d.expiresAt})`, key: `domain:${d.name}`, }); } for (const s of staleChecks) { push({ category: "expiry", severity: "info", type: "domains", message: `Couldn't refresh the registration for ${s.name} — the expiry shown may be stale: ${s.error}`, key: `domain-stale:${s.name}` }); } }), check("Integration failures", async () => { const { notifications } = await getSettings(); for (const f of getFailingSources()) { push({ category: "integration", severity: f.alerted ? "critical" : "warning", type: f.source, message: `${sourceLabel(f.source)} has failed its last ${plural(f.consecutiveFailures, "call")} in a row${f.alerted ? "" : ` (a notification goes out after ${notifications.integrationFailureThreshold})`} — see the Diagnostic Log`, key: `failing:${f.source}`, link: null, silenced: await isSourceInMaintenance(f.source), }); } }), ]); alerts.sort( (a, b) => Number(a.silenced) - Number(b.silenced) || SEVERITY_ORDER[a.severity] - SEVERITY_ORDER[b.severity] || a.source.localeCompare(b.source) || a.message.localeCompare(b.message), ); const active = alerts.filter((a) => !a.silenced); return { alerts, counts: { critical: active.filter((a) => a.severity === "critical").length, warning: active.filter((a) => a.severity === "warning").length, info: active.filter((a) => a.severity === "info").length, silenced: alerts.length - active.length, }, couldntCheck: [...blind.entries()].map(([name, v]) => ({ name, error: v.error, affects: [...v.affects] })), notes, generatedAt: new Date().toISOString(), }; } let cache: { at: number; report: AlertsReport } | null = null; let inFlight: Promise | null = null; /** The current alerts, reusing a recent result unless `force` asks for a fresh one (and even then not more than once every few seconds). */ export async function getAlerts(force: boolean): Promise { const age = cache ? Date.now() - cache.at : Infinity; if (cache && age < (force ? MIN_REFRESH_MS : CACHE_MS)) return { ...cache.report, cached: true }; inFlight ??= collectAlerts() .then((report) => { cache = { at: Date.now(), report }; return report; }) .finally(() => { inFlight = null; }); return { ...(await inFlight), cached: false }; }