import { eq } from "drizzle-orm"; import { db } from "../db/client.js"; import { integrations, servers } from "../db/schema.js"; import { loadIntegrationConfig } from "../integrations/loadIntegration.js"; import { createProxmoxAdapter, type ProxmoxNodeStats } from "../integrations/proxmox/adapter.js"; import { createSynologyAdapter, type SynologyStorageInfo } from "../integrations/synology/adapter.js"; import { notifyHealthIssues, notifyHealthRecovered } from "./notify.js"; import { getInternalFlag, getSettings, setInternalFlag } from "./settingsStore.js"; import { activeSubjects, subjectOfConditionKey } from "./maintenance.js"; const STATE_FLAG = "healthActiveConditions"; /** After a restart the agents haven't had a chance to report yet (they run every 15 min), so server-derived conditions are held rather than judged. */ const STARTUP_GRACE_MS = 20 * 60 * 1000; const PROCESS_START = Date.now(); // ─── Types ────────────────────────────────────────────────────────────────── export interface HealthCondition { /** Stable identity of the problem — the same problem always produces the same key, so it alerts once, not every run. */ key: string; /** Where the data came from, hierarchically ("server", "proxmox:3", "proxmox:3:pve1"). Used to hold a condition when its source can't be reached. */ source: string; message: string; /** How bad, for the Alerts page. Notifications don't use it. Left out means "warning". */ severity?: "critical" | "warning"; } export interface ServerSnapshot { id: number; name: string; lastSeenAt: string | null; disks: { mount: string; sizeBytes: number; usedBytes: number }[]; } export interface ProxmoxSnapshot { integrationId: number; integrationName: string; nodes: ProxmoxNodeStats[]; } export interface SynologySnapshot { integrationId: number; integrationName: string; storage: SynologyStorageInfo; } export interface HealthSnapshot { servers: ServerSnapshot[]; proxmox: ProxmoxSnapshot[]; synology: SynologySnapshot[]; } export interface Thresholds { serverOfflineMinutes: number; diskUsagePercent: number; } // ─── Formatting helpers ───────────────────────────────────────────────────── function formatBytes(bytes: number): string { const units = ["B", "KB", "MB", "GB", "TB", "PB"]; let value = bytes; let unit = 0; while (value >= 1024 && unit < units.length - 1) { value /= 1024; unit++; } return `${value.toFixed(value >= 100 || unit === 0 ? 0 : 1)} ${units[unit]}`; } function formatDuration(ms: number): string { const minutes = Math.floor(ms / 60_000); if (minutes < 60) return `${minutes} min`; const hours = Math.floor(minutes / 60); if (hours < 48) return `${hours} h`; return `${Math.floor(hours / 24)} days`; } /** Timestamps this app writes are ISO strings, but a bare SQLite "YYYY-MM-DD HH:MM:SS" (UTC, no zone) must not be read as local time. */ function parseTimestamp(value: string): number { return Date.parse(/[zZ]|[+-]\d{2}:?\d{2}$/.test(value) ? value : `${value.replace(" ", "T")}Z`); } function usage(used: number | null, total: number | null): number | null { if (used === null || total === null || total <= 0) return null; return (used / total) * 100; } // ─── Evaluation (pure) ────────────────────────────────────────────────────── /** * Turns a snapshot of what every source currently reports into the list of * problems that exist right now. Pure — no I/O, no clock other than `now` — * so the rules can be tested exactly. */ export function evaluateHealth(snapshot: HealthSnapshot, thresholds: Thresholds, now: number, opts: { skipServers?: boolean } = {}): HealthCondition[] { const out: HealthCondition[] = []; const limit = thresholds.diskUsagePercent; const pct = (n: number) => `${Math.round(n)}%`; /** Over the configured threshold is a warning; practically out of room is critical. */ const fullness = (p: number): "critical" | "warning" => (p >= 95 ? "critical" : "warning"); if (!opts.skipServers) { for (const s of snapshot.servers) { let offline = false; if (s.lastSeenAt) { const age = now - parseTimestamp(s.lastSeenAt); if (Number.isFinite(age) && age > thresholds.serverOfflineMinutes * 60_000) { offline = true; out.push({ key: `offline:server:${s.id}`, source: "server", message: `${s.name} hasn't reported for ${formatDuration(age)}`, severity: "critical", }); } } // A server that never reported has no agent to go quiet; and an offline one's disk figures are stale, so neither is judged on disks. if (s.lastSeenAt && !offline) { for (const d of s.disks) { const p = usage(d.usedBytes, d.sizeBytes); if (p !== null && p >= limit) { out.push({ key: `disk:server:${s.id}:${d.mount}`, source: "server", message: `${s.name}: ${d.mount} is ${pct(p)} full (${formatBytes(d.usedBytes)} of ${formatBytes(d.sizeBytes)})`, severity: fullness(p), }); } } } } } for (const px of snapshot.proxmox) { const seenShared = new Set(); for (const node of px.nodes) { if (node.error) continue; // couldn't be read — held by the caller, not judged const rootP = usage(node.rootfsUsedBytes, node.rootfsTotalBytes); if (rootP !== null && rootP >= limit) { out.push({ key: `disk:proxmox:${px.integrationId}:${node.node}:rootfs`, source: `proxmox:${px.integrationId}:${node.node}`, message: `${px.integrationName} / ${node.node}: root filesystem is ${pct(rootP)} full`, severity: fullness(rootP), }); } for (const st of node.storages) { if (!st.active) continue; const p = usage(st.usedBytes, st.totalBytes); if (p === null || p < limit) continue; if (st.shared) { // A shared storage is listed by every node — report it once, not once per node. if (seenShared.has(st.id)) continue; seenShared.add(st.id); out.push({ key: `disk:proxmox:${px.integrationId}:storage:${st.id}`, source: `proxmox:${px.integrationId}:shared`, message: `${px.integrationName}: shared storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`, severity: fullness(p), }); } else { out.push({ key: `disk:proxmox:${px.integrationId}:${node.node}:storage:${st.id}`, source: `proxmox:${px.integrationId}:${node.node}`, message: `${px.integrationName} / ${node.node}: storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`, severity: fullness(p), }); } } } } for (const syn of snapshot.synology) { const src = `synology:${syn.integrationId}`; for (const v of syn.storage.volumes) { if (v.status && v.status.toLowerCase() !== "normal") { out.push({ key: `synology-volume:${syn.integrationId}:${v.id}`, source: src, message: `${syn.integrationName}: volume ${v.id} status is "${v.status}"`, severity: "critical", }); } const p = usage(v.sizeUsed, v.sizeTotal); if (p !== null && p >= limit) { out.push({ key: `disk:synology:${syn.integrationId}:${v.id}`, source: src, message: `${syn.integrationName}: volume ${v.id} is ${pct(p)} full (${formatBytes(v.sizeUsed ?? 0)} of ${formatBytes(v.sizeTotal ?? 0)})`, severity: fullness(p), }); } } for (const d of syn.storage.disks) { const problems: string[] = []; if (d.status && d.status.toLowerCase() !== "normal") problems.push(`status "${d.status}"`); if (/warn|crit|fail|danger|bad/i.test(d.smartStatus ?? "")) problems.push(`SMART "${d.smartStatus}"`); if (d.exceedBadSectorThreshold) problems.push("bad sectors above threshold"); if (d.belowRemainLifeThreshold) problems.push("remaining life below threshold"); if (problems.length > 0) { out.push({ key: `synology-disk:${syn.integrationId}:${d.id}`, source: src, message: `${syn.integrationName}: disk ${d.name || d.id} — ${problems.join(", ")}`, // A disk that's no longer "normal" is failing; a SMART warning or a threshold crossing is the early notice. severity: d.status && d.status.toLowerCase() !== "normal" ? "critical" : "warning", }); } } } return out; } // ─── State diff (pure) ────────────────────────────────────────────────────── export type ActiveState = Record; /** `held` is a set of hierarchical sources that couldn't be read this run; matches the source itself or anything beneath it. */ function isHeld(source: string, held: Set): boolean { for (const h of held) if (source === h || source.startsWith(`${h}:`)) return true; return false; } /** * Compares this run's conditions with what was already being reported. * A condition present now but not before is new; one present before but not * now is resolved — except when its source couldn't be read this run, in * which case it's carried over untouched. Without that, one failed poll would * report a recovery and then re-alert the same problem on the next poll. */ export function diffConditions( previous: ActiveState, current: HealthCondition[], held: Set = new Set(), /** Subjects ("server:3", "integration:1") under an active maintenance window. */ silenced: Set = new Set(), ): { added: HealthCondition[]; resolved: { key: string; message: string }[]; next: ActiveState } { const isSilenced = (key: string) => { const subject = subjectOfConditionKey(key); return subject !== null && silenced.has(subject); }; // A silenced problem is not recorded as "known": if it's still there when the window ends it must // alert then, as new. (One that was already alerted before the window stays known, so it isn't repeated.) const visible = current.filter((c) => !isSilenced(c.key)); const next: ActiveState = {}; for (const c of visible) next[c.key] = { message: c.message, source: c.source }; const added = visible.filter((c) => !(c.key in previous)); const resolved: { key: string; message: string }[] = []; for (const [key, entry] of Object.entries(previous)) { if (key in next) continue; // Held (source unreadable) or silenced: leave it exactly as it was — neither cleared nor re-alerted. if (isHeld(entry.source, held) || isSilenced(key)) { next[key] = entry; } else { resolved.push({ key, message: entry.message }); } } return { added, resolved, next }; } // ─── Collection (I/O) ─────────────────────────────────────────────────────── /** How much longer, in ms, servers are exempt from being judged after a restart (their agents get a chance to report first). 0 once it's over. */ export function startupGraceRemainingMs(now: number = Date.now()): number { return Math.max(0, STARTUP_GRACE_MS - (now - PROCESS_START)); } export async function collectSnapshot(): Promise<{ snapshot: HealthSnapshot; held: Set }> { const held = new Set(); const snapshot: HealthSnapshot = { servers: [], proxmox: [], synology: [] }; for (const s of await db.select().from(servers)) { let disks: ServerSnapshot["disks"] = []; try { disks = s.disks ? JSON.parse(s.disks) : []; } catch { // a malformed report shouldn't stop every other server being checked } snapshot.servers.push({ id: s.id, name: s.name, lastSeenAt: s.lastSeenAt, disks }); } const rows = await db .select({ id: integrations.id, name: integrations.name, type: integrations.type }) .from(integrations) .where(eq(integrations.enabled, true)); for (const row of rows) { if (row.type !== "proxmox" && row.type !== "synology") continue; try { const loaded = await loadIntegrationConfig(row.id); if (!loaded) continue; if (row.type === "proxmox") { const nodes = await createProxmoxAdapter(loaded.config as any).listNodeStats(); snapshot.proxmox.push({ integrationId: row.id, integrationName: row.name, nodes }); for (const n of nodes) if (n.error) held.add(`proxmox:${row.id}:${n.node}`); if (nodes.length > 0 && nodes.every((n) => n.error)) held.add(`proxmox:${row.id}`); } else { const storage = await createSynologyAdapter(loaded.config as any).getStorageInfo(); snapshot.synology.push({ integrationId: row.id, integrationName: row.name, storage }); } } catch (err) { console.error(`[health] couldn't read ${row.type} integration ${row.id}:`, err instanceof Error ? err.message : err); held.add(`${row.type}:${row.id}`); } } return { snapshot, held }; } // ─── The scheduled pass ───────────────────────────────────────────────────── async function loadState(): Promise { const raw = await getInternalFlag(STATE_FLAG); if (!raw) return {}; try { return JSON.parse(raw); } catch { return {}; } } export async function runHealthCheck(now: number = Date.now()): Promise<{ added: number; resolved: number }> { const { healthChecks } = await getSettings(); const { snapshot, held } = await collectSnapshot(); const inGrace = now - PROCESS_START < STARTUP_GRACE_MS; if (inGrace) held.add("server"); const current = evaluateHealth(snapshot, healthChecks, now, { skipServers: inGrace }); const { added, resolved, next } = diffConditions(await loadState(), current, held, await activeSubjects(new Date(now))); await setInternalFlag(STATE_FLAG, JSON.stringify(next)); // State is tracked even when the alert toggle is off (the notify functions check it), // so turning alerts back on doesn't dump every long-standing condition at once. if (added.length > 0) await notifyHealthIssues(added); if (resolved.length > 0) await notifyHealthRecovered(resolved); return { added: added.length, resolved: resolved.length }; }