One list of the current problems across servers and integrations, instead of waiting for a notification or visiting each page: servers that stopped reporting, full or nearly full disks and volumes (critical from 95%), Synology volume/disk problems, failed or uncovered Proxmox backups, failed Proxmox Backup Server verifications, container image updates, expired or expiring secrets/domains/Tailscale keys, failed Semaphore and Gitea runs, Uptime Kuma monitors that are down, overdue osTicket tickets, and integrations whose calls keep failing. Visible to every role, with severity and kind filters, search, sorting, CSV export and "Check now". It runs the same checks that send the notifications rather than a second copy of them: the detection in the health, automation, Proxmox backup, PBS, Docker update and Tailscale key checks is pulled out into shared collectors that both the schedulers and the page call, so the two can't disagree about what counts as a problem. Notification behaviour is unchanged, including the scheduled backup checks skipping integrations under a maintenance window. Unlike the notifications the page ignores the on/off toggles, and keeps problems under a maintenance window, marked silenced and counted apart. It reads live, so a result is reused for a minute (and Refresh can't re-run everything more than once every ten seconds), and every source has a 20 s limit so one hung integration can't hang the page. Anything it couldn't read is called out at the top instead of looking like all clear, and server checks pause for the same 20 minutes after a restart as the notifications do, with a note saying so. Also gives the newer integrations (PBS, osTicket, Uptime Kuma, phpIPAM) proper names in "integration down" notifications instead of their ids. Verified through the real routes against a scratch database with fake backends (offline and full-disk servers, secrets and domains, a silenced server, a fake PBS with failed verification, a hanging integration, a refused one, a failing-calls streak, caching, the restart grace period, auth), and by rendering the real page against that data in a browser: filters, search, silenced toggle, sorting, Check now, dark mode. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
338 lines
15 KiB
TypeScript
338 lines
15 KiB
TypeScript
import { eq } from "drizzle-orm";
|
|
import { db } from "../db/client.js";
|
|
import { integrations, servers } from "../db/schema.js";
|
|
import { loadIntegrationConfig } from "../integrations/loadIntegration.js";
|
|
import { createProxmoxAdapter, type ProxmoxNodeStats } from "../integrations/proxmox/adapter.js";
|
|
import { createSynologyAdapter, type SynologyStorageInfo } from "../integrations/synology/adapter.js";
|
|
import { notifyHealthIssues, notifyHealthRecovered } from "./notify.js";
|
|
import { getInternalFlag, getSettings, setInternalFlag } from "./settingsStore.js";
|
|
import { activeSubjects, subjectOfConditionKey } from "./maintenance.js";
|
|
|
|
const STATE_FLAG = "healthActiveConditions";
|
|
/** After a restart the agents haven't had a chance to report yet (they run every 15 min), so server-derived conditions are held rather than judged. */
|
|
const STARTUP_GRACE_MS = 20 * 60 * 1000;
|
|
const PROCESS_START = Date.now();
|
|
|
|
// ─── Types ──────────────────────────────────────────────────────────────────
|
|
|
|
export interface HealthCondition {
|
|
/** Stable identity of the problem — the same problem always produces the same key, so it alerts once, not every run. */
|
|
key: string;
|
|
/** Where the data came from, hierarchically ("server", "proxmox:3", "proxmox:3:pve1"). Used to hold a condition when its source can't be reached. */
|
|
source: string;
|
|
message: string;
|
|
/** How bad, for the Alerts page. Notifications don't use it. Left out means "warning". */
|
|
severity?: "critical" | "warning";
|
|
}
|
|
|
|
export interface ServerSnapshot {
|
|
id: number;
|
|
name: string;
|
|
lastSeenAt: string | null;
|
|
disks: { mount: string; sizeBytes: number; usedBytes: number }[];
|
|
}
|
|
export interface ProxmoxSnapshot {
|
|
integrationId: number;
|
|
integrationName: string;
|
|
nodes: ProxmoxNodeStats[];
|
|
}
|
|
export interface SynologySnapshot {
|
|
integrationId: number;
|
|
integrationName: string;
|
|
storage: SynologyStorageInfo;
|
|
}
|
|
export interface HealthSnapshot {
|
|
servers: ServerSnapshot[];
|
|
proxmox: ProxmoxSnapshot[];
|
|
synology: SynologySnapshot[];
|
|
}
|
|
export interface Thresholds {
|
|
serverOfflineMinutes: number;
|
|
diskUsagePercent: number;
|
|
}
|
|
|
|
// ─── Formatting helpers ─────────────────────────────────────────────────────
|
|
|
|
function formatBytes(bytes: number): string {
|
|
const units = ["B", "KB", "MB", "GB", "TB", "PB"];
|
|
let value = bytes;
|
|
let unit = 0;
|
|
while (value >= 1024 && unit < units.length - 1) {
|
|
value /= 1024;
|
|
unit++;
|
|
}
|
|
return `${value.toFixed(value >= 100 || unit === 0 ? 0 : 1)} ${units[unit]}`;
|
|
}
|
|
|
|
function formatDuration(ms: number): string {
|
|
const minutes = Math.floor(ms / 60_000);
|
|
if (minutes < 60) return `${minutes} min`;
|
|
const hours = Math.floor(minutes / 60);
|
|
if (hours < 48) return `${hours} h`;
|
|
return `${Math.floor(hours / 24)} days`;
|
|
}
|
|
|
|
/** Timestamps this app writes are ISO strings, but a bare SQLite "YYYY-MM-DD HH:MM:SS" (UTC, no zone) must not be read as local time. */
|
|
function parseTimestamp(value: string): number {
|
|
return Date.parse(/[zZ]|[+-]\d{2}:?\d{2}$/.test(value) ? value : `${value.replace(" ", "T")}Z`);
|
|
}
|
|
|
|
function usage(used: number | null, total: number | null): number | null {
|
|
if (used === null || total === null || total <= 0) return null;
|
|
return (used / total) * 100;
|
|
}
|
|
|
|
// ─── Evaluation (pure) ──────────────────────────────────────────────────────
|
|
|
|
/**
|
|
* Turns a snapshot of what every source currently reports into the list of
|
|
* problems that exist right now. Pure — no I/O, no clock other than `now` —
|
|
* so the rules can be tested exactly.
|
|
*/
|
|
export function evaluateHealth(snapshot: HealthSnapshot, thresholds: Thresholds, now: number, opts: { skipServers?: boolean } = {}): HealthCondition[] {
|
|
const out: HealthCondition[] = [];
|
|
const limit = thresholds.diskUsagePercent;
|
|
const pct = (n: number) => `${Math.round(n)}%`;
|
|
/** Over the configured threshold is a warning; practically out of room is critical. */
|
|
const fullness = (p: number): "critical" | "warning" => (p >= 95 ? "critical" : "warning");
|
|
|
|
if (!opts.skipServers) {
|
|
for (const s of snapshot.servers) {
|
|
let offline = false;
|
|
if (s.lastSeenAt) {
|
|
const age = now - parseTimestamp(s.lastSeenAt);
|
|
if (Number.isFinite(age) && age > thresholds.serverOfflineMinutes * 60_000) {
|
|
offline = true;
|
|
out.push({
|
|
key: `offline:server:${s.id}`,
|
|
source: "server",
|
|
message: `${s.name} hasn't reported for ${formatDuration(age)}`,
|
|
severity: "critical",
|
|
});
|
|
}
|
|
}
|
|
// A server that never reported has no agent to go quiet; and an offline one's disk figures are stale, so neither is judged on disks.
|
|
if (s.lastSeenAt && !offline) {
|
|
for (const d of s.disks) {
|
|
const p = usage(d.usedBytes, d.sizeBytes);
|
|
if (p !== null && p >= limit) {
|
|
out.push({
|
|
key: `disk:server:${s.id}:${d.mount}`,
|
|
source: "server",
|
|
message: `${s.name}: ${d.mount} is ${pct(p)} full (${formatBytes(d.usedBytes)} of ${formatBytes(d.sizeBytes)})`,
|
|
severity: fullness(p),
|
|
});
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
for (const px of snapshot.proxmox) {
|
|
const seenShared = new Set<string>();
|
|
for (const node of px.nodes) {
|
|
if (node.error) continue; // couldn't be read — held by the caller, not judged
|
|
const rootP = usage(node.rootfsUsedBytes, node.rootfsTotalBytes);
|
|
if (rootP !== null && rootP >= limit) {
|
|
out.push({
|
|
key: `disk:proxmox:${px.integrationId}:${node.node}:rootfs`,
|
|
source: `proxmox:${px.integrationId}:${node.node}`,
|
|
message: `${px.integrationName} / ${node.node}: root filesystem is ${pct(rootP)} full`,
|
|
severity: fullness(rootP),
|
|
});
|
|
}
|
|
for (const st of node.storages) {
|
|
if (!st.active) continue;
|
|
const p = usage(st.usedBytes, st.totalBytes);
|
|
if (p === null || p < limit) continue;
|
|
if (st.shared) {
|
|
// A shared storage is listed by every node — report it once, not once per node.
|
|
if (seenShared.has(st.id)) continue;
|
|
seenShared.add(st.id);
|
|
out.push({
|
|
key: `disk:proxmox:${px.integrationId}:storage:${st.id}`,
|
|
source: `proxmox:${px.integrationId}:shared`,
|
|
message: `${px.integrationName}: shared storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`,
|
|
severity: fullness(p),
|
|
});
|
|
} else {
|
|
out.push({
|
|
key: `disk:proxmox:${px.integrationId}:${node.node}:storage:${st.id}`,
|
|
source: `proxmox:${px.integrationId}:${node.node}`,
|
|
message: `${px.integrationName} / ${node.node}: storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`,
|
|
severity: fullness(p),
|
|
});
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
for (const syn of snapshot.synology) {
|
|
const src = `synology:${syn.integrationId}`;
|
|
for (const v of syn.storage.volumes) {
|
|
if (v.status && v.status.toLowerCase() !== "normal") {
|
|
out.push({
|
|
key: `synology-volume:${syn.integrationId}:${v.id}`,
|
|
source: src,
|
|
message: `${syn.integrationName}: volume ${v.id} status is "${v.status}"`,
|
|
severity: "critical",
|
|
});
|
|
}
|
|
const p = usage(v.sizeUsed, v.sizeTotal);
|
|
if (p !== null && p >= limit) {
|
|
out.push({
|
|
key: `disk:synology:${syn.integrationId}:${v.id}`,
|
|
source: src,
|
|
message: `${syn.integrationName}: volume ${v.id} is ${pct(p)} full (${formatBytes(v.sizeUsed ?? 0)} of ${formatBytes(v.sizeTotal ?? 0)})`,
|
|
severity: fullness(p),
|
|
});
|
|
}
|
|
}
|
|
for (const d of syn.storage.disks) {
|
|
const problems: string[] = [];
|
|
if (d.status && d.status.toLowerCase() !== "normal") problems.push(`status "${d.status}"`);
|
|
if (/warn|crit|fail|danger|bad/i.test(d.smartStatus ?? "")) problems.push(`SMART "${d.smartStatus}"`);
|
|
if (d.exceedBadSectorThreshold) problems.push("bad sectors above threshold");
|
|
if (d.belowRemainLifeThreshold) problems.push("remaining life below threshold");
|
|
if (problems.length > 0) {
|
|
out.push({
|
|
key: `synology-disk:${syn.integrationId}:${d.id}`,
|
|
source: src,
|
|
message: `${syn.integrationName}: disk ${d.name || d.id} — ${problems.join(", ")}`,
|
|
// A disk that's no longer "normal" is failing; a SMART warning or a threshold crossing is the early notice.
|
|
severity: d.status && d.status.toLowerCase() !== "normal" ? "critical" : "warning",
|
|
});
|
|
}
|
|
}
|
|
}
|
|
|
|
return out;
|
|
}
|
|
|
|
// ─── State diff (pure) ──────────────────────────────────────────────────────
|
|
|
|
export type ActiveState = Record<string, { message: string; source: string }>;
|
|
|
|
/** `held` is a set of hierarchical sources that couldn't be read this run; matches the source itself or anything beneath it. */
|
|
function isHeld(source: string, held: Set<string>): boolean {
|
|
for (const h of held) if (source === h || source.startsWith(`${h}:`)) return true;
|
|
return false;
|
|
}
|
|
|
|
/**
|
|
* Compares this run's conditions with what was already being reported.
|
|
* A condition present now but not before is new; one present before but not
|
|
* now is resolved — except when its source couldn't be read this run, in
|
|
* which case it's carried over untouched. Without that, one failed poll would
|
|
* report a recovery and then re-alert the same problem on the next poll.
|
|
*/
|
|
export function diffConditions(
|
|
previous: ActiveState,
|
|
current: HealthCondition[],
|
|
held: Set<string> = new Set(),
|
|
/** Subjects ("server:3", "integration:1") under an active maintenance window. */
|
|
silenced: Set<string> = new Set(),
|
|
): { added: HealthCondition[]; resolved: { key: string; message: string }[]; next: ActiveState } {
|
|
const isSilenced = (key: string) => {
|
|
const subject = subjectOfConditionKey(key);
|
|
return subject !== null && silenced.has(subject);
|
|
};
|
|
// A silenced problem is not recorded as "known": if it's still there when the window ends it must
|
|
// alert then, as new. (One that was already alerted before the window stays known, so it isn't repeated.)
|
|
const visible = current.filter((c) => !isSilenced(c.key));
|
|
|
|
const next: ActiveState = {};
|
|
for (const c of visible) next[c.key] = { message: c.message, source: c.source };
|
|
|
|
const added = visible.filter((c) => !(c.key in previous));
|
|
const resolved: { key: string; message: string }[] = [];
|
|
for (const [key, entry] of Object.entries(previous)) {
|
|
if (key in next) continue;
|
|
// Held (source unreadable) or silenced: leave it exactly as it was — neither cleared nor re-alerted.
|
|
if (isHeld(entry.source, held) || isSilenced(key)) {
|
|
next[key] = entry;
|
|
} else {
|
|
resolved.push({ key, message: entry.message });
|
|
}
|
|
}
|
|
return { added, resolved, next };
|
|
}
|
|
|
|
// ─── Collection (I/O) ───────────────────────────────────────────────────────
|
|
|
|
/** How much longer, in ms, servers are exempt from being judged after a restart (their agents get a chance to report first). 0 once it's over. */
|
|
export function startupGraceRemainingMs(now: number = Date.now()): number {
|
|
return Math.max(0, STARTUP_GRACE_MS - (now - PROCESS_START));
|
|
}
|
|
|
|
export async function collectSnapshot(): Promise<{ snapshot: HealthSnapshot; held: Set<string> }> {
|
|
const held = new Set<string>();
|
|
const snapshot: HealthSnapshot = { servers: [], proxmox: [], synology: [] };
|
|
|
|
for (const s of await db.select().from(servers)) {
|
|
let disks: ServerSnapshot["disks"] = [];
|
|
try {
|
|
disks = s.disks ? JSON.parse(s.disks) : [];
|
|
} catch {
|
|
// a malformed report shouldn't stop every other server being checked
|
|
}
|
|
snapshot.servers.push({ id: s.id, name: s.name, lastSeenAt: s.lastSeenAt, disks });
|
|
}
|
|
|
|
const rows = await db
|
|
.select({ id: integrations.id, name: integrations.name, type: integrations.type })
|
|
.from(integrations)
|
|
.where(eq(integrations.enabled, true));
|
|
|
|
for (const row of rows) {
|
|
if (row.type !== "proxmox" && row.type !== "synology") continue;
|
|
try {
|
|
const loaded = await loadIntegrationConfig(row.id);
|
|
if (!loaded) continue;
|
|
if (row.type === "proxmox") {
|
|
const nodes = await createProxmoxAdapter(loaded.config as any).listNodeStats();
|
|
snapshot.proxmox.push({ integrationId: row.id, integrationName: row.name, nodes });
|
|
for (const n of nodes) if (n.error) held.add(`proxmox:${row.id}:${n.node}`);
|
|
if (nodes.length > 0 && nodes.every((n) => n.error)) held.add(`proxmox:${row.id}`);
|
|
} else {
|
|
const storage = await createSynologyAdapter(loaded.config as any).getStorageInfo();
|
|
snapshot.synology.push({ integrationId: row.id, integrationName: row.name, storage });
|
|
}
|
|
} catch (err) {
|
|
console.error(`[health] couldn't read ${row.type} integration ${row.id}:`, err instanceof Error ? err.message : err);
|
|
held.add(`${row.type}:${row.id}`);
|
|
}
|
|
}
|
|
return { snapshot, held };
|
|
}
|
|
|
|
// ─── The scheduled pass ─────────────────────────────────────────────────────
|
|
|
|
async function loadState(): Promise<ActiveState> {
|
|
const raw = await getInternalFlag(STATE_FLAG);
|
|
if (!raw) return {};
|
|
try {
|
|
return JSON.parse(raw);
|
|
} catch {
|
|
return {};
|
|
}
|
|
}
|
|
|
|
export async function runHealthCheck(now: number = Date.now()): Promise<{ added: number; resolved: number }> {
|
|
const { healthChecks } = await getSettings();
|
|
const { snapshot, held } = await collectSnapshot();
|
|
|
|
const inGrace = now - PROCESS_START < STARTUP_GRACE_MS;
|
|
if (inGrace) held.add("server");
|
|
|
|
const current = evaluateHealth(snapshot, healthChecks, now, { skipServers: inGrace });
|
|
const { added, resolved, next } = diffConditions(await loadState(), current, held, await activeSubjects(new Date(now)));
|
|
await setInternalFlag(STATE_FLAG, JSON.stringify(next));
|
|
|
|
// State is tracked even when the alert toggle is off (the notify functions check it),
|
|
// so turning alerts back on doesn't dump every long-standing condition at once.
|
|
if (added.length > 0) await notifyHealthIssues(added);
|
|
if (resolved.length > 0) await notifyHealthRecovered(resolved);
|
|
return { added: added.length, resolved: resolved.length };
|
|
}
|