diff --git a/README.md b/README.md index 2670842..4aaba9a 100644 --- a/README.md +++ b/README.md @@ -135,6 +135,13 @@ real API response (fixed to derive them from `connectedToControl` and `enabledRoutes`). See the git log for the full verification notes per integration. +Server and storage health is watched every 15 minutes: a server whose agent +stops reporting, a server disk / Proxmox storage / Synology volume passing a +usage threshold, and a Synology volume or disk that's degraded or failing each +raise one notification when the problem starts and one when it clears (both +thresholds are set under Settings → Notifications). Active problems are +remembered across restarts, so a rebuild doesn't re-alert them. + The app is installable as a PWA — "Install app" / "Add to Home Screen" from the browser gives it its own icon and a standalone window on phone or desktop. This needs the site to be served over HTTPS (browsers only offer install on secure diff --git a/server/src/index.ts b/server/src/index.ts index 10a1149..ea033a0 100644 --- a/server/src/index.ts +++ b/server/src/index.ts @@ -29,6 +29,7 @@ import { initLogRetentionScheduler } from "./services/logRetentionScheduler.js"; import { initDockerUpdateScheduler } from "./services/dockerUpdateScheduler.js"; import { initProxmoxBackupScheduler } from "./services/proxmoxBackupScheduler.js"; import { initQuietHoursScheduler } from "./services/quietHoursScheduler.js"; +import { initHealthScheduler } from "./services/healthScheduler.js"; warnIfAuthNotConfigured(); await runMigrations(); @@ -38,6 +39,7 @@ await initLogRetentionScheduler(); await initDockerUpdateScheduler(); await initProxmoxBackupScheduler(); await initQuietHoursScheduler(); +await initHealthScheduler(); const __dirname = dirname(fileURLToPath(import.meta.url)); const webDist = join(__dirname, "..", "..", "web", "dist"); diff --git a/server/src/routes/settings.ts b/server/src/routes/settings.ts index 160edd0..9342402 100644 --- a/server/src/routes/settings.ts +++ b/server/src/routes/settings.ts @@ -68,6 +68,7 @@ const updateSchema = z.object({ tailscaleKeyCheck: z.boolean(), dockerUpdateCheck: z.boolean(), proxmoxBackupCheck: z.boolean(), + healthAlerts: z.boolean(), secretCheckTime: z.string().regex(/^\d{2}:\d{2}$/), timezone: z.string(), integrationFailureAlerts: z.boolean(), @@ -85,6 +86,10 @@ const updateSchema = z.object({ .object({ enabled: z.boolean(), retentionDays: z.number().int().min(1).max(3650), intervalHours: z.number().int().min(1).max(720) }) .partial() .optional(), + healthChecks: z + .object({ serverOfflineMinutes: z.number().int().min(15).max(10080), diskUsagePercent: z.number().int().min(50).max(99) }) + .partial() + .optional(), quietHours: z .object({ enabled: z.boolean(), start: z.string().regex(/^\d{2}:\d{2}$/), end: z.string().regex(/^\d{2}:\d{2}$/) }) .partial() diff --git a/server/src/services/healthMonitor.ts b/server/src/services/healthMonitor.ts new file mode 100644 index 0000000..9cf968a --- /dev/null +++ b/server/src/services/healthMonitor.ts @@ -0,0 +1,307 @@ +import { eq } from "drizzle-orm"; +import { db } from "../db/client.js"; +import { integrations, servers } from "../db/schema.js"; +import { loadIntegrationConfig } from "../integrations/loadIntegration.js"; +import { createProxmoxAdapter, type ProxmoxNodeStats } from "../integrations/proxmox/adapter.js"; +import { createSynologyAdapter, type SynologyStorageInfo } from "../integrations/synology/adapter.js"; +import { notifyHealthIssues, notifyHealthRecovered } from "./notify.js"; +import { getInternalFlag, getSettings, setInternalFlag } from "./settingsStore.js"; + +const STATE_FLAG = "healthActiveConditions"; +/** After a restart the agents haven't had a chance to report yet (they run every 15 min), so server-derived conditions are held rather than judged. */ +const STARTUP_GRACE_MS = 20 * 60 * 1000; +const PROCESS_START = Date.now(); + +// ─── Types ────────────────────────────────────────────────────────────────── + +export interface HealthCondition { + /** Stable identity of the problem — the same problem always produces the same key, so it alerts once, not every run. */ + key: string; + /** Where the data came from, hierarchically ("server", "proxmox:3", "proxmox:3:pve1"). Used to hold a condition when its source can't be reached. */ + source: string; + message: string; +} + +export interface ServerSnapshot { + id: number; + name: string; + lastSeenAt: string | null; + disks: { mount: string; sizeBytes: number; usedBytes: number }[]; +} +export interface ProxmoxSnapshot { + integrationId: number; + integrationName: string; + nodes: ProxmoxNodeStats[]; +} +export interface SynologySnapshot { + integrationId: number; + integrationName: string; + storage: SynologyStorageInfo; +} +export interface HealthSnapshot { + servers: ServerSnapshot[]; + proxmox: ProxmoxSnapshot[]; + synology: SynologySnapshot[]; +} +export interface Thresholds { + serverOfflineMinutes: number; + diskUsagePercent: number; +} + +// ─── Formatting helpers ───────────────────────────────────────────────────── + +function formatBytes(bytes: number): string { + const units = ["B", "KB", "MB", "GB", "TB", "PB"]; + let value = bytes; + let unit = 0; + while (value >= 1024 && unit < units.length - 1) { + value /= 1024; + unit++; + } + return `${value.toFixed(value >= 100 || unit === 0 ? 0 : 1)} ${units[unit]}`; +} + +function formatDuration(ms: number): string { + const minutes = Math.floor(ms / 60_000); + if (minutes < 60) return `${minutes} min`; + const hours = Math.floor(minutes / 60); + if (hours < 48) return `${hours} h`; + return `${Math.floor(hours / 24)} days`; +} + +/** Timestamps this app writes are ISO strings, but a bare SQLite "YYYY-MM-DD HH:MM:SS" (UTC, no zone) must not be read as local time. */ +function parseTimestamp(value: string): number { + return Date.parse(/[zZ]|[+-]\d{2}:?\d{2}$/.test(value) ? value : `${value.replace(" ", "T")}Z`); +} + +function usage(used: number | null, total: number | null): number | null { + if (used === null || total === null || total <= 0) return null; + return (used / total) * 100; +} + +// ─── Evaluation (pure) ────────────────────────────────────────────────────── + +/** + * Turns a snapshot of what every source currently reports into the list of + * problems that exist right now. Pure — no I/O, no clock other than `now` — + * so the rules can be tested exactly. + */ +export function evaluateHealth(snapshot: HealthSnapshot, thresholds: Thresholds, now: number, opts: { skipServers?: boolean } = {}): HealthCondition[] { + const out: HealthCondition[] = []; + const limit = thresholds.diskUsagePercent; + const pct = (n: number) => `${Math.round(n)}%`; + + if (!opts.skipServers) { + for (const s of snapshot.servers) { + let offline = false; + if (s.lastSeenAt) { + const age = now - parseTimestamp(s.lastSeenAt); + if (Number.isFinite(age) && age > thresholds.serverOfflineMinutes * 60_000) { + offline = true; + out.push({ + key: `offline:server:${s.id}`, + source: "server", + message: `${s.name} hasn't reported for ${formatDuration(age)}`, + }); + } + } + // A server that never reported has no agent to go quiet; and an offline one's disk figures are stale, so neither is judged on disks. + if (s.lastSeenAt && !offline) { + for (const d of s.disks) { + const p = usage(d.usedBytes, d.sizeBytes); + if (p !== null && p >= limit) { + out.push({ + key: `disk:server:${s.id}:${d.mount}`, + source: "server", + message: `${s.name}: ${d.mount} is ${pct(p)} full (${formatBytes(d.usedBytes)} of ${formatBytes(d.sizeBytes)})`, + }); + } + } + } + } + } + + for (const px of snapshot.proxmox) { + const seenShared = new Set(); + for (const node of px.nodes) { + if (node.error) continue; // couldn't be read — held by the caller, not judged + const rootP = usage(node.rootfsUsedBytes, node.rootfsTotalBytes); + if (rootP !== null && rootP >= limit) { + out.push({ + key: `disk:proxmox:${px.integrationId}:${node.node}:rootfs`, + source: `proxmox:${px.integrationId}:${node.node}`, + message: `${px.integrationName} / ${node.node}: root filesystem is ${pct(rootP)} full`, + }); + } + for (const st of node.storages) { + if (!st.active) continue; + const p = usage(st.usedBytes, st.totalBytes); + if (p === null || p < limit) continue; + if (st.shared) { + // A shared storage is listed by every node — report it once, not once per node. + if (seenShared.has(st.id)) continue; + seenShared.add(st.id); + out.push({ + key: `disk:proxmox:${px.integrationId}:storage:${st.id}`, + source: `proxmox:${px.integrationId}:shared`, + message: `${px.integrationName}: shared storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`, + }); + } else { + out.push({ + key: `disk:proxmox:${px.integrationId}:${node.node}:storage:${st.id}`, + source: `proxmox:${px.integrationId}:${node.node}`, + message: `${px.integrationName} / ${node.node}: storage "${st.id}" is ${pct(p)} full (${formatBytes(st.usedBytes ?? 0)} of ${formatBytes(st.totalBytes ?? 0)})`, + }); + } + } + } + } + + for (const syn of snapshot.synology) { + const src = `synology:${syn.integrationId}`; + for (const v of syn.storage.volumes) { + if (v.status && v.status.toLowerCase() !== "normal") { + out.push({ + key: `synology-volume:${syn.integrationId}:${v.id}`, + source: src, + message: `${syn.integrationName}: volume ${v.id} status is "${v.status}"`, + }); + } + const p = usage(v.sizeUsed, v.sizeTotal); + if (p !== null && p >= limit) { + out.push({ + key: `disk:synology:${syn.integrationId}:${v.id}`, + source: src, + message: `${syn.integrationName}: volume ${v.id} is ${pct(p)} full (${formatBytes(v.sizeUsed ?? 0)} of ${formatBytes(v.sizeTotal ?? 0)})`, + }); + } + } + for (const d of syn.storage.disks) { + const problems: string[] = []; + if (d.status && d.status.toLowerCase() !== "normal") problems.push(`status "${d.status}"`); + if (/warn|crit|fail|danger|bad/i.test(d.smartStatus ?? "")) problems.push(`SMART "${d.smartStatus}"`); + if (d.exceedBadSectorThreshold) problems.push("bad sectors above threshold"); + if (d.belowRemainLifeThreshold) problems.push("remaining life below threshold"); + if (problems.length > 0) { + out.push({ + key: `synology-disk:${syn.integrationId}:${d.id}`, + source: src, + message: `${syn.integrationName}: disk ${d.name || d.id} — ${problems.join(", ")}`, + }); + } + } + } + + return out; +} + +// ─── State diff (pure) ────────────────────────────────────────────────────── + +export type ActiveState = Record; + +/** `held` is a set of hierarchical sources that couldn't be read this run; matches the source itself or anything beneath it. */ +function isHeld(source: string, held: Set): boolean { + for (const h of held) if (source === h || source.startsWith(`${h}:`)) return true; + return false; +} + +/** + * Compares this run's conditions with what was already being reported. + * A condition present now but not before is new; one present before but not + * now is resolved — except when its source couldn't be read this run, in + * which case it's carried over untouched. Without that, one failed poll would + * report a recovery and then re-alert the same problem on the next poll. + */ +export function diffConditions( + previous: ActiveState, + current: HealthCondition[], + held: Set = new Set(), +): { added: HealthCondition[]; resolved: { key: string; message: string }[]; next: ActiveState } { + const next: ActiveState = {}; + for (const c of current) next[c.key] = { message: c.message, source: c.source }; + + const added = current.filter((c) => !(c.key in previous)); + const resolved: { key: string; message: string }[] = []; + for (const [key, entry] of Object.entries(previous)) { + if (key in next) continue; + if (isHeld(entry.source, held)) { + next[key] = entry; + } else { + resolved.push({ key, message: entry.message }); + } + } + return { added, resolved, next }; +} + +// ─── Collection (I/O) ─────────────────────────────────────────────────────── + +async function collectSnapshot(): Promise<{ snapshot: HealthSnapshot; held: Set }> { + const held = new Set(); + const snapshot: HealthSnapshot = { servers: [], proxmox: [], synology: [] }; + + for (const s of await db.select().from(servers)) { + let disks: ServerSnapshot["disks"] = []; + try { + disks = s.disks ? JSON.parse(s.disks) : []; + } catch { + // a malformed report shouldn't stop every other server being checked + } + snapshot.servers.push({ id: s.id, name: s.name, lastSeenAt: s.lastSeenAt, disks }); + } + + const rows = await db + .select({ id: integrations.id, name: integrations.name, type: integrations.type }) + .from(integrations) + .where(eq(integrations.enabled, true)); + + for (const row of rows) { + if (row.type !== "proxmox" && row.type !== "synology") continue; + try { + const loaded = await loadIntegrationConfig(row.id); + if (!loaded) continue; + if (row.type === "proxmox") { + const nodes = await createProxmoxAdapter(loaded.config as any).listNodeStats(); + snapshot.proxmox.push({ integrationId: row.id, integrationName: row.name, nodes }); + for (const n of nodes) if (n.error) held.add(`proxmox:${row.id}:${n.node}`); + if (nodes.length > 0 && nodes.every((n) => n.error)) held.add(`proxmox:${row.id}`); + } else { + const storage = await createSynologyAdapter(loaded.config as any).getStorageInfo(); + snapshot.synology.push({ integrationId: row.id, integrationName: row.name, storage }); + } + } catch (err) { + console.error(`[health] couldn't read ${row.type} integration ${row.id}:`, err instanceof Error ? err.message : err); + held.add(`${row.type}:${row.id}`); + } + } + return { snapshot, held }; +} + +// ─── The scheduled pass ───────────────────────────────────────────────────── + +async function loadState(): Promise { + const raw = await getInternalFlag(STATE_FLAG); + if (!raw) return {}; + try { + return JSON.parse(raw); + } catch { + return {}; + } +} + +export async function runHealthCheck(now: number = Date.now()): Promise<{ added: number; resolved: number }> { + const { healthChecks } = await getSettings(); + const { snapshot, held } = await collectSnapshot(); + + const inGrace = now - PROCESS_START < STARTUP_GRACE_MS; + if (inGrace) held.add("server"); + + const current = evaluateHealth(snapshot, healthChecks, now, { skipServers: inGrace }); + const { added, resolved, next } = diffConditions(await loadState(), current, held); + await setInternalFlag(STATE_FLAG, JSON.stringify(next)); + + // State is tracked even when the alert toggle is off (the notify functions check it), + // so turning alerts back on doesn't dump every long-standing condition at once. + if (added.length > 0) await notifyHealthIssues(added); + if (resolved.length > 0) await notifyHealthRecovered(resolved); + return { added: added.length, resolved: resolved.length }; +} diff --git a/server/src/services/healthScheduler.ts b/server/src/services/healthScheduler.ts new file mode 100644 index 0000000..f8500ad --- /dev/null +++ b/server/src/services/healthScheduler.ts @@ -0,0 +1,22 @@ +import { runHealthCheck } from "./healthMonitor.js"; + +// 15 minutes matches the agent's default report interval, so a check never sees "stale" data that's just waiting for the next report. +const INTERVAL_MS = 15 * 60 * 1000; + +let timer: ReturnType | null = null; + +async function pass(): Promise { + try { + await runHealthCheck(); + } catch (err) { + console.error("[health] check failed:", err); + } +} + +export async function initHealthScheduler(): Promise { + if (timer) return; + timer = setInterval(() => void pass(), INTERVAL_MS); + // Not awaited: the first pass talks to Proxmox/Synology and must not delay server startup. + void pass(); + console.log(`[health] checking server/storage health every ${INTERVAL_MS / 60000} minutes`); +} diff --git a/server/src/services/notify.ts b/server/src/services/notify.ts index 70b0693..55dbb9b 100644 --- a/server/src/services/notify.ts +++ b/server/src/services/notify.ts @@ -277,6 +277,24 @@ export async function notifyProxmoxUncoveredGuests( ); } +export async function notifyHealthIssues(issues: { message: string }[]): Promise { + if (issues.length === 0) return; + if (!(await eventEnabled("healthAlerts"))) return; + await notify( + "Homelab Manager — Health Alert", + `${issues.length} new issue${issues.length !== 1 ? "s" : ""}:\n\n${issues.map((i) => `⚠ ${i.message}`).join("\n")}`, + ); +} + +export async function notifyHealthRecovered(resolved: { message: string }[]): Promise { + if (resolved.length === 0) return; + if (!(await eventEnabled("healthAlerts"))) return; + await notify( + "Homelab Manager — Health Recovered", + `${resolved.length} issue${resolved.length !== 1 ? "s" : ""} cleared:\n\n${resolved.map((i) => `✓ ${i.message}`).join("\n")}`, + ); +} + export async function notifyTailscaleKeyExpiry( expiring: { integrationName: string; deviceLabel: string; daysLeft: number }[], ): Promise { diff --git a/server/src/services/settingsStore.ts b/server/src/services/settingsStore.ts index fc5bbe3..161fdfc 100644 --- a/server/src/services/settingsStore.ts +++ b/server/src/services/settingsStore.ts @@ -42,6 +42,7 @@ export interface NotificationEvents { tailscaleKeyCheck: boolean; dockerUpdateCheck: boolean; proxmoxBackupCheck: boolean; + healthAlerts: boolean; secretCheckTime: string; // "HH:MM" — shared by the secret-expiry, Tailscale key-expiry, Docker update, and Proxmox backup checks timezone: string; integrationFailureAlerts: boolean; @@ -78,6 +79,13 @@ export interface QuietHoursSettings { end: string; } +export interface HealthCheckSettings { + /** A server whose agent hasn't reported for this long is considered offline (agent default interval is 15 min). */ + serverOfflineMinutes: number; + /** Disk / storage / volume usage at or above this percentage is reported. */ + diskUsagePercent: number; +} + export interface AppSettings { gotify: GotifySettings; ntfy: NtfySettings; @@ -89,6 +97,7 @@ export interface AppSettings { display: DisplaySettings; logRetention: LogRetentionSettings; quietHours: QuietHoursSettings; + healthChecks: HealthCheckSettings; } const DEFAULTS: AppSettings = { @@ -104,6 +113,7 @@ const DEFAULTS: AppSettings = { tailscaleKeyCheck: true, dockerUpdateCheck: true, proxmoxBackupCheck: true, + healthAlerts: true, secretCheckTime: "08:00", timezone: "UTC", integrationFailureAlerts: true, @@ -114,6 +124,7 @@ const DEFAULTS: AppSettings = { display: { dateFormat: "ymd", timeFormat: "24h", pageSize: 20 }, logRetention: { enabled: false, retentionDays: 90, intervalHours: 24 }, quietHours: { enabled: false, start: "22:00", end: "07:00" }, + healthChecks: { serverOfflineMinutes: 60, diskUsagePercent: 90 }, }; const KEYS = Object.keys(DEFAULTS) as (keyof AppSettings)[]; diff --git a/web/src/api/client.ts b/web/src/api/client.ts index b9fec51..7cc8ab4 100644 --- a/web/src/api/client.ts +++ b/web/src/api/client.ts @@ -162,6 +162,7 @@ export interface NotificationEvents { tailscaleKeyCheck: boolean; dockerUpdateCheck: boolean; proxmoxBackupCheck: boolean; + healthAlerts: boolean; secretCheckTime: string; timezone: string; integrationFailureAlerts: boolean; @@ -183,6 +184,11 @@ export interface LogRetentionSettings { intervalHours: number; } +export interface HealthCheckSettings { + serverOfflineMinutes: number; + diskUsagePercent: number; +} + export interface QuietHoursSettings { enabled: boolean; start: string; @@ -200,6 +206,7 @@ export interface AppSettings { display: DisplaySettings; logRetention: LogRetentionSettings; quietHours: QuietHoursSettings; + healthChecks: HealthCheckSettings; } export type AppSettingsPatch = { [K in keyof AppSettings]?: Partial }; diff --git a/web/src/pages/settings/NotificationSettings.tsx b/web/src/pages/settings/NotificationSettings.tsx index aeaffff..2aa41f6 100644 --- a/web/src/pages/settings/NotificationSettings.tsx +++ b/web/src/pages/settings/NotificationSettings.tsx @@ -8,6 +8,7 @@ import { type WebhookSettings, type NotificationEvents, type QuietHoursSettings, + type HealthCheckSettings, } from "../../api/client"; const TIMEZONES = [ @@ -45,12 +46,14 @@ const DEFAULT_NOTIFICATIONS: NotificationEvents = { tailscaleKeyCheck: true, dockerUpdateCheck: true, proxmoxBackupCheck: true, + healthAlerts: true, secretCheckTime: "08:00", timezone: "UTC", integrationFailureAlerts: true, integrationFailureThreshold: 3, }; const DEFAULT_QUIET_HOURS: QuietHoursSettings = { enabled: false, start: "22:00", end: "07:00" }; +const DEFAULT_HEALTH_CHECKS: HealthCheckSettings = { serverOfflineMinutes: 60, diskUsagePercent: 90 }; type TestResult = { ok: boolean; message: string } | null; @@ -76,6 +79,7 @@ export default function NotificationSettings() { const [webhook, setWebhook] = useState(DEFAULT_WEBHOOK); const [notifications, setNotifications] = useState(DEFAULT_NOTIFICATIONS); const [quietHours, setQuietHours] = useState(DEFAULT_QUIET_HOURS); + const [healthChecks, setHealthChecks] = useState(DEFAULT_HEALTH_CHECKS); const [queuedCount, setQueuedCount] = useState(null); const [flushing, setFlushing] = useState(false); const [flushResult, setFlushResult] = useState(null); @@ -99,6 +103,7 @@ export default function NotificationSettings() { setWebhook(res.settings.webhook); setNotifications(res.settings.notifications); setQuietHours(res.settings.quietHours); + setHealthChecks(res.settings.healthChecks); }) .catch((err) => setLoadError(err instanceof Error ? err.message : String(err))) .finally(() => setLoading(false)); @@ -113,7 +118,7 @@ export default function NotificationSettings() { setSaveError(null); setSaved(false); try { - await api.settings.update({ gotify, ntfy, smtp, webhook, notifications, quietHours }); + await api.settings.update({ gotify, ntfy, smtp, webhook, notifications, quietHours, healthChecks }); setSaved(true); setTimeout(() => setSaved(false), 3000); } catch (err) { @@ -520,6 +525,7 @@ export default function NotificationSettings() { { key: "tailscaleKeyCheck" as const, label: "Tailscale key expiry reminder" }, { key: "dockerUpdateCheck" as const, label: "Docker image update available" }, { key: "proxmoxBackupCheck" as const, label: "Proxmox backup failed or a guest has no coverage" }, + { key: "healthAlerts" as const, label: "Server offline, disk nearly full, or Synology volume/disk problem" }, { key: "integrationFailureAlerts" as const, label: "Integration/DNS provider failing repeatedly" }, ].map(({ key, label }) => (