import { getSettings } from "./settingsStore.js"; import { isSourceInMaintenance } from "./maintenance.js"; import { notifyIntegrationDown, notifyIntegrationRecovered } from "./notify.js"; interface SourceHealth { consecutiveFailures: number; /** True once a "down" notification has been sent for the current failure streak, so it isn't repeated on every subsequent failure. */ alerted: boolean; } /** * In-memory only (like the diag log's own ring buffer — resets on restart, * which is fine since this tracks a live streak, not history). Keyed by diag * log "source" (e.g. "proxmox", "cloudflare") — the same granularity the * Diagnostic Log itself filters by, so a homelab running two integrations of * the same type shares one health streak between them. */ const health = new Map(); /** Services whose most recent calls have been failing right now, for the Alerts page. `alerted` means a "down" notification has gone out for the streak. */ export function getFailingSources(): { source: string; consecutiveFailures: number; alerted: boolean }[] { return [...health.entries()] .filter(([, state]) => state.consecutiveFailures > 0) .map(([source, state]) => ({ source, consecutiveFailures: state.consecutiveFailures, alerted: state.alerted })); } /** Called after every diagnostic-log entry is recorded, to track consecutive failures per source and alert on threshold-cross / recovery. */ export async function trackIntegrationHealth(source: string, ok: boolean): Promise { const state = health.get(source) ?? { consecutiveFailures: 0, alerted: false }; if (ok) { if (state.alerted) { await notifyIntegrationRecovered(source); } health.set(source, { consecutiveFailures: 0, alerted: false }); return; } state.consecutiveFailures += 1; const { notifications } = await getSettings(); if (notifications.integrationFailureAlerts && !state.alerted && state.consecutiveFailures >= notifications.integrationFailureThreshold) { // During maintenance the failures keep being counted but `alerted` stays false, so if the service is // still failing once the window ends, the very next failed call alerts — a real outage isn't swallowed. if (!(await isSourceInMaintenance(source))) { state.alerted = true; await notifyIntegrationDown(source, state.consecutiveFailures); } } health.set(source, state); }