One list of the current problems across servers and integrations, instead of waiting for a notification or visiting each page: servers that stopped reporting, full or nearly full disks and volumes (critical from 95%), Synology volume/disk problems, failed or uncovered Proxmox backups, failed Proxmox Backup Server verifications, container image updates, expired or expiring secrets/domains/Tailscale keys, failed Semaphore and Gitea runs, Uptime Kuma monitors that are down, overdue osTicket tickets, and integrations whose calls keep failing. Visible to every role, with severity and kind filters, search, sorting, CSV export and "Check now". It runs the same checks that send the notifications rather than a second copy of them: the detection in the health, automation, Proxmox backup, PBS, Docker update and Tailscale key checks is pulled out into shared collectors that both the schedulers and the page call, so the two can't disagree about what counts as a problem. Notification behaviour is unchanged, including the scheduled backup checks skipping integrations under a maintenance window. Unlike the notifications the page ignores the on/off toggles, and keeps problems under a maintenance window, marked silenced and counted apart. It reads live, so a result is reused for a minute (and Refresh can't re-run everything more than once every ten seconds), and every source has a 20 s limit so one hung integration can't hang the page. Anything it couldn't read is called out at the top instead of looking like all clear, and server checks pause for the same 20 minutes after a restart as the notifications do, with a note saying so. Also gives the newer integrations (PBS, osTicket, Uptime Kuma, phpIPAM) proper names in "integration down" notifications instead of their ids. Verified through the real routes against a scratch database with fake backends (offline and full-disk servers, secrets and domains, a silenced server, a fake PBS with failed verification, a hanging integration, a refused one, a failing-calls streak, caching, the restart grace period, auth), and by rendering the real page against that data in a browser: filters, search, silenced toggle, sorting, Check now, dark mode. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
51 lines
2.4 KiB
TypeScript
51 lines
2.4 KiB
TypeScript
import { getSettings } from "./settingsStore.js";
|
|
import { isSourceInMaintenance } from "./maintenance.js";
|
|
import { notifyIntegrationDown, notifyIntegrationRecovered } from "./notify.js";
|
|
|
|
interface SourceHealth {
|
|
consecutiveFailures: number;
|
|
/** True once a "down" notification has been sent for the current failure streak, so it isn't repeated on every subsequent failure. */
|
|
alerted: boolean;
|
|
}
|
|
|
|
/**
|
|
* In-memory only (like the diag log's own ring buffer — resets on restart,
|
|
* which is fine since this tracks a live streak, not history). Keyed by diag
|
|
* log "source" (e.g. "proxmox", "cloudflare") — the same granularity the
|
|
* Diagnostic Log itself filters by, so a homelab running two integrations of
|
|
* the same type shares one health streak between them.
|
|
*/
|
|
const health = new Map<string, SourceHealth>();
|
|
|
|
/** Services whose most recent calls have been failing right now, for the Alerts page. `alerted` means a "down" notification has gone out for the streak. */
|
|
export function getFailingSources(): { source: string; consecutiveFailures: number; alerted: boolean }[] {
|
|
return [...health.entries()]
|
|
.filter(([, state]) => state.consecutiveFailures > 0)
|
|
.map(([source, state]) => ({ source, consecutiveFailures: state.consecutiveFailures, alerted: state.alerted }));
|
|
}
|
|
|
|
/** Called after every diagnostic-log entry is recorded, to track consecutive failures per source and alert on threshold-cross / recovery. */
|
|
export async function trackIntegrationHealth(source: string, ok: boolean): Promise<void> {
|
|
const state = health.get(source) ?? { consecutiveFailures: 0, alerted: false };
|
|
|
|
if (ok) {
|
|
if (state.alerted) {
|
|
await notifyIntegrationRecovered(source);
|
|
}
|
|
health.set(source, { consecutiveFailures: 0, alerted: false });
|
|
return;
|
|
}
|
|
|
|
state.consecutiveFailures += 1;
|
|
const { notifications } = await getSettings();
|
|
if (notifications.integrationFailureAlerts && !state.alerted && state.consecutiveFailures >= notifications.integrationFailureThreshold) {
|
|
// During maintenance the failures keep being counted but `alerted` stays false, so if the service is
|
|
// still failing once the window ends, the very next failed call alerts — a real outage isn't swallowed.
|
|
if (!(await isSourceInMaintenance(source))) {
|
|
state.alerted = true;
|
|
await notifyIntegrationDown(source, state.consecutiveFailures);
|
|
}
|
|
}
|
|
health.set(source, state);
|
|
}
|