Alert when an integration or DNS provider fails repeatedly

The Diagnostic Log already records every outbound call's success or
failure, but nothing acted on it — you'd only notice an integration
was down by happening to open its page. Adds a per-source consecutive-
failure counter (in-memory, reset on restart, same durability tier as
the diag log's own ring buffer) hooked into recordDiagEntry: crossing
the configurable threshold (default 3) sends one "down" notification
on every configured channel, and a "recovered" notification fires once
it succeeds again — no repeat spam while it stays down. New
"Integration/DNS provider failing repeatedly" toggle and threshold
field under Settings -> Notifications.

Verified end-to-end against an isolated scratch database with a real
local HTTP server standing in for the webhook channel: 5 consecutive
failures produced exactly one "Down" notification (at the 3rd
failure, correctly naming "3 calls"), a subsequent success produced
exactly one "Recovered" notification, and two more failures on a
fresh streak triggered nothing (below threshold) — confirmed the real
dev database's mtime was untouched throughout.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
bobbanandClaude Sonnet 5 committed 2026-09-19 14:51:00 +02:00
1 parent 10d123b18a
commit b5a4c6e2d9
7 files changed
+105

No files matched your search

+2
View File
@@ -63,6 +63,8 @@ const updateSchema = z.object({
tailscaleKeyCheck: z.boolean(),
secretCheckTime: z.string().regex(/^\d{2}:\d{2}$/),
timezone: z.string(),
integrationFailureAlerts: z.boolean(),
integrationFailureThreshold: z.number().int().min(1).max(20),
})
.partial()
.optional(),
+2
View File
@@ -1,6 +1,7 @@
import { and, desc, eq, lt, sql } from "drizzle-orm";
import { db } from "../db/client.js";
import { diagLog } from "../db/schema.js";
import { trackIntegrationHealth } from "./integrationHealthMonitor.js";
const MAX_ENTRIES = 500;
@@ -15,6 +16,7 @@ async function recordDiagEntry(entry: { source: string; operation: string; ok: b
if (cutoff) {
await db.delete(diagLog).where(lt(diagLog.id, cutoff.id));
}
await trackIntegrationHealth(entry.source, entry.ok);
} catch (err) {
console.error("[diagLog] failed to record entry:", err);
}
@@ -0,0 +1,38 @@
import { getSettings } from "./settingsStore.js";
import { notifyIntegrationDown, notifyIntegrationRecovered } from "./notify.js";
interface SourceHealth {
consecutiveFailures: number;
/** True once a "down" notification has been sent for the current failure streak, so it isn't repeated on every subsequent failure. */
alerted: boolean;
}
/**
* In-memory only (like the diag log's own ring buffer — resets on restart,
* which is fine since this tracks a live streak, not history). Keyed by diag
* log "source" (e.g. "proxmox", "cloudflare") — the same granularity the
* Diagnostic Log itself filters by, so a homelab running two integrations of
* the same type shares one health streak between them.
*/
const health = new Map<string, SourceHealth>();
/** Called after every diagnostic-log entry is recorded, to track consecutive failures per source and alert on threshold-cross / recovery. */
export async function trackIntegrationHealth(source: string, ok: boolean): Promise<void> {
const state = health.get(source) ?? { consecutiveFailures: 0, alerted: false };
if (ok) {
if (state.alerted) {
await notifyIntegrationRecovered(source);
}
health.set(source, { consecutiveFailures: 0, alerted: false });
return;
}
state.consecutiveFailures += 1;
const { notifications } = await getSettings();
if (notifications.integrationFailureAlerts && !state.alerted && state.consecutiveFailures >= notifications.integrationFailureThreshold) {
state.alerted = true;
await notifyIntegrationDown(source, state.consecutiveFailures);
}
health.set(source, state);
}
+33
View File
@@ -191,3 +191,36 @@ export async function notifyTailscaleKeyExpiry(
`${expiring.length} device key${expiring.length !== 1 ? "s" : ""} need attention:\n\n${lines.join("\n")}`,
);
}
// Diagnostic-log "source" values (see integrations/*/adapter.ts and dns/adapters/*.ts) to display names.
const SOURCE_LABELS: Record<string, string> = {
cloudflare: "Cloudflare",
loopia: "Loopia",
pihole: "Pi-hole",
azure: "Azure DNS",
cpanel: "cPanel",
technitium: "Technitium",
tailscale: "Tailscale",
proxmox: "Proxmox",
synology: "Synology",
semaphore: "Semaphore",
gitea: "Gitea",
dockhand: "Dockhand",
};
function sourceLabel(source: string): string {
return SOURCE_LABELS[source] ?? source;
}
export async function notifyIntegrationDown(source: string, consecutiveFailures: number): Promise<void> {
if (!(await eventEnabled("integrationFailureAlerts"))) return;
await notify(
"Homelab Manager — Integration Down",
`${sourceLabel(source)} has failed its last ${consecutiveFailures} call${consecutiveFailures !== 1 ? "s" : ""} in a row. Check the Diagnostic Log for details.`,
);
}
export async function notifyIntegrationRecovered(source: string): Promise<void> {
if (!(await eventEnabled("integrationFailureAlerts"))) return;
await notify("Homelab Manager — Integration Recovered", `${sourceLabel(source)} succeeded again after failing.`);
}
+5
View File
@@ -42,6 +42,9 @@ export interface NotificationEvents {
tailscaleKeyCheck: boolean;
secretCheckTime: string; // "HH:MM" — shared by the secret-expiry and Tailscale key-expiry checks
timezone: string;
integrationFailureAlerts: boolean;
/** Consecutive failed calls (from the Diagnostic Log) before an integration/DNS provider is considered down. */
integrationFailureThreshold: number;
}
export type ProviderColors = Record<string, string>;
@@ -90,6 +93,8 @@ const DEFAULTS: AppSettings = {
tailscaleKeyCheck: true,
secretCheckTime: "08:00",
timezone: "UTC",
integrationFailureAlerts: true,
integrationFailureThreshold: 3,
},
providerColors: {},
integrationColors: {},
+2
View File
@@ -142,6 +142,8 @@ export interface NotificationEvents {
tailscaleKeyCheck: boolean;
secretCheckTime: string;
timezone: string;
integrationFailureAlerts: boolean;
integrationFailureThreshold: number;
}
export type DateFormat = "ymd" | "dmy" | "mdy";
@@ -44,6 +44,8 @@ const DEFAULT_NOTIFICATIONS: NotificationEvents = {
tailscaleKeyCheck: true,
secretCheckTime: "08:00",
timezone: "UTC",
integrationFailureAlerts: true,
integrationFailureThreshold: 3,
};
type TestResult = { ok: boolean; message: string } | null;
@@ -489,6 +491,7 @@ export default function NotificationSettings() {
{ key: "dnsDelete" as const, label: "DNS record deleted" },
{ key: "secretCheck" as const, label: "Secret expiry reminder" },
{ key: "tailscaleKeyCheck" as const, label: "Tailscale key expiry reminder" },
{ key: "integrationFailureAlerts" as const, label: "Integration/DNS provider failing repeatedly" },
].map(({ key, label }) => (
<label key={key} className="form-check mb-2">
<input
@@ -500,6 +503,26 @@ export default function NotificationSettings() {
<span className="form-check-label">{label}</span>
</label>
))}
<div className="row g-2 mt-2">
<div className="col-6">
<label className="form-label">Alert after</label>
<div className="input-group">
<input
type="number"
className="form-control"
min={1}
max={20}
value={notifications.integrationFailureThreshold}
disabled={!notifications.integrationFailureAlerts}
onChange={(e) => setNotifications((n) => ({ ...n, integrationFailureThreshold: Number(e.target.value) }))}
/>
<span className="input-group-text">consecutive failures</span>
</div>
<div className="form-hint">
Based on the Diagnostic Log. A recovery notice is sent once it succeeds again.
</div>
</div>
</div>
{(() => {
const dailyChecksEnabled = notifications.secretCheck || notifications.tailscaleKeyCheck;
return (