The data was all being collected (agent last-seen, per-disk usage, Proxmox storage, Synology volume/disk health) but nothing acted on it, so a dead server or a full disk was only noticed by opening the right page. A new health pass runs every 15 minutes (matching the agent's default report interval) and raises one notification when a problem starts and one when it clears: a server's agent silent past a threshold (default 60 min), a server disk / Proxmox storage or root filesystem / Synology volume at or above a usage threshold (default 90%), and a Synology volume or disk that isn't "normal", has bad SMART, bad sectors past the threshold, or life remaining below it. Both thresholds and an on/off toggle live under Settings -> Notifications. The parts that make this trustworthy rather than noisy: - A problem is keyed by identity, so it alerts once and not every run; a shared Proxmox storage listed by every node is one problem, not one per node. - Active problems persist across restarts, so a rebuild doesn't re-alert everything already known. - If a source can't be read on a given run (Proxmox/Synology unreachable, one node lacking privileges) its existing problems are held, not reported "cleared" and then re-alerted when it comes back — the integration-failure alert already owns "the integration is down". - For 20 minutes after startup server-derived problems are held too: agents couldn't report while the app was down, so judging them then would report every server offline after any restart. - An offline server's disk figures are stale and are not judged; a server that never reported has no agent and raises nothing. - Tracking continues while the toggle is off (only sending is gated), so turning it back on doesn't dump every long-standing problem. Timestamps without a zone (SQLite's format) are read as UTC; the test runs on a UTC+2 machine, where reading them as local time gives a different answer. Verified with 32 checks: the evaluation rules and the state diff as pure functions (exact thresholds, the proxmox:1 vs proxmox:10 prefix trap, the flapping sequence), then a whole pass against a real Proxmox adapter talking to a fake HTTPS cluster (one node returning 403, the whole API down, a shared storage on two nodes, a node that recovers), a webhook receiver, the real DB, and the persisted state. Not exercised end-to-end: the Synology collection path — its rules are tested on data shaped exactly like the adapter's output types, but I did not stand up a fake DSM. Real dev database mtime untouched. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
183 lines
5.6 KiB
TypeScript
183 lines
5.6 KiB
TypeScript
import { sql } from "drizzle-orm";
|
|
import { db } from "../db/client.js";
|
|
import { settings } from "../db/schema.js";
|
|
|
|
export interface GotifySettings {
|
|
enabled: boolean;
|
|
url: string;
|
|
token: string;
|
|
priority: number;
|
|
}
|
|
|
|
export interface NtfySettings {
|
|
enabled: boolean;
|
|
url: string;
|
|
topic: string;
|
|
token: string;
|
|
priority: number;
|
|
}
|
|
|
|
export interface SmtpSettings {
|
|
enabled: boolean;
|
|
host: string;
|
|
port: number;
|
|
secure: boolean;
|
|
username: string;
|
|
password: string;
|
|
from: string;
|
|
to: string;
|
|
}
|
|
|
|
export interface WebhookSettings {
|
|
enabled: boolean;
|
|
url: string;
|
|
secret: string;
|
|
}
|
|
|
|
export interface NotificationEvents {
|
|
dnsAdd: boolean;
|
|
dnsUpdate: boolean;
|
|
dnsDelete: boolean;
|
|
secretCheck: boolean;
|
|
tailscaleKeyCheck: boolean;
|
|
dockerUpdateCheck: boolean;
|
|
proxmoxBackupCheck: boolean;
|
|
healthAlerts: boolean;
|
|
secretCheckTime: string; // "HH:MM" — shared by the secret-expiry, Tailscale key-expiry, Docker update, and Proxmox backup checks
|
|
timezone: string;
|
|
integrationFailureAlerts: boolean;
|
|
/** Consecutive failed calls (from the Diagnostic Log) before an integration/DNS provider is considered down. */
|
|
integrationFailureThreshold: number;
|
|
}
|
|
|
|
export type ProviderColors = Record<string, string>;
|
|
export type IntegrationColors = Record<string, string>;
|
|
|
|
export type DateFormat = "ymd" | "dmy" | "mdy";
|
|
export type TimeFormat = "24h" | "12h";
|
|
|
|
export interface DisplaySettings {
|
|
dateFormat: DateFormat;
|
|
timeFormat: TimeFormat;
|
|
/** Rows per page for every paginated table in the app. */
|
|
pageSize: number;
|
|
}
|
|
|
|
export interface LogRetentionSettings {
|
|
enabled: boolean;
|
|
/** Diagnostic log and audit log entries older than this are deleted. */
|
|
retentionDays: number;
|
|
/** How often the purge job runs, in hours. */
|
|
intervalHours: number;
|
|
}
|
|
|
|
export interface QuietHoursSettings {
|
|
enabled: boolean;
|
|
/** "HH:MM", local to notifications.timezone. Notifications from `start` to `end` are queued, not sent immediately. */
|
|
start: string;
|
|
/** "HH:MM" — when `start`..`end` wraps past midnight (e.g. 22:00 -> 07:00), the queue is flushed as one digest at this time. */
|
|
end: string;
|
|
}
|
|
|
|
export interface HealthCheckSettings {
|
|
/** A server whose agent hasn't reported for this long is considered offline (agent default interval is 15 min). */
|
|
serverOfflineMinutes: number;
|
|
/** Disk / storage / volume usage at or above this percentage is reported. */
|
|
diskUsagePercent: number;
|
|
}
|
|
|
|
export interface AppSettings {
|
|
gotify: GotifySettings;
|
|
ntfy: NtfySettings;
|
|
smtp: SmtpSettings;
|
|
webhook: WebhookSettings;
|
|
notifications: NotificationEvents;
|
|
providerColors: ProviderColors;
|
|
integrationColors: IntegrationColors;
|
|
display: DisplaySettings;
|
|
logRetention: LogRetentionSettings;
|
|
quietHours: QuietHoursSettings;
|
|
healthChecks: HealthCheckSettings;
|
|
}
|
|
|
|
const DEFAULTS: AppSettings = {
|
|
gotify: { enabled: false, url: "", token: "", priority: 5 },
|
|
ntfy: { enabled: false, url: "https://ntfy.sh", topic: "", token: "", priority: 3 },
|
|
smtp: { enabled: false, host: "", port: 587, secure: false, username: "", password: "", from: "", to: "" },
|
|
webhook: { enabled: false, url: "", secret: "" },
|
|
notifications: {
|
|
dnsAdd: true,
|
|
dnsUpdate: true,
|
|
dnsDelete: true,
|
|
secretCheck: true,
|
|
tailscaleKeyCheck: true,
|
|
dockerUpdateCheck: true,
|
|
proxmoxBackupCheck: true,
|
|
healthAlerts: true,
|
|
secretCheckTime: "08:00",
|
|
timezone: "UTC",
|
|
integrationFailureAlerts: true,
|
|
integrationFailureThreshold: 3,
|
|
},
|
|
providerColors: {},
|
|
integrationColors: {},
|
|
display: { dateFormat: "ymd", timeFormat: "24h", pageSize: 20 },
|
|
logRetention: { enabled: false, retentionDays: 90, intervalHours: 24 },
|
|
quietHours: { enabled: false, start: "22:00", end: "07:00" },
|
|
healthChecks: { serverOfflineMinutes: 60, diskUsagePercent: 90 },
|
|
};
|
|
|
|
const KEYS = Object.keys(DEFAULTS) as (keyof AppSettings)[];
|
|
|
|
export async function getSettings(): Promise<AppSettings> {
|
|
const rows = await db.select().from(settings);
|
|
const byKey = new Map(rows.map((r) => [r.key, r.value]));
|
|
|
|
const result = {} as AppSettings;
|
|
for (const key of KEYS) {
|
|
const raw = byKey.get(key);
|
|
let parsed: Record<string, unknown> = {};
|
|
if (raw) {
|
|
try {
|
|
parsed = JSON.parse(raw);
|
|
} catch {
|
|
parsed = {};
|
|
}
|
|
}
|
|
(result[key] as Record<string, unknown>) = { ...(DEFAULTS[key] as object), ...parsed };
|
|
}
|
|
return result;
|
|
}
|
|
|
|
export type AppSettingsPatch = { [K in keyof AppSettings]?: Partial<AppSettings[K]> };
|
|
|
|
export async function updateSettings(partial: AppSettingsPatch): Promise<AppSettings> {
|
|
const current = await getSettings();
|
|
|
|
for (const key of Object.keys(partial) as (keyof AppSettings)[]) {
|
|
if (!KEYS.includes(key)) continue;
|
|
const merged = { ...(current[key] as object), ...(partial[key] as object) };
|
|
const value = JSON.stringify(merged);
|
|
await db
|
|
.insert(settings)
|
|
.values({ key, value })
|
|
.onConflictDoUpdate({ target: settings.key, set: { value, updatedAt: sql`(current_timestamp)` } });
|
|
}
|
|
|
|
return getSettings();
|
|
}
|
|
|
|
/** Small internal key/value slot for scheduler bookkeeping, outside the AppSettings shape. */
|
|
export async function getInternalFlag(key: string): Promise<string | null> {
|
|
const [row] = await db.select().from(settings).where(sql`${settings.key} = ${`_internal:${key}`}`).limit(1);
|
|
return row?.value ?? null;
|
|
}
|
|
|
|
export async function setInternalFlag(key: string, value: string): Promise<void> {
|
|
const fullKey = `_internal:${key}`;
|
|
await db
|
|
.insert(settings)
|
|
.values({ key: fullKey, value })
|
|
.onConflictDoUpdate({ target: settings.key, set: { value, updatedAt: sql`(current_timestamp)` } });
|
|
}
|