mirror of
https://github.com/Studio-Saelix/sencho.git
synced 2026-08-20 23:32:19 +00:00
feat: auto-heal policies for unhealthy containers (#671)
* feat(db): add auto_heal_policies and auto_heal_history schema and CRUD Adds two new SQLite tables (auto_heal_policies, auto_heal_history) to DatabaseService.initSchema() and exposes CRUD methods: getAutoHealPolicies, getAutoHealPolicy, addAutoHealPolicy, updateAutoHealPolicy, deleteAutoHealPolicy, recordAutoHealHistory, getAutoHealHistory, incrementConsecutiveFailures, resetConsecutiveFailures, setPolicyEnabled. Also adds AutoHealPolicy and AutoHealHistoryEntry TypeScript interfaces. * feat(events): track health-status duration and expose state accessors - Add healthStatus and unhealthySince fields to InternalContainerState - onHealthStatus now records unhealthySince timestamp on first transition to unhealthy, and clears it when the container recovers or restarts - onStart resets both fields so a restarted container begins from 'starting' - Add listContainerStates() and getContainerState() public accessors for use by the upcoming AutoHealService evaluator * fix(auto-heal): key allowlist in updateAutoHealPolicy, cascade delete, extract ContainerHealthSnapshot * feat: add AutoHealService evaluator singleton Polls every 30 s, matches containers to enabled policies via Compose labels, and restarts containers that have been unhealthy beyond the configured threshold. Enforces cooldown, per-hour rate cap, and recent-user-action suppression; auto-disables policies after repeated consecutive failures. Also adds DockerEventManager.getService() accessor required by the evaluator. * fix(auto-heal): prune stale restartTimestamps, guard undefined policy id - Prune restartTimestamps entries for containers no longer running after each container list fetch, preventing unbounded map growth from dead container IDs. - Guard against policies with undefined id at the start of the per-policy loop; warn and skip rather than proceed with a non-null assertion. - Extract handleAutoDisable private helper to bring executeHeal under 30 lines and isolate the auto-disable side-effect sequence. - Move ContainerInfo type to module scope. * feat: add auto-heal API routes and wire AutoHealService lifecycle Registers five REST endpoints under /api/auto-heal/policies (list, create, patch, delete, history) with requirePaid + requireAdmin guards and Zod validation. Wires AutoHealService.start()/stop() into the server startup and graceful-shutdown blocks alongside MonitorService. * test: add AutoHealService and DatabaseService auto-heal unit tests - 15 unit tests for AutoHealService.shouldHeal covering all decision branches (healthy state, duration threshold, user-action suppression, cooldown, rate limiting, and correct skipReason values) - 13 integration tests for DatabaseService auto-heal CRUD: policy round-trip, stack-name filter, partial update, cascade delete, history ordering/limit, consecutive failure counters, and setPolicyEnabled toggle * fix: log AutoHealService shutdown errors consistently * fix(api): requireAdmin-first guard order and try/catch on auto-heal routes * feat(ui): add StackAutoHealSheet component * feat(ui): add Auto-Heal context menu item to EditorLayout * fix(ui): StackAutoHealSheet label, token, a11y, and useEffect fixes - Rename 'All services in stack' to 'All services' in combobox options and placeholder - Replace text-green-600 with text-success design token in actionColorClass - Add htmlFor/id pairs to all four numeric form inputs for accessibility - Inline fetch logic into useEffect, removing stale closure risk and eslint-disable comment - Remove now-unused fetchPolicies and fetchServices standalone functions - Update 'Auto-disable after' label to 'Auto-disable after (failures)' for clarity - Add toast.error in policy fetch failure path; services fetch silently skips as before * docs: add auto-heal-policies feature documentation * test(e2e): add auto-heal policies CRUD spec * fix(docs): correct auto-heal-policies nav position in docs.json
This commit is contained in:
@@ -0,0 +1,318 @@
|
||||
import { INTENTIONAL_KILL_WINDOW_MS } from './ContainerLifecycleClassifier';
|
||||
import { DatabaseService, AutoHealPolicy, AutoHealHistoryEntry } from './DatabaseService';
|
||||
import DockerController from './DockerController';
|
||||
import { DockerEventManager } from './DockerEventManager';
|
||||
import { ContainerHealthSnapshot } from './DockerEventService';
|
||||
import { NotificationService } from './NotificationService';
|
||||
|
||||
// Dockerode listContainers shape (subset used here)
|
||||
type ContainerInfo = {
|
||||
Id: string;
|
||||
Names?: string[];
|
||||
Labels?: Record<string, string>;
|
||||
};
|
||||
|
||||
const EVAL_INTERVAL_MS = 30_000;
|
||||
const INITIAL_DELAY_MS = 10_000;
|
||||
const RATE_LIMIT_WINDOW_MS = 60 * 60_000; // 1 hour
|
||||
|
||||
export class AutoHealService {
|
||||
private static instance: AutoHealService;
|
||||
private intervalId: NodeJS.Timeout | null = null;
|
||||
private initialTimer: NodeJS.Timeout | null = null;
|
||||
private isProcessing = false;
|
||||
private restartTimestamps = new Map<string, number[]>();
|
||||
|
||||
private constructor() {}
|
||||
|
||||
static getInstance(): AutoHealService {
|
||||
if (!AutoHealService.instance) {
|
||||
AutoHealService.instance = new AutoHealService();
|
||||
}
|
||||
return AutoHealService.instance;
|
||||
}
|
||||
|
||||
start(): void {
|
||||
this.initialTimer = setTimeout(() => {
|
||||
void this.evaluate();
|
||||
this.intervalId = setInterval(() => void this.evaluate(), EVAL_INTERVAL_MS);
|
||||
}, INITIAL_DELAY_MS);
|
||||
}
|
||||
|
||||
stop(): void {
|
||||
if (this.initialTimer) {
|
||||
clearTimeout(this.initialTimer);
|
||||
this.initialTimer = null;
|
||||
}
|
||||
if (this.intervalId) {
|
||||
clearInterval(this.intervalId);
|
||||
this.intervalId = null;
|
||||
}
|
||||
}
|
||||
|
||||
async evaluate(): Promise<void> {
|
||||
if (this.isProcessing) return;
|
||||
this.isProcessing = true;
|
||||
try {
|
||||
const db = DatabaseService.getInstance();
|
||||
const policies = db.getAutoHealPolicies().filter(p => p.enabled === 1);
|
||||
if (policies.length === 0) return;
|
||||
|
||||
// Evaluate only on local nodes (remote nodes self-monitor via their own instance)
|
||||
const nodes = db.getNodes().filter(n => n.type === 'local');
|
||||
for (const node of nodes) {
|
||||
await this.evaluateForNode(node.id, policies);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('[AutoHeal] evaluate error:', err instanceof Error ? err.message : err);
|
||||
} finally {
|
||||
this.isProcessing = false;
|
||||
}
|
||||
}
|
||||
|
||||
private async evaluateForNode(nodeId: number, policies: AutoHealPolicy[]): Promise<void> {
|
||||
let containers: ContainerInfo[];
|
||||
try {
|
||||
containers = await DockerController.getInstance(nodeId).getRunningContainers();
|
||||
} catch (err) {
|
||||
console.error(
|
||||
`[AutoHeal] failed to list containers on node ${nodeId}:`,
|
||||
err instanceof Error ? err.message : err,
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const db = DatabaseService.getInstance();
|
||||
const eventSvc = DockerEventManager.getInstance().getService(nodeId);
|
||||
const now = Date.now();
|
||||
|
||||
// Prune stale entries for containers no longer running on this node
|
||||
const liveIds = new Set(containers.map(c => c.Id));
|
||||
for (const [cid, timestamps] of this.restartTimestamps.entries()) {
|
||||
const recent = timestamps.filter(t => now - t < RATE_LIMIT_WINDOW_MS);
|
||||
if (recent.length === 0 || !liveIds.has(cid)) {
|
||||
this.restartTimestamps.delete(cid);
|
||||
} else {
|
||||
this.restartTimestamps.set(cid, recent);
|
||||
}
|
||||
}
|
||||
|
||||
for (const policy of policies) {
|
||||
if (policy.id === undefined) {
|
||||
console.warn('[AutoHeal] skipping policy without id:', policy.stack_name);
|
||||
continue;
|
||||
}
|
||||
const candidates = containers.filter(c => {
|
||||
const labels = c.Labels ?? {};
|
||||
if (labels['com.docker.compose.project'] !== policy.stack_name) return false;
|
||||
if (policy.service_name) {
|
||||
return labels['com.docker.compose.service'] === policy.service_name;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
|
||||
for (const container of candidates) {
|
||||
const containerName =
|
||||
container.Names?.[0]?.replace(/^\//, '') ?? container.Id.slice(0, 12);
|
||||
const serviceOverride = container.Labels?.['com.docker.compose.service'] ?? null;
|
||||
const state = eventSvc?.getContainerState(container.Id);
|
||||
const decision = this.shouldHeal(state, policy, container.Id, now);
|
||||
|
||||
if (!decision.heal) {
|
||||
if (
|
||||
decision.skipReason &&
|
||||
decision.skipReason !== 'not_unhealthy' &&
|
||||
decision.skipReason !== 'duration_not_met'
|
||||
) {
|
||||
db.recordAutoHealHistory({
|
||||
policy_id: policy.id!,
|
||||
stack_name: policy.stack_name,
|
||||
service_name: policy.service_name ?? serviceOverride,
|
||||
container_name: containerName,
|
||||
container_id: container.Id,
|
||||
action: decision.skipReason as AutoHealHistoryEntry['action'],
|
||||
reason: this.skipReasonText(decision.skipReason),
|
||||
success: 0,
|
||||
error: null,
|
||||
timestamp: now,
|
||||
});
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
await this.executeHeal(
|
||||
policy,
|
||||
nodeId,
|
||||
container.Id,
|
||||
containerName,
|
||||
policy.service_name ?? serviceOverride,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private shouldHeal(
|
||||
state: ContainerHealthSnapshot | undefined,
|
||||
policy: AutoHealPolicy,
|
||||
containerId: string,
|
||||
now: number,
|
||||
): { heal: boolean; skipReason?: string } {
|
||||
// No state tracked yet, or container is not unhealthy
|
||||
if (!state || state.healthStatus !== 'unhealthy' || !state.unhealthySince) {
|
||||
return { heal: false, skipReason: 'not_unhealthy' };
|
||||
}
|
||||
|
||||
// Duration threshold not yet met
|
||||
const unhealthyMs = now - state.unhealthySince;
|
||||
if (unhealthyMs < policy.unhealthy_duration_mins * 60_000) {
|
||||
return { heal: false, skipReason: 'duration_not_met' };
|
||||
}
|
||||
|
||||
// Suppress if user recently killed the container
|
||||
if (state.lastKillAt !== undefined && now - state.lastKillAt < INTENTIONAL_KILL_WINDOW_MS) {
|
||||
return { heal: false, skipReason: 'skipped_user_action' };
|
||||
}
|
||||
|
||||
// Cooldown: respect last_fired_at
|
||||
if (policy.last_fired_at > 0 && now - policy.last_fired_at < policy.cooldown_mins * 60_000) {
|
||||
return { heal: false, skipReason: 'skipped_cooldown' };
|
||||
}
|
||||
|
||||
// Rate limit: max restarts per hour
|
||||
const recentRestarts = (this.restartTimestamps.get(containerId) ?? []).filter(
|
||||
t => now - t < RATE_LIMIT_WINDOW_MS,
|
||||
);
|
||||
if (recentRestarts.length >= policy.max_restarts_per_hour) {
|
||||
return { heal: false, skipReason: 'skipped_rate_limit' };
|
||||
}
|
||||
|
||||
return { heal: true };
|
||||
}
|
||||
|
||||
private async executeHeal(
|
||||
policy: AutoHealPolicy,
|
||||
nodeId: number,
|
||||
containerId: string,
|
||||
containerName: string,
|
||||
serviceName: string | null,
|
||||
): Promise<void> {
|
||||
const db = DatabaseService.getInstance();
|
||||
const now = Date.now();
|
||||
const baseEntry = {
|
||||
policy_id: policy.id!,
|
||||
stack_name: policy.stack_name,
|
||||
service_name: serviceName,
|
||||
container_name: containerName,
|
||||
container_id: containerId,
|
||||
timestamp: now,
|
||||
};
|
||||
|
||||
try {
|
||||
await DockerController.getInstance(nodeId).restartContainer(containerId);
|
||||
|
||||
db.resetConsecutiveFailures(policy.id!);
|
||||
db.updateAutoHealPolicy(policy.id!, { last_fired_at: now });
|
||||
|
||||
const timestamps = this.restartTimestamps.get(containerId) ?? [];
|
||||
timestamps.push(now);
|
||||
this.restartTimestamps.set(
|
||||
containerId,
|
||||
timestamps.filter(t => now - t < RATE_LIMIT_WINDOW_MS),
|
||||
);
|
||||
|
||||
db.recordAutoHealHistory({
|
||||
...baseEntry,
|
||||
action: 'restarted',
|
||||
reason: `Container unhealthy for ${policy.unhealthy_duration_mins} minute(s); auto-restarted.`,
|
||||
success: 1,
|
||||
error: null,
|
||||
});
|
||||
|
||||
db.insertAuditLog({
|
||||
timestamp: now,
|
||||
username: 'system',
|
||||
method: 'POST',
|
||||
path: '/api/auto-heal/execute',
|
||||
status_code: 200,
|
||||
node_id: nodeId,
|
||||
ip_address: '127.0.0.1',
|
||||
summary: `Auto-healed container ${containerName} on stack ${policy.stack_name}`,
|
||||
});
|
||||
|
||||
NotificationService.getInstance()
|
||||
.dispatchAlert(
|
||||
'info',
|
||||
`Auto-Heal: Restarted ${containerName} on stack ${policy.stack_name} after being unhealthy for ${policy.unhealthy_duration_mins} minute(s).`,
|
||||
policy.stack_name,
|
||||
)
|
||||
.catch(err => console.error('[AutoHeal] notification dispatch failed:', err));
|
||||
} catch (err) {
|
||||
const errorMsg = err instanceof Error ? err.message : String(err);
|
||||
|
||||
db.incrementConsecutiveFailures(policy.id!);
|
||||
// Re-read to get updated consecutive_failures count
|
||||
const updated = db.getAutoHealPolicy(policy.id!);
|
||||
const failures = updated?.consecutive_failures ?? policy.consecutive_failures + 1;
|
||||
|
||||
db.recordAutoHealHistory({
|
||||
...baseEntry,
|
||||
action: 'failed',
|
||||
reason: `Restart failed: ${errorMsg}`,
|
||||
success: 0,
|
||||
error: errorMsg,
|
||||
});
|
||||
|
||||
NotificationService.getInstance()
|
||||
.dispatchAlert(
|
||||
'warning',
|
||||
`Auto-Heal: Failed to restart ${containerName} on stack ${policy.stack_name}. Error: ${errorMsg}`,
|
||||
policy.stack_name,
|
||||
)
|
||||
.catch(e => console.error('[AutoHeal] notification dispatch failed:', e));
|
||||
|
||||
// Auto-disable if failure threshold reached
|
||||
if (failures >= policy.auto_disable_after_failures) {
|
||||
this.handleAutoDisable(policy.id!, policy, baseEntry, failures);
|
||||
}
|
||||
|
||||
console.error(`[AutoHeal] restart failed for ${containerName} (${containerId}):`, errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
private handleAutoDisable(
|
||||
policyId: number,
|
||||
policy: AutoHealPolicy,
|
||||
baseEntry: Omit<Parameters<DatabaseService['recordAutoHealHistory']>[0], 'action' | 'reason' | 'success' | 'error'>,
|
||||
failures: number,
|
||||
): void {
|
||||
const db = DatabaseService.getInstance();
|
||||
db.setPolicyEnabled(policyId, false);
|
||||
db.recordAutoHealHistory({
|
||||
...baseEntry,
|
||||
action: 'policy_auto_disabled',
|
||||
reason: `Policy disabled after ${failures} consecutive restart failures. Check container logs and re-enable when resolved.`,
|
||||
success: 0,
|
||||
error: null,
|
||||
});
|
||||
NotificationService.getInstance()
|
||||
.dispatchAlert(
|
||||
'warning',
|
||||
`Auto-Heal: Policy for ${policy.stack_name}${policy.service_name ? '/' + policy.service_name : ''} has been auto-disabled after ${failures} consecutive failures.`,
|
||||
policy.stack_name,
|
||||
)
|
||||
.catch(e => console.error('[AutoHeal] notification dispatch failed:', e));
|
||||
}
|
||||
|
||||
private skipReasonText(reason: string): string {
|
||||
switch (reason) {
|
||||
case 'skipped_user_action':
|
||||
return 'Skipped: recent user action detected on this container.';
|
||||
case 'skipped_cooldown':
|
||||
return 'Skipped: cooldown period has not elapsed since last restart.';
|
||||
case 'skipped_rate_limit':
|
||||
return 'Skipped: hourly restart limit reached for this container.';
|
||||
default:
|
||||
return reason;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -28,6 +28,35 @@ export interface StackAlert {
|
||||
|
||||
export type NodeMode = 'proxy' | 'pilot_agent';
|
||||
|
||||
export interface AutoHealPolicy {
|
||||
id?: number;
|
||||
stack_name: string;
|
||||
service_name: string | null;
|
||||
unhealthy_duration_mins: number;
|
||||
cooldown_mins: number;
|
||||
max_restarts_per_hour: number;
|
||||
auto_disable_after_failures: number;
|
||||
enabled: number;
|
||||
consecutive_failures: number;
|
||||
last_fired_at: number;
|
||||
created_at: number;
|
||||
updated_at: number;
|
||||
}
|
||||
|
||||
export interface AutoHealHistoryEntry {
|
||||
id?: number;
|
||||
policy_id: number;
|
||||
stack_name: string;
|
||||
service_name: string | null;
|
||||
container_name: string;
|
||||
container_id: string;
|
||||
action: 'restarted' | 'skipped_user_action' | 'skipped_cooldown' | 'skipped_rate_limit' | 'failed' | 'policy_auto_disabled';
|
||||
reason: string;
|
||||
success: number;
|
||||
error: string | null;
|
||||
timestamp: number;
|
||||
}
|
||||
|
||||
export interface Node {
|
||||
id: number;
|
||||
name: string;
|
||||
@@ -807,6 +836,38 @@ export class DatabaseService {
|
||||
created_at INTEGER NOT NULL,
|
||||
updated_at INTEGER NOT NULL
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS auto_heal_policies (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
stack_name TEXT NOT NULL,
|
||||
service_name TEXT,
|
||||
unhealthy_duration_mins INTEGER NOT NULL,
|
||||
cooldown_mins INTEGER NOT NULL DEFAULT 5,
|
||||
max_restarts_per_hour INTEGER NOT NULL DEFAULT 3,
|
||||
auto_disable_after_failures INTEGER NOT NULL DEFAULT 5,
|
||||
enabled INTEGER NOT NULL DEFAULT 1,
|
||||
consecutive_failures INTEGER NOT NULL DEFAULT 0,
|
||||
last_fired_at INTEGER NOT NULL DEFAULT 0,
|
||||
created_at INTEGER NOT NULL,
|
||||
updated_at INTEGER NOT NULL
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS auto_heal_history (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
policy_id INTEGER NOT NULL,
|
||||
stack_name TEXT NOT NULL,
|
||||
service_name TEXT,
|
||||
container_name TEXT NOT NULL,
|
||||
container_id TEXT NOT NULL,
|
||||
action TEXT NOT NULL,
|
||||
reason TEXT NOT NULL,
|
||||
success INTEGER NOT NULL,
|
||||
error TEXT,
|
||||
timestamp INTEGER NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_auto_heal_history_policy_ts
|
||||
ON auto_heal_history(policy_id, timestamp DESC);
|
||||
`);
|
||||
|
||||
// Apply migrations safely (ignore if columns already exist)
|
||||
@@ -1205,6 +1266,94 @@ export class DatabaseService {
|
||||
stmt.run(timestamp, id);
|
||||
}
|
||||
|
||||
// --- Auto-Heal Policies ---
|
||||
|
||||
public getAutoHealPolicies(stackName?: string): AutoHealPolicy[] {
|
||||
if (stackName) {
|
||||
return this.db.prepare('SELECT * FROM auto_heal_policies WHERE stack_name = ?').all(stackName) as AutoHealPolicy[];
|
||||
}
|
||||
return this.db.prepare('SELECT * FROM auto_heal_policies').all() as AutoHealPolicy[];
|
||||
}
|
||||
|
||||
public getAutoHealPolicy(id: number): AutoHealPolicy | undefined {
|
||||
return this.db.prepare('SELECT * FROM auto_heal_policies WHERE id = ?').get(id) as AutoHealPolicy | undefined;
|
||||
}
|
||||
|
||||
public addAutoHealPolicy(policy: Omit<AutoHealPolicy, 'id'>): AutoHealPolicy {
|
||||
const stmt = this.db.prepare(
|
||||
'INSERT INTO auto_heal_policies (stack_name, service_name, unhealthy_duration_mins, cooldown_mins, max_restarts_per_hour, auto_disable_after_failures, enabled, consecutive_failures, last_fired_at, created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)'
|
||||
);
|
||||
const result = stmt.run(
|
||||
policy.stack_name,
|
||||
policy.service_name ?? null,
|
||||
policy.unhealthy_duration_mins,
|
||||
policy.cooldown_mins,
|
||||
policy.max_restarts_per_hour,
|
||||
policy.auto_disable_after_failures,
|
||||
policy.enabled,
|
||||
policy.consecutive_failures,
|
||||
policy.last_fired_at,
|
||||
policy.created_at,
|
||||
policy.updated_at
|
||||
);
|
||||
return this.db.prepare('SELECT * FROM auto_heal_policies WHERE id = ?').get(result.lastInsertRowid) as AutoHealPolicy;
|
||||
}
|
||||
|
||||
public updateAutoHealPolicy(id: number, patch: Partial<Omit<AutoHealPolicy, 'id' | 'stack_name' | 'created_at'>>): void {
|
||||
const ALLOWED_KEYS = new Set([
|
||||
'service_name', 'unhealthy_duration_mins', 'cooldown_mins',
|
||||
'max_restarts_per_hour', 'auto_disable_after_failures',
|
||||
'enabled', 'consecutive_failures', 'last_fired_at',
|
||||
]);
|
||||
const entries = Object.entries(patch).filter(([k, v]) => ALLOWED_KEYS.has(k) && v !== undefined);
|
||||
if (entries.length === 0) return;
|
||||
const fields = entries.map(([k]) => `${k} = ?`).join(', ');
|
||||
const values = entries.map(([, v]) => v);
|
||||
this.db.prepare(`UPDATE auto_heal_policies SET ${fields}, updated_at = ? WHERE id = ?`).run(...values, Date.now(), id);
|
||||
}
|
||||
|
||||
public deleteAutoHealPolicy(id: number): void {
|
||||
this.db.transaction(() => {
|
||||
this.db.prepare('DELETE FROM auto_heal_history WHERE policy_id = ?').run(id);
|
||||
this.db.prepare('DELETE FROM auto_heal_policies WHERE id = ?').run(id);
|
||||
})();
|
||||
}
|
||||
|
||||
public recordAutoHealHistory(entry: Omit<AutoHealHistoryEntry, 'id'>): void {
|
||||
this.db.prepare(
|
||||
'INSERT INTO auto_heal_history (policy_id, stack_name, service_name, container_name, container_id, action, reason, success, error, timestamp) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)'
|
||||
).run(
|
||||
entry.policy_id,
|
||||
entry.stack_name,
|
||||
entry.service_name ?? null,
|
||||
entry.container_name,
|
||||
entry.container_id,
|
||||
entry.action,
|
||||
entry.reason,
|
||||
entry.success,
|
||||
entry.error ?? null,
|
||||
entry.timestamp
|
||||
);
|
||||
}
|
||||
|
||||
public getAutoHealHistory(policyId: number, limit = 50): AutoHealHistoryEntry[] {
|
||||
return this.db.prepare(
|
||||
'SELECT * FROM auto_heal_history WHERE policy_id = ? ORDER BY timestamp DESC LIMIT ?'
|
||||
).all(policyId, limit) as AutoHealHistoryEntry[];
|
||||
}
|
||||
|
||||
public incrementConsecutiveFailures(policyId: number): void {
|
||||
this.db.prepare('UPDATE auto_heal_policies SET consecutive_failures = consecutive_failures + 1, updated_at = ? WHERE id = ?').run(Date.now(), policyId);
|
||||
}
|
||||
|
||||
public resetConsecutiveFailures(policyId: number): void {
|
||||
this.db.prepare('UPDATE auto_heal_policies SET consecutive_failures = 0, updated_at = ? WHERE id = ?').run(Date.now(), policyId);
|
||||
}
|
||||
|
||||
public setPolicyEnabled(policyId: number, enabled: boolean): void {
|
||||
this.db.prepare('UPDATE auto_heal_policies SET enabled = ?, updated_at = ? WHERE id = ?').run(enabled ? 1 : 0, Date.now(), policyId);
|
||||
}
|
||||
|
||||
// --- Notification History ---
|
||||
|
||||
public getNotificationHistory(limit = 50): NotificationHistory[] {
|
||||
|
||||
@@ -71,6 +71,11 @@ export class DockerEventManager {
|
||||
return Array.from(this.services.values()).map(s => s.getStatus());
|
||||
}
|
||||
|
||||
/** Returns the DockerEventService for a given local node, or undefined if not tracked. */
|
||||
public getService(nodeId: number): DockerEventService | undefined {
|
||||
return this.services.get(nodeId);
|
||||
}
|
||||
|
||||
// ========================================================================
|
||||
// Node lifecycle handlers
|
||||
// ========================================================================
|
||||
|
||||
@@ -24,6 +24,16 @@ import { getErrorMessage } from '../utils/errors';
|
||||
* See docs/features/alerts-notifications.mdx for user-facing behaviour.
|
||||
*/
|
||||
|
||||
/** Snapshot of a single container's health tracking state, exposed to AutoHealService. */
|
||||
export interface ContainerHealthSnapshot {
|
||||
id: string;
|
||||
name?: string;
|
||||
stackName?: string;
|
||||
healthStatus?: 'healthy' | 'unhealthy' | 'starting';
|
||||
unhealthySince?: number;
|
||||
lastKillAt?: number;
|
||||
}
|
||||
|
||||
/** Grace window after a `die` before classifying, to absorb out-of-order kill events. */
|
||||
const DIE_GRACE_WINDOW_MS = 500;
|
||||
|
||||
@@ -61,6 +71,8 @@ interface InternalContainerState extends ContainerLifecycleState {
|
||||
stackName?: string;
|
||||
lastCrashAlertAt?: number;
|
||||
lastActivityAt: number;
|
||||
healthStatus?: 'healthy' | 'unhealthy' | 'starting';
|
||||
unhealthySince?: number;
|
||||
}
|
||||
|
||||
interface DockerEventPayload {
|
||||
@@ -400,16 +412,29 @@ export class DockerEventService {
|
||||
}
|
||||
|
||||
private onHealthStatus(id: string, action: string, event: DockerEventPayload): void {
|
||||
if (!action.includes('unhealthy')) return;
|
||||
const state = this.getOrCreateState(id, event);
|
||||
state.lastActivityAt = Date.now();
|
||||
if (!this.isCrashAlertsEnabled()) return;
|
||||
const name = state.name ?? id.slice(0, 12);
|
||||
const stackName = state.stackName;
|
||||
void this.emitError(
|
||||
`Healthcheck failed: ${name} is unhealthy.`,
|
||||
stackName,
|
||||
);
|
||||
|
||||
if (action.includes('unhealthy')) {
|
||||
if (state.healthStatus !== 'unhealthy') {
|
||||
state.unhealthySince = Date.now();
|
||||
}
|
||||
state.healthStatus = 'unhealthy';
|
||||
if (!this.isCrashAlertsEnabled()) return;
|
||||
const name = state.name ?? id.slice(0, 12);
|
||||
const stackName = state.stackName;
|
||||
void this.emitError(
|
||||
`Healthcheck failed: ${name} is unhealthy.`,
|
||||
stackName,
|
||||
);
|
||||
} else {
|
||||
state.unhealthySince = undefined;
|
||||
if (action.includes('starting')) {
|
||||
state.healthStatus = 'starting';
|
||||
} else {
|
||||
state.healthStatus = 'healthy';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private onStart(id: string): void {
|
||||
@@ -419,6 +444,8 @@ export class DockerEventService {
|
||||
state.lastKillAt = undefined;
|
||||
state.oomPending = undefined;
|
||||
state.lastCrashAlertAt = undefined;
|
||||
state.unhealthySince = undefined;
|
||||
state.healthStatus = 'starting';
|
||||
state.lastActivityAt = Date.now();
|
||||
}
|
||||
|
||||
@@ -649,4 +676,32 @@ export class DockerEventService {
|
||||
trackedContainers: this.containerState.size,
|
||||
};
|
||||
}
|
||||
|
||||
// ========================================================================
|
||||
// Container state accessors (used by AutoHealService)
|
||||
// ========================================================================
|
||||
|
||||
public listContainerStates(): ContainerHealthSnapshot[] {
|
||||
return Array.from(this.containerState.entries()).map(([id, s]) => ({
|
||||
id,
|
||||
name: s.name,
|
||||
stackName: s.stackName,
|
||||
healthStatus: s.healthStatus,
|
||||
unhealthySince: s.unhealthySince,
|
||||
lastKillAt: s.lastKillAt,
|
||||
}));
|
||||
}
|
||||
|
||||
public getContainerState(id: string): ContainerHealthSnapshot | undefined {
|
||||
const s = this.containerState.get(id);
|
||||
if (!s) return undefined;
|
||||
return {
|
||||
id,
|
||||
name: s.name,
|
||||
stackName: s.stackName,
|
||||
healthStatus: s.healthStatus,
|
||||
unhealthySince: s.unhealthySince,
|
||||
lastKillAt: s.lastKillAt,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user