Files
sencho/backend/src/services/NotificationService.ts
T
Anso 80499ee18d feat(stack-activity): in-process metrics, structured diagnostic logs, docs (#1229)
Phase 3 + Phase 6 of the Stack Activity audit (PR 2 of 2):

- StackActivityMetricsService: in-process counters and ring-buffered
  latency histogram (1000 samples per nodeId/op pair). Mirrors the
  FileExplorerMetricsService pattern shipped in #1216. No external
  export. Records (nodeId, op) where op is read or write, with
  success/error counts and p50/p95 latency on demand.

- Admin endpoint GET /api/stack-activity-metrics returns the snapshot.
  Admin-only via requireAdmin, mounted next to the file-explorer
  metrics route. An operator debugging "why is the activity tab slow
  on this node?" can pull per-(nodeId, op) counts and latencies
  without scrolling logs.

- Diagnostic logs: route handler emits a structured [StackActivity:diag]
  read entry per request (stackName, nodeId, limit, before, beforeId,
  returned, elapsedMs); dispatchAlert emits a [StackActivity:diag] write
  entry per persisted notification (category, stackName, nodeId, actor,
  messageLen). Both gated on developer_mode via isDebugEnabled. Same
  namespace so a single grep covers reads and writes on the timeline
  path. Per-request and per-event, never inside a poll loop.

- Metric record points: the route's try/finally records a read metric
  with the outcome of the DB call; dispatchAlert records a write metric
  on both the success path and (before re-throwing) the failure path,
  so error rates from the insert path stay visible.

- docs/features/stack-activity.mdx: refreshed to reflect PR 1's
  retention behavior (30 days plus per-(node, stack) 500-row cap, 1000
  per-node unattached), composite (timestamp, id) cursor, error-vs-
  empty UI distinction, and "by username" vs "via Subsystem" actor
  rendering. Adds Troubleshooting entries for "Activity unavailable"
  (node disconnect or fetch failure), "expected event missing"
  (retention windows), and "same restart shows twice" (manual click
  vs Auto-Heal redeploy are distinct events).

No tier, role, or capability gate touched. The admin metrics endpoint
inherits the standard requireAdmin gate already used by /api/file-
explorer-metrics and /api/stack-metrics.
2026-05-25 21:25:59 -04:00

316 lines
13 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import WebSocket from 'ws';
import { DatabaseService, NotificationHistory } from './DatabaseService';
import { NodeRegistry } from './NodeRegistry';
import { isDebugEnabled } from '../utils/debug';
import { getErrorMessage } from '../utils/errors';
import { sanitizeForLog } from '../utils/safeLog';
import { sanitizeNotificationMessage } from '../utils/notificationMessage';
import { StackActivityMetricsService } from './StackActivityMetricsService';
export type NotificationCategory =
| 'deploy_success'
| 'deploy_failure'
| 'stack_started'
| 'stack_stopped'
| 'stack_restarted'
| 'image_update_available'
| 'image_update_applied'
| 'autoheal_triggered'
| 'monitor_alert'
| 'scan_finding'
| 'blueprint_deployed'
| 'blueprint_deployment_failed'
| 'blueprint_drift_detected'
| 'blueprint_drift_correction_failed'
| 'system';
export const ALL_NOTIFICATION_CATEGORIES: readonly NotificationCategory[] = [
'deploy_success', 'deploy_failure', 'stack_started', 'stack_stopped',
'stack_restarted', 'image_update_available', 'image_update_applied',
'autoheal_triggered', 'monitor_alert', 'scan_finding',
'blueprint_deployed', 'blueprint_deployment_failed',
'blueprint_drift_detected', 'blueprint_drift_correction_failed',
'system',
];
/** Webhook timeout: 10 seconds per external dispatch call. */
const WEBHOOK_TIMEOUT_MS = 10_000;
/** Valid notification channel types for defense-in-depth validation. */
const ALLOWED_CHANNEL_TYPES = new Set(['discord', 'slack', 'webhook']);
export class NotificationService {
private static instance: NotificationService;
private dbService: DatabaseService;
private readonly subscribers = new Set<WebSocket>();
private constructor() {
this.dbService = DatabaseService.getInstance();
}
public static getInstance(): NotificationService {
if (!NotificationService.instance) {
NotificationService.instance = new NotificationService();
}
return NotificationService.instance;
}
/**
* Register a WebSocket as a live-notification subscriber. Returns an
* unsubscribe function the caller should invoke on `'close'` / `'error'`
* (callers may guard against double-unsubscribe themselves; the Set
* handles repeated deletes safely either way).
*/
public subscribe(ws: WebSocket): () => void {
this.subscribers.add(ws);
return () => this.subscribers.delete(ws);
}
public getSubscriberCount(): number {
return this.subscribers.size;
}
/** Push a `{type,payload}` envelope to every currently-open subscriber. */
private broadcastToSubscribers(notification: NotificationHistory): void {
if (this.subscribers.size === 0) return;
const msg = JSON.stringify({ type: 'notification', payload: notification });
for (const ws of this.subscribers) {
if (ws.readyState === WebSocket.OPEN) {
ws.send(msg);
}
}
}
/**
* Broadcast an arbitrary non-notification event envelope to every
* currently-open subscriber WITHOUT writing it to the alerts history.
*
* Used by DockerEventService to push lightweight `state-invalidate`
* signals so the UI can refetch stack statuses on a real container event
* instead of waiting for the next polling tick. Persisting these would
* spam the notifications panel; they are pure ephemeral signals.
*/
public broadcastEvent(envelope: { type: string; [key: string]: unknown }): void {
if (this.subscribers.size === 0) return;
const msg = JSON.stringify(envelope);
for (const ws of this.subscribers) {
if (ws.readyState === WebSocket.OPEN) {
ws.send(msg);
}
}
}
/**
* Dispatch an alert: log to history, push via WebSocket, and route to
* external channels.
*
* Routing uses two tiers that coexist intentionally:
* - notification_routes (Admiral tier): per-stack pattern-based routing
* with priority ordering. If any route matches, global agents are skipped.
* - agents table (all tiers): global fallback channels used when no
* notification_routes match or when no stackName is provided.
*/
public async dispatchAlert(
level: 'info' | 'warning' | 'error',
category: NotificationCategory,
message: string,
options?: { stackName?: string; containerName?: string; actor?: string },
) {
const t0 = Date.now();
const { stackName, containerName, actor } = options ?? {};
// Internal writes use the middleware default so they share a row key
// with user-initiated requests; otherwise the UI and monitors split
// between different node_id buckets.
const localNodeId = NodeRegistry.getInstance().getDefaultNodeId();
// Use the full resolution chain (node.compose_dir, env, default)
// so messages mentioning a per-node compose override get collapsed.
const sanitized = sanitizeNotificationMessage(message, {
composeDir: NodeRegistry.getInstance().getComposeDir(localNodeId),
});
let notification: NotificationHistory;
try {
notification = this.dbService.addNotificationHistory(localNodeId, {
level,
category,
message: sanitized,
timestamp: Date.now(),
stack_name: stackName,
container_name: containerName,
actor_username: actor ?? null,
});
} catch (err) {
StackActivityMetricsService.getInstance().record(localNodeId, 'write', Date.now() - t0, false);
throw err;
}
StackActivityMetricsService.getInstance().record(localNodeId, 'write', Date.now() - t0, true);
// Separate [StackActivity:diag] namespace from the [Notify:diag] lines
// below so a single grep can pull every per-stack timeline write across
// route reads and dispatch writes.
if (isDebugEnabled()) {
console.log('[StackActivity:diag] write', {
category, stackName, nodeId: localNodeId, actor: actor ?? null, messageLen: sanitized.length,
});
}
// 2. Push to connected browser clients via WebSocket
this.broadcastToSubscribers(notification);
// 3. Check notification routing rules — always evaluated, matchers compose AND
const errors: string[] = [];
{
const routes = this.dbService.getEnabledNotificationRoutes();
const needsLabels = stackName !== undefined && routes.some(r => r.label_ids != null && r.label_ids.length > 0);
const stackLabelIds = needsLabels ? this.dbService.getStackLabelIds(localNodeId, stackName!) : [];
const matched = routes.filter(r => {
if (r.node_id != null && r.node_id !== localNodeId) return false;
if (r.stack_patterns.length > 0 && (stackName === undefined || !r.stack_patterns.includes(stackName))) return false;
if (r.label_ids != null && r.label_ids.length > 0 && !r.label_ids.some(id => stackLabelIds.includes(id))) return false;
if (r.categories != null && r.categories.length > 0 && !r.categories.includes(category)) return false;
return true;
});
if (matched.length > 0) {
if (isDebugEnabled()) console.log(`[Notify:diag] Matched ${matched.length} route(s) for stack "${sanitizeForLog(stackName ?? '(none)')}", category="${sanitizeForLog(category)}"`);
await Promise.allSettled(
matched.map(route =>
this.sendToChannel(route.channel_type, route.channel_url, level, sanitized)
.then(() => {
if (isDebugEnabled()) console.log(`[Notify:diag] Dispatched ${level} via route "${route.name}" (${route.channel_type})`);
})
.catch(error => {
console.error(`Failed to dispatch notification via route "${route.name}":`, error);
errors.push(`Route "${route.name}": ${getErrorMessage(error, String(error))}`);
})
)
);
this.recordDispatchErrors(notification.id!, errors);
return;
}
}
// 4. Fall back to this instance's agents (keyed by this instance's default node id).
const agents = this.dbService.getEnabledAgents(localNodeId);
if (agents.length === 0) {
if (isDebugEnabled()) console.log('[Notify:diag] No routes or agents matched; skipping external dispatch');
return;
}
if (isDebugEnabled()) console.log(`[Notify:diag] Falling back to ${agents.length} global agent(s)`);
await Promise.allSettled(
agents.map(agent =>
this.sendToChannel(agent.type, agent.url, level, sanitized)
.then(() => {
if (isDebugEnabled()) console.log(`[Notify:diag] Dispatched ${level} via global agent (${agent.type})`);
})
.catch(error => {
console.error(`Failed to dispatch notification to ${agent.type}:`, error);
errors.push(`${agent.type}: ${getErrorMessage(error, String(error))}`);
})
)
);
this.recordDispatchErrors(notification.id!, errors);
}
/** Persist dispatch errors to the notification record for user visibility. */
private recordDispatchErrors(notificationId: number, errors: string[]) {
if (errors.length > 0) {
try {
this.dbService.updateNotificationDispatchError(notificationId, errors.join('; '));
} catch (e) {
console.error('[Notify] Failed to record dispatch error:', e);
}
}
}
private async sendToChannel(type: string, url: string, level: 'info' | 'warning' | 'error', message: string): Promise<void> {
if (type === 'discord') {
await this.sendDiscordWebhook(url, level, message);
} else if (type === 'slack') {
await this.sendSlackWebhook(url, level, message);
} else if (type === 'webhook') {
await this.sendCustomWebhook(url, level, message);
} else {
throw new Error(`Unsupported channel type: ${type}`);
}
}
public async testDispatch(type: 'discord' | 'slack' | 'webhook', url: string) {
if (!ALLOWED_CHANNEL_TYPES.has(type)) throw new Error(`Invalid notification type: ${type}`);
if (!url || !url.startsWith('https://')) throw new Error('URL must use HTTPS');
await this.sendToChannel(type, url, 'info', '🔌 Test Notification from Sencho!');
}
private async sendDiscordWebhook(url: string, level: 'info' | 'warning' | 'error', message: string) {
const colorMap = {
info: 3447003, // Blue
warning: 16776960, // Yellow
error: 15158332 // Red
};
const payload = {
embeds: [{
title: `Sencho Alert [${level.toUpperCase()}]`,
description: message,
color: colorMap[level],
timestamp: new Date().toISOString()
}]
};
const response = await fetch(url, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(payload),
signal: AbortSignal.timeout(WEBHOOK_TIMEOUT_MS),
});
if (!response.ok) {
throw new Error(`Discord Webhook responded with ${response.status}`);
}
}
private async sendSlackWebhook(url: string, level: 'info' | 'warning' | 'error', message: string) {
const emojiMap = {
info: '️',
warning: '⚠️',
error: '🚨'
};
const payload = {
text: `${emojiMap[level]} *Sencho Alert [${level.toUpperCase()}]*\n${message}`
};
const response = await fetch(url, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(payload),
signal: AbortSignal.timeout(WEBHOOK_TIMEOUT_MS),
});
if (!response.ok) {
throw new Error(`Slack Webhook responded with ${response.status}`);
}
}
private async sendCustomWebhook(url: string, level: 'info' | 'warning' | 'error', message: string) {
const payload = {
level,
message,
timestamp: new Date().toISOString(),
source: 'sencho'
};
const response = await fetch(url, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(payload),
signal: AbortSignal.timeout(WEBHOOK_TIMEOUT_MS),
});
if (!response.ok) {
throw new Error(`Custom Webhook responded with ${response.status}`);
}
}
}