mirror of
https://github.com/Studio-Saelix/sencho.git
synced 2026-08-31 04:38:11 +00:00
fix(dashboard): harden real-time dashboard with bug fixes and design compliance (#517)
* fix(dashboard): harden real-time dashboard with bug fixes and design compliance - Fix host alert spam: add 5-minute cooldown for CPU/RAM/disk threshold alerts, preventing duplicate notifications every 30s during sustained breaches. Extract shared dispatchWithCooldown helper (also used by Docker janitor alerts). - Fix memory metric inflation: subtract filesystem cache from stored memory_mb values, matching the existing calculateMemoryPercent logic. - Fix crash detection reliability: replace fragile 'seconds ago' string matching with a tracked Set of alerted container IDs. Containers are only alerted once per crash event, with automatic cleanup when they start running again or after a 1-hour TTL. - Fix health status bar: exited containers now trigger 'degraded' state independently of unread error notifications. - Fix CPU chart Y-axis: auto-scale when aggregate container CPU exceeds 100% instead of silently clipping at the hardcoded domain ceiling. - Fix grammar: 'actives' to 'active' in container count label. - Add shadow-card-bevel to all dashboard cards per design system. - Update dashboard docs to reflect revised health status thresholds. * test(dashboard): update monitor service tests for new alert signatures - Add container Id fields to crash detection test fixtures - Update host alert assertions to match dispatchWithCooldown 3-arg call - Fix unhealthy container test to use State: 'unhealthy' instead of State: 'running' (running containers are now skipped in crash detect)
This commit is contained in:
@@ -38,13 +38,13 @@ function deriveHealth(stats: Stats, systemStats: SystemStats | null, notificatio
|
||||
if (disk >= 90) reasons.push(`Disk at ${disk.toFixed(1)}%`);
|
||||
else if (disk >= 80) reasons.push(`Disk at ${disk.toFixed(1)}%`);
|
||||
|
||||
if (stats.exited > 0 && unreadErrors > 0) reasons.push(`${stats.exited} exited container${stats.exited !== 1 ? 's' : ''}`);
|
||||
else if (unreadErrors > 0) reasons.push(`${unreadErrors} unread error${unreadErrors !== 1 ? 's' : ''}`);
|
||||
if (stats.exited > 0) reasons.push(`${stats.exited} exited container${stats.exited !== 1 ? 's' : ''}`);
|
||||
if (unreadErrors > 0) reasons.push(`${unreadErrors} unread error${unreadErrors !== 1 ? 's' : ''}`);
|
||||
|
||||
if (cpu >= 90 || ram >= 90 || disk >= 90 || (stats.exited > 0 && unreadErrors > 0)) {
|
||||
return { level: 'critical', reasons };
|
||||
}
|
||||
if (cpu >= 80 || ram >= 80 || disk >= 80 || unreadErrors > 0) {
|
||||
if (cpu >= 80 || ram >= 80 || disk >= 80 || stats.exited > 0 || unreadErrors > 0) {
|
||||
return { level: 'degraded', reasons };
|
||||
}
|
||||
return { level: 'healthy', reasons: ['All systems nominal'] };
|
||||
@@ -65,7 +65,7 @@ export function HealthStatusBar({ stats, systemStats, notifications, activeNodeN
|
||||
const unreadAlerts = notifications.filter(n => !n.is_read).length;
|
||||
|
||||
return (
|
||||
<Card className="bg-card px-4 py-3">
|
||||
<Card className="bg-card shadow-card-bevel px-4 py-3">
|
||||
<div className="flex items-center justify-between gap-4 flex-wrap">
|
||||
{/* Health badge */}
|
||||
<div className="flex items-center gap-3">
|
||||
|
||||
Reference in New Issue
Block a user