diff --git a/frontend-modern/src/components/Dashboard/Dashboard.tsx b/frontend-modern/src/components/Dashboard/Dashboard.tsx index 099f21d89..49c4a17ef 100644 --- a/frontend-modern/src/components/Dashboard/Dashboard.tsx +++ b/frontend-modern/src/components/Dashboard/Dashboard.tsx @@ -17,6 +17,7 @@ import { EmptyState } from '@/components/shared/EmptyState'; import { NodeGroupHeader } from '@/components/shared/NodeGroupHeader'; import { ProxmoxSectionNav } from '@/components/Proxmox/ProxmoxSectionNav'; import { isNodeOnline } from '@/utils/status'; +import { getNodeDisplayName } from '@/utils/nodes'; interface DashboardProps { vms: VM[]; @@ -56,8 +57,10 @@ export function Dashboard(props: DashboardProps) { return (a.clusterName || '').localeCompare(b.clusterName || ''); } - // Finally, sort by node name - return a.name.localeCompare(b.name); + // Finally, sort by node display name (falls back to hostname if unset) + const nameA = getNodeDisplayName(a); + const nameB = getNodeDisplayName(b); + return nameA.localeCompare(nameB); }); }); @@ -952,12 +955,14 @@ export function Dashboard(props: DashboardProps) { { - // Sort by node name for display, with instance ID as tiebreaker for duplicate hostnames + // Sort by friendly node name first, falling back to instance ID for stability const nodeA = nodeByInstance()[instanceIdA]; const nodeB = nodeByInstance()[instanceIdB]; - const nameCompare = (nodeA?.name || '').localeCompare(nodeB?.name || ''); + const labelA = nodeA ? getNodeDisplayName(nodeA) : instanceIdA; + const labelB = nodeB ? getNodeDisplayName(nodeB) : instanceIdB; + const nameCompare = labelA.localeCompare(labelB); if (nameCompare !== 0) return nameCompare; - // If names are equal (duplicate hostnames), sort by instance ID for stability + // If labels match (unlikely), fall back to the instance IDs return instanceIdA.localeCompare(instanceIdB); })} fallback={<>} diff --git a/frontend-modern/src/components/Storage/Storage.tsx b/frontend-modern/src/components/Storage/Storage.tsx index 0095f2935..c8bb28abc 100644 --- a/frontend-modern/src/components/Storage/Storage.tsx +++ b/frontend-modern/src/components/Storage/Storage.tsx @@ -11,6 +11,7 @@ import { Card } from '@/components/shared/Card'; import { EmptyState } from '@/components/shared/EmptyState'; import { NodeGroupHeader } from '@/components/shared/NodeGroupHeader'; import { ProxmoxSectionNav } from '@/components/Proxmox/ProxmoxSectionNav'; +import { getNodeDisplayName } from '@/utils/nodes'; const Storage: Component = () => { const { state, connected, activeAlerts, initialDataReceived } = useWebSocket(); @@ -736,11 +737,13 @@ const Storage: Component = () => { { - // Sort by node name for display when in node mode + // Sort by friendly node name for display when in node mode if (viewMode() === 'node') { const nodeA = nodeByInstance()[instanceIdA]; const nodeB = nodeByInstance()[instanceIdB]; - return (nodeA?.name || '').localeCompare(nodeB?.name || ''); + const labelA = nodeA ? getNodeDisplayName(nodeA) : instanceIdA; + const labelB = nodeB ? getNodeDisplayName(nodeB) : instanceIdB; + return labelA.localeCompare(labelB); } // Sort by storage name when in storage mode return instanceIdA.localeCompare(instanceIdB); diff --git a/frontend-modern/src/utils/nodes.ts b/frontend-modern/src/utils/nodes.ts index c902a979d..e04c601d4 100644 --- a/frontend-modern/src/utils/nodes.ts +++ b/frontend-modern/src/utils/nodes.ts @@ -16,12 +16,12 @@ const extractHostname = (value: string): string => { }; export function getNodeDisplayName(node: T): string { - const nameRaw = typeof node.name === 'string' ? node.name.trim() : ''; - if (nameRaw) return nameRaw; - const display = typeof node.displayName === 'string' ? node.displayName.trim() : ''; if (display) return display; + const nameRaw = typeof node.name === 'string' ? node.name.trim() : ''; + if (nameRaw) return nameRaw; + const hostRaw = typeof node.host === 'string' ? node.host.trim() : ''; const hostname = extractHostname(hostRaw); if (hostname) return hostname; diff --git a/internal/monitoring/monitor.go b/internal/monitoring/monitor.go index 47de1e4f2..29e41c854 100644 --- a/internal/monitoring/monitor.go +++ b/internal/monitoring/monitor.go @@ -6,6 +6,7 @@ import ( "fmt" "math" "net" + "net/url" "sort" "strconv" "strings" @@ -64,45 +65,142 @@ func getNodeDisplayName(instance *config.PVEInstance, nodeName string) string { } friendly := strings.TrimSpace(instance.Name) - if friendly == "" { + + if instance.IsCluster { + if endpointLabel := lookupClusterEndpointLabel(instance, nodeName); endpointLabel != "" { + return endpointLabel + } + + if baseName != "" && baseName != "unknown-node" { + return baseName + } + + if friendly != "" { + return friendly + } + return baseName } - if instance.IsCluster { - if strings.EqualFold(friendly, baseName) { - return baseName - } - return fmt.Sprintf("%s (%s)", friendly, baseName) + if friendly != "" { + return friendly } - return friendly + if baseName != "" && baseName != "unknown-node" { + return baseName + } + + if label := normalizeEndpointHost(instance.Host); label != "" && !isLikelyIPAddress(label) { + return label + } + + return baseName +} + +func lookupClusterEndpointLabel(instance *config.PVEInstance, nodeName string) string { + if instance == nil { + return "" + } + + for _, endpoint := range instance.ClusterEndpoints { + if !strings.EqualFold(endpoint.NodeName, nodeName) { + continue + } + + if host := strings.TrimSpace(endpoint.Host); host != "" { + if label := normalizeEndpointHost(host); label != "" && !isLikelyIPAddress(label) { + return label + } + } + + if nodeNameLabel := strings.TrimSpace(endpoint.NodeName); nodeNameLabel != "" { + return nodeNameLabel + } + + if ip := strings.TrimSpace(endpoint.IP); ip != "" { + return ip + } + } + + return "" +} + +func normalizeEndpointHost(raw string) string { + value := strings.TrimSpace(raw) + if value == "" { + return "" + } + + if parsed, err := url.Parse(value); err == nil && parsed.Host != "" { + host := parsed.Hostname() + if host != "" { + return host + } + return parsed.Host + } + + value = strings.TrimPrefix(value, "https://") + value = strings.TrimPrefix(value, "http://") + value = strings.TrimSpace(value) + if value == "" { + return "" + } + + if idx := strings.Index(value, "/"); idx >= 0 { + value = strings.TrimSpace(value[:idx]) + } + + if idx := strings.Index(value, ":"); idx >= 0 { + value = strings.TrimSpace(value[:idx]) + } + + return value +} + +func isLikelyIPAddress(value string) bool { + if value == "" { + return false + } + + if ip := net.ParseIP(value); ip != nil { + return true + } + + // Handle IPv6 with zone identifier (fe80::1%eth0) + if i := strings.Index(value, "%"); i > 0 { + if ip := net.ParseIP(value[:i]); ip != nil { + return true + } + } + + return false } // Monitor handles all monitoring operations type Monitor struct { - config *config.Config - state *models.State - pveClients map[string]PVEClientInterface - pbsClients map[string]*pbs.Client - tempCollector *TemperatureCollector // SSH-based temperature collector - mu sync.RWMutex - startTime time.Time - rateTracker *RateTracker - metricsHistory *MetricsHistory - alertManager *alerts.Manager - notificationMgr *notifications.NotificationManager - configPersist *config.ConfigPersistence - discoveryService *discovery.Service // Background discovery service - activePollCount int32 // Number of active polling operations - pollCounter int64 // Counter for polling cycles - authFailures map[string]int // Track consecutive auth failures per node - lastAuthAttempt map[string]time.Time // Track last auth attempt time - lastClusterCheck map[string]time.Time // Track last cluster check for standalone nodes - lastPhysicalDiskPoll map[string]time.Time // Track last physical disk poll time per instance - persistence *config.ConfigPersistence // Add persistence for saving updated configs - pbsBackupPollers map[string]bool // Track PBS backup polling goroutines per instance - runtimeCtx context.Context // Context used while monitor is running - wsHub *websocket.Hub // Hub used for broadcasting state + config *config.Config + state *models.State + pveClients map[string]PVEClientInterface + pbsClients map[string]*pbs.Client + tempCollector *TemperatureCollector // SSH-based temperature collector + mu sync.RWMutex + startTime time.Time + rateTracker *RateTracker + metricsHistory *MetricsHistory + alertManager *alerts.Manager + notificationMgr *notifications.NotificationManager + configPersist *config.ConfigPersistence + discoveryService *discovery.Service // Background discovery service + activePollCount int32 // Number of active polling operations + pollCounter int64 // Counter for polling cycles + authFailures map[string]int // Track consecutive auth failures per node + lastAuthAttempt map[string]time.Time // Track last auth attempt time + lastClusterCheck map[string]time.Time // Track last cluster check for standalone nodes + lastPhysicalDiskPoll map[string]time.Time // Track last physical disk poll time per instance + persistence *config.ConfigPersistence // Add persistence for saving updated configs + pbsBackupPollers map[string]bool // Track PBS backup polling goroutines per instance + runtimeCtx context.Context // Context used while monitor is running + wsHub *websocket.Hub // Hub used for broadcasting state } // safePercentage calculates percentage safely, returning 0 if divisor is 0 @@ -639,24 +737,24 @@ func New(cfg *config.Config) (*Monitor, error) { tempCollector := NewTemperatureCollector("root", "") m := &Monitor{ - config: cfg, - state: models.NewState(), - pveClients: make(map[string]PVEClientInterface), - pbsClients: make(map[string]*pbs.Client), - tempCollector: tempCollector, - startTime: time.Now(), - rateTracker: NewRateTracker(), - metricsHistory: NewMetricsHistory(1000, 24*time.Hour), // Keep up to 1000 points or 24 hours - alertManager: alerts.NewManager(), - notificationMgr: notifications.NewNotificationManager(cfg.PublicURL), - configPersist: config.NewConfigPersistence(cfg.DataPath), - discoveryService: nil, // Will be initialized in Start() - authFailures: make(map[string]int), - lastAuthAttempt: make(map[string]time.Time), - lastClusterCheck: make(map[string]time.Time), + config: cfg, + state: models.NewState(), + pveClients: make(map[string]PVEClientInterface), + pbsClients: make(map[string]*pbs.Client), + tempCollector: tempCollector, + startTime: time.Now(), + rateTracker: NewRateTracker(), + metricsHistory: NewMetricsHistory(1000, 24*time.Hour), // Keep up to 1000 points or 24 hours + alertManager: alerts.NewManager(), + notificationMgr: notifications.NewNotificationManager(cfg.PublicURL), + configPersist: config.NewConfigPersistence(cfg.DataPath), + discoveryService: nil, // Will be initialized in Start() + authFailures: make(map[string]int), + lastAuthAttempt: make(map[string]time.Time), + lastClusterCheck: make(map[string]time.Time), lastPhysicalDiskPoll: make(map[string]time.Time), - persistence: config.NewConfigPersistence(cfg.DataPath), - pbsBackupPollers: make(map[string]bool), + persistence: config.NewConfigPersistence(cfg.DataPath), + pbsBackupPollers: make(map[string]bool), } // Load saved configurations @@ -1866,126 +1964,126 @@ func (m *Monitor) pollPVEInstance(ctx context.Context, instanceName string, clie Dur("interval", pollingInterval). Msg("Starting disk health polling") - // Get existing disks from state to preserve data for offline nodes - currentState := m.state.GetSnapshot() - existingDisksMap := make(map[string]models.PhysicalDisk) - for _, disk := range currentState.PhysicalDisks { - if disk.Instance == instanceName { - existingDisksMap[disk.ID] = disk - } - } - - var allDisks []models.PhysicalDisk - polledNodes := make(map[string]bool) // Track which nodes we successfully polled - - for _, node := range nodes { - // Skip offline nodes but preserve their existing disk data - if node.Status != "online" { - log.Debug().Str("node", node.Node).Msg("Skipping disk poll for offline node - preserving existing data") - continue + // Get existing disks from state to preserve data for offline nodes + currentState := m.state.GetSnapshot() + existingDisksMap := make(map[string]models.PhysicalDisk) + for _, disk := range currentState.PhysicalDisks { + if disk.Instance == instanceName { + existingDisksMap[disk.ID] = disk + } } - // Get disk list for this node - log.Debug().Str("node", node.Node).Msg("Getting disk list for node") - disks, err := client.GetDisks(ctx, node.Node) - if err != nil { - // Check if it's a permission error or if the endpoint doesn't exist - if strings.Contains(err.Error(), "401") || strings.Contains(err.Error(), "403") { - log.Warn(). - Str("node", node.Node). - Err(err). - Msg("Insufficient permissions to access disk information - check API token permissions") - } else if strings.Contains(err.Error(), "404") || strings.Contains(err.Error(), "501") { - log.Info(). - Str("node", node.Node). - Msg("Disk monitoring not available on this node (may be using non-standard storage)") - } else { - log.Warn(). - Str("node", node.Node). - Err(err). - Msg("Failed to get disk list") - } - continue - } + var allDisks []models.PhysicalDisk + polledNodes := make(map[string]bool) // Track which nodes we successfully polled - log.Debug(). - Str("node", node.Node). - Int("diskCount", len(disks)). - Msg("Got disk list for node") + for _, node := range nodes { + // Skip offline nodes but preserve their existing disk data + if node.Status != "online" { + log.Debug().Str("node", node.Node).Msg("Skipping disk poll for offline node - preserving existing data") + continue + } - // Mark this node as successfully polled - polledNodes[node.Node] = true + // Get disk list for this node + log.Debug().Str("node", node.Node).Msg("Getting disk list for node") + disks, err := client.GetDisks(ctx, node.Node) + if err != nil { + // Check if it's a permission error or if the endpoint doesn't exist + if strings.Contains(err.Error(), "401") || strings.Contains(err.Error(), "403") { + log.Warn(). + Str("node", node.Node). + Err(err). + Msg("Insufficient permissions to access disk information - check API token permissions") + } else if strings.Contains(err.Error(), "404") || strings.Contains(err.Error(), "501") { + log.Info(). + Str("node", node.Node). + Msg("Disk monitoring not available on this node (may be using non-standard storage)") + } else { + log.Warn(). + Str("node", node.Node). + Err(err). + Msg("Failed to get disk list") + } + continue + } - // Check each disk for health issues and add to state - for _, disk := range disks { - // Create PhysicalDisk model - diskID := fmt.Sprintf("%s-%s-%s", instanceName, node.Node, strings.ReplaceAll(disk.DevPath, "/", "-")) - physicalDisk := models.PhysicalDisk{ - ID: diskID, - Node: node.Node, - Instance: instanceName, - DevPath: disk.DevPath, - Model: disk.Model, - Serial: disk.Serial, - Type: disk.Type, - Size: disk.Size, - Health: disk.Health, - Wearout: disk.Wearout, - RPM: disk.RPM, - Used: disk.Used, - LastChecked: time.Now(), - } - - allDisks = append(allDisks, physicalDisk) - - log.Debug(). - Str("node", node.Node). - Str("disk", disk.DevPath). - Str("model", disk.Model). - Str("health", disk.Health). - Int("wearout", disk.Wearout). - Msg("Checking disk health") - - normalizedHealth := strings.ToUpper(strings.TrimSpace(disk.Health)) - if normalizedHealth != "" && normalizedHealth != "UNKNOWN" && normalizedHealth != "PASSED" && normalizedHealth != "OK" { - // Disk has failed or is failing - alert manager will handle this - log.Warn(). - Str("node", node.Node). - Str("disk", disk.DevPath). - Str("model", disk.Model). - Str("health", disk.Health). - Int("wearout", disk.Wearout). - Msg("Disk health issue detected") - - // Pass disk info to alert manager - m.alertManager.CheckDiskHealth(instanceName, node.Node, disk) - } else if disk.Wearout > 0 && disk.Wearout < 10 { - // Low wearout warning (less than 10% life remaining) - log.Warn(). - Str("node", node.Node). - Str("disk", disk.DevPath). - Str("model", disk.Model). - Int("wearout", disk.Wearout). - Msg("SSD wearout critical - less than 10% life remaining") - - // Pass to alert manager for wearout alert - m.alertManager.CheckDiskHealth(instanceName, node.Node, disk) - } - } - } - - // Preserve existing disk data for nodes that weren't polled (offline or error) - for _, existingDisk := range existingDisksMap { - // Only preserve if we didn't poll this node - if !polledNodes[existingDisk.Node] { - // Keep the existing disk data but update the LastChecked to indicate it's stale - allDisks = append(allDisks, existingDisk) log.Debug(). - Str("node", existingDisk.Node). - Str("disk", existingDisk.DevPath). - Msg("Preserving existing disk data for unpolled node") + Str("node", node.Node). + Int("diskCount", len(disks)). + Msg("Got disk list for node") + + // Mark this node as successfully polled + polledNodes[node.Node] = true + + // Check each disk for health issues and add to state + for _, disk := range disks { + // Create PhysicalDisk model + diskID := fmt.Sprintf("%s-%s-%s", instanceName, node.Node, strings.ReplaceAll(disk.DevPath, "/", "-")) + physicalDisk := models.PhysicalDisk{ + ID: diskID, + Node: node.Node, + Instance: instanceName, + DevPath: disk.DevPath, + Model: disk.Model, + Serial: disk.Serial, + Type: disk.Type, + Size: disk.Size, + Health: disk.Health, + Wearout: disk.Wearout, + RPM: disk.RPM, + Used: disk.Used, + LastChecked: time.Now(), + } + + allDisks = append(allDisks, physicalDisk) + + log.Debug(). + Str("node", node.Node). + Str("disk", disk.DevPath). + Str("model", disk.Model). + Str("health", disk.Health). + Int("wearout", disk.Wearout). + Msg("Checking disk health") + + normalizedHealth := strings.ToUpper(strings.TrimSpace(disk.Health)) + if normalizedHealth != "" && normalizedHealth != "UNKNOWN" && normalizedHealth != "PASSED" && normalizedHealth != "OK" { + // Disk has failed or is failing - alert manager will handle this + log.Warn(). + Str("node", node.Node). + Str("disk", disk.DevPath). + Str("model", disk.Model). + Str("health", disk.Health). + Int("wearout", disk.Wearout). + Msg("Disk health issue detected") + + // Pass disk info to alert manager + m.alertManager.CheckDiskHealth(instanceName, node.Node, disk) + } else if disk.Wearout > 0 && disk.Wearout < 10 { + // Low wearout warning (less than 10% life remaining) + log.Warn(). + Str("node", node.Node). + Str("disk", disk.DevPath). + Str("model", disk.Model). + Int("wearout", disk.Wearout). + Msg("SSD wearout critical - less than 10% life remaining") + + // Pass to alert manager for wearout alert + m.alertManager.CheckDiskHealth(instanceName, node.Node, disk) + } + } + } + + // Preserve existing disk data for nodes that weren't polled (offline or error) + for _, existingDisk := range existingDisksMap { + // Only preserve if we didn't poll this node + if !polledNodes[existingDisk.Node] { + // Keep the existing disk data but update the LastChecked to indicate it's stale + allDisks = append(allDisks, existingDisk) + log.Debug(). + Str("node", existingDisk.Node). + Str("disk", existingDisk.DevPath). + Msg("Preserving existing disk data for unpolled node") + } } - } // Update physical disks in state log.Debug().