diff --git a/docs/release-control/v6/internal/subsystems/alerts.md b/docs/release-control/v6/internal/subsystems/alerts.md
index 37cb71d26..3722e139b 100644
--- a/docs/release-control/v6/internal/subsystems/alerts.md
+++ b/docs/release-control/v6/internal/subsystems/alerts.md
@@ -15,6 +15,12 @@
## Purpose
+The alerts overview offers the existing delivery-status refresh control when
+health is unavailable, including after a successful retained-queue action whose
+follow-up health read fails. The warning remains until a verified healthy read;
+a successful queue action alone is not evidence of delivery health. Normal
+degraded summary presentation continues to omit refresh.
+
Confirmed canonical metric recovery publishes the clearing evaluation's value,
observation time, and resolved metric wording in the snapshot consumed by
recent-resolution reads and notification callbacks. It must not reuse the last
@@ -71,6 +77,22 @@ their updated thresholds and may apply explicit resource-disable policies, but
it must not treat provider-owned incidents as missing thresholds. Unrelated
configuration saves preserve those incidents and their acknowledgement state
until their provider evaluator supplies recovery evidence.
+Storage configuration re-evaluation must use the same ordered resource ID and
+alias override lookup as polling. Static and forecast capacity alerts retain
+storage policy aliases in durable metadata, including through JSON and SQLite
+restore; a configuration reload must not fabricate recovery by substituting
+global defaults for a still-applicable datastore override. Older PBS snapshots
+without alias metadata may reconstruct the canonical datastore alias only when
+the recorded PBS instance, datastore name and complete legacy resource ID agree.
+Hyphenated names must not be split heuristically, and a same-named datastore on
+another instance must not inherit the override. Explicit policy changes retain
+normal resolution semantics.
+`TestPBSDatastoreOverrideLifecycleAcrossRestart` and
+`TestStoragePolicyAliasesLegacyIdentity` in
+`internal/alerts/canonical_stateful_test.go` pin restored incident identity,
+hysteresis, confirmed recovery, refiring, event counts and instance isolation
+for both persisted-alias and legacy snapshots.
+
When a VM or container stops, guest evaluation resolves only metric-threshold
alerts whose observations are no longer meaningful. Backup-age and snapshot
posture remain owned by their posture evaluator and may stay active while the
diff --git a/docs/release-control/v6/internal/subsystems/frontend-primitives.md b/docs/release-control/v6/internal/subsystems/frontend-primitives.md
index 26a89168b..0828f81b6 100644
--- a/docs/release-control/v6/internal/subsystems/frontend-primitives.md
+++ b/docs/release-control/v6/internal/subsystems/frontend-primitives.md
@@ -20,6 +20,12 @@
## Purpose
+The alerts overview offers the existing delivery-status refresh control when
+health is unavailable, including after a successful retained-queue action whose
+follow-up health read fails. The warning remains until a verified healthy read;
+a successful queue action alone is not evidence of delivery health. Normal
+degraded summary presentation continues to omit refresh.
+
Proxmox backup presentation treats every manifestless PBS artifact as
non-recoverable. It renders the artifact as `Running` when current writer
visibility is absent or a matching writer is active, and as danger-tone
diff --git a/frontend-modern/browser-verification.json b/frontend-modern/browser-verification.json
index c18e0751c..357d04b34 100644
--- a/frontend-modern/browser-verification.json
+++ b/frontend-modern/browser-verification.json
@@ -1,16 +1,16 @@
{
"version": 1,
- "base_sha": "2a833eccdfecf7248eed63eabf5b9184a9fa2390",
- "verified_at": "2026-09-05T23:57:09.414504Z",
+ "base_sha": "4cc53d10950f2effae720d251f92bf7d143d2093",
+ "verified_at": "2026-09-06T00:55:06.543758Z",
"result": "passed",
"changed_paths": [
- "frontend-modern/src/features/alerts/useNotificationDeliveryHealth.ts"
+ "frontend-modern/src/features/alerts/OverviewTab.tsx"
],
"content_sha256": {
- "frontend-modern/src/features/alerts/useNotificationDeliveryHealth.ts": "1eadf6df30b4f1130f868b8f139561ea60fdb87e1823deeeaaf2f8fbc699a00c"
+ "frontend-modern/src/features/alerts/OverviewTab.tsx": "8f7fdc04bd0546f86152c8bb392ff3e3fb755f1f0a1c28d1ee85da3e9fe23582"
},
"routes": [
- "/qualification (isolated caller/card fixture, not application routing)"
+ "/qualification (actual OverviewTab in isolated Solid Router fixture; not application shell)"
],
"viewports": [
{
@@ -23,12 +23,11 @@
}
],
"states": [
- "Real Chromium with current useAlertDestinationsTabState, useNotificationDeliveryHealth and AlertDeliveryHealthCard; scripted API promises and queue actions. No installed backend or provider receipt.",
- "Older healthy/newer degraded, older degraded/newer healthy, older error/newer healthy, older healthy/newer error. Warning presence and rendered state remain owned by newer request. Desktop and narrow screenshots inspected; this is not a full-shell or accessibility audit."
+ "Actual OverviewTab and delivery-health hook/card in Chromium, scripted API promises, empty active alerts. Not installed backend or recipient qualification.",
+ "Degraded health, successful Retry or Dismiss followed by unavailable health, pending manual refresh, verified healthy response. Desktop and narrow unavailable screenshots visually inspected; Refresh fits without clipping. Not a full accessibility or shell audit."
],
"interactions": [
- "pulse-heavy-run -- node scripts/check-delivery-health-ordering.mjs: 12 cases passed.",
- "Configuration Retry overlaps pending mount health. Resolve newer then older and compare rendered main text and alert presence.",
- "Retry retained deliveries and Dismiss retained failures each start post-action refresh while another read is pending. Older completion leaves loading true; latest healthy response clears warning. Confirmations and API responses are scripted."
+ "pulse-heavy-run -- node scripts/check-overview-health-refresh.mjs: four cases passed (two actions at two widths).",
+ "Retry and Dismiss each accepted once; failed health read preserves unavailable warning and offers Refresh; Refresh disabled while pending; healthy read removes warning without another queue action."
]
}
diff --git a/frontend-modern/src/features/alerts/AlertDeliveryHealthCard.test.tsx b/frontend-modern/src/features/alerts/AlertDeliveryHealthCard.test.tsx
index 9f3828c05..2651c01da 100644
--- a/frontend-modern/src/features/alerts/AlertDeliveryHealthCard.test.tsx
+++ b/frontend-modern/src/features/alerts/AlertDeliveryHealthCard.test.tsx
@@ -103,6 +103,26 @@ describe('AlertDeliveryHealthCard', () => {
expect(screen.getByRole('button', { name: 'Refresh delivery status' })).toBeDisabled();
});
+ it('allows unavailable summary health to be rechecked without a queue mutation', () => {
+ const onRefresh = vi.fn();
+ render(() => (
+
+ ));
+ expect(screen.getByRole('alert')).toHaveTextContent(
+ 'Notification delivery status is unavailable',
+ );
+ fireEvent.click(screen.getByRole('button', { name: 'Refresh delivery status' }));
+ expect(onRefresh).toHaveBeenCalledOnce();
+ expect(screen.getByRole('alert')).toBeTruthy();
+ });
+
it('keeps the overview treatment concise and points directly to delivery evidence', () => {
render(() => (
diff --git a/frontend-modern/src/features/alerts/OverviewTab.tsx b/frontend-modern/src/features/alerts/OverviewTab.tsx
index 219aeb25e..de8319f39 100644
--- a/frontend-modern/src/features/alerts/OverviewTab.tsx
+++ b/frontend-modern/src/features/alerts/OverviewTab.tsx
@@ -90,7 +90,7 @@ export function OverviewTab(props: {
onDismissFailures={() => void deliveryHealthState.dismissTerminalFailures()}
detailsHref="/alerts/notifications#notification-delivery-activity"
detailLevel="summary"
- showRefresh={false}
+ showRefresh={deliveryHealthState.deliveryHealthUnavailable()}
/>
diff --git a/frontend-modern/src/features/alerts/__tests__/OverviewTab.deliveryactions.test.tsx b/frontend-modern/src/features/alerts/__tests__/OverviewTab.deliveryactions.test.tsx
index c177bc613..f7724875b 100644
--- a/frontend-modern/src/features/alerts/__tests__/OverviewTab.deliveryactions.test.tsx
+++ b/frontend-modern/src/features/alerts/__tests__/OverviewTab.deliveryactions.test.tsx
@@ -1,5 +1,5 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
-import { cleanup, render, screen, waitFor } from '@solidjs/testing-library';
+import { cleanup, fireEvent, render, screen, waitFor } from '@solidjs/testing-library';
import { DEFAULT_LOCALE, setActiveLocale } from '@/i18n';
import type { Alert } from '@/types/api';
import type { NotificationHealth } from '@/api/notifications';
@@ -49,6 +49,8 @@ vi.mock('@/components/Alerts/InvestigateAlertButton', () => ({
InvestigateAlertButton: () => null,
}));
+import { notificationStore } from '@/stores/notifications';
+
import { OverviewTab } from '../OverviewTab';
function degradedHealth(): NotificationHealth {
@@ -107,10 +109,15 @@ describe('OverviewTab delivery health actions', () => {
getDeliveryDiagnoses.mockReset();
getDeliveryDiagnoses.mockResolvedValue([]);
getHealth.mockReset();
+ retryTerminalFailures.mockReset();
+ dismissTerminalFailures.mockReset();
+ vi.mocked(notificationStore.success).mockClear();
+ vi.mocked(notificationStore.error).mockClear();
});
afterEach(() => {
cleanup();
+ vi.restoreAllMocks();
setActiveLocale(DEFAULT_LOCALE);
});
@@ -127,4 +134,110 @@ describe('OverviewTab delivery health actions', () => {
expect(screen.queryByRole('button', { name: 'Refresh delivery status' })).toBeNull();
expect(screen.getByRole('alert')).toHaveTextContent('Most recent failures: connectivity (1).');
});
+ for (const action of [
+ { name: 'Retry retained deliveries', api: retryTerminalFailures },
+ { name: 'Dismiss retained failures', api: dismissTerminalFailures },
+ ]) {
+ it(`does not mutate or refresh when ${action.name} is cancelled`, async () => {
+ getHealth.mockResolvedValue(degradedHealth());
+ const confirmation = vi.spyOn(window, 'confirm').mockReturnValue(false);
+ render(() => );
+ fireEvent.click(await screen.findByRole('button', { name: action.name }));
+
+ expect(confirmation).toHaveBeenCalledOnce();
+ expect(action.api).not.toHaveBeenCalled();
+ expect(getHealth).toHaveBeenCalledTimes(1);
+ expect(screen.getByRole('alert')).toBeTruthy();
+ });
+
+ it(`retains attention and enables another attempt when ${action.name} fails`, async () => {
+ getHealth.mockResolvedValue(degradedHealth());
+ vi.spyOn(window, 'confirm').mockReturnValue(true);
+ action.api.mockRejectedValue(new Error('queue action unavailable'));
+ render(() => );
+ fireEvent.click(await screen.findByRole('button', { name: action.name }));
+
+ await waitFor(() => expect(notificationStore.error).toHaveBeenCalledOnce());
+ expect(notificationStore.success).not.toHaveBeenCalled();
+ expect(getHealth).toHaveBeenCalledTimes(1);
+ expect(screen.getByRole('alert')).toHaveTextContent('connectivity (1)');
+ expect(screen.getByRole('button', { name: action.name })).not.toBeDisabled();
+ });
+
+ it(`keeps health visibly unknown after ${action.name} succeeds but refresh fails`, async () => {
+ getHealth
+ .mockResolvedValueOnce(degradedHealth())
+ .mockRejectedValueOnce(new Error('health unavailable'));
+ action.api.mockResolvedValue({ affected: 85 });
+ vi.spyOn(window, 'confirm').mockReturnValue(true);
+ render(() => );
+ fireEvent.click(await screen.findByRole('button', { name: action.name }));
+
+ const refresh = await screen.findByRole('button', { name: 'Refresh delivery status' });
+ expect(screen.getByRole('alert')).toBeTruthy();
+ expect(notificationStore.success).toHaveBeenCalledOnce();
+ expect(notificationStore.error).not.toHaveBeenCalled();
+
+ const healthy = degradedHealth();
+ healthy.overallHealthy = true;
+ healthy.queue = {
+ ...healthy.queue,
+ status: 'healthy',
+ healthy: true,
+ attentionRequired: 0,
+ deadLetter: 0,
+ };
+ getHealth.mockResolvedValueOnce(healthy);
+ await waitFor(() => expect(refresh).not.toBeDisabled());
+ fireEvent.click(refresh);
+ await waitFor(() => expect(screen.queryByRole('alert')).toBeNull());
+ expect(getHealth).toHaveBeenCalledTimes(3);
+ expect(action.api).toHaveBeenCalledOnce();
+ });
+
+ it(`keeps both actions disabled until ${action.name} and its health refresh finish`, async () => {
+ const healthy = degradedHealth();
+ healthy.overallHealthy = true;
+ healthy.queue = {
+ ...healthy.queue,
+ status: 'healthy',
+ healthy: true,
+ attentionRequired: 0,
+ deadLetter: 0,
+ };
+ let completeAction!: (value: { affected: number }) => void;
+ let completeHealth!: (value: NotificationHealth) => void;
+ action.api.mockReturnValue(
+ new Promise((resolve) => {
+ completeAction = resolve;
+ }),
+ );
+ getHealth.mockResolvedValueOnce(degradedHealth()).mockReturnValueOnce(
+ new Promise((resolve) => {
+ completeHealth = resolve;
+ }),
+ );
+ vi.spyOn(window, 'confirm').mockReturnValue(true);
+ render(() => );
+ fireEvent.click(await screen.findByRole('button', { name: action.name }));
+
+ const expectActionsDisabled = () => {
+ const buttons = screen.getByRole('alert').querySelectorAll('button');
+ expect(buttons.length).toBe(2);
+ for (const button of buttons) expect(button).toBeDisabled();
+ };
+ expectActionsDisabled();
+ expect(getHealth).toHaveBeenCalledTimes(1);
+ completeAction({ affected: 85 });
+ await waitFor(() => expect(getHealth).toHaveBeenCalledTimes(2));
+ expectActionsDisabled();
+ expect(screen.getByRole('alert')).toBeTruthy();
+
+ completeHealth(healthy);
+ await waitFor(() => expect(screen.queryByRole('alert')).toBeNull());
+ expect(action.api).toHaveBeenCalledOnce();
+ expect(notificationStore.success).toHaveBeenCalledOnce();
+ expect(notificationStore.error).not.toHaveBeenCalled();
+ });
+ }
});
diff --git a/internal/alerts/canonical_stateful_test.go b/internal/alerts/canonical_stateful_test.go
index 5587e38c0..7920c9563 100644
--- a/internal/alerts/canonical_stateful_test.go
+++ b/internal/alerts/canonical_stateful_test.go
@@ -4,7 +4,9 @@ import (
"testing"
"time"
+ "github.com/rcourtman/pulse-go-rewrite/internal/alerts/eventlog"
alertspecs "github.com/rcourtman/pulse-go-rewrite/internal/alerts/specs"
+ "github.com/rcourtman/pulse-go-rewrite/internal/models"
"github.com/rcourtman/pulse-go-rewrite/internal/storagehealth"
"github.com/rcourtman/pulse-go-rewrite/internal/unifiedresources"
)
@@ -165,3 +167,115 @@ func TestStatefulAlertReFireCooldown(t *testing.T) {
}
})
}
+
+// A UI-keyed datastore override must govern actual incidents, not just the
+// threshold resolver. Exercise hysteresis and recurrence across SQLite reopen,
+// including a same-named datastore on another PBS instance.
+func TestPBSDatastoreOverrideLifecycleAcrossRestart(t *testing.T) {
+ for _, legacySnapshot := range []bool{false, true} {
+ name := "persisted aliases"
+ if legacySnapshot {
+ name = "legacy snapshot"
+ }
+ t.Run(name, func(t *testing.T) {
+ dir := t.TempDir()
+ start := func() *Manager {
+ m := NewManagerWithDataDir(dir)
+ t.Cleanup(m.Stop)
+ m.EnableEventLog()
+ if !m.activeStateAuthoritative.Load() {
+ t.Fatal("SQLite active state is not authoritative")
+ }
+ m.UpdateConfig(AlertConfig{Enabled: true, ActivationState: ActivationActive,
+ StorageDefault: HysteresisThreshold{Trigger: 95, Clear: 90},
+ Overrides: map[string]ThresholdConfig{"pbs-primary/backups": {Usage: &HysteresisThreshold{Trigger: 80, Clear: 70}}},
+ })
+ disableTestTimeThresholds(m)
+ return m
+ }
+ storage := func(instance string, usage float64) models.Storage {
+ return models.Storage{ID: instance + "-backups", AliasIDs: []string{instance + "/backups"}, Name: "backups", Instance: instance, Type: "pbs", Status: "online", Total: 1000, Used: int64(usage * 10), Free: int64(1000 - usage*10), Usage: usage}
+ }
+ observe := func(m *Manager, usage float64) {
+ t.Helper()
+ for range 5 {
+ m.CheckStorage(storage("pbs-primary", usage))
+ m.CheckStorage(storage("pbs-secondary", usage))
+ }
+ if testHasActiveAlert(t, m, canonicalMetricStateID("pbs-secondary-backups", "usage")) {
+ t.Fatal("override leaked to another PBS instance")
+ }
+ }
+ events := func(m *Manager, fired, resolved int) {
+ t.Helper()
+ for kind, want := range map[string]int{eventlog.TypeFired: fired, eventlog.TypeResolved: resolved} {
+ if got := len(queryAlertEvents(t, m, eventlog.Filter{Types: []string{kind}})); got != want {
+ t.Fatalf("%s events = %d, want %d", kind, got, want)
+ }
+ }
+ }
+ id := canonicalMetricStateID("pbs-primary-backups", "usage")
+ m := start()
+ observe(m, 85)
+ original := *testRequireActiveAlert(t, m, id)
+ events(m, 1, 0)
+ if legacySnapshot {
+ m.mu.Lock()
+ delete(m.activeAlerts[id].Metadata, storagePolicyAliasesKey)
+ m.mu.Unlock()
+ if err := m.SaveActiveAlerts(); err != nil {
+ t.Fatal(err)
+ }
+ }
+ m.Stop()
+
+ m = start()
+ observe(m, 75) // Below trigger, but not below the override's clear threshold.
+ if got := testRequireActiveAlert(t, m, id); !got.StartTime.Equal(original.StartTime) {
+ t.Fatal("restart replaced the firing incident")
+ }
+ events(m, 1, 0)
+ observe(m, 65)
+ if testHasActiveAlert(t, m, id) {
+ t.Fatal("override recovery did not clear incident")
+ }
+ events(m, 1, 1)
+ m.Stop()
+
+ m = start()
+ observe(m, 75)
+ if testHasActiveAlert(t, m, id) {
+ t.Fatal("resolved incident resurrected inside hysteresis band")
+ }
+ events(m, 1, 1)
+ observe(m, 85)
+ if got := testRequireActiveAlert(t, m, id); !got.StartTime.After(original.StartTime) {
+ t.Fatal("refire reused original incident start")
+ }
+ events(m, 2, 1)
+ })
+ }
+}
+
+func TestStoragePolicyAliasesLegacyIdentity(t *testing.T) {
+ for _, tc := range []struct {
+ name, instance, resource, datastore string
+ want bool
+ }{
+ {"hyphenated names", "pbs-backup-east", "pbs-backup-east-daily-store", "daily-store", true},
+ {"different instance", "pbs-backup-west", "pbs-backup-east-daily-store", "daily-store", false},
+ {"not PBS", "backup-east", "backup-east-daily-store", "daily-store", false},
+ {"missing datastore", "pbs-backup-east", "pbs-backup-east-", "", false},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ got := storagePolicyAliases(&Alert{Instance: tc.instance, ResourceID: tc.resource, ResourceName: tc.datastore})
+ if tc.want {
+ if len(got) != 1 || got[0] != tc.instance+"/"+tc.datastore {
+ t.Fatalf("aliases = %v", got)
+ }
+ } else if len(got) != 0 {
+ t.Fatalf("invented alias: %v", got)
+ }
+ })
+ }
+}
diff --git a/internal/alerts/capacity_forecast.go b/internal/alerts/capacity_forecast.go
index b435829b3..e7c2cc7d0 100644
--- a/internal/alerts/capacity_forecast.go
+++ b/internal/alerts/capacity_forecast.go
@@ -215,12 +215,13 @@ var capacityForecastMetadataKeys = []string{
func (m *Manager) evaluateStorageCapacity(storage models.Storage, thresholds ThresholdConfig, trend CapacityTrendObservation) {
input := &UnifiedResourceInput{
- ID: storage.ID,
- Type: "storage",
- Name: storage.Name,
- Node: storage.Node,
- Instance: storage.Instance,
- Disk: &UnifiedResourceMetric{Percent: storage.Usage},
+ ID: storage.ID,
+ StorageAliases: storage.AliasIDs,
+ Type: "storage",
+ Name: storage.Name,
+ Node: storage.Node,
+ Instance: storage.Instance,
+ Disk: &UnifiedResourceMetric{Percent: storage.Usage},
}
m.evaluateUnifiedCapacity(input, thresholds, trend, func() bool {
if !m.config.Enabled {
@@ -377,6 +378,9 @@ func (m *Manager) evaluateCapacityForecast(input *UnifiedResourceInput, threshol
"forecastBucketCount": trend.BucketCount,
"forecastCoverageSeconds": int64(trend.CoverageSpan / time.Second),
}
+ if len(input.StorageAliases) > 0 {
+ metadata[storagePolicyAliasesKey] = append([]string(nil), input.StorageAliases...)
+ }
_, _ = m.evaluateCanonicalLifecycleAlert(canonicalLifecycleAlertParams{
Spec: spec,
Evidence: alertspecs.AlertEvidence{
diff --git a/internal/alerts/config_runtime.go b/internal/alerts/config_runtime.go
index efeabf819..27202d3d0 100644
--- a/internal/alerts/config_runtime.go
+++ b/internal/alerts/config_runtime.go
@@ -503,7 +503,7 @@ func (m *Manager) reevaluateActiveAlertsLocked() {
alertsToResolve = append(alertsToResolve, alertID)
continue
}
- thresholds := m.resolveResourceThresholds("storage", resourceID)
+ thresholds := m.effectiveAlertPolicyNoLock(alertPolicyQuery{TypeKey: "storage", ResourceID: resourceID, StorageAliases: storagePolicyAliases(alert)}).Thresholds
if thresholds.Disabled {
alertsToResolve = append(alertsToResolve, alertID)
continue
diff --git a/internal/alerts/storage_override_identity.go b/internal/alerts/storage_override_identity.go
index a0c906114..00c19a86a 100644
--- a/internal/alerts/storage_override_identity.go
+++ b/internal/alerts/storage_override_identity.go
@@ -67,3 +67,32 @@ func (m *Manager) resolveStorageThresholdOverride(base ThresholdConfig, resource
func (m *Manager) resolveStorageThresholdsNoLock(storage models.Storage) ThresholdConfig {
return m.resolveStorageThresholdOverride(m.defaultThresholdsForResourceType("storage"), storage.ID, storage.AliasIDs)
}
+
+// Keep the polling identity through JSON/SQLite restore so configuration saves
+// cannot silently replace a datastore override with global storage defaults.
+const storagePolicyAliasesKey = "storagePolicyAliases"
+
+func storagePolicyAliases(alert *Alert) []string {
+ if alert == nil {
+ return nil
+ }
+ switch aliases := alert.Metadata[storagePolicyAliasesKey].(type) {
+ case []string:
+ return append([]string(nil), aliases...)
+ case []interface{}:
+ result := make([]string, 0, len(aliases))
+ for _, value := range aliases {
+ if alias, ok := value.(string); ok {
+ result = append(result, alias)
+ }
+ }
+ return result
+ }
+ // Older PBS snapshots predate alias metadata. The poller records the
+ // instance and datastore name separately: only reconstruct an alias when
+ // the complete legacy ID agrees, never split hyphenated names heuristically.
+ if strings.HasPrefix(alert.Instance, "pbs-") && alert.ResourceName != "" && alert.ResourceID == alert.Instance+"-"+alert.ResourceName {
+ return []string{alert.Instance + "/" + alert.ResourceName}
+ }
+ return nil
+}
diff --git a/internal/alerts/unified_eval.go b/internal/alerts/unified_eval.go
index 7e065cb9b..bcffe8100 100644
--- a/internal/alerts/unified_eval.go
+++ b/internal/alerts/unified_eval.go
@@ -55,6 +55,8 @@ type UnifiedResourceInput struct {
NetworkIn *UnifiedResourceMetric
NetworkOut *UnifiedResourceMetric
Temperature *UnifiedResourceMetric
+
+ StorageAliases []string // Durable policy lookup identities for storage alerts.
}
type unifiedMetricCandidate struct {
@@ -259,6 +261,19 @@ func (m *Manager) evaluateUnifiedMetrics(input *UnifiedResourceInput, thresholds
}
opts = metricOptionsWithTags(opts, input.Tags)
+ if len(input.StorageAliases) > 0 {
+ merged := metricOptions{}
+ if opts != nil {
+ merged = *opts
+ }
+ metadata := make(map[string]interface{}, len(merged.Metadata)+1)
+ for k, v := range merged.Metadata {
+ metadata[k] = v
+ }
+ metadata[storagePolicyAliasesKey] = append([]string(nil), input.StorageAliases...)
+ merged.Metadata = metadata
+ opts = &merged
+ }
for _, candidate := range buildUnifiedMetricCandidates(input, thresholds) {
m.checkMetricWithCanonicalSpec(candidate.Spec, input.Name, input.Node, input.Instance, unifiedAlertType(input.Type), candidate.Value, candidate.Threshold, opts)
}
diff --git a/internal/notifications/ntfy_outage_test.go b/internal/notifications/ntfy_outage_test.go
new file mode 100644
index 000000000..ad779027f
--- /dev/null
+++ b/internal/notifications/ntfy_outage_test.go
@@ -0,0 +1,122 @@
+package notifications
+
+import (
+ "encoding/json"
+ "io"
+ "net/http"
+ "strings"
+ "sync/atomic"
+ "testing"
+ "time"
+
+ "github.com/rcourtman/pulse-go-rewrite/internal/alerts"
+)
+
+// Provider rejection must preserve the firing receipt needed for recovery,
+// including when an operator retries the recovery after reopening SQLite.
+func TestQueuedNtfyRecoveryAfterProviderOutageAndRestart(t *testing.T) {
+ var unavailable atomic.Bool
+ var accepted atomic.Int32
+ server := newIPv4HTTPServer(t, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ body, err := io.ReadAll(r.Body)
+ if err != nil {
+ t.Errorf("read body: %v", err)
+ }
+ if unavailable.Load() {
+ w.WriteHeader(http.StatusServiceUnavailable)
+ return
+ }
+ if accepted.Add(1) == 2 {
+ if r.Header.Get("Priority") != "default" || r.Header.Get("Title") != "RESOLVED: database" ||
+ !strings.Contains(string(body), "is now healthy") {
+ t.Errorf("incorrect recovery: headers=%v body=%q", r.Header, body)
+ }
+ }
+ w.WriteHeader(http.StatusAccepted)
+ }))
+ defer server.Close()
+ dir := t.TempDir()
+ webhook := WebhookConfig{ID: "ops", Name: "ops", URL: server.URL + "/topic", Enabled: true, Service: "ntfy"}
+ open := func() *NotificationManager {
+ m := NewNotificationManagerWithDataDir("", dir)
+ m.webhookClient = server.Client()
+ if err := m.UpdateAllowedPrivateCIDRs("127.0.0.1/32"); err != nil {
+ t.Fatal(err)
+ }
+ m.AddWebhook(webhook)
+ t.Cleanup(m.Stop)
+ return m
+ }
+ m := open()
+ config, err := json.Marshal(webhook)
+ if err != nil {
+ t.Fatal(err)
+ }
+ alert := &alerts.Alert{ID: "cpu", ResourceName: "database", Type: "cpu", Level: alerts.AlertLevelCritical,
+ Message: "CPU above threshold", StartTime: time.Now().Add(-time.Minute)}
+ resolved := notificationDeliveryJob{Type: "webhook", Event: eventResolved, Alerts: []*alerts.Alert{alert}, WebhookConfig: &webhook}
+ wait := func(m *NotificationManager, id string, want NotificationQueueStatus, wantAudits int) {
+ t.Helper()
+ deadline := time.Now().Add(15 * time.Second)
+ for {
+ var status string
+ var audits int
+ if err := m.queue.db.QueryRow("SELECT status FROM notification_queue WHERE id = ?", id).Scan(&status); err != nil {
+ t.Fatal(err)
+ }
+ if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = ?", id).Scan(&audits); err != nil {
+ t.Fatal(err)
+ }
+ if status == string(want) && audits == wantAudits {
+ return
+ }
+ if time.Now().After(deadline) {
+ t.Fatalf("%s: status=%s audits=%d, want %s with audit", id, status, audits, want)
+ }
+ time.Sleep(time.Millisecond)
+ }
+ }
+ enqueue := func(id, kind string, payload *alerts.Alert) {
+ t.Helper()
+ if err := m.queue.Enqueue(&QueuedNotification{ID: id, Type: kind, Status: QueueStatusPending,
+ Config: config, Alerts: []*alerts.Alert{payload}, MaxAttempts: 1}); err != nil {
+ t.Fatal(err)
+ }
+ }
+ enqueue("firing", "webhook", alert.Clone())
+ wait(m, "firing", QueueStatusSent, 1)
+ unavailable.Store(true)
+ recovery := alert.Clone()
+ annotateResolvedMetadata(recovery, time.Now())
+ enqueue("recovery", "webhook_resolved", recovery)
+ wait(m, "recovery", QueueStatusDLQ, 1)
+ if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 1 {
+ t.Fatal("failed recovery consumed the firing receipt")
+ }
+ m.Stop()
+ m = open()
+ if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 1 {
+ t.Fatal("restart lost the firing receipt")
+ }
+ unavailable.Store(false)
+ if count, err := m.queue.RetryTerminalFailures(); err != nil || count != 1 {
+ t.Fatalf("retry = %d, %v; want one recovery", count, err)
+ }
+ wait(m, "recovery", QueueStatusSent, 2)
+ if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 0 {
+ t.Fatal("successful recovery did not consume the firing receipt")
+ }
+ if got := accepted.Load(); got != 2 {
+ t.Fatalf("accepted HTTP requests = %d, want firing and recovery only", got)
+ }
+ var failures, successes int
+ if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = 'recovery' AND success = 0").Scan(&failures); err != nil {
+ t.Fatal(err)
+ }
+ if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = 'recovery' AND success = 1").Scan(&successes); err != nil {
+ t.Fatal(err)
+ }
+ if failures != 1 || successes != 1 {
+ t.Fatalf("recovery audit failures=%d successes=%d, want 1 each", failures, successes)
+ }
+}
diff --git a/internal/notifications/ntfy_transition_test.go b/internal/notifications/ntfy_transition_test.go
index 162c7230f..398df6f3e 100644
--- a/internal/notifications/ntfy_transition_test.go
+++ b/internal/notifications/ntfy_transition_test.go
@@ -1,6 +1,7 @@
package notifications
import (
+ "encoding/json"
"io"
"net/http"
"reflect"
@@ -14,6 +15,16 @@ import (
// Exercise the shared destination across transitions: generated firing headers
// must not leak into the stored configuration and override recovery metadata.
func TestNtfySeverityRecoveryTransition(t *testing.T) {
+ for _, queued := range []bool{false, true} {
+ name := "direct"
+ if queued {
+ name = "queued"
+ }
+ t.Run(name, func(t *testing.T) { testNtfySeverityRecoveryTransition(t, queued) })
+ }
+}
+
+func testNtfySeverityRecoveryTransition(t *testing.T, queued bool) {
type receipt struct {
header http.Header
body string
@@ -28,7 +39,7 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
w.WriteHeader(http.StatusAccepted)
}))
defer server.Close()
- manager := NewNotificationManager("")
+ manager := NewNotificationManagerWithDataDir("", t.TempDir())
defer manager.Stop()
manager.webhookClient = server.Client()
if err := manager.UpdateAllowedPrivateCIDRs("127.0.0.1/32"); err != nil {
@@ -36,6 +47,7 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
}
webhook := WebhookConfig{Name: "transition", URL: server.URL + "/topic", Enabled: true, Service: "ntfy",
Headers: map[string]string{"X-Static": "preserved"}}
+ manager.AddWebhook(webhook)
originalHeaders := map[string]string{"X-Static": "preserved"}
alert := &alerts.Alert{ID: "transition", Type: "cpu", ResourceID: "vm-1", ResourceName: "database", Node: "node-a",
Message: "CPU above threshold", Value: 99, Threshold: 90, StartTime: time.Now().Add(-time.Minute)}
@@ -54,7 +66,25 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
alert.Level = step.level
before := *alert
var err error
- if step.resolved {
+ if queued {
+ config, marshalErr := json.Marshal(webhook)
+ if marshalErr != nil {
+ t.Fatal(marshalErr)
+ }
+ kind := "webhook"
+ payload := alert.Clone()
+ if step.resolved {
+ kind += "_resolved"
+ annotateResolvedMetadata(payload, time.Now())
+ }
+ if manager.queue == nil {
+ t.Fatal("notification queue unavailable")
+ }
+ err = manager.queue.Enqueue(&QueuedNotification{
+ ID: step.name, Type: kind, Status: QueueStatusPending,
+ Config: config, Alerts: []*alerts.Alert{payload}, MaxAttempts: 1,
+ })
+ } else if step.resolved {
err = manager.sendResolvedWebhook(webhook, []*alerts.Alert{alert}, time.Now())
} else {
err = manager.sendGroupedWebhook(webhook, []*alerts.Alert{alert})
@@ -86,6 +116,24 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
if !reflect.DeepEqual(*alert, before) {
t.Error("source alert mutated")
}
+ if queued {
+ // Receipt precedes the queue commit. Wait for completion before
+ // advancing the same alert identity to its next lifecycle state.
+ deadline := time.Now().Add(3 * time.Second)
+ for {
+ var status string
+ if err := manager.queue.db.QueryRow("SELECT status FROM notification_queue WHERE id = ?", step.name).Scan(&status); err != nil {
+ t.Fatal(err)
+ }
+ if status == string(QueueStatusSent) {
+ break
+ }
+ if time.Now().After(deadline) {
+ t.Fatalf("received HTTP request but queue status remained %s", status)
+ }
+ time.Sleep(time.Millisecond)
+ }
+ }
})
}
}
diff --git a/scripts/check-overview-health-refresh.mjs b/scripts/check-overview-health-refresh.mjs
new file mode 100644
index 000000000..494d67802
--- /dev/null
+++ b/scripts/check-overview-health-refresh.mjs
@@ -0,0 +1,98 @@
+// Isolated real-browser component qualification; no installed backend or delivery claim.
+import { createServer } from "../frontend-modern/node_modules/vite/dist/node/index.js";
+import solid from "../frontend-modern/node_modules/vite-plugin-solid/dist/esm/index.mjs";
+import { chromium } from "@playwright/test";
+import { resolve } from "node:path";
+import { mkdirSync } from "node:fs";
+import assert from "node:assert/strict";
+const root = resolve("frontend-modern");
+process.chdir(root);
+const fixture = `
+import { render } from 'solid-js/web';
+import { Router, Route } from '@solidjs/router';
+import { NotificationsAPI } from '/src/api/notifications';
+import { AlertsAPI } from '/src/api/alerts';
+import { OverviewTab } from '/src/features/alerts/OverviewTab';
+import '/src/index.css';
+const pending = [];
+NotificationsAPI.getHealth = () => new Promise((resolve, reject) => pending.push({resolve, reject}));
+AlertsAPI.getDeliveryDiagnoses = async () => [];
+window.actions = 0;
+NotificationsAPI.retryTerminalFailures = NotificationsAPI.dismissTerminalFailures = async () => { window.actions++; return {affected: 1}; };
+window.confirm = () => true;
+window.finish = (i, status) => status === 'error' ? pending[i].reject(new Error('scripted offline')) : pending[i].resolve({queue:{status, failed:0, deadLetter:status === 'healthy' ? 0 : 1, attentionRequired:status === 'healthy' ? 0 : 1}});
+window.count = () => pending.length;
+function Fixture() {
+return {}} showQuickTip={()=>false} dismissQuickTip={()=>{}} showAcknowledged={()=>true} setShowAcknowledged={()=>{}} alertsDisabled={()=>false}/>;
+}
+render(() => , document.getElementById('root'));
+`;
+const server = await createServer({
+ root,
+ configFile: false,
+ optimizeDeps: {
+ noDiscovery: true,
+ entries: [],
+ esbuildOptions: { target: "esnext" },
+ },
+ esbuild: { target: "esnext" },
+ plugins: [
+ solid(),
+ {
+ name: "ordering-fixture",
+ configureServer(s) {
+ s.middlewares.use((req, res, next) => {
+ if (req.url === "/qualification") {
+ res.setHeader("Content-Type", "text/html");
+ res.end(
+ '',
+ );
+ } else next();
+ });
+ },
+ resolveId(id) {
+ if (id === "/ordering-fixture.tsx") return id;
+ },
+ load(id) {
+ if (id === "/ordering-fixture.tsx") return fixture;
+ },
+ },
+ ],
+ resolve: { alias: { "@": resolve(root, "src") } },
+ server: { host: "127.0.0.1", port: 5197, strictPort: true },
+});
+let browser;
+try {
+ await server.listen();
+ browser = await chromium.launch({ headless: true });
+ mkdirSync('/tmp/pulse-overview-refresh', { recursive: true });
+ let cases = 0;
+ for (const width of [1440, 390]) {
+ for (const action of ['Retry retained deliveries', 'Dismiss retained failures']) {
+ const page = await browser.newPage({ viewport: { width, height: 900 } });
+ await page.goto('http://127.0.0.1:5197/qualification');
+ await page.waitForFunction(() => window.count?.() === 1);
+ await page.evaluate(() => window.finish(0, 'degraded'));
+ await page.getByRole('button', {name: action, exact: true}).click();
+ await page.waitForFunction(() => window.count() === 2);
+ await page.evaluate(() => window.finish(1, 'error'));
+ const refresh = page.getByRole('button', {name:'Refresh delivery status', exact:true});
+ await refresh.waitFor();
+ assert.match(await page.getByRole('alert').innerText(), /status is unavailable/);
+ await page.screenshot({path: `/tmp/pulse-overview-refresh/${width}-${cases}-unavailable.png`});
+ await refresh.click();
+ await page.waitForFunction(() => window.count() === 3);
+ assert.equal(await refresh.isDisabled(), true);
+ await page.evaluate(() => window.finish(2, 'healthy'));
+ await page.getByRole('alert').waitFor({state:'detached'});
+ assert.equal(await page.evaluate(() => window.actions), 1);
+ await page.screenshot({path: `/tmp/pulse-overview-refresh/${width}-${cases}-healthy.png`});
+ cases++;
+ await page.close();
+ }
+ }
+ console.log(`${cases} overview action/outage/refresh browser cases passed`);
+} finally {
+ await browser?.close();
+ await server.close();
+}