mirror of
https://github.com/rcourtman/Pulse.git
synced 2026-09-10 02:25:56 +00:00
Merge pull request #1923 from rcourtman/maintainer/20260906T013333Z
Keep alert recovery visible and storage thresholds stable after outages
This commit is contained in:
@@ -15,6 +15,12 @@
|
||||
|
||||
## Purpose
|
||||
|
||||
The alerts overview offers the existing delivery-status refresh control when
|
||||
health is unavailable, including after a successful retained-queue action whose
|
||||
follow-up health read fails. The warning remains until a verified healthy read;
|
||||
a successful queue action alone is not evidence of delivery health. Normal
|
||||
degraded summary presentation continues to omit refresh.
|
||||
|
||||
Confirmed canonical metric recovery publishes the clearing evaluation's value,
|
||||
observation time, and resolved metric wording in the snapshot consumed by
|
||||
recent-resolution reads and notification callbacks. It must not reuse the last
|
||||
@@ -71,6 +77,22 @@ their updated thresholds and may apply explicit resource-disable policies, but
|
||||
it must not treat provider-owned incidents as missing thresholds. Unrelated
|
||||
configuration saves preserve those incidents and their acknowledgement state
|
||||
until their provider evaluator supplies recovery evidence.
|
||||
Storage configuration re-evaluation must use the same ordered resource ID and
|
||||
alias override lookup as polling. Static and forecast capacity alerts retain
|
||||
storage policy aliases in durable metadata, including through JSON and SQLite
|
||||
restore; a configuration reload must not fabricate recovery by substituting
|
||||
global defaults for a still-applicable datastore override. Older PBS snapshots
|
||||
without alias metadata may reconstruct the canonical datastore alias only when
|
||||
the recorded PBS instance, datastore name and complete legacy resource ID agree.
|
||||
Hyphenated names must not be split heuristically, and a same-named datastore on
|
||||
another instance must not inherit the override. Explicit policy changes retain
|
||||
normal resolution semantics.
|
||||
`TestPBSDatastoreOverrideLifecycleAcrossRestart` and
|
||||
`TestStoragePolicyAliasesLegacyIdentity` in
|
||||
`internal/alerts/canonical_stateful_test.go` pin restored incident identity,
|
||||
hysteresis, confirmed recovery, refiring, event counts and instance isolation
|
||||
for both persisted-alias and legacy snapshots.
|
||||
|
||||
When a VM or container stops, guest evaluation resolves only metric-threshold
|
||||
alerts whose observations are no longer meaningful. Backup-age and snapshot
|
||||
posture remain owned by their posture evaluator and may stay active while the
|
||||
|
||||
@@ -20,6 +20,12 @@
|
||||
|
||||
## Purpose
|
||||
|
||||
The alerts overview offers the existing delivery-status refresh control when
|
||||
health is unavailable, including after a successful retained-queue action whose
|
||||
follow-up health read fails. The warning remains until a verified healthy read;
|
||||
a successful queue action alone is not evidence of delivery health. Normal
|
||||
degraded summary presentation continues to omit refresh.
|
||||
|
||||
Proxmox backup presentation treats every manifestless PBS artifact as
|
||||
non-recoverable. It renders the artifact as `Running` when current writer
|
||||
visibility is absent or a matching writer is active, and as danger-tone
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
{
|
||||
"version": 1,
|
||||
"base_sha": "2a833eccdfecf7248eed63eabf5b9184a9fa2390",
|
||||
"verified_at": "2026-09-05T23:57:09.414504Z",
|
||||
"base_sha": "4cc53d10950f2effae720d251f92bf7d143d2093",
|
||||
"verified_at": "2026-09-06T00:55:06.543758Z",
|
||||
"result": "passed",
|
||||
"changed_paths": [
|
||||
"frontend-modern/src/features/alerts/useNotificationDeliveryHealth.ts"
|
||||
"frontend-modern/src/features/alerts/OverviewTab.tsx"
|
||||
],
|
||||
"content_sha256": {
|
||||
"frontend-modern/src/features/alerts/useNotificationDeliveryHealth.ts": "1eadf6df30b4f1130f868b8f139561ea60fdb87e1823deeeaaf2f8fbc699a00c"
|
||||
"frontend-modern/src/features/alerts/OverviewTab.tsx": "8f7fdc04bd0546f86152c8bb392ff3e3fb755f1f0a1c28d1ee85da3e9fe23582"
|
||||
},
|
||||
"routes": [
|
||||
"/qualification (isolated caller/card fixture, not application routing)"
|
||||
"/qualification (actual OverviewTab in isolated Solid Router fixture; not application shell)"
|
||||
],
|
||||
"viewports": [
|
||||
{
|
||||
@@ -23,12 +23,11 @@
|
||||
}
|
||||
],
|
||||
"states": [
|
||||
"Real Chromium with current useAlertDestinationsTabState, useNotificationDeliveryHealth and AlertDeliveryHealthCard; scripted API promises and queue actions. No installed backend or provider receipt.",
|
||||
"Older healthy/newer degraded, older degraded/newer healthy, older error/newer healthy, older healthy/newer error. Warning presence and rendered state remain owned by newer request. Desktop and narrow screenshots inspected; this is not a full-shell or accessibility audit."
|
||||
"Actual OverviewTab and delivery-health hook/card in Chromium, scripted API promises, empty active alerts. Not installed backend or recipient qualification.",
|
||||
"Degraded health, successful Retry or Dismiss followed by unavailable health, pending manual refresh, verified healthy response. Desktop and narrow unavailable screenshots visually inspected; Refresh fits without clipping. Not a full accessibility or shell audit."
|
||||
],
|
||||
"interactions": [
|
||||
"pulse-heavy-run -- node scripts/check-delivery-health-ordering.mjs: 12 cases passed.",
|
||||
"Configuration Retry overlaps pending mount health. Resolve newer then older and compare rendered main text and alert presence.",
|
||||
"Retry retained deliveries and Dismiss retained failures each start post-action refresh while another read is pending. Older completion leaves loading true; latest healthy response clears warning. Confirmations and API responses are scripted."
|
||||
"pulse-heavy-run -- node scripts/check-overview-health-refresh.mjs: four cases passed (two actions at two widths).",
|
||||
"Retry and Dismiss each accepted once; failed health read preserves unavailable warning and offers Refresh; Refresh disabled while pending; healthy read removes warning without another queue action."
|
||||
]
|
||||
}
|
||||
|
||||
@@ -103,6 +103,26 @@ describe('AlertDeliveryHealthCard', () => {
|
||||
expect(screen.getByRole('button', { name: 'Refresh delivery status' })).toBeDisabled();
|
||||
});
|
||||
|
||||
it('allows unavailable summary health to be rechecked without a queue mutation', () => {
|
||||
const onRefresh = vi.fn();
|
||||
render(() => (
|
||||
<AlertDeliveryHealthCard
|
||||
health={null}
|
||||
unavailable
|
||||
refreshing={false}
|
||||
onRefresh={onRefresh}
|
||||
detailLevel="summary"
|
||||
showRefresh
|
||||
/>
|
||||
));
|
||||
expect(screen.getByRole('alert')).toHaveTextContent(
|
||||
'Notification delivery status is unavailable',
|
||||
);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Refresh delivery status' }));
|
||||
expect(onRefresh).toHaveBeenCalledOnce();
|
||||
expect(screen.getByRole('alert')).toBeTruthy();
|
||||
});
|
||||
|
||||
it('keeps the overview treatment concise and points directly to delivery evidence', () => {
|
||||
render(() => (
|
||||
<Router>
|
||||
|
||||
@@ -90,7 +90,7 @@ export function OverviewTab(props: {
|
||||
onDismissFailures={() => void deliveryHealthState.dismissTerminalFailures()}
|
||||
detailsHref="/alerts/notifications#notification-delivery-activity"
|
||||
detailLevel="summary"
|
||||
showRefresh={false}
|
||||
showRefresh={deliveryHealthState.deliveryHealthUnavailable()}
|
||||
/>
|
||||
</Show>
|
||||
<AlertOverviewStatsCards state={overviewState} />
|
||||
|
||||
+114
-1
@@ -1,5 +1,5 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { cleanup, render, screen, waitFor } from '@solidjs/testing-library';
|
||||
import { cleanup, fireEvent, render, screen, waitFor } from '@solidjs/testing-library';
|
||||
import { DEFAULT_LOCALE, setActiveLocale } from '@/i18n';
|
||||
import type { Alert } from '@/types/api';
|
||||
import type { NotificationHealth } from '@/api/notifications';
|
||||
@@ -49,6 +49,8 @@ vi.mock('@/components/Alerts/InvestigateAlertButton', () => ({
|
||||
InvestigateAlertButton: () => null,
|
||||
}));
|
||||
|
||||
import { notificationStore } from '@/stores/notifications';
|
||||
|
||||
import { OverviewTab } from '../OverviewTab';
|
||||
|
||||
function degradedHealth(): NotificationHealth {
|
||||
@@ -107,10 +109,15 @@ describe('OverviewTab delivery health actions', () => {
|
||||
getDeliveryDiagnoses.mockReset();
|
||||
getDeliveryDiagnoses.mockResolvedValue([]);
|
||||
getHealth.mockReset();
|
||||
retryTerminalFailures.mockReset();
|
||||
dismissTerminalFailures.mockReset();
|
||||
vi.mocked(notificationStore.success).mockClear();
|
||||
vi.mocked(notificationStore.error).mockClear();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
cleanup();
|
||||
vi.restoreAllMocks();
|
||||
setActiveLocale(DEFAULT_LOCALE);
|
||||
});
|
||||
|
||||
@@ -127,4 +134,110 @@ describe('OverviewTab delivery health actions', () => {
|
||||
expect(screen.queryByRole('button', { name: 'Refresh delivery status' })).toBeNull();
|
||||
expect(screen.getByRole('alert')).toHaveTextContent('Most recent failures: connectivity (1).');
|
||||
});
|
||||
for (const action of [
|
||||
{ name: 'Retry retained deliveries', api: retryTerminalFailures },
|
||||
{ name: 'Dismiss retained failures', api: dismissTerminalFailures },
|
||||
]) {
|
||||
it(`does not mutate or refresh when ${action.name} is cancelled`, async () => {
|
||||
getHealth.mockResolvedValue(degradedHealth());
|
||||
const confirmation = vi.spyOn(window, 'confirm').mockReturnValue(false);
|
||||
render(() => <OverviewTab {...defaultProps()} />);
|
||||
fireEvent.click(await screen.findByRole('button', { name: action.name }));
|
||||
|
||||
expect(confirmation).toHaveBeenCalledOnce();
|
||||
expect(action.api).not.toHaveBeenCalled();
|
||||
expect(getHealth).toHaveBeenCalledTimes(1);
|
||||
expect(screen.getByRole('alert')).toBeTruthy();
|
||||
});
|
||||
|
||||
it(`retains attention and enables another attempt when ${action.name} fails`, async () => {
|
||||
getHealth.mockResolvedValue(degradedHealth());
|
||||
vi.spyOn(window, 'confirm').mockReturnValue(true);
|
||||
action.api.mockRejectedValue(new Error('queue action unavailable'));
|
||||
render(() => <OverviewTab {...defaultProps()} />);
|
||||
fireEvent.click(await screen.findByRole('button', { name: action.name }));
|
||||
|
||||
await waitFor(() => expect(notificationStore.error).toHaveBeenCalledOnce());
|
||||
expect(notificationStore.success).not.toHaveBeenCalled();
|
||||
expect(getHealth).toHaveBeenCalledTimes(1);
|
||||
expect(screen.getByRole('alert')).toHaveTextContent('connectivity (1)');
|
||||
expect(screen.getByRole('button', { name: action.name })).not.toBeDisabled();
|
||||
});
|
||||
|
||||
it(`keeps health visibly unknown after ${action.name} succeeds but refresh fails`, async () => {
|
||||
getHealth
|
||||
.mockResolvedValueOnce(degradedHealth())
|
||||
.mockRejectedValueOnce(new Error('health unavailable'));
|
||||
action.api.mockResolvedValue({ affected: 85 });
|
||||
vi.spyOn(window, 'confirm').mockReturnValue(true);
|
||||
render(() => <OverviewTab {...defaultProps()} />);
|
||||
fireEvent.click(await screen.findByRole('button', { name: action.name }));
|
||||
|
||||
const refresh = await screen.findByRole('button', { name: 'Refresh delivery status' });
|
||||
expect(screen.getByRole('alert')).toBeTruthy();
|
||||
expect(notificationStore.success).toHaveBeenCalledOnce();
|
||||
expect(notificationStore.error).not.toHaveBeenCalled();
|
||||
|
||||
const healthy = degradedHealth();
|
||||
healthy.overallHealthy = true;
|
||||
healthy.queue = {
|
||||
...healthy.queue,
|
||||
status: 'healthy',
|
||||
healthy: true,
|
||||
attentionRequired: 0,
|
||||
deadLetter: 0,
|
||||
};
|
||||
getHealth.mockResolvedValueOnce(healthy);
|
||||
await waitFor(() => expect(refresh).not.toBeDisabled());
|
||||
fireEvent.click(refresh);
|
||||
await waitFor(() => expect(screen.queryByRole('alert')).toBeNull());
|
||||
expect(getHealth).toHaveBeenCalledTimes(3);
|
||||
expect(action.api).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
it(`keeps both actions disabled until ${action.name} and its health refresh finish`, async () => {
|
||||
const healthy = degradedHealth();
|
||||
healthy.overallHealthy = true;
|
||||
healthy.queue = {
|
||||
...healthy.queue,
|
||||
status: 'healthy',
|
||||
healthy: true,
|
||||
attentionRequired: 0,
|
||||
deadLetter: 0,
|
||||
};
|
||||
let completeAction!: (value: { affected: number }) => void;
|
||||
let completeHealth!: (value: NotificationHealth) => void;
|
||||
action.api.mockReturnValue(
|
||||
new Promise((resolve) => {
|
||||
completeAction = resolve;
|
||||
}),
|
||||
);
|
||||
getHealth.mockResolvedValueOnce(degradedHealth()).mockReturnValueOnce(
|
||||
new Promise((resolve) => {
|
||||
completeHealth = resolve;
|
||||
}),
|
||||
);
|
||||
vi.spyOn(window, 'confirm').mockReturnValue(true);
|
||||
render(() => <OverviewTab {...defaultProps()} />);
|
||||
fireEvent.click(await screen.findByRole('button', { name: action.name }));
|
||||
|
||||
const expectActionsDisabled = () => {
|
||||
const buttons = screen.getByRole('alert').querySelectorAll('button');
|
||||
expect(buttons.length).toBe(2);
|
||||
for (const button of buttons) expect(button).toBeDisabled();
|
||||
};
|
||||
expectActionsDisabled();
|
||||
expect(getHealth).toHaveBeenCalledTimes(1);
|
||||
completeAction({ affected: 85 });
|
||||
await waitFor(() => expect(getHealth).toHaveBeenCalledTimes(2));
|
||||
expectActionsDisabled();
|
||||
expect(screen.getByRole('alert')).toBeTruthy();
|
||||
|
||||
completeHealth(healthy);
|
||||
await waitFor(() => expect(screen.queryByRole('alert')).toBeNull());
|
||||
expect(action.api).toHaveBeenCalledOnce();
|
||||
expect(notificationStore.success).toHaveBeenCalledOnce();
|
||||
expect(notificationStore.error).not.toHaveBeenCalled();
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
@@ -4,7 +4,9 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/rcourtman/pulse-go-rewrite/internal/alerts/eventlog"
|
||||
alertspecs "github.com/rcourtman/pulse-go-rewrite/internal/alerts/specs"
|
||||
"github.com/rcourtman/pulse-go-rewrite/internal/models"
|
||||
"github.com/rcourtman/pulse-go-rewrite/internal/storagehealth"
|
||||
"github.com/rcourtman/pulse-go-rewrite/internal/unifiedresources"
|
||||
)
|
||||
@@ -165,3 +167,115 @@ func TestStatefulAlertReFireCooldown(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// A UI-keyed datastore override must govern actual incidents, not just the
|
||||
// threshold resolver. Exercise hysteresis and recurrence across SQLite reopen,
|
||||
// including a same-named datastore on another PBS instance.
|
||||
func TestPBSDatastoreOverrideLifecycleAcrossRestart(t *testing.T) {
|
||||
for _, legacySnapshot := range []bool{false, true} {
|
||||
name := "persisted aliases"
|
||||
if legacySnapshot {
|
||||
name = "legacy snapshot"
|
||||
}
|
||||
t.Run(name, func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
start := func() *Manager {
|
||||
m := NewManagerWithDataDir(dir)
|
||||
t.Cleanup(m.Stop)
|
||||
m.EnableEventLog()
|
||||
if !m.activeStateAuthoritative.Load() {
|
||||
t.Fatal("SQLite active state is not authoritative")
|
||||
}
|
||||
m.UpdateConfig(AlertConfig{Enabled: true, ActivationState: ActivationActive,
|
||||
StorageDefault: HysteresisThreshold{Trigger: 95, Clear: 90},
|
||||
Overrides: map[string]ThresholdConfig{"pbs-primary/backups": {Usage: &HysteresisThreshold{Trigger: 80, Clear: 70}}},
|
||||
})
|
||||
disableTestTimeThresholds(m)
|
||||
return m
|
||||
}
|
||||
storage := func(instance string, usage float64) models.Storage {
|
||||
return models.Storage{ID: instance + "-backups", AliasIDs: []string{instance + "/backups"}, Name: "backups", Instance: instance, Type: "pbs", Status: "online", Total: 1000, Used: int64(usage * 10), Free: int64(1000 - usage*10), Usage: usage}
|
||||
}
|
||||
observe := func(m *Manager, usage float64) {
|
||||
t.Helper()
|
||||
for range 5 {
|
||||
m.CheckStorage(storage("pbs-primary", usage))
|
||||
m.CheckStorage(storage("pbs-secondary", usage))
|
||||
}
|
||||
if testHasActiveAlert(t, m, canonicalMetricStateID("pbs-secondary-backups", "usage")) {
|
||||
t.Fatal("override leaked to another PBS instance")
|
||||
}
|
||||
}
|
||||
events := func(m *Manager, fired, resolved int) {
|
||||
t.Helper()
|
||||
for kind, want := range map[string]int{eventlog.TypeFired: fired, eventlog.TypeResolved: resolved} {
|
||||
if got := len(queryAlertEvents(t, m, eventlog.Filter{Types: []string{kind}})); got != want {
|
||||
t.Fatalf("%s events = %d, want %d", kind, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
id := canonicalMetricStateID("pbs-primary-backups", "usage")
|
||||
m := start()
|
||||
observe(m, 85)
|
||||
original := *testRequireActiveAlert(t, m, id)
|
||||
events(m, 1, 0)
|
||||
if legacySnapshot {
|
||||
m.mu.Lock()
|
||||
delete(m.activeAlerts[id].Metadata, storagePolicyAliasesKey)
|
||||
m.mu.Unlock()
|
||||
if err := m.SaveActiveAlerts(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
m.Stop()
|
||||
|
||||
m = start()
|
||||
observe(m, 75) // Below trigger, but not below the override's clear threshold.
|
||||
if got := testRequireActiveAlert(t, m, id); !got.StartTime.Equal(original.StartTime) {
|
||||
t.Fatal("restart replaced the firing incident")
|
||||
}
|
||||
events(m, 1, 0)
|
||||
observe(m, 65)
|
||||
if testHasActiveAlert(t, m, id) {
|
||||
t.Fatal("override recovery did not clear incident")
|
||||
}
|
||||
events(m, 1, 1)
|
||||
m.Stop()
|
||||
|
||||
m = start()
|
||||
observe(m, 75)
|
||||
if testHasActiveAlert(t, m, id) {
|
||||
t.Fatal("resolved incident resurrected inside hysteresis band")
|
||||
}
|
||||
events(m, 1, 1)
|
||||
observe(m, 85)
|
||||
if got := testRequireActiveAlert(t, m, id); !got.StartTime.After(original.StartTime) {
|
||||
t.Fatal("refire reused original incident start")
|
||||
}
|
||||
events(m, 2, 1)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestStoragePolicyAliasesLegacyIdentity(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name, instance, resource, datastore string
|
||||
want bool
|
||||
}{
|
||||
{"hyphenated names", "pbs-backup-east", "pbs-backup-east-daily-store", "daily-store", true},
|
||||
{"different instance", "pbs-backup-west", "pbs-backup-east-daily-store", "daily-store", false},
|
||||
{"not PBS", "backup-east", "backup-east-daily-store", "daily-store", false},
|
||||
{"missing datastore", "pbs-backup-east", "pbs-backup-east-", "", false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
got := storagePolicyAliases(&Alert{Instance: tc.instance, ResourceID: tc.resource, ResourceName: tc.datastore})
|
||||
if tc.want {
|
||||
if len(got) != 1 || got[0] != tc.instance+"/"+tc.datastore {
|
||||
t.Fatalf("aliases = %v", got)
|
||||
}
|
||||
} else if len(got) != 0 {
|
||||
t.Fatalf("invented alias: %v", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -215,12 +215,13 @@ var capacityForecastMetadataKeys = []string{
|
||||
|
||||
func (m *Manager) evaluateStorageCapacity(storage models.Storage, thresholds ThresholdConfig, trend CapacityTrendObservation) {
|
||||
input := &UnifiedResourceInput{
|
||||
ID: storage.ID,
|
||||
Type: "storage",
|
||||
Name: storage.Name,
|
||||
Node: storage.Node,
|
||||
Instance: storage.Instance,
|
||||
Disk: &UnifiedResourceMetric{Percent: storage.Usage},
|
||||
ID: storage.ID,
|
||||
StorageAliases: storage.AliasIDs,
|
||||
Type: "storage",
|
||||
Name: storage.Name,
|
||||
Node: storage.Node,
|
||||
Instance: storage.Instance,
|
||||
Disk: &UnifiedResourceMetric{Percent: storage.Usage},
|
||||
}
|
||||
m.evaluateUnifiedCapacity(input, thresholds, trend, func() bool {
|
||||
if !m.config.Enabled {
|
||||
@@ -377,6 +378,9 @@ func (m *Manager) evaluateCapacityForecast(input *UnifiedResourceInput, threshol
|
||||
"forecastBucketCount": trend.BucketCount,
|
||||
"forecastCoverageSeconds": int64(trend.CoverageSpan / time.Second),
|
||||
}
|
||||
if len(input.StorageAliases) > 0 {
|
||||
metadata[storagePolicyAliasesKey] = append([]string(nil), input.StorageAliases...)
|
||||
}
|
||||
_, _ = m.evaluateCanonicalLifecycleAlert(canonicalLifecycleAlertParams{
|
||||
Spec: spec,
|
||||
Evidence: alertspecs.AlertEvidence{
|
||||
|
||||
@@ -503,7 +503,7 @@ func (m *Manager) reevaluateActiveAlertsLocked() {
|
||||
alertsToResolve = append(alertsToResolve, alertID)
|
||||
continue
|
||||
}
|
||||
thresholds := m.resolveResourceThresholds("storage", resourceID)
|
||||
thresholds := m.effectiveAlertPolicyNoLock(alertPolicyQuery{TypeKey: "storage", ResourceID: resourceID, StorageAliases: storagePolicyAliases(alert)}).Thresholds
|
||||
if thresholds.Disabled {
|
||||
alertsToResolve = append(alertsToResolve, alertID)
|
||||
continue
|
||||
|
||||
@@ -67,3 +67,32 @@ func (m *Manager) resolveStorageThresholdOverride(base ThresholdConfig, resource
|
||||
func (m *Manager) resolveStorageThresholdsNoLock(storage models.Storage) ThresholdConfig {
|
||||
return m.resolveStorageThresholdOverride(m.defaultThresholdsForResourceType("storage"), storage.ID, storage.AliasIDs)
|
||||
}
|
||||
|
||||
// Keep the polling identity through JSON/SQLite restore so configuration saves
|
||||
// cannot silently replace a datastore override with global storage defaults.
|
||||
const storagePolicyAliasesKey = "storagePolicyAliases"
|
||||
|
||||
func storagePolicyAliases(alert *Alert) []string {
|
||||
if alert == nil {
|
||||
return nil
|
||||
}
|
||||
switch aliases := alert.Metadata[storagePolicyAliasesKey].(type) {
|
||||
case []string:
|
||||
return append([]string(nil), aliases...)
|
||||
case []interface{}:
|
||||
result := make([]string, 0, len(aliases))
|
||||
for _, value := range aliases {
|
||||
if alias, ok := value.(string); ok {
|
||||
result = append(result, alias)
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
// Older PBS snapshots predate alias metadata. The poller records the
|
||||
// instance and datastore name separately: only reconstruct an alias when
|
||||
// the complete legacy ID agrees, never split hyphenated names heuristically.
|
||||
if strings.HasPrefix(alert.Instance, "pbs-") && alert.ResourceName != "" && alert.ResourceID == alert.Instance+"-"+alert.ResourceName {
|
||||
return []string{alert.Instance + "/" + alert.ResourceName}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -55,6 +55,8 @@ type UnifiedResourceInput struct {
|
||||
NetworkIn *UnifiedResourceMetric
|
||||
NetworkOut *UnifiedResourceMetric
|
||||
Temperature *UnifiedResourceMetric
|
||||
|
||||
StorageAliases []string // Durable policy lookup identities for storage alerts.
|
||||
}
|
||||
|
||||
type unifiedMetricCandidate struct {
|
||||
@@ -259,6 +261,19 @@ func (m *Manager) evaluateUnifiedMetrics(input *UnifiedResourceInput, thresholds
|
||||
}
|
||||
|
||||
opts = metricOptionsWithTags(opts, input.Tags)
|
||||
if len(input.StorageAliases) > 0 {
|
||||
merged := metricOptions{}
|
||||
if opts != nil {
|
||||
merged = *opts
|
||||
}
|
||||
metadata := make(map[string]interface{}, len(merged.Metadata)+1)
|
||||
for k, v := range merged.Metadata {
|
||||
metadata[k] = v
|
||||
}
|
||||
metadata[storagePolicyAliasesKey] = append([]string(nil), input.StorageAliases...)
|
||||
merged.Metadata = metadata
|
||||
opts = &merged
|
||||
}
|
||||
for _, candidate := range buildUnifiedMetricCandidates(input, thresholds) {
|
||||
m.checkMetricWithCanonicalSpec(candidate.Spec, input.Name, input.Node, input.Instance, unifiedAlertType(input.Type), candidate.Value, candidate.Threshold, opts)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
package notifications
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/rcourtman/pulse-go-rewrite/internal/alerts"
|
||||
)
|
||||
|
||||
// Provider rejection must preserve the firing receipt needed for recovery,
|
||||
// including when an operator retries the recovery after reopening SQLite.
|
||||
func TestQueuedNtfyRecoveryAfterProviderOutageAndRestart(t *testing.T) {
|
||||
var unavailable atomic.Bool
|
||||
var accepted atomic.Int32
|
||||
server := newIPv4HTTPServer(t, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
body, err := io.ReadAll(r.Body)
|
||||
if err != nil {
|
||||
t.Errorf("read body: %v", err)
|
||||
}
|
||||
if unavailable.Load() {
|
||||
w.WriteHeader(http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
if accepted.Add(1) == 2 {
|
||||
if r.Header.Get("Priority") != "default" || r.Header.Get("Title") != "RESOLVED: database" ||
|
||||
!strings.Contains(string(body), "is now healthy") {
|
||||
t.Errorf("incorrect recovery: headers=%v body=%q", r.Header, body)
|
||||
}
|
||||
}
|
||||
w.WriteHeader(http.StatusAccepted)
|
||||
}))
|
||||
defer server.Close()
|
||||
dir := t.TempDir()
|
||||
webhook := WebhookConfig{ID: "ops", Name: "ops", URL: server.URL + "/topic", Enabled: true, Service: "ntfy"}
|
||||
open := func() *NotificationManager {
|
||||
m := NewNotificationManagerWithDataDir("", dir)
|
||||
m.webhookClient = server.Client()
|
||||
if err := m.UpdateAllowedPrivateCIDRs("127.0.0.1/32"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.AddWebhook(webhook)
|
||||
t.Cleanup(m.Stop)
|
||||
return m
|
||||
}
|
||||
m := open()
|
||||
config, err := json.Marshal(webhook)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
alert := &alerts.Alert{ID: "cpu", ResourceName: "database", Type: "cpu", Level: alerts.AlertLevelCritical,
|
||||
Message: "CPU above threshold", StartTime: time.Now().Add(-time.Minute)}
|
||||
resolved := notificationDeliveryJob{Type: "webhook", Event: eventResolved, Alerts: []*alerts.Alert{alert}, WebhookConfig: &webhook}
|
||||
wait := func(m *NotificationManager, id string, want NotificationQueueStatus, wantAudits int) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(15 * time.Second)
|
||||
for {
|
||||
var status string
|
||||
var audits int
|
||||
if err := m.queue.db.QueryRow("SELECT status FROM notification_queue WHERE id = ?", id).Scan(&status); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = ?", id).Scan(&audits); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if status == string(want) && audits == wantAudits {
|
||||
return
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("%s: status=%s audits=%d, want %s with audit", id, status, audits, want)
|
||||
}
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
}
|
||||
enqueue := func(id, kind string, payload *alerts.Alert) {
|
||||
t.Helper()
|
||||
if err := m.queue.Enqueue(&QueuedNotification{ID: id, Type: kind, Status: QueueStatusPending,
|
||||
Config: config, Alerts: []*alerts.Alert{payload}, MaxAttempts: 1}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
enqueue("firing", "webhook", alert.Clone())
|
||||
wait(m, "firing", QueueStatusSent, 1)
|
||||
unavailable.Store(true)
|
||||
recovery := alert.Clone()
|
||||
annotateResolvedMetadata(recovery, time.Now())
|
||||
enqueue("recovery", "webhook_resolved", recovery)
|
||||
wait(m, "recovery", QueueStatusDLQ, 1)
|
||||
if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 1 {
|
||||
t.Fatal("failed recovery consumed the firing receipt")
|
||||
}
|
||||
m.Stop()
|
||||
m = open()
|
||||
if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 1 {
|
||||
t.Fatal("restart lost the firing receipt")
|
||||
}
|
||||
unavailable.Store(false)
|
||||
if count, err := m.queue.RetryTerminalFailures(); err != nil || count != 1 {
|
||||
t.Fatalf("retry = %d, %v; want one recovery", count, err)
|
||||
}
|
||||
wait(m, "recovery", QueueStatusSent, 2)
|
||||
if got := m.filterResolvedJobsByDeliveryReceipt([]notificationDeliveryJob{resolved}); len(got) != 0 {
|
||||
t.Fatal("successful recovery did not consume the firing receipt")
|
||||
}
|
||||
if got := accepted.Load(); got != 2 {
|
||||
t.Fatalf("accepted HTTP requests = %d, want firing and recovery only", got)
|
||||
}
|
||||
var failures, successes int
|
||||
if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = 'recovery' AND success = 0").Scan(&failures); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := m.queue.db.QueryRow("SELECT count(*) FROM notification_audit WHERE notification_id = 'recovery' AND success = 1").Scan(&successes); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if failures != 1 || successes != 1 {
|
||||
t.Fatalf("recovery audit failures=%d successes=%d, want 1 each", failures, successes)
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package notifications
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"reflect"
|
||||
@@ -14,6 +15,16 @@ import (
|
||||
// Exercise the shared destination across transitions: generated firing headers
|
||||
// must not leak into the stored configuration and override recovery metadata.
|
||||
func TestNtfySeverityRecoveryTransition(t *testing.T) {
|
||||
for _, queued := range []bool{false, true} {
|
||||
name := "direct"
|
||||
if queued {
|
||||
name = "queued"
|
||||
}
|
||||
t.Run(name, func(t *testing.T) { testNtfySeverityRecoveryTransition(t, queued) })
|
||||
}
|
||||
}
|
||||
|
||||
func testNtfySeverityRecoveryTransition(t *testing.T, queued bool) {
|
||||
type receipt struct {
|
||||
header http.Header
|
||||
body string
|
||||
@@ -28,7 +39,7 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
|
||||
w.WriteHeader(http.StatusAccepted)
|
||||
}))
|
||||
defer server.Close()
|
||||
manager := NewNotificationManager("")
|
||||
manager := NewNotificationManagerWithDataDir("", t.TempDir())
|
||||
defer manager.Stop()
|
||||
manager.webhookClient = server.Client()
|
||||
if err := manager.UpdateAllowedPrivateCIDRs("127.0.0.1/32"); err != nil {
|
||||
@@ -36,6 +47,7 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
|
||||
}
|
||||
webhook := WebhookConfig{Name: "transition", URL: server.URL + "/topic", Enabled: true, Service: "ntfy",
|
||||
Headers: map[string]string{"X-Static": "preserved"}}
|
||||
manager.AddWebhook(webhook)
|
||||
originalHeaders := map[string]string{"X-Static": "preserved"}
|
||||
alert := &alerts.Alert{ID: "transition", Type: "cpu", ResourceID: "vm-1", ResourceName: "database", Node: "node-a",
|
||||
Message: "CPU above threshold", Value: 99, Threshold: 90, StartTime: time.Now().Add(-time.Minute)}
|
||||
@@ -54,7 +66,25 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
|
||||
alert.Level = step.level
|
||||
before := *alert
|
||||
var err error
|
||||
if step.resolved {
|
||||
if queued {
|
||||
config, marshalErr := json.Marshal(webhook)
|
||||
if marshalErr != nil {
|
||||
t.Fatal(marshalErr)
|
||||
}
|
||||
kind := "webhook"
|
||||
payload := alert.Clone()
|
||||
if step.resolved {
|
||||
kind += "_resolved"
|
||||
annotateResolvedMetadata(payload, time.Now())
|
||||
}
|
||||
if manager.queue == nil {
|
||||
t.Fatal("notification queue unavailable")
|
||||
}
|
||||
err = manager.queue.Enqueue(&QueuedNotification{
|
||||
ID: step.name, Type: kind, Status: QueueStatusPending,
|
||||
Config: config, Alerts: []*alerts.Alert{payload}, MaxAttempts: 1,
|
||||
})
|
||||
} else if step.resolved {
|
||||
err = manager.sendResolvedWebhook(webhook, []*alerts.Alert{alert}, time.Now())
|
||||
} else {
|
||||
err = manager.sendGroupedWebhook(webhook, []*alerts.Alert{alert})
|
||||
@@ -86,6 +116,24 @@ func TestNtfySeverityRecoveryTransition(t *testing.T) {
|
||||
if !reflect.DeepEqual(*alert, before) {
|
||||
t.Error("source alert mutated")
|
||||
}
|
||||
if queued {
|
||||
// Receipt precedes the queue commit. Wait for completion before
|
||||
// advancing the same alert identity to its next lifecycle state.
|
||||
deadline := time.Now().Add(3 * time.Second)
|
||||
for {
|
||||
var status string
|
||||
if err := manager.queue.db.QueryRow("SELECT status FROM notification_queue WHERE id = ?", step.name).Scan(&status); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if status == string(QueueStatusSent) {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("received HTTP request but queue status remained %s", status)
|
||||
}
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
// Isolated real-browser component qualification; no installed backend or delivery claim.
|
||||
import { createServer } from "../frontend-modern/node_modules/vite/dist/node/index.js";
|
||||
import solid from "../frontend-modern/node_modules/vite-plugin-solid/dist/esm/index.mjs";
|
||||
import { chromium } from "@playwright/test";
|
||||
import { resolve } from "node:path";
|
||||
import { mkdirSync } from "node:fs";
|
||||
import assert from "node:assert/strict";
|
||||
const root = resolve("frontend-modern");
|
||||
process.chdir(root);
|
||||
const fixture = `
|
||||
import { render } from 'solid-js/web';
|
||||
import { Router, Route } from '@solidjs/router';
|
||||
import { NotificationsAPI } from '/src/api/notifications';
|
||||
import { AlertsAPI } from '/src/api/alerts';
|
||||
import { OverviewTab } from '/src/features/alerts/OverviewTab';
|
||||
import '/src/index.css';
|
||||
const pending = [];
|
||||
NotificationsAPI.getHealth = () => new Promise((resolve, reject) => pending.push({resolve, reject}));
|
||||
AlertsAPI.getDeliveryDiagnoses = async () => [];
|
||||
window.actions = 0;
|
||||
NotificationsAPI.retryTerminalFailures = NotificationsAPI.dismissTerminalFailures = async () => { window.actions++; return {affected: 1}; };
|
||||
window.confirm = () => true;
|
||||
window.finish = (i, status) => status === 'error' ? pending[i].reject(new Error('scripted offline')) : pending[i].resolve({queue:{status, failed:0, deadLetter:status === 'healthy' ? 0 : 1, attentionRequired:status === 'healthy' ? 0 : 1}});
|
||||
window.count = () => pending.length;
|
||||
function Fixture() {
|
||||
return <main class="p-4"><OverviewTab overrides={[]} activeAlerts={{}} updateAlert={()=>{}} showQuickTip={()=>false} dismissQuickTip={()=>{}} showAcknowledged={()=>true} setShowAcknowledged={()=>{}} alertsDisabled={()=>false}/></main>;
|
||||
}
|
||||
render(() => <Router><Route path="/qualification" component={Fixture}/></Router>, document.getElementById('root'));
|
||||
`;
|
||||
const server = await createServer({
|
||||
root,
|
||||
configFile: false,
|
||||
optimizeDeps: {
|
||||
noDiscovery: true,
|
||||
entries: [],
|
||||
esbuildOptions: { target: "esnext" },
|
||||
},
|
||||
esbuild: { target: "esnext" },
|
||||
plugins: [
|
||||
solid(),
|
||||
{
|
||||
name: "ordering-fixture",
|
||||
configureServer(s) {
|
||||
s.middlewares.use((req, res, next) => {
|
||||
if (req.url === "/qualification") {
|
||||
res.setHeader("Content-Type", "text/html");
|
||||
res.end(
|
||||
'<div id="root"></div><script type="module" src="/ordering-fixture.tsx"></script>',
|
||||
);
|
||||
} else next();
|
||||
});
|
||||
},
|
||||
resolveId(id) {
|
||||
if (id === "/ordering-fixture.tsx") return id;
|
||||
},
|
||||
load(id) {
|
||||
if (id === "/ordering-fixture.tsx") return fixture;
|
||||
},
|
||||
},
|
||||
],
|
||||
resolve: { alias: { "@": resolve(root, "src") } },
|
||||
server: { host: "127.0.0.1", port: 5197, strictPort: true },
|
||||
});
|
||||
let browser;
|
||||
try {
|
||||
await server.listen();
|
||||
browser = await chromium.launch({ headless: true });
|
||||
mkdirSync('/tmp/pulse-overview-refresh', { recursive: true });
|
||||
let cases = 0;
|
||||
for (const width of [1440, 390]) {
|
||||
for (const action of ['Retry retained deliveries', 'Dismiss retained failures']) {
|
||||
const page = await browser.newPage({ viewport: { width, height: 900 } });
|
||||
await page.goto('http://127.0.0.1:5197/qualification');
|
||||
await page.waitForFunction(() => window.count?.() === 1);
|
||||
await page.evaluate(() => window.finish(0, 'degraded'));
|
||||
await page.getByRole('button', {name: action, exact: true}).click();
|
||||
await page.waitForFunction(() => window.count() === 2);
|
||||
await page.evaluate(() => window.finish(1, 'error'));
|
||||
const refresh = page.getByRole('button', {name:'Refresh delivery status', exact:true});
|
||||
await refresh.waitFor();
|
||||
assert.match(await page.getByRole('alert').innerText(), /status is unavailable/);
|
||||
await page.screenshot({path: `/tmp/pulse-overview-refresh/${width}-${cases}-unavailable.png`});
|
||||
await refresh.click();
|
||||
await page.waitForFunction(() => window.count() === 3);
|
||||
assert.equal(await refresh.isDisabled(), true);
|
||||
await page.evaluate(() => window.finish(2, 'healthy'));
|
||||
await page.getByRole('alert').waitFor({state:'detached'});
|
||||
assert.equal(await page.evaluate(() => window.actions), 1);
|
||||
await page.screenshot({path: `/tmp/pulse-overview-refresh/${width}-${cases}-healthy.png`});
|
||||
cases++;
|
||||
await page.close();
|
||||
}
|
||||
}
|
||||
console.log(`${cases} overview action/outage/refresh browser cases passed`);
|
||||
} finally {
|
||||
await browser?.close();
|
||||
await server.close();
|
||||
}
|
||||
Reference in New Issue
Block a user