mirror of
https://github.com/taylanbakircioglu/haproxy-openmanager.git
synced 2026-10-02 23:18:14 +00:00
Compare commits
46 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| dbb9189f16 | |||
| eee0a4716a | |||
| 3c8832330a | |||
| 81ab674072 | |||
| 6bf6d016f5 | |||
| 0a0226c758 | |||
| 8e534ef170 | |||
| 71786200dd | |||
| 33e3e8ef9d | |||
| 69e12f7459 | |||
| af07d72514 | |||
| a6166d11b9 | |||
| 882d25bb68 | |||
| 0ebf6583ea | |||
| 6be19f0bb5 | |||
| 9e5185c458 | |||
| 520b69a1c6 | |||
| 56107fa86f | |||
| f86a4331e8 | |||
| 1c47e246ec | |||
| d914f2398b | |||
| 9c1f3c811b | |||
| c79391cd13 | |||
| 9e2ea04777 | |||
| 60f4fa71ed | |||
| 97b2452bd2 | |||
| 23257b02cf | |||
| c8d144ca9d | |||
| 64d42663cd | |||
| 27fbe48c4b | |||
| e86e86a53c | |||
| 70ebc02e09 | |||
| c492b26bb1 | |||
| 8b07d7a6a3 | |||
| 428915998b | |||
| 1bc99c5fe7 | |||
| a1192e602d | |||
| 9d7a718671 | |||
| 62b1599354 | |||
| b34d7cf811 | |||
| c2ea424d70 | |||
| d9ef86f548 | |||
| f3d4fb11bb | |||
| 0692f26ebb | |||
| b0bb6a55c0 | |||
| f2b3df517f |
@@ -17,6 +17,34 @@ REDIS_URL=redis://redis:6379
|
||||
# Change this to a strong random string in production
|
||||
SECRET_KEY=your-secret-key-change-this-in-production
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Optional per-purpose encryption keys.
|
||||
#
|
||||
# Every secret the application stores is encrypted at rest with Fernet. Each class
|
||||
# derives its own key, so rotating one never affects another. If a variable below is
|
||||
# unset, that class's key is derived from SECRET_KEY via HKDF — which works, but means
|
||||
# rotating SECRET_KEY makes the existing values of that class UNDECRYPTABLE. Set an
|
||||
# explicit key (urlsafe-base64, 32 bytes) in production if you want independent
|
||||
# rotation. Generate one with:
|
||||
# python -c "from cryptography.fernet import Fernet; print(Fernet.generate_key().decode())"
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
# VRRP secrets for HA/VIP (Issue #27).
|
||||
# VIP_ENCRYPTION_KEY=
|
||||
|
||||
# TOTP secrets for multi-factor authentication (Issue #18).
|
||||
# MFA_ENCRYPTION_KEY=
|
||||
|
||||
# Per-account DNS provider credentials for ACME DNS-01 (Issue #35).
|
||||
# Rotating this without re-entering credentials makes DNS-01 renewals fail until the
|
||||
# affected accounts' credentials are re-saved in ACME Automation.
|
||||
# DNS_PROVIDER_ENCRYPTION_KEY=
|
||||
|
||||
# Private keys of PENDING CSRs, held only until the signed certificate is imported
|
||||
# (Issue #53). Rotating this while CSRs are out for signature makes those CSRs
|
||||
# unusable — they must be deleted and re-created.
|
||||
# CSR_ENCRYPTION_KEY=
|
||||
|
||||
# ============================================================================
|
||||
# PUBLIC URL CONFIGURATION
|
||||
# ============================================================================
|
||||
@@ -54,6 +82,15 @@ AGENT_HEARTBEAT_TIMEOUT_SECONDS=15
|
||||
# Config sync interval in seconds
|
||||
AGENT_CONFIG_SYNC_INTERVAL_SECONDS=30
|
||||
|
||||
# ============================================================================
|
||||
# BACKEND PERFORMANCE
|
||||
# ============================================================================
|
||||
# Number of uvicorn worker processes for the backend API (default: 1).
|
||||
# On multi-core hosts, setting this to the core count (e.g. 2) lets the API
|
||||
# use all cores. Safe to increase: background tasks are multi-replica safe
|
||||
# (the k8s deployment already runs 2+ replicas via HPA).
|
||||
UVICORN_WORKERS=1
|
||||
|
||||
# ============================================================================
|
||||
# CORS CONFIGURATION
|
||||
# ============================================================================
|
||||
|
||||
@@ -7,6 +7,8 @@ on:
|
||||
jobs:
|
||||
build_and_push:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: checkout
|
||||
@@ -19,27 +21,17 @@ jobs:
|
||||
- name: read product version
|
||||
id: prodversion
|
||||
run: |
|
||||
VERSION=$(jq -r .version version.json)
|
||||
VERSION=$(jq -r .version backend/version.json)
|
||||
if [ -z "$VERSION" ] || [ "$VERSION" = "null" ]; then
|
||||
echo "Failed to read product version from version.json" >&2
|
||||
echo "Failed to read product version from backend/version.json" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "VERSION=$VERSION" >> $GITHUB_OUTPUT
|
||||
|
||||
# The backend image is built with `context: ./backend`, so the
|
||||
# repo-root version.json is OUTSIDE the build context and never
|
||||
# reaches the container. Backend `main.py` falls back to a
|
||||
# compile-time constant when /app/version.json is missing, which
|
||||
# caused a real production drift: a redeploy of the v1.5.2 tree
|
||||
# silently still reported "v1.5.0" in `/api/version` because the
|
||||
# constant in main.py had been bumped but the file was not
|
||||
# available to read. Stage version.json into the backend
|
||||
# context here so the canonical file IS shipped and the
|
||||
# constant only serves as a defensive fallback. The staged file
|
||||
# is gitignored to keep `git status` clean for developers.
|
||||
- name: stage version.json into backend build context
|
||||
run: cp version.json backend/version.json
|
||||
|
||||
# version.json now lives at backend/version.json (inside the ./backend build
|
||||
# context), so `COPY . .` bakes it into the image directly — no staging step
|
||||
# is needed and the backend reports the correct version in every deployment,
|
||||
# not just this workflow's builds.
|
||||
- name: set up qemu
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
@@ -76,3 +68,30 @@ jobs:
|
||||
taylanbakircioglu/haproxy-openmanager-frontend:${{ steps.version.outputs.TAG }}
|
||||
taylanbakircioglu/haproxy-openmanager-frontend:${{ steps.prodversion.outputs.VERSION }}
|
||||
|
||||
# Keep the GitHub Releases/Tags in sync with version.json. The docker
|
||||
# images above are tagged with the product version, but nothing here
|
||||
# created the matching git tag, so the repo's Tags/Releases drifted
|
||||
# behind (stuck at the last manually-created tag). After the images are
|
||||
# pushed, cut a Release (which also creates the tag) for the current
|
||||
# version.json, but only if one does not already exist, so re-runs
|
||||
# without a version bump are a no-op.
|
||||
- name: create github release from version.json
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
VERSION="${{ steps.prodversion.outputs.VERSION }}"
|
||||
TAG="v${VERSION}"
|
||||
RELEASE_NAME=$(jq -r '.releaseName // empty' backend/version.json)
|
||||
if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
|
||||
echo "Release $TAG already exists, skipping."
|
||||
else
|
||||
TITLE="$TAG"
|
||||
[ -n "$RELEASE_NAME" ] && TITLE="$TAG — $RELEASE_NAME"
|
||||
gh release create "$TAG" \
|
||||
--repo "$GITHUB_REPOSITORY" \
|
||||
--target "$GITHUB_SHA" \
|
||||
--title "$TITLE" \
|
||||
--notes "Automated release for $TAG (from version.json)."
|
||||
echo "Created release $TAG"
|
||||
fi
|
||||
|
||||
|
||||
@@ -39,11 +39,6 @@ venv.bak/
|
||||
*.sqlite
|
||||
*.sqlite3
|
||||
|
||||
# Build-time staged version.json (CI `cp version.json backend/`).
|
||||
# The canonical file lives at repo root; this path is a transient
|
||||
# copy for the backend Docker build context.
|
||||
backend/version.json
|
||||
|
||||
# IDE
|
||||
.vscode/
|
||||
.idea/
|
||||
|
||||
@@ -45,6 +45,7 @@ Modern, web-based management interface for HAProxy load balancers with multi-clu
|
||||
- [ACME Auto SSL](#acme-auto-ssl---automated-certificate-management)
|
||||
- [WAF Management](#waf-management---web-application-firewall)
|
||||
- [IP Inventory](#ip-inventory---cross-cluster-ip-search--discovery)
|
||||
- [HA / VIP Management](#ha--vip-management---keepalived-virtual-ip-failover)
|
||||
- [User Management](#user-management---access-control--authentication)
|
||||
- [Settings](#settings---system-configuration)
|
||||
6. [Getting Started - First-Time Usage](#getting-started---first-time-usage)
|
||||
@@ -104,13 +105,15 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
✅ **Version Control & Rollback** - Every change versioned with one-click restore capability
|
||||
✅ **Real-Time Monitoring** - Live stats, health checks, and performance dashboards
|
||||
✅ **SSL Certificate Management** - Centralized SSL with expiration tracking
|
||||
✅ **CSR Creation** *(v1.9.0)* - Generate a private key + CSR in-app (RSA 2048/4096, ECDSA P-256/P-384, full subject + SANs), have it signed by any external CA, then import the signed certificate — the key never leaves the server
|
||||
✅ **ACME Auto SSL (Let's Encrypt)** - Automated certificate issuance, renewal, and deployment via ACME protocol
|
||||
✅ **ACME DNS-01 Challenge** *(v1.8.0)* - TXT-record validation for internal/isolated clusters (no public port 80) and wildcard certificates; pluggable DNS providers (Manual + Cloudflare + GoDaddy *(v1.10.0)*), opt-in, HTTP-01 unchanged
|
||||
✅ **ACME Certificate Diagnostic Panel** - Automated preflight that checks agent readiness, DNS resolution, port 80 reachability, and ACME challenge ACL before issuing certificates
|
||||
✅ **WAF Rules** - Web Application Firewall management and deployment
|
||||
✅ **Agent Script Versioning** - Update agents via UI (Monaco editor) with auto-upgrade
|
||||
✅ **Token-Based Agent Auth** - Secure token management with revoke/renew
|
||||
✅ **IP Inventory** - Cross-cluster IP search to identify agents, VIPs, and backend servers by IP
|
||||
✅ **Keepalived VRRP Detection** - Automatic MASTER/BACKUP state and VIP detection from keepalived
|
||||
✅ **HA / VIP (Keepalived) Management** - Create virtual IPs from the UI; the agent installs & configures Keepalived (unicast VRRP) with a HAProxy health-check so the VIP fails over automatically; live MASTER/BACKUP detection per node
|
||||
✅ **Role-Based User Management** - Admin and user roles with granular permissions and access control
|
||||
✅ **User Activity Audit Logs** - Complete audit trail of all system events
|
||||
✅ **REST API** - Full programmatic access for automation and CI/CD integration
|
||||
@@ -229,6 +232,7 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
- **Agent Token Management**: Secure token-based agent installation with token revoke and renew capabilities
|
||||
- **Agent Script Editor**: Monaco code editor for updating agent scripts via UI (not binary) - create new versions, rollback, and auto-upgrade all agents
|
||||
- **Platform-Specific Scripts**: Auto-generated installation scripts for Linux and macOS (x86_64/ARM64)
|
||||
- **HA / VIP (Keepalived) Management**: Create a virtual IP from the UI and the agents install & configure Keepalived (VRRP) with a HAProxy health-check so the VIP fails over automatically when HAProxy drops — opt-in, with live MASTER/BACKUP per node, multi-distro install, cluster-driven config path, and Apply/Reject staging
|
||||
|
||||
#### Version Control & Change Management
|
||||
- **Apply Management**: Centralized change tracking and deployment status monitoring across all agents
|
||||
@@ -252,10 +256,11 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
- **Stuck Order Detection** *(v1.4.0)*: Setup wizard surfaces orders that the CA has validated but not yet downloaded, with one-click `Complete` action and automatic 60-second retry
|
||||
- **Multi-Provider Support**: Configurable ACME directory URL supports Let's Encrypt, ZeroSSL, Google Trust Services, Buypass, and custom CAs
|
||||
- **HTTP-01 Challenge**: Built-in challenge responder with automatic HAProxy routing injection; reserved backend name `_acme_challenge_backend` is auto-managed and protected from manual edits / agent sync collisions
|
||||
- **DNS-01 Challenge** *(v1.8.0 — Issue #35)*: Validate via a DNS TXT record instead of HTTP on port 80, for **internal/isolated clusters with no public ingress** and for **wildcard** certificates (`*.example.com`). Pluggable per-account DNS provider (Manual + Cloudflare + GoDaddy *(v1.10.0)*; credentials encrypted at rest and verified on save), same PENDING → APPLIED pipeline, bounded automatic retry on propagation lag, and a DNS-01 event timeline. Opt-in via a global setting; HTTP-01 behaviour is unchanged. (See the *DNS-01 Challenge* subsection under ACME Auto SSL below.)
|
||||
- **ACME Account Management**: Register, view, and deactivate ACME accounts from the UI
|
||||
- **Staging Mode**: Test certificate issuance with Let's Encrypt staging environment before production
|
||||
- **Custom Staging Endpoint** *(v1.4.0)*: Optional `staging_url_override` setting lets you point staging mode at a private ACME test CA (e.g. Pebble) without touching the production directory URL
|
||||
- **External Account Binding (EAB)**: Support for CAs that require EAB (ZeroSSL, Google Trust Services)
|
||||
- **External Account Binding (EAB)**: Support for CAs that require EAB (ZeroSSL, Google Trust Services). Enter the EAB Key ID and HMAC Key globally in Settings, or per-account in the Register Account dialog (a per-account value overrides the global setting; leave it blank to use the global one)
|
||||
- **Structured Error Diagnostics** *(v1.4.0)*: All ACME failures (challenge, finalize, download) persist structured JSON to `letsencrypt_orders.error_detail` for clear post-mortem analysis
|
||||
- **Audit Logging** *(v1.4.0)*: Every ACME operation (request, revoke, CA-chain import, account ops) is captured in `user_activity_logs` for compliance review
|
||||
- **ACME Diagnostic Panel** *(v1.5.0 — Issue #13)*: Live pre-flight + post-failure diagnostics (DNS / port-80 / routing / account / agents) and merged event timeline (`acme_order_events` + correlated `user_activity_logs`) accessible from the ACME Automation page; humanized error rendering for 11+ RFC8555 problem types with backwards-compatible fallback for legacy plain-string `error_detail`; per-user 5/min rate-limit
|
||||
@@ -774,6 +779,15 @@ User Updates SSL in UI → All Agents Poll Backend (30s)
|
||||
→ Validate Config → Reload HAProxy (zero downtime)
|
||||
```
|
||||
|
||||
#### CSR Workflow (external / corporate CAs) — v1.9.0
|
||||
|
||||
For certificates signed by an external or corporate CA, the **CSR tab** on the SSL Certificates page covers the whole flow without the private key ever leaving the server:
|
||||
|
||||
1. **Create CSR**: pick a name (becomes the certificate name / on-agent file path), Common Name, optional SANs and subject fields (O/OU/L/ST/C/email), and a key algorithm (RSA 2048/4096 or ECDSA P-256/P-384). The backend generates the key + CSR; only the CSR PEM is shown (copy or download as `.csr`).
|
||||
2. **Get it signed**: submit the CSR to your Certificate Authority.
|
||||
3. **Import**: paste the signed certificate (+ optional chain), choose Global or cluster-specific scope and usage type. The backend verifies the certificate matches the stored key, rejects expired certs, warns on SAN drift, and creates a normal SSL certificate entry (source: `CSR`).
|
||||
4. **Deploy**: the imported certificate goes through the standard **PENDING → Apply Management → agent pull** pipeline like any other certificate.
|
||||
|
||||
#### Key Features
|
||||
- **Certificate Upload**: PEM format certificate and private key upload
|
||||
- **ACME Automation**: Automatic certificate issuance and renewal via Let's Encrypt / ACME protocol (see [ACME Auto SSL](#acme-auto-ssl---automated-certificate-management))
|
||||
@@ -964,6 +978,24 @@ Understanding how HTTP-01 challenges work in a distributed HAProxy environment i
|
||||
| Behind NAT/VIP | VIP: 1.2.3.4:80 | Internal:5000 | VIP address |
|
||||
| Multi-cluster | Multiple HAProxy nodes | Central OpenManager | Each domain → respective HAProxy |
|
||||
|
||||
#### DNS-01 Challenge — Internal/Isolated Clusters & Wildcards *(v1.8.0 — Issue #35)*
|
||||
|
||||
The default **HTTP-01** challenge validates over **port 80**, so the domain must resolve publicly to an HAProxy node with ACME Challenge Routing enabled. **DNS-01** validates via a **DNS TXT record** (`_acme-challenge.<domain>`) instead, so it needs **no inbound port 80 and no public ingress** to the HAProxy node. Use it for:
|
||||
|
||||
- **Internal / isolated clusters** (behind a VPN/firewall, no public port 80) where you still control the domain's DNS.
|
||||
- **Wildcard certificates** (`*.example.com`) — which can *only* be issued via DNS-01.
|
||||
|
||||
DNS-01 is **opt-in** and fully backward compatible: it is disabled until an administrator enables it, and existing HTTP-01 certificates are completely unaffected.
|
||||
|
||||
- **Enable it**: Settings → ACME / SSL Automation → **DNS-01 Challenge (advanced)** → turn on *Enable DNS-01 Challenge* and Save. While off, DNS-01 options are hidden and no DNS-01 orders can be created.
|
||||
- **Per-account provider**: in ACME Automation, create (or reconfigure) an ACME account with **Challenge Method = DNS-01** and a **DNS Provider**. Provider credentials are **verified before saving** and **encrypted at rest** (Fernet, mirroring the VRRP/MFA secret pattern); they are never returned by the API or written to logs.
|
||||
- **Supported providers**: **Manual** (publish the TXT record yourself in any DNS — including fully internal DNS — then click *Verify*; works everywhere but cannot auto-renew unattended), **Cloudflare** (API token with `Zone:DNS:Edit` + `Zone:Read`; the TXT record is created and cleaned up automatically and renews unattended), and **GoDaddy** *(v1.10.0)* (a **Production** API Key + Secret pair from `developer.godaddy.com/keys` — the first key that dashboard issues is an OTE/test key and is rejected; the zone must be in the same GoDaddy account, which needs at least one registered domain before GoDaddy allows DNS API access at all. A **Personal Access Token** works too: paste it as the API Key and leave the Secret blank — that is the forward path as GoDaddy retires the `sso-key` scheme. TXT records are created and cleaned up automatically and renew unattended). The provider interface is pluggable — more providers can be added without changing the issuance flow.
|
||||
- **Same pipeline**: after validation the certificate follows the normal PENDING → APPLIED flow (assign to clusters / Apply Management) and the agent serves it — identical to HTTP-01 from finalize onward, with **zero agent or rendered-config changes** for DNS-01.
|
||||
- **Manual flow**: the order detail shows the exact `_acme-challenge.<domain>` record name + TXT value (copyable); publish it and click *I've added the records — Verify*. For Cloudflare and GoDaddy it is automatic.
|
||||
- **Resilience**: a propagation-lag failure is recovered by a **bounded fresh-order retry chain** (1 original + 3 retries with increasing backoff, kept under Let's Encrypt's rate limits); any orphaned TXT record is cleaned up by a reconcile sweep. The order detail shows a DNS-01 event timeline (publish → validation → cleanup).
|
||||
- **Wildcards**: `*.example.com` is validated at `_acme-challenge.example.com`; it does **not** cover the apex — add `example.com` as a separate name if you need both (the providers handle the two coexisting TXT values automatically).
|
||||
- **Scope (this release)**: the Site Wizard remains HTTP-01-only; issue DNS-01 / wildcard certificates from **ACME Automation**.
|
||||
|
||||
#### ACME Quick Start Guide
|
||||
|
||||
Follow these steps to obtain your first Let's Encrypt certificate:
|
||||
@@ -1055,6 +1087,32 @@ The IP Inventory page provides a unified view of all IP addresses across every c
|
||||
- Locate a backend server IP across multiple clusters
|
||||
- Audit all IP addresses managed by the platform
|
||||
|
||||
### HA / VIP Management - Keepalived Virtual IP Failover
|
||||
|
||||
The HA / VIP page manages **highly-available virtual IPs** backed by Keepalived (VRRP) directly from the UI — no SSHing into nodes to install/configure Keepalived by hand. It builds on the agent pull-architecture: you define the VIP centrally, click Apply, and the agents converge.
|
||||
|
||||
**What it does:**
|
||||
- **VIP dashboard**: lists each virtual IP with its pool, VRID, per-node interface, and a **live MASTER / BACKUP / FAULT** column per node (refreshed every 30s from the existing keepalive-state heartbeat pipeline).
|
||||
- **Creation form**: enter the virtual IP (+ prefix), pick the pool, and select which of the pool's HAProxy nodes participate — each with a network interface (from the node's reported interfaces), a VRRP role (MASTER/BACKUP) and a priority. VRID auto-allocates per pool; an optional VRRP secret can be set.
|
||||
- **Hands-off automation**: on **Apply**, each member's `keepalived.conf` is rendered and delivered; the agent installs Keepalived if missing, writes the config + a HAProxy health-check (`vrrp_script` + `track_script`), validates with `keepalived -t`, and starts the service. When HAProxy drops on the active node, the health-check lowers VRRP priority and the **VIP fails over automatically** to a backup.
|
||||
|
||||
**How it works (pull-based, isolated):**
|
||||
- New tables `vip_instances` + `vip_members`; the agent polls `GET /api/agents/{name}/keepalived-config` (key-authenticated, agent-bound) and converges. It is fully **isolated** from the global HAProxy apply flow — VIP changes never regenerate `haproxy.cfg`.
|
||||
- **Cluster-driven path**: the `keepalived.conf` location is a cluster setting (`keepalived_config_path`, default `/etc/keepalived/keepalived.conf`) delivered to the agent dynamically — just like the HAProxy paths — so non-standard installs are supported with zero per-node effort. The default is what the keepalived service loads on every distro; if you set a non-default path, ensure the keepalived unit is configured to load it.
|
||||
|
||||
**Apply / Reject staging:**
|
||||
- VIP edits are staged as **PENDING** and only go live on **Apply**, which records an applied snapshot. **Reject** discards pending changes and **restores the previous applied state** (a never-applied VIP is discarded). Editing a live VIP never disrupts it until you re-Apply.
|
||||
|
||||
**Enterprise safety:**
|
||||
- **Opt-in & backward compatible** — nodes/clusters without a VIP do nothing new.
|
||||
- **Never clobbers a hand-managed Keepalived** — if an unmanaged `keepalived.conf` exists, the agent reports "externally managed" and leaves it untouched.
|
||||
- **Multi-distro install** — Debian/Ubuntu, RHEL/CentOS/Alma/Rocky, Fedora, SUSE/openSUSE, Alpine (apt/dnf/yum/zypper/apk).
|
||||
- **Reliable detection across platforms** — MASTER/BACKUP is detected via systemd journal, syslog, and a portable interface-based check (exact-match, all VIPs).
|
||||
- **Secure** — the VRRP secret is encrypted at rest and never returned by the API; the delivery endpoint requires the node's own API key.
|
||||
- **Scope** — VRRP targets bare-metal / VMware / on-prem L2 networks; the UI clearly notes that AWS/Azure/GCP don't honor VRRP/gratuitous-ARP. macOS agents can't run Keepalived and are excluded (the UI warns if one is selected).
|
||||
|
||||
**RBAC:** gated by the `vip` permission group (`vip.read/create/update/delete/apply`); admins bypass.
|
||||
|
||||
### User Management - Access Control & Authentication
|
||||
- **User Accounts**: Create, edit, and manage user accounts
|
||||
- **Role-based Access**: Admin, user, and custom role definitions
|
||||
@@ -1690,6 +1748,19 @@ Add new HAProxy clusters through the web interface or directly via API:
|
||||
}
|
||||
```
|
||||
|
||||
### Performance Tuning *(v1.8.6)*
|
||||
|
||||
The backend API defaults to a **single uvicorn worker process**, which uses one CPU core. Agents poll the API every 30 seconds (heartbeat, config, pending-requests, upgrade checks), so larger fleets add a constant baseline load. Two ways to scale:
|
||||
|
||||
- **Docker Compose — worker processes**: set `UVICORN_WORKERS` in your `.env` (default `1`). On a multi-core host, matching the core count (e.g. `UVICORN_WORKERS=2` on a 2-core machine) lets the API use all cores:
|
||||
```bash
|
||||
echo "UVICORN_WORKERS=2" >> .env && docker-compose up -d backend
|
||||
```
|
||||
- **Kubernetes/OpenShift — replicas**: the shipped manifests already include an HPA for the backend (2→10 replicas, `k8s/manifests/13-hpa.yaml`); raise `minReplicas`/`maxReplicas` as needed.
|
||||
|
||||
Both are safe: all background tasks (ACME completion, renewals, agent monitoring) are multi-replica safe by design (atomic claims via `FOR UPDATE SKIP LOCKED`, PostgreSQL advisory locks).
|
||||
|
||||
**Diagnosing slow requests**: every API response carries an `X-Response-Time` header, and the backend logs `Slow request detected` (WARNING) for any request taking longer than 1 second — check those log lines to pinpoint slow endpoints before tuning anything else.
|
||||
|
||||
## API Reference
|
||||
|
||||
@@ -1795,6 +1866,42 @@ GET /api/backends?cluster_id=1
|
||||
GET /api/frontends?cluster_id=1
|
||||
```
|
||||
|
||||
### SSL CSR API (v1.9.0)
|
||||
```bash
|
||||
# Create a CSR (generates the private key server-side; response contains the
|
||||
# CSR PEM — the private key is never returned by any endpoint)
|
||||
POST /api/ssl/csrs
|
||||
Authorization: Bearer <token>
|
||||
{
|
||||
"name": "www-example-com",
|
||||
"common_name": "www.example.com",
|
||||
"sans": ["api.example.com"],
|
||||
"key_algorithm": "rsa-2048", # rsa-2048 | rsa-4096 | ecdsa-p256 | ecdsa-p384
|
||||
"organization": "Example Corp",
|
||||
"country": "TR"
|
||||
}
|
||||
|
||||
# List CSRs (metadata only, no PEM)
|
||||
GET /api/ssl/csrs
|
||||
|
||||
# CSR detail (includes the CSR PEM)
|
||||
GET /api/ssl/csrs/{csr_id}
|
||||
|
||||
# Import the CA-signed certificate for a pending CSR
|
||||
POST /api/ssl/csrs/{csr_id}/import
|
||||
{
|
||||
"certificate_content": "-----BEGIN CERTIFICATE-----...",
|
||||
"chain_content": "-----BEGIN CERTIFICATE-----...", # optional
|
||||
"usage_type": "frontend", # frontend | server
|
||||
"is_global": false,
|
||||
"cluster_ids": [1, 2]
|
||||
}
|
||||
|
||||
# Delete a CSR (pending: permanently destroys the private key;
|
||||
# completed: removes history only — the imported certificate is unaffected)
|
||||
DELETE /api/ssl/csrs/{csr_id}
|
||||
```
|
||||
|
||||
### ACME / Let's Encrypt API
|
||||
```bash
|
||||
# List ACME accounts
|
||||
@@ -2021,7 +2128,7 @@ haproxy-openmanager/
|
||||
├── docker-compose.yml # Docker Compose configuration
|
||||
├── docker-compose.localtest.yml # Local development/testing overrides
|
||||
├── docker-compose.test.yml # Test environment
|
||||
├── version.json # Application version metadata
|
||||
├── backend/version.json # Application version metadata (single source of truth)
|
||||
├── build-images.sh # Build Docker images
|
||||
├── pytest.ini # Pytest configuration
|
||||
├── README.md # This file
|
||||
@@ -2367,6 +2474,40 @@ Developed with ❤️ for the HAProxy community
|
||||
|
||||
## Release Notes
|
||||
|
||||
- **v1.10.2** (2026-08-08) — **Dark mode fixes on Apply Management**: several panels on the Apply Management page were painted with light-mode colour literals, so in dark mode the **Pending Changes** box rendered as a cream panel with light text on it — measured contrast **1.03:1**, effectively unreadable, now **11.50:1**. The same bug affected the added/removed rows in the *View Change* diff (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME and pending-version panels, the VIP pending-delete row, and the agent-error recommendation box; all now derive from theme tokens. Separately, **static confirm dialogs came up white in dark mode**: in Ant Design 5 the static `Modal.confirm` / `message` / `notification` APIs render into their own detached root and never see the app's `ConfigProvider`, so they always used the light algorithm. Registering `ConfigProvider.config({ holderRender })` once at the app root fixes **every** static dialog in the application (12 components use them), not only this page. Light mode is byte-identical — each token resolves under the default algorithm to exactly the literal it replaced. Frontend only: no schema, API, environment or agent change.
|
||||
- **v1.10.1** (2026-08-08) — **CSR private key encrypted at rest** (Issue #53): the private key of a **pending** CSR is now Fernet-encrypted in the database instead of stored as PEM. It is the one key in the system worth protecting this way — it sits idle for the entire signing window (days to weeks), is never transmitted to an agent, and is destroyed the moment the signed certificate is imported; `ssl_certificates.private_key_content` and the ACME order keys are unchanged, because agents must receive those in plaintext on every poll. The token replaces the PEM in the **same column**, so there is **no schema change and no `SCHEMA_VERSION` bump** (and therefore no re-seed of the built-in roles). CSRs created before this release keep a raw PEM and are still read transparently, so anything already out for signature imports normally with no data migration. The key derives from `SECRET_KEY` via HKDF with its own info string, independent of the VIP/MFA/DNS keys, and an optional `CSR_ENCRYPTION_KEY` enables independent rotation — rotating `SECRET_KEY` without it makes pending CSR keys unrecoverable, which now fails with an explicit "delete and re-create this CSR" error rather than a misleading key-mismatch. `.env.template` now documents all four per-purpose encryption keys. No API, UI or agent change.
|
||||
- **v1.10.0** (2026-08-07) — **GoDaddy DNS provider for DNS-01** (Issue #35 follow-up): DNS-01 challenges can now be published and cleaned up automatically through **GoDaddy**, alongside the existing Manual and Cloudflare providers, so wildcard and internal-cluster certificates on GoDaddy-hosted zones **renew unattended**. Credentials are a **Production API Key + Secret** pair from `developer.godaddy.com/keys` (a **Personal Access Token** also works — paste it as the Key and leave the Secret blank, which is the forward path as GoDaddy retires `sso-key`); they are **verified against the GoDaddy API before being saved** and **encrypted at rest** (Fernet, the same path as Cloudflare), and are never returned by the API, logged, or written to an order event. GoDaddy's v1 API has **no per-value TXT write** — `PUT` replaces an entire RRset — so add/remove are read-modify-write with sibling values merged back, empty-`data` tombstones filtered out, and `DELETE` used for the last value (`PUT []` is rejected); this is what keeps the **apex + wildcard** case (two TXT values at one `_acme-challenge` name) working, and the record path is hard-gated so it can never collapse onto the zone-wide endpoint that would wipe SPF/DKIM/DMARC. Zone lookup probes the records API rather than the domain listing, so **delegated sub-zones** resolve and small accounts are not falsely rejected. Registry-only addition: one new provider module plus one registry line — no frontend change (the credential form is schema-driven). No schema, API-shape, agent, or rendered-config changes; Manual, Cloudflare and HTTP-01 are unaffected.
|
||||
- **v1.9.0** (2026-08-04) — **CSR creation** (in-app key + CSR generation and signed-certificate import): a new **CSR tab** on the SSL Certificates page generates a private key and Certificate Signing Request server-side (RSA 2048/4096 or ECDSA P-256/P-384; full subject — O/OU/L/ST/C/email — plus DNS SANs with wildcard support), for certificates signed by an **external or corporate CA**. The operator downloads/copies the CSR PEM, has it signed, then imports the signed certificate (+ optional chain): the backend verifies the certificate against the stored key (hard gate), rejects expired certs, warns on SAN drift, and creates a normal SSL certificate entry (source `CSR`) that flows through the standard **PENDING → Apply Management → agent pull** pipeline. The private key **never leaves the server** — no CSR endpoint returns it, and after import the CSR row's key copy is destroyed (the key then lives only on the certificate, like every other key). Additive schema change: one new table `ssl_csrs` (SCHEMA_VERSION 9 → 10, auto-migrated, no existing table altered); key generation runs off the event loop and is rate-limited per user; existing `ssl.*` permissions govern all new endpoints. No agent or rendered-config changes.
|
||||
- **v1.8.10** (2026-07-20) — **Security hardening** (GHSA-7rhv-c5pc-69r8, GHSA-3p5c-m5m4-mjpx, GHSA-3vh4): three advisory classes remediated, backend-only, no agent changes. (1) **RCE**: the agent script-template read/write endpoints now require the `agents.version` permission on top of authentication — a poisoned template is executed as root on every HAProxy node, so authentication alone was insufficient. (2) **Missing authentication**: operator/UI endpoints that were served without a JWT (dashboard stats, pool/cluster listings, agent inventory, WAF rules, config validate/optimize, SSL config-versions, health deep/agents/clusters) are now gated by a `require_authenticated_user` dependency, and agent data-plane endpoints that treated the `X-API-Key` header as *optional* (heartbeat, config, ssl-certificates, upgrade-status, pending-requests) now hard-reject a missing key. In every case the auth check was moved **ahead of** the handler's `try:` block so a 401 can no longer be rewritten into a 500 by the generic exception handler. (3) **SSRF**: a new `utils/ssrf_guard.py` (https-only, IPv4-pinned connector, all resolved addresses must be public, no redirects) protects the ACME directory fetch, the signed-request target and the ACME connection test, which accept operator- or DB-supplied URLs; the connection test also stopped reflecting arbitrary upstream JSON. Frontend dependency advisories patched in the same release. No schema, API-shape or rendered-config changes.
|
||||
- **v1.8.9** (2026-07-13) — **ACL `-f` pattern-file support** (Issue #38 follow-up): ACL definitions that reference a host-side pattern file (`acl … -f /etc/haproxy/lists/blocked.lst`) are accepted on import and edit instead of being rejected. The referenced file lives on the HAProxy node and cannot be validated from the manager, so the manager emits an **advisory warning** rather than a hard rejection and lets the agent's `haproxy -c` check be the fail-safe gate (a broken reference fails validation on the node and the previous config is restored). Consistent with the SPOE handling introduced in v1.8.8.
|
||||
- **v1.8.8** (2026-07-10) — **SPOE filter and frontend `log-format` preserved on import/edit** (Issue #38): importing an existing `haproxy.cfg` or editing a frontend silently dropped `filter spoe …` directives and custom `log-format` lines, so the next Apply pushed a config that had lost them. Both are now round-tripped through import and edit. As with `-f` pattern files, the SPOE engine config is a host-side file the manager cannot read, so it is preserved verbatim and reported as an advisory rather than validated centrally.
|
||||
- **v1.8.7** (2026-07-09) — **Version reporting single-source fix**: the version shown in the UI (backend-sourced via `/api/version`) could lag behind the real release. The canonical version lived in the repo-root `version.json`, but the backend image is built from the `./backend` context, so that file did not reach the container in every pipeline; the backend then fell back to a hardcoded constant in `main.py` that had to be bumped by hand and had drifted (it reported 1.8.4 after 1.8.5/1.8.6 shipped). The version now lives in a single file, `backend/version.json`, baked into every image automatically, and `main.py` no longer carries a real version literal (its fallback is a neutral "unknown"). A new test enforces that the version stays single-source and cannot drift. No functional or API change.
|
||||
- **v1.8.6** (2026-07-06) — **Performance: opt-in API workers + heartbeat micro-optimization** (Issue #35 follow-up): the backend container can now run multiple uvicorn worker processes via the new `UVICORN_WORKERS` environment variable (default **1** — behavior unchanged unless you opt in), letting the API use all cores on multi-core hosts; background tasks were already multi-replica safe, as exercised by the Kubernetes HPA deployment. The agent heartbeat handler now reads the agent's `status`/`version`/`upgrade_status` in one query instead of three (one round-trip per heartbeat, per agent, every 30s). Added a *Performance Tuning* section to the README (worker/replica scaling and how to use the `X-Response-Time` header and `Slow request detected` logs to pinpoint slow endpoints). Zero-risk release: no schema, API, or agent changes; defaults preserve existing behavior exactly.
|
||||
- **v1.8.5** (2026-07-03) — **ACME completion-task SQL fix** (Issue #35 follow-up): the background order-completion task (`complete_pending_acme_orders`, runs every 60s) died on **every cycle** with `syntax error at or near ")"` — an extra closing parenthesis introduced in v1.8.0's bounded DNS-01 retry claim query. Because that query is the task's first database call, **no background ACME work ran at all from v1.8.0 through v1.8.4**: orders were never claimed for finalize/download, the DNS-01 TXT record was never published (so DNS-01 with an automated provider such as Cloudflare could never validate), Site Wizard staged orders never left `wizard_staged`, and DNS-01 retry/TXT-cleanup never executed. The stray parenthesis is removed and a regression test now scans all ACME modules' SQL for unbalanced parentheses (the unit suite mocks the database, which is why a raw-SQL syntax error could slip through). One-line backend query fix; no schema, API, or agent changes — fully backward compatible.
|
||||
- **v1.8.4** (2026-06-27) — **Agent installer self-kill fix** (Issue #31): the Linux/macOS agent installer could abort during "pre-installation cleanup" (terminal showed `Killing processes matching: haproxy-agent` then `Killed`) when the install script's own filename contained "haproxy-agent". The cleanup killed processes by matching the bare string "haproxy-agent" against full command lines, which also matched the running installer (and a `sudo`/PAM ancestor the self-exclusion did not cover), so the installer terminated itself. Cleanup now targets only the installed agent (the `$INSTALL_DIR/haproxy-agent` binary and the agent service), never the bare string, and the UI now names the downloaded scripts `install-agent-<platform>.sh` / `uninstall-agent-<platform>.sh`. Installer-only change; the running agent and its privilege model (it runs as root for HAProxy reload, config writes, keepalived, and self-upgrade) are unchanged.
|
||||
- **v1.8.3** (2026-06-25) — **Agent heartbeat JSON fix** (Issue #31): a self-hosted agent could fail every heartbeat with `HTTP 400 Invalid JSON: Expecting property name enclosed in double quotes` when the system-info block it collects came back empty on an unusual host, leaving a stray comma in the hand-built heartbeat JSON. The agent script now substitutes a valid placeholder when that block is empty so it can no longer emit a stray comma, and the backend heartbeat endpoint now parses valid payloads as-is and, only when a body fails to parse, tolerates that specific malformed pattern (a leading or doubled comma) so an already-deployed agent recovers on its next heartbeat after this build is deployed. Backend + agent-script only; healthy agents of every version are byte-for-byte unaffected.
|
||||
- **v1.8.2** (2026-06-25) — **ACME nonce fix** (Issue #35 follow-up): the ACME client now scopes the anti-replay nonce **per certificate authority** so a nonce issued by one CA is never sent to another. This fixes ZeroSSL/Google account registration failing with `malformed: The Replay Nonce could not be base64url-decoded` (the client previously shared one nonce across CAs and only auto-retried on `badNonce`). Account registration now always uses a fresh nonce from the target CA, and the retry covers this case too. Backend-only; HTTP-01 and Let's Encrypt are unaffected.
|
||||
- **v1.8.1** (2026-06-24) — **ACME DNS-01 fixes** (Issue #35 follow-up): Cloudflare API tokens are now sanitized so a pasted token with quotes/spaces no longer fails with "Invalid request headers"; ZeroSSL/Google **External Account Binding (EAB)** can be entered per-account in the register dialog and EAB-required failures show a clear message; and **Apply Management** now categorizes cluster ACME enable/disable changes under their own "ACME Challenge Routing" section and **Apply/Reject All** correctly process them (previously "Rejected 0 HA/VIP change(s)"), consistent with every other entity. Fully backward compatible.
|
||||
- **v1.8.0** (2026-06-23) — **ACME DNS-01 challenge support** (Issue #35): Auto SSL can now validate via a **DNS TXT record** (`_acme-challenge.<domain>`) instead of HTTP-01 on port 80, enabling certificates for **internal/isolated clusters with no public ingress** and **wildcard** certificates (`*.example.com`). Pluggable **per-account DNS provider** (Manual + Cloudflare to start; credentials verified on save and **encrypted at rest**, never returned by the API or logged), the same **PENDING → APPLIED** pipeline, a **bounded automatic retry** on propagation lag, and a **DNS-01 event timeline** in the order detail. **Opt-in** via Settings → ACME (global switch, default off); **HTTP-01 is byte-for-byte unchanged**, with **zero agent or rendered-config changes**. Manual DNS-01 certificates cannot auto-renew unattended; the UI states this and disables auto-renew for them.
|
||||
- **v1.7.8** (2026-06-07) — HA / VIP apply progress now shows **per-node** convergence: a multi-node VIP's apply popup reads "Syncing HA/VIP… 1/2 node(s) converged" (matching the HA/VIP table) instead of a coarse per-change count. Frontend-only.
|
||||
- **v1.7.7** (2026-06-07) — HA / VIP apply-progress consistency: applying a VIP change (or approving a delete) used to flash the progress popup green instantly while the HA/VIP page still showed `SYNCING (0/1)` for a couple of minutes. The popup now **keeps showing "Syncing HA/VIP… X/Y node(s) converged"** until each member node reports the VIP `ACTIVE` (create/edit) or fully torn down (delete) — exactly like the HAProxy agent-sync widget — then completes green. It's a fire-and-forget background poll (the Apply button is released immediately), bounded at ~5 min so an offline node can't spin forever (then it completes with an informational "still converging — track on the HA/VIP page"). Frontend-only; no backend/agent/schema change.
|
||||
- **v1.7.6** (2026-06-07) — HA / VIP UX + accuracy polish: (1) the on-prem/L2 cloud caveat is now a **subtle, collapsed-by-default "Network requirements" info link** instead of a prominent yellow warning. (2) The delete dialog is simplified — deletion is **always a graceful teardown** (stop & disable keepalived, remove our config, release the VIP, keep the package); the confusing "also uninstall the package" checkbox was removed (it was a no-op on any node whose keepalived predates the install marker, and package removal is better handled as a deliberate node-decommission step — the `purge_package` API remains for that). (3) The agent now reports keepalived **FAULT** state (e.g. when the chosen interface has no usable IPv4) instead of misreporting it as BACKUP, so a misconfigured VIP shows red/FAULT in the UI. (1)+(2) are frontend-only; (3) is an additive agent-script change — push it via **Agent Script Management → Reset to Defaults**, then **Upgrade**.
|
||||
- **v1.7.5** (2026-06-07) — HA / VIP "View Change" fix: editing a VIP (e.g. priority + virtual IP) now shows a **real line diff — only the lines that actually changed** — in Apply Management's View Change, instead of rendering the whole `keepalived.conf` as "added". Also fixes a doubled `+ +` prefix (the VIP diff now stores lines without a +/- prefix, matching the standard config diff). View-only; no schema, agent, apply, or render change.
|
||||
- **v1.7.4** (2026-06-06) — HAProxy config-generator robustness fix: a frontend that uses a stick counter (`track-sc<N>` or an `sc_*_rate(...)` fetch, e.g. a rate-limit `http-request deny if { sc_http_req_rate(0) gt N }`) but declares **no `stick-table`** caused HAProxy to fatally reject the whole cluster config with *"table '<frontend>' used but not configured"*. This happened where rate-limit directives had been baked into a frontend's stored `request_headers`/`options` (by an older version or a config import). The generator now **auto-injects a default `stick-table`** in that case. Purely additive — it only fires when a counter is used and no table exists (a config that was already invalid), so frontends that already declare a stick-table or don't rate-limit are byte-unchanged.
|
||||
- **v1.7.3** (2026-06-06) — HA / VIP backward-compat fix: the v1.7.2 deletion-tracking list now only resurfaces VIPs deleted through the **new approval flow** (gated on `last_config_status='APPLIED'`), so a VIP soft-deleted under an earlier version's immediate-delete is no longer shown as `DELETING`. Display-only; no schema or agent change.
|
||||
- **v1.7.2** (2026-06-06) — HA / VIP safety & visibility follow-up:
|
||||
- **Approval-gated deletion (safety).** Deleting a *running* VIP from the UI no longer takes effect immediately — it is **staged for Apply Management** and the VIP **keeps running, untouched**, until you **Approve** it (Reject keeps it). The agent is told to tear keepalived down **only after approval**, so a misclick can never tear down a production VIP — an agent never deletes without an explicit human approval. The deletion is trackable through the standard Apply popup and a **DELETING** status on the HA/VIP tab. (A VIP that was never applied is removed at once — nothing is running to tear down.)
|
||||
- **Diagnostics view** — a per-VIP search-icon modal showing each node's keepalived deploy state, the message it reported, last-ack time and live VRRP state (handy while a fresh install is SYNCING, ~30s), plus the exact node-side log commands.
|
||||
- **Opt-in package uninstall** — the safe default keeps the keepalived package (stop & disable, remove our config, release the VIP); a default-off checkbox additionally uninstalls the package **only on nodes where OpenManager installed it** (an admin's pre-existing keepalived is never removed, tracked via an install marker).
|
||||
- Adds additive `purge_on_teardown` + `pending_delete` columns; `SCHEMA_VERSION` 5 → 7 (idempotent; existing data/passwords unaffected). Agent script updated — push it via **Agent Script Management → Reset to Defaults**, then **Upgrade** agents.
|
||||
- **v1.7.1** (2026-06-06) — HA / VIP follow-up: a VIP can now be created on a **single node** — a Keepalived-managed floating IP **without** failover (e.g. a one-box HAProxy that wants a stable address, or before a second node is added). Add a second node anytime for real VRRP failover. A single-node VIP renders a clean **multicast** config (no bare `unicast_src_ip`, which `keepalived -t` rejects); multi-node behaviour is unchanged. The Create-VIP form now auto-selects the first chosen node as **MASTER** and shows an in-UI notice that, on Apply, Keepalived is **installed automatically from the node's OS package repositories** (apt/dnf/yum/zypper/apk — the node must reach its repos / an internal mirror), while a hand-managed Keepalived is still left untouched. No schema change.
|
||||
- **v1.7.0** (2026-06-05) — Feature (Issue #27): **HA / VIP (Keepalived) management** from the UI. A new "HA / VIP" tab lets you create a virtual IP, pick a per-node interface, and select which pool nodes participate (with MASTER/BACKUP roles + priorities); on Apply, the agent installs & configures Keepalived (unicast VRRP, cloud-safe default) with a HAProxy health-check so the VIP fails over automatically when HAProxy drops, and the tab shows live MASTER/BACKUP per node. Fully **opt-in and backward compatible** — nodes/clusters without a VIP are untouched, and a node already running a hand-managed Keepalived is detected and never overwritten (reported as "externally managed"). Pending VIP changes can be **Rejected** to fully restore the last applied state. Keepalived is installed across the major distros (Debian/Ubuntu, RHEL/CentOS/Alma/Rocky, Fedora, SUSE/openSUSE, Alpine), and live MASTER/BACKUP detection works across distros/init systems (journald + log files + portable interface-based detection). On-prem/L2 scope (VRRP); a clear in-UI notice covers the cloud caveat. Adds two new tables (`vip_instances`, `vip_members`) — `SCHEMA_VERSION` bumps to 3 (idempotent re-run; existing data and passwords unaffected). A new "HA / VIP" tab lets you create a virtual IP, pick a per-node interface, and select which pool nodes participate (with MASTER/BACKUP roles + priorities); on Apply, the agent installs & configures Keepalived (unicast VRRP, cloud-safe default) with a HAProxy health-check so the VIP fails over automatically when HAProxy drops, and the tab shows live MASTER/BACKUP per node. Fully **opt-in and backward compatible** — nodes/clusters without a VIP are untouched, and a node already running a hand-managed Keepalived is detected and never overwritten (reported as "externally managed"). Pending VIP changes can be **Rejected** to fully restore the last applied state. Keepalived is installed across the major distros (Debian/Ubuntu, RHEL/CentOS/Alma/Rocky, Fedora, SUSE/openSUSE, Alpine), and live MASTER/BACKUP detection works across distros/init systems (journald + log files + portable interface-based detection). On-prem/L2 scope (VRRP); a clear in-UI notice covers the cloud caveat. Adds two new tables (`vip_instances`, `vip_members`) — `SCHEMA_VERSION` bumps to 3 (idempotent re-run; existing data and passwords unaffected).
|
||||
- **v1.6.5** (2026-06-02) — Security: re-pinned the bundled nginx reverse-proxy image to `nginx:1.31.1-alpine` (mainline patched release) for the nginx "poolslip" advisory (fixed in mainline 1.31.1+ / stable 1.30.2+). Supersedes the v1.6.4 stable pin. No config, schema, or behavior changes.
|
||||
- **v1.6.4** (2026-06-02) — Security: pinned the bundled nginx reverse-proxy image to a patched stable release (`nginx:1.30.2-alpine`) for the nginx "poolslip" advisory (mainline ≤ 1.31.0 affected; fixed in stable 1.30.2+). The product's nginx config uses no `rewrite` capture groups, so the config-level mitigation did not apply — the fix is the version pin. No config, schema, or behavior changes.
|
||||
- **v1.6.3** (2026-06-01) — Bugfix: a backend **server** toggled OFF (`is_active=false`) disappeared from the UI with no way to reactivate it. `GET /api/backends` now honors `include_inactive` for servers (previously only backends), so disabled servers stay visible with an OFF switch + "Inactive" tag and can be re-enabled; soft-deleted (pending-delete) servers stay hidden. A Reject of a server toggle now correctly rolls back `is_active` (entity snapshot). Startup migrations are hardened for multiple replicas / rolling deploys (advisory lock + schema-version gate, so an already-current schema isn't re-migrated under a serving peer's load). Version is reported consistently across all layers. No schema changes; config generation unchanged (disabled servers stay commented out).
|
||||
- **v1.6.2** (2026-05-30) — Bugfixes: (1) agents (which authenticate with their `X-API-Key` token) could not reach `GET /api/clusters` / `/api/clusters/{id}` after the v1.5.x cluster-read hardening, breaking agent assignment ("401: Authorization header missing"); these endpoints now accept either a user JWT or an agent token (anonymous access is still rejected). (2) The uninstall-script generator returned 400 for macOS agents (which report platform `darwin`); it now normalizes the platform the same way the install generator does. No UI or schema changes.
|
||||
- **v1.6.1** (2026-05-21) — Security patch: bump `axios` to 1.16.x (prototype-pollution hardening, header-injection fix, keep-alive memory leak fix) and `fast-uri` to 3.1.2 (GHSA-v39h-62p7-jpjc). No functional changes.
|
||||
|
||||
For full release notes and the list of features delivered in each version (v1.5.x Site Wizard + ACME Diagnostic Panel, v1.4.0 ACME stability + enterprise audit, v1.3.0, ...) see the [GitHub Releases](https://github.com/taylanbakircioglu/haproxy-openmanager/releases) page.
|
||||
|
||||
---
|
||||
|
||||
@@ -1,3 +1,173 @@
|
||||
# Upgrade Notes — v1.10.2 (Dark mode fixes on Apply Management)
|
||||
|
||||
**Frontend only. Nothing to do on upgrade.** No schema, no `SCHEMA_VERSION` bump, no API change,
|
||||
no environment variable, zero agent impact. Light mode is byte-identical: every colour swapped in
|
||||
this release resolves, under the default algorithm, to exactly the literal it replaced
|
||||
(`colorWarningBg` → `#fffbe6`, `colorSuccessBg` → `#f6ffed`, `colorErrorBg` → `#fff2f0`, …), so
|
||||
only dark mode changes.
|
||||
|
||||
- **Apply Management panels** were painted with light-mode colour literals, so in dark mode the
|
||||
"Pending Changes" box rendered as a cream panel with light text on it. Measured contrast was
|
||||
**1.03:1** — effectively invisible. It is now **11.50:1**. The same class of bug affected the
|
||||
diff rows in *View Change* (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME/pending version
|
||||
panels, the VIP pending-delete row and the agent-error recommendation box.
|
||||
- **Static confirm dialogs came up white in dark mode.** In Ant Design 5 the static
|
||||
`Modal.confirm` / `message` / `notification` APIs render into their own detached root and never
|
||||
see the app's `ConfigProvider`, so they always used the light algorithm. This release registers
|
||||
`ConfigProvider.config({ holderRender })` once at the app root, which fixes **every** static
|
||||
dialog in the app (12 components use them), not just Apply Management.
|
||||
- **Rollback:** downgrade freely. This release changes rendering only.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.1 (CSR private key encrypted at rest)
|
||||
|
||||
**Backward compatible.** Nothing to do on upgrade, and nothing changes for existing clusters,
|
||||
agents or certificates:
|
||||
|
||||
- **Schema:** **no `SCHEMA_VERSION` bump and no migration.** The Fernet token replaces the PEM
|
||||
inside the *existing* `ssl_csrs.private_key_pem` TEXT column. As in v1.10.0, this means the
|
||||
four built-in roles are **not** re-seeded, so any customization of `super_admin` / `operator` /
|
||||
`security_admin` / `viewer` survives.
|
||||
- **Existing pending CSRs keep working.** Rows written before this release hold a raw PEM and are
|
||||
still read transparently, so a CSR that is already out for signature can be imported normally
|
||||
after the upgrade. There is no data migration and no downtime step. Those rows stay plaintext
|
||||
until they are imported (which NULLs the key) — if you want everything encrypted immediately,
|
||||
delete and re-create any long-pending CSRs.
|
||||
- **Scope:** this covers the PENDING CSR key only. It is the one key in the system that sits idle
|
||||
for the whole signing window and is never transmitted. `ssl_certificates.private_key_content`
|
||||
and the ACME order keys are unchanged, because agents must receive those in plaintext on every
|
||||
poll.
|
||||
- **Optional env:** `CSR_ENCRYPTION_KEY` (see `.env.template`). If unset, the key is derived from
|
||||
`SECRET_KEY` via HKDF with its own info string, so it is independent of the VIP, MFA and DNS
|
||||
provider keys.
|
||||
- **⚠️ Rotating `SECRET_KEY` while `CSR_ENCRYPTION_KEY` is unset makes pending CSR keys
|
||||
unrecoverable.** Import then fails with an explicit "delete this CSR and create a new one"
|
||||
error rather than a misleading key-mismatch. Set an explicit `CSR_ENCRYPTION_KEY` if you
|
||||
rotate `SECRET_KEY`. Certificates already imported are unaffected — their key lives on the
|
||||
certificate row.
|
||||
- **API / UI / agents:** unchanged. No CSR endpoint ever returned the private key before or now,
|
||||
and nothing about the CSR tab changes.
|
||||
- **Rollback:** the application downgrades cleanly — 1.10.0 starts normally against the same
|
||||
database and every other feature is unaffected. The one casualty is a CSR **created on 1.10.1
|
||||
and still pending**: 1.10.0 has no decrypt step, so it hands the Fernet token straight to the
|
||||
key-pairing check. Measured on a real downgrade, the import then fails with
|
||||
`HTTP 500 — Could not verify the certificate/key pair: key parse failed (encrypted?)`; it does
|
||||
**not** silently pair the wrong key, and it does not corrupt anything. Import or delete CSRs
|
||||
created on 1.10.1 before downgrading. Certificates already imported are unaffected, since their
|
||||
key lives on the certificate row, and CSRs created before 1.10.1 are plaintext and still work.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.0 (GoDaddy DNS-01 provider)
|
||||
|
||||
**Backward compatible & additive.** Nothing changes unless you select **GoDaddy** as an ACME
|
||||
account's DNS provider:
|
||||
|
||||
- **Schema:** **no `SCHEMA_VERSION` bump.** The GoDaddy credentials (API Key + Secret) are stored
|
||||
as two keys inside the *existing* encrypted
|
||||
`letsencrypt_account_dns_credentials.credentials_encrypted` blob — no new table, no new column,
|
||||
no migration.
|
||||
- **✅ Built-in roles are NOT re-seeded.** The re-seed warning in the v1.9.0 notes below is
|
||||
triggered by a `SCHEMA_VERSION` bump. This release does not bump it, so any customization you
|
||||
made to `super_admin` / `operator` / `security_admin` / `viewer` survives untouched.
|
||||
- **Permissions / API shape:** unchanged. `GET /api/letsencrypt/dns-providers` simply returns one
|
||||
extra entry in its `providers` array; every request and response shape is identical, and the
|
||||
credential form is rendered from that schema, so there is no frontend behaviour change either.
|
||||
- **Environment:** no new variable. GoDaddy credentials use the same Fernet-at-rest path as
|
||||
Cloudflare (`DNS_PROVIDER_ENCRYPTION_KEY`, falling back to a key derived from `SECRET_KEY`).
|
||||
- **Agents:** zero agent changes. DNS-01 is invisible to agents; an issued certificate follows the
|
||||
normal PENDING → Apply Management → agent pull pipeline exactly as before.
|
||||
- **Using it:** the API Key must be a **Production** key from `developer.godaddy.com/keys` (the
|
||||
first key that dashboard issues is an OTE/test key and is rejected), the zone must be in the same
|
||||
GoDaddy account, and that account needs at least one registered domain before GoDaddy permits DNS
|
||||
API access. A Personal Access Token also works — paste it as the Key and leave the Secret blank.
|
||||
Credentials are checked against the GoDaddy API before they are stored, so an invalid, OTE or
|
||||
ineligible key fails at save time. Note the check is a **read**: a Personal Access Token that has
|
||||
`domains.domain:read` but not `domains.dns:update` saves successfully and only fails at the first
|
||||
publish, with a 403 in the order timeline.
|
||||
- **Rollback:** don't select GoDaddy. Existing Manual and Cloudflare accounts and all HTTP-01
|
||||
issuance are untouched. **Downgrading after adopting GoDaddy is not a no-op**: on 1.9.0
|
||||
`godaddy` is not a known provider, so any account still set to it degrades to the manual-confirm
|
||||
path (in-flight DNS-01 orders wait for a confirmation nobody can give and expire after 48h, and
|
||||
renewals stop), and the cleanup sweep marks published TXT records cleaned without removing them.
|
||||
Before downgrading, switch affected accounts back to Manual or Cloudflare and let the reconcile
|
||||
sweep remove outstanding `_acme-challenge` records first. The stored credential row itself is
|
||||
inert — an encrypted blob for an unknown provider.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.9.0 (CSR creation)
|
||||
|
||||
**Backward compatible & additive.** Upgrading to v1.9.0 changes nothing for existing
|
||||
clusters/agents until you create a CSR:
|
||||
|
||||
- **Schema:** `SCHEMA_VERSION` bumps to `10`, so on first start the (idempotent)
|
||||
migration sequence re-runs once and adds **one new table** (`ssl_csrs`) plus its
|
||||
indexes. **No existing table is altered**, existing rows are untouched, and the
|
||||
**admin password is not reset** (the default-user seeding is guarded by an
|
||||
existence check, not an upsert). No new permission strings are introduced — all
|
||||
CSR endpoints are governed by the existing `ssl.create` / `ssl.read` /
|
||||
`ssl.delete` permissions.
|
||||
- **⚠️ Built-in roles are re-seeded to their defaults (pre-existing behaviour of
|
||||
every `SCHEMA_VERSION` bump — verified in a v1.8.10 → v1.9.0 upgrade drill).**
|
||||
Because the version gate re-runs the whole sequence, `update_system_roles_to_enterprise_rbac()`
|
||||
issues an unconditional `UPDATE roles SET … permissions = <defaults> WHERE name = …`
|
||||
for the four **built-in** roles (`super_admin`, `operator`, `security_admin`,
|
||||
`viewer`). **Any customization you made to a built-in role is reverted.** In the
|
||||
drill, an `operator` role that had been narrowed by removing `apply.execute` and
|
||||
`config.bulk_import` came back with both restored (57 → 59 permissions).
|
||||
- **Roles you created yourself are NOT affected** — the re-seed matches on the four
|
||||
built-in names only.
|
||||
- This is not new in v1.9.0: it happens on every release that bumps
|
||||
`SCHEMA_VERSION` (v1.7.0, v1.8.0, v1.8.8 …). It is documented as intentional at
|
||||
`backend/database/migrations.py` (the "BUMP THIS … OR seeded/role data" note) —
|
||||
the migration is treated as the authority on built-in-role contents.
|
||||
- **If you have hardened a built-in role, do this:** export it before upgrading
|
||||
(`GET /api/roles`), then re-apply your changes after the first start
|
||||
(`PUT /api/roles/{id}`) — or, preferably, move your customization into a
|
||||
purpose-made custom role, which survives every upgrade.
|
||||
- **Key storage:** CSR private keys are stored in the database like every other key
|
||||
in the system (`ssl_certificates.private_key_content` and the ACME order keys).
|
||||
The key is never returned by any CSR API endpoint, and after a successful import
|
||||
the CSR row's key copy is set to NULL (the key then lives only on the certificate
|
||||
row).
|
||||
- **Agents:** zero agent changes. Agents never read the new table; a CSR becomes
|
||||
visible to agents only after its signed certificate is imported **and** applied via
|
||||
Apply Management (the standard PENDING pipeline).
|
||||
- **Rollback:** simply don't use the CSR tab. The `ssl_csrs` table is inert when
|
||||
empty; downgrading the application leaves it as an ignored extra table.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.7.0 (HA / VIP Keepalived management, Issue #27)
|
||||
|
||||
**Backward compatible & opt-in.** Upgrading to v1.7.0 changes nothing for existing
|
||||
clusters/agents until you create a VIP:
|
||||
|
||||
- **Schema:** `SCHEMA_VERSION` bumps to `3`, so on first start the (idempotent)
|
||||
migration sequence re-runs once and adds two **new** tables (`vip_instances`,
|
||||
`vip_members`) plus an additive `vip_instances.applied_snapshot` column (enables
|
||||
rejecting a pending VIP change and restoring the previous applied state). No existing
|
||||
table is altered. Existing rows and the **admin password
|
||||
are not reset** (default users are create-if-missing). The only data effect is that
|
||||
the **four built-in system roles** (`super_admin`/`operator`/`security_admin`/`viewer`)
|
||||
are re-seeded to their canonical permission sets **plus** the new `vip.*` permissions —
|
||||
this is the long-standing behavior of the role seeder; **custom roles are untouched**.
|
||||
- **Agents:** the agent script gains an opt-in keepalived deploy that is a **no-op** on
|
||||
any node without an applied VIP, and it **never overwrites a hand-managed
|
||||
`/etc/keepalived/keepalived.conf`** (it reports "externally managed" instead).
|
||||
- **Scope:** VRRP VIPs target bare-metal / VMware / on-prem L2 networks. On AWS/Azure/GCP
|
||||
the cloud fabric doesn't honor VRRP/gratuitous-ARP; the UI surfaces this. Ensure host
|
||||
firewalls permit VRRP (IP protocol 112).
|
||||
- **Optional env:** `VIP_ENCRYPTION_KEY` (see `.env.template`) — if unset, the VRRP secret
|
||||
encryption key is derived from `SECRET_KEY` (like MFA).
|
||||
|
||||
No rollback steps are required to *disable* the feature: simply don't create VIPs (or
|
||||
delete them — agents tear down their managed keepalived on the next poll).
|
||||
|
||||
---
|
||||
|
||||
# Agent Upgrade Guide - Dashboard Stats Fix
|
||||
|
||||
## Problem
|
||||
|
||||
+10
-2
@@ -37,5 +37,13 @@ USER appuser
|
||||
# Expose port
|
||||
EXPOSE 8000
|
||||
|
||||
# Run the application in production mode (without --reload)
|
||||
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"]
|
||||
# Run the application in production mode (without --reload).
|
||||
# UVICORN_WORKERS (default 1) opts into multiple worker processes on multi-core
|
||||
# hosts; with 1 worker uvicorn runs in-process, identical to the flagless CMD
|
||||
# this replaces. Falls back to WEB_CONCURRENCY when UVICORN_WORKERS is unset
|
||||
# because flagless uvicorn honored WEB_CONCURRENCY (uvicorn config.py) — this
|
||||
# keeps any deployment that relied on it byte-for-byte compatible. Background
|
||||
# tasks are multi-replica safe (FOR UPDATE SKIP LOCKED / advisory locks), as
|
||||
# already exercised by the k8s HPA deployment. `exec` keeps uvicorn as PID 1
|
||||
# so signal handling is unchanged.
|
||||
CMD ["sh", "-c", "exec uvicorn main:app --host 0.0.0.0 --port 8000 --workers ${UVICORN_WORKERS:-${WEB_CONCURRENCY:-1}}"]
|
||||
@@ -1,4 +1,4 @@
|
||||
from fastapi import HTTPException, status
|
||||
from fastapi import HTTPException, status, Header
|
||||
from typing import Optional, Dict, Any
|
||||
from jose import jwt
|
||||
import logging
|
||||
@@ -92,6 +92,18 @@ async def get_current_user_from_token(authorization: Optional[str] = None) -> Op
|
||||
detail="Authentication failed"
|
||||
)
|
||||
|
||||
async def require_authenticated_user(authorization: Optional[str] = Header(None)) -> Dict[str, Any]:
|
||||
"""FastAPI dependency: require a valid operator JWT, else 401.
|
||||
|
||||
Reads the Authorization header itself, so it can be attached at router or
|
||||
route level to gate operator/UI endpoints that must not be public:
|
||||
APIRouter(..., dependencies=[Depends(require_authenticated_user)])
|
||||
@router.get(..., dependencies=[Depends(require_authenticated_user)])
|
||||
Any authenticated user passes (no fine-grained RBAC here) — this restores the
|
||||
pre-existing "logged-in users only" expectation without changing role access.
|
||||
"""
|
||||
return await get_current_user_from_token(authorization)
|
||||
|
||||
async def get_current_user_from_token_no_exception(authorization: Optional[str] = None) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Get current user from JWT token without raising HTTPException.
|
||||
|
||||
@@ -50,9 +50,62 @@ async def ensure_agents_table():
|
||||
await conn.execute("ALTER TYPE config_status ADD VALUE IF NOT EXISTS 'DELETION';")
|
||||
logger.info("Ensured REJECTED and DELETION values exist in config_status enum.")
|
||||
|
||||
# First, create essential tables if they don't exist
|
||||
await create_essential_tables(conn)
|
||||
|
||||
# First, create essential tables if they don't exist.
|
||||
#
|
||||
# Rolling-restart resilience: create_essential_tables() runs idempotent
|
||||
# CREATE ... IF NOT EXISTS statements on every startup. Its CREATE INDEX
|
||||
# block needs a SHARE lock that conflicts with concurrent writes (e.g.
|
||||
# agent heartbeats updating backend_servers/agents). During a redeploy a
|
||||
# writer can hold that lock, so the DDL blocked for the full 60s
|
||||
# command_timeout -> TimeoutError -> startup crash -> crash-loop.
|
||||
#
|
||||
# Fix: fail fast on locks (lock_timeout), retry briefly, and on
|
||||
# persistent contention SKIP the idempotent bootstrap and continue — on
|
||||
# an established DB the objects already exist; a fresh DB has no writers
|
||||
# so the first attempt always succeeds. lock_timeout is scoped to this
|
||||
# call and RESET afterwards, so every other migration below keeps its
|
||||
# original (wait-indefinitely) behavior. Non-lock errors still propagate
|
||||
# (genuine schema problems must NOT be masked).
|
||||
import asyncio as _asyncio
|
||||
_lock_excs = (_asyncio.TimeoutError,)
|
||||
try:
|
||||
import asyncpg as _asyncpg
|
||||
_lock_excs = _lock_excs + (
|
||||
_asyncpg.exceptions.LockNotAvailableError,
|
||||
_asyncpg.exceptions.QueryCanceledError,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
await conn.execute("SET lock_timeout = '10s'")
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
for _attempt in range(1, 4):
|
||||
try:
|
||||
await create_essential_tables(conn)
|
||||
break
|
||||
except _lock_excs as _lock_err:
|
||||
if _attempt < 3:
|
||||
logger.warning(
|
||||
f"create_essential_tables: lock contention "
|
||||
f"(attempt {_attempt}/3), retrying in 3s "
|
||||
f"({type(_lock_err).__name__})"
|
||||
)
|
||||
await _asyncio.sleep(3)
|
||||
else:
|
||||
logger.warning(
|
||||
"create_essential_tables: persistent lock contention; "
|
||||
"skipping idempotent schema bootstrap and continuing "
|
||||
"startup (objects already exist on an established DB) "
|
||||
f"({type(_lock_err).__name__})"
|
||||
)
|
||||
finally:
|
||||
try:
|
||||
await conn.execute("RESET lock_timeout")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Ensure status column exists in config_versions table
|
||||
status_column_exists = await conn.fetchval("""
|
||||
SELECT 1 FROM information_schema.columns
|
||||
@@ -141,7 +194,14 @@ async def ensure_agents_table():
|
||||
'use_backend_rules': "ALTER TABLE frontends ADD COLUMN use_backend_rules JSONB DEFAULT '[]'::jsonb;",
|
||||
'request_headers': "ALTER TABLE frontends ADD COLUMN request_headers TEXT;",
|
||||
'response_headers': "ALTER TABLE frontends ADD COLUMN response_headers TEXT;",
|
||||
'maxconn': "ALTER TABLE frontends ADD COLUMN maxconn INTEGER;"
|
||||
'maxconn': "ALTER TABLE frontends ADD COLUMN maxconn INTEGER;",
|
||||
# Issue #38: SPOE filter directives (e.g. Coraza WAF) and frontend
|
||||
# log-format were silently dropped on bulk-import / manual edit
|
||||
# because the parser recognised only a fixed set of directives.
|
||||
# These nullable TEXT columns persist them verbatim (multi-line for
|
||||
# `filters`), mirroring the request_headers/options passthrough.
|
||||
'log_format': "ALTER TABLE frontends ADD COLUMN log_format TEXT;",
|
||||
'filters': "ALTER TABLE frontends ADD COLUMN filters TEXT;"
|
||||
}
|
||||
|
||||
for col, query in frontend_columns.items():
|
||||
@@ -431,6 +491,7 @@ async def ensure_agents_table():
|
||||
connection_type VARCHAR(50) DEFAULT 'agent',
|
||||
stats_socket_path VARCHAR(500) DEFAULT '/run/haproxy/admin.sock',
|
||||
haproxy_config_path VARCHAR(500) DEFAULT '/etc/haproxy/haproxy.cfg',
|
||||
keepalived_config_path VARCHAR(500) DEFAULT '/etc/keepalived/keepalived.conf',
|
||||
pool_id INTEGER REFERENCES haproxy_cluster_pools(id) ON DELETE SET NULL,
|
||||
haproxy_user VARCHAR(255) DEFAULT 'haproxy',
|
||||
haproxy_group VARCHAR(255) DEFAULT 'haproxy',
|
||||
@@ -1298,6 +1359,7 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'apply.read', 'apply.execute', 'apply.reject', 'apply.history', 'apply.bulk', 'apply.emergency',
|
||||
'agents.read', 'agents.create', 'agents.update', 'agents.delete', 'agents.script', 'agents.toggle', 'agents.upgrade', 'agents.version', 'agents.logs',
|
||||
'clusters.read', 'clusters.create', 'clusters.update', 'clusters.delete', 'clusters.switch', 'clusters.config',
|
||||
'vip.read', 'vip.create', 'vip.update', 'vip.delete', 'vip.apply',
|
||||
'config.read', 'config.update', 'config.download', 'config.upload', 'config.backup', 'config.restore', 'config.history', 'config.bulk_import', 'config.view_request', 'config.download_request',
|
||||
'users.read', 'users.create', 'users.update', 'users.delete', 'users.password', 'users.roles',
|
||||
'roles.read', 'roles.create', 'roles.update', 'roles.delete', 'roles.permissions',
|
||||
@@ -1319,6 +1381,7 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'apply.read', 'apply.execute', 'apply.reject', 'apply.history', 'apply.bulk',
|
||||
'agents.read', 'agents.update', 'agents.toggle', 'agents.upgrade', 'agents.version', 'agents.logs',
|
||||
'clusters.read', 'clusters.switch', 'clusters.config',
|
||||
'vip.read', 'vip.create', 'vip.update', 'vip.delete', 'vip.apply',
|
||||
'config.read', 'config.update', 'config.download', 'config.history', 'config.bulk_import', 'config.view_request', 'config.download_request',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health',
|
||||
'activity.read'
|
||||
@@ -1336,6 +1399,7 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'apply.read', 'apply.execute', 'apply.reject', 'apply.history',
|
||||
'agents.read', 'agents.version', 'agents.logs',
|
||||
'clusters.read', 'clusters.switch',
|
||||
'vip.read',
|
||||
'config.read', 'config.history', 'config.view_request', 'config.download_request',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health',
|
||||
'activity.read', 'activity.all', 'activity.export',
|
||||
@@ -1354,6 +1418,7 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'apply.read', 'apply.history',
|
||||
'agents.read',
|
||||
'clusters.read', 'clusters.switch',
|
||||
'vip.read',
|
||||
'config.read', 'config.history', 'config.view_request',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health',
|
||||
'activity.read',
|
||||
@@ -1639,13 +1704,136 @@ async def ensure_agent_activity_logs_table():
|
||||
await close_database_connection(conn)
|
||||
# Don't raise - this is not critical for system operation
|
||||
|
||||
# Schema-version gate for the migration runner.
|
||||
#
|
||||
# >>> BUMP THIS whenever you add/modify ANY step in _run_all_migrations_inner()
|
||||
# >>> that changes the schema (table/column/index/constraint) OR seeded/role data
|
||||
# >>> (e.g. update_system_roles_to_enterprise_rbac). Otherwise the new step will
|
||||
# >>> NOT run on databases already marked at the current version.
|
||||
#
|
||||
# When the DB already records >= this version, run_all_migrations() skips the
|
||||
# whole (lock-heavy) idempotent sequence, so redeploys/scale-ups issue NO DDL and
|
||||
# a concurrently-serving replica's traffic cannot block ALTER / CREATE INDEX (the
|
||||
# rolling-deploy startup crash that motivated this gate).
|
||||
#
|
||||
# Backward compatibility (the product runs at many versions across companies):
|
||||
# - First start on this code: no marker -> applied_version is NULL -> the FULL
|
||||
# sequence runs (upgrades any prior version), THEN the marker is written. So
|
||||
# upgrading from any older version is unaffected.
|
||||
# - The marker is written ONLY after _run_all_migrations_inner() completes with
|
||||
# no exception, so an interrupted/failed migration never marks an incomplete
|
||||
# schema as done — the next start retries.
|
||||
# - Behavior change vs the historical "re-run every idempotent ensure_* on every
|
||||
# start": once marked, same-version restarts no longer re-run (and therefore no
|
||||
# longer auto-repair manual drift). To force a re-run, bump SCHEMA_VERSION or
|
||||
# delete the schema_migrations row.
|
||||
#
|
||||
# v1.7.0 (Issue #27 — HA/VIP Keepalived management): bumped 1 -> 2 so the new
|
||||
# additive ensure_vip_tables() step (two brand-new tables) actually runs on
|
||||
# databases already marked at version 1. The whole re-run is idempotent.
|
||||
# v1.7.0 self-review: bumped 2 -> 3 so the additive `applied_snapshot` column on
|
||||
# vip_instances (enables VIP reject/restore-to-previous) lands on DBs marked at 2.
|
||||
# v1.7.0 self-review: bumped 3 -> 4 for the additive `keepalived_config_path` column on
|
||||
# haproxy_clusters (cluster-driven keepalived.conf path, like haproxy_config_path).
|
||||
# v1.7.0 self-review: bumped 4 -> 5 to drop the table-level UNIQUE on vip_instances.name
|
||||
# and replace it with a partial unique index (active rows only), so a soft-deleted VIP's
|
||||
# name is reusable — consistent with the address/VRID partial indexes. Idempotent re-run.
|
||||
# v1.7.2: bumped 5 -> 6 for the additive `purge_on_teardown` column on vip_instances
|
||||
# (opt-in "also uninstall the keepalived package on delete"; default FALSE keeps the safe
|
||||
# graceful-teardown behaviour). Additive + idempotent.
|
||||
# v1.7.2: bumped 6 -> 7 for the additive `pending_delete` column on vip_instances
|
||||
# (approval-gated VIP deletion: a delete is staged for Apply Management and the VIP keeps
|
||||
# running until APPROVED, so an agent never tears down without explicit human approval).
|
||||
# v1.8.0 (Issue #35 — ACME DNS-01 challenge support): bumped 7 -> 8 for additive DNS-01
|
||||
# columns on letsencrypt_accounts/letsencrypt_orders/acme_challenges and the brand-new
|
||||
# letsencrypt_account_dns_credentials table (ensure_letsencrypt_dns_credentials step).
|
||||
# All additive + idempotent; default challenge_type 'http-01' keeps existing flows byte-identical.
|
||||
# v1.8.8 (Issue #38 — SPOE filter + frontend log-format): bumped 8 -> 9 for the additive
|
||||
# `log_format` + `filters` TEXT columns on `frontends` (frontend_columns loop). Without this
|
||||
# bump, already-deployed databases (version >= 8) skip the whole migration run and never gain
|
||||
# the columns, so the frontends SELECT/INSERT would fail. Additive + idempotent + nullable;
|
||||
# existing rows stay NULL and render byte-identical.
|
||||
# v1.9.0 (CSR creation): bumped 9 -> 10 for the brand-new `ssl_csrs` table
|
||||
# (ensure_ssl_csrs_table step). Holds a locally generated private key + CSR PEM
|
||||
# until the operator imports the CA-signed certificate; the import creates a
|
||||
# normal ssl_certificates row and NULLs the key copy here. Additive + idempotent;
|
||||
# no existing table is altered, agents never read this table.
|
||||
SCHEMA_VERSION = 10
|
||||
|
||||
|
||||
async def run_all_migrations():
|
||||
"""Run all database migrations"""
|
||||
"""Run all database migrations.
|
||||
|
||||
Hardened for multiple backend replicas / rolling deploys:
|
||||
- A session-level advisory lock serializes the run so only one pod migrates
|
||||
at a time (others wait, then hit the version gate and skip). It is
|
||||
session-scoped, so it auto-releases if a pod dies mid-migration.
|
||||
- A schema-version marker (schema_migrations) gates the run: when the DB is
|
||||
already at SCHEMA_VERSION the whole idempotent sequence is skipped, so no
|
||||
DDL is issued and a serving replica's traffic can't block it.
|
||||
If the advisory lock or marker can't be used, we fall back to running the
|
||||
(idempotent) migrations rather than crashing startup.
|
||||
"""
|
||||
logger.info("Starting database migrations...")
|
||||
|
||||
MIGRATION_ADVISORY_LOCK_KEY = 1836016242 # single-key advisory space ("migr"); distinct from the (ns,id) locks used elsewhere
|
||||
lock_conn = None
|
||||
lock_acquired = False
|
||||
try:
|
||||
lock_conn = await get_database_connection()
|
||||
try:
|
||||
await lock_conn.execute("SELECT pg_advisory_lock($1)", MIGRATION_ADVISORY_LOCK_KEY)
|
||||
lock_acquired = True
|
||||
logger.info("Acquired migration advisory lock (migrations serialized across pods)")
|
||||
except Exception as _lock_e:
|
||||
logger.warning(f"Could not acquire migration advisory lock; proceeding (migrations are idempotent): {_lock_e}")
|
||||
|
||||
# Schema-version gate: skip the lock-heavy sequence if the DB is current.
|
||||
applied_version = None
|
||||
try:
|
||||
await lock_conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS schema_migrations (
|
||||
id INTEGER PRIMARY KEY DEFAULT 1,
|
||||
version INTEGER NOT NULL,
|
||||
applied_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT schema_migrations_singleton CHECK (id = 1)
|
||||
)
|
||||
""")
|
||||
applied_version = await lock_conn.fetchval("SELECT version FROM schema_migrations WHERE id = 1")
|
||||
except Exception as _mk_e:
|
||||
logger.warning(f"schema_migrations marker unavailable; running full migrations: {_mk_e}")
|
||||
applied_version = None
|
||||
|
||||
if applied_version is not None and applied_version >= SCHEMA_VERSION:
|
||||
logger.info(f"Schema already at version {applied_version} (>= {SCHEMA_VERSION}); skipping migration run.")
|
||||
return
|
||||
|
||||
await _run_all_migrations_inner()
|
||||
|
||||
try:
|
||||
await lock_conn.execute("""
|
||||
INSERT INTO schema_migrations (id, version, applied_at)
|
||||
VALUES (1, $1, CURRENT_TIMESTAMP)
|
||||
ON CONFLICT (id) DO UPDATE SET version = EXCLUDED.version, applied_at = EXCLUDED.applied_at
|
||||
""", SCHEMA_VERSION)
|
||||
logger.info(f"Recorded schema version {SCHEMA_VERSION} in schema_migrations.")
|
||||
except Exception as _wr_e:
|
||||
logger.warning(f"Could not record schema version marker (migrations still applied): {_wr_e}")
|
||||
finally:
|
||||
if lock_acquired and lock_conn is not None:
|
||||
try:
|
||||
await lock_conn.execute("SELECT pg_advisory_unlock($1)", MIGRATION_ADVISORY_LOCK_KEY)
|
||||
except Exception:
|
||||
pass
|
||||
if lock_conn is not None:
|
||||
await close_database_connection(lock_conn)
|
||||
|
||||
|
||||
async def _run_all_migrations_inner():
|
||||
"""The full idempotent migration sequence. Runs under the migration advisory
|
||||
lock and is gated by the schema-version marker in run_all_migrations()."""
|
||||
# First, ensure basic database schema exists
|
||||
await run_init_sql()
|
||||
|
||||
|
||||
# Then run additional migrations
|
||||
await ensure_agents_table()
|
||||
await ensure_config_versions_metadata_column()
|
||||
@@ -1684,6 +1872,9 @@ async def run_all_migrations():
|
||||
await ensure_system_settings_table()
|
||||
await ensure_acme_tables()
|
||||
await ensure_acme_columns_on_existing_tables()
|
||||
# Issue #35 (v1.8.0 — ACME DNS-01): per-account encrypted DNS provider credentials.
|
||||
# MUST run after ensure_acme_tables() (FK references letsencrypt_accounts).
|
||||
await ensure_letsencrypt_dns_credentials()
|
||||
# Issue #11 cleanup: must run AFTER acme_tables/columns to ensure FK refs exist
|
||||
await cleanup_orphan_acme_challenge_backend()
|
||||
# v1.5.0 Feature A (ACME diagnostics) + Feature B (site wizard)
|
||||
@@ -1703,9 +1894,85 @@ async def run_all_migrations():
|
||||
# Issue #18 — TOTP MFA (v1.6.0): additive columns + 3 new tables
|
||||
await ensure_mfa_columns()
|
||||
|
||||
# Issue #27 — HA/VIP Keepalived management (v1.7.0): two brand-new tables.
|
||||
# MUST run after its FK targets (haproxy_cluster_pools/agents/users), all created above.
|
||||
await ensure_vip_tables()
|
||||
|
||||
# v1.9.0 — CSR creation: brand-new ssl_csrs table. FK-references
|
||||
# ssl_certificates/users, both created above.
|
||||
await ensure_ssl_csrs_table()
|
||||
|
||||
logger.info("Database migrations completed successfully.")
|
||||
|
||||
|
||||
async def ensure_ssl_csrs_table():
|
||||
"""v1.9.0 — CSR (Certificate Signing Request) creation. Additive only:
|
||||
one brand-new table (ssl_csrs) + indexes. No ALTER of any existing table,
|
||||
so the entire current fleet is byte-identical. Fully idempotent
|
||||
(CREATE TABLE/INDEX IF NOT EXISTS). FK targets (ssl_certificates, users)
|
||||
are created earlier in the sequence.
|
||||
|
||||
A CSR row holds a locally generated private key + CSR PEM until the
|
||||
operator imports the CA-signed certificate. The import creates a normal
|
||||
ssl_certificates row (source='csr', last_config_status='PENDING') and
|
||||
NULLs the private_key_pem copy here — the key then lives only on the
|
||||
certificate row, like every other key in the system. Agents never read
|
||||
this table: the agent SSL delivery endpoint selects from
|
||||
ssl_certificates only, so a pending CSR can never leak to an agent.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS ssl_csrs (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(100) NOT NULL,
|
||||
common_name VARCHAR(253) NOT NULL,
|
||||
subject JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
sans JSONB NOT NULL DEFAULT '[]'::jsonb,
|
||||
key_algorithm VARCHAR(20) NOT NULL DEFAULT 'rsa-2048',
|
||||
csr_pem TEXT NOT NULL,
|
||||
private_key_pem TEXT,
|
||||
status VARCHAR(20) NOT NULL DEFAULT 'pending',
|
||||
ssl_certificate_id INTEGER REFERENCES ssl_certificates(id) ON DELETE SET NULL,
|
||||
completed_at TIMESTAMP,
|
||||
created_by INTEGER REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT ssl_csrs_status_check CHECK (status IN ('pending', 'completed'))
|
||||
);
|
||||
""")
|
||||
|
||||
# Only PENDING CSRs reserve their name: the name becomes the
|
||||
# ssl_certificates.name (and thus /etc/ssl/haproxy/{name}.pem on every
|
||||
# agent) at import time, so two open CSRs must not target the same
|
||||
# cert name. Completed CSRs are history and may share a name across
|
||||
# reissues — mirrors the uq_vip_name_active partial-index rationale.
|
||||
await conn.execute(
|
||||
"CREATE UNIQUE INDEX IF NOT EXISTS uq_ssl_csrs_name_pending ON ssl_csrs(name) WHERE status = 'pending';"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_ssl_csrs_status ON ssl_csrs(status);"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_ssl_csrs_cert ON ssl_csrs(ssl_certificate_id);"
|
||||
)
|
||||
|
||||
logger.info("ssl_csrs table ensured (v1.9.0 CSR creation)")
|
||||
except Exception as e:
|
||||
logger.error(f"Error ensuring ssl_csrs table: {e}")
|
||||
# Re-raise (ensure_ssl_cluster_junction_table precedent): this step is
|
||||
# part of the SCHEMA_VERSION=10 bump, and run_all_migrations() records
|
||||
# the marker only after the inner sequence completes cleanly. Swallowing
|
||||
# a failure here would stamp version 10 with no ssl_csrs table, and the
|
||||
# version gate would then skip every future retry — permanently.
|
||||
raise
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def ensure_mfa_columns():
|
||||
"""Issue #18 — TOTP MFA (v1.6.0): additive columns on users + 3 new tables.
|
||||
|
||||
@@ -1781,6 +2048,127 @@ async def ensure_mfa_columns():
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
async def ensure_vip_tables():
|
||||
"""Issue #27 — HA/VIP (Keepalived) management (v1.7.0). Additive only:
|
||||
two brand-new tables (vip_instances, vip_members) + indexes. No ALTER of any
|
||||
existing table, so the entire current fleet is byte-identical. Fully idempotent
|
||||
(CREATE TABLE/INDEX IF NOT EXISTS). FK targets (haproxy_cluster_pools, agents,
|
||||
users) are created earlier in the sequence — this function is registered LAST.
|
||||
|
||||
Backward-compat: a cluster/agent with no VIP row is unaffected; the agent
|
||||
delivery endpoint returns 'not_configured' for every node without a membership.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
# Cluster-driven keepalived.conf path (mirrors haproxy_config_path): additive +
|
||||
# idempotent, with a universal default so operators need set nothing. The agent
|
||||
# pulls this from its cluster, exactly like the HAProxy paths.
|
||||
await conn.execute(
|
||||
"ALTER TABLE haproxy_clusters ADD COLUMN IF NOT EXISTS keepalived_config_path "
|
||||
"VARCHAR(500) DEFAULT '/etc/keepalived/keepalived.conf';")
|
||||
|
||||
# VIP instance: one row per virtual IP (one VRRP group), anchored to a pool.
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS vip_instances (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name VARCHAR(255) NOT NULL,
|
||||
description TEXT,
|
||||
pool_id INTEGER NOT NULL REFERENCES haproxy_cluster_pools(id) ON DELETE CASCADE,
|
||||
virtual_ip VARCHAR(45) NOT NULL,
|
||||
prefix_length INTEGER NOT NULL DEFAULT 24,
|
||||
virtual_router_id INTEGER NOT NULL,
|
||||
advert_int INTEGER NOT NULL DEFAULT 1,
|
||||
auth_pass_encrypted TEXT,
|
||||
use_unicast BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
track_haproxy BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
is_active BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
last_config_status VARCHAR(20) NOT NULL DEFAULT 'PENDING',
|
||||
applied_snapshot JSONB,
|
||||
created_by INTEGER REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT vip_vrid_range CHECK (virtual_router_id BETWEEN 1 AND 255)
|
||||
);
|
||||
""")
|
||||
# Additive (idempotent) for DBs that created vip_instances before applied_snapshot
|
||||
# existed (v1.7.0 self-review): holds the field-level state as of the last Apply so
|
||||
# a pending edit can be rejected and fully reverted to the previous applied state.
|
||||
await conn.execute("ALTER TABLE vip_instances ADD COLUMN IF NOT EXISTS applied_snapshot JSONB;")
|
||||
# Opt-in package removal (v1.7.2): when an operator deletes a VIP and explicitly ticks
|
||||
# "also uninstall keepalived from the node(s)", we set this flag so the teardown
|
||||
# delivery tells the agent to purge the OS package. Default FALSE = the safe enterprise
|
||||
# default (stop+disable+remove our config, but KEEP the package). Additive + idempotent;
|
||||
# a node we never managed stays untouched regardless.
|
||||
await conn.execute("ALTER TABLE vip_instances ADD COLUMN IF NOT EXISTS purge_on_teardown BOOLEAN NOT NULL DEFAULT FALSE;")
|
||||
# Approval-gated deletion (v1.7.2): deleting a RUNNING VIP from the UI does NOT take
|
||||
# effect immediately — it sets pending_delete=TRUE and stages a vip-*-delete version
|
||||
# for Apply Management. The VIP stays is_active=TRUE (agents keep serving it, NOTHING
|
||||
# is torn down) until the operator APPROVES; only then does apply flip is_active=FALSE
|
||||
# and the agents tear down. Reject clears the flag and the VIP keeps running untouched.
|
||||
# This guarantees an agent never tears a VIP down without an explicit human approval —
|
||||
# protecting production. Additive + idempotent.
|
||||
await conn.execute("ALTER TABLE vip_instances ADD COLUMN IF NOT EXISTS pending_delete BOOLEAN NOT NULL DEFAULT FALSE;")
|
||||
# Uniqueness as PARTIAL indexes on active rows so a soft-deleted VIP frees its
|
||||
# name/address/VRID for immediate reuse (a table-level UNIQUE would keep blocking it).
|
||||
# NAME (v1.7.0 self-review): the original CREATE used a table-level UNIQUE on name,
|
||||
# which left a soft-deleted VIP's name blocked (you couldn't re-create a VIP with the
|
||||
# same name) — inconsistent with addr/VRID. Drop that constraint and use a partial
|
||||
# index instead. Idempotent: no-op on a fresh table (no inline UNIQUE) and on re-run.
|
||||
await conn.execute("ALTER TABLE vip_instances DROP CONSTRAINT IF EXISTS vip_instances_name_key;")
|
||||
await conn.execute(
|
||||
"CREATE UNIQUE INDEX IF NOT EXISTS uq_vip_name_active ON vip_instances(name) WHERE is_active=TRUE;"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE UNIQUE INDEX IF NOT EXISTS uq_vip_addr_active ON vip_instances(virtual_ip) WHERE is_active=TRUE;"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE UNIQUE INDEX IF NOT EXISTS uq_vip_vrid_active ON vip_instances(pool_id, virtual_router_id) WHERE is_active=TRUE;"
|
||||
)
|
||||
|
||||
# Per-node membership: which agents participate + their VRRP role/priority,
|
||||
# the applied (delivered) config snapshot, and the agent's deploy ack.
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS vip_members (
|
||||
id SERIAL PRIMARY KEY,
|
||||
vip_id INTEGER NOT NULL REFERENCES vip_instances(id) ON DELETE CASCADE,
|
||||
agent_id INTEGER NOT NULL REFERENCES agents(id) ON DELETE CASCADE,
|
||||
network_interface VARCHAR(64) NOT NULL,
|
||||
role VARCHAR(10) NOT NULL DEFAULT 'BACKUP',
|
||||
priority INTEGER NOT NULL DEFAULT 100,
|
||||
applied_config_content TEXT,
|
||||
applied_config_hash VARCHAR(64),
|
||||
last_deploy_state VARCHAR(24),
|
||||
last_deploy_message TEXT,
|
||||
last_deploy_hash VARCHAR(64),
|
||||
last_deploy_at TIMESTAMP,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT vip_member_role CHECK (role IN ('MASTER','BACKUP')),
|
||||
CONSTRAINT vip_member_priority_range CHECK (priority BETWEEN 1 AND 254),
|
||||
CONSTRAINT vip_member_unique UNIQUE (vip_id, agent_id)
|
||||
);
|
||||
""")
|
||||
# Last line of defense against split-brain: at most one MASTER per VIP.
|
||||
await conn.execute(
|
||||
"CREATE UNIQUE INDEX IF NOT EXISTS uq_vip_one_master ON vip_members(vip_id) WHERE role='MASTER';"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_members_vip ON vip_members(vip_id);"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_members_agent ON vip_members(agent_id);"
|
||||
)
|
||||
|
||||
logger.info("✅ VIP tables ensured (Issue #27 — HA/VIP Keepalived management)")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to ensure VIP tables: {e}")
|
||||
# Don't raise — follow the same defensive pattern as ensure_mfa_columns
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
async def add_ssl_certificate_id_to_backend_servers():
|
||||
"""Add ssl_certificate_id column to backend_servers table for SSL certificate management"""
|
||||
conn = None
|
||||
@@ -3357,6 +3745,23 @@ async def ensure_acme_columns_on_existing_tables():
|
||||
('last_attempt_at', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS last_attempt_at TIMESTAMPTZ"),
|
||||
# Commit 3a: track auto-completion task lock/poll timestamps for atomic claim across replicas
|
||||
('orders_updated_at_idx', "CREATE INDEX IF NOT EXISTS idx_letsencrypt_orders_status_updated ON letsencrypt_orders(status, updated_at) WHERE status = 'valid' AND ssl_certificate_id IS NULL"),
|
||||
# Issue #35 (v1.8.0 — ACME DNS-01): per-account challenge method + DNS provider selection
|
||||
('acct_challenge_type', "ALTER TABLE letsencrypt_accounts ADD COLUMN IF NOT EXISTS challenge_type VARCHAR(20) DEFAULT 'http-01'"),
|
||||
('acct_dns_provider', "ALTER TABLE letsencrypt_accounts ADD COLUMN IF NOT EXISTS dns_provider VARCHAR(50)"),
|
||||
# per-order challenge method + bounded DNS-01 retry chain (dns01_parent_order_id is a PLAIN INTEGER, not a FK,
|
||||
# to avoid a self-referential cascade interacting with account/order bulk DELETEs)
|
||||
('order_challenge_type', "ALTER TABLE letsencrypt_orders ADD COLUMN IF NOT EXISTS challenge_type VARCHAR(20) DEFAULT 'http-01'"),
|
||||
('order_dns01_attempts', "ALTER TABLE letsencrypt_orders ADD COLUMN IF NOT EXISTS dns01_attempts INTEGER DEFAULT 0"),
|
||||
('order_dns01_last_attempt_at', "ALTER TABLE letsencrypt_orders ADD COLUMN IF NOT EXISTS dns01_last_attempt_at TIMESTAMPTZ"),
|
||||
('order_dns01_parent_order_id', "ALTER TABLE letsencrypt_orders ADD COLUMN IF NOT EXISTS dns01_parent_order_id INTEGER"),
|
||||
('order_dns01_retry_claimed', "ALTER TABLE letsencrypt_orders ADD COLUMN IF NOT EXISTS dns01_retry_claimed BOOLEAN DEFAULT FALSE"),
|
||||
# per-challenge DNS-01 lifecycle state
|
||||
('chal_challenge_type', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS challenge_type VARCHAR(20) DEFAULT 'http-01'"),
|
||||
('chal_dns_txt_value', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS dns_txt_value TEXT"),
|
||||
('chal_dns_record_published', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS dns_record_published BOOLEAN DEFAULT FALSE"),
|
||||
('chal_dns_record_cleaned', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS dns_record_cleaned BOOLEAN DEFAULT FALSE"),
|
||||
('chal_dns_published_at', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS dns_published_at TIMESTAMPTZ"),
|
||||
('chal_manual_confirm_deadline', "ALTER TABLE acme_challenges ADD COLUMN IF NOT EXISTS manual_confirm_deadline TIMESTAMPTZ"),
|
||||
]:
|
||||
try:
|
||||
await conn.execute(sql)
|
||||
@@ -3372,6 +3777,34 @@ async def ensure_acme_columns_on_existing_tables():
|
||||
logger.error(f"Error adding ACME columns: {e}")
|
||||
|
||||
|
||||
async def ensure_letsencrypt_dns_credentials():
|
||||
"""Issue #35 (v1.8.0 — ACME DNS-01): per-account encrypted DNS provider credentials.
|
||||
|
||||
Idempotent (CREATE TABLE IF NOT EXISTS). FK to letsencrypt_accounts (created earlier by
|
||||
ensure_acme_tables). Credentials are Fernet-encrypted at rest (backend/utils/dns_credentials.py);
|
||||
only the provider name + timestamps are ever surfaced to the API.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS letsencrypt_account_dns_credentials (
|
||||
id SERIAL PRIMARY KEY,
|
||||
account_id INTEGER NOT NULL UNIQUE REFERENCES letsencrypt_accounts(id) ON DELETE CASCADE,
|
||||
dns_provider VARCHAR(50) NOT NULL,
|
||||
credentials_encrypted TEXT NOT NULL,
|
||||
created_at TIMESTAMPTZ DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMPTZ DEFAULT CURRENT_TIMESTAMP
|
||||
)
|
||||
""")
|
||||
logger.info("Ensured letsencrypt_account_dns_credentials table")
|
||||
await close_database_connection(conn)
|
||||
except Exception as e:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
logger.error(f"Error ensuring letsencrypt_account_dns_credentials: {e}")
|
||||
|
||||
|
||||
async def cleanup_orphan_acme_challenge_backend():
|
||||
"""
|
||||
Issue #11: One-time cleanup of orphan `_acme_challenge_backend` rows that may
|
||||
|
||||
+74
-9
@@ -8,8 +8,13 @@ import redis
|
||||
import asyncio
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
_version_info = {"version": "1.6.0", "releaseName": "Multi-Factor Authentication (MFA)", "releaseDate": "2026-05-18"}
|
||||
for _vpath in ["/app/version.json", os.path.join(os.path.dirname(__file__), "..", "version.json")]:
|
||||
# Single source of truth: backend/version.json, which sits next to this module and is baked into
|
||||
# every image by `COPY . .` (build context ./backend) — no pipeline staging needed. The literal
|
||||
# below is only a last-resort "file missing" marker; it is deliberately NOT a real version so it can
|
||||
# never silently drift out of sync (this exact drift showed a stale version after v1.8.5/v1.8.6).
|
||||
# Keep the canonical version ONLY in backend/version.json — test_version_consistency.py enforces it.
|
||||
_version_info = {"version": "unknown", "releaseName": "unknown", "releaseDate": ""}
|
||||
for _vpath in [os.path.join(os.path.dirname(__file__), "version.json"), "/app/version.json"]:
|
||||
try:
|
||||
with open(_vpath) as _vf:
|
||||
_version_info = json.load(_vf)
|
||||
@@ -41,6 +46,8 @@ from routers.letsencrypt import router as letsencrypt_router
|
||||
from routers.acme_diagnostics import router as acme_diagnostics_router
|
||||
from routers.site_wizard import router as site_wizard_router
|
||||
from routers.mfa import router as mfa_router
|
||||
from routers.vip import router as vip_router # Issue #27 — HA/VIP (Keepalived) management
|
||||
from routers.csr import router as csr_router # v1.9.0 — CSR creation (in-app key+CSR generation, signed-cert import)
|
||||
|
||||
# Production logging configuration
|
||||
from utils.logging_config import setup_production_logging
|
||||
@@ -303,6 +310,21 @@ async def complete_pending_acme_orders():
|
||||
WHERE (
|
||||
status IN ('pending', 'processing', 'ready')
|
||||
OR (status = 'valid' AND ssl_certificate_id IS NULL)
|
||||
-- Issue #35: bounded DNS-01 retry. ONLY dns-01 invalids with remaining
|
||||
-- budget + elapsed backoff are claimed; http-01 invalids are NEVER matched
|
||||
-- (their existing skip-and-log is preserved).
|
||||
OR (
|
||||
status = 'invalid' AND challenge_type = 'dns-01'
|
||||
AND ssl_certificate_id IS NULL
|
||||
AND COALESCE(dns01_retry_claimed, FALSE) = FALSE
|
||||
AND COALESCE(dns01_attempts, 0) < 3
|
||||
AND (
|
||||
dns01_last_attempt_at IS NULL
|
||||
OR dns01_last_attempt_at < NOW() - (
|
||||
(CASE COALESCE(dns01_attempts, 0) WHEN 0 THEN 15 WHEN 1 THEN 30 ELSE 60 END)
|
||||
|| ' minutes')::INTERVAL
|
||||
)
|
||||
)
|
||||
)
|
||||
AND created_at > NOW() - INTERVAL '7 days'
|
||||
AND (updated_at IS NULL OR updated_at < NOW() - INTERVAL '30 seconds')
|
||||
@@ -336,9 +358,17 @@ async def complete_pending_acme_orders():
|
||||
continue
|
||||
|
||||
logger.info(f"[ACME-COMPLETE] Claimed {len(claimed_ids)} order(s) for completion: {claimed_ids}")
|
||||
|
||||
|
||||
from services.dns01_orchestrator import (
|
||||
advance_dns01_order, retry_invalid_dns01, reconcile_dns01_cleanup,
|
||||
)
|
||||
|
||||
for oid in claimed_ids:
|
||||
try:
|
||||
# Issue #35: advance the DNS-01 publish->confirm->respond state machine for
|
||||
# pending dns-01 orders (no-op for http-01 or non-pending orders).
|
||||
await advance_dns01_order(oid)
|
||||
|
||||
status_info = await acme_svc.check_order_status(oid)
|
||||
current_status = status_info.get('status')
|
||||
|
||||
@@ -351,12 +381,21 @@ async def complete_pending_acme_orders():
|
||||
result = await _complete_certificate(oid)
|
||||
logger.info(f"[ACME-COMPLETE] Order {oid} completed - {result.get('message', '')}")
|
||||
elif current_status == 'invalid':
|
||||
logger.warning(f"[ACME-COMPLETE] Order {oid} is invalid, skipping")
|
||||
# Issue #35: bounded DNS-01 fresh-order retry (no-op for http-01).
|
||||
await retry_invalid_dns01(oid)
|
||||
logger.warning(f"[ACME-COMPLETE] Order {oid} is invalid")
|
||||
elif current_status in ('pending', 'processing'):
|
||||
logger.info(f"[ACME-COMPLETE] Order {oid} still {current_status}, will retry next cycle")
|
||||
except Exception as poll_err:
|
||||
logger.error(f"[ACME-COMPLETE] Failed to complete order {oid}: {poll_err}")
|
||||
|
||||
# Issue #35: best-effort cleanup of TXT records left published on terminal orders
|
||||
# (covers a failed cleanup or the kill-switch being flipped off). NOT gated by the switch.
|
||||
try:
|
||||
await reconcile_dns01_cleanup()
|
||||
except Exception as rec_err:
|
||||
logger.debug(f"[ACME-COMPLETE] DNS-01 reconcile skipped: {rec_err}")
|
||||
|
||||
# NOTE: v1.5.0 wizard-staged processing now runs BEFORE the
|
||||
# claimed_ids early-continue above (Bulgu #2 fix), so it executes
|
||||
# every cycle regardless of pending/processing volume.
|
||||
@@ -672,7 +711,9 @@ async def check_letsencrypt_renewals():
|
||||
skip = False
|
||||
try:
|
||||
order = await conn2.fetchrow(
|
||||
"SELECT account_id, domains, cluster_ids FROM letsencrypt_orders WHERE id = $1",
|
||||
"SELECT o.account_id, o.domains, o.cluster_ids, o.challenge_type, a.dns_provider "
|
||||
"FROM letsencrypt_orders o JOIN letsencrypt_accounts a ON o.account_id = a.id "
|
||||
"WHERE o.id = $1",
|
||||
order_id
|
||||
)
|
||||
if order:
|
||||
@@ -688,15 +729,37 @@ async def check_letsencrypt_renewals():
|
||||
if existing:
|
||||
logger.info(f"[ACME-RENEWAL] Skipping cert {cert['id']} - order {existing['id']} already in progress")
|
||||
skip = True
|
||||
elif (order['challenge_type'] == 'dns-01'):
|
||||
# Issue #35: manual DNS-01 cannot auto-renew unattended; and for an
|
||||
# automated provider, don't re-mint hourly if a recent retry chain already
|
||||
# exhausted its budget (avoids tripping the CA new-order rate limit).
|
||||
if (order['dns_provider'] or 'manual') == 'manual':
|
||||
logger.warning(f"[ACME-RENEWAL] cert {cert['id']} uses manual DNS-01; cannot auto-renew unattended (publish the TXT and renew manually)")
|
||||
skip = True
|
||||
else:
|
||||
exhausted = await conn2.fetchrow("""
|
||||
SELECT id FROM letsencrypt_orders
|
||||
WHERE domains::text = $1::text AND challenge_type = 'dns-01'
|
||||
AND status = 'invalid' AND COALESCE(dns01_attempts, 0) >= 3
|
||||
AND created_at > NOW() - INTERVAL '24 hours'
|
||||
LIMIT 1
|
||||
""", json.dumps(domains))
|
||||
if exhausted:
|
||||
logger.warning(f"[ACME-RENEWAL] cert {cert['id']} DNS-01 renewal recently failed (check DNS); skipping re-mint for 24h")
|
||||
skip = True
|
||||
finally:
|
||||
await close_database_connection(conn2)
|
||||
|
||||
if not order or skip:
|
||||
continue
|
||||
|
||||
new_order = await acme_svc.create_order(order['account_id'], domains, cluster_ids)
|
||||
await acme_svc.respond_to_challenges(new_order['order_id'])
|
||||
logger.info(f"[ACME-RENEWAL] Initiated renewal order {new_order['order_id']} for cert {cert['id']} ({cert['name']})")
|
||||
challenge_type = order['challenge_type'] or 'http-01'
|
||||
new_order = await acme_svc.create_order(order['account_id'], domains, cluster_ids, challenge_type=challenge_type)
|
||||
# http-01 responds immediately (token served continuously); dns-01 is driven by the
|
||||
# orchestrator AFTER the TXT is published (never respond before publish).
|
||||
if challenge_type != 'dns-01':
|
||||
await acme_svc.respond_to_challenges(new_order['order_id'])
|
||||
logger.info(f"[ACME-RENEWAL] Initiated renewal order {new_order['order_id']} ({challenge_type}) for cert {cert['id']} ({cert['name']})")
|
||||
except Exception as cert_err:
|
||||
logger.error(f"[ACME-RENEWAL] Failed to initiate renewal for cert {cert['id']}: {cert_err}")
|
||||
|
||||
@@ -830,12 +893,14 @@ app.include_router(dashboard_stats_router) # HAProxy stats dashboard
|
||||
app.include_router(agent_router)
|
||||
app.include_router(waf_router)
|
||||
app.include_router(ssl_router)
|
||||
app.include_router(csr_router) # v1.9.0: CSR creation (in-app key+CSR generation, signed-cert import)
|
||||
app.include_router(security_router)
|
||||
app.include_router(configuration_router)
|
||||
app.include_router(settings_router)
|
||||
app.include_router(letsencrypt_router)
|
||||
app.include_router(acme_diagnostics_router) # v1.5.0 Issue #13: ACME Diagnostic Panel
|
||||
app.include_router(site_wizard_router) # v1.5.0 Issue #14: New Site Setup Wizard
|
||||
app.include_router(vip_router) # v1.7.0 Issue #27: HA/VIP (Keepalived) management
|
||||
|
||||
|
||||
# Legacy URL alias: /api/proxied-hosts/* → 308 redirect to /api/sites/*.
|
||||
@@ -1053,7 +1118,7 @@ async def serve_acme_challenge(token: str):
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
row = await conn.fetchrow(
|
||||
"SELECT key_authorization FROM acme_challenges WHERE token = $1 AND (status IN ('pending', 'processing') OR status IS NULL) LIMIT 1",
|
||||
"SELECT key_authorization FROM acme_challenges WHERE token = $1 AND (status IN ('pending', 'processing') OR status IS NULL) AND (challenge_type = 'http-01' OR challenge_type IS NULL) LIMIT 1",
|
||||
token
|
||||
)
|
||||
if row:
|
||||
|
||||
@@ -8,6 +8,7 @@ class HAProxyClusterCreate(BaseModel):
|
||||
stats_socket_path: str = "/run/haproxy/admin.sock"
|
||||
haproxy_config_path: str = "/etc/haproxy/haproxy.cfg"
|
||||
haproxy_bin_path: str = "/usr/sbin/haproxy" # HAProxy binary path
|
||||
keepalived_config_path: str = "/etc/keepalived/keepalived.conf" # HA/VIP: keepalived.conf path (Issue #27)
|
||||
pool_id: Optional[int] = None # Which pool this cluster belongs to
|
||||
|
||||
class HAProxyClusterUpdate(BaseModel):
|
||||
@@ -17,6 +18,7 @@ class HAProxyClusterUpdate(BaseModel):
|
||||
stats_socket_path: Optional[str] = None
|
||||
haproxy_config_path: Optional[str] = None
|
||||
haproxy_bin_path: Optional[str] = None
|
||||
keepalived_config_path: Optional[str] = None
|
||||
pool_id: Optional[int] = None
|
||||
is_active: Optional[bool] = None
|
||||
acme_enabled: Optional[bool] = None
|
||||
|
||||
@@ -0,0 +1,251 @@
|
||||
"""
|
||||
Pydantic models for the CSR (Certificate Signing Request) feature (v1.9.0).
|
||||
|
||||
A CSR row is the precursor of an ssl_certificates row: the backend generates
|
||||
the private key + CSR locally, the operator has the CSR signed by an external
|
||||
CA and then imports the signed certificate. The CSR `name` therefore obeys the
|
||||
exact same path-traversal contract as the SSL certificate name (Bulgu #21) —
|
||||
at import time it becomes /etc/ssl/haproxy/{name}.pem on every agent and is
|
||||
shell-processed by the agent script as root.
|
||||
|
||||
The import model deliberately has NO private key field: the key never leaves
|
||||
the server. It is stored on the ssl_csrs row at generation time and paired
|
||||
with the signed certificate server-side.
|
||||
"""
|
||||
|
||||
import re
|
||||
from typing import List, Optional
|
||||
|
||||
from pydantic import BaseModel, field_validator, model_validator
|
||||
|
||||
KEY_ALGORITHMS = ('rsa-2048', 'rsa-4096', 'ecdsa-p256', 'ecdsa-p384')
|
||||
|
||||
# RFC 1035 LDH hostname, lowercase, optional single leftmost wildcard label.
|
||||
# Single-label names are allowed (internal CAs routinely sign bare hostnames).
|
||||
_DNS_NAME_PATTERN = re.compile(
|
||||
r'^(\*\.)?[a-z0-9]([a-z0-9-]{0,61}[a-z0-9])?'
|
||||
r'(\.[a-z0-9]([a-z0-9-]{0,61}[a-z0-9])?)*$'
|
||||
)
|
||||
|
||||
# Reject control characters in free-text subject fields: they would be
|
||||
# persisted, echoed into the UI / issuer column, and printed into agent logs
|
||||
# via `openssl -subject` output.
|
||||
_CONTROL_CHARS_PATTERN = re.compile(r'[\x00-\x1f\x7f]')
|
||||
|
||||
_MAX_SANS = 100
|
||||
_MAX_CERT_PEM_BYTES = 64 * 1024 # a leaf certificate is ~2 KB; 64 KB is generous
|
||||
_MAX_CHAIN_PEM_BYTES = 256 * 1024 # agents re-download all cert content every poll
|
||||
|
||||
|
||||
def _validate_dns_name(value: str, field_label: str) -> str:
|
||||
v = (value or '').strip().lower()
|
||||
if not v:
|
||||
raise ValueError(f'{field_label} must not be empty')
|
||||
if len(v) > 253:
|
||||
raise ValueError(f'{field_label} must be 253 characters or fewer')
|
||||
if not _DNS_NAME_PATTERN.match(v):
|
||||
raise ValueError(
|
||||
f'{field_label} {value!r} is not a valid DNS name — lowercase '
|
||||
'letters, digits, hyphens and dots only; a wildcard is allowed '
|
||||
'only as the leftmost label (e.g. *.example.com).'
|
||||
)
|
||||
return v
|
||||
|
||||
|
||||
def _validate_subject_text(value: Optional[str], field_label: str, max_len: int = 64) -> Optional[str]:
|
||||
if value is None:
|
||||
return None
|
||||
v = value.strip()
|
||||
if not v:
|
||||
return None
|
||||
if len(v) > max_len:
|
||||
raise ValueError(f'{field_label} must be {max_len} characters or fewer')
|
||||
if _CONTROL_CHARS_PATTERN.search(v):
|
||||
raise ValueError(f'{field_label} must not contain control characters')
|
||||
return v
|
||||
|
||||
|
||||
def _validate_csr_name(v: str) -> str:
|
||||
"""Mirror of SSLCertificateCreate.validate_name_no_path_traversal (Bulgu #21)
|
||||
with one deliberate tightening: max length 100, matching the
|
||||
ssl_certificates.name VARCHAR(100) column (the historical 200-char limit
|
||||
overflows the column and 500s — not replicated here)."""
|
||||
if v is None:
|
||||
raise ValueError('CSR name is required')
|
||||
stripped = v.strip()
|
||||
if not stripped:
|
||||
raise ValueError('CSR name must not be empty')
|
||||
if stripped != v:
|
||||
raise ValueError('CSR name must not contain leading/trailing whitespace')
|
||||
if len(stripped) > 100:
|
||||
raise ValueError('CSR name must be 100 characters or fewer')
|
||||
if not re.match(r'^[A-Za-z0-9_.-]+$', stripped):
|
||||
raise ValueError(
|
||||
f'CSR name={v!r} contains forbidden characters — only letters, '
|
||||
'digits, underscore, hyphen, and dot are allowed (the name becomes '
|
||||
'a filename component under /etc/ssl/haproxy/ at import).'
|
||||
)
|
||||
if '..' in stripped:
|
||||
raise ValueError(f'CSR name={v!r} must not contain ".." (path traversal)')
|
||||
if stripped.startswith('.'):
|
||||
raise ValueError(f'CSR name={v!r} must not start with "." (hidden filename)')
|
||||
if stripped.startswith('-'):
|
||||
raise ValueError(f'CSR name={v!r} must not start with "-" (CLI flag confusion)')
|
||||
return stripped
|
||||
|
||||
|
||||
class SSLCSRCreate(BaseModel):
|
||||
name: str # becomes the certificate name at import
|
||||
common_name: str
|
||||
organization: Optional[str] = None # O
|
||||
organizational_unit: Optional[str] = None # OU
|
||||
locality: Optional[str] = None # L
|
||||
state: Optional[str] = None # ST
|
||||
country: Optional[str] = None # C — exactly 2 letters
|
||||
email: Optional[str] = None # emailAddress
|
||||
sans: List[str] = [] # DNS names; CN is auto-added server-side
|
||||
key_algorithm: str = 'rsa-2048'
|
||||
|
||||
@field_validator('name')
|
||||
@classmethod
|
||||
def validate_name(cls, v):
|
||||
return _validate_csr_name(v)
|
||||
|
||||
@field_validator('common_name')
|
||||
@classmethod
|
||||
def validate_common_name(cls, v):
|
||||
v = _validate_dns_name(v, 'Common Name')
|
||||
# RFC 5280 ub-common-name — many CAs reject CNs longer than 64 chars.
|
||||
if len(v) > 64:
|
||||
raise ValueError(
|
||||
'Common Name must be 64 characters or fewer (RFC 5280 upper '
|
||||
'bound) — put longer names in the SAN list instead.'
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator('sans')
|
||||
@classmethod
|
||||
def validate_sans(cls, v):
|
||||
if not v:
|
||||
return []
|
||||
if len(v) > _MAX_SANS:
|
||||
raise ValueError(f'At most {_MAX_SANS} SAN entries are allowed')
|
||||
seen = set()
|
||||
result = []
|
||||
for entry in v:
|
||||
normalised = _validate_dns_name(entry, 'SAN entry')
|
||||
if normalised not in seen:
|
||||
seen.add(normalised)
|
||||
result.append(normalised)
|
||||
return result
|
||||
|
||||
@field_validator('organization')
|
||||
@classmethod
|
||||
def validate_organization(cls, v):
|
||||
return _validate_subject_text(v, 'Organization (O)')
|
||||
|
||||
@field_validator('organizational_unit')
|
||||
@classmethod
|
||||
def validate_organizational_unit(cls, v):
|
||||
return _validate_subject_text(v, 'Organizational Unit (OU)')
|
||||
|
||||
@field_validator('locality')
|
||||
@classmethod
|
||||
def validate_locality(cls, v):
|
||||
return _validate_subject_text(v, 'Locality (L)')
|
||||
|
||||
@field_validator('state')
|
||||
@classmethod
|
||||
def validate_state(cls, v):
|
||||
return _validate_subject_text(v, 'State/Province (ST)')
|
||||
|
||||
@field_validator('country')
|
||||
@classmethod
|
||||
def validate_country(cls, v):
|
||||
# cryptography raises a bare ValueError for a non-2-char COUNTRY_NAME;
|
||||
# pre-validate so the operator gets a friendly 422 instead of a 500.
|
||||
if v is None:
|
||||
return None
|
||||
v = v.strip()
|
||||
if not v:
|
||||
return None
|
||||
if not re.match(r'^[A-Za-z]{2}$', v):
|
||||
raise ValueError('Country (C) must be exactly 2 letters (ISO 3166-1 alpha-2, e.g. TR, US)')
|
||||
return v.upper()
|
||||
|
||||
@field_validator('email')
|
||||
@classmethod
|
||||
def validate_email(cls, v):
|
||||
v = _validate_subject_text(v, 'Email', max_len=254)
|
||||
if v is not None and ('@' not in v or v.startswith('@') or v.endswith('@')):
|
||||
raise ValueError('Email must be a valid address (missing or misplaced "@")')
|
||||
return v
|
||||
|
||||
@field_validator('key_algorithm')
|
||||
@classmethod
|
||||
def validate_key_algorithm(cls, v):
|
||||
if v not in KEY_ALGORITHMS:
|
||||
raise ValueError(
|
||||
f'key_algorithm must be one of: {", ".join(KEY_ALGORITHMS)}'
|
||||
)
|
||||
return v
|
||||
|
||||
|
||||
class SSLCSRImport(BaseModel):
|
||||
"""Import the CA-signed certificate for a pending CSR. The private key is
|
||||
NOT part of the request — it is already stored on the CSR row."""
|
||||
certificate_content: str # PEM
|
||||
chain_content: Optional[str] = None # PEM, optional
|
||||
usage_type: str = 'frontend' # "frontend" or "server"
|
||||
is_global: bool = False
|
||||
cluster_ids: Optional[List[int]] = None
|
||||
# Escape hatch for name collisions that appeared AFTER the CSR was
|
||||
# created: overrides the CSR's reserved name for the certificate row.
|
||||
name: Optional[str] = None
|
||||
|
||||
@field_validator('certificate_content')
|
||||
@classmethod
|
||||
def validate_certificate(cls, v):
|
||||
if not v or not v.strip():
|
||||
raise ValueError('Certificate content is required')
|
||||
v = v.strip()
|
||||
if len(v.encode('utf-8', errors='ignore')) > _MAX_CERT_PEM_BYTES:
|
||||
raise ValueError('Certificate content exceeds the 64 KB limit')
|
||||
if '-----BEGIN CERTIFICATE-----' not in v or '-----END CERTIFICATE-----' not in v:
|
||||
raise ValueError('Certificate must be in PEM format')
|
||||
return v
|
||||
|
||||
@field_validator('chain_content')
|
||||
@classmethod
|
||||
def validate_chain(cls, v):
|
||||
if v and v.strip():
|
||||
v = v.strip()
|
||||
if len(v.encode('utf-8', errors='ignore')) > _MAX_CHAIN_PEM_BYTES:
|
||||
raise ValueError('Certificate chain exceeds the 256 KB limit')
|
||||
if '-----BEGIN CERTIFICATE-----' not in v or '-----END CERTIFICATE-----' not in v:
|
||||
raise ValueError('Certificate chain must be in PEM format')
|
||||
return v
|
||||
return None
|
||||
|
||||
@field_validator('usage_type')
|
||||
@classmethod
|
||||
def validate_usage_type(cls, v):
|
||||
if v not in ['frontend', 'server']:
|
||||
raise ValueError('usage_type must be either "frontend" or "server"')
|
||||
return v
|
||||
|
||||
@field_validator('name')
|
||||
@classmethod
|
||||
def validate_name(cls, v):
|
||||
if v is None or not str(v).strip():
|
||||
return None
|
||||
return _validate_csr_name(v)
|
||||
|
||||
@model_validator(mode='after')
|
||||
def validate_cluster_selection(self):
|
||||
if not self.is_global and not self.cluster_ids:
|
||||
raise ValueError(
|
||||
'cluster_ids is required when is_global is false — pick at '
|
||||
'least one cluster or import the certificate as global.'
|
||||
)
|
||||
return self
|
||||
+28
-45
@@ -85,6 +85,11 @@ class FrontendConfig(BaseModel):
|
||||
response_headers: Optional[str] = None
|
||||
options: Optional[str] = None
|
||||
tcp_request_rules: Optional[str] = None
|
||||
# Issue #38: SPOE filter directives (Coraza WAF etc.) + frontend log-format.
|
||||
# Passthrough TEXT (no validator) — SPOE `filter ... config <path>` legitimately
|
||||
# references an operator-managed file, so the ACL `-f` guard must NOT apply here.
|
||||
log_format: Optional[str] = None
|
||||
filters: Optional[str] = None
|
||||
timeout_client: Optional[int] = None
|
||||
timeout_http_request: Optional[int] = None
|
||||
rate_limit: Optional[int] = None
|
||||
@@ -451,26 +456,17 @@ class FrontendConfig(BaseModel):
|
||||
if any(dangerous in rule.lower() for dangerous in ['$(', '`']):
|
||||
raise ValueError(f'ACL rule contains potentially dangerous content: "{rule}"')
|
||||
|
||||
# Phase K Phase D follow-up (Bulgu #12 round 3) — reject
|
||||
# the HAProxy `-f <file>` pattern-file flag here too so the
|
||||
# manual Frontend API mirrors the wizard's parity rule.
|
||||
# HAProxy OpenManager does not provision pattern files
|
||||
# onto the HAProxy node filesystem, so any `-f /path/...`
|
||||
# reference will fail HAProxy's `-c` parse at apply time
|
||||
# with "failed to open pattern file". Reject up-front so
|
||||
# operators get the same actionable error from both the
|
||||
# manual page and the wizard.
|
||||
if re.search(r"(^|\s)-f(\s|$)", rule):
|
||||
raise ValueError(
|
||||
f'ACL rule "{rule}" uses the HAProxy `-f <file>` '
|
||||
"pattern-file flag, which is not supported in "
|
||||
"HAProxy OpenManager: the product does not "
|
||||
"provision pattern files onto the HAProxy node "
|
||||
"filesystem, so the reference would fail at "
|
||||
"reload time. Use inline values instead "
|
||||
"(e.g. `src 10.0.0.0/24` rather than "
|
||||
"`src -f /etc/haproxy/admins.lst`)."
|
||||
)
|
||||
# Issue #38 follow-up — the `-f <file>` pattern-file flag
|
||||
# is ACCEPTED here (the Bulgu #12 hard reject was removed).
|
||||
# Pattern files are operator-managed host files, exactly
|
||||
# like the SPOE `filter ... config <path>` reference this
|
||||
# release started preserving: bulk import always accepted
|
||||
# `-f`, the free-form fields (request_headers,
|
||||
# tcp_request_rules) always accepted it, and the agent
|
||||
# runs `haproxy -c` before every reload so a missing file
|
||||
# fails safely (previous config keeps running). The route
|
||||
# handlers surface a non-blocking warning listing the
|
||||
# referenced pattern files instead.
|
||||
|
||||
validated_rules.append(rule)
|
||||
|
||||
@@ -510,20 +506,12 @@ class FrontendConfig(BaseModel):
|
||||
if not any(rule.startswith(redirect_type) for redirect_type in valid_redirects):
|
||||
raise ValueError(f'Invalid redirect rule: "{rule}". Must start with: location, prefix, or scheme.')
|
||||
|
||||
# Phase K Phase D follow-up (Bulgu #12 round 3) —
|
||||
# mirror the wizard's `-f <file>` guard here. The
|
||||
# `X !X` contradiction check used to live alongside
|
||||
# this guard, but Bulgu #62 (round-22 audit) moved
|
||||
# it into the route handler so updates can grandfather
|
||||
# legacy rules created before the contradiction guard
|
||||
# landed. See `routers/frontend.py::_collect_routing_rule_contradictions`.
|
||||
if re.search(r"(^|\s)-f(\s|$)", rule):
|
||||
raise ValueError(
|
||||
f'Redirect rule "{rule}" uses the HAProxy `-f <file>` '
|
||||
"pattern-file flag, which is not supported in "
|
||||
"HAProxy OpenManager: the product does not provision "
|
||||
"pattern files onto the HAProxy node filesystem."
|
||||
)
|
||||
# Issue #38 follow-up — `-f <file>` pattern-file references
|
||||
# are ACCEPTED (Bulgu #12 hard reject removed; see
|
||||
# validate_acl_rules for the full rationale). The `X !X`
|
||||
# contradiction check lives in the route handler
|
||||
# (`routers/frontend.py::_collect_routing_rule_contradictions`,
|
||||
# Bulgu #62) and is unchanged.
|
||||
|
||||
validated_rules.append(rule)
|
||||
|
||||
@@ -531,9 +519,12 @@ class FrontendConfig(BaseModel):
|
||||
|
||||
@validator('use_backend_rules')
|
||||
def validate_use_backend_rules_syntax(cls, v):
|
||||
"""Phase K Phase D follow-up (Bulgu #12 round 3) — manual
|
||||
Frontend API parity guard: reject `-f <file>` references
|
||||
and dangerous shell patterns.
|
||||
"""Manual Frontend API guard for dangerous shell patterns.
|
||||
|
||||
Issue #38 follow-up — the Bulgu #12 `-f <file>` hard reject
|
||||
was removed (see validate_acl_rules for the rationale);
|
||||
pattern-file references are operator-managed host files and
|
||||
are surfaced as non-blocking warnings by the route handlers.
|
||||
|
||||
Bulgu #62 (round-22 audit) — the `X !X` contradiction check
|
||||
previously lived here but moved into the route handler so
|
||||
@@ -563,13 +554,5 @@ class FrontendConfig(BaseModel):
|
||||
f'use_backend rule contains potentially dangerous '
|
||||
f'content: "{rule}"'
|
||||
)
|
||||
if re.search(r"(^|\s)-f(\s|$)", rule):
|
||||
raise ValueError(
|
||||
f'use_backend rule "{rule}" uses the HAProxy '
|
||||
"`-f <file>` pattern-file flag, which is not "
|
||||
"supported in HAProxy OpenManager: the product "
|
||||
"does not provision pattern files onto the HAProxy "
|
||||
"node filesystem."
|
||||
)
|
||||
validated_rules.append(rule)
|
||||
return validated_rules
|
||||
@@ -179,35 +179,17 @@ _MAX_RULE_STRING_LEN = 4096
|
||||
# attempts.
|
||||
_DANGEROUS_RULE_PATTERNS = ("$(", "`")
|
||||
|
||||
# Phase K Phase D follow-up (Bulgu #12 round 3) — the HAProxy `-f
|
||||
# <file>` ACL/condition flag instructs HAProxy to load match patterns
|
||||
# from a server-side file at parse time. HAProxy OpenManager is a
|
||||
# fully-managed product: we do NOT provision pattern files onto the
|
||||
# HAProxy node's filesystem, and operators have no UI to upload one.
|
||||
# A `-f /some/path` reference therefore ALWAYS resolves to
|
||||
# "file not found" when HAProxy's real `-c` parse runs at apply
|
||||
# time, producing exactly the operator-reported failure mode:
|
||||
# [ALERT] parsing ACL 'acl1' : failed to open pattern file </path>.
|
||||
# [ALERT] parsing switching rule : no such ACL : 'acl1'.
|
||||
#
|
||||
# Surface this BEFORE persist by rejecting `-f` in any rule string
|
||||
# that comes through the wizard / manual frontend API. Reject ALL
|
||||
# variants (` -f `, leading `-f `, trailing `... -f`) defensively so
|
||||
# operators cannot slip the flag through with creative spacing.
|
||||
# The check is anchored to ACL/condition rule strings only; raw
|
||||
# HAProxy snippet fields (tcp_request_rules, request_headers, ...)
|
||||
# are NOT touched because those are inherently free-form and
|
||||
# advanced operators may legitimately reference pre-provisioned
|
||||
# pattern files there.
|
||||
_ACL_FILE_FLAG_PATTERN = re.compile(r"(^|\s)-f(\s|$)")
|
||||
_ACL_FILE_FLAG_MESSAGE = (
|
||||
"pattern-file references with '-f <file>' are not supported in ACL / "
|
||||
"use_backend / redirect rules: HAProxy OpenManager does not provision "
|
||||
"pattern files onto the HAProxy node's filesystem, so the reference "
|
||||
"would always fail at HAProxy reload time. Use inline values "
|
||||
"instead (e.g. `acl is_admin src 10.0.0.0/24` rather than "
|
||||
"`acl is_admin src -f /etc/haproxy/admins.lst`)."
|
||||
)
|
||||
# Issue #38 follow-up — the HAProxy `-f <file>` ACL/condition flag
|
||||
# loads match patterns from a file on the HAProxy host. The Bulgu #12
|
||||
# hard reject (`_ACL_FILE_FLAG_PATTERN`/`_ACL_FILE_FLAG_MESSAGE`) was
|
||||
# removed: pattern files are operator-managed host files (exactly like
|
||||
# the SPOE `filter ... config <path>` reference preserved since
|
||||
# v1.8.8), bulk import and the free-form fields (tcp_request_rules,
|
||||
# request_headers) always accepted them, and the agent runs
|
||||
# `haproxy -c` before every reload so a missing file fails safely
|
||||
# (the previous config keeps running). The manual frontend route
|
||||
# handlers emit a non-blocking warning listing referenced pattern
|
||||
# files (`routers/frontend.py::_pattern_file_warnings`).
|
||||
|
||||
# Phase K Phase D follow-up (Bulgu #13) — detect a routing /
|
||||
# redirect rule whose condition references the SAME ACL in both
|
||||
@@ -305,13 +287,10 @@ def _validate_haproxy_directive_string(
|
||||
f"{field_label} entry contains potentially dangerous content: "
|
||||
f"{pattern!r}"
|
||||
)
|
||||
# Phase K Phase D follow-up (Bulgu #12 round 3) — reject the
|
||||
# HAProxy `-f <file>` pattern-file flag because OpenManager does
|
||||
# not manage the HAProxy node filesystem. See the module-level
|
||||
# `_ACL_FILE_FLAG_PATTERN` docstring for the full operator-
|
||||
# reported failure mode this guards against.
|
||||
if _ACL_FILE_FLAG_PATTERN.search(stripped):
|
||||
raise ValueError(f"{field_label}: {_ACL_FILE_FLAG_MESSAGE}")
|
||||
# Issue #38 follow-up — `-f <file>` pattern-file references are
|
||||
# ACCEPTED (Bulgu #12 hard reject removed; see the module-level
|
||||
# `_ACL_FILE_FLAG_PATTERN` comment). The route handlers surface
|
||||
# a non-blocking pattern-file warning instead.
|
||||
# Phase K Phase D follow-up (Bulgu #13) — for routing /
|
||||
# redirect rules (not ACL definitions themselves), reject a
|
||||
# condition that contains the same ACL in both positive and
|
||||
@@ -1083,21 +1062,12 @@ class FrontendStep(BaseModel):
|
||||
normalised: List[Union[str, dict]] = []
|
||||
for el in v:
|
||||
if isinstance(el, dict):
|
||||
# Phase K Phase D follow-up (Bulgu #12 round 3
|
||||
# extension) — dict-shaped redirect rules emit their
|
||||
# `condition` / `target` fields VERBATIM into the
|
||||
# rendered HAProxy directive. A dict with
|
||||
# `condition: "if { src -f /etc/haproxy/x.lst }"`
|
||||
# would slip past the string-only validator above
|
||||
# and trigger the same operator-reported "failed to
|
||||
# open pattern file" rejection at apply time. Reject
|
||||
# `-f` in any string-shaped value the dict carries.
|
||||
for field_name in ("condition", "target", "type"):
|
||||
val = el.get(field_name)
|
||||
if isinstance(val, str) and _ACL_FILE_FLAG_PATTERN.search(val):
|
||||
raise ValueError(
|
||||
f"redirect_rules.{field_name}: {_ACL_FILE_FLAG_MESSAGE}"
|
||||
)
|
||||
# Issue #38 follow-up — dict-shaped redirect rules may
|
||||
# carry `-f <file>` pattern-file references in their
|
||||
# `condition`/`target` values; these are ACCEPTED now
|
||||
# (Bulgu #12 hard reject removed — operator-managed
|
||||
# host files, fail-safe apply; see module-level
|
||||
# `_ACL_FILE_FLAG_PATTERN` comment).
|
||||
# Bulgu #13 extension — same contradiction guard
|
||||
# for dict-shaped redirect conditions.
|
||||
cond_val = el.get("condition")
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Issue #27 — HA/VIP (Keepalived) management (v1.7.0).
|
||||
|
||||
Pydantic request/response models for the VIP management API. Field validation
|
||||
is strict because several values flow into a generated keepalived.conf and into
|
||||
root-run agent commands — we reuse the same FORBIDDEN-metacharacter discipline as
|
||||
models/agent.py (Bulgu #81) and validate the VIP as a real IPv4 address.
|
||||
|
||||
Pydantic idiom: the project runs pydantic>=2.5; this module uses the v2-native
|
||||
@field_validator/@model_validator style (matching models/ssl.py).
|
||||
"""
|
||||
import ipaddress
|
||||
from typing import List, Optional
|
||||
|
||||
from pydantic import BaseModel, field_validator, model_validator
|
||||
|
||||
# Shell/keepalived.conf metacharacters that must never appear in a value that
|
||||
# reaches the generated config or a root-run agent command (mirrors
|
||||
# models/agent.py:202, the Bulgu #81 convention).
|
||||
_FORBIDDEN = set('$`;&|<>"\'\\\n\r\x00*?')
|
||||
|
||||
|
||||
def _validate_iface(v: str) -> str:
|
||||
if not isinstance(v, str) or not v.strip():
|
||||
raise ValueError('network_interface must be a non-empty string')
|
||||
s = v.strip()
|
||||
if len(s) > 64:
|
||||
raise ValueError('network_interface too long (max 64 chars)')
|
||||
if any(c in _FORBIDDEN for c in s):
|
||||
raise ValueError('network_interface contains a forbidden character')
|
||||
# Linux iface names: letters, digits, and . _ - : @ (vlans/aliases/altnames)
|
||||
import re as _re
|
||||
if not _re.match(r'^[A-Za-z0-9][A-Za-z0-9._:@-]{0,63}$', s):
|
||||
raise ValueError(
|
||||
'network_interface must start alphanumeric and contain only '
|
||||
'letters, digits, and . _ - : @'
|
||||
)
|
||||
return s
|
||||
|
||||
|
||||
def _validate_ipv4(v: str) -> str:
|
||||
if not isinstance(v, str) or not v.strip():
|
||||
raise ValueError('virtual_ip must be a non-empty string')
|
||||
s = v.strip()
|
||||
try:
|
||||
addr = ipaddress.ip_address(s)
|
||||
except ValueError:
|
||||
raise ValueError(f'virtual_ip={v!r} is not a valid IP address')
|
||||
if addr.version != 4:
|
||||
raise ValueError('virtual_ip must be IPv4 — IPv6 VIPs are not supported yet')
|
||||
return s
|
||||
|
||||
|
||||
class VIPMemberIn(BaseModel):
|
||||
agent_id: int
|
||||
network_interface: str
|
||||
role: str = 'BACKUP'
|
||||
priority: int = 100
|
||||
|
||||
@field_validator('network_interface')
|
||||
@classmethod
|
||||
def _iface(cls, v):
|
||||
return _validate_iface(v)
|
||||
|
||||
@field_validator('role')
|
||||
@classmethod
|
||||
def _role(cls, v):
|
||||
u = (v or '').strip().upper()
|
||||
if u not in ('MASTER', 'BACKUP'):
|
||||
raise ValueError("role must be 'MASTER' or 'BACKUP'")
|
||||
return u
|
||||
|
||||
@field_validator('priority')
|
||||
@classmethod
|
||||
def _priority(cls, v):
|
||||
if not isinstance(v, int) or not (1 <= v <= 254):
|
||||
raise ValueError('priority must be an integer between 1 and 254')
|
||||
return v
|
||||
|
||||
|
||||
def _validate_members(members: List['VIPMemberIn']) -> List['VIPMemberIn']:
|
||||
# >=1 node: a single-node VIP is a keepalived-managed floating IP without failover
|
||||
# (valid, e.g. a one-box HAProxy that wants a stable VIP, or before a 2nd node is added).
|
||||
# Two or more nodes give actual VRRP failover. The UI flags the single-node case.
|
||||
if not members:
|
||||
raise ValueError('a VIP needs at least 1 member node')
|
||||
agent_ids = [m.agent_id for m in members]
|
||||
if len(set(agent_ids)) != len(agent_ids):
|
||||
raise ValueError('each node may appear at most once in a VIP')
|
||||
masters = [m for m in members if m.role == 'MASTER']
|
||||
if len(masters) != 1:
|
||||
raise ValueError('exactly one member must be MASTER')
|
||||
master_prio = masters[0].priority
|
||||
if any(m.role == 'BACKUP' and m.priority >= master_prio for m in members):
|
||||
raise ValueError('the MASTER must have a strictly higher priority than every BACKUP')
|
||||
return members
|
||||
|
||||
|
||||
def _validate_auth_pass(v: Optional[str]) -> Optional[str]:
|
||||
if v is None or v == '':
|
||||
return None
|
||||
# keepalived PASS auth_pass is silently truncated to 8 chars (B-3) — reject longer
|
||||
# so MASTER/BACKUP never silently disagree.
|
||||
if not (1 <= len(v) <= 8):
|
||||
raise ValueError('auth_pass must be 1-8 characters (keepalived PASS limit)')
|
||||
if any(c in _FORBIDDEN for c in v):
|
||||
raise ValueError('auth_pass contains a forbidden character')
|
||||
# No whitespace: keepalived PASS auth_pass is a single token, and a whitespace-containing
|
||||
# secret would only partially redact in the masked config diff (review HIGH-2).
|
||||
if any(c.isspace() for c in v):
|
||||
raise ValueError('auth_pass must not contain whitespace')
|
||||
return v
|
||||
|
||||
|
||||
class VIPCreate(BaseModel):
|
||||
name: str
|
||||
description: Optional[str] = None
|
||||
pool_id: int
|
||||
virtual_ip: str
|
||||
prefix_length: int = 24
|
||||
virtual_router_id: Optional[int] = None # auto-allocated within the pool when omitted
|
||||
advert_int: int = 1
|
||||
auth_pass: Optional[str] = None # plaintext on the wire; stored Fernet-encrypted
|
||||
use_unicast: bool = True
|
||||
track_haproxy: bool = True
|
||||
members: List[VIPMemberIn]
|
||||
|
||||
@field_validator('name')
|
||||
@classmethod
|
||||
def _name(cls, v):
|
||||
if not isinstance(v, str) or not v.strip():
|
||||
raise ValueError('name must be a non-empty string')
|
||||
s = v.strip()
|
||||
if len(s) > 255:
|
||||
raise ValueError('name too long (max 255 chars)')
|
||||
if any(c in _FORBIDDEN for c in s):
|
||||
raise ValueError('name contains a forbidden character')
|
||||
return s
|
||||
|
||||
@field_validator('virtual_ip')
|
||||
@classmethod
|
||||
def _vip(cls, v):
|
||||
return _validate_ipv4(v)
|
||||
|
||||
@field_validator('prefix_length')
|
||||
@classmethod
|
||||
def _prefix(cls, v):
|
||||
if not isinstance(v, int) or not (1 <= v <= 32):
|
||||
raise ValueError('prefix_length must be an integer between 1 and 32 (IPv4)')
|
||||
return v
|
||||
|
||||
@field_validator('virtual_router_id')
|
||||
@classmethod
|
||||
def _vrid(cls, v):
|
||||
if v is None:
|
||||
return v
|
||||
if not isinstance(v, int) or not (1 <= v <= 255):
|
||||
raise ValueError('virtual_router_id must be an integer between 1 and 255')
|
||||
return v
|
||||
|
||||
@field_validator('advert_int')
|
||||
@classmethod
|
||||
def _advert(cls, v):
|
||||
if not isinstance(v, int) or not (1 <= v <= 255):
|
||||
raise ValueError('advert_int must be an integer between 1 and 255 (seconds)')
|
||||
return v
|
||||
|
||||
@field_validator('auth_pass')
|
||||
@classmethod
|
||||
def _auth(cls, v):
|
||||
return _validate_auth_pass(v)
|
||||
|
||||
@model_validator(mode='after')
|
||||
def _members_consistent(self):
|
||||
_validate_members(self.members)
|
||||
return self
|
||||
|
||||
|
||||
class VIPUpdate(BaseModel):
|
||||
"""All fields optional — only provided fields are changed. Any change sets the
|
||||
VIP back to last_config_status='PENDING' (router-side)."""
|
||||
name: Optional[str] = None
|
||||
description: Optional[str] = None
|
||||
virtual_ip: Optional[str] = None
|
||||
prefix_length: Optional[int] = None
|
||||
virtual_router_id: Optional[int] = None
|
||||
advert_int: Optional[int] = None
|
||||
auth_pass: Optional[str] = None # provide only to rotate; omit to keep existing
|
||||
use_unicast: Optional[bool] = None
|
||||
track_haproxy: Optional[bool] = None
|
||||
members: Optional[List[VIPMemberIn]] = None
|
||||
|
||||
@field_validator('name')
|
||||
@classmethod
|
||||
def _name(cls, v):
|
||||
if v is None:
|
||||
return v
|
||||
if not v.strip():
|
||||
raise ValueError('name must be a non-empty string')
|
||||
s = v.strip()
|
||||
if len(s) > 255 or any(c in _FORBIDDEN for c in s):
|
||||
raise ValueError('name invalid (too long or forbidden character)')
|
||||
return s
|
||||
|
||||
@field_validator('virtual_ip')
|
||||
@classmethod
|
||||
def _vip(cls, v):
|
||||
return _validate_ipv4(v) if v is not None else v
|
||||
|
||||
@field_validator('prefix_length')
|
||||
@classmethod
|
||||
def _prefix(cls, v):
|
||||
if v is None:
|
||||
return v
|
||||
if not (1 <= v <= 32):
|
||||
raise ValueError('prefix_length must be 1-32 (IPv4)')
|
||||
return v
|
||||
|
||||
@field_validator('virtual_router_id')
|
||||
@classmethod
|
||||
def _vrid(cls, v):
|
||||
if v is None:
|
||||
return v
|
||||
if not (1 <= v <= 255):
|
||||
raise ValueError('virtual_router_id must be 1-255')
|
||||
return v
|
||||
|
||||
@field_validator('advert_int')
|
||||
@classmethod
|
||||
def _advert(cls, v):
|
||||
if v is None:
|
||||
return v
|
||||
if not (1 <= v <= 255):
|
||||
raise ValueError('advert_int must be 1-255 seconds')
|
||||
return v
|
||||
|
||||
@field_validator('auth_pass')
|
||||
@classmethod
|
||||
def _auth(cls, v):
|
||||
return _validate_auth_pass(v)
|
||||
|
||||
@model_validator(mode='after')
|
||||
def _members_consistent(self):
|
||||
if self.members is not None:
|
||||
_validate_members(self.members)
|
||||
return self
|
||||
@@ -79,7 +79,7 @@ async def _load_order(conn, order_id: int) -> dict:
|
||||
"""
|
||||
SELECT id, account_id, status, domains, cluster_ids, error_detail,
|
||||
post_completion_actions, pending_apply_version_name,
|
||||
wizard_staged_until, created_by
|
||||
wizard_staged_until, created_by, challenge_type
|
||||
FROM letsencrypt_orders
|
||||
WHERE id = $1
|
||||
""",
|
||||
@@ -220,6 +220,7 @@ async def run_diagnostics(order_id: int, authorization: str = Header(None)):
|
||||
domains=domains,
|
||||
cluster_ids=cluster_ids,
|
||||
account_id=order["account_id"],
|
||||
challenge_type=(order.get("challenge_type") or "http-01"),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — diagnostic boundary
|
||||
# run_checks now wraps individual checks, but a top-level
|
||||
@@ -335,6 +336,7 @@ async def rerun_diagnostic_check(
|
||||
cluster_ids=cluster_ids,
|
||||
account_id=order["account_id"],
|
||||
only=[check_id],
|
||||
challenge_type=(order.get("challenge_type") or "http-01"),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — diagnostic boundary
|
||||
logger.exception(
|
||||
|
||||
+401
-138
@@ -29,6 +29,40 @@ AGENT_VERSIONS = {
|
||||
"linux": "2.0.0"
|
||||
}
|
||||
|
||||
|
||||
def _sanitize_agent_json(body_str: str):
|
||||
"""Repair the common malformed-JSON patterns a hand-built agent heartbeat can emit.
|
||||
|
||||
Agents assemble their heartbeat JSON as text in bash, so an empty interpolated value can leave
|
||||
a structurally-invalid comma (issue #31). Returns (possibly_repaired_str, was_changed). The
|
||||
repairs are conservative and target only structural artifacts an agent produces; they never
|
||||
alter this endpoint's legitimate string values (the agent emits no string containing ',,' —
|
||||
haproxy_stats_csv is base64/comma-free and the rest are constrained os/kernel/ip/version text).
|
||||
"""
|
||||
import re
|
||||
sanitized = False
|
||||
# Fix 1: empty value before a comma ("server_statuses": ,)
|
||||
if re.search(r':\s*,', body_str):
|
||||
body_str = re.sub(r':\s*,', ': null,', body_str); sanitized = True
|
||||
# Fix 2: empty value before a closing brace ("field":})
|
||||
if re.search(r':\s*}', body_str):
|
||||
body_str = re.sub(r':\s*}', ': null}', body_str); sanitized = True
|
||||
# Fix 3: trailing comma before } or ]
|
||||
if re.search(r',(\s*[}\]])', body_str):
|
||||
body_str = re.sub(r',(\s*[}\]])', r'\1', body_str); sanitized = True
|
||||
# Fix 4: leading comma run right after an opening brace/bracket (issue #31): an empty
|
||||
# $system_info as the first member collapses to '{ , "name": ...'. The ': ,' fix above cannot
|
||||
# catch this because there is no key/colon before the comma.
|
||||
if re.search(r'([{\[])(\s*,)+', body_str):
|
||||
body_str = re.sub(r'([{\[])(\s*,)+', r'\1', body_str); sanitized = True
|
||||
# Fix 5: a run of commas between members (issue #31): an empty $system_info between two fields
|
||||
# produces '"version": "x",\n ,\n "haproxy_status": ...'. Runs after Fix 1/3 so only
|
||||
# structural commas remain; collapse any comma run to a single comma.
|
||||
if re.search(r',(\s*,)+', body_str):
|
||||
body_str = re.sub(r',(\s*,)+', ',', body_str); sanitized = True
|
||||
return body_str, sanitized
|
||||
|
||||
|
||||
def get_platform_key(agent_platform: str) -> str:
|
||||
"""Convert agent platform to standardized platform key - fixed empty platform fallback"""
|
||||
platform = agent_platform.lower() if agent_platform else 'unknown'
|
||||
@@ -217,7 +251,7 @@ def calculate_agent_health(status, last_seen):
|
||||
return "offline"
|
||||
|
||||
@router.get("", summary="Get All Agents", response_description="List of all agents")
|
||||
async def get_agents(pool_id: Optional[int] = None, authorization: str = Header(None)):
|
||||
async def get_agents(pool_id: Optional[int] = None, authorization: str = Header(None), x_api_key: Optional[str] = Header(None)):
|
||||
"""
|
||||
# Get All Agents
|
||||
|
||||
@@ -269,9 +303,22 @@ async def get_agents(pool_id: Optional[int] = None, authorization: str = Header(
|
||||
- **haproxy_status**: Status of HAProxy service on agent's server
|
||||
- **last_seen**: Last heartbeat timestamp
|
||||
"""
|
||||
# SECURITY (GHSA-3p5c-m5m4-mjpx): the agent inventory (names, hostnames, IPs,
|
||||
# pools, OS) is operator data and was previously served unauthenticated — it is
|
||||
# also the read-back channel used in the RCE exfil PoC. Require EITHER a valid
|
||||
# operator JWT OR a valid agent X-API-Key: deployed agents poll this endpoint
|
||||
# (with their key, not a JWT) to read their own applied_config_version and avoid
|
||||
# re-applying config on restart, so a JWT-only gate would break them. Checked
|
||||
# before the try so the 401 is not swallowed by the generic handler.
|
||||
if authorization:
|
||||
current_user = await get_current_user_from_token(authorization) # raises 401 on invalid JWT
|
||||
else:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
if not await validate_agent_api_key(x_api_key):
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
|
||||
try:
|
||||
if pool_id:
|
||||
agents = await conn.fetch("""
|
||||
@@ -729,15 +776,24 @@ async def generate_uninstall_script(platform: str, authorization: str = Header(N
|
||||
sudo ./uninstall-agent.sh
|
||||
```
|
||||
"""
|
||||
# SECURITY (GHSA-3p5c-m5m4-mjpx): require authentication (operator JWT or agent
|
||||
# key), consistent with generate-install-script. The uninstall script itself is
|
||||
# generic (no secrets/topology), but an agent-management endpoint should not be
|
||||
# anonymously reachable. Checked before the try so the 401 is not swallowed.
|
||||
if authorization:
|
||||
await get_current_user_from_token(authorization)
|
||||
elif not await validate_agent_api_key(x_api_key):
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
# Validate platform
|
||||
platform_lower = platform.lower()
|
||||
if platform_lower not in ['linux', 'macos']:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Invalid platform: {platform}. Must be 'linux' or 'macos'"
|
||||
)
|
||||
|
||||
# Normalize platform to a canonical key (always 'linux' or 'macos').
|
||||
# macOS agents register with platform 'darwin' (from `uname -s`), so the
|
||||
# strict ['linux','macos'] check used to 400 on the UI's uninstall flow
|
||||
# (GET /generate-uninstall-script/darwin). Reuse the same get_platform_key()
|
||||
# helper the install-script generator uses, so darwin/osx/mac and the
|
||||
# linux distro variants all resolve correctly. Backward-compatible:
|
||||
# 'linux'/'macos' still map to themselves.
|
||||
platform_lower = get_platform_key(platform)
|
||||
|
||||
# Read uninstall script from agent_scripts directory (same as install scripts)
|
||||
import os
|
||||
script_filename = f"uninstall-agent-{platform_lower}.sh"
|
||||
@@ -861,7 +917,25 @@ async def delete_agent(agent_id: int, authorization: str = Header(None)):
|
||||
# Validate cluster access if agent belongs to a cluster
|
||||
if agent['cluster_id']:
|
||||
await validate_user_cluster_access(current_user['id'], agent['cluster_id'], conn)
|
||||
|
||||
|
||||
# HA/VIP (Issue #27): block deleting a node that's still a member of an active VIP.
|
||||
# Otherwise the CASCADE would silently drop it from the VIP (breaking the one-MASTER
|
||||
# topology with no signal) and the still-running node would keep advertising the VIP
|
||||
# with no way to be told to tear down (review MED-3). Make the operator remove it
|
||||
# from the VIP first — that stages a clean PENDING change they can apply.
|
||||
try:
|
||||
vip_member = await conn.fetchrow(
|
||||
"SELECT v.name FROM vip_members vm JOIN vip_instances v ON v.id = vm.vip_id "
|
||||
"WHERE vm.agent_id = $1 AND v.is_active = TRUE LIMIT 1", agent_id)
|
||||
except Exception: # noqa: BLE001 — vip_* may not exist on older schemas; don't block delete
|
||||
vip_member = None
|
||||
if vip_member:
|
||||
await close_database_connection(conn)
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=(f"This node is a member of VIP '{vip_member['name']}'. Remove it from the "
|
||||
f"VIP on the HA / VIP page (and apply) before deleting the agent."))
|
||||
|
||||
await conn.execute("DELETE FROM agents WHERE id = $1", agent_id)
|
||||
|
||||
await close_database_connection(conn)
|
||||
@@ -905,14 +979,24 @@ def _extract_agent_ip(heartbeat_data: AgentHeartbeat) -> Optional[str]:
|
||||
return None
|
||||
|
||||
@router.post("/{agent_id}/heartbeat")
|
||||
async def agent_heartbeat(agent_id: int, heartbeat_data: AgentHeartbeat):
|
||||
async def agent_heartbeat(agent_id: int, heartbeat_data: AgentHeartbeat, x_api_key: Optional[str] = Header(None)):
|
||||
"""Receive agent heartbeat and update status."""
|
||||
# Agent authentication is MANDATORY (GHSA-3p5c-m5m4-mjpx). This legacy by-ID
|
||||
# heartbeat previously had NO auth, allowing unauthenticated state spoofing of
|
||||
# any agent row. Deployed agents use the by-name heartbeat; a valid global
|
||||
# agent token is now required here too. NOTE: raised BEFORE the try below so
|
||||
# the 401 is not swallowed by the generic `except Exception` handler.
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key on by-id heartbeat for agent ID {agent_id}")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
|
||||
await conn.execute("""
|
||||
UPDATE agents
|
||||
SET status = 'online',
|
||||
UPDATE agents
|
||||
SET status = 'online',
|
||||
last_seen = CURRENT_TIMESTAMP,
|
||||
hostname = COALESCE($2, hostname),
|
||||
haproxy_status = COALESCE($3, haproxy_status),
|
||||
@@ -1035,6 +1119,8 @@ async def agent_config_applied_notification(agent_name: str, notification_data:
|
||||
await close_database_connection(conn)
|
||||
return {"status": "ok", "message": "Config applied notification received"}
|
||||
|
||||
except HTTPException:
|
||||
raise # let auth 401/403 propagate (do not turn it into a 200 error body)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to process config applied notification from agent '{agent_name}': {e}")
|
||||
return {"status": "error", "message": str(e)}
|
||||
@@ -1130,6 +1216,8 @@ async def agent_config_validation_failed(agent_name: str, notification_data: dic
|
||||
|
||||
return {"status": "ok", "message": "Validation error notification received"}
|
||||
|
||||
except HTTPException:
|
||||
raise # let auth 401/403 propagate (do not turn it into a 200 error body)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to process validation error notification from agent '{agent_name}': {e}")
|
||||
return {"status": "error", "message": str(e)}
|
||||
@@ -1419,6 +1507,8 @@ async def agent_config_sync(agent_name: str, sync_data: dict, x_api_key: Optiona
|
||||
logger.info(f"CONFIG SYNC: Agent '{agent_name}' synced {len(active_backends)} backends, {len(active_frontends)} frontends, {len(active_servers)} servers with database")
|
||||
return {"status": "ok", "message": f"Config synced - {len(active_backends)} backends, {len(active_frontends)} frontends, {len(active_servers)} servers processed"}
|
||||
|
||||
except HTTPException:
|
||||
raise # let auth 401/403 propagate (do not turn it into a 200 error body)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to process config sync from agent '{agent_name}': {e}")
|
||||
return {"status": "error", "message": str(e)}
|
||||
@@ -1436,47 +1526,33 @@ async def agent_heartbeat_by_name(
|
||||
import json
|
||||
from pydantic import ValidationError
|
||||
|
||||
# Read raw body and sanitize common JSON errors from agents
|
||||
# Read raw body. Parse VALID JSON as-is (the normal case for every agent version) and only
|
||||
# fall back to the malformed-JSON repair when the body does not parse. This guarantees a healthy
|
||||
# heartbeat from any agent version is byte-for-byte untouched — the repair regexes can never run
|
||||
# against a well-formed payload (issue #31; strictly safer than repairing unconditionally).
|
||||
try:
|
||||
raw_body = await request.body()
|
||||
body_str = raw_body.decode('utf-8')
|
||||
|
||||
# Sanitize common malformed JSON patterns from agents
|
||||
original_body = body_str
|
||||
sanitized = False
|
||||
|
||||
# Fix 1: Empty values before comma (most common: "server_statuses": ,)
|
||||
if re.search(r':\s*,', body_str):
|
||||
body_str = re.sub(r':\s*,', ': null,', body_str)
|
||||
sanitized = True
|
||||
|
||||
# Fix 2: Empty values before closing brace
|
||||
if re.search(r':\s*}', body_str):
|
||||
body_str = re.sub(r':\s*}', ': null}', body_str)
|
||||
sanitized = True
|
||||
|
||||
# Fix 3: Trailing commas
|
||||
if re.search(r',(\s*[}\]])', body_str):
|
||||
body_str = re.sub(r',(\s*[}\]])', r'\1', body_str)
|
||||
sanitized = True
|
||||
|
||||
if sanitized:
|
||||
# Extract agent name for logging
|
||||
agent_name = "unknown"
|
||||
try:
|
||||
name_match = re.search(r'"name"\s*:\s*"([^"]+)"', body_str)
|
||||
if name_match:
|
||||
agent_name = name_match.group(1)
|
||||
except:
|
||||
pass
|
||||
|
||||
logger.info(f"Sanitized malformed JSON from agent '{agent_name}' - fixed empty values and trailing commas")
|
||||
logger.debug(f"Original JSON (preview): {original_body[:300]}")
|
||||
logger.debug(f"Sanitized JSON (preview): {body_str[:300]}")
|
||||
|
||||
# Parse sanitized JSON into Pydantic model
|
||||
heartbeat_dict = json.loads(body_str)
|
||||
|
||||
|
||||
try:
|
||||
heartbeat_dict = json.loads(body_str)
|
||||
except json.JSONDecodeError:
|
||||
# Malformed body (would otherwise be a hard 400). Attempt a conservative repair of the
|
||||
# comma artifacts a hand-built agent heartbeat can emit, then re-parse.
|
||||
repaired, changed = _sanitize_agent_json(body_str)
|
||||
if changed:
|
||||
agent_name = "unknown"
|
||||
try:
|
||||
name_match = re.search(r'"name"\s*:\s*"([^"]+)"', repaired)
|
||||
if name_match:
|
||||
agent_name = name_match.group(1)
|
||||
except Exception:
|
||||
pass
|
||||
logger.info(f"Repaired malformed JSON from agent '{agent_name}' before parsing")
|
||||
logger.debug(f"Original JSON (preview): {body_str[:300]}")
|
||||
logger.debug(f"Repaired JSON (preview): {repaired[:300]}")
|
||||
heartbeat_dict = json.loads(repaired) # may still raise -> handled as 400 below
|
||||
|
||||
# DEBUG: Log cluster_id for auto-register troubleshooting
|
||||
if heartbeat_dict.get('name'):
|
||||
logger.info(f"HEARTBEAT DEBUG: agent={heartbeat_dict.get('name')}, cluster_id={heartbeat_dict.get('cluster_id')}, has_cluster_id={bool(heartbeat_dict.get('cluster_id'))}")
|
||||
@@ -1493,21 +1569,25 @@ async def agent_heartbeat_by_name(
|
||||
logger.error(f"Unexpected error processing heartbeat: {e}")
|
||||
raise HTTPException(status_code=500, detail="Internal server error")
|
||||
|
||||
# Agent authentication is MANDATORY (GHSA-3p5c-m5m4-mjpx). A valid global agent
|
||||
# token is required to heartbeat OR auto-register. Deployed agents always send
|
||||
# X-API-Key; an absent/invalid key is an unauthenticated caller. This is done
|
||||
# OUTSIDE the processing try below (whose generic `except Exception` would
|
||||
# otherwise convert the 401 into a 500), and before opening a DB connection
|
||||
# (validate_agent_api_key(None) needs no DB). Closes keyless heartbeat spoofing
|
||||
# and keyless rogue-agent auto-registration (the `elif not agent` keyless path
|
||||
# below is now unreachable, since agent_auth is guaranteed truthy past here).
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key on heartbeat for agent '{heartbeat_data.name}'")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
|
||||
# Continue with normal heartbeat processing
|
||||
try:
|
||||
# Validate agent API key for security
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
|
||||
conn = await get_database_connection()
|
||||
agent_name = heartbeat_data.name
|
||||
|
||||
# If API key provided, validate it exists but allow placeholder agent updates
|
||||
if x_api_key and not agent_auth:
|
||||
await close_database_connection(conn)
|
||||
logger.warning(f"Invalid API key provided by agent '{agent_name}'")
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
agent = await conn.fetchrow("SELECT id, pool_id, api_key FROM agents WHERE name = $1", agent_name)
|
||||
|
||||
# If agent exists and is using a different API key, update the token association
|
||||
@@ -1656,9 +1736,13 @@ async def agent_heartbeat_by_name(
|
||||
""", x_api_key)
|
||||
|
||||
# Check if agent was in upgrading status and version has changed
|
||||
current_agent_status = await conn.fetchval("SELECT status FROM agents WHERE id = $1", agent_id)
|
||||
current_agent_version = await conn.fetchval("SELECT version FROM agents WHERE id = $1", agent_id)
|
||||
current_upgrade_status = await conn.fetchval("SELECT upgrade_status FROM agents WHERE id = $1", agent_id)
|
||||
# (v1.8.6: one round-trip instead of three; row is None exactly when the
|
||||
# per-column fetchvals would each have returned None)
|
||||
current_agent_row = await conn.fetchrow(
|
||||
"SELECT status, version, upgrade_status FROM agents WHERE id = $1", agent_id)
|
||||
current_agent_status = current_agent_row['status'] if current_agent_row else None
|
||||
current_agent_version = current_agent_row['version'] if current_agent_row else None
|
||||
current_upgrade_status = current_agent_row['upgrade_status'] if current_agent_row else None
|
||||
|
||||
# Determine new status - preserve upgrading status unless version actually changed
|
||||
new_status = current_agent_status or 'online'
|
||||
@@ -1749,9 +1833,14 @@ async def agent_heartbeat_by_name(
|
||||
heartbeat_data.operating_system, heartbeat_data.kernel_version,
|
||||
heartbeat_data.uptime, heartbeat_data.cpu_count, heartbeat_data.memory_total,
|
||||
heartbeat_data.disk_space,
|
||||
# Convert lists to JSON for JSONB columns
|
||||
heartbeat_data.network_interfaces if isinstance(heartbeat_data.network_interfaces, str) else json.dumps(heartbeat_data.network_interfaces or []),
|
||||
heartbeat_data.capabilities if isinstance(heartbeat_data.capabilities, str) else json.dumps(heartbeat_data.capabilities or []),
|
||||
# Convert lists to JSON for JSONB columns. Don't WIPE network_interfaces/capabilities
|
||||
# when a heartbeat omits them: send NULL so COALESCE keeps the existing value (a bare
|
||||
# `or []` would store "[]" and erase the reported NICs / keepalived_management on every
|
||||
# daemon heartbeat that doesn't include them — issue #27 corporate test).
|
||||
(heartbeat_data.network_interfaces if isinstance(heartbeat_data.network_interfaces, str)
|
||||
else (json.dumps(heartbeat_data.network_interfaces) if heartbeat_data.network_interfaces else None)),
|
||||
(heartbeat_data.capabilities if isinstance(heartbeat_data.capabilities, str)
|
||||
else (json.dumps(heartbeat_data.capabilities) if heartbeat_data.capabilities else None)),
|
||||
agent_ip, heartbeat_data.haproxy_status, heartbeat_data.haproxy_version,
|
||||
heartbeat_data.applied_config_version, new_status, update_applied_version,
|
||||
heartbeat_data.keepalive_state, heartbeat_data.keepalive_ip)
|
||||
@@ -1907,9 +1996,19 @@ async def agent_heartbeat_by_name(
|
||||
@router.get("/{agent_name}/config")
|
||||
async def get_agent_config(agent_name: str, x_api_key: Optional[str] = Header(None)):
|
||||
"""Get HAProxy configuration for specific agent"""
|
||||
# Validate agent API key — MANDATORY (GHSA-3p5c-m5m4-mjpx). Checked BEFORE any
|
||||
# DB work and before the existence check, so an unauthenticated caller learns
|
||||
# neither the full haproxy.cfg nor whether the agent exists. Raised before the
|
||||
# try so it is not swallowed by the generic handler; validate_agent_api_key(None)
|
||||
# returns None without touching the DB.
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key for agent '{agent_name}' config fetch")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
|
||||
# Get agent info first to check pool
|
||||
# CRITICAL: Include cluster's haproxy_bin_path, haproxy_config_path, stats_socket_path
|
||||
# These are needed for dynamic validation - cluster admin can change paths without reinstalling agent
|
||||
@@ -1921,30 +2020,19 @@ async def get_agent_config(agent_name: str, x_api_key: Optional[str] = Header(No
|
||||
LEFT JOIN haproxy_clusters hc ON hc.pool_id = a.pool_id
|
||||
WHERE a.name = $1
|
||||
""", agent_name)
|
||||
|
||||
|
||||
if not agent_info:
|
||||
await close_database_connection(conn)
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
|
||||
# Validate agent API key
|
||||
# API key is global - can be used for multiple agents
|
||||
if x_api_key:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
|
||||
if not agent_auth:
|
||||
await close_database_connection(conn)
|
||||
logger.warning(f"Invalid API key provided for agent '{agent_name}' config fetch")
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.info(f"Agent '{agent_name}' fetching config using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' fetching config using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"Config fetch authorized for agent '{agent_name}'")
|
||||
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.info(f"Agent '{agent_name}' fetching config using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' fetching config using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"Config fetch authorized for agent '{agent_name}'")
|
||||
|
||||
if not agent_info['enabled']:
|
||||
await close_database_connection(conn)
|
||||
return {
|
||||
@@ -2035,9 +2123,18 @@ async def get_agent_config(agent_name: str, x_api_key: Optional[str] = Header(No
|
||||
@router.get("/{agent_name}/ssl-certificates")
|
||||
async def get_agent_ssl_certificates(agent_name: str, since: Optional[str] = None, x_api_key: Optional[str] = Header(None)):
|
||||
"""Get SSL certificates for specific agent's cluster"""
|
||||
# Validate agent API key — MANDATORY (GHSA-3p5c-m5m4-mjpx). This response
|
||||
# returns SSL private_key_content, so authentication is checked BEFORE any DB
|
||||
# work and before the existence check. Raised before the try so the 401 is not
|
||||
# swallowed; validate_agent_api_key(None) returns None without a DB hit.
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key for agent '{agent_name}' SSL certificates")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
|
||||
# Get agent and cluster info first
|
||||
agent_info = await conn.fetchrow("""
|
||||
SELECT a.id, a.name, a.pool_id, hc.id as cluster_id, hc.name as cluster_name,
|
||||
@@ -2046,29 +2143,18 @@ async def get_agent_ssl_certificates(agent_name: str, since: Optional[str] = Non
|
||||
LEFT JOIN haproxy_clusters hc ON hc.pool_id = a.pool_id
|
||||
WHERE a.name = $1
|
||||
""", agent_name)
|
||||
|
||||
|
||||
if not agent_info:
|
||||
await close_database_connection(conn)
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
|
||||
# Validate agent API key
|
||||
# API key is global - can be used for multiple agents
|
||||
if x_api_key:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
|
||||
if not agent_auth:
|
||||
await close_database_connection(conn)
|
||||
logger.warning(f"Invalid API key provided for agent '{agent_name}' SSL certificates")
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.info(f"Agent '{agent_name}' fetching SSL certificates using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' fetching SSL certificates using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"SSL fetch authorized for agent '{agent_name}'")
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.info(f"Agent '{agent_name}' fetching SSL certificates using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' fetching SSL certificates using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"SSL fetch authorized for agent '{agent_name}'")
|
||||
|
||||
if not agent_info['enabled']:
|
||||
await close_database_connection(conn)
|
||||
@@ -2180,13 +2266,171 @@ async def get_agent_ssl_certificates(agent_name: str, since: Optional[str] = Non
|
||||
logger.info(f"SSL INCREMENTAL: No certificates updated since {since}")
|
||||
|
||||
return response_data
|
||||
|
||||
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"SSL certificates retrieval failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
@router.get("/{agent_name}/keepalived-config")
|
||||
async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str] = Header(None)):
|
||||
"""Issue #27 (v1.7.0) — deliver this agent's APPLIED keepalived snapshot, or a
|
||||
teardown / no-op signal.
|
||||
|
||||
Snapshot-based (T-1): keys off vip_instances.is_active + the member's
|
||||
applied_config_content, NOT the live last_config_status — so a PENDING edit never
|
||||
flips a running member to not_configured (no mid-edit teardown). Auth is MANDATORY (a
|
||||
valid agent key is required), but — like the /config and /ssl-certificates endpoints —
|
||||
the agent API key is a SHARED/global install token, so the config is resolved by the
|
||||
requested agent_name and a token/name mismatch is an advisory audit log, NOT a 403
|
||||
(a hard 403 would break every VIP member whose name isn't the one row the shared token
|
||||
resolves to — review HIGH-1). Any unexpected error degrades to not_configured (B-7) so
|
||||
the agent stays inert; a node with no membership row always gets not_configured.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
# Auth FIRST and MANDATORY: a valid agent key is REQUIRED (the response carries the
|
||||
# VRRP secret). The token is shared/global, so resolve by agent_name and only LOG a
|
||||
# name mismatch — do not 403 (HIGH-1). Done before the agent lookup so an
|
||||
# unauthenticated caller can't probe which agent names exist.
|
||||
from auth_middleware import validate_agent_api_key
|
||||
if not x_api_key:
|
||||
raise HTTPException(status_code=401, detail="Agent API key required")
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
# The agent API key is a SHARED/global install token (many agent rows per token),
|
||||
# so validate_agent_api_key resolves it to one arbitrary agent for that token. Resolve
|
||||
# the keepalived config strictly by the requested agent_name and treat a token/name
|
||||
# mismatch as an advisory audit log — exactly like the /config and /ssl-certificates
|
||||
# endpoints. (A hard 403 here would reject every VIP member whose name isn't the one
|
||||
# row the shared token happens to return, so the VIP could never converge — review HIGH-1.)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.info(f"Agent '{agent_name}' fetching keepalived config using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' fetching keepalived config using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
conn = await get_database_connection()
|
||||
# Resolve the agent + its cluster's keepalived.conf path (cluster-driven, like the
|
||||
# HAProxy paths). config_path is returned in EVERY response so the agent knows where
|
||||
# to write/own-marker-check even on not_configured/teardown.
|
||||
agent = await conn.fetchrow("""
|
||||
SELECT a.id, a.name, COALESCE(a.enabled, TRUE) AS enabled,
|
||||
hc.keepalived_config_path
|
||||
FROM agents a
|
||||
LEFT JOIN haproxy_clusters hc ON hc.pool_id = a.pool_id
|
||||
WHERE a.name = $1
|
||||
""", agent_name)
|
||||
if not agent:
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
config_path = agent['keepalived_config_path'] or '/etc/keepalived/keepalived.conf'
|
||||
if not agent['enabled']:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
|
||||
row = await conn.fetchrow("""
|
||||
SELECT v.id AS vip_id, v.name AS vip_name, v.is_active, v.track_haproxy,
|
||||
v.purge_on_teardown,
|
||||
m.applied_config_content, m.applied_config_hash
|
||||
FROM vip_members m JOIN vip_instances v ON v.id = m.vip_id
|
||||
WHERE m.agent_id = $1
|
||||
-- Active VIP first (an agent has at most one). With NO active VIP, pick the most
|
||||
-- RECENTLY updated inactive membership so a teardown reflects the latest delete
|
||||
-- (incl. its purge flag) — not a stale older VIP the node was once part of.
|
||||
ORDER BY v.is_active DESC, v.updated_at DESC, v.id DESC
|
||||
LIMIT 1
|
||||
""", agent['id'])
|
||||
|
||||
if not row:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
if not row['is_active']:
|
||||
# Soft-deleted VIP → teardown. purge carries the operator's opt-in package removal;
|
||||
# the agent still only purges on nodes where IT installed keepalived (install marker).
|
||||
return {"agent_name": agent_name, "status": "teardown", "vip_id": row['vip_id'],
|
||||
"config_path": config_path, "keepalived": None,
|
||||
"purge": bool(row['purge_on_teardown'])}
|
||||
if not row['applied_config_content']:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
|
||||
from services.keepalived_config import build_haproxy_check_script
|
||||
check_script = build_haproxy_check_script() if row['track_haproxy'] else ""
|
||||
return {
|
||||
"agent_name": agent_name,
|
||||
"status": "available",
|
||||
"config_path": config_path,
|
||||
"keepalived": {
|
||||
"desired_state": "enabled",
|
||||
"install_if_missing": True,
|
||||
"vip_id": row['vip_id'],
|
||||
"vip_name": row['vip_name'],
|
||||
"config_content": row['applied_config_content'],
|
||||
"config_hash": row['applied_config_hash'],
|
||||
"check_script": check_script,
|
||||
},
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"keepalived-config delivery failed for '{agent_name}': {e}")
|
||||
# Degrade to no-op rather than 500 (B-7) — keeps the fleet inert on any error.
|
||||
return {"agent_name": agent_name, "status": "not_configured",
|
||||
"config_path": "/etc/keepalived/keepalived.conf", "keepalived": None}
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/{agent_name}/keepalived-status")
|
||||
async def agent_keepalived_status(agent_name: str, status_data: dict, x_api_key: Optional[str] = Header(None)):
|
||||
"""Issue #27 (v1.7.0) — agent reports the outcome of a keepalived deploy/teardown.
|
||||
|
||||
Auth mirrors config-applied's post-Bulgu-#75 guard (reject a MISSING key — never the
|
||||
`and` short-circuit that accepted no-key requests). The token is a shared/global install
|
||||
token, so the status is recorded strictly for the requested agent_name and a token/name
|
||||
mismatch is an advisory audit log (like /config-applied), not a 403 — review HIGH-1.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not x_api_key or not agent_auth:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
if agent_auth['name'] != agent_name:
|
||||
logger.info(f"Agent '{agent_name}' reporting keepalived status using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
conn = await get_database_connection()
|
||||
agent = await conn.fetchrow("SELECT id FROM agents WHERE name = $1", agent_name)
|
||||
if not agent:
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
|
||||
vip_id = status_data.get("vip_id")
|
||||
state = (status_data.get("state") or "").strip()[:24]
|
||||
config_hash = (status_data.get("config_hash") or "")[:64]
|
||||
message = status_data.get("message")
|
||||
if vip_id is None:
|
||||
# No specific VIP (e.g. a teardown ack) — update all this agent's memberships.
|
||||
await conn.execute("""
|
||||
UPDATE vip_members SET last_deploy_state=$2, last_deploy_message=$3,
|
||||
last_deploy_hash=$4, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
WHERE agent_id=$1
|
||||
""", agent['id'], state, message, config_hash)
|
||||
else:
|
||||
await conn.execute("""
|
||||
UPDATE vip_members SET last_deploy_state=$3, last_deploy_message=$4,
|
||||
last_deploy_hash=$5, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
WHERE agent_id=$1 AND vip_id=$2
|
||||
""", agent['id'], int(vip_id), state, message, config_hash)
|
||||
return {"status": "ok"}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"keepalived-status update failed for '{agent_name}': {e}")
|
||||
raise HTTPException(status_code=500, detail="keepalived-status update failed")
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
@router.get("/script-version")
|
||||
async def get_latest_script_version(platform: str = "macos"):
|
||||
"""Get the latest available agent script version for specified platform"""
|
||||
@@ -2234,16 +2478,24 @@ async def get_latest_script_version(platform: str = "macos"):
|
||||
@router.get("/{agent_name}/upgrade-status")
|
||||
async def get_agent_upgrade_status(agent_name: str, x_api_key: Optional[str] = Header(None)):
|
||||
"""Get agent upgrade status - used by agents to check if they should upgrade"""
|
||||
# Validate agent API key — MANDATORY (GHSA-3p5c-m5m4-mjpx). Checked before any
|
||||
# DB work; deployed agents always send X-API-Key. Raised before the try so the
|
||||
# 401 is not swallowed; validate_agent_api_key(None) returns None without a DB hit.
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key for agent '{agent_name}' upgrade status")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
|
||||
# Check if agent has upgrade pending (include platform and pool for validation)
|
||||
agent = await conn.fetchrow("""
|
||||
SELECT status, version as current_version, platform, pool_id
|
||||
FROM agents
|
||||
FROM agents
|
||||
WHERE name = $1
|
||||
""", agent_name)
|
||||
|
||||
|
||||
if not agent:
|
||||
await close_database_connection(conn)
|
||||
return {
|
||||
@@ -2251,26 +2503,15 @@ async def get_agent_upgrade_status(agent_name: str, x_api_key: Optional[str] = H
|
||||
"target_version": "",
|
||||
"message": "Agent not found"
|
||||
}
|
||||
|
||||
# Validate agent API key
|
||||
# API key is global - can be used for multiple agents
|
||||
if x_api_key:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
|
||||
if not agent_auth:
|
||||
await close_database_connection(conn)
|
||||
logger.warning(f"Invalid API key provided for agent '{agent_name}' upgrade status")
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.debug(f"Agent '{agent_name}' checking upgrade status using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' checking upgrade status using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"Upgrade status check authorized for agent '{agent_name}'")
|
||||
|
||||
|
||||
# Log which agent's API key was used (for audit trail)
|
||||
if agent_auth['name'] == agent_name:
|
||||
logger.debug(f"Agent '{agent_name}' checking upgrade status using its own API key")
|
||||
else:
|
||||
logger.info(f"Agent '{agent_name}' checking upgrade status using API key from agent '{agent_auth['name']}'")
|
||||
|
||||
logger.debug(f"Upgrade status check authorized for agent '{agent_name}'")
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
# Agent should upgrade if status is 'upgrading'
|
||||
@@ -2704,7 +2945,17 @@ async def get_agent_script_template(platform: str, authorization: str = Header(N
|
||||
"""Get the latest script template for specified platform from database"""
|
||||
try:
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
|
||||
|
||||
# SECURITY (GHSA-7rhv-c5pc-69r8): the raw install/upgrade script is a
|
||||
# version-management surface. Gate reads with agents.version too, matching
|
||||
# the write path above (operator/security_admin/super_admin retain access).
|
||||
has_permission = await check_user_permission(current_user["id"], "agents", "version")
|
||||
if not has_permission:
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail="Insufficient permissions: agents.version required"
|
||||
)
|
||||
|
||||
conn = await get_database_connection()
|
||||
|
||||
# Get latest script template for platform
|
||||
@@ -2755,7 +3006,19 @@ async def save_agent_script_template(platform: str, template_data: dict, authori
|
||||
"""Save updated script template to database using shared helper function"""
|
||||
try:
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
|
||||
|
||||
# SECURITY (GHSA-7rhv-c5pc-69r8): agent script templates become the
|
||||
# install/self-upgrade script executed as root on HAProxy nodes. A poisoned
|
||||
# template is RCE. Authentication alone is NOT enough — require the same
|
||||
# agents.version permission as POST /versions; otherwise any JWT holder
|
||||
# (including viewer) could overwrite the active script.
|
||||
has_permission = await check_user_permission(current_user["id"], "agents", "version")
|
||||
if not has_permission:
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail="Insufficient permissions: agents.version required"
|
||||
)
|
||||
|
||||
script_content = template_data.get('script_content', '')
|
||||
version = template_data.get('version', '')
|
||||
|
||||
|
||||
+53
-22
@@ -333,31 +333,38 @@ async def get_backends(
|
||||
""")
|
||||
|
||||
result = []
|
||||
# Issue #24: servers must honor include_inactive exactly like the backend
|
||||
# queries above. Pre-fix these sub-queries hardcoded `is_active = TRUE`
|
||||
# (added in f34a6ee to hide soft-deleted entities), so a server toggled
|
||||
# OFF (is_active=false) vanished from the UI with no way to reactivate it.
|
||||
# Default callers (include_inactive=false) keep the is_active filter →
|
||||
# byte-identical behavior; include_inactive=true now also returns inactive
|
||||
# (disabled / soft-deleted) servers. last_config_status is selected so the
|
||||
# frontend can tell DISABLED (re-enableable) from DELETION (pending delete).
|
||||
server_active_filter = "" if include_inactive else "AND is_active = TRUE"
|
||||
for backend in backends:
|
||||
# Get servers for this backend with cluster_id (ONLY show active servers)
|
||||
# CRITICAL FIX: Add is_active = TRUE filter to prevent soft-deleted servers from appearing
|
||||
if cluster_id:
|
||||
servers = await conn.fetch("""
|
||||
servers = await conn.fetch(f"""
|
||||
SELECT id, server_name, server_address, server_port, weight, maxconn,
|
||||
check_enabled, check_port, backup_server, ssl_enabled, ssl_verify, ssl_certificate_id,
|
||||
ssl_sni, ssl_min_ver, ssl_max_ver, ssl_ciphers,
|
||||
cookie_value, inter, fall, rise,
|
||||
is_active, cluster_id,
|
||||
is_active, cluster_id, last_config_status,
|
||||
haproxy_status, haproxy_status_updated_at, backend_name
|
||||
FROM backend_servers
|
||||
WHERE backend_name = $1 AND cluster_id = $2 AND is_active = TRUE ORDER BY server_name
|
||||
FROM backend_servers
|
||||
WHERE backend_name = $1 AND cluster_id = $2 {server_active_filter} ORDER BY server_name
|
||||
""", backend["name"], cluster_id)
|
||||
else:
|
||||
servers = await conn.fetch("""
|
||||
servers = await conn.fetch(f"""
|
||||
SELECT id, server_name, server_address, server_port, weight, maxconn,
|
||||
check_enabled, check_port, backup_server, ssl_enabled, ssl_verify, ssl_certificate_id,
|
||||
ssl_sni, ssl_min_ver, ssl_max_ver, ssl_ciphers,
|
||||
cookie_value, inter, fall, rise,
|
||||
is_active, cluster_id,
|
||||
is_active, cluster_id, last_config_status,
|
||||
haproxy_status, haproxy_status_updated_at, backend_name
|
||||
FROM backend_servers
|
||||
WHERE backend_name = $1 AND is_active = TRUE ORDER BY server_name
|
||||
""", backend["name"])
|
||||
FROM backend_servers
|
||||
WHERE backend_name = $1 {server_active_filter} ORDER BY server_name
|
||||
""", backend["name"])
|
||||
|
||||
# Prepare server list with real-time HAProxy status from agents
|
||||
server_list = []
|
||||
@@ -413,6 +420,7 @@ async def get_backends(
|
||||
"fall": s.get("fall"),
|
||||
"rise": s.get("rise"),
|
||||
"is_active": s["is_active"],
|
||||
"last_config_status": s.get("last_config_status") or "APPLIED", # Issue #24: lets UI distinguish DISABLED (re-enableable) from DELETION
|
||||
"status": server_status,
|
||||
"status_age_minutes": status_age_minutes,
|
||||
"last_status_update": s.get("haproxy_status_updated_at").isoformat().replace('+00:00', 'Z') if s.get("haproxy_status_updated_at") else None,
|
||||
@@ -1909,12 +1917,12 @@ async def toggle_server(server_id: int, request: Request, authorization: str = H
|
||||
|
||||
conn = await get_database_connection()
|
||||
|
||||
# Get server info
|
||||
# Get server info. Issue #24: fetch the FULL row (not just 5 columns) so we
|
||||
# can snapshot the pre-toggle state for reject-rollback (see config version below).
|
||||
server = await conn.fetchrow("""
|
||||
SELECT id, server_name, backend_name, is_active, cluster_id
|
||||
FROM backend_servers WHERE id = $1
|
||||
SELECT * FROM backend_servers WHERE id = $1
|
||||
""", server_id)
|
||||
|
||||
|
||||
if not server:
|
||||
await close_database_connection(conn)
|
||||
raise HTTPException(status_code=404, detail="Server not found")
|
||||
@@ -1975,22 +1983,45 @@ async def toggle_server(server_id: int, request: Request, authorization: str = H
|
||||
|
||||
# Generate new HAProxy config (after database commit)
|
||||
config_content = await generate_haproxy_config_for_cluster(cluster_id)
|
||||
|
||||
|
||||
# Create new config version
|
||||
config_hash = hashlib.sha256(config_content.encode()).hexdigest()
|
||||
version_name = f"server-{server_id}-toggle-{int(time.time())}"
|
||||
|
||||
|
||||
# Get system admin user ID for created_by
|
||||
conn2 = await get_database_connection()
|
||||
admin_user_id = await conn2.fetchval("SELECT id FROM users WHERE username = 'admin' LIMIT 1") or 1
|
||||
|
||||
|
||||
# Issue #24: persist an entity snapshot so a Reject of this toggle
|
||||
# rolls back is_active to its pre-toggle value. Pre-fix the toggle's
|
||||
# config version carried NO metadata, so reject only reset
|
||||
# last_config_status and the server stayed disabled (out of sync with
|
||||
# the still-active live config). Mirrors the server-edit snapshot path;
|
||||
# reject's rollback_entity_from_snapshot('server') restores is_active.
|
||||
import json
|
||||
from utils.entity_snapshot import save_entity_snapshot
|
||||
entity_snapshot_metadata = await save_entity_snapshot(
|
||||
conn=conn2,
|
||||
entity_type="server",
|
||||
entity_id=server_id,
|
||||
old_values=dict(server), # full pre-toggle row
|
||||
new_values={"is_active": new_status},
|
||||
operation="UPDATE",
|
||||
)
|
||||
old_config = await conn2.fetchval("""
|
||||
SELECT config_content FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'APPLIED' AND is_active = TRUE
|
||||
ORDER BY created_at DESC LIMIT 1
|
||||
""", cluster_id)
|
||||
metadata = {"pre_apply_snapshot": old_config or "", **entity_snapshot_metadata}
|
||||
|
||||
# Create PENDING config version
|
||||
config_version_id = await conn2.fetchval("""
|
||||
INSERT INTO config_versions
|
||||
(cluster_id, version_name, config_content, checksum, created_by, is_active, status)
|
||||
VALUES ($1, $2, $3, $4, $5, FALSE, 'PENDING')
|
||||
INSERT INTO config_versions
|
||||
(cluster_id, version_name, config_content, checksum, created_by, is_active, status, metadata)
|
||||
VALUES ($1, $2, $3, $4, $5, FALSE, 'PENDING', $6)
|
||||
RETURNING id
|
||||
""", cluster_id, version_name, config_content, config_hash, admin_user_id)
|
||||
""", cluster_id, version_name, config_content, config_hash, admin_user_id, json.dumps(metadata))
|
||||
|
||||
logger.error(f"SERVER TOGGLE DEBUG: Created PENDING config version {version_name} for cluster {cluster_id}")
|
||||
|
||||
|
||||
+196
-40
@@ -292,12 +292,14 @@ async def create_cluster(cluster: HAProxyClusterCreate, authorization: str = Hea
|
||||
|
||||
# Create cluster
|
||||
cluster_id = await conn.fetchval("""
|
||||
INSERT INTO haproxy_clusters (name, description, connection_type, is_active,
|
||||
stats_socket_path, haproxy_config_path, haproxy_bin_path, pool_id)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7)
|
||||
INSERT INTO haproxy_clusters (name, description, connection_type, is_active,
|
||||
stats_socket_path, haproxy_config_path, haproxy_bin_path,
|
||||
keepalived_config_path, pool_id)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7, $8)
|
||||
RETURNING id
|
||||
""", cluster.name, cluster.description, cluster.connection_type,
|
||||
cluster.stats_socket_path, cluster.haproxy_config_path, cluster.haproxy_bin_path, cluster.pool_id)
|
||||
cluster.stats_socket_path, cluster.haproxy_config_path, cluster.haproxy_bin_path,
|
||||
cluster.keepalived_config_path, cluster.pool_id)
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -436,7 +438,12 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
update_fields.append(f"haproxy_bin_path = ${param_counter}")
|
||||
update_values.append(cluster.haproxy_bin_path)
|
||||
param_counter += 1
|
||||
|
||||
|
||||
if cluster.keepalived_config_path is not None:
|
||||
update_fields.append(f"keepalived_config_path = ${param_counter}")
|
||||
update_values.append(cluster.keepalived_config_path)
|
||||
param_counter += 1
|
||||
|
||||
if cluster.pool_id is not None:
|
||||
update_fields.append(f"pool_id = ${param_counter}")
|
||||
update_values.append(cluster.pool_id)
|
||||
@@ -517,7 +524,7 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
|
||||
|
||||
@router.get("/{cluster_id}", summary="Get Cluster by ID", response_description="Cluster details")
|
||||
async def get_cluster(cluster_id: int, authorization: str = Header(None)):
|
||||
async def get_cluster(cluster_id: int, authorization: str = Header(None), x_api_key: Optional[str] = Header(None)):
|
||||
"""
|
||||
# Get Specific HAProxy Cluster
|
||||
|
||||
@@ -528,8 +535,12 @@ async def get_cluster(cluster_id: int, authorization: str = Header(None)):
|
||||
|
||||
## Example Request
|
||||
```bash
|
||||
# User (UI) authentication:
|
||||
curl -X GET "{BASE_URL}/api/clusters/1" \\
|
||||
-H "Authorization: Bearer eyJhbGciOiJIUz..."
|
||||
# Agent authentication (agent token in X-API-Key):
|
||||
curl -X GET "{BASE_URL}/api/clusters/1" \\
|
||||
-H "X-API-Key: hap_..."
|
||||
```
|
||||
|
||||
## Example Response
|
||||
@@ -554,20 +565,30 @@ async def get_cluster(cluster_id: int, authorization: str = Header(None)):
|
||||
- **404**: Cluster not found
|
||||
- **500**: Server error
|
||||
"""
|
||||
try:
|
||||
# R18c audit fix (round 6 final convergence): authenticate
|
||||
# the caller before fetching cluster topology by ID. Pre-fix
|
||||
# this sibling of GET /api/clusters was anonymous, so an
|
||||
# attacker could iterate cluster IDs to enumerate the same
|
||||
# info (stats socket, paths, ACME flags, pool identity) the
|
||||
# list endpoint just locked down. Closes the asymmetry.
|
||||
# R18c audit fix (round 6 final convergence): authenticate the caller
|
||||
# before fetching cluster topology by ID. Pre-fix this sibling of
|
||||
# GET /api/clusters was anonymous, so an attacker could iterate cluster
|
||||
# IDs to enumerate the same info (stats socket, paths, ACME flags, pool
|
||||
# identity) the list endpoint just locked down.
|
||||
# Issue #22: agents send their token in the X-API-Key header (not a user
|
||||
# JWT), so accept either credential — mirrors the dual-auth on
|
||||
# POST /api/agents/generate-install-script. Anonymous is still rejected.
|
||||
if authorization:
|
||||
from auth_middleware import get_current_user_from_token
|
||||
await get_current_user_from_token(authorization)
|
||||
elif x_api_key:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
if not await validate_agent_api_key(x_api_key):
|
||||
raise HTTPException(status_code=401, detail="Invalid agent API key")
|
||||
else:
|
||||
raise HTTPException(status_code=401, detail="Authorization header or X-API-Key required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
cluster = await conn.fetchrow("""
|
||||
SELECT c.id, c.name, c.description, c.connection_type, c.is_active,
|
||||
SELECT c.id, c.name, c.description, c.connection_type, c.is_active,
|
||||
c.created_at, c.stats_socket_path, c.haproxy_config_path, c.haproxy_bin_path,
|
||||
c.keepalived_config_path,
|
||||
c.pool_id, c.is_default, c.acme_enabled, c.acme_backend_url,
|
||||
p.name as pool_name
|
||||
FROM haproxy_clusters c
|
||||
@@ -591,6 +612,7 @@ async def get_cluster(cluster_id: int, authorization: str = Header(None)):
|
||||
"stats_socket_path": cluster["stats_socket_path"],
|
||||
"haproxy_config_path": cluster["haproxy_config_path"],
|
||||
"haproxy_bin_path": cluster["haproxy_bin_path"],
|
||||
"keepalived_config_path": cluster.get("keepalived_config_path"),
|
||||
"pool_id": cluster["pool_id"],
|
||||
"pool_name": cluster["pool_name"],
|
||||
"acme_enabled": cluster.get("acme_enabled", False),
|
||||
@@ -602,7 +624,7 @@ async def get_cluster(cluster_id: int, authorization: str = Header(None)):
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("", summary="Get All Clusters", response_description="List of all clusters")
|
||||
async def get_clusters(authorization: str = Header(None)):
|
||||
async def get_clusters(authorization: str = Header(None), x_api_key: Optional[str] = Header(None)):
|
||||
"""
|
||||
# Get All HAProxy Clusters
|
||||
|
||||
@@ -610,8 +632,12 @@ async def get_clusters(authorization: str = Header(None)):
|
||||
|
||||
## Example Request
|
||||
```bash
|
||||
# User (UI) authentication:
|
||||
curl -X GET "{BASE_URL}/api/clusters" \\
|
||||
-H "Authorization: Bearer eyJhbGciOiJIUz..."
|
||||
# Agent authentication (agent token in X-API-Key):
|
||||
curl -X GET "{BASE_URL}/api/clusters" \\
|
||||
-H "X-API-Key: hap_..."
|
||||
```
|
||||
|
||||
## Example Response
|
||||
@@ -662,23 +688,34 @@ async def get_clusters(authorization: str = Header(None)):
|
||||
## Error Responses
|
||||
- **500**: Server error
|
||||
"""
|
||||
try:
|
||||
# R18c audit fix (round 6 #3 — KRITIK info leak): require an
|
||||
# authenticated caller. Pre-fix the endpoint accepted
|
||||
# anonymous GETs and returned cluster topology including
|
||||
# internal HAProxy paths (stats socket, config path, bin
|
||||
# path), pool ids, ACME flags, and agent counts. This is
|
||||
# both reconnaissance for an attacker and the spine of the
|
||||
# cluster-scoped RBAC the rest of the platform builds on,
|
||||
# so guarding it at the read layer is essential after R18c
|
||||
# round 5's roster + role guards.
|
||||
# R18c audit fix (round 6 #3 — KRITIK info leak): require an authenticated
|
||||
# caller. Pre-fix the endpoint accepted anonymous GETs and returned cluster
|
||||
# topology including internal HAProxy paths (stats socket, config path, bin
|
||||
# path), pool ids, ACME flags, and agent counts. This is both reconnaissance
|
||||
# for an attacker and the spine of the cluster-scoped RBAC the rest of the
|
||||
# platform builds on, so guarding it at the read layer is essential.
|
||||
# Issue #22: agents send their token in the X-API-Key header (not a user
|
||||
# JWT), so accept either credential — mirrors the dual-auth on
|
||||
# POST /api/agents/generate-install-script. Anonymous is still rejected.
|
||||
# (Guard kept OUTSIDE the try below: get_clusters' broad `except Exception`
|
||||
# re-wraps raised HTTPExceptions into 500, which produced the "500 - 401"
|
||||
# in the issue log; raising here yields a clean 401.)
|
||||
if authorization:
|
||||
from auth_middleware import get_current_user_from_token
|
||||
await get_current_user_from_token(authorization)
|
||||
elif x_api_key:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
if not await validate_agent_api_key(x_api_key):
|
||||
raise HTTPException(status_code=401, detail="Invalid agent API key")
|
||||
else:
|
||||
raise HTTPException(status_code=401, detail="Authorization header or X-API-Key required")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
clusters = await conn.fetch("""
|
||||
SELECT c.id, c.name, c.description, c.connection_type, c.is_active,
|
||||
SELECT c.id, c.name, c.description, c.connection_type, c.is_active,
|
||||
c.created_at, c.stats_socket_path, c.haproxy_config_path, c.haproxy_bin_path,
|
||||
c.keepalived_config_path,
|
||||
c.pool_id, c.is_default, c.acme_enabled, c.acme_backend_url,
|
||||
p.name as pool_name,
|
||||
COALESCE(agent_counts.total_agents, 0) as total_agents,
|
||||
@@ -737,6 +774,7 @@ async def get_clusters(authorization: str = Header(None)):
|
||||
"stats_socket_path": cluster["stats_socket_path"],
|
||||
"haproxy_config_path": cluster["haproxy_config_path"],
|
||||
"haproxy_bin_path": cluster["haproxy_bin_path"],
|
||||
"keepalived_config_path": cluster.get("keepalived_config_path"),
|
||||
"pool_id": cluster["pool_id"],
|
||||
"pool_name": cluster["pool_name"],
|
||||
"acme_enabled": cluster.get("acme_enabled", False),
|
||||
@@ -1313,6 +1351,8 @@ async def list_cluster_config_versions(cluster_id: int, authorization: str = Hea
|
||||
version_type = "WAF Rule"
|
||||
elif "ssl-" in version['version_name']:
|
||||
version_type = "SSL Certificate"
|
||||
elif "vip-" in version['version_name']:
|
||||
version_type = "HA / VIP"
|
||||
|
||||
# Parse validation error if present
|
||||
validation_error = version.get("validation_error")
|
||||
@@ -1417,10 +1457,14 @@ async def apply_pending_changes(
|
||||
# Users will handle configuration completeness through the centralized Apply Management page
|
||||
|
||||
# Get all pending config versions for this cluster
|
||||
# HA/VIP (Issue #27): vip-* versions are owned by the VIP apply/reject endpoints
|
||||
# (keepalived is not part of haproxy.cfg). Exclude them so a generic cluster apply
|
||||
# from any entity page never marks a VIP version APPLIED without enacting it. This
|
||||
# is a no-op for every non-VIP cluster (no vip-* rows exist).
|
||||
pending_versions = await conn.fetch("""
|
||||
SELECT id, version_name, created_at, config_content, checksum, metadata
|
||||
FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'PENDING'
|
||||
FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'PENDING' AND version_name NOT LIKE 'vip-%'
|
||||
ORDER BY created_at ASC
|
||||
""", cluster_id)
|
||||
|
||||
@@ -2539,6 +2583,92 @@ async def get_config_version_diff(cluster_id: int, version_id: int, authorizatio
|
||||
}
|
||||
}
|
||||
|
||||
# HA/VIP (Issue #27): vip-{id}-{action} versions show the generated keepalived.conf
|
||||
# each member node will deploy as the change content (VRRP secret masked). Mirrors
|
||||
# the ssl-* special case above so VIP uses the STANDARD View Change diff modal.
|
||||
vip_match = re.search(r'vip-(\d+)-(create|update|delete)', current_version['version_name'])
|
||||
if vip_match:
|
||||
vip_id = int(vip_match.group(1))
|
||||
vip_action = vip_match.group(2)
|
||||
rendered, vip_meta = None, None
|
||||
old_content = ""
|
||||
try:
|
||||
from routers.vip import render_vip_config_masked
|
||||
rendered, vip_meta = await render_vip_config_masked(conn, vip_id)
|
||||
# For a create/update diff, fetch the PREVIOUS applied vip-* config for this VIP
|
||||
# (the last-deployed keepalived.conf) so an EDIT shows ONLY the changed lines
|
||||
# instead of the whole config as "added". Both sides are already secret-masked.
|
||||
if vip_action != "delete":
|
||||
prev_applied = await conn.fetchrow(
|
||||
"SELECT config_content FROM config_versions WHERE version_name LIKE $1 "
|
||||
"AND status='APPLIED' AND cluster_id=$2 AND id < $3 ORDER BY id DESC LIMIT 1",
|
||||
f"vip-{vip_id}-%", cluster_id, current_version['id'])
|
||||
if prev_applied and prev_applied['config_content']:
|
||||
old_content = prev_applied['config_content']
|
||||
except Exception as vip_err:
|
||||
logger.warning(f"VIP DIFF: render failed for vip {vip_id}: {vip_err}")
|
||||
await close_database_connection(conn)
|
||||
|
||||
changes = []
|
||||
line_number = 1
|
||||
if vip_action == "delete" or rendered is None:
|
||||
title = (vip_meta or {}).get("name") or f"VIP {vip_id}"
|
||||
for line in [f"# HA/VIP change: {title}",
|
||||
"# keepalived will be stopped and the virtual IP released on each member node."
|
||||
if vip_action == "delete" else
|
||||
"# (configuration is not available to render yet)"]:
|
||||
changes.append({"type": "context", "line": line, "line_number": line_number})
|
||||
line_number += 1
|
||||
summary = {"added": 0, "removed": 1 if vip_action == "delete" else 0,
|
||||
"total_changes": 1 if vip_action == "delete" else 0}
|
||||
else:
|
||||
# Real line diff. Match the STANDARD haproxy diff format: the line is stored
|
||||
# WITHOUT a +/- prefix (the UI adds it from `type` — the old `+ {line}` here
|
||||
# caused the doubled "+ +"). A CREATE (no previous applied config) shows
|
||||
# everything as added; an UPDATE shows ONLY the lines that actually changed.
|
||||
import difflib
|
||||
new_content = current_version['config_content'] or rendered or ""
|
||||
added_count = 0
|
||||
removed_count = 0
|
||||
line_number = 0
|
||||
if not old_content:
|
||||
for i, l in enumerate(new_content.split('\n')):
|
||||
changes.append({"type": "added", "line": l, "line_number": i + 1})
|
||||
added_count += 1
|
||||
else:
|
||||
for dl in difflib.unified_diff(old_content.split('\n'), new_content.split('\n'),
|
||||
lineterm='', n=3):
|
||||
if dl.startswith('@@'):
|
||||
mm = re.search(r'@@ -(\d+),?\d* \+(\d+),?\d* @@', dl)
|
||||
if mm:
|
||||
line_number = int(mm.group(2))
|
||||
continue
|
||||
if dl.startswith('---') or dl.startswith('+++'):
|
||||
continue
|
||||
if dl.startswith('+'):
|
||||
changes.append({"type": "added", "line": dl[1:], "line_number": line_number})
|
||||
added_count += 1
|
||||
line_number += 1
|
||||
elif dl.startswith('-'):
|
||||
changes.append({"type": "removed", "line": dl[1:], "line_number": line_number})
|
||||
removed_count += 1
|
||||
elif dl.startswith(' '):
|
||||
changes.append({"type": "context", "line": dl[1:], "line_number": line_number})
|
||||
line_number += 1
|
||||
summary = {"added": added_count, "removed": removed_count,
|
||||
"total_changes": added_count + removed_count}
|
||||
|
||||
return {
|
||||
"current_version": {
|
||||
"id": current_version['id'],
|
||||
"version_name": current_version['version_name'],
|
||||
"created_at": current_version['created_at'].isoformat().replace('+00:00', 'Z')
|
||||
},
|
||||
"previous_version": None,
|
||||
"changes": changes,
|
||||
"summary": summary,
|
||||
}
|
||||
|
||||
# Check if current version has config content
|
||||
if not current_version['config_content']:
|
||||
# Special handling for restore versions - they might not have content yet
|
||||
@@ -3566,17 +3696,19 @@ async def confirm_restore_config_version(
|
||||
# UPDATE existing frontend (ALL 8 parsed fields)
|
||||
# CRITICAL FIX: Include maxconn and timeout_client so UI shows restored values
|
||||
await conn.execute("""
|
||||
UPDATE frontends
|
||||
SET bind_address = $1, bind_port = $2, default_backend = $3,
|
||||
UPDATE frontends
|
||||
SET bind_address = $1, bind_port = $2, default_backend = $3,
|
||||
mode = $4, ssl_enabled = $5, ssl_port = $6,
|
||||
maxconn = $7, timeout_client = $8,
|
||||
log_format = $11, filters = $12,
|
||||
updated_at = CURRENT_TIMESTAMP, last_config_status = 'PENDING'
|
||||
WHERE id = $9 AND cluster_id = $10
|
||||
""",
|
||||
""",
|
||||
parsed_fe.bind_address, parsed_fe.bind_port, parsed_fe.default_backend,
|
||||
parsed_fe.mode, parsed_fe.ssl_enabled, parsed_fe.ssl_port,
|
||||
parsed_fe.maxconn, parsed_fe.timeout_client,
|
||||
fe_id, cluster_id
|
||||
fe_id, cluster_id,
|
||||
parsed_fe.log_format, parsed_fe.filters # Issue #38
|
||||
)
|
||||
changes_summary["frontends_updated"] += 1
|
||||
logger.info(f"RESTORE: Updated frontend '{parsed_fe.name}' (SSL: {parsed_fe.ssl_enabled}, maxconn: {parsed_fe.maxconn})")
|
||||
@@ -3584,16 +3716,17 @@ async def confirm_restore_config_version(
|
||||
# CREATE new frontend (ALL 8 parsed fields)
|
||||
# CRITICAL FIX: Include maxconn and timeout_client so UI shows restored values
|
||||
await conn.execute("""
|
||||
INSERT INTO frontends
|
||||
INSERT INTO frontends
|
||||
(name, bind_address, bind_port, default_backend, mode, ssl_enabled, ssl_port,
|
||||
maxconn, timeout_client,
|
||||
cluster_id, is_active, last_config_status, created_at, updated_at)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, TRUE, 'PENDING', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
|
||||
""",
|
||||
cluster_id, log_format, filters, is_active, last_config_status, created_at, updated_at)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, TRUE, 'PENDING', CURRENT_TIMESTAMP, CURRENT_TIMESTAMP)
|
||||
""",
|
||||
parsed_fe.name, parsed_fe.bind_address, parsed_fe.bind_port,
|
||||
parsed_fe.default_backend, parsed_fe.mode, parsed_fe.ssl_enabled, parsed_fe.ssl_port,
|
||||
parsed_fe.maxconn, parsed_fe.timeout_client,
|
||||
cluster_id
|
||||
cluster_id,
|
||||
parsed_fe.log_format, parsed_fe.filters # Issue #38
|
||||
)
|
||||
changes_summary["frontends_created"] += 1
|
||||
logger.info(f"RESTORE: Created frontend '{parsed_fe.name}' (SSL: {parsed_fe.ssl_enabled}, maxconn: {parsed_fe.maxconn})")
|
||||
@@ -4226,6 +4359,20 @@ async def undo_reject_config_version(
|
||||
),
|
||||
)
|
||||
|
||||
# HA/VIP (Issue #27): vip-* versions are owned by the VIP entity, so undo is handled
|
||||
# by the VIP router — it re-stages the rejected change as PENDING from the version's
|
||||
# captured pending_state (reactivating the VIP if a rejected create soft-deleted it,
|
||||
# or re-applying a rejected edit). Returns an error string if it can't (e.g. the
|
||||
# name/address/VRID was reused since reject) -> surface a clean 409.
|
||||
if version_name.startswith('vip-'):
|
||||
from routers.vip import restore_vip_from_rejected_version
|
||||
err = await restore_vip_from_rejected_version(conn, version_id)
|
||||
await close_database_connection(conn)
|
||||
if err:
|
||||
raise HTTPException(status_code=409, detail=f"Cannot undo this VIP change — {err}.")
|
||||
return {"message": "VIP change restored to PENDING — review and Apply it from Apply Management",
|
||||
"version_name": version_name}
|
||||
|
||||
async with conn.transaction():
|
||||
# Mark the version as PENDING again
|
||||
await conn.execute("""
|
||||
@@ -4902,9 +5049,17 @@ async def reject_all_pending_changes(cluster_id: int, authorization: str = Heade
|
||||
await validate_user_cluster_access(current_user['id'], cluster_id, conn)
|
||||
|
||||
# Get all pending config versions for this cluster (CRITICAL: Include metadata for rollback!)
|
||||
# HA/VIP (Issue #27): exclude vip-* versions — they are rejected/reverted by the
|
||||
# VIP reject endpoint (which restores keepalived state), not the generic rollback.
|
||||
# ORDER BY created_at ASC: the rollback loop dedups per entity and keeps the FIRST-processed
|
||||
# snapshot, so the OLDEST snapshot must win — its old_values hold the true pre-change state.
|
||||
# Critical when one entity has multiple pending versions (e.g. cluster ACME enable->disable->enable):
|
||||
# rolling back to the oldest restores the original acme_enabled. (Matches the apply SELECT, which
|
||||
# already orders created_at ASC.)
|
||||
pending_versions = await conn.fetch("""
|
||||
SELECT id, version_name, metadata FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'PENDING'
|
||||
WHERE cluster_id = $1 AND status = 'PENDING' AND version_name NOT LIKE 'vip-%'
|
||||
ORDER BY created_at ASC
|
||||
""", cluster_id)
|
||||
|
||||
# CRITICAL FIX: Detect and clean orphan config versions
|
||||
@@ -5116,11 +5271,12 @@ async def reject_all_pending_changes(cluster_id: int, authorization: str = Heade
|
||||
)
|
||||
|
||||
# Mark all pending config versions as REJECTED (don't delete them)
|
||||
# HA/VIP (Issue #27): leave vip-* versions to the VIP reject endpoint.
|
||||
rejected_count = len(pending_versions)
|
||||
await conn.execute("""
|
||||
UPDATE config_versions
|
||||
UPDATE config_versions
|
||||
SET status = 'REJECTED'
|
||||
WHERE cluster_id = $1 AND status = 'PENDING'
|
||||
WHERE cluster_id = $1 AND status = 'PENDING' AND version_name NOT LIKE 'vip-%'
|
||||
""", cluster_id)
|
||||
|
||||
# Update WAF rules status to APPLIED (rolled back)
|
||||
|
||||
+101
-12
@@ -3,7 +3,7 @@ Configuration Management and Validation API
|
||||
Provides endpoints for HAProxy configuration validation, templates, and optimization
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Header, Request
|
||||
from fastapi import APIRouter, HTTPException, Header, Request, Depends
|
||||
from pydantic import BaseModel
|
||||
from typing import Dict, List, Any, Optional
|
||||
import logging
|
||||
@@ -18,7 +18,7 @@ from utils.config_templates import (
|
||||
)
|
||||
from utils.haproxy_config_parser import parse_haproxy_config
|
||||
from utils.logging_config import log_with_correlation, PerformanceLogger
|
||||
from auth_middleware import get_current_user_from_token
|
||||
from auth_middleware import get_current_user_from_token, require_authenticated_user
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
|
||||
router = APIRouter(prefix="/api/config", tags=["Configuration Management"])
|
||||
@@ -62,7 +62,7 @@ class ConfigOptimizationRequest(BaseModel):
|
||||
optimization_level: str = "balanced" # conservative, balanced, aggressive
|
||||
target_environment: str = "production" # development, staging, production
|
||||
|
||||
@router.post("/validate")
|
||||
@router.post("/validate", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): was optional-auth; runs HAProxy validator on caller input
|
||||
async def validate_configuration(
|
||||
request: ConfigValidationRequest,
|
||||
current_user: dict = None,
|
||||
@@ -203,7 +203,7 @@ async def get_template_details(template_id: str):
|
||||
)
|
||||
raise HTTPException(status_code=500, detail=f"Failed to get template: {str(e)}")
|
||||
|
||||
@router.post("/templates/{template_id}/generate")
|
||||
@router.post("/templates/{template_id}/generate", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): was optional-auth
|
||||
async def generate_configuration(
|
||||
template_id: str,
|
||||
request: TemplateGenerationRequest,
|
||||
@@ -271,7 +271,7 @@ async def generate_configuration(
|
||||
detail=f"Configuration generation failed: {str(e)}"
|
||||
)
|
||||
|
||||
@router.post("/optimize")
|
||||
@router.post("/optimize", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): was optional-auth
|
||||
async def optimize_configuration(
|
||||
request: ConfigOptimizationRequest,
|
||||
current_user: dict = None,
|
||||
@@ -855,6 +855,9 @@ async def parse_bulk_config(
|
||||
"response_headers": frontend.response_headers,
|
||||
"options": frontend.options,
|
||||
"tcp_request_rules": frontend.tcp_request_rules,
|
||||
# Issue #38: SPOE filters + frontend log-format
|
||||
"log_format": frontend.log_format,
|
||||
"filters": frontend.filters,
|
||||
# CRITICAL: SSL Advanced Options (parsed from bind directive)
|
||||
"ssl_alpn": frontend.ssl_alpn,
|
||||
"ssl_npn": frontend.ssl_npn,
|
||||
@@ -1089,7 +1092,69 @@ async def parse_bulk_config(
|
||||
|
||||
# Add auto-assignment info at the beginning
|
||||
enhanced_warnings = ssl_auto_assign_info + enhanced_warnings
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# Issue #38: SPOE pre-flight advisories. Surface, at preview time, the
|
||||
# SPOE configurations that would FAIL HAProxy's `haproxy -c` at apply so
|
||||
# the operator sees them BEFORE importing. Cluster-aware: the referenced
|
||||
# SPOE engine config (e.g. coraza.cfg) is a sibling of the cluster's
|
||||
# haproxy_config_path, which HAProxy OpenManager does not provision.
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
try:
|
||||
_cfg_path = await conn.fetchval(
|
||||
"SELECT haproxy_config_path FROM haproxy_clusters WHERE id = $1",
|
||||
request.cluster_id,
|
||||
) or "/etc/haproxy/haproxy.cfg"
|
||||
_cfg_dir = _cfg_path.rsplit("/", 1)[0] or "/etc/haproxy"
|
||||
for _fe in frontends_data:
|
||||
_rh = _fe.get("request_headers") or ""
|
||||
_filters = _fe.get("filters") or ""
|
||||
# engines declared by `filter spoe engine <name> config <path>`
|
||||
_declared_engines = set(re.findall(
|
||||
r"filter\s+spoe\s+engine\s+(\S+)", _filters, re.IGNORECASE))
|
||||
# engines referenced by `... send-spoe-group <name> <group>`
|
||||
_used_engines = set(re.findall(
|
||||
r"send-spoe-group\s+(\S+)", _rh, re.IGNORECASE))
|
||||
_missing = _used_engines - _declared_engines
|
||||
if _missing:
|
||||
enhanced_warnings.append(
|
||||
f"⚠️ Frontend '{_fe['name']}': 'send-spoe-group' references SPOE "
|
||||
f"engine(s) {', '.join(sorted(_missing))} but no matching "
|
||||
f"'filter spoe engine <name> ...' line was found. HAProxy will "
|
||||
f"reject this at apply with \"unable to find SPOE engine\". Add the "
|
||||
f"filter line to this frontend."
|
||||
)
|
||||
for _path in re.findall(
|
||||
r"filter\s+spoe\s+engine\s+\S+\s+config\s+(\S+)",
|
||||
_filters, re.IGNORECASE):
|
||||
enhanced_warnings.append(
|
||||
f"ℹ️ Frontend '{_fe['name']}': SPOE engine config '{_path}' and its "
|
||||
f"SPOA backend must exist on the HAProxy host (cluster config dir: "
|
||||
f"{_cfg_dir}). HAProxy OpenManager preserves the filter directive but "
|
||||
f"does not provision these files; otherwise 'haproxy -c' fails at apply."
|
||||
)
|
||||
# Issue #38 follow-up: ACL `-f <file>` pattern-file advisory.
|
||||
# Scan only the structured rule fields (acl/use_backend) —
|
||||
# request_headers/tcp_request_rules were always free-form and
|
||||
# warning on them now would add new noise for existing users.
|
||||
_pattern_paths = []
|
||||
for _rule in (_fe.get("acl_rules") or []) + (_fe.get("use_backend_rules") or []):
|
||||
if isinstance(_rule, str):
|
||||
_pattern_paths.extend(
|
||||
re.findall(r"(?:^|\s)-f\s+(\S+)", _rule))
|
||||
if _pattern_paths:
|
||||
_uniq = sorted(set(_pattern_paths))
|
||||
enhanced_warnings.append(
|
||||
f"ℹ️ Frontend '{_fe['name']}': ACL/routing rules reference pattern "
|
||||
f"file(s) {', '.join(_uniq)}. Each file must exist at that exact path "
|
||||
f"on every HAProxy host in the cluster (cluster config dir: {_cfg_dir}) "
|
||||
f"— HAProxy OpenManager does not create or distribute pattern files. "
|
||||
f"A missing file fails safely at 'haproxy -c' (previous config keeps "
|
||||
f"running)."
|
||||
)
|
||||
except Exception as _spoe_adv_err:
|
||||
logger.warning(f"SPOE advisory generation skipped: {_spoe_adv_err}")
|
||||
|
||||
# BULK IMPORT MVP: Check existing entities for UPSERT detection
|
||||
# Mark each entity as new or update for UI display
|
||||
# CRITICAL: Only mark as UPDATE if there are actual field changes
|
||||
@@ -1150,7 +1215,17 @@ async def parse_bulk_config(
|
||||
if frontend.get("tcp_request_rules") and frontend["tcp_request_rules"] != existing["tcp_request_rules"]:
|
||||
has_changes = True
|
||||
changes["tcp_request_rules"] = {"old": existing["tcp_request_rules"], "new": frontend["tcp_request_rules"]}
|
||||
|
||||
# Issue #38: SPOE filters + log-format change detection. REQUIRED for
|
||||
# persistence (not just display): without it, an import that only adds
|
||||
# a `filter`/`log-format` to an existing frontend would be flagged
|
||||
# "no change" and the directive would never be written to the DB.
|
||||
if frontend.get("log_format") and frontend["log_format"] != existing.get("log_format"):
|
||||
has_changes = True
|
||||
changes["log_format"] = {"old": existing.get("log_format"), "new": frontend["log_format"]}
|
||||
if frontend.get("filters") and frontend["filters"] != existing.get("filters"):
|
||||
has_changes = True
|
||||
changes["filters"] = {"old": existing.get("filters"), "new": frontend["filters"]}
|
||||
|
||||
# CRITICAL: SSL Advanced Options change detection
|
||||
if frontend.get("ssl_alpn") is not None and frontend.get("ssl_alpn") != existing.get("ssl_alpn"):
|
||||
has_changes = True
|
||||
@@ -2094,7 +2169,18 @@ async def bulk_create_entities(
|
||||
update_fields.append(f"options = ${param_index}")
|
||||
update_values.append(frontend_data["options"])
|
||||
param_index += 1
|
||||
|
||||
|
||||
# Issue #38: SPOE filters + frontend log-format (merge strategy)
|
||||
if frontend_data.get("log_format") and frontend_data["log_format"] != existing_full.get("log_format"):
|
||||
update_fields.append(f"log_format = ${param_index}")
|
||||
update_values.append(frontend_data["log_format"])
|
||||
param_index += 1
|
||||
|
||||
if frontend_data.get("filters") and frontend_data["filters"] != existing_full.get("filters"):
|
||||
update_fields.append(f"filters = ${param_index}")
|
||||
update_values.append(frontend_data["filters"])
|
||||
param_index += 1
|
||||
|
||||
# CRITICAL FIX: Update SSL advanced options (alpn, npn, ciphers, etc.)
|
||||
# These are parsed from bind directive and should be preserved in database
|
||||
if "ssl_alpn" in frontend_data and frontend_data.get("ssl_alpn") != existing_full.get("ssl_alpn"):
|
||||
@@ -2214,9 +2300,10 @@ async def bulk_create_entities(
|
||||
timeout_client, timeout_http_request, maxconn,
|
||||
request_headers, response_headers, tcp_request_rules, options,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
cluster_id, acl_rules, use_backend_rules, redirect_rules, updated_at
|
||||
)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, CURRENT_TIMESTAMP)
|
||||
cluster_id, acl_rules, use_backend_rules, redirect_rules,
|
||||
log_format, filters, updated_at
|
||||
)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, CURRENT_TIMESTAMP)
|
||||
RETURNING id
|
||||
""",
|
||||
frontend_data["name"],
|
||||
@@ -2257,7 +2344,9 @@ async def bulk_create_entities(
|
||||
request.cluster_id,
|
||||
json.dumps(frontend_data.get("acl_rules", [])), # acl_rules
|
||||
json.dumps(frontend_data.get("use_backend_rules", [])), # use_backend_rules
|
||||
json.dumps([]) # redirect_rules
|
||||
json.dumps([]), # redirect_rules
|
||||
frontend_data.get("log_format"), # Issue #38
|
||||
frontend_data.get("filters") # Issue #38
|
||||
)
|
||||
|
||||
created_entities["frontends"].append({
|
||||
|
||||
@@ -201,13 +201,15 @@ async def get_pending_config_requests(agent_name: str, x_api_key: Optional[str]
|
||||
Called during heartbeat.
|
||||
"""
|
||||
try:
|
||||
# Validate agent API key
|
||||
# Validate agent API key — MANDATORY (GHSA-3p5c-m5m4-mjpx). Deployed agents
|
||||
# always send X-API-Key; an absent/invalid key is unauthenticated. This
|
||||
# endpoint also mutates state (marks requests 'processing'), so a keyless
|
||||
# caller could otherwise starve the real agent.
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
|
||||
if x_api_key and not agent_auth:
|
||||
logger.warning(f"Invalid API key provided by agent '{agent_name}' for pending requests")
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
if not agent_auth:
|
||||
logger.warning(f"Missing/invalid API key from '{agent_name}' for pending requests")
|
||||
raise HTTPException(status_code=401, detail="Authentication required")
|
||||
|
||||
conn = await get_database_connection()
|
||||
|
||||
# Get pending requests
|
||||
|
||||
@@ -0,0 +1,375 @@
|
||||
"""
|
||||
CSR (Certificate Signing Request) endpoints (v1.9.0).
|
||||
|
||||
Generate a private key + CSR in-app, download the CSR PEM, have it signed by
|
||||
an external CA, then import the signed certificate — which creates a normal
|
||||
ssl_certificates row that flows through the existing pipeline
|
||||
(config version → Apply Management → agent pull).
|
||||
|
||||
Security posture:
|
||||
- All endpoints enforce ssl.* permissions explicitly (including the read
|
||||
endpoints — deliberately stricter than the legacy cert detail route).
|
||||
- The private key is NEVER returned by any endpoint here; after import it is
|
||||
reachable only via the existing certificate detail route.
|
||||
- Key generation is offloaded to a thread (RSA-4096 takes seconds; the
|
||||
backend runs a single-worker event loop by default) and rate-limited
|
||||
per user via the user_activity_logs COUNT pattern (acme_diagnostics
|
||||
precedent — slowapi is not registered on the app).
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Request, Header
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from auth_middleware import get_current_user_from_token, check_user_permission
|
||||
from models.csr import SSLCSRCreate, SSLCSRImport
|
||||
from services import csr_service, ssl_service
|
||||
from routers.ssl import _assert_safe_cert_name, validate_user_cluster_access
|
||||
from utils.activity_log import log_user_activity
|
||||
|
||||
router = APIRouter(prefix="/api/ssl/csrs", tags=["SSL CSRs"])
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_RATE_LIMIT_CREATE_PER_MIN = 10
|
||||
|
||||
# Columns exposed to the API — private_key_pem is deliberately absent so a
|
||||
# future `SELECT *` refactor cannot silently start leaking it.
|
||||
_CSR_LIST_COLUMNS = """
|
||||
c.id, c.name, c.common_name, c.subject, c.sans, c.key_algorithm,
|
||||
c.status, c.ssl_certificate_id, c.completed_at, c.created_at, c.updated_at,
|
||||
s.name AS certificate_name, u.username AS created_by_username
|
||||
"""
|
||||
|
||||
_INT32_MAX = 2_147_483_647
|
||||
|
||||
|
||||
def _client_ip(request: Optional[Request]) -> Optional[str]:
|
||||
try:
|
||||
return str(request.client.host) if request and request.client else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _user_agent(request: Optional[Request]) -> Optional[str]:
|
||||
try:
|
||||
return request.headers.get("user-agent") if request else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
async def _require(authorization: Optional[str], action: str):
|
||||
"""Authenticate + enforce ssl.<action>; returns current_user or raises 401/403."""
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
ok = await check_user_permission(current_user["id"], "ssl", action, current_user=current_user)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=403, detail=f"Insufficient permissions: ssl.{action} required")
|
||||
return current_user
|
||||
|
||||
|
||||
def _assert_int32_id(csr_id: int) -> None:
|
||||
"""ssl_csrs.id is int4 — an out-of-range path param would surface as an
|
||||
asyncpg DataError 500 (Bulgu #96 precedent); return a clean 404 instead."""
|
||||
if csr_id < 1 or csr_id > _INT32_MAX:
|
||||
raise HTTPException(status_code=404, detail="CSR not found")
|
||||
|
||||
|
||||
def _assert_valid_cluster_id(cluster_id: int) -> None:
|
||||
"""Same int4 guard for body-supplied cluster ids: haproxy_clusters.id is
|
||||
SERIAL/int4, so an out-of-range value would raise asyncpg DataError inside
|
||||
validate_user_cluster_access and surface as a 500 with the raw driver
|
||||
error. Fail with the same clean 404 the cluster lookup itself produces."""
|
||||
if not isinstance(cluster_id, int) or cluster_id < 1 or cluster_id > _INT32_MAX:
|
||||
raise HTTPException(status_code=404, detail="Cluster not found")
|
||||
|
||||
|
||||
async def _enforce_create_rate_limit(conn, user_id: int) -> None:
|
||||
"""Per-user per-minute limit on key generation, counted against the
|
||||
csr_create audit-log action (acme_diagnostics _enforce_rate_limit pattern,
|
||||
backed by the (user_id, action, created_at DESC) composite index)."""
|
||||
cnt = await conn.fetchval(
|
||||
"""
|
||||
SELECT COUNT(*)
|
||||
FROM user_activity_logs
|
||||
WHERE user_id = $1
|
||||
AND action = 'csr_create'
|
||||
AND created_at >= NOW() - INTERVAL '60 seconds'
|
||||
""",
|
||||
user_id,
|
||||
)
|
||||
if cnt is not None and cnt >= _RATE_LIMIT_CREATE_PER_MIN:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
f"Rate limit exceeded: at most {_RATE_LIMIT_CREATE_PER_MIN} "
|
||||
"CSRs may be created per minute"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@router.post("")
|
||||
async def create_csr(payload: SSLCSRCreate, request: Request, authorization: Optional[str] = Header(None)):
|
||||
"""Generate a private key + CSR. Returns the CSR PEM immediately (so the
|
||||
UI can show copy/download in one round trip) — never the private key."""
|
||||
current_user = await _require(authorization, "create")
|
||||
conn = None
|
||||
try:
|
||||
# Belt and braces on top of the model validator — same duplication
|
||||
# convention as the certificate create route.
|
||||
_assert_safe_cert_name(payload.name)
|
||||
|
||||
conn = await get_database_connection()
|
||||
await _enforce_create_rate_limit(conn, current_user["id"])
|
||||
|
||||
# Fail fast on a taken name BEFORE burning CPU on key generation;
|
||||
# insert_csr_row re-checks and the partial unique index closes the race.
|
||||
await csr_service.assert_csr_name_available(conn, payload.name)
|
||||
|
||||
bundle = await asyncio.to_thread(csr_service.generate_csr_bundle, payload)
|
||||
csr_id = await csr_service.insert_csr_row(conn, payload, bundle, current_user["id"])
|
||||
|
||||
row = await conn.fetchrow(
|
||||
f"""
|
||||
SELECT {_CSR_LIST_COLUMNS}, c.csr_pem
|
||||
FROM ssl_csrs c
|
||||
LEFT JOIN ssl_certificates s ON c.ssl_certificate_id = s.id
|
||||
LEFT JOIN users u ON c.created_by = u.id
|
||||
WHERE c.id = $1
|
||||
""",
|
||||
csr_id,
|
||||
)
|
||||
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"],
|
||||
action='csr_create',
|
||||
resource_type='ssl_csr',
|
||||
resource_id=str(csr_id),
|
||||
details={
|
||||
'csr_name': payload.name,
|
||||
'common_name': payload.common_name,
|
||||
'sans': bundle['sans'],
|
||||
'key_algorithm': payload.key_algorithm,
|
||||
},
|
||||
ip_address=_client_ip(request),
|
||||
user_agent=_user_agent(request),
|
||||
)
|
||||
|
||||
return {
|
||||
"message": f"CSR '{payload.name}' created successfully",
|
||||
"csr": csr_service.csr_row_to_dict(row, include_pem=True),
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error creating CSR: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_csrs(authorization: Optional[str] = Header(None)):
|
||||
"""List CSRs (no PEM payloads — fetch the detail route for the CSR PEM).
|
||||
Cluster-agnostic: a CSR binds to clusters only at import time."""
|
||||
await _require(authorization, "read")
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
rows = await conn.fetch(
|
||||
f"""
|
||||
SELECT {_CSR_LIST_COLUMNS}
|
||||
FROM ssl_csrs c
|
||||
LEFT JOIN ssl_certificates s ON c.ssl_certificate_id = s.id
|
||||
LEFT JOIN users u ON c.created_by = u.id
|
||||
ORDER BY c.created_at DESC
|
||||
"""
|
||||
)
|
||||
return [csr_service.csr_row_to_dict(r) for r in rows]
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error listing CSRs: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{csr_id}")
|
||||
async def get_csr(csr_id: int, authorization: Optional[str] = Header(None)):
|
||||
"""CSR detail including the CSR PEM. The private key is never included."""
|
||||
await _require(authorization, "read")
|
||||
_assert_int32_id(csr_id)
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
row = await conn.fetchrow(
|
||||
f"""
|
||||
SELECT {_CSR_LIST_COLUMNS}, c.csr_pem
|
||||
FROM ssl_csrs c
|
||||
LEFT JOIN ssl_certificates s ON c.ssl_certificate_id = s.id
|
||||
LEFT JOIN users u ON c.created_by = u.id
|
||||
WHERE c.id = $1
|
||||
""",
|
||||
csr_id,
|
||||
)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404, detail="CSR not found")
|
||||
return csr_service.csr_row_to_dict(row, include_pem=True)
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error fetching CSR {csr_id}: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/{csr_id}/import")
|
||||
async def import_csr_certificate(
|
||||
csr_id: int,
|
||||
payload: SSLCSRImport,
|
||||
request: Request,
|
||||
authorization: Optional[str] = Header(None),
|
||||
):
|
||||
"""Import the CA-signed certificate for a pending CSR. Creates an
|
||||
ssl_certificates row (source='csr', PENDING) and stages one config
|
||||
version per affected cluster — the operator applies manually."""
|
||||
current_user = await _require(authorization, "create")
|
||||
_assert_int32_id(csr_id)
|
||||
conn = None
|
||||
try:
|
||||
if payload.name:
|
||||
_assert_safe_cert_name(payload.name)
|
||||
|
||||
conn = await get_database_connection()
|
||||
|
||||
if not payload.is_global:
|
||||
for cluster_id in payload.cluster_ids or []:
|
||||
_assert_valid_cluster_id(cluster_id)
|
||||
await validate_user_cluster_access(current_user["id"], cluster_id, conn)
|
||||
|
||||
result = await csr_service.import_signed_certificate(
|
||||
conn, csr_id, payload, current_user["id"]
|
||||
)
|
||||
cert_id = result["certificate_id"]
|
||||
|
||||
if payload.is_global:
|
||||
cluster_rows = await conn.fetch(
|
||||
"SELECT id FROM haproxy_clusters WHERE is_active = TRUE"
|
||||
)
|
||||
affected_clusters = [r['id'] for r in cluster_rows]
|
||||
else:
|
||||
affected_clusters = payload.cluster_ids or []
|
||||
|
||||
# Post-commit staging — a config-generation failure never rolls back
|
||||
# the certificate (same semantics as the manual create flow).
|
||||
sync_results = await ssl_service.stage_ssl_config_versions(
|
||||
conn, cert_id, affected_clusters, action='create',
|
||||
created_by=current_user["id"],
|
||||
)
|
||||
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"],
|
||||
action='create',
|
||||
resource_type='ssl_certificate',
|
||||
resource_id=str(cert_id),
|
||||
details={
|
||||
'certificate_name': result['certificate_name'],
|
||||
'domain': result.get('primary_domain', 'unknown'),
|
||||
'via': 'csr',
|
||||
'csr_id': csr_id,
|
||||
'usage_type': payload.usage_type,
|
||||
'is_global': payload.is_global,
|
||||
'cluster_ids': payload.cluster_ids,
|
||||
'warnings': result['warnings'],
|
||||
},
|
||||
ip_address=_client_ip(request),
|
||||
user_agent=_user_agent(request),
|
||||
)
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"],
|
||||
action='csr_import',
|
||||
resource_type='ssl_csr',
|
||||
resource_id=str(csr_id),
|
||||
details={
|
||||
'certificate_id': cert_id,
|
||||
'certificate_name': result['certificate_name'],
|
||||
},
|
||||
ip_address=_client_ip(request),
|
||||
user_agent=_user_agent(request),
|
||||
)
|
||||
|
||||
return {
|
||||
"message": (
|
||||
f"Certificate '{result['certificate_name']}' imported "
|
||||
"successfully. Go to Apply Management to deploy."
|
||||
),
|
||||
"certificate_id": cert_id,
|
||||
"warnings": result["warnings"],
|
||||
"sync_results": sync_results,
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error importing signed certificate for CSR {csr_id}: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.delete("/{csr_id}")
|
||||
async def delete_csr(csr_id: int, request: Request, authorization: Optional[str] = Header(None)):
|
||||
"""Hard delete. For a pending CSR this permanently destroys the private
|
||||
key (any certificate later signed from that CSR becomes unusable); for a
|
||||
completed CSR it only removes history — the imported certificate is not
|
||||
affected (the FK points csr → cert)."""
|
||||
current_user = await _require(authorization, "delete")
|
||||
_assert_int32_id(csr_id)
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
async with conn.transaction():
|
||||
# FOR UPDATE serialises against an in-flight import of the same CSR.
|
||||
row = await conn.fetchrow(
|
||||
"SELECT id, name, status FROM ssl_csrs WHERE id = $1 FOR UPDATE",
|
||||
csr_id,
|
||||
)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404, detail="CSR not found")
|
||||
await conn.execute("DELETE FROM ssl_csrs WHERE id = $1", csr_id)
|
||||
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"],
|
||||
action='delete',
|
||||
resource_type='ssl_csr',
|
||||
resource_id=str(csr_id),
|
||||
details={'csr_name': row['name'], 'status': row['status']},
|
||||
ip_address=_client_ip(request),
|
||||
user_agent=_user_agent(request),
|
||||
)
|
||||
|
||||
if row['status'] == 'pending':
|
||||
message = (
|
||||
f"CSR '{row['name']}' deleted — its private key has been "
|
||||
"permanently destroyed."
|
||||
)
|
||||
else:
|
||||
message = (
|
||||
f"CSR '{row['name']}' deleted (history only) — the imported "
|
||||
"certificate is not affected."
|
||||
)
|
||||
return {"message": message}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error deleting CSR {csr_id}: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
@@ -1,4 +1,5 @@
|
||||
from fastapi import APIRouter, HTTPException, Header
|
||||
from fastapi import APIRouter, HTTPException, Header, Depends
|
||||
from auth_middleware import require_authenticated_user
|
||||
from typing import Optional
|
||||
from datetime import datetime
|
||||
import logging
|
||||
@@ -11,7 +12,7 @@ from agent_notifications import get_cluster_agents_status
|
||||
router = APIRouter(prefix="/api", tags=["dashboard", "pools"])
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@router.get("/dashboard/overview")
|
||||
@router.get("/dashboard/overview", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): leaked cluster/pool/agent stats, names, health & alerts anonymously (auth was optional)
|
||||
async def get_dashboard_overview(cluster_id: Optional[int] = None, authorization: str = Header(None)):
|
||||
"""Get dashboard overview with comprehensive statistics, optionally filtered by cluster"""
|
||||
try:
|
||||
@@ -238,7 +239,7 @@ async def get_dashboard_overview(cluster_id: Optional[int] = None, authorization
|
||||
logger.error(f"Error fetching dashboard overview: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/dashboard/stats")
|
||||
@router.get("/dashboard/stats", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): aggregate cluster/agent counts
|
||||
async def get_dashboard_stats():
|
||||
"""Get dashboard statistics"""
|
||||
try:
|
||||
@@ -279,7 +280,7 @@ async def get_dashboard_stats():
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/pools")
|
||||
@router.get("/pools", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): pool names/env/counts
|
||||
async def get_pools():
|
||||
"""Get all HAProxy cluster pools"""
|
||||
try:
|
||||
@@ -350,7 +351,7 @@ async def get_pools():
|
||||
logger.error(f"Error fetching pools: {e}")
|
||||
return {"pools": []}
|
||||
|
||||
@router.get("/haproxy-cluster-pools")
|
||||
@router.get("/haproxy-cluster-pools", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c)
|
||||
async def get_haproxy_cluster_pools():
|
||||
"""Get all HAProxy cluster pools (legacy endpoint)"""
|
||||
# Just call the main pools endpoint
|
||||
@@ -474,7 +475,7 @@ async def update_pool(pool_id: int, pool: PoolUpdate, authorization: str = Heade
|
||||
logger.error(f"Failed to update pool: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Failed to update pool: {str(e)}")
|
||||
|
||||
@router.get("/haproxy-cluster-pools/{pool_id}/agents")
|
||||
@router.get("/haproxy-cluster-pools/{pool_id}/agents", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): full agent inventory — same class as GET /api/agents
|
||||
async def get_pool_agents(pool_id: int):
|
||||
"""Get all agents for a specific pool"""
|
||||
try:
|
||||
@@ -561,7 +562,7 @@ async def get_pool_agents(pool_id: int):
|
||||
logger.error(f"Error fetching pool agents: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Failed to fetch pool agents: {str(e)}")
|
||||
|
||||
@router.get("/haproxy/stats")
|
||||
@router.get("/haproxy/stats", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c)
|
||||
async def get_haproxy_stats(cluster_id: Optional[int] = None):
|
||||
"""Get HAProxy statistics"""
|
||||
try:
|
||||
|
||||
@@ -3,13 +3,22 @@ Dashboard Stats Router
|
||||
API endpoints for HAProxy statistics dashboard
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
from fastapi import APIRouter, HTTPException, Query, Depends
|
||||
from typing import Optional, List
|
||||
import logging
|
||||
|
||||
from services.dashboard_stats_service import dashboard_stats_service
|
||||
from auth_middleware import require_authenticated_user
|
||||
|
||||
router = APIRouter(prefix="/api/dashboard-stats", tags=["dashboard-stats"])
|
||||
# SECURITY (GHSA-3p5c-m5m4-mjpx): this entire router (traffic metrics, backend
|
||||
# health, cluster topology, agent status) was mounted without authentication.
|
||||
# Require a valid JWT on every route. The frontend Dashboard already sends the
|
||||
# operator JWT on these calls, so this is transparent to the UI.
|
||||
router = APIRouter(
|
||||
prefix="/api/dashboard-stats",
|
||||
tags=["dashboard-stats"],
|
||||
dependencies=[Depends(require_authenticated_user)],
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
||||
+69
-12
@@ -117,6 +117,41 @@ def _rule_contradiction_text(rule: Any) -> Optional[str]:
|
||||
return None
|
||||
|
||||
|
||||
def _pattern_file_warnings(
|
||||
acl_rules: Optional[List[Any]] = None,
|
||||
use_backend_rules: Optional[List[Any]] = None,
|
||||
redirect_rules: Optional[List[Any]] = None,
|
||||
) -> List[str]:
|
||||
"""Issue #38 follow-up — non-blocking `-f <file>` pattern-file
|
||||
advisory for the manual frontend API.
|
||||
|
||||
The Bulgu #12 hard reject was removed from the Pydantic models:
|
||||
pattern files are operator-managed host files (same policy as the
|
||||
SPOE `filter ... config <path>` reference preserved since v1.8.8)
|
||||
and the agent's pre-reload `haproxy -c` makes a missing file fail
|
||||
safely. This helper returns one warning listing the unique file
|
||||
paths referenced across the rule fields, or [] when no rule uses
|
||||
`-f` — operators who don't use pattern files see no change.
|
||||
"""
|
||||
paths: List[str] = []
|
||||
for rules in (acl_rules, use_backend_rules, redirect_rules):
|
||||
for rule in rules or []:
|
||||
text = rule if isinstance(rule, str) else (
|
||||
rule.get("condition") if isinstance(rule, dict) else None)
|
||||
if isinstance(text, str):
|
||||
paths.extend(re.findall(r"(?:^|\s)-f\s+(\S+)", text))
|
||||
if not paths:
|
||||
return []
|
||||
uniq = sorted(set(paths))
|
||||
return [
|
||||
f"ACL/routing rules reference pattern file(s) {', '.join(uniq)}. "
|
||||
f"Each file must exist at that exact path on every HAProxy host "
|
||||
f"in the cluster — HAProxy OpenManager does not create or "
|
||||
f"distribute pattern files. A missing file fails safely at "
|
||||
f"'haproxy -c' (the previous config keeps running)."
|
||||
]
|
||||
|
||||
|
||||
def _collect_routing_rule_contradictions(
|
||||
rules: List[Any], origin_label: str,
|
||||
) -> List[Tuple[str, Any]]:
|
||||
@@ -408,6 +443,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -424,6 +460,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -453,6 +490,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -465,6 +503,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -480,6 +519,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -492,6 +532,7 @@ async def get_frontends(
|
||||
ssl_alpn, ssl_npn, ssl_ciphers, ssl_ciphersuites, ssl_min_ver, ssl_max_ver, ssl_strict_sni,
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules,
|
||||
log_format, filters,
|
||||
timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
maxconn, is_active, created_at, updated_at, cluster_id, last_config_status
|
||||
@@ -601,6 +642,8 @@ async def get_frontends(
|
||||
"response_headers": f.get("response_headers"),
|
||||
"options": f.get("options"),
|
||||
"tcp_request_rules": f.get("tcp_request_rules"),
|
||||
"log_format": f.get("log_format"), # Issue #38
|
||||
"filters": f.get("filters"), # Issue #38
|
||||
"timeout_client": f.get("timeout_client"),
|
||||
"timeout_http_request": f.get("timeout_http_request"),
|
||||
"rate_limit": f.get("rate_limit"),
|
||||
@@ -735,18 +778,18 @@ async def create_frontend(frontend: FrontendConfig, request: Request, authorizat
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules, timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
cluster_id, maxconn, updated_at
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, CURRENT_TIMESTAMP)
|
||||
cluster_id, maxconn, log_format, filters, updated_at
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, CURRENT_TIMESTAMP)
|
||||
RETURNING id
|
||||
""", frontend.name, frontend.bind_address, frontend.bind_port,
|
||||
""", frontend.name, frontend.bind_address, frontend.bind_port,
|
||||
frontend.default_backend, frontend.mode, frontend.ssl_enabled,
|
||||
frontend.ssl_certificate_id, ssl_cert_ids_json, frontend.ssl_port, frontend.ssl_cert_path, frontend.ssl_cert, frontend.ssl_verify,
|
||||
frontend.ssl_alpn, frontend.ssl_npn, frontend.ssl_ciphers, frontend.ssl_ciphersuites,
|
||||
frontend.ssl_alpn, frontend.ssl_npn, frontend.ssl_ciphers, frontend.ssl_ciphersuites,
|
||||
frontend.ssl_min_ver, frontend.ssl_max_ver, frontend.ssl_strict_sni,
|
||||
json.dumps(frontend.acl_rules or []), json.dumps(frontend.redirect_rules or []), json.dumps(frontend.use_backend_rules or []),
|
||||
frontend.request_headers, frontend.response_headers, filtered_options, frontend.tcp_request_rules, frontend.timeout_client, frontend.timeout_http_request,
|
||||
frontend.rate_limit, frontend.compression, frontend.log_separate, frontend.monitor_uri,
|
||||
frontend.cluster_id, frontend.maxconn)
|
||||
frontend.cluster_id, frontend.maxconn, frontend.log_format, frontend.filters)
|
||||
|
||||
# If cluster_id provided, create new config version for agents
|
||||
sync_results = []
|
||||
@@ -825,12 +868,19 @@ async def create_frontend(frontend: FrontendConfig, request: Request, authorizat
|
||||
user_agent=request.headers.get('user-agent')
|
||||
)
|
||||
|
||||
return {
|
||||
response: dict = {
|
||||
"message": f"Frontend '{frontend.name}' created successfully",
|
||||
"id": frontend_id,
|
||||
"frontend": frontend.dict(),
|
||||
"sync_results": sync_results
|
||||
}
|
||||
# Issue #38 follow-up — non-blocking pattern-file advisory
|
||||
# (additive field; absent when no rule references `-f`).
|
||||
pattern_warnings = _pattern_file_warnings(
|
||||
frontend.acl_rules, frontend.use_backend_rules, frontend.redirect_rules)
|
||||
if pattern_warnings:
|
||||
response["warnings"] = pattern_warnings
|
||||
return response
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1060,9 +1110,10 @@ async def update_frontend(frontend_id: int, frontend: FrontendConfig, request: R
|
||||
acl_rules = $20, redirect_rules = $21, use_backend_rules = $22,
|
||||
request_headers = $23, response_headers = $24, options = $25, tcp_request_rules = $26, timeout_client = $27, timeout_http_request = $28,
|
||||
rate_limit = $29, compression = $30, log_separate = $31, monitor_uri = $32,
|
||||
cluster_id = $33, maxconn = $34, updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $35
|
||||
""", frontend.name, frontend.bind_address, frontend.bind_port,
|
||||
cluster_id = $33, maxconn = $34, log_format = $35, filters = $36,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $37
|
||||
""", frontend.name, frontend.bind_address, frontend.bind_port,
|
||||
frontend.default_backend, frontend.mode, ssl_enabled,
|
||||
ssl_certificate_id, ssl_cert_ids_json, ssl_port, ssl_cert_path, ssl_cert, ssl_verify,
|
||||
frontend.ssl_alpn, frontend.ssl_npn, frontend.ssl_ciphers, frontend.ssl_ciphersuites,
|
||||
@@ -1070,7 +1121,7 @@ async def update_frontend(frontend_id: int, frontend: FrontendConfig, request: R
|
||||
json.dumps(frontend.acl_rules or []), json.dumps(frontend.redirect_rules or []), json.dumps(frontend.use_backend_rules or []),
|
||||
frontend.request_headers, frontend.response_headers, filtered_options, frontend.tcp_request_rules, frontend.timeout_client, frontend.timeout_http_request,
|
||||
frontend.rate_limit, frontend.compression, frontend.log_separate, frontend.monitor_uri,
|
||||
frontend.cluster_id, frontend.maxconn, frontend_id)
|
||||
frontend.cluster_id, frontend.maxconn, frontend.log_format, frontend.filters, frontend_id)
|
||||
|
||||
# Debug: Check what was actually saved
|
||||
updated_frontend = await conn.fetchrow("""
|
||||
@@ -1124,6 +1175,8 @@ async def update_frontend(frontend_id: int, frontend: FrontendConfig, request: R
|
||||
"response_headers": frontend.response_headers,
|
||||
"options": filtered_options,
|
||||
"tcp_request_rules": frontend.tcp_request_rules,
|
||||
"log_format": frontend.log_format, # Issue #38
|
||||
"filters": frontend.filters, # Issue #38
|
||||
"timeout_client": frontend.timeout_client,
|
||||
"timeout_http_request": frontend.timeout_http_request,
|
||||
"rate_limit": frontend.rate_limit,
|
||||
@@ -1235,8 +1288,12 @@ async def update_frontend(frontend_id: int, frontend: FrontendConfig, request: R
|
||||
# yellow toast on the next refresh. The save SUCCEEDED; the
|
||||
# warnings only flag latent legacy data the operator may
|
||||
# want to clean up at their convenience.
|
||||
if contradiction_warnings:
|
||||
response["warnings"] = contradiction_warnings
|
||||
# Issue #38 follow-up — append the pattern-file advisory to
|
||||
# the same list (additive; empty when no rule uses `-f`).
|
||||
all_warnings = list(contradiction_warnings or []) + _pattern_file_warnings(
|
||||
frontend.acl_rules, frontend.use_backend_rules, frontend.redirect_rules)
|
||||
if all_warnings:
|
||||
response["warnings"] = all_warnings
|
||||
return response
|
||||
except HTTPException:
|
||||
raise
|
||||
|
||||
@@ -3,8 +3,9 @@ Production-Ready Health Check and Monitoring Endpoints
|
||||
Provides comprehensive system health monitoring for Kubernetes and production environments
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from fastapi import APIRouter, HTTPException, Depends
|
||||
from fastapi.responses import JSONResponse
|
||||
from auth_middleware import require_authenticated_user
|
||||
import logging
|
||||
import asyncio
|
||||
import time
|
||||
@@ -74,7 +75,7 @@ async def readiness_probe():
|
||||
logger.error(f"Readiness probe failed: {e}")
|
||||
raise HTTPException(status_code=503, detail=f"Not ready: {str(e)}")
|
||||
|
||||
@router.get("/deep")
|
||||
@router.get("/deep", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): leaks host CPU/mem/disk, DB/redis/python versions, PID
|
||||
async def deep_health_check():
|
||||
"""Comprehensive health check with detailed system information"""
|
||||
global _health_cache
|
||||
@@ -197,7 +198,7 @@ async def deep_health_check():
|
||||
|
||||
raise HTTPException(status_code=503, detail=error_response)
|
||||
|
||||
@router.get("/agents")
|
||||
@router.get("/agents", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): leaks agent names/hostnames
|
||||
async def agents_health():
|
||||
"""Monitor agent connectivity and health status"""
|
||||
try:
|
||||
@@ -261,7 +262,7 @@ async def agents_health():
|
||||
logger.error(f"Agent health check failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Agent health check failed: {str(e)}")
|
||||
|
||||
@router.get("/clusters")
|
||||
@router.get("/clusters", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): leaks cluster names, HAProxy versions, pending counts
|
||||
async def clusters_health():
|
||||
"""Monitor HAProxy cluster health and configuration status"""
|
||||
try:
|
||||
@@ -329,7 +330,7 @@ async def clusters_health():
|
||||
logger.error(f"Cluster health check failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Cluster health check failed: {str(e)}")
|
||||
|
||||
@router.get("/errors")
|
||||
@router.get("/errors", dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): app error metrics, sibling of /deep,/agents,/clusters
|
||||
async def error_statistics():
|
||||
"""Get application error statistics and metrics"""
|
||||
try:
|
||||
|
||||
+438
-48
@@ -1,6 +1,7 @@
|
||||
from fastapi import APIRouter, HTTPException, Header
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from typing import Optional, List
|
||||
from pydantic import BaseModel, Field, field_validator, model_validator
|
||||
from typing import Optional, List, Dict
|
||||
import base64
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
@@ -10,6 +11,40 @@ from datetime import datetime
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from services.acme_service import acme_service
|
||||
from services.haproxy_config import generate_haproxy_config_for_cluster
|
||||
from services.dns_providers import list_providers, is_supported, get_provider, DnsProviderError
|
||||
from utils.dns_credentials import encrypt_dns_credentials, decrypt_dns_credentials
|
||||
|
||||
# Issue #35: DNS-01 challenge methods.
|
||||
_CHALLENGE_TYPES = ("http-01", "dns-01")
|
||||
|
||||
|
||||
async def _dns01_enabled() -> bool:
|
||||
"""Global kill-switch (system_settings acme.dns01_enabled, default False). Read via the ACME
|
||||
settings dict so non-admins never need the admin-only /api/settings/acme endpoint."""
|
||||
try:
|
||||
settings = await acme_service._get_settings()
|
||||
val = settings.get('dns01_enabled')
|
||||
if isinstance(val, str):
|
||||
return val.strip().lower() in ('1', 'true', 'yes', 'on')
|
||||
return bool(val)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
# Per-user sliding-window rate limit for the manual dns-confirm action (soft anti-abuse so a user
|
||||
# can't spam the CA via the confirm button). Per-process; sufficient for a manual UI action.
|
||||
_DNS_CONFIRM_RL: Dict[int, list] = {}
|
||||
_DNS_CONFIRM_LIMIT = 5
|
||||
_DNS_CONFIRM_WINDOW = 60.0
|
||||
|
||||
|
||||
async def _enforce_dns_confirm_rate_limit(user_id: int) -> None:
|
||||
now = time.time()
|
||||
bucket = [t for t in _DNS_CONFIRM_RL.get(user_id, []) if now - t < _DNS_CONFIRM_WINDOW]
|
||||
if len(bucket) >= _DNS_CONFIRM_LIMIT:
|
||||
raise HTTPException(status_code=429, detail="Rate limit exceeded: dns-confirm allowed 5 requests per minute")
|
||||
bucket.append(now)
|
||||
_DNS_CONFIRM_RL[user_id] = bucket
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -25,8 +60,81 @@ class AccountCreate(BaseModel):
|
||||
email: str
|
||||
directory_url: Optional[str] = None
|
||||
tos_agreed: bool = True
|
||||
eab_kid: Optional[str] = None
|
||||
eab_hmac_key: Optional[str] = None
|
||||
# EAB (External Account Binding) for CAs that require it (ZeroSSL, Google). The KID is opaque
|
||||
# (bound only); the HMAC key must be base64url so newAccount's _b64url_decode won't raise a
|
||||
# cryptic binascii error (a common copy mistake is standard-base64 '+'/'/' vs urlsafe '-'/'_').
|
||||
eab_kid: Optional[str] = Field(default=None, max_length=256)
|
||||
eab_hmac_key: Optional[str] = Field(default=None, max_length=512)
|
||||
# Issue #35: per-account default challenge method + DNS provider (for dns-01).
|
||||
challenge_type: str = "http-01"
|
||||
dns_provider: Optional[str] = None
|
||||
|
||||
@field_validator('challenge_type')
|
||||
@classmethod
|
||||
def _validate_challenge_type(cls, v):
|
||||
if v not in _CHALLENGE_TYPES:
|
||||
raise ValueError(f"challenge_type must be one of {_CHALLENGE_TYPES}")
|
||||
return v
|
||||
|
||||
@field_validator('eab_hmac_key')
|
||||
@classmethod
|
||||
def _validate_eab_hmac_key(cls, v):
|
||||
if not v:
|
||||
return v
|
||||
try:
|
||||
base64.urlsafe_b64decode(v + '=' * (-len(v) % 4))
|
||||
except Exception:
|
||||
raise ValueError("eab_hmac_key is not valid base64; copy it exactly from your CA account.")
|
||||
return v
|
||||
|
||||
@field_validator('directory_url')
|
||||
@classmethod
|
||||
def _validate_directory_url(cls, v):
|
||||
# SECURITY (GHSA-3vh4-gvxx-wm2p): reject non-https URLs and literal
|
||||
# non-public IP hosts at the API boundary. The full DNS-based SSRF check
|
||||
# runs at fetch time (acme_service.get_directory -> ssrf_guard).
|
||||
if not v:
|
||||
return v
|
||||
from urllib.parse import urlparse
|
||||
import ipaddress
|
||||
from utils.ssrf_guard import is_public_ip
|
||||
parsed = urlparse(v.strip())
|
||||
if parsed.scheme.lower() != 'https':
|
||||
raise ValueError("directory_url must be an https URL")
|
||||
host = parsed.hostname
|
||||
if not host:
|
||||
raise ValueError("directory_url has no host")
|
||||
try:
|
||||
ipaddress.ip_address(host)
|
||||
is_ip_literal = True
|
||||
except ValueError:
|
||||
is_ip_literal = False
|
||||
if is_ip_literal and not is_public_ip(host):
|
||||
raise ValueError("directory_url must not point to a private/loopback IP address")
|
||||
return v
|
||||
|
||||
@model_validator(mode='after')
|
||||
def _require_provider_for_dns01(self):
|
||||
if self.challenge_type == 'dns-01' and not (self.dns_provider or '').strip():
|
||||
raise ValueError("dns_provider is required when challenge_type is 'dns-01'")
|
||||
return self
|
||||
|
||||
|
||||
class DnsCredentialsUpsert(BaseModel):
|
||||
dns_provider: str = Field(..., min_length=1, max_length=50)
|
||||
credentials: Dict[str, str] = Field(default_factory=dict)
|
||||
|
||||
@field_validator('credentials')
|
||||
@classmethod
|
||||
def _validate_credentials(cls, v):
|
||||
if len(v) > 20:
|
||||
raise ValueError("Too many credential fields")
|
||||
for key, val in v.items():
|
||||
if not isinstance(key, str) or not re.match(r'^[a-zA-Z0-9_]{1,50}$', key):
|
||||
raise ValueError(f"Invalid credential field name: {key!r}")
|
||||
if not isinstance(val, str) or len(val) > 4000:
|
||||
raise ValueError(f"Credential value for {key!r} is missing or too long")
|
||||
return v
|
||||
|
||||
|
||||
class CertificateRequest(BaseModel):
|
||||
@@ -38,6 +146,8 @@ class CertificateRequest(BaseModel):
|
||||
account_id: Optional[int] = None
|
||||
cluster_ids: List[int] = Field(default_factory=list)
|
||||
auto_renew: bool = True
|
||||
# Issue #35: optional override; when None the account's default method is used.
|
||||
challenge_type: Optional[str] = None
|
||||
|
||||
@field_validator('domains')
|
||||
@classmethod
|
||||
@@ -54,7 +164,30 @@ class CertificateRequest(BaseModel):
|
||||
if not _DOMAIN_REGEX.match(d_norm):
|
||||
raise ValueError(f"Invalid domain format: '{d}'")
|
||||
normalized.append(d_norm)
|
||||
return normalized
|
||||
# De-duplicate (case/whitespace variants normalize to the same value) while preserving order,
|
||||
# so we don't submit a redundant SAN to the CA or render duplicate-keyed tags in the UI.
|
||||
return list(dict.fromkeys(normalized))
|
||||
|
||||
@field_validator('challenge_type')
|
||||
@classmethod
|
||||
def _validate_challenge_type(cls, v):
|
||||
if v is not None and v not in _CHALLENGE_TYPES:
|
||||
raise ValueError(f"challenge_type must be one of {_CHALLENGE_TYPES}")
|
||||
return v
|
||||
|
||||
@model_validator(mode='after')
|
||||
def _wildcard_requires_dns01(self):
|
||||
# Static cross-field guard: a wildcard SAN can ONLY be issued via dns-01 (the CA rejects
|
||||
# wildcard over http-01). The runtime dns01_enabled gate + provider resolution happen in the
|
||||
# endpoint (validators can't do async/DB). When challenge_type is None here, the effective
|
||||
# method is resolved from the account in the endpoint, which re-checks this.
|
||||
if any((d or '').startswith('*.') for d in (self.domains or [])):
|
||||
# Only reject when the caller EXPLICITLY chose a non-dns-01 method. When challenge_type is
|
||||
# None, the effective method is resolved from the account in the endpoint, which re-checks
|
||||
# wildcard-requires-dns-01 — so account-default dns-01 inheritance still works for wildcards.
|
||||
if self.challenge_type is not None and self.challenge_type != 'dns-01':
|
||||
raise ValueError("Wildcard certificates require challenge_type 'dns-01'")
|
||||
return self
|
||||
|
||||
|
||||
# --- Account management ---
|
||||
@@ -77,7 +210,7 @@ async def list_accounts(authorization: str = Header(None)):
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
rows = await conn.fetch(
|
||||
"SELECT id, email, directory_url, account_url, status, tos_agreed, eab_kid, created_at, updated_at FROM letsencrypt_accounts ORDER BY id"
|
||||
"SELECT id, email, directory_url, account_url, status, tos_agreed, eab_kid, created_at, updated_at, challenge_type, dns_provider FROM letsencrypt_accounts ORDER BY id"
|
||||
)
|
||||
return [dict(r) for r in rows]
|
||||
finally:
|
||||
@@ -107,16 +240,36 @@ async def create_account(body: AccountCreate, authorization: str = Header(None))
|
||||
eab_kid = body.eab_kid or settings.get('eab_kid', '') or None
|
||||
eab_hmac_key = body.eab_hmac_key or settings.get('eab_hmac_key', '') or None
|
||||
|
||||
# Issue #35: a dns-01 account must name a supported DNS provider.
|
||||
if body.challenge_type == 'dns-01':
|
||||
if not await _dns01_enabled():
|
||||
raise HTTPException(status_code=409, detail="DNS-01 is disabled by an administrator (enable it in Settings).")
|
||||
if not is_supported((body.dns_provider or '').strip()):
|
||||
raise HTTPException(status_code=422, detail=f"Unsupported DNS provider: {body.dns_provider}")
|
||||
|
||||
result = await acme_service.register_account(
|
||||
email=body.email,
|
||||
directory_url=directory_url,
|
||||
tos_agreed=body.tos_agreed,
|
||||
eab_kid=eab_kid,
|
||||
eab_hmac_key=eab_hmac_key,
|
||||
challenge_type=body.challenge_type,
|
||||
dns_provider=(body.dns_provider or None),
|
||||
)
|
||||
return result
|
||||
except HTTPException:
|
||||
# Preserve deliberate status codes (e.g. 409 DNS-01 disabled, 422 unsupported provider) —
|
||||
# the broad except below would otherwise downgrade them all to 400.
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"ACME account registration failed: {e}")
|
||||
# Humanize the common EAB-required failure (ZeroSSL/Google). The ACME error propagates as a
|
||||
# string ("Account registration failed: {<dict>}"), so match the URN substring in str(e).
|
||||
if 'externalaccountrequired' in str(e).lower():
|
||||
raise HTTPException(status_code=400, detail=(
|
||||
"This CA requires External Account Binding (EAB). Enter the EAB Key ID and HMAC Key "
|
||||
"from your ZeroSSL/Google account and retry."
|
||||
))
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
|
||||
@@ -186,6 +339,147 @@ async def remove_account(account_id: int, authorization: str = Header(None)):
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# --- Issue #35: DNS-01 providers + per-account DNS credentials ---
|
||||
|
||||
@router.get("/dns-providers")
|
||||
async def get_dns_providers(authorization: str = Header(None)):
|
||||
"""List supported DNS providers + their credential-field schema (for the UI). Also returns the
|
||||
global dns01_enabled gate so a non-admin cert UI can read it without the admin-only settings API.
|
||||
Authenticated (any user); not admin-only."""
|
||||
from auth_middleware import get_current_user_from_token
|
||||
await get_current_user_from_token(authorization)
|
||||
return {"dns01_enabled": await _dns01_enabled(), "providers": list_providers()}
|
||||
|
||||
|
||||
@router.get("/accounts/{account_id}/dns-credentials")
|
||||
async def get_dns_credentials(account_id: int, authorization: str = Header(None)):
|
||||
"""Masked metadata only — provider + which credential fields are set + updated_at. NEVER returns
|
||||
the ciphertext or any plaintext token. Read-only for any authenticated user (matches list_accounts)."""
|
||||
from auth_middleware import get_current_user_from_token
|
||||
await get_current_user_from_token(authorization)
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
row = await conn.fetchrow(
|
||||
"SELECT dns_provider, credentials_encrypted, updated_at FROM letsencrypt_account_dns_credentials WHERE account_id = $1",
|
||||
account_id,
|
||||
)
|
||||
if not row:
|
||||
return {"configured": False, "dns_provider": None, "credential_fields_present": [], "updated_at": None}
|
||||
present = []
|
||||
decrypted = decrypt_dns_credentials(row["credentials_encrypted"])
|
||||
if isinstance(decrypted, dict):
|
||||
present = sorted(decrypted.keys())
|
||||
return {
|
||||
"configured": True,
|
||||
"dns_provider": row["dns_provider"],
|
||||
"credential_fields_present": present,
|
||||
"updated_at": row["updated_at"],
|
||||
}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.put("/accounts/{account_id}/dns-credentials")
|
||||
async def upsert_dns_credentials(account_id: int, body: DnsCredentialsUpsert, authorization: str = Header(None)):
|
||||
"""Store (encrypted) DNS provider credentials for an account. Admin-only. Verifies the
|
||||
credentials against the provider BEFORE persisting; returns a sanitized result (never the token)."""
|
||||
from auth_middleware import get_current_user_from_token
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
if not current_user.get('is_admin', False):
|
||||
raise HTTPException(status_code=403, detail="Admin access required")
|
||||
provider_name = body.dns_provider.strip()
|
||||
if not is_supported(provider_name):
|
||||
raise HTTPException(status_code=422, detail=f"Unsupported DNS provider: {provider_name}")
|
||||
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
exists = await conn.fetchval("SELECT 1 FROM letsencrypt_accounts WHERE id = $1", account_id)
|
||||
if not exists:
|
||||
raise HTTPException(status_code=404, detail="ACME account not found")
|
||||
|
||||
# Verify credentials synchronously; only persist on success. The detail is user-safe.
|
||||
try:
|
||||
provider = get_provider(provider_name, dict(body.credentials))
|
||||
verify = await provider.verify_credentials()
|
||||
except DnsProviderError as exc:
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception:
|
||||
# Defensive: never let a provider-internal exception string (which could echo creds in a
|
||||
# future provider) reach the client. Always a sanitized 422.
|
||||
raise HTTPException(status_code=422, detail="DNS provider credential verification failed")
|
||||
if not verify.get("ok"):
|
||||
raise HTTPException(status_code=422, detail=verify.get("detail") or "DNS provider credential verification failed")
|
||||
|
||||
token = encrypt_dns_credentials(dict(body.credentials))
|
||||
await conn.execute(
|
||||
"""INSERT INTO letsencrypt_account_dns_credentials (account_id, dns_provider, credentials_encrypted, updated_at)
|
||||
VALUES ($1, $2, $3, NOW())
|
||||
ON CONFLICT (account_id) DO UPDATE SET
|
||||
dns_provider = EXCLUDED.dns_provider,
|
||||
credentials_encrypted = EXCLUDED.credentials_encrypted,
|
||||
updated_at = NOW()""",
|
||||
account_id, provider_name, token,
|
||||
)
|
||||
# Keep the account's provider selection in sync.
|
||||
await conn.execute(
|
||||
"UPDATE letsencrypt_accounts SET dns_provider = $1, updated_at = NOW() WHERE id = $2",
|
||||
provider_name, account_id,
|
||||
)
|
||||
return {"ok": True, "dns_provider": provider_name, "detail": verify.get("detail", "Credentials stored.")}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.delete("/accounts/{account_id}/dns-credentials")
|
||||
async def delete_dns_credentials(account_id: int, authorization: str = Header(None)):
|
||||
"""Remove an account's stored DNS credentials. Admin-only."""
|
||||
from auth_middleware import get_current_user_from_token
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
if not current_user.get('is_admin', False):
|
||||
raise HTTPException(status_code=403, detail="Admin access required")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
await conn.execute("DELETE FROM letsencrypt_account_dns_credentials WHERE account_id = $1", account_id)
|
||||
return {"ok": True}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/orders/{order_id}/dns-confirm")
|
||||
async def confirm_dns_order(order_id: int, authorization: str = Header(None)):
|
||||
"""Manual DNS-01 only: the user asserts the TXT record is published; tell the CA to validate.
|
||||
Requires ssl.create + a per-user rate limit; acts only on a dns-01 + manual + pending order."""
|
||||
from auth_middleware import get_current_user_from_token, check_user_permission
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
has_perm = await check_user_permission(current_user['id'], 'ssl', 'create')
|
||||
if not has_perm:
|
||||
raise HTTPException(status_code=403, detail="Insufficient permissions: ssl.create required")
|
||||
await _enforce_dns_confirm_rate_limit(current_user['id'])
|
||||
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
order = await conn.fetchrow(
|
||||
"""SELECT o.id, o.status, o.challenge_type, a.dns_provider
|
||||
FROM letsencrypt_orders o JOIN letsencrypt_accounts a ON o.account_id = a.id
|
||||
WHERE o.id = $1""",
|
||||
order_id,
|
||||
)
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
if not order:
|
||||
raise HTTPException(status_code=404, detail="Order not found")
|
||||
if order['challenge_type'] != 'dns-01' or (order['dns_provider'] or 'manual') != 'manual':
|
||||
raise HTTPException(status_code=409, detail="This order is not a manual DNS-01 order")
|
||||
if order['status'] not in ('pending', 'processing'):
|
||||
raise HTTPException(status_code=409, detail=f"Order is '{order['status']}' and cannot be confirmed")
|
||||
|
||||
from services.dns01_orchestrator import confirm_manual_dns01
|
||||
await confirm_manual_dns01(order_id)
|
||||
return {"ok": True, "message": "DNS-01 confirmation submitted; the CA will validate shortly."}
|
||||
|
||||
|
||||
# --- Certificate operations ---
|
||||
|
||||
@router.post("/certificates")
|
||||
@@ -222,32 +516,61 @@ async def request_certificate(body: CertificateRequest, authorization: str = Hea
|
||||
logger.info(f"ACME: Using account_id={account_id} for certificate request")
|
||||
|
||||
warnings = []
|
||||
# Audit Tur 5 / Commit 8c: empty cluster_ids in UI means "global certificate".
|
||||
# Resolve to all ACME-enabled active clusters; only fail if NONE exist.
|
||||
|
||||
# Issue #35: resolve the effective challenge method (request override, else account default).
|
||||
conn_acct = await get_database_connection()
|
||||
try:
|
||||
acct = await conn_acct.fetchrow(
|
||||
"SELECT challenge_type, dns_provider FROM letsencrypt_accounts WHERE id = $1", account_id
|
||||
)
|
||||
finally:
|
||||
await close_database_connection(conn_acct)
|
||||
effective_challenge = (body.challenge_type or (acct['challenge_type'] if acct else None) or 'http-01')
|
||||
dns_provider = (acct['dns_provider'] if acct else None)
|
||||
is_dns01 = (effective_challenge == 'dns-01')
|
||||
has_wildcard = any(d.startswith('*.') for d in body.domains)
|
||||
|
||||
if is_dns01:
|
||||
if not await _dns01_enabled():
|
||||
raise HTTPException(status_code=409, detail="DNS-01 is disabled by an administrator (enable it in Settings).")
|
||||
if not is_supported((dns_provider or '').strip()):
|
||||
raise HTTPException(status_code=422, detail="The selected ACME account has no DNS provider configured for DNS-01.")
|
||||
if (dns_provider or 'manual') == 'manual':
|
||||
# Manual DNS-01 cannot be renewed unattended; auto-renew is forced off on the issued
|
||||
# certificate (see _complete_certificate). Tell the requester so it isn't a surprise.
|
||||
warnings.append(
|
||||
"Manual DNS-01 certificates cannot auto-renew unattended. Auto-renew will be disabled; "
|
||||
"re-publish the TXT record and request renewal before expiry."
|
||||
)
|
||||
elif has_wildcard:
|
||||
raise HTTPException(status_code=422, detail="Wildcard certificates require a DNS-01 account.")
|
||||
|
||||
# Empty cluster_ids = "global certificate". For DNS-01 no ACME Challenge Routing / port 80 is
|
||||
# needed, so resolve to ALL active clusters; http-01 still requires acme_enabled clusters.
|
||||
if not body.cluster_ids:
|
||||
conn_resolve = await get_database_connection()
|
||||
try:
|
||||
acme_clusters_resolved = await conn_resolve.fetch(
|
||||
"SELECT id FROM haproxy_clusters WHERE acme_enabled = TRUE AND is_active = TRUE"
|
||||
)
|
||||
if is_dns01:
|
||||
resolved = await conn_resolve.fetch("SELECT id FROM haproxy_clusters WHERE is_active = TRUE")
|
||||
else:
|
||||
resolved = await conn_resolve.fetch("SELECT id FROM haproxy_clusters WHERE acme_enabled = TRUE AND is_active = TRUE")
|
||||
finally:
|
||||
await close_database_connection(conn_resolve)
|
||||
if not acme_clusters_resolved:
|
||||
if not resolved:
|
||||
if is_dns01:
|
||||
raise HTTPException(status_code=422, detail="Cannot issue certificate: no active clusters configured.")
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail="Cannot issue certificate: no ACME-enabled clusters configured. "
|
||||
"Enable ACME Challenge Routing on at least one cluster in Cluster Management, "
|
||||
"Apply the configuration change, then retry."
|
||||
)
|
||||
body.cluster_ids = [c['id'] for c in acme_clusters_resolved]
|
||||
body.cluster_ids = [c['id'] for c in resolved]
|
||||
warnings.append(
|
||||
f"No clusters specified — applied to all ACME-enabled cluster(s) ({len(body.cluster_ids)})"
|
||||
f"No clusters specified — applied to all {'active' if is_dns01 else 'ACME-enabled'} cluster(s) ({len(body.cluster_ids)})"
|
||||
)
|
||||
logger.warning(
|
||||
f"ACME: Empty cluster_ids → global cert fallback to {len(body.cluster_ids)} ACME-enabled cluster(s)"
|
||||
)
|
||||
else:
|
||||
# Validate that referenced clusters exist + are ACME-enabled (warn-only).
|
||||
elif not is_dns01:
|
||||
# http-01 only: warn if no cluster has ACME Challenge Routing enabled.
|
||||
try:
|
||||
conn_warn = await get_database_connection()
|
||||
try:
|
||||
@@ -255,7 +578,6 @@ async def request_certificate(body: CertificateRequest, authorization: str = Hea
|
||||
"SELECT COUNT(*) FROM haproxy_clusters WHERE acme_enabled = TRUE AND is_active = TRUE"
|
||||
)
|
||||
if acme_clusters == 0:
|
||||
logger.warning("ACME: No clusters with ACME Challenge Routing enabled - certificate validation will likely fail")
|
||||
warnings.append(
|
||||
"No clusters have ACME Challenge Routing enabled. "
|
||||
"Certificate validation will fail. Enable it in Cluster Management and Apply Changes first."
|
||||
@@ -269,14 +591,30 @@ async def request_certificate(body: CertificateRequest, authorization: str = Hea
|
||||
account_id=account_id,
|
||||
domains=body.domains,
|
||||
cluster_ids=body.cluster_ids,
|
||||
challenge_type=effective_challenge,
|
||||
created_by=current_user['id'],
|
||||
)
|
||||
|
||||
challenges = await acme_service.respond_to_challenges(order['order_id'])
|
||||
# Audit trail: record who requested the certificate + the method (esp. for DNS-01/wildcard,
|
||||
# which has a wider blast radius than http-01). Never raises into the request path.
|
||||
try:
|
||||
from utils.activity_log import record_event
|
||||
await record_event(
|
||||
order['order_id'], "acme.order.requested",
|
||||
message=f"Certificate requested ({effective_challenge}) by user {current_user['id']}",
|
||||
details={"user_id": current_user['id'], "challenge_type": effective_challenge,
|
||||
"dns_provider": dns_provider, "domains": body.domains},
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# http-01 posts the challenge response immediately (token served continuously). dns-01 is
|
||||
# driven by the orchestrator AFTER the TXT is published (manual waits for dns-confirm), so we
|
||||
# must NOT respond here.
|
||||
challenges = []
|
||||
if not is_dns01:
|
||||
challenges = await acme_service.respond_to_challenges(order['order_id'])
|
||||
|
||||
# Commit 5b: re-fetch the order's *current* status from DB. The status
|
||||
# returned by create_order() reflects the moment of creation; after
|
||||
# respond_to_challenges() the CA may have already advanced it (e.g. to
|
||||
# 'processing'). Surfacing stale status leads UI to under-poll.
|
||||
conn_status = await get_database_connection()
|
||||
try:
|
||||
fresh_status = await conn_status.fetchval(
|
||||
@@ -287,14 +625,23 @@ async def request_certificate(body: CertificateRequest, authorization: str = Hea
|
||||
await close_database_connection(conn_status)
|
||||
effective_status = fresh_status or order['status']
|
||||
|
||||
logger.info(f"ACME: Order {order['order_id']} created, {len(challenges)} challenge(s) posted, status={effective_status}")
|
||||
if is_dns01:
|
||||
msg = ("Order created. Publish the DNS TXT record shown for each domain, then confirm."
|
||||
if dns_provider == 'manual'
|
||||
else "Order created. The DNS TXT record(s) will be published automatically; waiting for CA validation.")
|
||||
else:
|
||||
msg = "Order created. ACME challenges have been posted. Waiting for CA validation."
|
||||
|
||||
logger.info(f"ACME: Order {order['order_id']} created ({effective_challenge}), status={effective_status}")
|
||||
|
||||
return {
|
||||
"order_id": order['order_id'],
|
||||
"status": effective_status,
|
||||
"domains": body.domains,
|
||||
"challenge_type": effective_challenge,
|
||||
"dns_provider": dns_provider,
|
||||
"challenges": challenges,
|
||||
"message": "Order created. ACME challenges have been posted. Waiting for CA validation.",
|
||||
"message": msg,
|
||||
"warnings": warnings,
|
||||
}
|
||||
except HTTPException:
|
||||
@@ -316,7 +663,8 @@ async def list_orders(authorization: str = Header(None)):
|
||||
rows = await conn.fetch("""
|
||||
SELECT o.id, o.account_id, o.order_url, o.status, o.domains,
|
||||
o.ssl_certificate_id, o.cluster_ids, o.error_detail,
|
||||
o.created_at, o.updated_at, a.email as account_email
|
||||
o.created_at, o.updated_at, o.challenge_type, a.email as account_email,
|
||||
a.dns_provider
|
||||
FROM letsencrypt_orders o
|
||||
JOIN letsencrypt_accounts a ON o.account_id = a.id
|
||||
ORDER BY o.created_at DESC
|
||||
@@ -345,7 +693,8 @@ async def get_order(order_id: int, authorization: str = Header(None)):
|
||||
SELECT o.id, o.account_id, o.order_url, o.status, o.domains,
|
||||
o.certificate_url, o.finalize_url, o.expires_at,
|
||||
o.error_detail, o.ssl_certificate_id, o.cluster_ids,
|
||||
o.created_at, o.updated_at, a.email as account_email
|
||||
o.created_at, o.updated_at, o.challenge_type, a.email as account_email,
|
||||
a.dns_provider
|
||||
FROM letsencrypt_orders o
|
||||
JOIN letsencrypt_accounts a ON o.account_id = a.id
|
||||
WHERE o.id = $1
|
||||
@@ -353,13 +702,23 @@ async def get_order(order_id: int, authorization: str = Header(None)):
|
||||
if not order:
|
||||
raise HTTPException(status_code=404, detail="Order not found")
|
||||
|
||||
# Issue #35: include challenge_type + dns_txt_value (PUBLIC DNS data — NOT key_authorization,
|
||||
# NOT the API token) so the UI can render manual DNS-01 instructions. Explicit column list.
|
||||
challenges = await conn.fetch(
|
||||
"SELECT id, order_id, domain, token, challenge_url, status, validated_at, created_at FROM acme_challenges WHERE order_id = $1 ORDER BY domain", order_id
|
||||
"SELECT id, order_id, domain, token, challenge_url, status, validated_at, created_at, "
|
||||
"challenge_type, dns_txt_value FROM acme_challenges WHERE order_id = $1 ORDER BY domain", order_id
|
||||
)
|
||||
result = dict(order)
|
||||
result['domains'] = json.loads(result['domains']) if isinstance(result['domains'], str) else result['domains']
|
||||
result['cluster_ids'] = json.loads(result['cluster_ids']) if isinstance(result['cluster_ids'], str) else result['cluster_ids']
|
||||
result['challenges'] = [dict(c) for c in challenges]
|
||||
ch_list = []
|
||||
for c in challenges:
|
||||
cd = dict(c)
|
||||
if cd.get('challenge_type') == 'dns-01':
|
||||
# Server computes the record name (wildcard *.-stripping lives server-side).
|
||||
cd['dns_record_name'] = acme_service._challenge_dns_name(cd['domain'])
|
||||
ch_list.append(cd)
|
||||
result['challenges'] = ch_list
|
||||
return result
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
@@ -470,18 +829,23 @@ async def renew_order(order_id: int, authorization: str = Header(None)):
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
order = await conn.fetchrow(
|
||||
"SELECT account_id, domains, cluster_ids FROM letsencrypt_orders WHERE id = $1", order_id
|
||||
"SELECT account_id, domains, cluster_ids, challenge_type FROM letsencrypt_orders WHERE id = $1", order_id
|
||||
)
|
||||
if not order:
|
||||
raise HTTPException(status_code=404, detail="Order not found")
|
||||
domains = json.loads(order['domains']) if isinstance(order['domains'], str) else order['domains']
|
||||
cluster_ids = json.loads(order['cluster_ids']) if isinstance(order['cluster_ids'], str) else order['cluster_ids']
|
||||
challenge_type = order['challenge_type'] or 'http-01'
|
||||
|
||||
new_order = await acme_service.create_order(
|
||||
account_id=order['account_id'], domains=domains, cluster_ids=cluster_ids
|
||||
account_id=order['account_id'], domains=domains, cluster_ids=cluster_ids,
|
||||
challenge_type=challenge_type, created_by=current_user['id'],
|
||||
)
|
||||
challenges = await acme_service.respond_to_challenges(new_order['order_id'])
|
||||
return {"message": "Renewal order created", "new_order_id": new_order['order_id'], "challenges": challenges}
|
||||
# dns-01 is driven by the orchestrator after the TXT is published; only http-01 responds here.
|
||||
challenges = []
|
||||
if challenge_type != 'dns-01':
|
||||
challenges = await acme_service.respond_to_challenges(new_order['order_id'])
|
||||
return {"message": "Renewal order created", "new_order_id": new_order['order_id'], "challenge_type": challenge_type, "challenges": challenges}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -611,11 +975,17 @@ async def get_renewal_schedule(authorization: str = Header(None)):
|
||||
raise HTTPException(status_code=403, detail="Insufficient permissions: ssl.read required")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
# Issue #35: expose the challenge method/provider (via the originating order/account) so the
|
||||
# UI can distinguish manual DNS-01 certs, which cannot auto-renew unattended. LEFT JOINs keep
|
||||
# legacy certs (no order link / pre-DNS-01 columns) rendering as http-01.
|
||||
certs = await conn.fetch("""
|
||||
SELECT id, name, primary_domain, expiry_date, auto_renew, days_until_expiry
|
||||
FROM ssl_certificates
|
||||
WHERE source = 'letsencrypt' AND is_active = TRUE
|
||||
ORDER BY expiry_date ASC NULLS LAST
|
||||
SELECT c.id, c.name, c.primary_domain, c.expiry_date, c.auto_renew, c.days_until_expiry,
|
||||
o.challenge_type, a.dns_provider
|
||||
FROM ssl_certificates c
|
||||
LEFT JOIN letsencrypt_orders o ON o.id = c.letsencrypt_order_id
|
||||
LEFT JOIN letsencrypt_accounts a ON a.id = o.account_id
|
||||
WHERE c.source = 'letsencrypt' AND c.is_active = TRUE
|
||||
ORDER BY c.expiry_date ASC NULLS LAST
|
||||
""")
|
||||
return [dict(c) for c in certs]
|
||||
finally:
|
||||
@@ -663,6 +1033,18 @@ async def _complete_certificate(order_id: int) -> dict:
|
||||
cluster_ids = json.loads(order['cluster_ids']) if isinstance(order['cluster_ids'], str) else order['cluster_ids']
|
||||
primary_domain = domains[0] if domains else 'unknown'
|
||||
|
||||
# Issue #35: a manual DNS-01 certificate cannot be auto-renewed unattended (the renewal task
|
||||
# skips it — see main.py), so persist auto_renew=FALSE rather than storing a misleading
|
||||
# "Enabled" that the user trusts while the cert silently expires. http-01 and automated
|
||||
# DNS-01 (e.g. Cloudflare) keep auto_renew=TRUE, preserving existing behaviour.
|
||||
auto_renew_value = True
|
||||
if order.get('challenge_type') == 'dns-01':
|
||||
acct_provider = await conn.fetchval(
|
||||
"SELECT dns_provider FROM letsencrypt_accounts WHERE id = $1", order['account_id']
|
||||
)
|
||||
if (acct_provider or 'manual') == 'manual':
|
||||
auto_renew_value = False
|
||||
|
||||
# Commit 5g: guard against empty cert_private_key.
|
||||
# Inserting an SSL certificate row with an empty private_key would silently
|
||||
# produce an unusable certificate (HAProxy would fail to load on Apply, or
|
||||
@@ -749,12 +1131,13 @@ async def _complete_certificate(order_id: int) -> dict:
|
||||
certificate_content = $1, private_key_content = $2, chain_content = $3,
|
||||
all_domains = $4::jsonb, expiry_date = $5, days_until_expiry = $6,
|
||||
issuer = $7, fingerprint = $8, letsencrypt_order_id = $9,
|
||||
auto_renew = TRUE, is_active = TRUE, last_config_status = 'PENDING',
|
||||
auto_renew = $11, is_active = TRUE, last_config_status = 'PENDING',
|
||||
updated_at = NOW()
|
||||
WHERE id = $10
|
||||
""", cert_data['certificate_pem'], private_key_pem,
|
||||
cert_data.get('chain_pem', ''), all_domains,
|
||||
expiry_date, days_until_expiry, issuer, fingerprint, order_id, cert_id)
|
||||
expiry_date, days_until_expiry, issuer, fingerprint, order_id, cert_id,
|
||||
auto_renew_value)
|
||||
logger.info(f"ACME RENEWAL: Updated certificate {cert_id} for {primary_domain}, status=PENDING")
|
||||
else:
|
||||
cert_row = await conn.fetchrow("""
|
||||
@@ -764,11 +1147,12 @@ async def _complete_certificate(order_id: int) -> dict:
|
||||
usage_type, source, letsencrypt_order_id, auto_renew, is_active,
|
||||
last_config_status)
|
||||
VALUES ($1, $2, $3, $4, $5, $6::jsonb, $7, $8, $9, $10,
|
||||
'frontend', 'letsencrypt', $11, TRUE, TRUE, 'PENDING')
|
||||
'frontend', 'letsencrypt', $11, $12, TRUE, 'PENDING')
|
||||
RETURNING id
|
||||
""", f"le-{primary_domain}", cert_data['certificate_pem'], private_key_pem,
|
||||
cert_data.get('chain_pem', ''), primary_domain, all_domains,
|
||||
expiry_date, days_until_expiry, issuer, fingerprint, order_id)
|
||||
expiry_date, days_until_expiry, issuer, fingerprint, order_id,
|
||||
auto_renew_value)
|
||||
cert_id = cert_row['id']
|
||||
|
||||
await conn.execute(
|
||||
@@ -815,12 +1199,18 @@ async def _complete_certificate(order_id: int) -> dict:
|
||||
if mapped:
|
||||
effective_cluster_ids = [r['cluster_id'] for r in mapped]
|
||||
else:
|
||||
# Audit Tur 6 / Commit 5j: only fall back to ACME-enabled clusters.
|
||||
# Applying renewal to ACME-disabled clusters could re-introduce Issue #11
|
||||
# patterns and risks deploying certs to clusters where they can't be renewed.
|
||||
all_clusters = await conn.fetch(
|
||||
"SELECT id FROM haproxy_clusters WHERE is_active = TRUE AND acme_enabled = TRUE"
|
||||
# Audit Tur 6 / Commit 5j: http-01 only falls back to ACME-enabled clusters.
|
||||
# Issue #35: a dns-01 cert needs NO challenge routing / port 80, so it can renew on
|
||||
# any active cluster — resolve to all active clusters for dns-01.
|
||||
ch_type = await conn.fetchval(
|
||||
"SELECT challenge_type FROM letsencrypt_orders WHERE id = $1", order_id
|
||||
)
|
||||
if ch_type == 'dns-01':
|
||||
all_clusters = await conn.fetch("SELECT id FROM haproxy_clusters WHERE is_active = TRUE")
|
||||
else:
|
||||
all_clusters = await conn.fetch(
|
||||
"SELECT id FROM haproxy_clusters WHERE is_active = TRUE AND acme_enabled = TRUE"
|
||||
)
|
||||
effective_cluster_ids = [r['id'] for r in all_clusters]
|
||||
if effective_cluster_ids:
|
||||
logger.info(f"ACME RENEWAL: Resolved {len(effective_cluster_ids)} cluster(s) for global cert {cert_id}")
|
||||
|
||||
@@ -104,16 +104,40 @@ async def test_acme_connection(authorization: str = Header(None), directory_url:
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
# SECURITY (GHSA-3vh4-gvxx-wm2p): validate the URL before any outbound request
|
||||
# (https-only; block loopback/RFC1918/link-local/cloud-metadata after DNS),
|
||||
# pin the connector to IPv4, and never follow redirects. Also do NOT reflect
|
||||
# arbitrary upstream JSON keys back to the caller — that was an information-
|
||||
# disclosure oracle. Only report presence of the FIXED, known ACME directory
|
||||
# field names (never attacker-controlled data).
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector, SSRFValidationError
|
||||
|
||||
directory_url = str(directory_url)
|
||||
try:
|
||||
await assert_public_url(directory_url)
|
||||
except SSRFValidationError as e:
|
||||
return {"success": False, "error": f"Refused to fetch directory URL: {e}"}
|
||||
|
||||
_KNOWN_ACME_FIELDS = ["newNonce", "newAccount", "newOrder", "newAuthz", "revokeCert", "keyChange"]
|
||||
try:
|
||||
import aiohttp
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(str(directory_url), timeout=aiohttp.ClientTimeout(total=10)) as resp:
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.get(
|
||||
directory_url,
|
||||
timeout=aiohttp.ClientTimeout(total=10),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if resp.status == 200:
|
||||
data = await resp.json()
|
||||
data = await resp.json(content_type=None)
|
||||
if not isinstance(data, dict):
|
||||
return {"success": False, "error": "Directory URL did not return a JSON object"}
|
||||
present = [k for k in _KNOWN_ACME_FIELDS if k in data]
|
||||
if not present:
|
||||
return {"success": False, "error": "Response is not a valid ACME directory"}
|
||||
return {
|
||||
"success": True,
|
||||
"directory": str(directory_url),
|
||||
"endpoints": list(data.keys()) if isinstance(data, dict) else []
|
||||
"directory": directory_url,
|
||||
"endpoints": present,
|
||||
}
|
||||
else:
|
||||
return {"success": False, "error": f"HTTP {resp.status} from directory URL"}
|
||||
|
||||
@@ -9,7 +9,7 @@ from datetime import datetime, timezone
|
||||
|
||||
# Import database and models
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from auth_middleware import get_current_user_from_token
|
||||
from auth_middleware import get_current_user_from_token, require_authenticated_user
|
||||
from models.ssl import SSLCertificate, SSLCertificateCreate, SSLCertificateUpdate, SSLCertificateResponse
|
||||
from utils.ssl_parser import parse_ssl_certificate, validate_private_key, validate_certificate_chain, format_certificate_info
|
||||
from utils.activity_log import log_user_activity
|
||||
@@ -825,7 +825,8 @@ async def get_ssl_certificate(cert_id: int, authorization: str = Header(None)):
|
||||
logger.error(f"Error getting SSL certificate: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/certificates/{cert_id}/config-versions")
|
||||
@router.get("/certificates/{cert_id}/config-versions",
|
||||
dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): was unauthenticated
|
||||
async def get_ssl_certificate_config_versions(cert_id: int):
|
||||
"""Get config version history for specific SSL certificate"""
|
||||
try:
|
||||
|
||||
@@ -0,0 +1,987 @@
|
||||
"""Issue #27 — HA/VIP (Keepalived) management API (v1.7.0).
|
||||
|
||||
Isolated router: it owns vip_instances / vip_members only. It NEVER imports or calls
|
||||
the global HAProxy apply flow (cluster.py::apply_pending_changes) or the haproxy.cfg
|
||||
generator — VIP "Apply" is a standalone verb that renders per-node keepalived.conf
|
||||
snapshots into vip_members and flips the VIP to APPLIED.
|
||||
|
||||
For Apply-Management consistency it ALSO stages a standard `config_versions` row per
|
||||
change (version_name `vip-{id}-{action}`, status PENDING, is_active=FALSE — exactly like
|
||||
ssl-* versions) so VIP changes appear in the right-panel Pending Versions list with the
|
||||
product's standard "View Change" diff. is_active stays FALSE so a VIP version can never
|
||||
be served to an agent as haproxy.cfg (the agent config query requires is_active=TRUE);
|
||||
cluster.py keeps vip-* versions out of the generic haproxy apply/reject (NOT LIKE 'vip-%').
|
||||
|
||||
Every DB access is wrapped so a (pathological) missing vip_* relation degrades to an
|
||||
empty/None result instead of a 500 (B-7) — preserving the fleet-wide no-op guarantee.
|
||||
"""
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import time
|
||||
import uuid
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Header, HTTPException, Request
|
||||
|
||||
from auth_middleware import check_user_permission, get_current_user_from_token
|
||||
from database.connection import close_database_connection, get_database_connection
|
||||
from models.vip import VIPCreate, VIPUpdate
|
||||
from services.keepalived_config import (
|
||||
build_haproxy_check_script,
|
||||
decrypt_vrrp_secret,
|
||||
encrypt_vrrp_secret,
|
||||
render_keepalived_conf,
|
||||
)
|
||||
from utils.activity_log import log_user_activity
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/api/vip", tags=["HA / VIP"])
|
||||
|
||||
|
||||
def _client_ip(request: Optional[Request]) -> Optional[str]:
|
||||
try:
|
||||
return request.client.host if request and request.client else None
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
def _user_agent(request: Optional[Request]) -> Optional[str]:
|
||||
try:
|
||||
return request.headers.get("user-agent") if request else None
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
async def _require(authorization: Optional[str], action: str):
|
||||
"""Authenticate + enforce vip.<action>; returns current_user or raises 401/403."""
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
ok = await check_user_permission(current_user["id"], "vip", action, current_user=current_user)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=403, detail=f"vip.{action} permission required")
|
||||
return current_user
|
||||
|
||||
|
||||
async def _alloc_free_vrid(conn, pool_id: int, requested: Optional[int]) -> int:
|
||||
"""Use the requested VRID if free in the pool, else the lowest free 1..255."""
|
||||
used = {r["virtual_router_id"] for r in await conn.fetch(
|
||||
"SELECT virtual_router_id FROM vip_instances WHERE pool_id=$1 AND is_active=TRUE", pool_id)}
|
||||
if requested is not None:
|
||||
if requested in used:
|
||||
raise HTTPException(status_code=409, detail=f"VRID {requested} already used in this pool")
|
||||
return requested
|
||||
for cand in range(1, 256):
|
||||
if cand not in used:
|
||||
return cand
|
||||
raise HTTPException(status_code=409, detail="No free VRID (1-255) left in this pool")
|
||||
|
||||
|
||||
def _md5(text: str) -> str:
|
||||
return hashlib.md5(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
# The VRRP secret is rendered into keepalived.conf as `auth_pass <value>`. The "secret
|
||||
# never leaves the server in cleartext" rule means any config we hand back to the UI
|
||||
# (e.g. the version diff) must mask it — only the at-rest Fernet token and the
|
||||
# agent-delivery endpoint ever see the real value. Mask the WHOLE remainder of the line
|
||||
# (not just up to the first space) so a secret containing whitespace can't partially leak.
|
||||
_AUTH_PASS_RE = re.compile(r"(auth_pass\s+).*")
|
||||
|
||||
|
||||
def _redact_secret(conf: Optional[str]) -> str:
|
||||
return _AUTH_PASS_RE.sub(r"\1********", conf or "")
|
||||
|
||||
|
||||
def _derive_deploy_status(last_config_status: str, members, pending_delete: bool = False,
|
||||
is_active: bool = True) -> tuple:
|
||||
"""Display status that reflects ACTUAL agent convergence, not just the staging flag.
|
||||
|
||||
Other entities only read APPLIED once their agents acknowledge the new config; the
|
||||
VIP now mirrors that so the table never claims a VIP is live before its member nodes
|
||||
have deployed keepalived and acked the current config hash (issue #27 follow-up).
|
||||
|
||||
PENDING_DELETE — a deletion is STAGED and awaiting approval in Apply Management; the
|
||||
VIP keeps running until approved (nothing is torn down)
|
||||
DELETING — deletion APPROVED; member nodes are tearing keepalived down (not yet acked)
|
||||
PENDING — staged (created/edited); apply from Apply Management
|
||||
SYNCING — applied; an ONLINE member is still converging (deploying/acking)
|
||||
AWAITING — applied, but the un-converged members' agents are all OFFLINE, so
|
||||
nothing can deploy yet (bring the node's agent online) — not a hang
|
||||
ACTIVE — applied AND every member deployed & acked the current config hash
|
||||
ERROR — a member reported a deploy error
|
||||
ATTENTION — a member found a hand-managed keepalived (externally_managed)
|
||||
|
||||
Returns (status, synced_count, total_count).
|
||||
"""
|
||||
# Approval-gated delete: staged (still running) vs approved (tearing down).
|
||||
if pending_delete:
|
||||
return "PENDING_DELETE", 0, len(members)
|
||||
if not is_active:
|
||||
total = len(members)
|
||||
torn = sum(1 for m in members if m["last_deploy_state"] == "disabled")
|
||||
return ("DELETED" if total and torn == total else "DELETING"), torn, total
|
||||
if last_config_status == "PENDING":
|
||||
return "PENDING", 0, len(members)
|
||||
total = len(members)
|
||||
if total == 0:
|
||||
return "APPLIED", 0, 0
|
||||
|
||||
def _in_sync(m) -> bool:
|
||||
return (m["last_deploy_state"] == "enabled"
|
||||
and m["applied_config_hash"] is not None
|
||||
and m["last_deploy_hash"] == m["applied_config_hash"])
|
||||
|
||||
synced = sum(1 for m in members if _in_sync(m))
|
||||
if any(m["last_deploy_state"] == "error" for m in members):
|
||||
return "ERROR", synced, total
|
||||
if any(m["last_deploy_state"] == "externally_managed" for m in members):
|
||||
return "ATTENTION", synced, total
|
||||
if synced == total:
|
||||
return "ACTIVE", synced, total
|
||||
# Not fully converged: distinguish "actively converging" (an online agent will deploy
|
||||
# on its next poll) from "waiting on offline agents" (nothing will happen until the
|
||||
# operator brings the node's agent online) — the latter must NOT read as a live spinner.
|
||||
not_synced = [m for m in members if not _in_sync(m)]
|
||||
if not_synced and all((m["agent_status"] or "offline") != "online" for m in not_synced):
|
||||
return "AWAITING", synced, total
|
||||
return "SYNCING", synced, total
|
||||
|
||||
|
||||
async def render_vip_config_masked(conn, vip_id: int):
|
||||
"""Render the keepalived.conf each member node would deploy, as ONE masked text block
|
||||
(VRRP secret never in cleartext). Used by the standard config-version diff for vip-*
|
||||
versions (cluster.py) and to populate the staged version's config_content — so VIP
|
||||
changes show the product's standard "View Change" instead of a bespoke screen.
|
||||
|
||||
Returns (text, {"name": ...}) or (None, None) if the VIP is gone. Best-effort: a
|
||||
per-node render error becomes an inline comment rather than raising.
|
||||
"""
|
||||
v = await conn.fetchrow("SELECT * FROM vip_instances WHERE id=$1", vip_id)
|
||||
if not v:
|
||||
return None, None
|
||||
members = await conn.fetch("""
|
||||
SELECT m.agent_id, m.network_interface, m.role, m.priority,
|
||||
a.name AS agent_name, a.ip_address
|
||||
FROM vip_members m LEFT JOIN agents a ON a.id = m.agent_id
|
||||
WHERE m.vip_id=$1 ORDER BY m.priority DESC
|
||||
""", vip_id)
|
||||
auth_plain = decrypt_vrrp_secret(v["auth_pass_encrypted"]) if v["auth_pass_encrypted"] else None
|
||||
vip_dict = {"id": v["id"], "name": v["name"], "virtual_ip": v["virtual_ip"],
|
||||
"prefix_length": v["prefix_length"], "virtual_router_id": v["virtual_router_id"],
|
||||
"advert_int": v["advert_int"], "use_unicast": v["use_unicast"],
|
||||
"track_haproxy": v["track_haproxy"]}
|
||||
member_dicts = [{"role": m["role"], "priority": m["priority"],
|
||||
"network_interface": m["network_interface"], "agent_id": m["agent_id"],
|
||||
"ip_address": str(m["ip_address"]) if m["ip_address"] else ""} for m in members]
|
||||
blocks = []
|
||||
for m in members:
|
||||
this_agent = next(d for d in member_dicts if d["agent_id"] == m["agent_id"])
|
||||
peer_ips = [d["ip_address"] for d in member_dicts
|
||||
if d["agent_id"] != m["agent_id"] and d["ip_address"]]
|
||||
try:
|
||||
conf = render_keepalived_conf(vip=vip_dict, members=member_dicts,
|
||||
this_agent=this_agent, peer_ips=peer_ips,
|
||||
auth_pass_plain=auth_plain)
|
||||
except Exception as exc: # noqa: BLE001 — best-effort preview
|
||||
conf = f"# cannot render this node yet: {exc}\n"
|
||||
header = (f"# ===== node: {m['agent_name'] or m['agent_id']} "
|
||||
f"({m['ip_address'] or 'no IP'}) — {m['role']} priority {m['priority']} =====")
|
||||
blocks.append(header + "\n" + _redact_secret(conf))
|
||||
text = "\n\n".join(blocks) if blocks else "# (no participating nodes selected yet)\n"
|
||||
return text, {"name": v["name"]}
|
||||
|
||||
|
||||
async def _stage_vip_version(conn, vip_id: int, action: str, created_by: Optional[int]):
|
||||
"""Stage a standard PENDING config_versions row for a VIP change so it shows in Apply
|
||||
Management exactly like other entities. One row per cluster in the VIP's pool;
|
||||
is_active=FALSE so it is NEVER delivered as haproxy.cfg. Best-effort — a staging
|
||||
failure never fails the VIP operation (logged); the VIP still applies via its own flow.
|
||||
"""
|
||||
try:
|
||||
cluster_ids = [r["id"] for r in await conn.fetch(
|
||||
"SELECT id FROM haproxy_clusters WHERE pool_id = "
|
||||
"(SELECT pool_id FROM vip_instances WHERE id=$1)", vip_id)]
|
||||
if not cluster_ids:
|
||||
return
|
||||
content, _meta = await render_vip_config_masked(conn, vip_id)
|
||||
content = content or f"# VIP {vip_id} ({action})\n"
|
||||
checksum = _md5(content)
|
||||
# uuid suffix makes the name collision-proof even for sub-second same-action re-edits
|
||||
# (UNIQUE(cluster_id, version_name) would otherwise reject a same-second retry; review LOW-1).
|
||||
version_name = f"vip-{vip_id}-{action}-{int(time.time())}-{uuid.uuid4().hex[:6]}"
|
||||
# Capture this change's PENDING state so undo-reject can faithfully re-stage it
|
||||
# (restore the VIP — reactivating it if a rejected create soft-deleted it).
|
||||
pending_state = await _capture_vip_pending_state(conn, vip_id)
|
||||
metadata = json.dumps({"pending_state": pending_state}) if pending_state else None
|
||||
for cid in cluster_ids:
|
||||
# Collapse repeated pre-apply edits to a single PENDING row per VIP+cluster.
|
||||
await conn.execute(
|
||||
"DELETE FROM config_versions WHERE cluster_id=$1 AND status='PENDING' "
|
||||
"AND version_name LIKE $2", cid, f"vip-{vip_id}-%")
|
||||
await conn.execute("""
|
||||
INSERT INTO config_versions
|
||||
(cluster_id, version_name, config_content, checksum, created_by, is_active, status, metadata)
|
||||
VALUES ($1,$2,$3,$4,$5,FALSE,'PENDING',$6::jsonb)
|
||||
""", cid, version_name, content, checksum, created_by, metadata)
|
||||
except Exception as e: # noqa: BLE001 — versioning is a UI convenience, never block the op
|
||||
logger.warning(f"_stage_vip_version({vip_id},{action}) failed: {e}")
|
||||
|
||||
|
||||
async def _capture_vip_pending_state(conn, vip_id: int):
|
||||
"""Snapshot a VIP's current (pending) field + member state for undo-reject. The VRRP
|
||||
secret travels only as its encrypted token (never plaintext)."""
|
||||
v = await conn.fetchrow("SELECT * FROM vip_instances WHERE id=$1", vip_id)
|
||||
if not v:
|
||||
return None
|
||||
members = await conn.fetch(
|
||||
"SELECT agent_id, network_interface, role, priority FROM vip_members WHERE vip_id=$1", vip_id)
|
||||
return {
|
||||
"vip": {"name": v["name"], "description": v["description"], "virtual_ip": v["virtual_ip"],
|
||||
"prefix_length": v["prefix_length"], "virtual_router_id": v["virtual_router_id"],
|
||||
"advert_int": v["advert_int"], "use_unicast": v["use_unicast"],
|
||||
"track_haproxy": v["track_haproxy"], "auth_pass_encrypted": v["auth_pass_encrypted"]},
|
||||
"members": [{"agent_id": m["agent_id"], "network_interface": m["network_interface"],
|
||||
"role": m["role"], "priority": m["priority"]} for m in members],
|
||||
}
|
||||
|
||||
|
||||
async def restore_vip_from_rejected_version(conn, version_id: int):
|
||||
"""Undo-reject for a vip-* version: re-stage the rejected change as PENDING from the
|
||||
version's captured `pending_state`. Reactivates the VIP if a rejected create soft-deleted
|
||||
it, or re-applies a rejected edit's members — preserving each remaining member's
|
||||
last-applied snapshot so a running VIP is never torn down (T-1). Returns None on success
|
||||
or a human-readable error string (e.g. the name/address/VRID was reused since reject).
|
||||
"""
|
||||
row = await conn.fetchrow("SELECT version_name, metadata FROM config_versions WHERE id=$1", version_id)
|
||||
if not row:
|
||||
return "version not found"
|
||||
m = re.match(r"vip-(\d+)-", row["version_name"] or "")
|
||||
if not m:
|
||||
return "not a VIP version"
|
||||
vip_id = int(m.group(1))
|
||||
# Undo the reject of a DELETE request → re-arm the staged deletion (the VIP keeps running
|
||||
# until re-approved; nothing on the node changes). Requires the VIP to still be active.
|
||||
if "-delete-" in (row["version_name"] or ""):
|
||||
if not await conn.fetchrow("SELECT 1 FROM vip_instances WHERE id=$1 AND is_active=TRUE", vip_id):
|
||||
return "the VIP is no longer active — nothing to re-stage for deletion"
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET pending_delete=TRUE, last_config_status='PENDING', "
|
||||
"updated_at=CURRENT_TIMESTAMP WHERE id=$1", vip_id)
|
||||
await conn.execute(
|
||||
"UPDATE config_versions SET status='PENDING', is_active=FALSE, updated_at=CURRENT_TIMESTAMP "
|
||||
"WHERE version_name=$1 AND status='REJECTED'", row["version_name"])
|
||||
return None
|
||||
meta = row["metadata"]
|
||||
meta = json.loads(meta) if isinstance(meta, str) else meta
|
||||
ps = (meta or {}).get("pending_state")
|
||||
if not ps:
|
||||
return ("this VIP change predates undo support — re-create or re-edit the VIP "
|
||||
"from the HA/VIP page")
|
||||
if not await conn.fetchrow("SELECT 1 FROM vip_instances WHERE id=$1", vip_id):
|
||||
return "the VIP no longer exists — re-create it from the HA/VIP page"
|
||||
sv, sm = ps["vip"], ps.get("members", [])
|
||||
# One-VIP-per-agent must STILL hold after reactivation: if a member was meanwhile added
|
||||
# to another active VIP (while this one was rejected/soft-deleted), undoing here would
|
||||
# create a double-membership — and since the delivery endpoint serves one VIP per agent,
|
||||
# reactivating this (never-applied → not_configured) VIP would tear down the other live
|
||||
# VIP on that node. Block it with a clear message (review round-3 FINDING 2).
|
||||
member_ids = [m["agent_id"] for m in sm]
|
||||
if member_ids:
|
||||
dup = await conn.fetchrow(
|
||||
"SELECT a.name AS agent_name, v.name AS vip_name FROM vip_members vm "
|
||||
"JOIN vip_instances v ON v.id = vm.vip_id JOIN agents a ON a.id = vm.agent_id "
|
||||
"WHERE vm.agent_id = ANY($1) AND v.is_active = TRUE AND v.id <> $2 LIMIT 1",
|
||||
member_ids, vip_id)
|
||||
if dup:
|
||||
return (f"node '{dup['agent_name']}' now belongs to active VIP '{dup['vip_name']}' — "
|
||||
f"remove it there first, then undo")
|
||||
# Preserve the running (applied) snapshot for members that remain, so undo of an edit
|
||||
# never momentarily flips a live node to not_configured (T-1).
|
||||
prev = {r["agent_id"]: r for r in await conn.fetch(
|
||||
"SELECT agent_id, applied_config_content, applied_config_hash, last_deploy_state, "
|
||||
"last_deploy_message, last_deploy_hash, last_deploy_at FROM vip_members WHERE vip_id=$1", vip_id)}
|
||||
try:
|
||||
async with conn.transaction():
|
||||
await conn.execute("""
|
||||
UPDATE vip_instances SET name=$2, description=$3, virtual_ip=$4, prefix_length=$5,
|
||||
virtual_router_id=$6, advert_int=$7, use_unicast=$8, track_haproxy=$9,
|
||||
auth_pass_encrypted=$10, is_active=TRUE, last_config_status='PENDING',
|
||||
updated_at=CURRENT_TIMESTAMP
|
||||
WHERE id=$1
|
||||
""", vip_id, sv["name"], sv.get("description"), sv["virtual_ip"], sv["prefix_length"],
|
||||
sv["virtual_router_id"], sv["advert_int"], sv["use_unicast"], sv["track_haproxy"],
|
||||
sv.get("auth_pass_encrypted"))
|
||||
await conn.execute("DELETE FROM vip_members WHERE vip_id=$1", vip_id)
|
||||
for mm in sm:
|
||||
o = prev.get(mm["agent_id"])
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members (vip_id, agent_id, network_interface, role, priority,
|
||||
applied_config_content, applied_config_hash,
|
||||
last_deploy_state, last_deploy_message, last_deploy_hash, last_deploy_at)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11)
|
||||
""", vip_id, mm["agent_id"], mm["network_interface"], mm["role"], mm["priority"],
|
||||
o["applied_config_content"] if o else None, o["applied_config_hash"] if o else None,
|
||||
o["last_deploy_state"] if o else None, o["last_deploy_message"] if o else None,
|
||||
o["last_deploy_hash"] if o else None, o["last_deploy_at"] if o else None)
|
||||
# A multi-cluster pool stages one row per cluster under the SAME version_name;
|
||||
# flip them all back to PENDING so every affected cluster's Apply Management shows
|
||||
# the restored change (mirrors the SSL auto-undo). is_active stays FALSE (review MED-2).
|
||||
await conn.execute(
|
||||
"UPDATE config_versions SET status='PENDING', is_active=FALSE, "
|
||||
"updated_at=CURRENT_TIMESTAMP WHERE version_name=$1 AND status='REJECTED'",
|
||||
row["version_name"])
|
||||
except Exception as e: # noqa: BLE001
|
||||
if "unique" in str(e).lower() or "duplicate" in str(e).lower():
|
||||
return "the VIP's name, address or VRID was reused after it was rejected"
|
||||
raise
|
||||
return None
|
||||
|
||||
|
||||
async def _transition_vip_versions(conn, vip_id: int, new_status: str):
|
||||
"""Move this VIP's PENDING config_versions to APPLIED/REJECTED (is_active stays FALSE).
|
||||
Called by the VIP apply/reject endpoints so the right-panel version tracks the VIP."""
|
||||
try:
|
||||
await conn.execute(
|
||||
"UPDATE config_versions SET status=$2, is_active=FALSE, updated_at=CURRENT_TIMESTAMP "
|
||||
"WHERE status='PENDING' AND version_name LIKE $1", f"vip-{vip_id}-%", new_status)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning(f"_transition_vip_versions({vip_id},{new_status}) failed: {e}")
|
||||
|
||||
|
||||
def _jsonb_list(v):
|
||||
"""agents.capabilities/network_interfaces are JSONB; asyncpg returns them as a raw
|
||||
JSON string (no codec). Normalize to a Python list."""
|
||||
if isinstance(v, list):
|
||||
return v
|
||||
if isinstance(v, str):
|
||||
try:
|
||||
parsed = json.loads(v)
|
||||
return parsed if isinstance(parsed, list) else []
|
||||
except Exception: # noqa: BLE001
|
||||
return []
|
||||
return []
|
||||
|
||||
|
||||
def _capable(capabilities) -> bool:
|
||||
return "keepalived_management" in _jsonb_list(capabilities)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# List / read
|
||||
# ---------------------------------------------------------------------------
|
||||
@router.get("")
|
||||
async def list_vips(cluster_id: Optional[int] = None, authorization: str = Header(None)):
|
||||
"""List VIPs with their members + live MASTER/BACKUP state (never the secret).
|
||||
|
||||
Optional cluster_id scopes to VIPs in that cluster's pool — used by the Apply
|
||||
Management page (which is cluster-scoped) to surface pending VIP changes.
|
||||
"""
|
||||
await _require(authorization, "read")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
# Show active VIPs PLUS approved-but-still-tearing-down ones (is_active=FALSE with a
|
||||
# member that hasn't acked 'disabled' yet) so the operator can TRACK a deletion through
|
||||
# to completion; a VIP drops off only once every member has torn keepalived down.
|
||||
# The teardown-tracking clause is gated on last_config_status='APPLIED' so it ONLY shows
|
||||
# deletions APPROVED via the new flow — a VIP soft-deleted under the old immediate-delete
|
||||
# (pre-1.7.2: is_active=FALSE, last_config_status='PENDING') is NOT resurfaced (backward
|
||||
# compat). Rejected never-applied creates (also PENDING) are likewise excluded.
|
||||
_visible = ("(v.is_active = TRUE OR (v.last_config_status = 'APPLIED' AND EXISTS ("
|
||||
"SELECT 1 FROM vip_members mm WHERE mm.vip_id = v.id "
|
||||
"AND mm.applied_config_hash IS NOT NULL "
|
||||
"AND mm.last_deploy_state IS DISTINCT FROM 'disabled')))")
|
||||
if cluster_id is not None:
|
||||
vips = await conn.fetch(f"""
|
||||
SELECT v.id, v.name, v.description, v.pool_id, v.virtual_ip, v.prefix_length,
|
||||
v.virtual_router_id, v.advert_int, v.use_unicast, v.track_haproxy,
|
||||
v.is_active, v.last_config_status, v.pending_delete,
|
||||
(v.auth_pass_encrypted IS NOT NULL) AS auth_pass_set,
|
||||
v.created_at, v.updated_at, p.name AS pool_name
|
||||
FROM vip_instances v
|
||||
LEFT JOIN haproxy_cluster_pools p ON p.id = v.pool_id
|
||||
WHERE {_visible}
|
||||
AND v.pool_id = (SELECT pool_id FROM haproxy_clusters WHERE id = $1)
|
||||
ORDER BY v.name
|
||||
""", cluster_id)
|
||||
else:
|
||||
vips = await conn.fetch(f"""
|
||||
SELECT v.id, v.name, v.description, v.pool_id, v.virtual_ip, v.prefix_length,
|
||||
v.virtual_router_id, v.advert_int, v.use_unicast, v.track_haproxy,
|
||||
v.is_active, v.last_config_status, v.pending_delete,
|
||||
(v.auth_pass_encrypted IS NOT NULL) AS auth_pass_set,
|
||||
v.created_at, v.updated_at, p.name AS pool_name
|
||||
FROM vip_instances v
|
||||
LEFT JOIN haproxy_cluster_pools p ON p.id = v.pool_id
|
||||
WHERE {_visible}
|
||||
ORDER BY v.name
|
||||
""")
|
||||
result = []
|
||||
for v in vips:
|
||||
members = await conn.fetch("""
|
||||
SELECT m.id, m.agent_id, m.network_interface, m.role, m.priority,
|
||||
m.applied_config_hash, m.last_deploy_state, m.last_deploy_hash, m.last_deploy_at,
|
||||
a.name AS agent_name, a.status AS agent_status,
|
||||
a.keepalive_state, a.keepalive_ip, a.capabilities,
|
||||
a.ip_address
|
||||
FROM vip_members m
|
||||
LEFT JOIN agents a ON a.id = m.agent_id
|
||||
WHERE m.vip_id = $1
|
||||
ORDER BY m.priority DESC
|
||||
""", v["id"])
|
||||
deploy_status, deploy_synced, deploy_total = _derive_deploy_status(
|
||||
v["last_config_status"], members, v["pending_delete"], v["is_active"])
|
||||
result.append({
|
||||
**dict(v),
|
||||
# Convergence-aware status for the table (issue #27 follow-up); the raw
|
||||
# last_config_status is kept above for the PENDING gate / Apply button.
|
||||
"deploy_status": deploy_status,
|
||||
"deploy_synced": deploy_synced,
|
||||
"deploy_total": deploy_total,
|
||||
"members": [{
|
||||
"id": m["id"], "agent_id": m["agent_id"], "agent_name": m["agent_name"],
|
||||
"network_interface": m["network_interface"], "role": m["role"],
|
||||
"priority": m["priority"], "agent_status": m["agent_status"],
|
||||
"keepalive_state": m["keepalive_state"], "keepalive_ip": m["keepalive_ip"],
|
||||
"ip_address": str(m["ip_address"]) if m["ip_address"] else None,
|
||||
"keepalived_capable": _capable(m["capabilities"]),
|
||||
"last_deploy_state": m["last_deploy_state"],
|
||||
"last_deploy_at": m["last_deploy_at"].isoformat() if m["last_deploy_at"] else None,
|
||||
} for m in members],
|
||||
})
|
||||
return {"vips": result}
|
||||
except Exception as e: # noqa: BLE001 — degrade to empty rather than 500 (B-7)
|
||||
logger.error(f"list_vips failed: {e}")
|
||||
return {"vips": []}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{vip_id}")
|
||||
async def get_vip(vip_id: int, authorization: str = Header(None)):
|
||||
await _require(authorization, "read")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow("""
|
||||
SELECT v.id, v.name, v.description, v.pool_id, v.virtual_ip, v.prefix_length,
|
||||
v.virtual_router_id, v.advert_int, v.use_unicast, v.track_haproxy,
|
||||
v.is_active, v.last_config_status,
|
||||
(v.auth_pass_encrypted IS NOT NULL) AS auth_pass_set,
|
||||
v.created_at, v.updated_at
|
||||
FROM vip_instances v WHERE v.id = $1 AND v.is_active = TRUE
|
||||
""", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
members = await conn.fetch("""
|
||||
SELECT m.agent_id, m.network_interface, m.role, m.priority,
|
||||
m.last_deploy_state, m.last_deploy_at,
|
||||
a.name AS agent_name, a.keepalive_state, a.keepalive_ip
|
||||
FROM vip_members m LEFT JOIN agents a ON a.id = m.agent_id
|
||||
WHERE m.vip_id = $1 ORDER BY m.priority DESC
|
||||
""", vip_id)
|
||||
return {**dict(v), "members": [dict(m) for m in members]}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Create / update / delete
|
||||
# ---------------------------------------------------------------------------
|
||||
async def _validate_members_against_pool(conn, pool_id: int, members: List, vip_virtual_ip: str,
|
||||
exclude_vip_id: Optional[int] = None):
|
||||
"""Members must belong to the VIP's pool; VIP must not collide with an agent IP
|
||||
or another live VIP (T-4/SQL-1)."""
|
||||
# pool membership
|
||||
for m in members:
|
||||
row = await conn.fetchrow("SELECT pool_id FROM agents WHERE id=$1", m.agent_id)
|
||||
if not row:
|
||||
raise HTTPException(status_code=400, detail=f"agent {m.agent_id} not found")
|
||||
if row["pool_id"] != pool_id:
|
||||
raise HTTPException(status_code=400,
|
||||
detail=f"agent {m.agent_id} is not in pool {pool_id}")
|
||||
# One active VIP per agent (v1): a node deploys a single keepalived.conf, and the
|
||||
# delivery endpoint serves one VIP per agent — a second active membership would never
|
||||
# converge (stuck SYNCING) instead of erroring. Reject it up front (review MED-2).
|
||||
agent_ids = [m.agent_id for m in members]
|
||||
if agent_ids:
|
||||
dq = ("SELECT a.name AS agent_name, v.name AS vip_name "
|
||||
"FROM vip_members vm JOIN vip_instances v ON v.id = vm.vip_id "
|
||||
"JOIN agents a ON a.id = vm.agent_id "
|
||||
"WHERE vm.agent_id = ANY($1) AND v.is_active = TRUE")
|
||||
dparams = [agent_ids]
|
||||
if exclude_vip_id is not None:
|
||||
dq += " AND v.id <> $2"
|
||||
dparams.append(exclude_vip_id)
|
||||
dup = await conn.fetchrow(dq + " LIMIT 1", *dparams)
|
||||
if dup:
|
||||
raise HTTPException(status_code=409,
|
||||
detail=f"node '{dup['agent_name']}' is already a member of VIP "
|
||||
f"'{dup['vip_name']}' — a node can belong to only one VIP")
|
||||
# VIP must not be an existing agent's primary IP (INET cast, SQL-1)
|
||||
clash = await conn.fetchval("SELECT 1 FROM agents WHERE ip_address = $1::inet LIMIT 1", vip_virtual_ip)
|
||||
if clash:
|
||||
raise HTTPException(status_code=409, detail=f"{vip_virtual_ip} is already a node's IP")
|
||||
# …or another live VIP's address
|
||||
q = "SELECT 1 FROM vip_instances WHERE virtual_ip=$1 AND is_active=TRUE"
|
||||
params = [vip_virtual_ip]
|
||||
if exclude_vip_id is not None:
|
||||
q += " AND id <> $2"
|
||||
params.append(exclude_vip_id)
|
||||
if await conn.fetchval(q, *params):
|
||||
raise HTTPException(status_code=409, detail=f"{vip_virtual_ip} is already used by another VIP")
|
||||
|
||||
|
||||
@router.post("")
|
||||
async def create_vip(payload: VIPCreate, request: Request, authorization: str = Header(None)):
|
||||
current_user = await _require(authorization, "create")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
pool = await conn.fetchrow("SELECT id FROM haproxy_cluster_pools WHERE id=$1", payload.pool_id)
|
||||
if not pool:
|
||||
raise HTTPException(status_code=400, detail=f"pool {payload.pool_id} not found")
|
||||
await _validate_members_against_pool(conn, payload.pool_id, payload.members, payload.virtual_ip)
|
||||
|
||||
enc = encrypt_vrrp_secret(payload.auth_pass) if payload.auth_pass else None
|
||||
# Allocate VRID + insert, retrying once on a unique-violation race (B-1).
|
||||
last_err = None
|
||||
for _attempt in range(2):
|
||||
vrid = await _alloc_free_vrid(conn, payload.pool_id, payload.virtual_router_id)
|
||||
try:
|
||||
async with conn.transaction():
|
||||
vip_id = await conn.fetchval("""
|
||||
INSERT INTO vip_instances
|
||||
(name, description, pool_id, virtual_ip, prefix_length, virtual_router_id,
|
||||
advert_int, auth_pass_encrypted, use_unicast, track_haproxy,
|
||||
is_active, last_config_status, created_by)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,TRUE,'PENDING',$11)
|
||||
RETURNING id
|
||||
""", payload.name, payload.description, payload.pool_id, payload.virtual_ip,
|
||||
payload.prefix_length, vrid, payload.advert_int, enc,
|
||||
payload.use_unicast, payload.track_haproxy, current_user["id"])
|
||||
for m in payload.members:
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members (vip_id, agent_id, network_interface, role, priority)
|
||||
VALUES ($1,$2,$3,$4,$5)
|
||||
""", vip_id, m.agent_id, m.network_interface, m.role, m.priority)
|
||||
last_err = None
|
||||
break
|
||||
except Exception as ie: # noqa: BLE001
|
||||
if "unique" in str(ie).lower() or "duplicate" in str(ie).lower():
|
||||
last_err = ie
|
||||
if payload.virtual_router_id is not None:
|
||||
raise HTTPException(status_code=409, detail="VIP name/address/VRID already in use")
|
||||
continue # auto-VRID race → retry allocation
|
||||
raise
|
||||
if last_err is not None:
|
||||
raise HTTPException(status_code=409, detail="VIP create conflict (name/address/VRID)")
|
||||
|
||||
# Stage a standard PENDING config_version so the change shows in Apply Management.
|
||||
await _stage_vip_version(conn, vip_id, "create", current_user["id"])
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="create", resource_type="vip",
|
||||
resource_id=str(vip_id),
|
||||
details={"name": payload.name, "virtual_ip": payload.virtual_ip,
|
||||
"pool_id": payload.pool_id, "vrid": vrid, "members": len(payload.members)},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"id": vip_id, "message": "VIP created (PENDING — apply from Apply Management)"}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.put("/{vip_id}")
|
||||
async def update_vip(vip_id: int, payload: VIPUpdate, request: Request, authorization: str = Header(None)):
|
||||
current_user = await _require(authorization, "update")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow("SELECT * FROM vip_instances WHERE id=$1 AND is_active=TRUE", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
new_ip = payload.virtual_ip or v["virtual_ip"]
|
||||
members = payload.members if payload.members is not None else None
|
||||
if members is not None:
|
||||
await _validate_members_against_pool(conn, v["pool_id"], members, new_ip, exclude_vip_id=vip_id)
|
||||
elif payload.virtual_ip:
|
||||
# Address-only change: re-check the VIP isn't a node IP or another live VIP.
|
||||
clash = await conn.fetchval("SELECT 1 FROM agents WHERE ip_address=$1::inet LIMIT 1", new_ip)
|
||||
if clash:
|
||||
raise HTTPException(status_code=409, detail=f"{new_ip} is already a node's IP")
|
||||
dup = await conn.fetchval(
|
||||
"SELECT 1 FROM vip_instances WHERE virtual_ip=$1 AND is_active=TRUE AND id<>$2", new_ip, vip_id)
|
||||
if dup:
|
||||
raise HTTPException(status_code=409, detail=f"{new_ip} is already used by another VIP")
|
||||
|
||||
enc_set = payload.auth_pass is not None and payload.auth_pass != ""
|
||||
try:
|
||||
async with conn.transaction():
|
||||
await conn.execute("""
|
||||
UPDATE vip_instances SET
|
||||
name = COALESCE($2, name),
|
||||
description = COALESCE($3, description),
|
||||
virtual_ip = COALESCE($4, virtual_ip),
|
||||
prefix_length = COALESCE($5, prefix_length),
|
||||
virtual_router_id = COALESCE($6, virtual_router_id),
|
||||
advert_int = COALESCE($7, advert_int),
|
||||
use_unicast = COALESCE($8, use_unicast),
|
||||
track_haproxy = COALESCE($9, track_haproxy),
|
||||
auth_pass_encrypted = CASE WHEN $10 THEN $11 ELSE auth_pass_encrypted END,
|
||||
last_config_status = 'PENDING',
|
||||
-- Editing a VIP means you are KEEPING and changing it, so it cancels any
|
||||
-- staged deletion (otherwise a later Apply would delete instead of applying
|
||||
-- the edit). The edit then becomes the pending change to approve.
|
||||
pending_delete = FALSE,
|
||||
purge_on_teardown = FALSE,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $1
|
||||
""", vip_id, payload.name, payload.description, payload.virtual_ip,
|
||||
payload.prefix_length, payload.virtual_router_id, payload.advert_int,
|
||||
payload.use_unicast, payload.track_haproxy,
|
||||
enc_set, encrypt_vrrp_secret(payload.auth_pass) if enc_set else None)
|
||||
if members is not None:
|
||||
# T-1: preserve the APPLIED snapshot + deploy ack for members that REMAIN,
|
||||
# so a pending member edit never momentarily flips a running node to
|
||||
# not_configured (which would self-heal-teardown a live VIP). The new
|
||||
# config only goes live on the next Apply. A REMOVED member loses its row
|
||||
# → delivery returns not_configured → marker self-heal teardown (correct:
|
||||
# it's no longer part of the VIP). A NEW member starts with a NULL snapshot.
|
||||
prev = {r["agent_id"]: r for r in await conn.fetch(
|
||||
"SELECT agent_id, applied_config_content, applied_config_hash, "
|
||||
"last_deploy_state, last_deploy_message, last_deploy_hash, last_deploy_at "
|
||||
"FROM vip_members WHERE vip_id=$1", vip_id)}
|
||||
await conn.execute("DELETE FROM vip_members WHERE vip_id=$1", vip_id)
|
||||
for m in members:
|
||||
o = prev.get(m.agent_id)
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members (vip_id, agent_id, network_interface, role, priority,
|
||||
applied_config_content, applied_config_hash,
|
||||
last_deploy_state, last_deploy_message, last_deploy_hash, last_deploy_at)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11)
|
||||
""", vip_id, m.agent_id, m.network_interface, m.role, m.priority,
|
||||
o["applied_config_content"] if o else None,
|
||||
o["applied_config_hash"] if o else None,
|
||||
o["last_deploy_state"] if o else None,
|
||||
o["last_deploy_message"] if o else None,
|
||||
o["last_deploy_hash"] if o else None,
|
||||
o["last_deploy_at"] if o else None)
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as ie: # noqa: BLE001
|
||||
if "unique" in str(ie).lower() or "duplicate" in str(ie).lower():
|
||||
raise HTTPException(status_code=409, detail="VIP name/address/VRID already in use")
|
||||
raise
|
||||
# Re-stage the standard PENDING config_version reflecting the edited config.
|
||||
await _stage_vip_version(conn, vip_id, "update", current_user["id"])
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="update", resource_type="vip",
|
||||
resource_id=str(vip_id), details={"vip_id": vip_id},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": "VIP updated (PENDING — apply from Apply Management)"}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.delete("/{vip_id}")
|
||||
async def delete_vip(vip_id: int, request: Request, purge_package: bool = False,
|
||||
authorization: str = Header(None)):
|
||||
"""Request VIP deletion — APPROVAL-GATED for safety.
|
||||
|
||||
Deleting a *running* (already-applied) VIP does NOT take effect immediately: it is STAGED
|
||||
for Apply Management (pending_delete=TRUE + a vip-*-delete version) and the VIP keeps
|
||||
running — is_active stays TRUE, agents keep serving it, NOTHING is torn down — until the
|
||||
operator APPROVES the deletion. Rejecting it leaves the VIP running, untouched. Only the
|
||||
approval flips is_active=FALSE and lets the agents tear keepalived down. So a misclick can
|
||||
never tear down a production VIP, and an agent never deletes without an explicit approval.
|
||||
|
||||
A VIP that was NEVER applied (not deployed to any node) is removed immediately — there is
|
||||
nothing running to tear down. purge_package opts into uninstalling the keepalived package
|
||||
on teardown (default keeps it) and is honoured only once the deletion is approved, and only
|
||||
on nodes where WE installed it (the agent's install marker), never an admin's package.
|
||||
"""
|
||||
current_user = await _require(authorization, "delete")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow(
|
||||
"SELECT id, name, applied_snapshot FROM vip_instances WHERE id=$1 AND is_active=TRUE", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
|
||||
if v["applied_snapshot"] is None:
|
||||
# Never deployed to any node — removing it affects nothing, so do it at once.
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET is_active=FALSE, last_config_status='PENDING', "
|
||||
"pending_delete=FALSE, purge_on_teardown=$2, updated_at=CURRENT_TIMESTAMP WHERE id=$1",
|
||||
vip_id, bool(purge_package))
|
||||
await conn.execute(
|
||||
"DELETE FROM config_versions WHERE status='PENDING' AND version_name LIKE $1",
|
||||
f"vip-{vip_id}-%")
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="delete", resource_type="vip",
|
||||
resource_id=str(vip_id), details={"name": v["name"], "never_applied": True},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": "VIP removed — it was never applied, so no node was affected.",
|
||||
"staged": False}
|
||||
|
||||
# Running VIP → STAGE the deletion for approval. is_active stays TRUE (no teardown yet);
|
||||
# the agent keeps serving the VIP until the operator approves in Apply Management.
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET pending_delete=TRUE, purge_on_teardown=$2, "
|
||||
"last_config_status='PENDING', updated_at=CURRENT_TIMESTAMP WHERE id=$1",
|
||||
vip_id, bool(purge_package))
|
||||
await _stage_vip_version(conn, vip_id, "delete", current_user["id"])
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="delete-requested", resource_type="vip",
|
||||
resource_id=str(vip_id), details={"name": v["name"], "purge_package": bool(purge_package)},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": ("Deletion staged for approval — the VIP keeps running until you APPROVE it "
|
||||
"in Apply Management; reject to keep it. Nothing changes on the node until "
|
||||
"you approve."),
|
||||
"staged": True, "purge_package": bool(purge_package)}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Apply (isolated) + status
|
||||
# ---------------------------------------------------------------------------
|
||||
@router.post("/{vip_id}/apply")
|
||||
async def apply_vip(vip_id: int, request: Request, authorization: str = Header(None)):
|
||||
current_user = await _require(authorization, "apply")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow("SELECT * FROM vip_instances WHERE id=$1 AND is_active=TRUE", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
# APPROVED DELETION: if a deletion was staged for this VIP, approving it here performs the
|
||||
# actual delete — flip is_active=FALSE so the agents tear keepalived down on their next
|
||||
# poll (honouring purge_on_teardown). Until this moment the VIP kept running untouched, so
|
||||
# the teardown happens ONLY after this explicit human approval.
|
||||
if v["pending_delete"]:
|
||||
async with conn.transaction():
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET is_active=FALSE, pending_delete=FALSE, "
|
||||
"last_config_status='APPLIED', updated_at=CURRENT_TIMESTAMP WHERE id=$1", vip_id)
|
||||
await _transition_vip_versions(conn, vip_id, "APPLIED")
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="delete", resource_type="vip", resource_id=str(vip_id),
|
||||
details={"name": v["name"], "approved_delete": True,
|
||||
"purge_package": bool(v["purge_on_teardown"])},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
msg = "VIP deletion approved — member nodes will stop keepalived and release the VIP on their next poll"
|
||||
if v["purge_on_teardown"]:
|
||||
msg += "; the keepalived package will be uninstalled on nodes where we installed it"
|
||||
return {"message": msg, "deleted": True}
|
||||
members = await conn.fetch("""
|
||||
SELECT m.id, m.agent_id, m.network_interface, m.role, m.priority, a.ip_address
|
||||
FROM vip_members m LEFT JOIN agents a ON a.id = m.agent_id
|
||||
WHERE m.vip_id = $1
|
||||
""", vip_id)
|
||||
if len(members) < 1:
|
||||
raise HTTPException(status_code=400, detail="VIP needs at least 1 member")
|
||||
masters = [m for m in members if m["role"] == "MASTER"]
|
||||
if len(masters) != 1:
|
||||
raise HTTPException(status_code=400, detail="exactly one member must be MASTER")
|
||||
if any(m["ip_address"] is None for m in members):
|
||||
raise HTTPException(status_code=400,
|
||||
detail="every member must have a reported IP before apply (unicast peers)")
|
||||
master_prio = masters[0]["priority"]
|
||||
if any(m["role"] == "BACKUP" and m["priority"] >= master_prio for m in members):
|
||||
raise HTTPException(status_code=400, detail="MASTER priority must exceed every BACKUP")
|
||||
|
||||
auth_plain = decrypt_vrrp_secret(v["auth_pass_encrypted"]) if v["auth_pass_encrypted"] else None
|
||||
# If a secret is set but can't be decrypted (SECRET_KEY/VIP_ENCRYPTION_KEY rotated or
|
||||
# drifted between pods), FAIL the apply rather than silently rendering a config with NO
|
||||
# VRRP authentication — that would be a silent security downgrade and a guaranteed
|
||||
# MASTER/BACKUP auth mismatch with any node still holding the old config (review MED-1).
|
||||
if v["auth_pass_encrypted"] and not auth_plain:
|
||||
raise HTTPException(status_code=409,
|
||||
detail="VRRP secret could not be decrypted (encryption key changed?) — "
|
||||
"re-enter the VRRP secret on the VIP, then apply again")
|
||||
check_script = build_haproxy_check_script() if v["track_haproxy"] else ""
|
||||
member_dicts = [{"role": m["role"], "priority": m["priority"],
|
||||
"network_interface": m["network_interface"],
|
||||
"agent_id": m["agent_id"],
|
||||
"ip_address": str(m["ip_address"])} for m in members]
|
||||
vip_dict = {"id": v["id"], "name": v["name"], "virtual_ip": v["virtual_ip"],
|
||||
"prefix_length": v["prefix_length"], "virtual_router_id": v["virtual_router_id"],
|
||||
"advert_int": v["advert_int"], "use_unicast": v["use_unicast"],
|
||||
"track_haproxy": v["track_haproxy"]}
|
||||
|
||||
snapshot_members = []
|
||||
async with conn.transaction():
|
||||
for m in members:
|
||||
this_agent = next(d for d in member_dicts if d["agent_id"] == m["agent_id"])
|
||||
peer_ips = [d["ip_address"] for d in member_dicts if d["agent_id"] != m["agent_id"]]
|
||||
conf = render_keepalived_conf(vip=vip_dict, members=member_dicts,
|
||||
this_agent=this_agent, peer_ips=peer_ips,
|
||||
auth_pass_plain=auth_plain)
|
||||
chash = _md5(conf)
|
||||
await conn.execute("""
|
||||
UPDATE vip_members
|
||||
SET applied_config_content=$2, applied_config_hash=$3, updated_at=CURRENT_TIMESTAMP
|
||||
WHERE id=$1
|
||||
""", m["id"], conf, chash)
|
||||
snapshot_members.append({
|
||||
"agent_id": m["agent_id"], "network_interface": m["network_interface"],
|
||||
"role": m["role"], "priority": m["priority"],
|
||||
"applied_config_content": conf, "applied_config_hash": chash})
|
||||
# Capture the field-level applied state so a later pending edit can be REJECTED
|
||||
# and fully reverted to exactly this state (auth secret stored encrypted, never plain).
|
||||
applied_snapshot = {
|
||||
"vip": {"name": v["name"], "description": v["description"], "virtual_ip": v["virtual_ip"],
|
||||
"prefix_length": v["prefix_length"], "virtual_router_id": v["virtual_router_id"],
|
||||
"advert_int": v["advert_int"], "use_unicast": v["use_unicast"],
|
||||
"track_haproxy": v["track_haproxy"], "auth_pass_encrypted": v["auth_pass_encrypted"]},
|
||||
"members": snapshot_members}
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET last_config_status='APPLIED', "
|
||||
"applied_snapshot=$2::jsonb, updated_at=CURRENT_TIMESTAMP WHERE id=$1",
|
||||
vip_id, json.dumps(applied_snapshot))
|
||||
|
||||
# Move the standard config_version PENDING → APPLIED (stays is_active=FALSE so it is
|
||||
# never served as haproxy.cfg). Keeps the right-panel version in lockstep with the VIP.
|
||||
await _transition_vip_versions(conn, vip_id, "APPLIED")
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="apply", resource_type="vip",
|
||||
resource_id=str(vip_id),
|
||||
details={"name": v["name"], "virtual_ip": v["virtual_ip"], "members": len(members)},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
# check_script is rendered but not stored on the vip row; the agent gets it via
|
||||
# the delivery endpoint (which rebuilds it). Returned here only for visibility.
|
||||
return {"message": "VIP applied — agents will converge on next poll",
|
||||
"members_rendered": len(members), "tracks_haproxy": bool(check_script)}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/{vip_id}/reject")
|
||||
async def reject_vip(vip_id: int, request: Request, authorization: str = Header(None)):
|
||||
"""Discard a VIP's PENDING changes and fully restore the last-APPLIED state — the
|
||||
isolated equivalent of the product's reject -> restore-to-previous. Restores both the
|
||||
vip_instances fields and the exact member set (with their delivered config snapshots)
|
||||
from `applied_snapshot`, so the agents keep running what they already have (no churn).
|
||||
A never-applied PENDING VIP (no snapshot) is SOFT-deleted (is_active=FALSE) and its
|
||||
staged version marked REJECTED — so the change is reversible from the Rejected tab
|
||||
(undo-reject reactivates it). Never touches the global haproxy.cfg apply flow.
|
||||
"""
|
||||
current_user = await _require(authorization, "update")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow("SELECT * FROM vip_instances WHERE id=$1 AND is_active=TRUE", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
# REJECT A STAGED DELETION: cancel it — the VIP keeps running exactly as before (it was
|
||||
# never touched; is_active was never flipped). Clear the delete + purge intent and mark
|
||||
# the staged version REJECTED. This is the "nothing happened" path the operator expects.
|
||||
if v["pending_delete"]:
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET pending_delete=FALSE, purge_on_teardown=FALSE, "
|
||||
"last_config_status='APPLIED', updated_at=CURRENT_TIMESTAMP WHERE id=$1", vip_id)
|
||||
await _transition_vip_versions(conn, vip_id, "REJECTED")
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="reject", resource_type="vip", resource_id=str(vip_id),
|
||||
details={"name": v["name"], "delete_cancelled": True},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": "Deletion rejected — the VIP keeps running unchanged."}
|
||||
if v["last_config_status"] != "PENDING":
|
||||
return {"message": "Nothing to reject — no pending changes"}
|
||||
|
||||
snap_raw = v["applied_snapshot"]
|
||||
snap = json.loads(snap_raw) if isinstance(snap_raw, str) else snap_raw
|
||||
if not snap:
|
||||
# Created but never applied -> reject SOFT-deletes the VIP (is_active=FALSE) and
|
||||
# marks its staged version REJECTED. The row + members are kept so undo-reject can
|
||||
# reactivate the exact VIP (no orphan). The partial unique indexes free its
|
||||
# name/address/VRID for reuse while it's inactive.
|
||||
await _transition_vip_versions(conn, vip_id, "REJECTED")
|
||||
await conn.execute(
|
||||
"UPDATE vip_instances SET is_active=FALSE, last_config_status='PENDING', "
|
||||
"updated_at=CURRENT_TIMESTAMP WHERE id=$1", vip_id)
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="reject", resource_type="vip", resource_id=str(vip_id),
|
||||
details={"name": v["name"], "discarded": True},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": "Pending VIP rejected (undo from the Rejected tab to restore it)"}
|
||||
|
||||
sv = snap["vip"]
|
||||
sm = snap.get("members", [])
|
||||
try:
|
||||
async with conn.transaction():
|
||||
await conn.execute("""
|
||||
UPDATE vip_instances SET
|
||||
name=$2, description=$3, virtual_ip=$4, prefix_length=$5, virtual_router_id=$6,
|
||||
advert_int=$7, use_unicast=$8, track_haproxy=$9, auth_pass_encrypted=$10,
|
||||
last_config_status='APPLIED', updated_at=CURRENT_TIMESTAMP
|
||||
WHERE id=$1
|
||||
""", vip_id, sv["name"], sv.get("description"), sv["virtual_ip"], sv["prefix_length"],
|
||||
sv["virtual_router_id"], sv["advert_int"], sv["use_unicast"], sv["track_haproxy"],
|
||||
sv.get("auth_pass_encrypted"))
|
||||
await conn.execute("DELETE FROM vip_members WHERE vip_id=$1", vip_id)
|
||||
for m in sm:
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members (vip_id, agent_id, network_interface, role, priority,
|
||||
applied_config_content, applied_config_hash)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7)
|
||||
""", vip_id, m["agent_id"], m["network_interface"], m["role"], m["priority"],
|
||||
m.get("applied_config_content"), m.get("applied_config_hash"))
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as ie: # noqa: BLE001
|
||||
if "unique" in str(ie).lower() or "duplicate" in str(ie).lower():
|
||||
raise HTTPException(status_code=409,
|
||||
detail="Cannot restore — the previous address/VRID was taken in the meantime")
|
||||
raise
|
||||
# Mark the standard config_version REJECTED (history); the VIP is back to APPLIED.
|
||||
await _transition_vip_versions(conn, vip_id, "REJECTED")
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="reject", resource_type="vip", resource_id=str(vip_id),
|
||||
details={"name": v["name"], "restored": True},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
return {"message": "Pending changes rejected — VIP restored to its last applied state"}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{vip_id}/status")
|
||||
async def vip_status(vip_id: int, authorization: str = Header(None)):
|
||||
await _require(authorization, "read")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
v = await conn.fetchrow("SELECT id, name, is_active, last_config_status FROM vip_instances WHERE id=$1", vip_id)
|
||||
if not v:
|
||||
raise HTTPException(status_code=404, detail="VIP not found")
|
||||
members = await conn.fetch("""
|
||||
SELECT m.agent_id, m.role, m.priority, m.network_interface,
|
||||
m.last_deploy_state, m.last_deploy_message, m.last_deploy_at, m.applied_config_hash,
|
||||
a.name AS agent_name, a.keepalive_state, a.keepalive_ip, a.status AS agent_status,
|
||||
a.capabilities
|
||||
FROM vip_members m LEFT JOIN agents a ON a.id = m.agent_id
|
||||
WHERE m.vip_id=$1 ORDER BY m.priority DESC
|
||||
""", vip_id)
|
||||
out = []
|
||||
for m in members:
|
||||
applied = m["applied_config_hash"]
|
||||
deploy = m["last_deploy_state"]
|
||||
capable = _capable(m["capabilities"])
|
||||
if applied and not deploy:
|
||||
converge = "awaiting agent (upgrade may be required)" if not capable else "converging"
|
||||
else:
|
||||
converge = deploy or "pending"
|
||||
out.append({
|
||||
"agent_name": m["agent_name"], "role": m["role"], "priority": m["priority"],
|
||||
"network_interface": m["network_interface"],
|
||||
"agent_status": m["agent_status"], "keepalive_state": m["keepalive_state"],
|
||||
"keepalive_ip": m["keepalive_ip"], "keepalived_capable": capable,
|
||||
"deploy_state": deploy, "deploy_message": m["last_deploy_message"],
|
||||
"deploy_at": m["last_deploy_at"].isoformat() if m["last_deploy_at"] else None,
|
||||
"convergence": converge,
|
||||
})
|
||||
return {"id": v["id"], "name": v["name"], "is_active": v["is_active"],
|
||||
"last_config_status": v["last_config_status"], "members": out}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# NOTE: the keepalived config preview/diff is served by the STANDARD config-version diff
|
||||
# endpoint (cluster.py get_config_version_diff, vip-* branch) via render_vip_config_masked
|
||||
# above — there is no bespoke VIP preview endpoint, so VIP changes use the product's
|
||||
# standard "View Change" like every other entity (issue #27 follow-up).
|
||||
@@ -1,4 +1,5 @@
|
||||
from fastapi import APIRouter, HTTPException, Request, Header
|
||||
from fastapi import APIRouter, HTTPException, Request, Header, Depends
|
||||
from auth_middleware import require_authenticated_user
|
||||
from typing import Optional
|
||||
import logging
|
||||
import time
|
||||
@@ -182,7 +183,8 @@ async def get_waf_stats(cluster_id: Optional[int] = None, authorization: str = H
|
||||
logger.error(f"Error fetching WAF stats: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/rules", summary="Get WAF Rules", response_description="List of WAF rules")
|
||||
@router.get("/rules", summary="Get WAF Rules", response_description="List of WAF rules",
|
||||
dependencies=[Depends(require_authenticated_user)]) # SECURITY (GHSA-3p5c): WAF rule definitions aid bypass crafting
|
||||
async def get_waf_rules(cluster_id: Optional[int] = None):
|
||||
"""
|
||||
# Get WAF Rules
|
||||
|
||||
@@ -116,6 +116,10 @@ _PROBLEM_HUMANIZED: Dict[str, Dict[str, str]] = {
|
||||
"title": "HTTP-01 challenge response mismatch",
|
||||
"hint": "The CA fetched the challenge URL but received the wrong key authorization. Confirm the challenge was served from the right backend.",
|
||||
},
|
||||
"urn:ietf:params:acme:error:externalAccountRequired": {
|
||||
"title": "External Account Binding (EAB) required",
|
||||
"hint": "This CA (e.g. ZeroSSL, Google) requires EAB. Enter the EAB Key ID and HMAC Key from your CA account when registering.",
|
||||
},
|
||||
"urn:ietf:params:acme:error:invalidContact": {
|
||||
"title": "Invalid contact email",
|
||||
"hint": "The ACME account email is malformed. Update the LE account email.",
|
||||
@@ -199,6 +203,23 @@ def humanize_error_detail(error_detail: Any) -> Dict[str, Any]:
|
||||
status = parsed.get("status")
|
||||
subproblems = parsed.get("subproblems") or []
|
||||
|
||||
# Issue #35: DNS-01 failures are recorded as {stage, reason, timestamp} (no RFC8555 "type"),
|
||||
# so without this fallback the humanized alert would show a bare "ACME error" with no message.
|
||||
# Surface the reason and a targeted hint so the operator knows exactly what to fix.
|
||||
if not problem_type and parsed.get("reason"):
|
||||
reason = str(parsed.get("reason"))
|
||||
message = message or reason
|
||||
title = "DNS-01 validation failed"
|
||||
rlow = reason.lower()
|
||||
if "decrypt" in rlow or "credential" in rlow:
|
||||
hint = hint or "Re-enter the DNS provider credentials for this account in ACME Automation."
|
||||
elif "zone" in rlow:
|
||||
hint = hint or "Confirm the domain's DNS zone is managed by the configured provider and the token has access to it."
|
||||
elif "deadline" in rlow or "expired" in rlow or "confirm" in rlow:
|
||||
hint = hint or "The manual confirmation window passed. Create a new certificate request and publish the TXT record promptly."
|
||||
else:
|
||||
hint = hint or "Check the DNS TXT record and provider credentials, then retry."
|
||||
|
||||
out = {
|
||||
"title": title,
|
||||
"message": message,
|
||||
@@ -694,6 +715,7 @@ async def run_checks(
|
||||
cluster_ids: List[int],
|
||||
account_id: Optional[int],
|
||||
only: Optional[List[str]] = None,
|
||||
challenge_type: str = "http-01",
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Execute the full pre-flight check suite. `only` lets callers re-run a
|
||||
subset (per-check rerun in the UI).
|
||||
@@ -714,12 +736,29 @@ async def run_checks(
|
||||
except (TypeError, ValueError):
|
||||
safe_account_id = None
|
||||
|
||||
# Issue #35: DNS-01 validates via a TXT record, so the HTTP-01 reachability checks
|
||||
# (public A record, inbound port 80, ACME Challenge Routing) do not apply — report them
|
||||
# as `skipped` rather than failing an internal/isolated host that is actually fine.
|
||||
is_dns01 = (challenge_type == "dns-01")
|
||||
if "dns" in selected:
|
||||
results.append(await _safe_check("dns", "DNS resolution", check_dns(safe_domains)))
|
||||
if is_dns01:
|
||||
results.append(_check_result("dns", "DNS resolution", "skipped",
|
||||
"DNS-01: a public A record is not required (validation is via a TXT record).",
|
||||
severity="info"))
|
||||
else:
|
||||
results.append(await _safe_check("dns", "DNS resolution", check_dns(safe_domains)))
|
||||
if "port80" in selected:
|
||||
results.append(await _safe_check("port80", "Port 80 reachability", check_port80(safe_domains)))
|
||||
if is_dns01:
|
||||
results.append(_check_result("port80", "Port 80 reachability", "skipped",
|
||||
"DNS-01: inbound port 80 is not required.", severity="info"))
|
||||
else:
|
||||
results.append(await _safe_check("port80", "Port 80 reachability", check_port80(safe_domains)))
|
||||
if "routing" in selected:
|
||||
results.append(await _safe_check("routing", "HAProxy routing", check_routing(conn, safe_domains, safe_cluster_ids)))
|
||||
if is_dns01:
|
||||
results.append(_check_result("routing", "HAProxy routing", "skipped",
|
||||
"DNS-01: ACME Challenge Routing is not required.", severity="info"))
|
||||
else:
|
||||
results.append(await _safe_check("routing", "HAProxy routing", check_routing(conn, safe_domains, safe_cluster_ids)))
|
||||
if "account" in selected:
|
||||
results.append(await _safe_check("account", "ACME account", check_account(conn, safe_account_id)))
|
||||
if "agents" in selected:
|
||||
|
||||
@@ -28,7 +28,7 @@ def _b64url(data: bytes) -> str:
|
||||
|
||||
|
||||
def _b64url_decode(s: str) -> bytes:
|
||||
s += '=' * (4 - len(s) % 4)
|
||||
s += '=' * (-len(s) % 4) # pad to a multiple of 4 (0 pad when already aligned)
|
||||
return base64.urlsafe_b64decode(s)
|
||||
|
||||
|
||||
@@ -37,7 +37,11 @@ class ACMEService:
|
||||
|
||||
def __init__(self):
|
||||
self._directory_cache: Dict[str, dict] = {}
|
||||
self._nonce: Optional[str] = None
|
||||
# Anti-replay nonces are scoped PER CA (directory_url). A Replay-Nonce issued by one ACME
|
||||
# server must never be sent in a JWS to another, or the second server rejects it (e.g. ZeroSSL
|
||||
# "malformed: The Replay Nonce could not be base64url-decoded"). This client is a process-wide
|
||||
# singleton shared across CAs, so a single shared nonce was leaking across them.
|
||||
self._nonce_by_dir: Dict[str, str] = {}
|
||||
|
||||
async def _get_settings(self) -> dict:
|
||||
conn = await get_database_connection()
|
||||
@@ -65,25 +69,43 @@ class ACMEService:
|
||||
if cached.get('_fetched_at', 0) > time.time() - 3600:
|
||||
return cached
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(directory_url, timeout=aiohttp.ClientTimeout(total=15)) as resp:
|
||||
# SECURITY (GHSA-3vh4-gvxx-wm2p): directory_url can come from a stored
|
||||
# account row; validate it (https + public IP, no redirects) before the
|
||||
# server-side fetch so it cannot be pointed at internal/metadata targets.
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
await assert_public_url(directory_url)
|
||||
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.get(directory_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
if resp.status != 200:
|
||||
raise Exception(f"Failed to fetch ACME directory: HTTP {resp.status}")
|
||||
data = await resp.json()
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce = resp.headers['Replay-Nonce']
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
data['_fetched_at'] = time.time()
|
||||
self._directory_cache[directory_url] = data
|
||||
return data
|
||||
|
||||
async def _get_nonce(self, directory_url: str) -> str:
|
||||
if self._nonce:
|
||||
nonce = self._nonce
|
||||
self._nonce = None
|
||||
return nonce
|
||||
# Use a cached nonce for THIS CA only; otherwise fetch a fresh one from THIS CA's newNonce.
|
||||
cached = self._nonce_by_dir.pop(directory_url, None)
|
||||
if cached:
|
||||
return cached
|
||||
directory = await self.get_directory(directory_url)
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.head(directory['newNonce']) as resp:
|
||||
# get_directory may have just captured a nonce for this CA from the directory response.
|
||||
cached = self._nonce_by_dir.pop(directory_url, None)
|
||||
if cached:
|
||||
return cached
|
||||
# SECURITY (GHSA-3vh4-gvxx-wm2p): newNonce is taken from the (attacker-
|
||||
# influenceable) directory JSON and is fetched here BEFORE the guarded
|
||||
# _signed_request POST, so it must be guarded too — otherwise a directory
|
||||
# that returns an internal newNonce (and omits Replay-Nonce) is a live SSRF.
|
||||
# https + public IP only, IPv4-pinned connector, no redirects, bounded timeout.
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
nonce_url = directory['newNonce']
|
||||
await assert_public_url(nonce_url)
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.head(nonce_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
return resp.headers['Replay-Nonce']
|
||||
|
||||
def _generate_account_key(self) -> Tuple[str, dict]:
|
||||
@@ -128,6 +150,20 @@ class ACMEService:
|
||||
digest = hashlib.sha256(ordered.encode('utf-8')).digest()
|
||||
return _b64url(digest)
|
||||
|
||||
@staticmethod
|
||||
def _dns_txt_value(key_authorization: str) -> str:
|
||||
"""RFC 8555 §8.4: the DNS-01 TXT value is base64url(SHA256(key_authorization)) over the
|
||||
RAW 32-byte digest (NOT the hexdigest)."""
|
||||
return _b64url(hashlib.sha256(key_authorization.encode('utf-8')).digest())
|
||||
|
||||
@staticmethod
|
||||
def _challenge_dns_name(identifier: str) -> str:
|
||||
"""The `_acme-challenge.<base>` record name for an ACME identifier. A leading wildcard
|
||||
`*.` is stripped, so both `*.example.com` and bare `example.com` map to the SAME name
|
||||
`_acme-challenge.example.com` (which is why apex+wildcard need two coexisting TXT values)."""
|
||||
base = identifier[2:] if identifier.startswith('*.') else identifier
|
||||
return f"_acme-challenge.{base}"
|
||||
|
||||
def _sign_jws(self, private_key, protected: dict, payload: Any) -> dict:
|
||||
protected_b64 = _b64url(json.dumps(protected).encode('utf-8'))
|
||||
if payload == "":
|
||||
@@ -165,20 +201,34 @@ class ACMEService:
|
||||
|
||||
body = self._sign_jws(private_key, protected, payload)
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
# SECURITY (GHSA-3vh4-gvxx-wm2p): `url` is taken from the CA directory /
|
||||
# order responses. The directory is already fetched from a validated
|
||||
# public CA, but guard the follow-up POST target too (defence in depth)
|
||||
# so a tampered/malicious directory cannot steer the request internally.
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
await assert_public_url(url)
|
||||
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
for attempt in range(3):
|
||||
async with session.post(
|
||||
url,
|
||||
json=body,
|
||||
headers={"Content-Type": "application/jose+json"},
|
||||
timeout=aiohttp.ClientTimeout(total=30),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce = resp.headers['Replay-Nonce']
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
|
||||
if resp.status == 400:
|
||||
if resp.status == 400 and attempt < 2:
|
||||
err = await resp.json()
|
||||
if err.get('type') == 'urn:ietf:params:acme:error:badNonce' and attempt < 2:
|
||||
etype = (err.get('type') or '')
|
||||
edetail = (err.get('detail') or '').lower()
|
||||
# Retry on badNonce, and on any nonce-related malformed rejection (e.g.
|
||||
# "The Replay Nonce could not be base64url-decoded") — refetch a FRESH nonce
|
||||
# from the target CA and resign. With per-CA scoping the cross-CA cause is gone;
|
||||
# this is defense-in-depth so a stale/rejected nonce always self-heals.
|
||||
if etype.endswith('badNonce') or 'nonce' in edetail:
|
||||
nonce = resp.headers.get('Replay-Nonce') or await self._get_nonce(directory_url)
|
||||
protected['nonce'] = nonce
|
||||
body = self._sign_jws(private_key, protected, payload)
|
||||
@@ -208,6 +258,8 @@ class ACMEService:
|
||||
tos_agreed: bool = True,
|
||||
eab_kid: Optional[str] = None,
|
||||
eab_hmac_key: Optional[str] = None,
|
||||
challenge_type: str = 'http-01',
|
||||
dns_provider: Optional[str] = None,
|
||||
) -> dict:
|
||||
directory = await self.get_directory(directory_url)
|
||||
pem, jwk = self._generate_account_key()
|
||||
@@ -250,13 +302,14 @@ class ACMEService:
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
row = await conn.fetchrow("""
|
||||
INSERT INTO letsencrypt_accounts (email, directory_url, account_url, jwk_private_key, status, tos_agreed, eab_kid)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7)
|
||||
INSERT INTO letsencrypt_accounts (email, directory_url, account_url, jwk_private_key, status, tos_agreed, eab_kid, challenge_type, dns_provider)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
||||
ON CONFLICT (email, directory_url) DO UPDATE SET
|
||||
account_url = $3, jwk_private_key = $4, status = $5, tos_agreed = $6, updated_at = NOW()
|
||||
RETURNING id, email, directory_url, account_url, status, tos_agreed, created_at
|
||||
account_url = $3, jwk_private_key = $4, status = $5, tos_agreed = $6,
|
||||
challenge_type = $8, dns_provider = $9, updated_at = NOW()
|
||||
RETURNING id, email, directory_url, account_url, status, tos_agreed, created_at, challenge_type, dns_provider
|
||||
""", email, directory_url, account_url, pem,
|
||||
data.get('status') or 'valid', tos_agreed, eab_kid)
|
||||
data.get('status') or 'valid', tos_agreed, eab_kid, challenge_type, dns_provider)
|
||||
return dict(row)
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
@@ -302,8 +355,10 @@ class ACMEService:
|
||||
account_id: int,
|
||||
domains: List[str],
|
||||
cluster_ids: Optional[List[int]] = None,
|
||||
challenge_type: str = 'http-01',
|
||||
created_by: Optional[int] = None,
|
||||
) -> dict:
|
||||
logger.info(f"ACME: Creating order for domains={domains}, account_id={account_id}")
|
||||
logger.info(f"ACME: Creating order for domains={domains}, account_id={account_id}, challenge_type={challenge_type}")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
account = await conn.fetchrow(
|
||||
@@ -341,12 +396,12 @@ class ACMEService:
|
||||
|
||||
order_row = await conn.fetchrow("""
|
||||
INSERT INTO letsencrypt_orders
|
||||
(account_id, order_url, status, domains, finalize_url, expires_at, cluster_ids)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7)
|
||||
(account_id, order_url, status, domains, finalize_url, expires_at, cluster_ids, challenge_type, created_by)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)
|
||||
RETURNING id
|
||||
""", account_id, order_url, data.get('status') or 'pending',
|
||||
json.dumps(domains), data.get('finalize') or '', expires_at,
|
||||
json.dumps(cluster_ids or []))
|
||||
json.dumps(cluster_ids or []), challenge_type, created_by)
|
||||
|
||||
order_id = order_row['id']
|
||||
|
||||
@@ -383,18 +438,24 @@ class ACMEService:
|
||||
domain = (auth_data.get('identifier') or {}).get('value', '')
|
||||
http01_for_domain = False
|
||||
for challenge in (auth_data.get('challenges') or []):
|
||||
if challenge.get('type') == 'http-01':
|
||||
# Store only the challenge of the CHOSEN method (default 'http-01' keeps the
|
||||
# existing behaviour byte-identical; 'dns-01' selects the TXT challenge instead).
|
||||
if challenge.get('type') == challenge_type:
|
||||
token = challenge['token']
|
||||
jwk = self._get_jwk(private_key)
|
||||
thumbprint = self._jwk_thumbprint(jwk)
|
||||
key_auth = f"{token}.{thumbprint}"
|
||||
dns_txt = self._dns_txt_value(key_auth) if challenge_type == 'dns-01' else None
|
||||
|
||||
await conn.execute("""
|
||||
INSERT INTO acme_challenges (order_id, domain, token, key_authorization, challenge_url, status)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)
|
||||
INSERT INTO acme_challenges
|
||||
(order_id, domain, token, key_authorization, challenge_url, status,
|
||||
challenge_type, dns_txt_value)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7, $8)
|
||||
""", order_id, domain, token, key_auth,
|
||||
challenge.get('url') or '', challenge.get('status') or 'pending')
|
||||
logger.info(f"ACME: Challenge stored for domain={domain}, token={token[:20]}..., challenge_url={(challenge.get('url') or '')[:60]}")
|
||||
challenge.get('url') or '', challenge.get('status') or 'pending',
|
||||
challenge_type, dns_txt)
|
||||
logger.info(f"ACME: {challenge_type} challenge stored for domain={domain}, token={token[:20]}..., challenge_url={(challenge.get('url') or '')[:60]}")
|
||||
http01_for_domain = True
|
||||
if http01_for_domain and domain:
|
||||
domains_with_http01.add(domain)
|
||||
@@ -413,7 +474,7 @@ class ACMEService:
|
||||
error_payload, order_id
|
||||
)
|
||||
raise Exception(
|
||||
f"ACME order {order_id} created but no http-01 challenges available "
|
||||
f"ACME order {order_id} created but no {challenge_type} challenges available "
|
||||
f"(auth fetch failures: {len(auth_fetch_failures)}). See order.error_detail for diagnostics."
|
||||
)
|
||||
elif auth_fetch_failures:
|
||||
@@ -449,10 +510,15 @@ class ACMEService:
|
||||
try:
|
||||
# Issue #12 / Commit 5a: include 'failed' challenges so they can be retried,
|
||||
# but rate-limit per challenge: max 5 attempts in last 5 minutes.
|
||||
# DNS-01 skip-gate (the single safe choke point): never POST a challenge response for a
|
||||
# dns-01 row whose TXT record has not been published yet — that would make the CA validate
|
||||
# against a missing record and burn the order. http-01 rows (challenge_type 'http-01'/NULL)
|
||||
# are never excluded, so the existing flow is byte-identical.
|
||||
challenges = await conn.fetch(
|
||||
"""SELECT * FROM acme_challenges
|
||||
WHERE order_id = $1
|
||||
AND (status IN ('pending', 'failed') OR status IS NULL)""",
|
||||
AND (status IN ('pending', 'failed') OR status IS NULL)
|
||||
AND NOT (COALESCE(challenge_type, 'http-01') = 'dns-01' AND COALESCE(dns_record_published, FALSE) = FALSE)""",
|
||||
order_id
|
||||
)
|
||||
order = await conn.fetchrow(
|
||||
|
||||
@@ -0,0 +1,461 @@
|
||||
"""
|
||||
csr_service: CSR (Certificate Signing Request) generation + signed-certificate
|
||||
import (v1.9.0).
|
||||
|
||||
Flow:
|
||||
1. `generate_csr_bundle` builds a private key + CSR locally (pure crypto,
|
||||
no DB/IO — callers MUST run it via `asyncio.to_thread`: RSA-4096
|
||||
generation takes seconds and would stall the single-worker event loop).
|
||||
2. The bundle is persisted to `ssl_csrs` (`insert_csr_row`); the operator
|
||||
downloads the CSR PEM and has it signed by an external CA.
|
||||
3. `import_signed_certificate` pairs the CA response with the stored key,
|
||||
creates a normal `ssl_certificates` row (source='csr',
|
||||
last_config_status='PENDING' — agents never see it before Apply) and
|
||||
NULLs the key copy on the CSR row.
|
||||
|
||||
The CSR builder generalises the in-repo ACME reference
|
||||
(services/acme_service.py finalize_order): PEM output instead of DER, full
|
||||
subject instead of CN-only, ECDSA support, same PKCS8/NoEncryption key
|
||||
serialisation (the agent concatenates cert+key+chain into one PEM and HAProxy
|
||||
cannot read passphrase-protected keys).
|
||||
|
||||
Private keys are ENCRYPTED AT REST from v1.10.1 (Issue #53): the Fernet token
|
||||
replaces the PEM in the same `ssl_csrs.private_key_pem` column, so there is no
|
||||
schema change and no SCHEMA_VERSION bump. Rows written earlier hold a raw PEM
|
||||
and are still read transparently — see utils/csr_key_crypto.py for the format
|
||||
discriminator and the key-rotation caveat. The pending CSR key is the one key
|
||||
in the system worth encrypting: it sits idle for the whole signing window and
|
||||
is never transmitted, unlike ssl_certificates.private_key_content and the ACME
|
||||
order keys, which agents must receive in plaintext on every poll.
|
||||
The key is NEVER returned by any CSR API response — `csr_row_to_dict` strips
|
||||
it unconditionally.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
from types import SimpleNamespace
|
||||
|
||||
import asyncpg
|
||||
from fastapi import HTTPException
|
||||
|
||||
from cryptography import x509
|
||||
from cryptography.hazmat.primitives import hashes, serialization
|
||||
from cryptography.hazmat.primitives.asymmetric import ec, rsa
|
||||
from cryptography.x509.oid import NameOID
|
||||
|
||||
from services import ssl_service
|
||||
from utils.csr_key_crypto import decrypt_csr_private_key, encrypt_csr_private_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
_KEY_FACTORIES = {
|
||||
'rsa-2048': lambda: rsa.generate_private_key(public_exponent=65537, key_size=2048),
|
||||
'rsa-4096': lambda: rsa.generate_private_key(public_exponent=65537, key_size=4096),
|
||||
'ecdsa-p256': lambda: ec.generate_private_key(ec.SECP256R1()),
|
||||
'ecdsa-p384': lambda: ec.generate_private_key(ec.SECP384R1()),
|
||||
}
|
||||
|
||||
# (payload attribute, x509 OID, subject-JSON key)
|
||||
_SUBJECT_OID_MAP = [
|
||||
('organization', NameOID.ORGANIZATION_NAME, 'O'),
|
||||
('organizational_unit', NameOID.ORGANIZATIONAL_UNIT_NAME, 'OU'),
|
||||
('locality', NameOID.LOCALITY_NAME, 'L'),
|
||||
('state', NameOID.STATE_OR_PROVINCE_NAME, 'ST'),
|
||||
('country', NameOID.COUNTRY_NAME, 'C'),
|
||||
('email', NameOID.EMAIL_ADDRESS, 'emailAddress'),
|
||||
]
|
||||
|
||||
|
||||
def generate_csr_bundle(payload: Any) -> Dict[str, Any]:
|
||||
"""Generate a private key + CSR for a validated SSLCSRCreate payload.
|
||||
|
||||
Pure CPU-bound crypto — no DB, no network. Callers must offload via
|
||||
`asyncio.to_thread` (see module docstring).
|
||||
|
||||
Returns {'csr_pem', 'private_key_pem', 'sans', 'subject'}.
|
||||
"""
|
||||
key = _KEY_FACTORIES[payload.key_algorithm]()
|
||||
|
||||
attrs = [x509.NameAttribute(NameOID.COMMON_NAME, payload.common_name)]
|
||||
subject_json: Dict[str, str] = {}
|
||||
for attr_name, oid, json_key in _SUBJECT_OID_MAP:
|
||||
value = getattr(payload, attr_name, None)
|
||||
if value and str(value).strip():
|
||||
cleaned = str(value).strip()
|
||||
attrs.append(x509.NameAttribute(oid, cleaned))
|
||||
subject_json[json_key] = cleaned
|
||||
|
||||
# CN always first in the SAN list, then the extra names, deduped with
|
||||
# order preserved (mirrors the ACME flow where domains[0] is the CN).
|
||||
sans = list(dict.fromkeys([payload.common_name, *(payload.sans or [])]))
|
||||
|
||||
builder = (
|
||||
x509.CertificateSigningRequestBuilder()
|
||||
.subject_name(x509.Name(attrs))
|
||||
.add_extension(
|
||||
x509.SubjectAlternativeName([x509.DNSName(d) for d in sans]),
|
||||
critical=False,
|
||||
)
|
||||
)
|
||||
csr = builder.sign(key, hashes.SHA256())
|
||||
|
||||
return {
|
||||
'csr_pem': csr.public_bytes(serialization.Encoding.PEM).decode('utf-8'),
|
||||
'private_key_pem': key.private_bytes(
|
||||
serialization.Encoding.PEM,
|
||||
serialization.PrivateFormat.PKCS8,
|
||||
serialization.NoEncryption(),
|
||||
).decode('utf-8'),
|
||||
'sans': sans,
|
||||
'subject': subject_json,
|
||||
}
|
||||
|
||||
|
||||
def diff_domains(csr_sans: Optional[List[str]], cert_domains: Optional[List[str]]) -> List[str]:
|
||||
"""Human-readable warnings for SAN drift between the CSR and the signed
|
||||
certificate (case-insensitive set diff). CAs legitimately add/normalise
|
||||
SANs, so drift is WARN-only — the hard gate is the key match."""
|
||||
csr_set = {d.lower() for d in (csr_sans or []) if d}
|
||||
cert_set = {d.lower() for d in (cert_domains or []) if d}
|
||||
warnings: List[str] = []
|
||||
added = sorted(cert_set - csr_set)
|
||||
dropped = sorted(csr_set - cert_set)
|
||||
if added:
|
||||
warnings.append(
|
||||
f"The CA added domains that were not in the CSR: {', '.join(added)}"
|
||||
)
|
||||
if dropped:
|
||||
warnings.append(
|
||||
f"The CA dropped domains that were requested in the CSR: {', '.join(dropped)}"
|
||||
)
|
||||
return warnings
|
||||
|
||||
|
||||
def _maybe_json_list(value: Any) -> List[str]:
|
||||
"""asyncpg returns JSONB columns as str unless a codec is registered."""
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
parsed = json.loads(value)
|
||||
return parsed if isinstance(parsed, list) else []
|
||||
except Exception:
|
||||
return []
|
||||
return list(value) if value else []
|
||||
|
||||
|
||||
def csr_row_to_dict(row: Any, include_pem: bool = False) -> Dict[str, Any]:
|
||||
"""Row → API dict. ALWAYS strips private_key_pem — the key never leaves
|
||||
the server via a CSR endpoint. csr_pem included only on demand
|
||||
(detail/create responses, not lists)."""
|
||||
d = dict(row)
|
||||
d.pop('private_key_pem', None)
|
||||
if not include_pem:
|
||||
d.pop('csr_pem', None)
|
||||
for key in ('subject', 'sans'):
|
||||
if key in d and isinstance(d[key], str):
|
||||
try:
|
||||
d[key] = json.loads(d[key])
|
||||
except Exception:
|
||||
pass
|
||||
return d
|
||||
|
||||
|
||||
async def assert_csr_name_available(conn, name: str) -> None:
|
||||
"""Reject a CSR name that is already taken by an ACTIVE certificate or
|
||||
another PENDING CSR. Called BEFORE key generation (cheap fail-fast) and
|
||||
re-run inside `insert_csr_row` (the unique index closes the race)."""
|
||||
existing_cert = await conn.fetchval(
|
||||
"SELECT id FROM ssl_certificates WHERE name = $1 AND is_active = TRUE",
|
||||
name,
|
||||
)
|
||||
if existing_cert:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
f"An active SSL certificate named '{name}' already exists. "
|
||||
"The CSR name becomes the certificate name at import — choose "
|
||||
"a different name or remove the existing certificate first."
|
||||
),
|
||||
)
|
||||
existing_csr = await conn.fetchval(
|
||||
"SELECT id FROM ssl_csrs WHERE name = $1 AND status = 'pending'",
|
||||
name,
|
||||
)
|
||||
if existing_csr:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
f"A pending CSR named '{name}' already exists (id={existing_csr}). "
|
||||
"Import or delete it first, or choose a different name."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
async def insert_csr_row(conn, payload: Any, bundle: Dict[str, Any], user_id: Optional[int]) -> int:
|
||||
"""Persist a freshly generated CSR bundle. Returns the new csr id.
|
||||
|
||||
Issue #53 (v1.10.1): the private key is Fernet-encrypted before it is stored. The token goes
|
||||
into the SAME private_key_pem TEXT column — no schema change — and is only ever decrypted
|
||||
in-process by import_signed_certificate. No CSR endpoint returns the column either way.
|
||||
"""
|
||||
await assert_csr_name_available(conn, payload.name)
|
||||
stored_key = encrypt_csr_private_key(bundle['private_key_pem'])
|
||||
try:
|
||||
csr_id = await conn.fetchval(
|
||||
"""
|
||||
INSERT INTO ssl_csrs
|
||||
(name, common_name, subject, sans, key_algorithm, csr_pem,
|
||||
private_key_pem, status, created_by)
|
||||
VALUES ($1, $2, $3::jsonb, $4::jsonb, $5, $6, $7, 'pending', $8)
|
||||
RETURNING id
|
||||
""",
|
||||
payload.name,
|
||||
payload.common_name,
|
||||
json.dumps(bundle['subject']),
|
||||
json.dumps(bundle['sans']),
|
||||
payload.key_algorithm,
|
||||
bundle['csr_pem'],
|
||||
stored_key,
|
||||
user_id,
|
||||
)
|
||||
except asyncpg.exceptions.UniqueViolationError:
|
||||
# uq_ssl_csrs_name_pending — a concurrent request won the name.
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
f"A pending CSR named '{payload.name}' was just created by a "
|
||||
"concurrent request — choose a different name."
|
||||
),
|
||||
)
|
||||
return csr_id
|
||||
|
||||
|
||||
async def import_signed_certificate(conn, csr_id: int, imp: Any, user_id: Optional[int]) -> Dict[str, Any]:
|
||||
"""Pair the CA-signed certificate with the stored CSR key and create the
|
||||
ssl_certificates row. Atomic: cert row + CSR state change commit together.
|
||||
|
||||
Returns {'certificate_id', 'certificate_name', 'primary_domain',
|
||||
'warnings', 'reactivated'}. Raises HTTPException on every failure
|
||||
(404 missing, 409 already completed, 400 validation).
|
||||
"""
|
||||
async with conn.transaction():
|
||||
# Row lock serialises concurrent imports AND a concurrent DELETE of
|
||||
# the same CSR; works across multiple uvicorn workers (DB-level lock).
|
||||
row = await conn.fetchrow(
|
||||
"SELECT * FROM ssl_csrs WHERE id = $1 FOR UPDATE", csr_id
|
||||
)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404, detail="CSR not found")
|
||||
if row['status'] == 'completed':
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=(
|
||||
f"CSR '{row['name']}' is already completed — certificate "
|
||||
f"id {row['ssl_certificate_id']} was imported from it. "
|
||||
"Create a new CSR to reissue."
|
||||
),
|
||||
)
|
||||
if not row['private_key_pem']:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
"Stored CSR private key is missing — the CSR row is "
|
||||
"corrupt. Delete it and create a new CSR."
|
||||
),
|
||||
)
|
||||
# Issue #53: the column holds a Fernet token from v1.10.1 on, and a raw PEM for rows
|
||||
# written before it. decrypt_csr_private_key accepts both, so no data migration is
|
||||
# needed. A None here means the token cannot be decrypted — SECRET_KEY was rotated
|
||||
# without CSR_ENCRYPTION_KEY set. Fail loudly: the key is gone, so the CA's certificate
|
||||
# can never be paired with it, and silently falling through would surface as the far
|
||||
# more confusing "certificate does not match this CSR's private key".
|
||||
stored_key = decrypt_csr_private_key(row['private_key_pem'])
|
||||
if not stored_key:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
f"The stored private key for CSR '{row['name']}' cannot be decrypted. This "
|
||||
"happens when SECRET_KEY was rotated while CSR_ENCRYPTION_KEY was not set. "
|
||||
"The key is unrecoverable, so this CSR can no longer be completed — delete "
|
||||
"it and create a new one (then have the new CSR signed)."
|
||||
),
|
||||
)
|
||||
|
||||
effective_name = getattr(imp, 'name', None) or row['name']
|
||||
|
||||
# Parse the pasted certificate FIRST so a malformed/truncated CA
|
||||
# response gets the manual flow's 400, not a 500 from the key-match
|
||||
# step below (verify_certificate_key_match reports an unparseable
|
||||
# cert as match=None, which we treat as an integrity failure).
|
||||
from utils.ssl_parser import parse_ssl_certificate, verify_certificate_key_match
|
||||
precheck = parse_ssl_certificate(imp.certificate_content)
|
||||
if precheck.get('error'):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Invalid SSL certificate: {precheck['error']}",
|
||||
)
|
||||
|
||||
# THE defining check of this feature: the CA response must match the
|
||||
# key we generated. Deliberately stricter than create_cert_row's
|
||||
# lenient fallback — we generated this key ourselves, so an
|
||||
# unverifiable pair is an integrity failure, not operator input.
|
||||
match_result = verify_certificate_key_match(imp.certificate_content, stored_key)
|
||||
if match_result.get('match') is False:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
"The signed certificate does not match this CSR's private "
|
||||
"key — the CA response likely belongs to a different "
|
||||
"CSR/key. Verify you pasted the certificate that was "
|
||||
"issued for this exact CSR."
|
||||
),
|
||||
)
|
||||
if match_result.get('match') is not True:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
"Could not verify the certificate/key pair: "
|
||||
f"{match_result.get('reason', 'unknown')}"
|
||||
),
|
||||
)
|
||||
|
||||
# Full parse/validation pipeline shared with the manual + wizard
|
||||
# flows: invalid PEM, bad chain and already-expired certs all 400.
|
||||
payload = SimpleNamespace(
|
||||
name=effective_name,
|
||||
certificate_content=imp.certificate_content,
|
||||
private_key_content=stored_key,
|
||||
chain_content=getattr(imp, 'chain_content', None),
|
||||
usage_type=getattr(imp, 'usage_type', 'frontend') or 'frontend',
|
||||
)
|
||||
fields = ssl_service._prepare_cert_fields(payload)
|
||||
|
||||
# Global name uniqueness (ssl_certificates.cluster_id is always NULL
|
||||
# under the R38 schema, so name is effectively a global namespace).
|
||||
existing = await conn.fetchrow(
|
||||
"SELECT id, is_active FROM ssl_certificates WHERE name = $1 LIMIT 1",
|
||||
effective_name,
|
||||
)
|
||||
if existing and existing['is_active']:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
f"An active SSL certificate named '{effective_name}' "
|
||||
"already exists (created after this CSR). Delete or "
|
||||
"rename it, or pass a different `name` in the import "
|
||||
"request — the CSR stays pending and can be re-imported."
|
||||
),
|
||||
)
|
||||
|
||||
reactivated = False
|
||||
if existing and not existing['is_active']:
|
||||
# Reactivate the soft-deleted row (mirrors create_cert_row):
|
||||
# preserves the row id so historical references keep working.
|
||||
await conn.execute(
|
||||
"DELETE FROM ssl_certificate_clusters WHERE ssl_certificate_id = $1",
|
||||
existing['id'],
|
||||
)
|
||||
await conn.execute(
|
||||
"""
|
||||
UPDATE ssl_certificates
|
||||
SET is_active = TRUE,
|
||||
last_config_status = 'PENDING',
|
||||
certificate_content = $2,
|
||||
private_key_content = $3,
|
||||
chain_content = $4,
|
||||
primary_domain = $5,
|
||||
all_domains = $6::jsonb,
|
||||
expiry_date = $7,
|
||||
usage_type = $8,
|
||||
issuer = $9,
|
||||
fingerprint = $10,
|
||||
status = $11,
|
||||
days_until_expiry = $12,
|
||||
source = 'csr',
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $1
|
||||
""",
|
||||
existing['id'],
|
||||
fields['cert_content'],
|
||||
fields['private_key_content'],
|
||||
fields['chain_content'],
|
||||
fields['primary_domain'],
|
||||
json.dumps(fields['all_domains']),
|
||||
fields['expiry_date'],
|
||||
fields['usage_type'],
|
||||
fields['issuer'],
|
||||
fields['fingerprint'],
|
||||
fields['status'],
|
||||
fields['days_until_expiry'],
|
||||
)
|
||||
cert_id = existing['id']
|
||||
reactivated = True
|
||||
logger.info(
|
||||
f"csr_service.import_signed_certificate: reactivated "
|
||||
f"soft-deleted cert '{effective_name}' (id={cert_id}) for CSR {csr_id}"
|
||||
)
|
||||
else:
|
||||
cert_id = await conn.fetchval(
|
||||
"""
|
||||
INSERT INTO ssl_certificates (
|
||||
name, primary_domain, certificate_content, private_key_content,
|
||||
chain_content, expiry_date, issuer, fingerprint, status,
|
||||
days_until_expiry, all_domains, is_active, cluster_id,
|
||||
last_config_status, usage_type, source
|
||||
) VALUES (
|
||||
$1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11::jsonb,
|
||||
TRUE, NULL, 'PENDING', $12, 'csr'
|
||||
)
|
||||
RETURNING id
|
||||
""",
|
||||
effective_name,
|
||||
fields['primary_domain'],
|
||||
fields['cert_content'],
|
||||
fields['private_key_content'],
|
||||
fields['chain_content'],
|
||||
fields['expiry_date'],
|
||||
fields['issuer'],
|
||||
fields['fingerprint'],
|
||||
fields['status'],
|
||||
fields['days_until_expiry'],
|
||||
json.dumps(fields['all_domains']),
|
||||
fields['usage_type'],
|
||||
)
|
||||
|
||||
# Cluster bindings: global = zero junction rows (existing convention).
|
||||
if not getattr(imp, 'is_global', False):
|
||||
for cluster_id in (getattr(imp, 'cluster_ids', None) or []):
|
||||
await ssl_service.ensure_cluster_junction(conn, cert_id, cluster_id)
|
||||
|
||||
# Complete the CSR and destroy the key copy — the key now lives on
|
||||
# the certificate row only, like every other key in the system.
|
||||
await conn.execute(
|
||||
"""
|
||||
UPDATE ssl_csrs
|
||||
SET status = 'completed',
|
||||
ssl_certificate_id = $2,
|
||||
private_key_pem = NULL,
|
||||
completed_at = CURRENT_TIMESTAMP,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $1
|
||||
""",
|
||||
csr_id,
|
||||
cert_id,
|
||||
)
|
||||
|
||||
warnings = diff_domains(_maybe_json_list(row['sans']), fields['all_domains'])
|
||||
if reactivated:
|
||||
warnings.append(
|
||||
f"A soft-deleted certificate named '{effective_name}' was "
|
||||
f"reactivated (row id {cert_id}) — existing entities that still "
|
||||
"reference that id now serve the newly imported certificate."
|
||||
)
|
||||
|
||||
return {
|
||||
'certificate_id': cert_id,
|
||||
'certificate_name': effective_name,
|
||||
'primary_domain': fields['primary_domain'],
|
||||
'warnings': warnings,
|
||||
'reactivated': reactivated,
|
||||
}
|
||||
@@ -0,0 +1,321 @@
|
||||
"""Issue #35 — ACME DNS-01 (v1.8.0): non-blocking per-cycle orchestration.
|
||||
|
||||
Driven by the existing `complete_pending_acme_orders` background task (which already claims
|
||||
in-progress orders with `FOR UPDATE SKIP LOCKED`). For a dns-01 order this module advances AT MOST
|
||||
ONE step per 60s cycle — publish TXT, then (after a short min-age) respond — so the serial claim
|
||||
loop is never blocked by a multi-minute wait, and NO DNS library is needed (the CA is the source of
|
||||
truth; a propagation-lag `invalid` is recovered by a bounded fresh-order chain).
|
||||
|
||||
Design invariants (from the hardening review):
|
||||
- Additive at the RRset level: publish/cleanup operate on a single (name, value), so wildcard+apex
|
||||
(two values at one name) coexist.
|
||||
- No new order-status value: a failed order stays `invalid`; a boolean `dns01_retry_claimed` does the
|
||||
winner-only CAS + claim exclusion, so existing `status` consumers are untouched.
|
||||
- Secrets (provider API tokens) are NEVER logged or written to error_detail/events.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from services.acme_service import acme_service as acme_svc, ACMEService
|
||||
from services.dns_providers import get_provider, is_supported, DnsProviderError
|
||||
from utils.dns_credentials import decrypt_dns_credentials
|
||||
from utils.activity_log import record_event
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Tunables (kept conservative vs Let's Encrypt rate limits: 5 failed-validations/host/hour,
|
||||
# 300 new-orders/account/3h).
|
||||
PROPAGATION_GRACE_SECONDS = 25 # min age before we tell the CA to validate
|
||||
MANUAL_CONFIRM_TTL = timedelta(hours=48)
|
||||
MAX_RETRIES = 3 # bounded fresh-order chain (1 original + 3 retries = 4 orders)
|
||||
# Retry backoff floor (minutes) indexed by the order's current dns01_attempts: [15, 30, 60].
|
||||
# The AUTHORITATIVE implementation is the SQL CASE in main.py's claim query
|
||||
# (complete_pending_acme_orders), so the backoff is evaluated atomically with the
|
||||
# FOR UPDATE SKIP LOCKED claim. Documented here only — do not reintroduce a second copy.
|
||||
|
||||
|
||||
def _now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _aware(dt) -> Optional[datetime]:
|
||||
if dt is None:
|
||||
return None
|
||||
return dt if dt.tzinfo else dt.replace(tzinfo=timezone.utc)
|
||||
|
||||
|
||||
async def _load_credentials(conn, account_id: int) -> Tuple[Optional[Dict[str, str]], bool]:
|
||||
"""Return (credentials_dict_or_None, row_exists). credentials None + row_exists True means the
|
||||
stored token could not be decrypted (e.g. SECRET_KEY rotated)."""
|
||||
row = await conn.fetchrow(
|
||||
"SELECT credentials_encrypted FROM letsencrypt_account_dns_credentials WHERE account_id = $1",
|
||||
account_id,
|
||||
)
|
||||
if not row:
|
||||
return None, False
|
||||
return decrypt_dns_credentials(row["credentials_encrypted"]), True
|
||||
|
||||
|
||||
async def _fail_order(conn, order_id: int, reason: str) -> None:
|
||||
"""Mark an order invalid with a sanitized reason (no secrets) + event."""
|
||||
import json
|
||||
payload = json.dumps({"stage": "dns01", "reason": reason, "timestamp": _now().isoformat()})
|
||||
await conn.execute(
|
||||
"UPDATE letsencrypt_orders SET status = 'invalid', error_detail = $1, updated_at = NOW() WHERE id = $2",
|
||||
payload, order_id,
|
||||
)
|
||||
await record_event(order_id, "acme.dns01.validation", severity="ERROR", message=reason, conn=conn)
|
||||
|
||||
|
||||
async def advance_dns01_order(order_id: int) -> None:
|
||||
"""One non-blocking step for a claimed pending/processing dns-01 order. No-op for http-01 or
|
||||
orders not in a publishable state. Safe to call every cycle (idempotent via CAS flags)."""
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
order = await conn.fetchrow(
|
||||
"""SELECT o.id, o.status, o.challenge_type, o.account_id, a.dns_provider
|
||||
FROM letsencrypt_orders o JOIN letsencrypt_accounts a ON o.account_id = a.id
|
||||
WHERE o.id = $1""",
|
||||
order_id,
|
||||
)
|
||||
if not order or order["challenge_type"] != "dns-01":
|
||||
return
|
||||
if order["status"] not in ("pending", "processing"):
|
||||
return
|
||||
|
||||
challenges = await conn.fetch(
|
||||
"SELECT * FROM acme_challenges WHERE order_id = $1 AND challenge_type = 'dns-01'", order_id
|
||||
)
|
||||
if not challenges:
|
||||
return
|
||||
|
||||
provider_name = (order["dns_provider"] or "manual").strip()
|
||||
|
||||
# --- Manual provider: the user publishes + confirms; we only enforce the deadline. ---
|
||||
if provider_name == "manual" or not is_supported(provider_name):
|
||||
deadline = _aware(challenges[0]["manual_confirm_deadline"])
|
||||
if deadline is None:
|
||||
new_deadline = _now() + MANUAL_CONFIRM_TTL
|
||||
await conn.execute(
|
||||
"UPDATE acme_challenges SET manual_confirm_deadline = $1 WHERE order_id = $2 AND manual_confirm_deadline IS NULL",
|
||||
new_deadline, order_id,
|
||||
)
|
||||
elif _now() > deadline:
|
||||
await _fail_order(conn, order_id,
|
||||
"Manual DNS-01 confirmation deadline passed without confirmation.")
|
||||
return # respond happens via the dns-confirm endpoint
|
||||
|
||||
# --- Automated provider (e.g. Cloudflare). ---
|
||||
creds, row_exists = await _load_credentials(conn, order["account_id"])
|
||||
if not row_exists:
|
||||
await _fail_order(conn, order_id,
|
||||
f"No DNS provider credentials configured for provider '{provider_name}'.")
|
||||
return
|
||||
if creds is None:
|
||||
await _fail_order(conn, order_id,
|
||||
"DNS provider credentials could not be decrypted; re-enter them in Settings.")
|
||||
return
|
||||
|
||||
provider = get_provider(provider_name, creds)
|
||||
|
||||
# Publish any not-yet-published challenge (CAS so two replicas can't double-publish).
|
||||
for ch in challenges:
|
||||
if ch["dns_record_published"]:
|
||||
continue
|
||||
flipped = await conn.fetchval(
|
||||
"""UPDATE acme_challenges SET dns_record_published = TRUE, dns_published_at = NOW()
|
||||
WHERE id = $1 AND dns_record_published = FALSE RETURNING id""",
|
||||
ch["id"],
|
||||
)
|
||||
if not flipped:
|
||||
continue
|
||||
name = ACMEService._challenge_dns_name(ch["domain"])
|
||||
try:
|
||||
await provider.add_txt_record(name, ch["dns_txt_value"])
|
||||
await record_event(order_id, "acme.dns01.publish",
|
||||
message=f"Published TXT {name}; waiting for DNS propagation before asking the CA to validate.",
|
||||
details={"name": name, "provider": provider_name}, conn=conn)
|
||||
except DnsProviderError as exc:
|
||||
# Revert so the next cycle retries the publish; keep the order pending. exc is sanitized.
|
||||
await conn.execute(
|
||||
"UPDATE acme_challenges SET dns_record_published = FALSE, dns_published_at = NULL WHERE id = $1",
|
||||
ch["id"],
|
||||
)
|
||||
await record_event(order_id, "acme.dns01.publish", severity="WARNING",
|
||||
message=f"Publish failed for {name}: {exc}",
|
||||
details={"name": name, "provider": provider_name}, conn=conn)
|
||||
return
|
||||
|
||||
# All published? Then respond once the min-age gate has elapsed (across cycles, no sleep).
|
||||
rows = await conn.fetch(
|
||||
"SELECT dns_record_published, dns_published_at FROM acme_challenges WHERE order_id = $1 AND challenge_type = 'dns-01'",
|
||||
order_id,
|
||||
)
|
||||
if any(not r["dns_record_published"] for r in rows):
|
||||
return
|
||||
published_ats = [_aware(r["dns_published_at"]) for r in rows if r["dns_published_at"]]
|
||||
if not published_ats:
|
||||
return
|
||||
if (_now() - min(published_ats)).total_seconds() < PROPAGATION_GRACE_SECONDS:
|
||||
return # wait one more cycle
|
||||
|
||||
# Only POST the challenge response if something still needs validating — avoids re-POSTing
|
||||
# every cycle (and bumping dns01_last_attempt_at) once the CA already has them processing.
|
||||
still_pending = await conn.fetchval(
|
||||
"""SELECT 1 FROM acme_challenges WHERE order_id = $1 AND challenge_type = 'dns-01'
|
||||
AND (status IN ('pending', 'failed') OR status IS NULL) LIMIT 1""",
|
||||
order_id,
|
||||
)
|
||||
if not still_pending:
|
||||
return
|
||||
|
||||
await acme_svc.respond_to_challenges(order_id)
|
||||
await conn.execute("UPDATE letsencrypt_orders SET dns01_last_attempt_at = NOW() WHERE id = $1", order_id)
|
||||
await record_event(order_id, "acme.dns01.responded",
|
||||
message="Told the CA to validate the DNS-01 challenge(s).", conn=conn)
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def confirm_manual_dns01(order_id: int) -> Dict:
|
||||
"""Called by POST /orders/{id}/dns-confirm for the manual provider: mark the TXT published and
|
||||
tell the CA to validate. Returns a small status dict."""
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
await conn.execute(
|
||||
"""UPDATE acme_challenges SET dns_record_published = TRUE, dns_published_at = COALESCE(dns_published_at, NOW())
|
||||
WHERE order_id = $1 AND challenge_type = 'dns-01'""",
|
||||
order_id,
|
||||
)
|
||||
await acme_svc.respond_to_challenges(order_id)
|
||||
await conn.execute("UPDATE letsencrypt_orders SET dns01_last_attempt_at = NOW() WHERE id = $1", order_id)
|
||||
await record_event(order_id, "acme.dns01.responded",
|
||||
message="Manual DNS-01 confirmed; told the CA to validate.", conn=conn)
|
||||
return {"ok": True}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def retry_invalid_dns01(order_id: int) -> None:
|
||||
"""Bounded fresh-order recovery for a dns-01 order that went `invalid` (e.g. propagation lag).
|
||||
Winner-only CAS on `dns01_retry_claimed`; cleans the old TXT, mints a child order. No-op for
|
||||
http-01 or when the budget is exhausted."""
|
||||
conn = await get_database_connection()
|
||||
child_created = False
|
||||
try:
|
||||
order = await conn.fetchrow(
|
||||
"""SELECT o.*, a.dns_provider FROM letsencrypt_orders o
|
||||
JOIN letsencrypt_accounts a ON o.account_id = a.id WHERE o.id = $1""",
|
||||
order_id,
|
||||
)
|
||||
if not order or order["challenge_type"] != "dns-01":
|
||||
return
|
||||
if (order["dns01_attempts"] or 0) >= MAX_RETRIES:
|
||||
return # budget exhausted; stays terminal invalid
|
||||
|
||||
# Winner-only claim (closes the cross-replica double-mint race).
|
||||
claimed = await conn.fetchval(
|
||||
"""UPDATE letsencrypt_orders SET dns01_retry_claimed = TRUE, updated_at = NOW()
|
||||
WHERE id = $1 AND status = 'invalid' AND dns01_retry_claimed = FALSE RETURNING id""",
|
||||
order_id,
|
||||
)
|
||||
if not claimed:
|
||||
return
|
||||
|
||||
provider_name = (order["dns_provider"] or "manual").strip()
|
||||
# Best-effort cleanup of this order's TXT before minting the replacement.
|
||||
if provider_name != "manual" and is_supported(provider_name):
|
||||
creds, _exists = await _load_credentials(conn, order["account_id"])
|
||||
if creds:
|
||||
provider = get_provider(provider_name, creds)
|
||||
chs = await conn.fetch(
|
||||
"SELECT domain, dns_txt_value FROM acme_challenges WHERE order_id = $1 AND challenge_type = 'dns-01'",
|
||||
order_id,
|
||||
)
|
||||
for ch in chs:
|
||||
try:
|
||||
await provider.remove_txt_record(ACMEService._challenge_dns_name(ch["domain"]), ch["dns_txt_value"])
|
||||
except DnsProviderError:
|
||||
pass # tolerate; the reconcile sweep will retry
|
||||
await conn.execute(
|
||||
"UPDATE acme_challenges SET dns_record_cleaned = TRUE WHERE order_id = $1 AND dns_record_published = TRUE",
|
||||
order_id,
|
||||
)
|
||||
|
||||
import json
|
||||
domains = json.loads(order["domains"]) if isinstance(order["domains"], str) else (order["domains"] or [])
|
||||
cluster_ids = json.loads(order["cluster_ids"]) if isinstance(order["cluster_ids"], str) else (order["cluster_ids"] or [])
|
||||
next_attempts = (order["dns01_attempts"] or 0) + 1
|
||||
child = await acme_svc.create_order(
|
||||
order["account_id"], domains, cluster_ids, challenge_type="dns-01",
|
||||
created_by=order["created_by"],
|
||||
)
|
||||
child_created = True
|
||||
await conn.execute(
|
||||
"""UPDATE letsencrypt_orders
|
||||
SET dns01_attempts = $1, dns01_parent_order_id = $2, dns01_last_attempt_at = NOW()
|
||||
WHERE id = $3""",
|
||||
next_attempts, order_id, child["order_id"],
|
||||
)
|
||||
await record_event(order_id, "acme.dns01.validation", severity="WARNING",
|
||||
message=f"DNS-01 order invalid; minted retry #{next_attempts} (order {child['order_id']}).",
|
||||
details={"child_order_id": child["order_id"], "attempt": next_attempts}, conn=conn)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error(f"[DNS01-RETRY] order {order_id}: {exc}")
|
||||
# A transient failure (e.g. CA rate limit) BEFORE the child was minted must NOT permanently
|
||||
# burn the retry slot — reset the claim so the next cycle can retry. If the child was already
|
||||
# created, leave the claim set (resetting would double-mint).
|
||||
if not child_created:
|
||||
try:
|
||||
await conn.execute(
|
||||
"UPDATE letsencrypt_orders SET dns01_retry_claimed = FALSE WHERE id = $1 AND status = 'invalid'",
|
||||
order_id,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def reconcile_dns01_cleanup() -> None:
|
||||
"""Best-effort sweep that removes any TXT records left published for terminal orders (covers a
|
||||
cleanup that failed, or the kill-switch being flipped off mid-flight). NOT gated by the
|
||||
kill-switch. Runs once per completion cycle."""
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
rows = await conn.fetch(
|
||||
"""SELECT c.id AS chal_id, c.order_id, c.domain, c.dns_txt_value, o.account_id, a.dns_provider
|
||||
FROM acme_challenges c
|
||||
JOIN letsencrypt_orders o ON c.order_id = o.id
|
||||
JOIN letsencrypt_accounts a ON o.account_id = a.id
|
||||
WHERE c.challenge_type = 'dns-01'
|
||||
AND c.dns_record_published = TRUE
|
||||
AND COALESCE(c.dns_record_cleaned, FALSE) = FALSE
|
||||
AND o.status IN ('valid', 'invalid', 'cancelled')
|
||||
LIMIT 50""",
|
||||
)
|
||||
for r in rows:
|
||||
provider_name = (r["dns_provider"] or "manual").strip()
|
||||
if provider_name == "manual" or not is_supported(provider_name):
|
||||
# Manual: nothing to call; mark cleaned so we stop revisiting.
|
||||
await conn.execute("UPDATE acme_challenges SET dns_record_cleaned = TRUE WHERE id = $1", r["chal_id"])
|
||||
continue
|
||||
creds, _exists = await _load_credentials(conn, r["account_id"])
|
||||
if creds is None:
|
||||
continue # can't clean without creds; leave for a later pass
|
||||
provider = get_provider(provider_name, creds)
|
||||
try:
|
||||
await provider.remove_txt_record(ACMEService._challenge_dns_name(r["domain"]), r["dns_txt_value"])
|
||||
await conn.execute("UPDATE acme_challenges SET dns_record_cleaned = TRUE WHERE id = $1", r["chal_id"])
|
||||
await record_event(r["order_id"], "acme.dns01.cleanup",
|
||||
message=f"Cleaned up TXT for {r['domain']}", conn=conn)
|
||||
except DnsProviderError:
|
||||
pass # retry next sweep
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug(f"[DNS01-RECONCILE] skipped: {exc}")
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Issue #35 — ACME DNS-01 (v1.8.0): pluggable DNS provider package.
|
||||
|
||||
A small adapter layer so DNS-01 challenges can publish/clean up the
|
||||
`_acme-challenge.<domain>` TXT record via different DNS providers. The interface is
|
||||
additive at the RRset level (add/remove a single value by name+content, never
|
||||
overwrite-by-name) so multiple coexisting values at one name (wildcard + apex) work.
|
||||
|
||||
Providers: manual (user publishes the TXT themselves), Cloudflare, and GoDaddy (v1.10.0). New
|
||||
providers plug in via the registry without touching the orchestration.
|
||||
"""
|
||||
from .base import DnsProvider, DnsProviderError
|
||||
from .registry import get_provider, list_providers, is_supported
|
||||
|
||||
__all__ = ["DnsProvider", "DnsProviderError", "get_provider", "list_providers", "is_supported"]
|
||||
@@ -0,0 +1,53 @@
|
||||
"""Abstract DNS provider interface for ACME DNS-01 (Issue #35)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Dict, List
|
||||
|
||||
|
||||
class DnsProviderError(Exception):
|
||||
"""A DNS provider failure with a SANITIZED, user-safe message.
|
||||
|
||||
The message must NEVER contain API tokens, request headers, or other secrets — it is
|
||||
persisted to acme_order_events / order error_detail and shown in the UI. Raise this (not a
|
||||
raw aiohttp/json error) so credentials can't leak into logs or the order timeline.
|
||||
"""
|
||||
|
||||
|
||||
class DnsProvider(ABC):
|
||||
"""Base class for a pluggable DNS provider.
|
||||
|
||||
RRset semantics are ADDITIVE: ``add_txt_record`` ensures a (name, value) TXT exists WITHOUT
|
||||
removing other values at the same name, and ``remove_txt_record`` deletes ONLY the record
|
||||
matching (name, value). This is required because a cert for ``example.com`` + ``*.example.com``
|
||||
publishes two distinct values at the SAME name ``_acme-challenge.example.com``.
|
||||
"""
|
||||
|
||||
# Stable machine name (used in DB + API); human label; whether the provider automates publishing.
|
||||
name: str = "base"
|
||||
label: str = "Base"
|
||||
automated: bool = True
|
||||
|
||||
# Declarative schema the UI renders to collect credentials. Each field:
|
||||
# {"key", "label", "type" ("text"|"password"), "required" (bool), "max_length" (int), "help" (str)}
|
||||
credential_fields: List[Dict] = []
|
||||
|
||||
def __init__(self, credentials: Dict[str, str] | None = None):
|
||||
self.credentials = credentials or {}
|
||||
|
||||
@abstractmethod
|
||||
async def verify_credentials(self) -> Dict:
|
||||
"""Validate the stored credentials against the provider. Returns
|
||||
``{"ok": bool, "detail": str}`` (detail is user-safe). Must not raise on auth failure —
|
||||
return ``ok=False`` with a sanitized detail; may raise DnsProviderError on transport errors.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
async def add_txt_record(self, name: str, value: str) -> None:
|
||||
"""Ensure a TXT record (name, value) exists. Idempotent; must not remove other values
|
||||
at the same name. Raise DnsProviderError (sanitized) on failure."""
|
||||
|
||||
@abstractmethod
|
||||
async def remove_txt_record(self, name: str, value: str) -> None:
|
||||
"""Remove ONLY the TXT record matching (name, value). Tolerate 'already gone'.
|
||||
Raise DnsProviderError (sanitized) on a real failure."""
|
||||
@@ -0,0 +1,179 @@
|
||||
"""Cloudflare DNS provider for ACME DNS-01 (Issue #35).
|
||||
|
||||
Uses the Cloudflare API v4 over aiohttp (no new dependency). The base URL is a hardcoded
|
||||
constant and redirects are not followed (no user-controlled URL — only the already-validated
|
||||
domain name influences which zone is used). Errors are wrapped in DnsProviderError with a
|
||||
sanitized message so the API token never reaches logs / order events.
|
||||
|
||||
Token scope required: Zone:DNS:Edit + Zone:Read.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
from urllib.parse import quote
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .base import DnsProvider, DnsProviderError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
CLOUDFLARE_API_BASE = "https://api.cloudflare.com/client/v4"
|
||||
_TIMEOUT = aiohttp.ClientTimeout(total=20)
|
||||
|
||||
# Characters NOT valid in an HTTP bearer credential (RFC 6750 token68: A-Za-z0-9-._~+/=).
|
||||
# Cloudflare API tokens are a strict subset of this set, so removing anything outside it can
|
||||
# never corrupt a valid token, but it does strip the paste artifacts that make Cloudflare
|
||||
# reject the Authorization header with HTTP 400 "Invalid request headers" (CF code 6003):
|
||||
# surrounding/embedded quotes, interior spaces/tabs, zero-width/unicode chars, and CR/LF
|
||||
# (the latter would otherwise make aiohttp raise client-side before the request is even sent).
|
||||
_NON_TOKEN68 = re.compile(r"[^A-Za-z0-9._~+/=-]")
|
||||
|
||||
|
||||
def _strip_quotes(s: str) -> str:
|
||||
s = (s or "").strip()
|
||||
if len(s) >= 2 and s[0] == '"' and s[-1] == '"':
|
||||
return s[1:-1]
|
||||
return s
|
||||
|
||||
|
||||
def _sanitize_token(s: str) -> str:
|
||||
"""Strip surrounding quotes/whitespace, then drop every character outside the token68 set."""
|
||||
return _NON_TOKEN68.sub("", _strip_quotes(s))
|
||||
|
||||
|
||||
class CloudflareDNSProvider(DnsProvider):
|
||||
name = "cloudflare"
|
||||
label = "Cloudflare"
|
||||
automated = True
|
||||
credential_fields: List[Dict] = [
|
||||
{
|
||||
"key": "api_token",
|
||||
"label": "API Token",
|
||||
"type": "password",
|
||||
"required": True,
|
||||
"max_length": 200,
|
||||
"help": "Scoped API token with Zone:DNS:Edit and Zone:Read permissions.",
|
||||
}
|
||||
]
|
||||
|
||||
def __init__(self, credentials: Dict[str, str] | None = None):
|
||||
super().__init__(credentials)
|
||||
self._raw_token = (self.credentials.get("api_token") or "").strip()
|
||||
# Sanitize to the token68 set so a pasted token with quotes/spaces/control/unicode chars
|
||||
# cannot produce an invalid Authorization header (CF 6003 "Invalid request headers").
|
||||
self._token = _sanitize_token(self._raw_token)
|
||||
|
||||
def _headers(self) -> Dict[str, str]:
|
||||
return {"Authorization": f"Bearer {self._token}", "Content-Type": "application/json"}
|
||||
|
||||
async def _request(self, session: aiohttp.ClientSession, method: str, path: str, **kwargs) -> dict:
|
||||
"""One Cloudflare API call. Returns the parsed JSON body. Raises a SANITIZED
|
||||
DnsProviderError on transport/HTTP/API error (never echoes the token or raw headers)."""
|
||||
url = f"{CLOUDFLARE_API_BASE}{path}"
|
||||
try:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
body = await resp.json()
|
||||
except Exception: # noqa: BLE001
|
||||
body = {}
|
||||
if resp.status in (401, 403):
|
||||
raise DnsProviderError("Cloudflare rejected the API token (check it has Zone:DNS:Edit + Zone:Read).")
|
||||
if resp.status >= 400 or not body.get("success", False):
|
||||
# Cloudflare returns {"errors":[{"code":..,"message":..}]} — surface only the
|
||||
# human message text, never the request (which carries the token header).
|
||||
msgs = "; ".join(
|
||||
str(e.get("message")) for e in (body.get("errors") or []) if e.get("message")
|
||||
)
|
||||
raise DnsProviderError(
|
||||
f"Cloudflare API error (HTTP {resp.status}){': ' + msgs if msgs else ''}"
|
||||
)
|
||||
return body
|
||||
except DnsProviderError:
|
||||
raise
|
||||
except aiohttp.ClientError as exc:
|
||||
# Do NOT include exc verbatim everywhere; aiohttp client errors are URL/transport only
|
||||
# (no token), but keep the message generic and stable.
|
||||
raise DnsProviderError(f"Could not reach the Cloudflare API ({type(exc).__name__}).")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise DnsProviderError(f"Unexpected Cloudflare API failure ({type(exc).__name__}).")
|
||||
|
||||
async def verify_credentials(self) -> Dict:
|
||||
if not self._token:
|
||||
return {"ok": False, "detail": "No Cloudflare API token provided."}
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
body = await self._request(session, "GET", "/zones?per_page=1")
|
||||
total = ((body.get("result_info") or {}).get("total_count"))
|
||||
detail = "Cloudflare token valid."
|
||||
if isinstance(total, int):
|
||||
detail = f"Cloudflare token valid; {total} zone(s) visible."
|
||||
return {"ok": True, "detail": detail}
|
||||
except DnsProviderError as exc:
|
||||
# Always surface the real Cloudflare reason (e.g. token scope). If sanitizing also changed
|
||||
# the token, append a hint that stray characters were stripped (never echo the token).
|
||||
detail = str(exc)
|
||||
if self._raw_token != self._token:
|
||||
detail += (" Note: the token contained characters that were stripped; if it still "
|
||||
"fails, re-copy it from Cloudflare without quotes or spaces.")
|
||||
return {"ok": False, "detail": detail}
|
||||
except Exception: # noqa: BLE001 — never leak an internal/transport error verbatim
|
||||
return {"ok": False, "detail": "Could not verify the Cloudflare token."}
|
||||
|
||||
async def _resolve_zone(self, session: aiohttp.ClientSession, record_name: str) -> Tuple[str, str]:
|
||||
"""Find the most-specific (longest-suffix) managed zone for a record name.
|
||||
Returns (zone_id, zone_name). Raises DnsProviderError if no zone matches."""
|
||||
labels = record_name.split(".")
|
||||
# Walk suffixes from longest to shortest; a zone needs at least 2 labels.
|
||||
for i in range(len(labels) - 1):
|
||||
candidate = ".".join(labels[i:])
|
||||
if candidate.count(".") < 1:
|
||||
break
|
||||
body = await self._request(
|
||||
session, "GET", f"/zones?name={quote(candidate)}&status=active&per_page=50"
|
||||
)
|
||||
results = body.get("result") or []
|
||||
if results:
|
||||
return results[0]["id"], candidate
|
||||
raise DnsProviderError(f"No managed Cloudflare zone found for {record_name}.")
|
||||
|
||||
async def _find_record_id(
|
||||
self, session: aiohttp.ClientSession, zone_id: str, name: str, value: str
|
||||
) -> Optional[str]:
|
||||
body = await self._request(
|
||||
session, "GET", f"/zones/{zone_id}/dns_records?type=TXT&name={quote(name)}&per_page=100"
|
||||
)
|
||||
for rec in body.get("result") or []:
|
||||
if _strip_quotes(rec.get("content", "")) == value:
|
||||
return rec.get("id")
|
||||
return None
|
||||
|
||||
async def add_txt_record(self, name: str, value: str) -> None:
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
zone_id, _zone_name = await self._resolve_zone(session, name)
|
||||
# Idempotent: only create if (name, value) is not already present (preserves coexisting values).
|
||||
existing = await self._find_record_id(session, zone_id, name, value)
|
||||
if existing:
|
||||
return
|
||||
await self._request(
|
||||
session,
|
||||
"POST",
|
||||
f"/zones/{zone_id}/dns_records",
|
||||
json={"type": "TXT", "name": name, "content": value, "ttl": 120},
|
||||
)
|
||||
|
||||
async def remove_txt_record(self, name: str, value: str) -> None:
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
try:
|
||||
zone_id, _zone_name = await self._resolve_zone(session, name)
|
||||
except DnsProviderError:
|
||||
# Zone gone / not resolvable — nothing we can clean up.
|
||||
return
|
||||
record_id = await self._find_record_id(session, zone_id, name, value)
|
||||
if not record_id:
|
||||
return # already gone — tolerate
|
||||
await self._request(session, "DELETE", f"/zones/{zone_id}/dns_records/{record_id}")
|
||||
@@ -0,0 +1,512 @@
|
||||
"""GoDaddy DNS provider for ACME DNS-01 (Issue #35 follow-up, v1.10.0).
|
||||
|
||||
Uses the GoDaddy Domains API v1 over aiohttp (no new dependency). The base URL is a hardcoded
|
||||
constant and redirects are not followed (no user-controlled URL — only the already-validated
|
||||
domain name selects which zone is touched), which is the same reason cloudflare.py is exempt from
|
||||
utils/ssrf_guard.py. Every failure is wrapped in DnsProviderError with a SANITIZED message: the
|
||||
API Key and Secret are scrubbed out of any text that could reach a log, an order event, or
|
||||
letsencrypt_orders.error_detail.
|
||||
|
||||
Two GoDaddy-specific hazards drive the shape of this module — neither exists on Cloudflare:
|
||||
|
||||
1. NO PER-VALUE WRITE. `PUT /v1/domains/{d}/records/TXT/{name}` REPLACES the entire RRset at that
|
||||
type+name; it does not merge. A certificate for `example.com` + `*.example.com` publishes two
|
||||
DIFFERENT TXT values at the SAME name `_acme-challenge.example.com` (base.py's additive
|
||||
contract), so a naive single-value PUT would silently destroy the sibling and fail the wildcard
|
||||
authorization. Every mutation here is therefore read-modify-write: GET the current RRset, merge,
|
||||
PUT the whole list back. An EMPTY array is rejected (422 INVALID_BODY, "Records must be
|
||||
specified"), so removing the LAST value must use DELETE — never `PUT []`.
|
||||
|
||||
2. ZONE-DESTRUCTIVE SIBLING PATHS. `PUT /v1/domains/{d}/records/TXT` (three segments, no name)
|
||||
wipes EVERY TXT in the zone — SPF, DKIM, DMARC, Microsoft/Google verification — and
|
||||
`PUT /v1/domains/{d}/records` wipes the whole zone (this is dehydrated issue #430 verbatim).
|
||||
The record path is built only by _rrset_path(), which refuses an empty zone or relative name so
|
||||
a URL can never collapse onto one of those endpoints.
|
||||
|
||||
Concurrency: v1 has no ETag, no If-Match and no per-record id, so read-modify-write can lose an
|
||||
update if two mutations at one name overlap. Today they cannot: orders are advanced sequentially
|
||||
(`for oid in claimed_ids: await advance_dns01_order(oid)` in main.py) and an order's challenges are
|
||||
published sequentially (`for ch in challenges: await provider.add_txt_record(...)` in
|
||||
dns01_orchestrator.py), so the apex+wildcard pair is strictly ordered and the second publish sees
|
||||
the first. _rrset_lock() makes that safety structural rather than incidental. Across REPLICAS the
|
||||
window is real but narrow (two orders publishing at the same record name in overlapping cycles) and
|
||||
self-healing: a lost publish ends `invalid` and the bounded retry chain mints a fresh order, a lost
|
||||
cleanup is retried by the reconcile sweep, and an orphaned `_acme-challenge` TXT is inert. The real
|
||||
fix is the v3 API (POST + DELETE by recordId, natively per-value), which is PAT-only and a
|
||||
follow-up; it is deliberately not used here because v1 + sso-key is what operators can use today.
|
||||
|
||||
Credentials: an API Key + Secret pair from https://developer.godaddy.com/keys. It must be a
|
||||
PRODUCTION key — the first key the dashboard issues is an OTE (test) key and an OTE credential
|
||||
against api.godaddy.com returns 401. A Personal Access Token also works: paste it as the API Key
|
||||
and leave the Secret blank, and the Authorization header becomes `Bearer <token>`. That path is not
|
||||
cosmetic — GoDaddy marks sso-key "deprecated, supported through 2026" and the current v1 OpenAPI
|
||||
advertises only bearer auth, so the PAT is the migration target, not an alternative.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from urllib.parse import quote
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .base import DnsProvider, DnsProviderError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
GODADDY_API_BASE = "https://api.godaddy.com/v1"
|
||||
_TIMEOUT = aiohttp.ClientTimeout(total=20)
|
||||
|
||||
# GoDaddy enforces a 600s (10 min) TTL floor at request time. The published v1 OpenAPI declares no
|
||||
# minimum, so a smaller value is not caught by the schema — it fails with
|
||||
# 422 {"code":"INVALID_BODY","fields":[{"message":"must have a minimum value of 600", ...}]}.
|
||||
# Pin the floor; DNS-01 has no reason to want anything longer.
|
||||
_TXT_TTL = 600
|
||||
|
||||
# Read-modify-write serialization, keyed by the RRset (record name), not the zone — the RRset is the
|
||||
# actual unit of contention, and keying on it avoids serializing unrelated subdomains of one zone.
|
||||
# The orchestrator is sequential today (see the module docstring), so this is defence in depth: it
|
||||
# is what stops a future `asyncio.gather()` over the publish loop from silently breaking every
|
||||
# wildcard+apex certificate. Bounded in practice by the certificate inventory of one process, so
|
||||
# there is no eviction; the entries are empty Lock objects.
|
||||
_RRSET_LOCKS: Dict[str, asyncio.Lock] = {}
|
||||
|
||||
|
||||
def _rrset_lock(record_name: str) -> asyncio.Lock:
|
||||
key = (record_name or "").rstrip(".").lower()
|
||||
lock = _RRSET_LOCKS.get(key)
|
||||
if lock is None:
|
||||
# Safe without a guard: a single event loop never preempts between the get and the assign.
|
||||
lock = _RRSET_LOCKS[key] = asyncio.Lock()
|
||||
return lock
|
||||
|
||||
|
||||
def _scrub(text: str, *secrets: str) -> str:
|
||||
"""Remove credential substrings from a message before it can reach a log or an order event.
|
||||
|
||||
GoDaddy error bodies do not echo the Authorization header, so this is belt-and-braces — but it
|
||||
makes base.py's "never leak a secret" invariant structural instead of a matter of care. Short
|
||||
strings are skipped so a 1-2 char credential fragment cannot blank out ordinary prose.
|
||||
"""
|
||||
out = text or ""
|
||||
for secret in secrets:
|
||||
if secret and len(secret) >= 4:
|
||||
out = out.replace(secret, "***")
|
||||
return out[:300]
|
||||
|
||||
|
||||
def _relative_name(fqdn: str, zone: str) -> str:
|
||||
"""Convert an absolute record name to the zone-relative form GoDaddy's API requires.
|
||||
|
||||
GoDaddy record names are RELATIVE to the zone with NO trailing dot, and the zone apex is the
|
||||
literal "@" — never an empty string (which would collapse the URL onto the zone-wide TXT
|
||||
endpoint) and never the domain name itself.
|
||||
|
||||
("_acme-challenge.example.com", "example.com") -> "_acme-challenge"
|
||||
("_acme-challenge.foo.bar.example.com", "example.com") -> "_acme-challenge.foo.bar"
|
||||
("example.com", "example.com") -> "@"
|
||||
"""
|
||||
f = (fqdn or "").rstrip(".").lower()
|
||||
z = (zone or "").rstrip(".").lower()
|
||||
if z and f == z:
|
||||
return "@"
|
||||
if z and f.endswith("." + z):
|
||||
return f[: -(len(z) + 1)]
|
||||
# Defensive: callers always pass a zone that _resolve_domain derived from this very name.
|
||||
return f or "@"
|
||||
|
||||
|
||||
def _rrset_path(zone: str, rel_name: str) -> str:
|
||||
"""Build the 4-segment record path `/domains/{zone}/records/TXT/{name}`.
|
||||
|
||||
SAFETY GATE: an empty rel_name would collapse the URL to `/domains/{zone}/records/TXT` — the
|
||||
endpoint that replaces EVERY TXT record in the zone (SPF, DKIM, DMARC, domain verifications).
|
||||
A "." or ".." segment does the same thing one step later: `quote()` leaves both untouched
|
||||
(they are unreserved) and yarl normalizes dot segments away when it builds the URL, so
|
||||
".../records/TXT/.." would resolve to ".../records" — the whole-zone endpoint. Refuse both
|
||||
rather than build them. `safe=''` percent-encodes the apex "@" as "%40" (accepted bare too,
|
||||
but safer through proxies); "_", "-" and "." are unreserved and pass through unchanged, so a
|
||||
multi-label relative name stays one readable path segment.
|
||||
"""
|
||||
if not zone or not rel_name:
|
||||
raise DnsProviderError("Internal error: refusing to build a zone-wide GoDaddy TXT record path.")
|
||||
if rel_name.strip(".") == "" or any(part in (".", "..") for part in rel_name.split("/")):
|
||||
raise DnsProviderError("Internal error: refusing to build a GoDaddy TXT path from a dot segment.")
|
||||
return f"/domains/{quote(zone, safe='')}/records/TXT/{quote(rel_name, safe='')}"
|
||||
|
||||
|
||||
def _live_values(records: List[Dict]) -> List[str]:
|
||||
"""The non-empty `data` values in an RRset read.
|
||||
|
||||
GoDaddy leaves tombstone rows with `"data": ""` behind at a name after some removals. Echoing
|
||||
one back in a PUT body is rejected with 422 INVALID_BODY, so every field implementation
|
||||
(lego, acme.sh, Posh-ACME) filters them independently — so do we.
|
||||
"""
|
||||
out: List[str] = []
|
||||
for rec in records or []:
|
||||
data = (rec or {}).get("data") or ""
|
||||
if data:
|
||||
out.append(data)
|
||||
return out
|
||||
|
||||
|
||||
def _merge_add(existing: List[Dict], value: str) -> Optional[List[Dict]]:
|
||||
"""PUT body that adds `value` while preserving every coexisting sibling value.
|
||||
|
||||
Returns None when `value` is already present — an idempotent no-op, which is where an ACME
|
||||
retry cycle lands.
|
||||
"""
|
||||
live = _live_values(existing)
|
||||
if value in live:
|
||||
return None
|
||||
return [{"data": d, "ttl": _TXT_TTL} for d in live] + [{"data": value, "ttl": _TXT_TTL}]
|
||||
|
||||
|
||||
def _merge_remove(existing: List[Dict], value: str) -> Optional[List[Dict]]:
|
||||
"""PUT body that removes ONLY `value`, keeping every sibling.
|
||||
|
||||
Three-state result, because GoDaddy needs three different calls:
|
||||
None -> `value` is not there; already gone, tolerate (base.py's remove contract).
|
||||
[] -> it was the last value; the caller must DELETE, since `PUT []` is rejected.
|
||||
list -> PUT this body.
|
||||
"""
|
||||
live = _live_values(existing)
|
||||
if value not in live:
|
||||
return None
|
||||
return [{"data": d, "ttl": _TXT_TTL} for d in live if d != value]
|
||||
|
||||
|
||||
def _require_rrset(body: Any) -> List[Dict]:
|
||||
"""The RRset read, or a refusal.
|
||||
|
||||
FAIL CLOSED. A read that did not come back as a JSON array must never be treated as "the RRset
|
||||
is empty" — the very next call is a full-RRset PUT, so coercing an unreadable read to [] would
|
||||
replace every coexisting sibling value with just ours. Failing instead is free: the orchestrator
|
||||
reverts the publish flag and retries next cycle, while a destructive PUT is unrecoverable.
|
||||
"""
|
||||
if not isinstance(body, list):
|
||||
raise DnsProviderError(
|
||||
"GoDaddy returned an unreadable TXT record list; refusing to replace the record set."
|
||||
)
|
||||
return body
|
||||
|
||||
|
||||
def _error_fields(body: Any) -> Tuple[str, str]:
|
||||
"""The whitelisted (code, message) pair from a GoDaddy error body.
|
||||
|
||||
Only these two string fields are ever read; the raw body is never interpolated into a
|
||||
user-facing message.
|
||||
"""
|
||||
if not isinstance(body, dict):
|
||||
return "", ""
|
||||
code = body.get("code")
|
||||
message = body.get("message")
|
||||
return (code if isinstance(code, str) else ""), (message if isinstance(message, str) else "")
|
||||
|
||||
|
||||
def _retry_after_seconds(headers, body: Any) -> int:
|
||||
"""Seconds to wait after a 429.
|
||||
|
||||
The current platform sends `Retry-After` and `ratelimit-reset` headers with no body, while the
|
||||
legacy v1 OpenAPI documents an `ErrorLimit` body carrying `retryAfterSec`. All three shapes are
|
||||
live in the wild — and so is none of them, hence the 60s default.
|
||||
"""
|
||||
for key in ("Retry-After", "ratelimit-reset"):
|
||||
raw = (headers or {}).get(key)
|
||||
if raw:
|
||||
try:
|
||||
return max(1, int(str(raw).strip()))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if isinstance(body, dict):
|
||||
raw = body.get("retryAfterSec")
|
||||
if isinstance(raw, int) and raw > 0:
|
||||
return raw
|
||||
return 60
|
||||
|
||||
|
||||
class _GoDaddyHTTPError(DnsProviderError):
|
||||
"""A DnsProviderError that also carries the HTTP status and GoDaddy `code`.
|
||||
|
||||
Callers INSIDE this module branch on the status (tolerate a 404 read-back, fall through a
|
||||
zone probe), while everything outside — dns01_orchestrator, letsencrypt.py — still sees a
|
||||
plain sanitized DnsProviderError and needs no change.
|
||||
"""
|
||||
|
||||
def __init__(self, message: str, status: int, code: str = ""):
|
||||
super().__init__(message)
|
||||
self.status = status
|
||||
self.code = code
|
||||
|
||||
|
||||
class GoDaddyDNSProvider(DnsProvider):
|
||||
name = "godaddy"
|
||||
label = "GoDaddy"
|
||||
automated = True
|
||||
credential_fields: List[Dict] = [
|
||||
{
|
||||
"key": "api_key",
|
||||
"label": "API Key",
|
||||
"type": "password",
|
||||
"required": True,
|
||||
"max_length": 200,
|
||||
"help": ("Production API Key from developer.godaddy.com/keys — the first key the dashboard "
|
||||
"issues is an OTE (test) key and will be rejected. A Personal Access Token also "
|
||||
"works: paste it here and leave the Secret blank."),
|
||||
},
|
||||
{
|
||||
"key": "api_secret",
|
||||
"label": "API Secret",
|
||||
"type": "password",
|
||||
"required": False,
|
||||
"max_length": 200,
|
||||
"help": ("The Secret half of the same API Key pair. Leave blank ONLY if the field above "
|
||||
"holds a Personal Access Token. The account also needs at least one registered "
|
||||
"domain for GoDaddy to allow DNS API access at all."),
|
||||
},
|
||||
]
|
||||
|
||||
def __init__(self, credentials: Dict[str, str] | None = None):
|
||||
super().__init__(credentials)
|
||||
# Normalize, never validate: dns01_orchestrator.py calls get_provider() OUTSIDE any
|
||||
# DnsProviderError guard, so a constructor that raised on malformed credentials would escape
|
||||
# as an unhandled exception in the 60s background cycle. The UI drops blank fields before
|
||||
# submitting, so a left-blank field arrives as a MISSING key rather than "" — `.get() or ""`
|
||||
# covers both.
|
||||
self._api_key = (self.credentials.get("api_key") or "").strip()
|
||||
self._api_secret = (self.credentials.get("api_secret") or "").strip()
|
||||
# Per-INSTANCE zone cache. A module-level cache would leak one ACME account's zone visibility
|
||||
# into another's; an instance lives for exactly one orchestrator step, which is precisely the
|
||||
# scope where caching pays off (apex + wildcard resolve the same zone from the same name).
|
||||
self._zone_cache: Dict[str, str] = {}
|
||||
|
||||
def _auth_header(self) -> str:
|
||||
"""`sso-key <key>:<secret>` when a Secret is present, else `Bearer <token>` for a PAT.
|
||||
|
||||
Literal prefix, one space, a single colon — no base64, no URL-encoding, no quotes. Keeping
|
||||
this as one swappable string is what makes GoDaddy's sso-key sunset a credential change
|
||||
rather than a code change.
|
||||
"""
|
||||
if self._api_secret:
|
||||
return f"sso-key {self._api_key}:{self._api_secret}"
|
||||
return f"Bearer {self._api_key}"
|
||||
|
||||
def _headers(self) -> Dict[str, str]:
|
||||
# Accept is not optional: these endpoints content-negotiate application/xml and
|
||||
# text/javascript. Content-Type is required on every write or GoDaddy answers 400/415.
|
||||
return {
|
||||
"Authorization": self._auth_header(),
|
||||
"Accept": "application/json",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
|
||||
def _http_error(self, status: int, code: str, message: str, retry_after: Optional[int]) -> _GoDaddyHTTPError:
|
||||
"""Map an HTTP status to a sanitized, operator-actionable DnsProviderError.
|
||||
|
||||
These strings land in acme_order_events and letsencrypt_orders.error_detail and are shown
|
||||
in the order timeline, so each one names what to fix. GoDaddy's own `code`/`message` is
|
||||
appended when present because the two 403 causes — account not eligible for the DNS API vs.
|
||||
a PAT missing `domains.dns:update` — are indistinguishable by status alone. Scrubbing
|
||||
happens HERE, at the single point where provider-supplied text enters a message, so a new
|
||||
caller cannot forget it.
|
||||
"""
|
||||
code = _scrub(code, self._api_key, self._api_secret)
|
||||
message = _scrub(message, self._api_key, self._api_secret)
|
||||
if status == 401:
|
||||
detail = ("GoDaddy rejected the API credentials. Check they are a PRODUCTION Key/Secret pair "
|
||||
"from developer.godaddy.com/keys — the first key the dashboard issues is an OTE "
|
||||
"(test) key and is not valid here.")
|
||||
elif status == 403:
|
||||
detail = ("GoDaddy denied access to the DNS API. The account needs at least one registered "
|
||||
"domain, and a Personal Access Token needs the domains.domain:read and "
|
||||
"domains.dns:update scopes.")
|
||||
elif status == 404:
|
||||
detail = ("GoDaddy has no zone for this domain (check it is registered in this account and "
|
||||
"uses GoDaddy nameservers).")
|
||||
elif status == 409:
|
||||
detail = "GoDaddy reports this domain is not eligible to have its DNS records changed."
|
||||
elif status == 422:
|
||||
detail = "GoDaddy rejected the record change as invalid (HTTP 422)."
|
||||
elif status == 429:
|
||||
detail = f"GoDaddy rate limit reached; retry in ~{retry_after or 60}s."
|
||||
else:
|
||||
detail = f"GoDaddy API error (HTTP {status})."
|
||||
if code or message:
|
||||
detail += f" (GoDaddy: {code}{': ' + message if message else ''})"
|
||||
return _GoDaddyHTTPError(detail, status=status, code=code)
|
||||
|
||||
async def _request(self, session: aiohttp.ClientSession, method: str, path: str, **kwargs) -> Any:
|
||||
"""One GoDaddy API call. Returns the parsed JSON body, or None for the empty-bodied writes.
|
||||
|
||||
Raises a SANITIZED _GoDaddyHTTPError / DnsProviderError — never the credentials, never the
|
||||
request, never a response body verbatim.
|
||||
"""
|
||||
url = f"{GODADDY_API_BASE}{path}"
|
||||
try:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
# content_type=None: every GoDaddy write answers 200/204 with an EMPTY body, and
|
||||
# aiohttp would otherwise raise on the missing/other content type before parsing.
|
||||
body = await resp.json(content_type=None)
|
||||
except ValueError:
|
||||
# ONLY a decode failure (JSONDecodeError subclasses ValueError) is swallowed —
|
||||
# an empty write body, or an HTML error page on a >=400. A transport failure
|
||||
# mid-read (ClientPayloadError, TimeoutError) must NOT land here: it would look
|
||||
# identical to "empty body", and a caller that reads an RRset would then see
|
||||
# None and could mistake it for an empty RRset. Those propagate to the handlers
|
||||
# below and become a real DnsProviderError.
|
||||
body = None
|
||||
# 2xx only. Redirects are deliberately not followed (aiohttp would forward the
|
||||
# Authorization header), so a 3xx is a failed call — treating `< 400` as success
|
||||
# would report a redirected write as a silent no-op.
|
||||
if 200 <= resp.status < 300:
|
||||
return body
|
||||
code, message = _error_fields(body)
|
||||
retry_after = _retry_after_seconds(resp.headers, body) if resp.status == 429 else None
|
||||
raise self._http_error(resp.status, code, message, retry_after)
|
||||
except DnsProviderError:
|
||||
raise
|
||||
except aiohttp.ClientError as exc:
|
||||
# Only the exception TYPE is interpolated: an aiohttp client error's str() can carry the
|
||||
# request URL, and the message is persisted to the order timeline.
|
||||
raise DnsProviderError(f"Could not reach the GoDaddy API ({type(exc).__name__}).")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise DnsProviderError(f"Unexpected GoDaddy API failure ({type(exc).__name__}).")
|
||||
|
||||
async def verify_credentials(self) -> Dict:
|
||||
if not self._api_key:
|
||||
return {"ok": False, "detail": "No GoDaddy API Key provided."}
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
# Cheapest read-only check: one request, no zone needed. Deliberately NOT
|
||||
# GET /v1/domains/{domain} — GoDaddy has rejected that details call for small
|
||||
# accounts since 2024-05 while record-level calls keep working, so verifying with it
|
||||
# produces false negatives on accounts where DNS-01 would succeed.
|
||||
body = await self._request(session, "GET", "/domains?limit=1")
|
||||
if not isinstance(body, list):
|
||||
return {"ok": False, "detail": "GoDaddy returned an unexpected response to the credential check."}
|
||||
if not body:
|
||||
# An empty list is NOT a failure: sub-zones delegated to GoDaddy nameservers are
|
||||
# manageable via the records API but never appear in the domain listing.
|
||||
return {"ok": True, "detail": ("GoDaddy credentials valid, but no domains are visible in this "
|
||||
"account — the domain you validate must be registered here, or "
|
||||
"be a zone delegated to GoDaddy nameservers.")}
|
||||
return {"ok": True, "detail": "GoDaddy credentials valid."}
|
||||
except DnsProviderError as exc:
|
||||
detail = str(exc)
|
||||
if not self._api_secret:
|
||||
# The Bearer path is silent otherwise, and a half-filled form is the likeliest cause.
|
||||
detail += (" Note: no API Secret was entered, so the API Key was sent as a Personal Access "
|
||||
"Token (Bearer). If you have a Key + Secret pair, enter both halves.")
|
||||
return {"ok": False, "detail": detail}
|
||||
except Exception: # noqa: BLE001 — never leak an internal/transport error verbatim
|
||||
return {"ok": False, "detail": "Could not verify the GoDaddy credentials."}
|
||||
|
||||
async def _resolve_domain(self, session: aiohttp.ClientSession, record_name: str) -> str:
|
||||
"""Find the most-specific (longest-suffix) GoDaddy-managed zone for an absolute record name.
|
||||
|
||||
GoDaddy has no `/zones?name=` equivalent, so this walks suffixes longest-to-shortest and
|
||||
probes `GET /v1/domains/{candidate}/records/NS`. That probe (rather than the domain listing
|
||||
or the domain-details call) is deliberate: it finds sub-zones delegated to GoDaddy
|
||||
nameservers, which never appear in `GET /v1/domains` at all, and it does not depend on the
|
||||
details endpoint that small accounts are rejected from.
|
||||
"""
|
||||
cached = self._zone_cache.get(record_name)
|
||||
if cached:
|
||||
return cached
|
||||
labels = record_name.rstrip(".").lower().split(".")
|
||||
for i in range(len(labels) - 1):
|
||||
candidate = ".".join(labels[i:])
|
||||
if candidate.count(".") < 1:
|
||||
break # a zone needs at least two labels
|
||||
try:
|
||||
body = await self._request(
|
||||
session, "GET", f"/domains/{quote(candidate, safe='')}/records/NS"
|
||||
)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status in (404, 422):
|
||||
continue # not a zone in this account — keep walking
|
||||
# 401/403/409/429/5xx are credential, eligibility or platform failures, not
|
||||
# "wrong zone". Continuing would burn the rate-limit budget re-failing on every
|
||||
# remaining suffix and would bury the real cause under "no managed domain".
|
||||
raise
|
||||
if isinstance(body, list) and body:
|
||||
self._zone_cache[record_name] = candidate
|
||||
return candidate
|
||||
raise DnsProviderError(f"No managed GoDaddy domain found for {record_name}.")
|
||||
|
||||
async def add_txt_record(self, name: str, value: str) -> None:
|
||||
async with _rrset_lock(name):
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
zone = await self._resolve_domain(session, name)
|
||||
path = _rrset_path(zone, _relative_name(name, zone))
|
||||
try:
|
||||
existing = await self._request(session, "GET", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status != 404:
|
||||
raise
|
||||
# Some accounts 404 reading back a record set in a zone whose WRITES succeed
|
||||
# (acme.sh #6517). Reachable only when the NS probe resolved the zone but the
|
||||
# TXT read 404s — if the NS probe itself 404s we never get here and the caller
|
||||
# sees "No managed GoDaddy domain found", which is the honest answer. We cannot
|
||||
# merge what we cannot read, and a single-value PUT would destroy any coexisting
|
||||
# sibling, so PATCH is the only correct recovery: it is the one genuinely
|
||||
# ADDITIVE primitive in v1 ("Appends DNS records ... Existing records with the
|
||||
# same type and name are preserved"). It cannot dedupe, but a duplicate
|
||||
# identical TXT is harmless for validation and cleanup removes the whole RRset.
|
||||
await self._request(
|
||||
session, "PATCH", f"/domains/{quote(zone, safe='')}/records",
|
||||
json=[{"type": "TXT", "name": _relative_name(name, zone),
|
||||
"data": value, "ttl": _TXT_TTL}],
|
||||
)
|
||||
return
|
||||
body = _merge_add(_require_rrset(existing), value)
|
||||
if body is None:
|
||||
return # already published — idempotent, this is where ACME retries land
|
||||
await self._request(session, "PUT", path, json=body)
|
||||
|
||||
async def remove_txt_record(self, name: str, value: str) -> None:
|
||||
async with _rrset_lock(name):
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
try:
|
||||
zone = await self._resolve_domain(session, name)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
# Raise only what a later sweep could plausibly succeed at. reconcile_dns01_cleanup
|
||||
# swallows the error and leaves dns_record_cleaned FALSE, so the row is re-selected
|
||||
# every cycle — and its query takes a bare LIMIT 50, so rows that can NEVER succeed
|
||||
# (revoked key, account lost DNS-API eligibility) would monopolise the whole
|
||||
# cleanup budget and starve every other account. For those terminal statuses we
|
||||
# give up quietly: the orphaned `_acme-challenge` TXT is inert, and the same
|
||||
# credential failure is already loud on the publish path, where it is actionable.
|
||||
if exc.status == 429 or exc.status >= 500:
|
||||
raise
|
||||
return
|
||||
except DnsProviderError:
|
||||
return # zone genuinely not resolvable — nothing we could clean up
|
||||
path = _rrset_path(zone, _relative_name(name, zone))
|
||||
try:
|
||||
existing = await self._request(session, "GET", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status == 404:
|
||||
return # RRset (or the read) is gone — tolerate
|
||||
raise
|
||||
body = _merge_remove(_require_rrset(existing), value)
|
||||
if body is None:
|
||||
return # our value is not there — already gone, tolerate
|
||||
if not body:
|
||||
# The LAST value at this name. `PUT []` is rejected (422 INVALID_BODY, "Records
|
||||
# must be specified"), so emptying an RRset REQUIRES DELETE. This removes only
|
||||
# TXT at this exact name; other names and other record types are preserved.
|
||||
# Do NOT fall back to the "write an empty string to delete" folklore — that hack
|
||||
# is what creates the tombstone rows _live_values has to filter.
|
||||
try:
|
||||
await self._request(session, "DELETE", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status == 404:
|
||||
return # raced with another cleanup — tolerate
|
||||
raise
|
||||
return
|
||||
await self._request(session, "PUT", path, json=body)
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Manual DNS provider for ACME DNS-01 (Issue #35).
|
||||
|
||||
The user publishes the `_acme-challenge` TXT record in their own DNS (any provider, including
|
||||
fully internal/isolated DNS that no API can reach) and then confirms via the UI. There is no API
|
||||
to call, so add/remove are no-ops and the orchestration waits for an explicit `dns-confirm`.
|
||||
CNAME delegation works implicitly here: the CA follows a CNAME, so a user who delegates
|
||||
`_acme-challenge` elsewhere just publishes the value there and confirms.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Dict, List
|
||||
|
||||
from .base import DnsProvider
|
||||
|
||||
|
||||
class ManualDNSProvider(DnsProvider):
|
||||
name = "manual"
|
||||
label = "Manual (publish the TXT record yourself)"
|
||||
automated = False
|
||||
credential_fields: List[Dict] = [] # no credentials needed
|
||||
|
||||
async def verify_credentials(self) -> Dict:
|
||||
return {"ok": True, "detail": "Manual mode needs no credentials. You will publish the TXT record yourself."}
|
||||
|
||||
async def add_txt_record(self, name: str, value: str) -> None:
|
||||
# No-op: the user publishes the record and confirms via the UI.
|
||||
return None
|
||||
|
||||
async def remove_txt_record(self, name: str, value: str) -> None:
|
||||
# No-op: the user may remove the record manually after issuance.
|
||||
return None
|
||||
@@ -0,0 +1,46 @@
|
||||
"""DNS provider registry for ACME DNS-01 (Issue #35).
|
||||
|
||||
Single source of truth mapping a provider name -> class. The API serves the credential-field
|
||||
schema from here (so the UI has no hardcoded provider fields) and validates inbound provider
|
||||
names against this allow-list. Adding a provider = add it here; nothing else changes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Dict, List, Type
|
||||
|
||||
from .base import DnsProvider
|
||||
from .cloudflare import CloudflareDNSProvider
|
||||
from .godaddy import GoDaddyDNSProvider
|
||||
from .manual import ManualDNSProvider
|
||||
|
||||
_PROVIDERS: Dict[str, Type[DnsProvider]] = {
|
||||
ManualDNSProvider.name: ManualDNSProvider,
|
||||
CloudflareDNSProvider.name: CloudflareDNSProvider,
|
||||
GoDaddyDNSProvider.name: GoDaddyDNSProvider,
|
||||
}
|
||||
|
||||
|
||||
def is_supported(name: str) -> bool:
|
||||
return name in _PROVIDERS
|
||||
|
||||
|
||||
def get_provider(name: str, credentials: Dict[str, str] | None = None) -> DnsProvider:
|
||||
cls = _PROVIDERS.get(name)
|
||||
if cls is None:
|
||||
raise ValueError(f"Unsupported DNS provider: {name}")
|
||||
return cls(credentials or {})
|
||||
|
||||
|
||||
def list_providers() -> List[Dict]:
|
||||
"""Return the UI-facing provider catalog: name, label, automated flag, and credential schema."""
|
||||
out: List[Dict] = []
|
||||
for name, cls in _PROVIDERS.items():
|
||||
out.append(
|
||||
{
|
||||
"name": cls.name,
|
||||
"label": cls.label,
|
||||
"automated": cls.automated,
|
||||
"credential_fields": cls.credential_fields,
|
||||
}
|
||||
)
|
||||
return out
|
||||
@@ -56,14 +56,14 @@ async def create_frontend_row(
|
||||
acl_rules, redirect_rules, use_backend_rules,
|
||||
request_headers, response_headers, options, tcp_request_rules, timeout_client, timeout_http_request,
|
||||
rate_limit, compression, log_separate, monitor_uri,
|
||||
cluster_id, maxconn, updated_at
|
||||
cluster_id, maxconn, log_format, filters, updated_at
|
||||
) VALUES (
|
||||
$1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12,
|
||||
$13, $14, $15, $16, $17, $18, $19,
|
||||
$20, $21, $22,
|
||||
$23, $24, $25, $26, $27, $28,
|
||||
$29, $30, $31, $32,
|
||||
$33, $34, CURRENT_TIMESTAMP
|
||||
$33, $34, $35, $36, CURRENT_TIMESTAMP
|
||||
)
|
||||
RETURNING id
|
||||
""",
|
||||
@@ -101,6 +101,8 @@ async def create_frontend_row(
|
||||
getattr(payload, "monitor_uri", None),
|
||||
cluster_id,
|
||||
getattr(payload, "maxconn", None),
|
||||
getattr(payload, "log_format", None), # Issue #38
|
||||
getattr(payload, "filters", None), # Issue #38
|
||||
)
|
||||
|
||||
if mark_pending:
|
||||
|
||||
@@ -393,6 +393,12 @@ def _categorize_haproxy_directive(line: str) -> str:
|
||||
return "prelude"
|
||||
if s.startswith("acl "):
|
||||
return "acl"
|
||||
# Issue #38: SPOE (and other) `filter` directives must be declared BEFORE
|
||||
# the `http-request send-spoe-group` rules that use them, otherwise HAProxy
|
||||
# fails with "unable to find SPOE engine". Own bucket, flushed right after
|
||||
# `prelude` and before tcp_req/acl/http_req (see flush order below).
|
||||
if s.startswith("filter "):
|
||||
return "filter"
|
||||
if s.startswith("stick-table") or s.startswith("stick "):
|
||||
return "stick"
|
||||
if s.startswith("tcp-request"):
|
||||
@@ -414,6 +420,7 @@ def _categorize_haproxy_directive(line: str) -> str:
|
||||
or s.startswith("compression ")
|
||||
or s.startswith("monitor-uri")
|
||||
or s.startswith("log ")
|
||||
or s.startswith("log-format") # Issue #38: log-format / log-format-sd
|
||||
or s.startswith("description ")
|
||||
or s.startswith("disabled")
|
||||
or s.startswith("enabled")
|
||||
@@ -903,11 +910,18 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
# "stick-table already declared").
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
_fe_buckets: Dict[str, List[str]] = {
|
||||
"prelude": [], "stick": [], "tcp_req": [],
|
||||
"prelude": [], "filter": [], "stick": [], "tcp_req": [],
|
||||
"acl": [], "http_req": [], "http_resp": [],
|
||||
"redirect": [], "use_be": [], "default_be": [],
|
||||
}
|
||||
_stick_table_emitted = False
|
||||
# Whether this frontend emits ANY stick-counter usage (`track-sc<N>` or an
|
||||
# `sc_*_rate(...)` fetch). If it does but no `stick-table` is declared, HAProxy
|
||||
# fatally rejects the WHOLE cluster config with "table '<frontend>' used but not
|
||||
# configured". This happens with rate-limit directives baked into a frontend's
|
||||
# stored fields (request_headers/options) by an older version or a config import.
|
||||
# We track it here and inject a default stick-table before flushing if needed.
|
||||
_sc_counter_used = False
|
||||
# Phase K Phase D follow-up (Bulgu #13) — same dedup
|
||||
# contract for `http-request track-sc<N> <fetch>` lines.
|
||||
# HAProxy only NEEDS one tracking call per
|
||||
@@ -931,9 +945,13 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
correct frontend-block bucket. Idempotent for stick-table
|
||||
lines (R3.3 dedup) AND http-request track-sc<N> lines
|
||||
(Bulgu #13 dedup)."""
|
||||
nonlocal _stick_table_emitted
|
||||
nonlocal _stick_table_emitted, _sc_counter_used
|
||||
cat = _categorize_haproxy_directive(line)
|
||||
stripped = line.strip()
|
||||
# Any stick-counter usage (track-sc<N> write, or an sc_*_rate(...) fetch like
|
||||
# sc_http_req_rate(0)) requires a stick-table in this frontend.
|
||||
if "track-sc" in stripped or ("sc_" in stripped and "_rate(" in stripped):
|
||||
_sc_counter_used = True
|
||||
if cat == "stick":
|
||||
if _stick_table_emitted and stripped.startswith("stick-table"):
|
||||
logger.debug(
|
||||
@@ -985,6 +1003,17 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
if line_stripped and line_stripped not in ('[]', '{}', 'null', 'None'):
|
||||
_emit_fe(f" {line_stripped}")
|
||||
|
||||
# Issue #38: emit frontend log-format and SPOE (etc.) filter directives.
|
||||
# `log_format` routes to the `prelude` bucket, `filters` to the `filter`
|
||||
# bucket (both via _emit_fe → _categorize_haproxy_directive), guaranteeing
|
||||
# `filter ...` is rendered before the `http-request send-spoe-group` rules.
|
||||
for _fld in ('log_format', 'filters'):
|
||||
if frontend.get(_fld):
|
||||
for line in frontend[_fld].split('\n'):
|
||||
line_stripped = line.strip()
|
||||
if line_stripped and line_stripped not in ('[]', '{}', 'null', 'None'):
|
||||
_emit_fe(f" {line_stripped}")
|
||||
|
||||
# CRITICAL: Validate frontend-backend mode compatibility
|
||||
if frontend.get('default_backend'):
|
||||
default_backend_name = frontend['default_backend'].strip() if frontend['default_backend'] else ''
|
||||
@@ -1188,6 +1217,22 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
else:
|
||||
logger.warning(f"Config Generation: No config lines generated for WAF rule '{waf_rule['name']}' (ID: {waf_rule['id']}, Type: {waf_rule['rule_type']})")
|
||||
|
||||
# Robustness fix: if this frontend uses a stick counter (track-sc<N> or an
|
||||
# sc_*_rate(...) fetch) but declared NO stick-table, inject a default one so HAProxy
|
||||
# doesn't fatally reject the whole cluster config with "table '<frontend>' used but
|
||||
# not configured". This rescues rate-limit directives baked into a frontend's stored
|
||||
# request_headers/options by an older version or import. Purely additive — it only
|
||||
# fires when a counter is used AND no table exists (a config that is invalid today),
|
||||
# so it never changes a frontend that already has a stick-table or doesn't rate-limit.
|
||||
if _sc_counter_used and not _stick_table_emitted:
|
||||
_fe_buckets["stick"].insert(
|
||||
0, " stick-table type ip size 100k expire 30s store http_req_rate(10s)")
|
||||
_stick_table_emitted = True
|
||||
logger.info(
|
||||
f"STICK-TABLE AUTO-INJECT: frontend '{frontend['name']}' uses a stick "
|
||||
f"counter (track-sc/sc_*_rate) but declared no stick-table; injected a "
|
||||
f"default so the config stays valid.")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────
|
||||
# Flush the per-frontend buckets in canonical HAProxy order.
|
||||
# The order below is the single source of truth for emit
|
||||
@@ -1196,6 +1241,7 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
# ─────────────────────────────────────────────────────────────
|
||||
for _bucket_key in (
|
||||
"prelude",
|
||||
"filter",
|
||||
"stick",
|
||||
"tcp_req",
|
||||
"acl",
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
"""Issue #27 — HA/VIP (Keepalived) management (v1.7.0).
|
||||
|
||||
Standalone, DB-free renderer for a node's /etc/keepalived/keepalived.conf and the
|
||||
HAProxy health-check script, plus Fernet at-rest encryption for the VRRP secret.
|
||||
|
||||
The router fetches DB rows and calls these pure functions; nothing here touches the
|
||||
database or logs secrets. Trivially unit-testable (see tests/test_keepalived_config.py).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from typing import List, Optional
|
||||
|
||||
from cryptography.fernet import Fernet, InvalidToken
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
from config import SECRET_KEY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Written into every file we manage so the agent can tell "ours" from a
|
||||
# hand-maintained keepalived setup (ownership guard, B-3/T-2). Must match the
|
||||
# string the agent greps for in linux_install.sh.
|
||||
OWNERSHIP_MARKER = "# Managed by HAProxy OpenManager"
|
||||
CHECK_SCRIPT_PATH = "/etc/keepalived/check_haproxy.sh"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VRRP secret at rest (mirrors backend/services/mfa_service.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
_fernet_instance: Optional[Fernet] = None
|
||||
|
||||
|
||||
def _resolve_fernet_key() -> bytes:
|
||||
"""Prefer an explicit VIP_ENCRYPTION_KEY; else derive from SECRET_KEY via HKDF
|
||||
with a versioned info string (so the secret survives restarts, like MFA)."""
|
||||
explicit = os.getenv("VIP_ENCRYPTION_KEY", "").strip()
|
||||
if explicit:
|
||||
try:
|
||||
Fernet(explicit.encode())
|
||||
return explicit.encode()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("VIP_ENCRYPTION_KEY env var present but invalid: %s", exc)
|
||||
hkdf = HKDF(algorithm=hashes.SHA256(), length=32, salt=None, info=b"vip-vrrp-secret-v1")
|
||||
derived = hkdf.derive(SECRET_KEY.encode("utf-8"))
|
||||
return base64.urlsafe_b64encode(derived)
|
||||
|
||||
|
||||
def _get_fernet() -> Fernet:
|
||||
global _fernet_instance
|
||||
if _fernet_instance is None:
|
||||
_fernet_instance = Fernet(_resolve_fernet_key())
|
||||
return _fernet_instance
|
||||
|
||||
|
||||
def reset_fernet_for_tests() -> None:
|
||||
"""Test-only hook to force re-resolution after env mutation."""
|
||||
global _fernet_instance
|
||||
_fernet_instance = None
|
||||
|
||||
|
||||
def encrypt_vrrp_secret(secret_plain: str) -> str:
|
||||
return _get_fernet().encrypt(secret_plain.encode("utf-8")).decode("utf-8")
|
||||
|
||||
|
||||
def decrypt_vrrp_secret(secret_encrypted: str) -> Optional[str]:
|
||||
try:
|
||||
return _get_fernet().decrypt(secret_encrypted.encode("utf-8")).decode("utf-8")
|
||||
except InvalidToken:
|
||||
logger.warning("Failed to decrypt VRRP secret (invalid Fernet token)")
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Unexpected error decrypting VRRP secret: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Renderers
|
||||
# ---------------------------------------------------------------------------
|
||||
def build_haproxy_check_script(*, bin_path: Optional[str] = None,
|
||||
config_path: Optional[str] = None) -> str:
|
||||
"""Render the health-check the agent writes to CHECK_SCRIPT_PATH.
|
||||
|
||||
Derives the process name from the HAProxy binary basename (B-4) rather than a
|
||||
blind hardcoded 'haproxy'. Returns non-zero when HAProxy isn't running so the
|
||||
VRRP track_script lowers this node's priority and the VIP fails over.
|
||||
"""
|
||||
proc = "haproxy"
|
||||
if bin_path:
|
||||
base = os.path.basename(bin_path.strip())
|
||||
if re.match(r'^[A-Za-z0-9._-]{1,64}$', base):
|
||||
proc = base
|
||||
return (
|
||||
"#!/bin/sh\n"
|
||||
f"{OWNERSHIP_MARKER} — DO NOT EDIT\n"
|
||||
"# Exits 0 while HAProxy is up; non-zero triggers VRRP failover.\n"
|
||||
f"pidof {proc} >/dev/null 2>&1 || exit 1\n"
|
||||
"exit 0\n"
|
||||
)
|
||||
|
||||
|
||||
def _vrrp_instance_name(vip_id: int) -> str:
|
||||
return f"VI_{int(vip_id)}"
|
||||
|
||||
|
||||
def _failover_weight(members: List[dict]) -> int:
|
||||
"""Negative weight so a failed MASTER drops strictly below every healthy BACKUP
|
||||
(B-6). master_priority + weight < min(backup_priority)."""
|
||||
master = next((m for m in members if str(m.get("role", "")).upper() == "MASTER"), None)
|
||||
backups = [int(m["priority"]) for m in members if str(m.get("role", "")).upper() != "MASTER"]
|
||||
if not master or not backups:
|
||||
return -20
|
||||
return -((int(master["priority"]) - min(backups)) + 1)
|
||||
|
||||
|
||||
def render_keepalived_conf(*, vip: dict, members: List[dict], this_agent: dict,
|
||||
peer_ips: List[str], auth_pass_plain: Optional[str]) -> str:
|
||||
"""Render one node's keepalived.conf from the VIP + member rows.
|
||||
|
||||
`vip` keys: id, name, virtual_ip, prefix_length, virtual_router_id, advert_int,
|
||||
use_unicast, track_haproxy.
|
||||
`this_agent` keys: role, priority, network_interface, ip_address (str).
|
||||
`peer_ips`: the OTHER members' ip_address strings (already str()'d by the caller).
|
||||
Caller must never log the returned string (it may contain auth_pass).
|
||||
"""
|
||||
role = str(this_agent["role"]).upper()
|
||||
iface = this_agent["network_interface"]
|
||||
prio = int(this_agent["priority"])
|
||||
track = bool(vip.get("track_haproxy", True))
|
||||
use_unicast = bool(vip.get("use_unicast", True))
|
||||
vrid = int(vip["virtual_router_id"])
|
||||
advert = int(vip.get("advert_int", 1))
|
||||
name = str(vip.get("name", ""))
|
||||
inst = _vrrp_instance_name(vip["id"])
|
||||
|
||||
lines: List[str] = []
|
||||
lines.append(f"{OWNERSHIP_MARKER} — DO NOT EDIT")
|
||||
lines.append(f'# VIP "{name}" (id={vip["id"]}) — role {role}')
|
||||
lines.append("global_defs {")
|
||||
lines.append(" enable_script_security")
|
||||
lines.append(" script_user root")
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
|
||||
if track:
|
||||
weight = _failover_weight(members)
|
||||
lines.append("vrrp_script chk_haproxy {")
|
||||
lines.append(f' script "{CHECK_SCRIPT_PATH}"')
|
||||
lines.append(" interval 2")
|
||||
lines.append(" fall 2")
|
||||
lines.append(" rise 2")
|
||||
lines.append(f" weight {weight}")
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
|
||||
lines.append(f"vrrp_instance {inst} {{")
|
||||
lines.append(f" state {role}")
|
||||
lines.append(f" interface {iface}")
|
||||
lines.append(f" virtual_router_id {vrid}")
|
||||
lines.append(f" priority {prio}")
|
||||
lines.append(f" advert_int {advert}")
|
||||
if auth_pass_plain:
|
||||
lines.append(" authentication {")
|
||||
lines.append(" auth_type PASS")
|
||||
lines.append(f" auth_pass {auth_pass_plain}")
|
||||
lines.append(" }")
|
||||
# Unicast only makes sense with at least one peer. For a single-node VIP (no peers)
|
||||
# we deliberately omit the unicast block: keepalived treats a bare `unicast_src_ip`
|
||||
# with no `unicast_peer` as deprecated, warns, and silently falls back to multicast —
|
||||
# and `keepalived -t` flags it. Omitting it yields a clean multicast config that holds
|
||||
# the VIP with no peer to talk to. Multi-node behaviour (peers present) is unchanged.
|
||||
if use_unicast and peer_ips:
|
||||
src = this_agent.get("ip_address")
|
||||
if src:
|
||||
lines.append(f" unicast_src_ip {src}")
|
||||
lines.append(" unicast_peer {")
|
||||
for p in peer_ips:
|
||||
lines.append(f" {p}")
|
||||
lines.append(" }")
|
||||
lines.append(" virtual_ipaddress {")
|
||||
lines.append(f' {vip["virtual_ip"]}/{int(vip.get("prefix_length", 24))} dev {iface}')
|
||||
lines.append(" }")
|
||||
if track:
|
||||
lines.append(" track_script {")
|
||||
lines.append(" chk_haproxy")
|
||||
lines.append(" }")
|
||||
lines.append("}")
|
||||
return "\n".join(lines) + "\n"
|
||||
+139
-13
@@ -40,10 +40,12 @@ flow.
|
||||
(callers translate to wizard step-jumpback toasts).
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Optional
|
||||
from typing import Any, List, Optional
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
@@ -106,23 +108,22 @@ def _recompute_status_from_expiry(
|
||||
return cert_info_status or "valid", cert_info_days or 0
|
||||
|
||||
|
||||
async def create_cert_row(
|
||||
conn,
|
||||
payload: Any,
|
||||
cluster_id: int,
|
||||
) -> int:
|
||||
"""Insert a row into ssl_certificates (always cluster_id=NULL) + junction
|
||||
binding to the given cluster_id. Returns new ssl_certificate_id.
|
||||
def _prepare_cert_fields(payload: Any) -> dict:
|
||||
"""Parse + validate the PEM material on `payload` and derive every
|
||||
ssl_certificates column value from it (v1.9.0 extraction — shared by
|
||||
`create_cert_row` and the CSR import flow in services/csr_service.py,
|
||||
byte-identical to the former inline body of `create_cert_row`).
|
||||
|
||||
payload is expected to expose:
|
||||
name, certificate_content, private_key_content, chain_content,
|
||||
usage_type (optional, default 'frontend').
|
||||
|
||||
All cert metadata (primary_domain, all_domains, expiry_date,
|
||||
issuer, fingerprint, status, days_until_expiry) is now parsed
|
||||
FROM the PEM content via `parse_ssl_certificate` — operator-
|
||||
supplied values on the payload are accepted as a graceful
|
||||
fallback only when parsing fails (which itself raises 400).
|
||||
Raises HTTPException(400) on any parse/validation failure (invalid PEM,
|
||||
bad private key, cert/key mismatch, bad chain, already-expired cert).
|
||||
|
||||
Returns a dict with keys: cert_content, private_key_content,
|
||||
chain_content, cert_info, primary_domain, all_domains, expiry_date,
|
||||
issuer, fingerprint, status, days_until_expiry, usage_type.
|
||||
"""
|
||||
cert_content = getattr(payload, "certificate_content", None) or ""
|
||||
if not cert_content.strip():
|
||||
@@ -213,6 +214,53 @@ async def create_cert_row(
|
||||
)
|
||||
usage_type = getattr(payload, "usage_type", "frontend") or "frontend"
|
||||
|
||||
return {
|
||||
"cert_content": cert_content,
|
||||
"private_key_content": private_key_content,
|
||||
"chain_content": chain_content,
|
||||
"cert_info": cert_info,
|
||||
"primary_domain": primary_domain,
|
||||
"all_domains": all_domains,
|
||||
"expiry_date": expiry_date,
|
||||
"issuer": issuer,
|
||||
"fingerprint": fingerprint,
|
||||
"status": status,
|
||||
"days_until_expiry": days_until_expiry,
|
||||
"usage_type": usage_type,
|
||||
}
|
||||
|
||||
|
||||
async def create_cert_row(
|
||||
conn,
|
||||
payload: Any,
|
||||
cluster_id: int,
|
||||
) -> int:
|
||||
"""Insert a row into ssl_certificates (always cluster_id=NULL) + junction
|
||||
binding to the given cluster_id. Returns new ssl_certificate_id.
|
||||
|
||||
payload is expected to expose:
|
||||
name, certificate_content, private_key_content, chain_content,
|
||||
usage_type (optional, default 'frontend').
|
||||
|
||||
All cert metadata (primary_domain, all_domains, expiry_date,
|
||||
issuer, fingerprint, status, days_until_expiry) is now parsed
|
||||
FROM the PEM content via `parse_ssl_certificate` — operator-
|
||||
supplied values on the payload are accepted as a graceful
|
||||
fallback only when parsing fails (which itself raises 400).
|
||||
"""
|
||||
fields = _prepare_cert_fields(payload)
|
||||
cert_content = fields["cert_content"]
|
||||
private_key_content = fields["private_key_content"]
|
||||
chain_content = fields["chain_content"]
|
||||
expiry_date = fields["expiry_date"]
|
||||
primary_domain = fields["primary_domain"]
|
||||
all_domains = fields["all_domains"]
|
||||
issuer = fields["issuer"]
|
||||
fingerprint = fields["fingerprint"]
|
||||
status = fields["status"]
|
||||
days_until_expiry = fields["days_until_expiry"]
|
||||
usage_type = fields["usage_type"]
|
||||
|
||||
existing = await conn.fetchrow(
|
||||
"""
|
||||
SELECT s.id, s.is_active
|
||||
@@ -408,3 +456,81 @@ async def validate_server_ca_bundle_eligibility(
|
||||
cluster_id,
|
||||
)
|
||||
return row is not None
|
||||
|
||||
|
||||
async def stage_ssl_config_versions(
|
||||
conn,
|
||||
cert_id: int,
|
||||
cluster_ids: List[int],
|
||||
action: str = "create",
|
||||
created_by: Optional[int] = None,
|
||||
) -> List[dict]:
|
||||
"""Stage one PENDING config version per affected cluster after an SSL
|
||||
certificate mutation (v1.9.0 — distilled from the routers/ssl.py POST
|
||||
/certificates staging loop; used by the CSR import flow).
|
||||
|
||||
Uses the EXACT `ssl-{cert_id}-{action}-{timestamp}` version-name scheme of
|
||||
the manual SSL flow so Apply Management, the `has_pending_config`
|
||||
LIKE-filter ('ssl-' || id || '-%'), and the agent delivery predicates
|
||||
treat CSR-imported certificates identically to manually uploaded ones.
|
||||
Agents are NOT notified here — the operator applies manually.
|
||||
|
||||
Per-cluster failures are caught and reported in the returned
|
||||
sync_results list (the DB save has already succeeded — same semantics as
|
||||
the manual flow, where a config-generation failure never rolls back the
|
||||
certificate row).
|
||||
"""
|
||||
# Local import: keeps services/haproxy_config free to import ssl helpers
|
||||
# without a module-level cycle.
|
||||
from services.haproxy_config import generate_haproxy_config_for_cluster
|
||||
|
||||
sync_results: List[dict] = []
|
||||
for cluster_id in cluster_ids:
|
||||
try:
|
||||
config_content = await generate_haproxy_config_for_cluster(cluster_id)
|
||||
config_hash = hashlib.sha256(config_content.encode()).hexdigest()
|
||||
version_name = f"ssl-{cert_id}-{action}-{int(time.time())}"
|
||||
|
||||
version_created_by = created_by
|
||||
if version_created_by is None:
|
||||
version_created_by = await conn.fetchval(
|
||||
"SELECT id FROM users WHERE username = 'admin' LIMIT 1"
|
||||
) or 1
|
||||
|
||||
await conn.fetchval(
|
||||
"""
|
||||
INSERT INTO config_versions
|
||||
(cluster_id, version_name, config_content, checksum, created_by, is_active, status)
|
||||
VALUES ($1, $2, $3, $4, $5, FALSE, 'PENDING')
|
||||
RETURNING id
|
||||
""",
|
||||
cluster_id,
|
||||
version_name,
|
||||
config_content,
|
||||
config_hash,
|
||||
version_created_by,
|
||||
)
|
||||
logger.info(
|
||||
f"APPLY WORKFLOW: Created PENDING config version {version_name} "
|
||||
f"for cluster {cluster_id} (ssl_service.stage_ssl_config_versions)"
|
||||
)
|
||||
sync_results.append({
|
||||
'node': 'pending',
|
||||
'success': True,
|
||||
'cluster_id': cluster_id,
|
||||
'version': version_name,
|
||||
'status': 'PENDING',
|
||||
'message': 'SSL certificate staged. Click Apply to activate.',
|
||||
})
|
||||
except Exception as e:
|
||||
logger.error(
|
||||
f"Cluster config staging failed for SSL certificate {cert_id} "
|
||||
f"on cluster {cluster_id}: {e}"
|
||||
)
|
||||
sync_results.append({
|
||||
'node': 'cluster',
|
||||
'success': False,
|
||||
'cluster_id': cluster_id,
|
||||
'error': str(e),
|
||||
})
|
||||
return sync_results
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
"""Issue #38 follow-up — ACL `-f <file>` pattern-file support (v1.8.9).
|
||||
|
||||
The Bulgu #12 hard rejects were removed: pattern files are
|
||||
operator-managed host files (same policy as the SPOE
|
||||
`filter ... config <path>` reference preserved since v1.8.8), bulk
|
||||
import always accepted `-f`, and the agent runs `haproxy -c` before
|
||||
every reload so a missing file fails safely. These tests pin:
|
||||
|
||||
1. ACCEPT — the manual FrontendConfig model and the wizard models
|
||||
accept `-f` in every rule field (string + dict shapes).
|
||||
2. GUARDS KEPT — `$(`/backtick shell-substitution rejects and the
|
||||
`X !X` contradiction machinery are unchanged.
|
||||
3. WARNINGS — `_pattern_file_warnings` emits exactly one advisory
|
||||
listing the referenced files, and NOTHING for `-f`-free rules
|
||||
(zero-noise: existing users see no new output).
|
||||
4. ADVISORY — the bulk-import preview advisory block scans
|
||||
acl/use_backend rules (and only those fields).
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from models.frontend import FrontendConfig # noqa: E402
|
||||
from routers.frontend import _pattern_file_warnings # noqa: E402
|
||||
|
||||
|
||||
ACL_F = "blacklisted src -f /etc/haproxy/blacklist.lst"
|
||||
UB_F = "be-secure if { src -f /etc/haproxy/allowlist.lst }"
|
||||
REDIR_F = "location /blocked if { src -f /etc/haproxy/blacklist.lst }"
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 1. ACCEPT — manual FrontendConfig model
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_frontend_config_accepts_acl_file_flag():
|
||||
fe = FrontendConfig(name="fe1", bind_port=80, mode="http", acl_rules=[ACL_F])
|
||||
assert fe.acl_rules == [ACL_F]
|
||||
|
||||
|
||||
def test_frontend_config_accepts_use_backend_file_flag():
|
||||
fe = FrontendConfig(
|
||||
name="fe1", bind_port=80, mode="http", use_backend_rules=[UB_F])
|
||||
assert fe.use_backend_rules == [UB_F]
|
||||
|
||||
|
||||
def test_frontend_config_accepts_redirect_string_file_flag():
|
||||
fe = FrontendConfig(
|
||||
name="fe1", bind_port=80, mode="http", redirect_rules=[REDIR_F])
|
||||
assert fe.redirect_rules == [REDIR_F]
|
||||
|
||||
|
||||
def test_frontend_config_accepts_redirect_dict_file_flag():
|
||||
rule = {"type": "scheme", "scheme": "https",
|
||||
"condition": "if { src -f /etc/haproxy/blacklist.lst }"}
|
||||
fe = FrontendConfig(
|
||||
name="fe1", bind_port=80, mode="http", redirect_rules=[rule])
|
||||
assert fe.redirect_rules == [rule]
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 2. GUARDS KEPT — dangerous-content rejects unchanged
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad_rule", [
|
||||
"acl1 path $(rm -rf /)",
|
||||
"acl1 path `id`",
|
||||
])
|
||||
def test_acl_shell_substitution_still_rejected(bad_rule):
|
||||
from pydantic import ValidationError
|
||||
with pytest.raises(ValidationError):
|
||||
FrontendConfig(name="fe1", bind_port=80, mode="http", acl_rules=[bad_rule])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad_rule", [
|
||||
"be1 if $(whoami)",
|
||||
"be1 if `id`",
|
||||
])
|
||||
def test_use_backend_shell_substitution_still_rejected(bad_rule):
|
||||
from pydantic import ValidationError
|
||||
with pytest.raises(ValidationError):
|
||||
FrontendConfig(
|
||||
name="fe1", bind_port=80, mode="http", use_backend_rules=[bad_rule])
|
||||
|
||||
|
||||
def test_contradiction_detection_still_works_on_file_flag_rules():
|
||||
"""Interaction guard: a `-f` rule with an `X !X` contradiction is
|
||||
still caught by the handler-level contradiction machinery — the
|
||||
`-f` relaxation must not weaken that gate."""
|
||||
from models.frontend import _frontend_has_acl_contradiction
|
||||
assert _frontend_has_acl_contradiction(
|
||||
"be1 if blacklisted !blacklisted") is True
|
||||
# And a normal -f rule is NOT a contradiction.
|
||||
assert _frontend_has_acl_contradiction(UB_F) is False
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 3. WARNINGS — _pattern_file_warnings (zero-noise contract)
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_pattern_file_warnings_lists_unique_paths():
|
||||
warnings = _pattern_file_warnings(
|
||||
acl_rules=[ACL_F, "other src -f /etc/haproxy/blacklist.lst"],
|
||||
use_backend_rules=[UB_F],
|
||||
redirect_rules=[{"condition": "if { src -f /etc/haproxy/geo.lst }"}],
|
||||
)
|
||||
assert len(warnings) == 1
|
||||
w = warnings[0]
|
||||
assert "/etc/haproxy/blacklist.lst" in w
|
||||
assert "/etc/haproxy/allowlist.lst" in w
|
||||
assert "/etc/haproxy/geo.lst" in w
|
||||
# Duplicate path listed once.
|
||||
assert w.count("/etc/haproxy/blacklist.lst") == 1
|
||||
# Non-blocking framing: mentions fail-safe haproxy -c.
|
||||
assert "haproxy -c" in w
|
||||
|
||||
|
||||
def test_pattern_file_warnings_empty_without_file_flag():
|
||||
"""Zero-noise: operators who don't use `-f` must see NO warning."""
|
||||
assert _pattern_file_warnings(
|
||||
acl_rules=["is_api path_beg /api", "is_admin src 10.0.0.0/24"],
|
||||
use_backend_rules=["be-api if is_api"],
|
||||
redirect_rules=[{"type": "scheme", "scheme": "https",
|
||||
"condition": "if !{ ssl_fc }"}],
|
||||
) == []
|
||||
assert _pattern_file_warnings() == []
|
||||
|
||||
|
||||
def test_pattern_file_warnings_ignores_dash_f_substrings():
|
||||
"""`-file`/`-foo` substrings must not trigger the advisory."""
|
||||
assert _pattern_file_warnings(
|
||||
acl_rules=["is_self path_beg /self-config-file",
|
||||
"is_foo path_beg /foo -m beg"],
|
||||
) == []
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 4. Wizard models accept `-f` (string + dict) — parity
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_wizard_models_accept_file_flag():
|
||||
from models.site_wizard import FrontendStep
|
||||
|
||||
fe = FrontendStep(
|
||||
name="fe1", mode="http", bind_address="*", bind_port=80,
|
||||
acl_rules=[ACL_F],
|
||||
use_backend_rules=["be-x if blacklisted"],
|
||||
redirect_rules=[{"type": "scheme", "target": "https",
|
||||
"condition": "if { src -f /etc/haproxy/x.lst }"}],
|
||||
)
|
||||
assert fe.acl_rules == [ACL_F]
|
||||
assert fe.redirect_rules[0]["condition"] == "if { src -f /etc/haproxy/x.lst }"
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# 5. Bulk-import preview advisory — source-level pin
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_parse_bulk_advisory_scans_only_structured_rule_fields():
|
||||
"""The preview advisory scans acl_rules/use_backend_rules but NOT
|
||||
request_headers/tcp_request_rules (always-free-form fields —
|
||||
warning there would add new noise for existing users)."""
|
||||
src = Path(__file__).resolve().parents[1] / "routers" / "config.py"
|
||||
text = src.read_text()
|
||||
block_start = text.index("pattern-file advisory")
|
||||
block = text[block_start:block_start + 1200]
|
||||
assert 'acl_rules' in block
|
||||
assert 'use_backend_rules' in block
|
||||
assert 'request_headers' not in block.split("_pattern_paths")[1], (
|
||||
"advisory must not scan request_headers")
|
||||
|
||||
|
||||
def test_no_dash_f_reject_left_in_models():
|
||||
"""No model file may still hard-reject the `-f` flag."""
|
||||
for rel in ("models/frontend.py", "models/site_wizard.py"):
|
||||
text = (Path(__file__).resolve().parents[1] / rel).read_text()
|
||||
for m in re.finditer(r"-f\(\\s\|\$\)", text):
|
||||
ctx = text[max(0, m.start() - 400):m.start() + 400]
|
||||
assert "raise ValueError" not in ctx, (
|
||||
f"{rel}: a `-f` reject regex still sits next to a raise")
|
||||
@@ -81,6 +81,7 @@ def test_legacy_plain_string_with_brace_but_invalid_json_falls_back():
|
||||
("urn:ietf:params:acme:error:rejectedIdentifier", "blacklisted", "rejected"),
|
||||
("urn:ietf:params:acme:error:serverInternal", "internal err", "ACME server"),
|
||||
("urn:ietf:params:acme:error:userActionRequired", "agree to ToS", "User action"),
|
||||
("urn:ietf:params:acme:error:externalAccountRequired", "EAB required", "External Account Binding"),
|
||||
])
|
||||
def test_known_problem_types_are_humanized(problem_type, detail_text, expected_title_contains):
|
||||
payload = json.dumps({"type": problem_type, "detail": detail_text, "status": 400})
|
||||
|
||||
@@ -8,7 +8,31 @@ from pydantic import ValidationError
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from routers.letsencrypt import CertificateRequest
|
||||
from routers.letsencrypt import CertificateRequest, AccountCreate
|
||||
|
||||
|
||||
class TestAccountCreateEAB:
|
||||
"""Issue #35 follow-up: EAB HMAC key must be valid base64url; empty/None passes through
|
||||
(falls back to global Settings) so non-EAB accounts (HTTP-01 / LE / Cloudflare) are unaffected."""
|
||||
|
||||
def test_no_eab_is_allowed(self):
|
||||
acc = AccountCreate(email="a@b.com")
|
||||
assert acc.eab_hmac_key is None and acc.eab_kid is None
|
||||
|
||||
def test_valid_base64url_hmac_accepted(self):
|
||||
# urlsafe base64, unpadded and padded — both accepted.
|
||||
AccountCreate(email="a@b.com", eab_kid="kid-1", eab_hmac_key="YWJjZGVmZ2g")
|
||||
AccountCreate(email="a@b.com", eab_kid="kid-1", eab_hmac_key="YWJjZA==")
|
||||
|
||||
def test_invalid_base64_hmac_rejected(self):
|
||||
# 5 base64 chars (count ≡ 1 mod 4) is undecodable — the exact shape that would otherwise
|
||||
# make register_account's _b64url_decode raise a cryptic binascii error.
|
||||
with pytest.raises(ValidationError):
|
||||
AccountCreate(email="a@b.com", eab_kid="kid-1", eab_hmac_key="AAAAA")
|
||||
|
||||
def test_oversized_hmac_rejected(self):
|
||||
with pytest.raises(ValidationError):
|
||||
AccountCreate(email="a@b.com", eab_kid="kid-1", eab_hmac_key="A" * 600)
|
||||
|
||||
|
||||
class TestCertificateRequestDomains:
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
"""Issue #31 — agent heartbeat JSON sanitizer.
|
||||
|
||||
A self-hosted agent builds its heartbeat JSON as text in bash. When a collected value is empty,
|
||||
the payload can contain a structurally-invalid comma that broke the heartbeat with
|
||||
`HTTP 400 Invalid JSON: Expecting property name enclosed in double quotes`. The backend now
|
||||
repairs that pattern in `_sanitize_agent_json` so an already-deployed agent recovers without a
|
||||
re-install. These tests pin that behaviour and prove the repair never corrupts a healthy payload.
|
||||
"""
|
||||
import json
|
||||
|
||||
from routers.agent import _sanitize_agent_json
|
||||
|
||||
|
||||
def _assert_parses(raw: str) -> dict:
|
||||
out, _ = _sanitize_agent_json(raw)
|
||||
return json.loads(out) # raises if the repair did not produce valid JSON
|
||||
|
||||
|
||||
def test_reporter_empty_system_info_bare_comma():
|
||||
# The exact shape the reporter hit: an empty $system_info collapses ' $system_info,' to a
|
||||
# bare comma between two members -> '"version": "x",\n ,\n "haproxy_status": ...'.
|
||||
raw = (
|
||||
'{\n'
|
||||
' "name": "test",\n'
|
||||
' "hostname": "h",\n'
|
||||
' "status": "online",\n'
|
||||
' "version": "2.0.0",\n'
|
||||
' ,\n'
|
||||
' "haproxy_status": "running",\n'
|
||||
' "cluster_id": 1\n'
|
||||
'}'
|
||||
)
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["name"] == "test"
|
||||
assert parsed["status"] == "online"
|
||||
assert parsed["haproxy_status"] == "running"
|
||||
|
||||
|
||||
def test_empty_numeric_subfield_before_comma():
|
||||
# An empty unquoted numeric ("memory_total": ,) — covered by the pre-existing Fix 1.
|
||||
raw = '{ "name": "t", "cpu_count": , "memory_total": , "status": "online" }'
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["cpu_count"] is None and parsed["memory_total"] is None
|
||||
assert parsed["status"] == "online"
|
||||
|
||||
|
||||
def test_empty_value_before_closing_brace():
|
||||
raw = '{ "name": "t", "status": "online", "applied_config_version": }'
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["applied_config_version"] is None
|
||||
|
||||
|
||||
def test_leading_comma_first_member():
|
||||
# Empty $system_info as the FIRST member -> '{ , "name": ... }'.
|
||||
raw = '{\n ,\n "name": "t",\n "status": "online"\n}'
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["name"] == "t"
|
||||
|
||||
|
||||
def test_comma_run_two_empty_fields():
|
||||
# Two empties in a row (odd-length comma run) must still collapse to valid JSON.
|
||||
raw = '{ "a": 1,\n ,\n ,\n "b": 2 }'
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["a"] == 1 and parsed["b"] == 2
|
||||
|
||||
|
||||
def test_trailing_comma_regression():
|
||||
# Pre-existing Fix 3 must still hold after the new fixes were added.
|
||||
raw = '{ "name": "t", "status": "online", }'
|
||||
parsed = _assert_parses(raw)
|
||||
assert parsed["name"] == "t"
|
||||
|
||||
|
||||
def test_healthy_payload_is_untouched():
|
||||
# A well-formed agent payload must pass through unchanged (sanitized=False) and its values —
|
||||
# including the base64 stats CSV and the nested server_statuses — must be byte-identical.
|
||||
payload = {
|
||||
"name": "agent-1",
|
||||
"status": "online",
|
||||
"cluster_id": 1,
|
||||
"server_statuses": {"be_app": {"s1": "UP", "s2": "DOWN"}},
|
||||
"network_interfaces": ["eth0", "eth1"],
|
||||
"haproxy_stats_csv": "IyBwdmJjLGJhY2tlbmQsZnJvbnRlbmQs", # base64: contains commas only inside a quoted string is impossible (base64 has none)
|
||||
"applied_config_version": "cluster-1-v42",
|
||||
}
|
||||
raw = json.dumps(payload)
|
||||
out, changed = _sanitize_agent_json(raw)
|
||||
assert changed is False
|
||||
assert out == raw # byte-identical
|
||||
assert json.loads(out) == payload
|
||||
|
||||
|
||||
def test_idempotent_on_already_clean_minimal():
|
||||
raw = '{"name": "t", "status": "online"}'
|
||||
out, changed = _sanitize_agent_json(raw)
|
||||
assert changed is False
|
||||
assert out == raw
|
||||
@@ -0,0 +1,61 @@
|
||||
"""Issue #31 — agent-script hardening guard (static).
|
||||
|
||||
The agent install scripts hand-build the heartbeat JSON, so if `collect_system_info` ever yields
|
||||
nothing the `$system_info,` line collapses to a bare comma and the whole heartbeat is invalid JSON
|
||||
(HTTP 400). The fix adds a guard at every fragment-form call site that substitutes a single valid
|
||||
key when system_info is empty. This static check enforces that the guard is present AND kept in
|
||||
sync across BOTH platform scripts — the project requires the two agent-script copies to stay in
|
||||
lockstep. (Empty numeric subfields like "memory_total": , are a separate, milder case already
|
||||
repaired by the backend sanitizer, so they are intentionally NOT guarded in the script — guarding
|
||||
them with a strict integer test would wrongly reject the scientific-notation that mawk emits for
|
||||
multi-GB sizes on Debian/Ubuntu.)
|
||||
"""
|
||||
import os
|
||||
|
||||
_SCRIPT_DIR = os.path.join(
|
||||
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), # backend/
|
||||
"utils", "agent_scripts",
|
||||
)
|
||||
|
||||
|
||||
def _read(name: str) -> str:
|
||||
with open(os.path.join(_SCRIPT_DIR, name), "r") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
LINUX = _read("linux_install.sh")
|
||||
MACOS = _read("macos_install.sh")
|
||||
|
||||
# The empty-system_info guard — present at BOTH fragment call sites (register_agent + send_heartbeat).
|
||||
_B2_GUARD = '[[ "$system_info" != *\'"\'* ]] && system_info=\'"operating_system": "unknown"\''
|
||||
|
||||
|
||||
def test_b2_guard_present_and_in_sync():
|
||||
# Two fragment-form call sites per script (register_agent + send_heartbeat), identical wording.
|
||||
assert LINUX.count(_B2_GUARD) == 2, "linux_install.sh missing/duplicated empty-system_info guard"
|
||||
assert MACOS.count(_B2_GUARD) == 2, "macos_install.sh missing/duplicated empty-system_info guard"
|
||||
|
||||
|
||||
def test_b2_guard_precedes_every_fragment_system_info_use():
|
||||
# Every ' $system_info,' fragment line (the one that breaks on an empty value) must be in a
|
||||
# function whose system_info was guarded. We assert the count of guards matches the count of
|
||||
# fragment-form interpolations' call sites: each script has exactly one register + one
|
||||
# send_heartbeat fragment builder feeding those lines, both guarded above.
|
||||
for name, script in (("linux", LINUX), ("macos", MACOS)):
|
||||
assert script.count(" $system_info,") >= 1, f"{name}: fragment heartbeat form unexpectedly gone"
|
||||
assert script.count(_B2_GUARD) == 2, f"{name}: each fragment call site must carry the guard"
|
||||
|
||||
|
||||
def test_cleanup_does_not_self_kill_via_bare_haproxy_agent_pattern():
|
||||
# Issue #31 (v1.8.4): the pre-installation cleanup kills processes by pgrep -f "$pattern". A bare
|
||||
# "haproxy-agent" pattern also matches the installer's OWN path (install-haproxy-agent-*.sh) and a
|
||||
# sudo/PAM ancestor, so the installer killed itself. The kill loop must target ONLY the installed
|
||||
# agent (binary path + service/label), never the bare string.
|
||||
for name, script in (("linux", LINUX), ("macos", MACOS)):
|
||||
assert 'for pattern in "haproxy-agent"' not in script, (
|
||||
f"{name}: pre-install cleanup uses the bare 'haproxy-agent' kill pattern -> self-kill (issue #31)"
|
||||
)
|
||||
# The narrowed, installer-safe pattern must be present (binary path via $INSTALL_DIR).
|
||||
assert 'for pattern in "$INSTALL_DIR/haproxy-agent"' in script, (
|
||||
f"{name}: cleanup must match the installed binary path, not a bare substring"
|
||||
)
|
||||
@@ -0,0 +1,469 @@
|
||||
"""
|
||||
v1.9.0 CSR creation — unit tests for the signed-certificate import flow and
|
||||
config-version staging (pattern: test_ssl_service_extraction.py, AsyncMock conn).
|
||||
|
||||
Pins the security-relevant invariants:
|
||||
- key match is a HARD gate: match=False → 400 before any INSERT, and
|
||||
match=None (unverifiable) → 500, never a lenient pass (we generated the
|
||||
key ourselves — deliberate divergence from create_cert_row's fallback).
|
||||
- the new cert row is cluster_id=NULL / last_config_status='PENDING' /
|
||||
source='csr' (PENDING keeps it invisible to agents until Apply).
|
||||
- completing the CSR NULLs the private key copy.
|
||||
- staging reuses the exact `ssl-{id}-create-{ts}` version-name scheme.
|
||||
"""
|
||||
import json
|
||||
from contextlib import contextmanager
|
||||
from datetime import datetime, timezone
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from fastapi import HTTPException
|
||||
|
||||
from models.csr import SSLCSRImport
|
||||
from services.csr_service import (
|
||||
assert_csr_name_available,
|
||||
import_signed_certificate,
|
||||
insert_csr_row,
|
||||
)
|
||||
from services.ssl_service import stage_ssl_config_versions
|
||||
|
||||
|
||||
_VALID_PARSE = {
|
||||
"primary_domain": "www.example.com",
|
||||
"all_domains": ["www.example.com"],
|
||||
"expiry_date": datetime(2099, 1, 1, tzinfo=timezone.utc),
|
||||
"issuer": "CN=Test CA",
|
||||
"fingerprint": "AA:BB:CC",
|
||||
"status": "valid",
|
||||
"days_until_expiry": 365,
|
||||
}
|
||||
|
||||
_FAKE_CERT = "-----BEGIN CERTIFICATE-----\nX\n-----END CERTIFICATE-----"
|
||||
_FAKE_KEY = "-----BEGIN PRIVATE KEY-----\nY\n-----END PRIVATE KEY-----"
|
||||
|
||||
|
||||
def _csr_row(**overrides):
|
||||
row = {
|
||||
"id": 5,
|
||||
"name": "csr-www",
|
||||
"common_name": "www.example.com",
|
||||
"subject": "{}",
|
||||
"sans": json.dumps(["www.example.com"]),
|
||||
"key_algorithm": "rsa-2048",
|
||||
"csr_pem": "-----BEGIN CERTIFICATE REQUEST-----\nZ\n-----END CERTIFICATE REQUEST-----",
|
||||
"private_key_pem": _FAKE_KEY,
|
||||
"status": "pending",
|
||||
"ssl_certificate_id": None,
|
||||
}
|
||||
row.update(overrides)
|
||||
return row
|
||||
|
||||
|
||||
def _mk_conn():
|
||||
conn = AsyncMock()
|
||||
# asyncpg's conn.transaction() is a SYNC call returning an async CM.
|
||||
conn.transaction = MagicMock()
|
||||
return conn
|
||||
|
||||
|
||||
def _import_payload(**overrides):
|
||||
base = dict(
|
||||
certificate_content=_FAKE_CERT,
|
||||
chain_content=None,
|
||||
usage_type="frontend",
|
||||
is_global=False,
|
||||
cluster_ids=[1, 2],
|
||||
name=None,
|
||||
)
|
||||
base.update(overrides)
|
||||
return SSLCSRImport(**base)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _patched(match=None, parse=None):
|
||||
"""Patch every parser touchpoint of the import path: the function-local
|
||||
imports in csr_service (utils.ssl_parser.*) and the module-level imports
|
||||
in ssl_service._prepare_cert_fields (services.ssl_service.*)."""
|
||||
match_result = match if match is not None else {"match": True}
|
||||
parse_result = dict(parse or _VALID_PARSE)
|
||||
with patch("utils.ssl_parser.verify_certificate_key_match", return_value=match_result), \
|
||||
patch("utils.ssl_parser.parse_ssl_certificate", return_value=dict(parse_result)), \
|
||||
patch("services.ssl_service.parse_ssl_certificate", return_value=dict(parse_result)), \
|
||||
patch("services.ssl_service.validate_private_key", return_value=True), \
|
||||
patch("services.ssl_service.validate_certificate_chain", return_value=True):
|
||||
yield
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# import_signed_certificate
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_happy_path_inserts_pending_csr_sourced_cert():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(), None] # FOR UPDATE row, no name clash
|
||||
conn.fetchval.return_value = 42 # INSERT ... RETURNING id
|
||||
|
||||
with _patched():
|
||||
result = await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert result["certificate_id"] == 42
|
||||
assert result["reactivated"] is False
|
||||
|
||||
# Concurrency invariants: everything runs inside a transaction and the
|
||||
# CSR row is locked FOR UPDATE (serialises double-import and delete-races).
|
||||
assert conn.transaction.call_count == 1
|
||||
lock_sql = conn.fetchrow.call_args_list[0].args[0]
|
||||
assert "FOR UPDATE" in lock_sql
|
||||
|
||||
insert_sql, *insert_args = conn.fetchval.call_args.args
|
||||
assert "INSERT INTO ssl_certificates" in insert_sql
|
||||
assert "NULL, 'PENDING'" in insert_sql, "cert must stay invisible to agents until Apply"
|
||||
assert "'csr'" in insert_sql, "source column must record the CSR origin"
|
||||
# The stored CSR key — not any request-supplied key — must be persisted.
|
||||
assert _FAKE_KEY in insert_args
|
||||
|
||||
# One junction row per requested cluster.
|
||||
junction_calls = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "ssl_certificate_clusters" in c.args[0] and "INSERT" in c.args[0]
|
||||
]
|
||||
assert len(junction_calls) == 2
|
||||
assert {c.args[2] for c in junction_calls} == {1, 2}
|
||||
|
||||
# CSR completion must destroy the key copy.
|
||||
completion_calls = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "UPDATE ssl_csrs" in c.args[0]
|
||||
]
|
||||
assert len(completion_calls) == 1
|
||||
assert "private_key_pem = NULL" in completion_calls[0].args[0]
|
||||
assert "status = 'completed'" in completion_calls[0].args[0]
|
||||
assert completion_calls[0].args[1] == 5 # csr_id
|
||||
assert completion_calls[0].args[2] == 42 # cert_id
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_global_creates_zero_junction_rows():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(), None]
|
||||
conn.fetchval.return_value = 42
|
||||
|
||||
with _patched():
|
||||
await import_signed_certificate(
|
||||
conn, 5, _import_payload(is_global=True, cluster_ids=None), user_id=7
|
||||
)
|
||||
|
||||
junction_calls = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "ssl_certificate_clusters" in c.args[0] and "INSERT" in c.args[0]
|
||||
]
|
||||
assert junction_calls == [], "global cert = zero junction rows (existing convention)"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_key_mismatch_rejected_400_before_any_write():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row()]
|
||||
|
||||
with _patched(match={"match": False, "reason": "public key mismatch"}):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "does not match" in exc_info.value.detail
|
||||
assert not conn.fetchval.await_count, "nothing must be inserted on mismatch"
|
||||
assert not conn.execute.await_count
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_unverifiable_key_match_is_hard_error_not_lenient():
|
||||
"""match=None means OUR stored key is unreadable — integrity failure,
|
||||
never the lenient pass create_cert_row historically allows."""
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row()]
|
||||
|
||||
with _patched(match={"match": None, "reason": "key could not be parsed"}):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 500
|
||||
assert not conn.fetchval.await_count
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_expired_certificate_rejected_400():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row()]
|
||||
|
||||
expired = dict(_VALID_PARSE)
|
||||
expired["status"] = "expired"
|
||||
expired["days_until_expiry"] = -10
|
||||
with _patched(parse=expired):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "expired" in exc_info.value.detail.lower()
|
||||
assert not conn.fetchval.await_count
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_malformed_certificate_rejected_400_not_500():
|
||||
"""A cert with PEM markers but unparseable content (truncated CA response)
|
||||
is OPERATOR INPUT — it must get the manual flow's 400, not the 500 that
|
||||
the strict key-match branch reserves for a corrupt STORED key."""
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row()]
|
||||
|
||||
with _patched(parse={"error": "Could not parse certificate"}):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "Invalid SSL certificate" in exc_info.value.detail
|
||||
assert not conn.fetchval.await_count
|
||||
assert not conn.execute.await_count
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_completed_csr_conflicts_409():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(status="completed", ssl_certificate_id=42)]
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 409
|
||||
assert "already completed" in exc_info.value.detail
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_missing_csr_404():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [None]
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 999, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_active_name_collision_rejected_with_hint():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(), {"id": 9, "is_active": True}]
|
||||
|
||||
with _patched():
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "already exists" in exc_info.value.detail
|
||||
assert "name" in exc_info.value.detail # points at the override escape hatch
|
||||
assert not conn.fetchval.await_count
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_name_override_is_used_for_the_cert_row():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(), None]
|
||||
conn.fetchval.return_value = 42
|
||||
|
||||
with _patched():
|
||||
result = await import_signed_certificate(
|
||||
conn, 5, _import_payload(name="renamed-cert"), user_id=7
|
||||
)
|
||||
|
||||
assert result["certificate_name"] == "renamed-cert"
|
||||
_, *insert_args = conn.fetchval.call_args.args
|
||||
assert "renamed-cert" in insert_args
|
||||
# And the collision check must have run against the override, not csr.name.
|
||||
name_lookup = conn.fetchrow.call_args_list[1]
|
||||
assert name_lookup.args[1] == "renamed-cert"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_reactivates_soft_deleted_name_and_warns():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [_csr_row(), {"id": 77, "is_active": False}]
|
||||
|
||||
with _patched():
|
||||
result = await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert result["certificate_id"] == 77
|
||||
assert result["reactivated"] is True
|
||||
assert any("reactivated" in w for w in result["warnings"])
|
||||
assert not conn.fetchval.await_count, "reactivation must UPDATE, not INSERT"
|
||||
update_calls = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "UPDATE ssl_certificates" in c.args[0]
|
||||
]
|
||||
assert len(update_calls) == 1
|
||||
update_sql = update_calls[0].args[0]
|
||||
assert "source = 'csr'" in update_sql
|
||||
# The reactivated row must come back to life invisible to agents until
|
||||
# Apply, with the row itself active again.
|
||||
assert "last_config_status = 'PENDING'" in update_sql
|
||||
assert "is_active = TRUE" in update_sql
|
||||
# Old cluster bindings must be wiped before re-binding to the new scope.
|
||||
junction_deletes = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "DELETE FROM ssl_certificate_clusters" in c.args[0]
|
||||
]
|
||||
assert len(junction_deletes) == 1
|
||||
assert junction_deletes[0].args[1] == 77
|
||||
# …and the importer's requested clusters re-bound via the junction.
|
||||
junction_inserts = [
|
||||
c for c in conn.execute.call_args_list
|
||||
if c.args and "INSERT INTO ssl_certificate_clusters" in c.args[0]
|
||||
]
|
||||
assert {c.args[2] for c in junction_inserts} == {1, 2}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_san_drift_warns_but_succeeds():
|
||||
conn = _mk_conn()
|
||||
conn.fetchrow.side_effect = [
|
||||
_csr_row(sans=json.dumps(["www.example.com", "api.example.com"])),
|
||||
None,
|
||||
]
|
||||
conn.fetchval.return_value = 42
|
||||
|
||||
drifted = dict(_VALID_PARSE)
|
||||
drifted["all_domains"] = ["www.example.com", "cdn.example.com"]
|
||||
with _patched(parse=drifted):
|
||||
result = await import_signed_certificate(conn, 5, _import_payload(), user_id=7)
|
||||
|
||||
assert result["certificate_id"] == 42
|
||||
assert any("added" in w and "cdn.example.com" in w for w in result["warnings"])
|
||||
assert any("dropped" in w and "api.example.com" in w for w in result["warnings"])
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# insert_csr_row / assert_csr_name_available
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_csr_name_taken_by_active_cert_rejected():
|
||||
conn = _mk_conn()
|
||||
conn.fetchval.side_effect = [11] # active cert with the name exists
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await assert_csr_name_available(conn, "taken")
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "certificate" in exc_info.value.detail.lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_csr_name_taken_by_pending_csr_rejected():
|
||||
conn = _mk_conn()
|
||||
conn.fetchval.side_effect = [None, 12] # no cert, but a pending CSR
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await assert_csr_name_available(conn, "taken")
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "pending CSR" in exc_info.value.detail
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_insert_csr_row_translates_unique_violation_to_400():
|
||||
"""The uq_ssl_csrs_name_pending partial index closes the create/create
|
||||
race — the loser must get a clean 400, not a 500."""
|
||||
import asyncpg as _asyncpg
|
||||
|
||||
conn = _mk_conn()
|
||||
# availability checks pass, INSERT hits the unique index
|
||||
conn.fetchval.side_effect = [
|
||||
None, None, _asyncpg.exceptions.UniqueViolationError("dup"),
|
||||
]
|
||||
payload = SimpleNamespace(
|
||||
name="raced", common_name="www.example.com", key_algorithm="rsa-2048"
|
||||
)
|
||||
bundle = {"subject": {}, "sans": ["www.example.com"], "csr_pem": "PEM", "private_key_pem": "KEY"}
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await insert_csr_row(conn, payload, bundle, user_id=1)
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "concurrent" in exc_info.value.detail
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# router-level guards
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cluster_id_int32_guard_rejects_out_of_range_with_404():
|
||||
"""Body-supplied cluster ids must never reach asyncpg out of int4 range
|
||||
(DataError → raw 500) — same Bulgu #96 hygiene as the csr_id path param."""
|
||||
from routers.csr import _assert_valid_cluster_id
|
||||
|
||||
_assert_valid_cluster_id(1)
|
||||
_assert_valid_cluster_id(2_147_483_647)
|
||||
for bad in (0, -1, 2_147_483_648, 99_999_999_999):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
_assert_valid_cluster_id(bad)
|
||||
assert exc_info.value.status_code == 404
|
||||
assert "Cluster not found" in exc_info.value.detail
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# stage_ssl_config_versions
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stage_creates_one_pending_version_per_cluster_with_ssl_naming():
|
||||
import re
|
||||
|
||||
conn = _mk_conn()
|
||||
conn.fetchval.return_value = 1001 # config_versions INSERT RETURNING id
|
||||
|
||||
with patch(
|
||||
"services.haproxy_config.generate_haproxy_config_for_cluster",
|
||||
new=AsyncMock(return_value="# cfg"),
|
||||
):
|
||||
results = await stage_ssl_config_versions(conn, 42, [1, 2], created_by=7)
|
||||
|
||||
assert len(results) == 2
|
||||
assert all(r["success"] for r in results)
|
||||
assert [r["cluster_id"] for r in results] == [1, 2]
|
||||
|
||||
insert_calls = [
|
||||
c for c in conn.fetchval.call_args_list
|
||||
if c.args and "INSERT INTO config_versions" in c.args[0]
|
||||
]
|
||||
assert len(insert_calls) == 2
|
||||
for call in insert_calls:
|
||||
sql = call.args[0]
|
||||
assert "FALSE, 'PENDING'" in sql, "staged versions must be inactive + PENDING"
|
||||
version_name = call.args[2]
|
||||
# EXACT manual-flow scheme: Apply Management + has_pending_config
|
||||
# LIKE-filters key off 'ssl-{id}-...'.
|
||||
assert re.match(r"^ssl-42-create-\d+$", version_name), version_name
|
||||
assert call.args[5] == 7 # created_by honours the importing user
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stage_reports_per_cluster_failure_without_raising():
|
||||
conn = _mk_conn()
|
||||
conn.fetchval.return_value = 1001
|
||||
|
||||
async def _gen(cluster_id):
|
||||
if cluster_id == 2:
|
||||
raise RuntimeError("config generation exploded")
|
||||
return "# cfg"
|
||||
|
||||
with patch(
|
||||
"services.haproxy_config.generate_haproxy_config_for_cluster",
|
||||
new=AsyncMock(side_effect=_gen),
|
||||
):
|
||||
results = await stage_ssl_config_versions(conn, 42, [1, 2], created_by=7)
|
||||
|
||||
assert len(results) == 2
|
||||
assert results[0]["success"] is True
|
||||
assert results[1]["success"] is False
|
||||
assert "exploded" in results[1]["error"]
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Issue #53 (v1.10.1) — at-rest encryption for the pending CSR private key.
|
||||
|
||||
Pure logic: no DB, no network. Covers the round-trip, the backward-compatible read of rows
|
||||
written before this release, the unrecoverable-key path after a key rotation, and a static
|
||||
assertion that the write path can no longer store a raw PEM.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-csr-encryption-unit-tests")
|
||||
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
from utils.csr_key_crypto import (
|
||||
decrypt_csr_private_key,
|
||||
encrypt_csr_private_key,
|
||||
is_encrypted,
|
||||
reset_fernet_for_tests,
|
||||
)
|
||||
|
||||
_SAMPLE_PEM = (
|
||||
"-----BEGIN PRIVATE KEY-----\n"
|
||||
"MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj\n"
|
||||
"-----END PRIVATE KEY-----\n"
|
||||
)
|
||||
|
||||
|
||||
def test_roundtrip_and_ciphertext_does_not_contain_the_key():
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# The stored form must not be the PEM, and must not leak any recognisable fragment of it.
|
||||
assert token != _SAMPLE_PEM
|
||||
assert "-----BEGIN" not in token
|
||||
assert "MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj" not in token
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_is_encrypted_discriminates_token_from_legacy_pem():
|
||||
reset_fernet_for_tests()
|
||||
assert is_encrypted(encrypt_csr_private_key(_SAMPLE_PEM)) is True
|
||||
assert is_encrypted(_SAMPLE_PEM) is False
|
||||
assert is_encrypted("") is False
|
||||
assert is_encrypted(None) is False
|
||||
|
||||
|
||||
def test_legacy_plaintext_row_is_read_unchanged():
|
||||
# Rows written before v1.10.1 hold a raw PEM. They must keep working with NO data migration,
|
||||
# otherwise upgrading would strand every CSR that is out for signature.
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(_SAMPLE_PEM) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_empty_or_missing_value_returns_none():
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(None) is None
|
||||
assert decrypt_csr_private_key("") is None
|
||||
|
||||
|
||||
def test_key_rotation_makes_the_stored_key_unrecoverable_rather_than_wrong():
|
||||
"""After a rotation the caller must get None, never a silently wrong key."""
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
|
||||
# Rotate: an explicit, different CSR_ENCRYPTION_KEY takes precedence over the derived one.
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = Fernet.generate_key().decode()
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) is None
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_explicit_env_key_is_used_and_survives_reset():
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
key = Fernet.generate_key().decode()
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = key
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# Decryptable with the same explicit key from a fresh instance...
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
# ...and independently verifiable with the raw Fernet key.
|
||||
assert Fernet(key.encode()).decrypt(token.encode()).decode() == _SAMPLE_PEM
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_derivation_uses_its_own_hkdf_info_string():
|
||||
"""Each secret class derives an independent key, so rotating one never affects another."""
|
||||
src = (Path(__file__).resolve().parent.parent / "utils" / "csr_key_crypto.py").read_text()
|
||||
assert b"csr-private-key-v1".decode() in src
|
||||
# Must NOT reuse another class's info string.
|
||||
for foreign in ("dns-provider-creds-v1", "vip-vrrp-secret-v1", "mfa-totp-secret-v1"):
|
||||
assert foreign not in src, f"CSR key derivation must not reuse the {foreign} info string"
|
||||
|
||||
|
||||
def test_write_path_stores_the_encrypted_form_not_the_pem():
|
||||
"""Static pin: insert_csr_row must encrypt before the INSERT.
|
||||
|
||||
A future refactor that passed bundle['private_key_pem'] straight through would silently
|
||||
reintroduce plaintext storage, and no unit test with a mocked connection would notice.
|
||||
"""
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
insert_fn = src[src.index("async def insert_csr_row("):]
|
||||
insert_fn = insert_fn[: insert_fn.index("\nasync def ")]
|
||||
assert "encrypt_csr_private_key(bundle['private_key_pem'])" in insert_fn
|
||||
# The raw PEM must not be a bind parameter of the INSERT itself.
|
||||
assert not re.search(r"^\s*bundle\['private_key_pem'\],\s*$", insert_fn, re.M)
|
||||
|
||||
|
||||
def test_import_path_decrypts_and_fails_closed_on_unrecoverable_key():
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
fn = src[src.index("async def import_signed_certificate("):]
|
||||
assert "decrypt_csr_private_key(row['private_key_pem'])" in fn
|
||||
# A None decrypt must raise rather than fall through to the key-match comparison.
|
||||
assert "cannot be decrypted" in fn
|
||||
@@ -0,0 +1,104 @@
|
||||
"""
|
||||
v1.9.0 CSR creation — static source assertions (pattern: test_vip_purge.py).
|
||||
|
||||
Guards the migration wiring that a unit test cannot exercise without a real
|
||||
database: the SCHEMA_VERSION bump (without it, deployed installs skip the
|
||||
whole migration run and the ssl_csrs table never appears), the migration
|
||||
registration, the security-relevant DDL, and the router registration.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
|
||||
_BACKEND_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _read(rel_path: str) -> str:
|
||||
with open(os.path.join(_BACKEND_DIR, rel_path), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_schema_version_bumped_to_10():
|
||||
src = _read(os.path.join("database", "migrations.py"))
|
||||
m = re.search(r"^SCHEMA_VERSION\s*=\s*(\d+)", src, re.MULTILINE)
|
||||
assert m, "SCHEMA_VERSION constant not found in migrations.py"
|
||||
assert int(m.group(1)) >= 10, (
|
||||
"SCHEMA_VERSION must be >= 10 for the v1.9.0 ssl_csrs table — "
|
||||
"without the bump, existing installs (version >= 9) skip the whole "
|
||||
"migration run and never gain the table."
|
||||
)
|
||||
|
||||
|
||||
def test_ssl_csrs_migration_defined_and_registered():
|
||||
src = _read(os.path.join("database", "migrations.py"))
|
||||
assert "async def ensure_ssl_csrs_table" in src
|
||||
|
||||
inner = src.split("async def _run_all_migrations_inner", 1)[1]
|
||||
inner = inner.split("\nasync def ", 1)[0] # body of the runner only
|
||||
assert "await ensure_ssl_csrs_table()" in inner, (
|
||||
"ensure_ssl_csrs_table must be invoked from _run_all_migrations_inner"
|
||||
)
|
||||
|
||||
|
||||
def test_ssl_csrs_ddl_essentials():
|
||||
src = _read(os.path.join("database", "migrations.py"))
|
||||
ddl_start = src.index("CREATE TABLE IF NOT EXISTS ssl_csrs")
|
||||
ddl = src[ddl_start:ddl_start + 2500]
|
||||
|
||||
assert "private_key_pem TEXT" in ddl
|
||||
assert "name VARCHAR(100) NOT NULL" in ddl, (
|
||||
"ssl_csrs.name must align with ssl_certificates.name VARCHAR(100)"
|
||||
)
|
||||
assert "ssl_certificate_id INTEGER REFERENCES ssl_certificates(id) ON DELETE SET NULL" in ddl, (
|
||||
"deleting the imported cert must not cascade into CSR history"
|
||||
)
|
||||
# Partial unique index: only PENDING CSRs reserve their target cert name.
|
||||
assert "uq_ssl_csrs_name_pending" in src
|
||||
assert re.search(
|
||||
r"uq_ssl_csrs_name_pending\s+ON\s+ssl_csrs\(name\)\s+WHERE\s+status\s*=\s*'pending'",
|
||||
src,
|
||||
), "name uniqueness must be scoped to pending CSRs (partial index)"
|
||||
|
||||
|
||||
def test_csr_router_registered_in_main():
|
||||
src = _read("main.py")
|
||||
assert "from routers.csr import router as csr_router" in src
|
||||
assert "app.include_router(csr_router)" in src
|
||||
|
||||
|
||||
def test_csr_endpoint_permission_mapping():
|
||||
"""Pin which ssl.<action> permission each endpoint enforces: a regression
|
||||
that dropped or weakened a _require() call would otherwise pass the
|
||||
auth-rejection tests (they only assert 401/403 for unauthenticated calls)."""
|
||||
src = _read(os.path.join("routers", "csr.py"))
|
||||
|
||||
def _handler_body(decorator):
|
||||
start = src.index(decorator)
|
||||
nxt = src.find("@router.", start + 1)
|
||||
return src[start:nxt if nxt != -1 else len(src)]
|
||||
|
||||
expectations = [
|
||||
('@router.post("")', '"create"'),
|
||||
('@router.get("")', '"read"'),
|
||||
('@router.get("/{csr_id}")', '"read"'),
|
||||
('@router.post("/{csr_id}/import")', '"create"'),
|
||||
('@router.delete("/{csr_id}")', '"delete"'),
|
||||
]
|
||||
for decorator, action in expectations:
|
||||
body = _handler_body(decorator)
|
||||
assert f"_require(authorization, {action})" in body, (
|
||||
f"endpoint {decorator} must enforce ssl.{action.strip(chr(34))}"
|
||||
)
|
||||
|
||||
|
||||
def test_csr_router_never_selects_private_key():
|
||||
"""The CSR endpoints must use the explicit column list — a bare
|
||||
`SELECT *` into an API response is how the key would leak. The one place
|
||||
SELECT * is allowed is the service-layer FOR UPDATE row (it needs the key
|
||||
to pair with the cert); the router itself must not touch the column."""
|
||||
src = _read(os.path.join("routers", "csr.py"))
|
||||
code_only = re.sub(r"#.*", "", src) # strip comments; the column name may
|
||||
# legitimately appear there as documentation
|
||||
assert "private_key_pem" not in code_only, (
|
||||
"routers/csr.py must never reference private_key_pem in code"
|
||||
)
|
||||
assert "SELECT *" not in code_only, "routers/csr.py must use explicit column lists"
|
||||
@@ -0,0 +1,172 @@
|
||||
"""
|
||||
v1.9.0 CSR creation — Pydantic model validation tests (models/csr.py).
|
||||
|
||||
The CSR name shares the SSL certificate name's path-traversal contract
|
||||
(Bulgu #21) with one deliberate tightening: max 100 chars, matching the
|
||||
ssl_certificates.name VARCHAR(100) column.
|
||||
"""
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
from models.csr import SSLCSRCreate, SSLCSRImport
|
||||
|
||||
_CERT_PEM = "-----BEGIN CERTIFICATE-----\nX\n-----END CERTIFICATE-----"
|
||||
|
||||
|
||||
def _create(**overrides):
|
||||
base = dict(name="my-csr", common_name="www.example.com")
|
||||
base.update(overrides)
|
||||
return SSLCSRCreate(**base)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# SSLCSRCreate
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_minimal_valid_create():
|
||||
m = _create()
|
||||
assert m.name == "my-csr"
|
||||
assert m.common_name == "www.example.com"
|
||||
assert m.key_algorithm == "rsa-2048"
|
||||
assert m.sans == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad_name", [
|
||||
"../../etc/cron.d/evil", # path traversal
|
||||
"a..b", # embedded ..
|
||||
".hidden", # hidden filename
|
||||
"-flag", # CLI flag confusion
|
||||
"has space",
|
||||
"wild*card",
|
||||
"",
|
||||
"x" * 101, # VARCHAR(100) alignment — 200 is NOT allowed here
|
||||
])
|
||||
def test_name_rejects_unsafe_values(bad_name):
|
||||
with pytest.raises(ValidationError):
|
||||
_create(name=bad_name)
|
||||
|
||||
|
||||
def test_name_accepts_100_chars():
|
||||
assert _create(name="x" * 100).name == "x" * 100
|
||||
|
||||
|
||||
def test_common_name_wildcard_accepted_and_lowercased():
|
||||
m = _create(common_name="*.Example.COM")
|
||||
assert m.common_name == "*.example.com"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("bad_cn", [
|
||||
"",
|
||||
"under_score.example.com", # _ is not LDH
|
||||
"*.*.example.com", # wildcard only as leftmost single label
|
||||
"-leading.example.com",
|
||||
"a" * 70 + ".example.com", # label > 63
|
||||
"cn-longer-than-64-chars-" + "x" * 45 + ".example.com", # CN > 64 total
|
||||
])
|
||||
def test_common_name_rejects_invalid(bad_cn):
|
||||
with pytest.raises(ValidationError):
|
||||
_create(common_name=bad_cn)
|
||||
|
||||
|
||||
def test_sans_normalised_deduped_and_capped():
|
||||
m = _create(sans=["API.example.com", "api.example.com", "cdn.example.com"])
|
||||
assert m.sans == ["api.example.com", "cdn.example.com"]
|
||||
|
||||
with pytest.raises(ValidationError):
|
||||
_create(sans=[f"h{i}.example.com" for i in range(101)])
|
||||
|
||||
|
||||
def test_country_normalised_or_rejected():
|
||||
assert _create(country="tr").country == "TR"
|
||||
assert _create(country=None).country is None
|
||||
for bad in ("TUR", "T", "1A"):
|
||||
with pytest.raises(ValidationError):
|
||||
_create(country=bad)
|
||||
|
||||
|
||||
def test_subject_fields_reject_control_characters():
|
||||
with pytest.raises(ValidationError):
|
||||
_create(organization="Evil\x00Corp")
|
||||
with pytest.raises(ValidationError):
|
||||
_create(locality="line\nbreak")
|
||||
|
||||
|
||||
def test_subject_fields_reject_overlength():
|
||||
with pytest.raises(ValidationError):
|
||||
_create(organization="x" * 65)
|
||||
|
||||
|
||||
def test_key_algorithm_strict_enum():
|
||||
for good in ("rsa-2048", "rsa-4096", "ecdsa-p256", "ecdsa-p384"):
|
||||
assert _create(key_algorithm=good).key_algorithm == good
|
||||
for bad in ("rsa-1024", "rsa-8192", "ed25519", "2048", ""):
|
||||
with pytest.raises(ValidationError):
|
||||
_create(key_algorithm=bad)
|
||||
|
||||
|
||||
def test_email_basic_validation():
|
||||
assert _create(email="ops@example.com").email == "ops@example.com"
|
||||
with pytest.raises(ValidationError):
|
||||
_create(email="not-an-email")
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# SSLCSRImport
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_import_minimal_global():
|
||||
m = SSLCSRImport(certificate_content=_CERT_PEM, is_global=True)
|
||||
assert m.usage_type == "frontend"
|
||||
assert m.name is None
|
||||
|
||||
|
||||
def test_import_requires_clusters_when_not_global():
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content=_CERT_PEM, is_global=False)
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content=_CERT_PEM, is_global=False, cluster_ids=[])
|
||||
m = SSLCSRImport(certificate_content=_CERT_PEM, is_global=False, cluster_ids=[1])
|
||||
assert m.cluster_ids == [1]
|
||||
|
||||
|
||||
def test_import_certificate_must_be_pem():
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content="not a pem", is_global=True)
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content="", is_global=True)
|
||||
|
||||
|
||||
def test_import_certificate_size_capped():
|
||||
huge = _CERT_PEM + "A" * (64 * 1024 + 1)
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content=huge, is_global=True)
|
||||
|
||||
|
||||
def test_import_chain_optional_but_validated():
|
||||
m = SSLCSRImport(certificate_content=_CERT_PEM, is_global=True, chain_content=" ")
|
||||
assert m.chain_content is None
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(
|
||||
certificate_content=_CERT_PEM, is_global=True, chain_content="garbage"
|
||||
)
|
||||
|
||||
|
||||
def test_import_name_override_shares_the_name_contract():
|
||||
m = SSLCSRImport(certificate_content=_CERT_PEM, is_global=True, name="renamed")
|
||||
assert m.name == "renamed"
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content=_CERT_PEM, is_global=True, name="../evil")
|
||||
# Empty override collapses to None (falls back to the CSR's own name).
|
||||
m2 = SSLCSRImport(certificate_content=_CERT_PEM, is_global=True, name=" ")
|
||||
assert m2.name is None
|
||||
|
||||
|
||||
def test_import_usage_type_enum():
|
||||
for good in ("frontend", "server"):
|
||||
assert SSLCSRImport(
|
||||
certificate_content=_CERT_PEM, is_global=True, usage_type=good
|
||||
).usage_type == good
|
||||
with pytest.raises(ValidationError):
|
||||
SSLCSRImport(certificate_content=_CERT_PEM, is_global=True, usage_type="both")
|
||||
@@ -0,0 +1,66 @@
|
||||
"""
|
||||
v1.9.0 CSR creation — behavioral auth tests for /api/ssl/csrs endpoints
|
||||
(pattern: test_ssl_list_endpoint_auth.py).
|
||||
|
||||
Every CSR endpoint must refuse unauthenticated / garbage-token requests.
|
||||
The CSR detail route additionally must never 200 without auth because it
|
||||
returns the CSR PEM; no endpoint ever returns the private key, but auth is
|
||||
the first line regardless.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
_VALID_CREATE_BODY = {
|
||||
"name": "auth-test-csr",
|
||||
"common_name": "www.example.com",
|
||||
}
|
||||
|
||||
_VALID_IMPORT_BODY = {
|
||||
"certificate_content": (
|
||||
"-----BEGIN CERTIFICATE-----\nX\n-----END CERTIFICATE-----"
|
||||
),
|
||||
"is_global": True,
|
||||
}
|
||||
|
||||
_ENDPOINTS = [
|
||||
("get", "/api/ssl/csrs", None),
|
||||
("get", "/api/ssl/csrs/1", None),
|
||||
("post", "/api/ssl/csrs", _VALID_CREATE_BODY),
|
||||
("post", "/api/ssl/csrs/1/import", _VALID_IMPORT_BODY),
|
||||
("delete", "/api/ssl/csrs/1", None),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method,path,body", _ENDPOINTS)
|
||||
def test_csr_endpoint_unauthenticated_rejected(client, method, path, body):
|
||||
"""No Authorization header → endpoint must refuse the request."""
|
||||
res = getattr(client, method)(path, json=body) if body is not None else getattr(client, method)(path)
|
||||
assert res.status_code in (401, 403, 422), (
|
||||
f"{method.upper()} {path} without Authorization returned "
|
||||
f"{res.status_code} — anonymous access to CSR data must not be "
|
||||
f"possible. Body: {res.text[:200]}"
|
||||
)
|
||||
if res.status_code == 200: # defensive, mirrors the R18 test style
|
||||
data = res.json()
|
||||
assert not data, "CSR endpoint returned data without auth"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("method,path,body", _ENDPOINTS)
|
||||
def test_csr_endpoint_invalid_token_rejected(client, method, path, body):
|
||||
"""Garbage token → endpoint must refuse the request."""
|
||||
headers = {"Authorization": "Bearer not-a-valid-jwt"}
|
||||
if body is not None:
|
||||
res = getattr(client, method)(path, json=body, headers=headers)
|
||||
else:
|
||||
res = getattr(client, method)(path, headers=headers)
|
||||
assert res.status_code in (401, 403, 422), (
|
||||
f"{method.upper()} {path} with an invalid token returned {res.status_code}"
|
||||
)
|
||||
|
||||
|
||||
def test_csr_routes_are_registered(client):
|
||||
"""The router must actually be mounted — a 404 would make the auth tests
|
||||
above pass vacuously."""
|
||||
res = client.get("/api/ssl/csrs")
|
||||
assert res.status_code != 404, (
|
||||
"GET /api/ssl/csrs returned 404 — csr_router is not registered in main.py"
|
||||
)
|
||||
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
v1.9.0 CSR creation — pure-crypto tests for services/csr_service.py.
|
||||
|
||||
No mocks: every algorithm's output must parse with `cryptography` and the
|
||||
CSR's public key must match the generated private key (the property the
|
||||
whole import flow depends on).
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from cryptography import x509
|
||||
from cryptography.hazmat.primitives import serialization
|
||||
from cryptography.hazmat.primitives.asymmetric import ec, rsa
|
||||
from cryptography.x509.oid import ExtensionOID, NameOID
|
||||
|
||||
from services.csr_service import csr_row_to_dict, diff_domains, generate_csr_bundle
|
||||
|
||||
|
||||
def _payload(**overrides):
|
||||
base = dict(
|
||||
name="test-csr",
|
||||
common_name="www.example.com",
|
||||
organization=None,
|
||||
organizational_unit=None,
|
||||
locality=None,
|
||||
state=None,
|
||||
country=None,
|
||||
email=None,
|
||||
sans=[],
|
||||
key_algorithm="rsa-2048",
|
||||
)
|
||||
base.update(overrides)
|
||||
return SimpleNamespace(**base)
|
||||
|
||||
|
||||
def _spki(key):
|
||||
return key.public_key().public_bytes(
|
||||
serialization.Encoding.DER,
|
||||
serialization.PublicFormat.SubjectPublicKeyInfo,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"algo,key_cls,key_check",
|
||||
[
|
||||
("rsa-2048", rsa.RSAPrivateKey, lambda k: k.key_size == 2048),
|
||||
("rsa-4096", rsa.RSAPrivateKey, lambda k: k.key_size == 4096),
|
||||
("ecdsa-p256", ec.EllipticCurvePrivateKey, lambda k: k.curve.name == "secp256r1"),
|
||||
("ecdsa-p384", ec.EllipticCurvePrivateKey, lambda k: k.curve.name == "secp384r1"),
|
||||
],
|
||||
)
|
||||
def test_generate_bundle_all_algorithms(algo, key_cls, key_check):
|
||||
bundle = generate_csr_bundle(_payload(key_algorithm=algo))
|
||||
|
||||
csr = x509.load_pem_x509_csr(bundle["csr_pem"].encode())
|
||||
key = serialization.load_pem_private_key(
|
||||
bundle["private_key_pem"].encode(), password=None
|
||||
)
|
||||
|
||||
assert isinstance(key, key_cls)
|
||||
assert key_check(key)
|
||||
# The CSR must be signed by exactly this key.
|
||||
csr_spki = csr.public_key().public_bytes(
|
||||
serialization.Encoding.DER,
|
||||
serialization.PublicFormat.SubjectPublicKeyInfo,
|
||||
)
|
||||
assert csr_spki == _spki(key)
|
||||
assert csr.is_signature_valid
|
||||
# PKCS8, unencrypted — the agent concatenates cert+key into one PEM and
|
||||
# HAProxy cannot read passphrase-protected keys.
|
||||
assert bundle["private_key_pem"].startswith("-----BEGIN PRIVATE KEY-----")
|
||||
|
||||
|
||||
def test_subject_contains_all_provided_fields():
|
||||
bundle = generate_csr_bundle(_payload(
|
||||
organization="Example Corp",
|
||||
organizational_unit="IT",
|
||||
locality="Istanbul",
|
||||
state="Marmara",
|
||||
country="TR",
|
||||
email="ops@example.com",
|
||||
))
|
||||
csr = x509.load_pem_x509_csr(bundle["csr_pem"].encode())
|
||||
|
||||
def _one(oid):
|
||||
attrs = csr.subject.get_attributes_for_oid(oid)
|
||||
return attrs[0].value if attrs else None
|
||||
|
||||
assert _one(NameOID.COMMON_NAME) == "www.example.com"
|
||||
assert _one(NameOID.ORGANIZATION_NAME) == "Example Corp"
|
||||
assert _one(NameOID.ORGANIZATIONAL_UNIT_NAME) == "IT"
|
||||
assert _one(NameOID.LOCALITY_NAME) == "Istanbul"
|
||||
assert _one(NameOID.STATE_OR_PROVINCE_NAME) == "Marmara"
|
||||
assert _one(NameOID.COUNTRY_NAME) == "TR"
|
||||
assert _one(NameOID.EMAIL_ADDRESS) == "ops@example.com"
|
||||
assert bundle["subject"] == {
|
||||
"O": "Example Corp", "OU": "IT", "L": "Istanbul",
|
||||
"ST": "Marmara", "C": "TR", "emailAddress": "ops@example.com",
|
||||
}
|
||||
|
||||
|
||||
def test_subject_omits_empty_fields():
|
||||
bundle = generate_csr_bundle(_payload())
|
||||
csr = x509.load_pem_x509_csr(bundle["csr_pem"].encode())
|
||||
assert not csr.subject.get_attributes_for_oid(NameOID.ORGANIZATION_NAME)
|
||||
assert bundle["subject"] == {}
|
||||
|
||||
|
||||
def test_sans_cn_first_and_deduped():
|
||||
bundle = generate_csr_bundle(_payload(
|
||||
common_name="www.example.com",
|
||||
sans=["api.example.com", "www.example.com", "api.example.com", "cdn.example.com"],
|
||||
))
|
||||
assert bundle["sans"] == ["www.example.com", "api.example.com", "cdn.example.com"]
|
||||
|
||||
csr = x509.load_pem_x509_csr(bundle["csr_pem"].encode())
|
||||
san_ext = csr.extensions.get_extension_for_oid(
|
||||
ExtensionOID.SUBJECT_ALTERNATIVE_NAME
|
||||
)
|
||||
dns_names = san_ext.value.get_values_for_type(x509.DNSName)
|
||||
assert dns_names == ["www.example.com", "api.example.com", "cdn.example.com"]
|
||||
|
||||
|
||||
def test_wildcard_common_name_flows_into_san():
|
||||
bundle = generate_csr_bundle(_payload(common_name="*.example.com"))
|
||||
csr = x509.load_pem_x509_csr(bundle["csr_pem"].encode())
|
||||
san_ext = csr.extensions.get_extension_for_oid(
|
||||
ExtensionOID.SUBJECT_ALTERNATIVE_NAME
|
||||
)
|
||||
assert san_ext.value.get_values_for_type(x509.DNSName) == ["*.example.com"]
|
||||
|
||||
|
||||
def test_diff_domains_reports_added_and_dropped():
|
||||
warnings = diff_domains(
|
||||
["www.example.com", "api.example.com"],
|
||||
["WWW.example.com", "cdn.example.com"],
|
||||
)
|
||||
assert len(warnings) == 2
|
||||
added = next(w for w in warnings if "added" in w)
|
||||
dropped = next(w for w in warnings if "dropped" in w)
|
||||
assert "cdn.example.com" in added
|
||||
assert "api.example.com" in dropped
|
||||
# Case-insensitive: www must NOT be reported in either direction.
|
||||
assert "www.example.com" not in added
|
||||
assert "www.example.com" not in dropped
|
||||
|
||||
|
||||
def test_diff_domains_identical_sets_yield_no_warnings():
|
||||
assert diff_domains(["a.example.com"], ["A.EXAMPLE.COM"]) == []
|
||||
assert diff_domains([], []) == []
|
||||
|
||||
|
||||
def test_csr_row_to_dict_never_exposes_private_key():
|
||||
row = {
|
||||
"id": 1,
|
||||
"name": "x",
|
||||
"private_key_pem": "-----BEGIN PRIVATE KEY-----\nSECRET\n-----END PRIVATE KEY-----",
|
||||
"csr_pem": "-----BEGIN CERTIFICATE REQUEST-----\nX\n-----END CERTIFICATE REQUEST-----",
|
||||
"subject": '{"O": "Example"}',
|
||||
"sans": '["a.example.com"]',
|
||||
}
|
||||
out = csr_row_to_dict(row)
|
||||
assert "private_key_pem" not in out
|
||||
assert "csr_pem" not in out # lists exclude the PEM
|
||||
assert out["subject"] == {"O": "Example"}
|
||||
assert out["sans"] == ["a.example.com"]
|
||||
|
||||
detail = csr_row_to_dict(row, include_pem=True)
|
||||
assert "private_key_pem" not in detail # NEVER, even on detail
|
||||
assert detail["csr_pem"].startswith("-----BEGIN CERTIFICATE REQUEST-----")
|
||||
@@ -0,0 +1,615 @@
|
||||
"""Issue #35 — ACME DNS-01: focused unit tests for the pure logic (no DB/network).
|
||||
|
||||
Covers the TXT-value math (RFC 8555 §8.4 — raw SHA-256 digest, base64url, NOT hex),
|
||||
the _acme-challenge record-name derivation (wildcard stripping), credential encryption
|
||||
round-trip + tamper handling, the DNS provider registry/allow-list, and (v1.10.0) the
|
||||
GoDaddy provider's zone-relative name derivation and additive RRset merge math.
|
||||
"""
|
||||
import base64
|
||||
import hashlib
|
||||
import os
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-dns01-unit-tests")
|
||||
|
||||
from services.acme_service import ACMEService
|
||||
from services.dns_providers import list_providers, is_supported, get_provider, DnsProviderError
|
||||
from utils.dns_credentials import (
|
||||
encrypt_dns_credentials, decrypt_dns_credentials, reset_fernet_for_tests,
|
||||
)
|
||||
|
||||
|
||||
def _b64url(b: bytes) -> str:
|
||||
return base64.urlsafe_b64encode(b).rstrip(b"=").decode("ascii")
|
||||
|
||||
|
||||
def test_dns_txt_value_is_raw_sha256_base64url():
|
||||
key_auth = "token123.thumbprintABC"
|
||||
expected = _b64url(hashlib.sha256(key_auth.encode("utf-8")).digest())
|
||||
assert ACMEService._dns_txt_value(key_auth) == expected
|
||||
# Must NOT be the (classic-mistake) base64url of the HEX digest.
|
||||
hex_based = _b64url(hashlib.sha256(key_auth.encode("utf-8")).hexdigest().encode("utf-8"))
|
||||
assert ACMEService._dns_txt_value(key_auth) != hex_based
|
||||
|
||||
|
||||
def test_challenge_dns_name_derivation():
|
||||
assert ACMEService._challenge_dns_name("example.com") == "_acme-challenge.example.com"
|
||||
# Wildcard: the '*.' is stripped, so apex + wildcard share the SAME record name.
|
||||
assert ACMEService._challenge_dns_name("*.example.com") == "_acme-challenge.example.com"
|
||||
assert ACMEService._challenge_dns_name("foo.bar.example.com") == "_acme-challenge.foo.bar.example.com"
|
||||
|
||||
|
||||
def test_credential_encryption_roundtrip():
|
||||
reset_fernet_for_tests()
|
||||
creds = {"api_token": "super-secret-token-value"}
|
||||
token = encrypt_dns_credentials(creds)
|
||||
assert token != "super-secret-token-value"
|
||||
assert "super-secret-token-value" not in token # ciphertext, not plaintext
|
||||
assert decrypt_dns_credentials(token) == creds
|
||||
|
||||
|
||||
def test_decrypt_invalid_token_returns_none():
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_dns_credentials("not-a-valid-fernet-token") is None
|
||||
|
||||
|
||||
def test_provider_registry_and_allow_list():
|
||||
names = {p["name"] for p in list_providers()}
|
||||
assert {"manual", "cloudflare", "godaddy"} <= names
|
||||
assert is_supported("manual") and is_supported("cloudflare") and is_supported("godaddy")
|
||||
assert not is_supported("route53") # not in the allow-list
|
||||
|
||||
assert get_provider("manual").automated is False
|
||||
cf = get_provider("cloudflare", {"api_token": "x"})
|
||||
assert cf.automated is True
|
||||
assert any(f["key"] == "api_token" for f in cf.credential_fields)
|
||||
|
||||
raised = False
|
||||
try:
|
||||
get_provider("definitely-not-a-provider")
|
||||
except ValueError:
|
||||
raised = True
|
||||
assert raised
|
||||
|
||||
|
||||
def test_cloudflare_token_sanitize():
|
||||
# Issue #35 follow-up: a pasted token with quotes/spaces/control/unicode chars produced an
|
||||
# invalid Authorization header (CF 6003 "Invalid request headers"). The sanitizer strips them.
|
||||
from services.dns_providers.cloudflare import _sanitize_token, CloudflareDNSProvider
|
||||
|
||||
# Surrounding double quotes stripped.
|
||||
assert _sanitize_token('"abc123-_def"') == 'abc123-_def'
|
||||
# Interior spaces / tabs / newlines removed.
|
||||
assert _sanitize_token('abc 123\tdef\n') == 'abc123def'
|
||||
# A clean token68 string is unchanged (cannot corrupt a valid Cloudflare token).
|
||||
clean = 'A1b2-_C3.d4~e5+f6/g7=='
|
||||
assert _sanitize_token(clean) == clean
|
||||
# Single quotes and a zero-width char removed.
|
||||
assert _sanitize_token("'tok" + chr(0x200b) + "en'") == 'token'
|
||||
|
||||
# The provider constructor sanitizes into _token and keeps the raw input for diagnostics.
|
||||
p = CloudflareDNSProvider({"api_token": '"my-token_123"'})
|
||||
assert p._token == 'my-token_123'
|
||||
assert p._raw_token == '"my-token_123"'
|
||||
|
||||
|
||||
# --- v1.10.0: GoDaddy provider (pure logic only — no network, no DB) ---
|
||||
|
||||
|
||||
def test_godaddy_credential_fields_schema():
|
||||
# Re-assert DnsCredentialsUpsert's validator rules directly against the declared schema, so the
|
||||
# UI can never render a field whose submission the API would reject with a 422.
|
||||
import re
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
fields = GoDaddyDNSProvider.credential_fields
|
||||
assert [f["key"] for f in fields] == ["api_key", "api_secret"]
|
||||
for f in fields:
|
||||
assert re.match(r"^[a-zA-Z0-9_]{1,50}$", f["key"]) # DnsCredentialsUpsert key regex
|
||||
assert f["type"] == "password" # renders Input.Password, not Input
|
||||
assert isinstance(f["max_length"], int) and 0 < f["max_length"] <= 4000 # validator value cap
|
||||
assert f["help"] and isinstance(f["help"], str) # shown in the Form.Item `extra` slot
|
||||
# api_secret is optional on purpose: leaving it blank is how a Personal Access Token is used
|
||||
# (Bearer), which is the migration path off the sso-key scheme GoDaddy is retiring.
|
||||
assert fields[0]["required"] is True and fields[1]["required"] is False
|
||||
# Must not reuse Cloudflare's field name: the register modal's credential Form.Items are named
|
||||
# cred_<key> in a SHARED form and are not cleared when the provider dropdown changes.
|
||||
assert "api_token" not in {f["key"] for f in fields}
|
||||
|
||||
|
||||
def test_godaddy_provider_is_automated():
|
||||
p = get_provider("godaddy", {"api_key": "k", "api_secret": "s"})
|
||||
assert p.automated is True # else the orchestrator takes the manual-confirm branch
|
||||
assert p.name == "godaddy" and 1 <= len(p.name) <= 50 # dns_provider Field(min_length=1, max_length=50)
|
||||
assert p.label == "GoDaddy"
|
||||
|
||||
|
||||
def test_godaddy_missing_credentials_returns_not_ok():
|
||||
# verify_credentials must RETURN {"ok": False}, never raise: the router turns any non-
|
||||
# DnsProviderError into the information-free generic 422 and the user never sees the reason.
|
||||
import asyncio
|
||||
|
||||
for creds in ({}, {"api_secret": "s"}): # blank UI fields arrive as MISSING keys, not ""
|
||||
r = asyncio.run(get_provider("godaddy", creds).verify_credentials())
|
||||
assert r["ok"] is False and r["detail"]
|
||||
# Short-circuits before any request, so this touches no network.
|
||||
|
||||
|
||||
def test_godaddy_auth_header_formats_and_secret_never_leaks():
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider, _scrub
|
||||
|
||||
sentinel = "SENTINEL-SECRET-DO-NOT-LEAK"
|
||||
p = GoDaddyDNSProvider({"api_key": "KEY123", "api_secret": sentinel})
|
||||
# Literal prefix, one space, a single colon — no base64, no quoting.
|
||||
assert p._auth_header() == f"sso-key KEY123:{sentinel}"
|
||||
# No secret -> Personal Access Token. This one branch is the whole sso-key-sunset migration.
|
||||
assert GoDaddyDNSProvider({"api_key": "PAT"})._auth_header() == "Bearer PAT"
|
||||
# _scrub removes credential substrings from anything bound for a log or an order event.
|
||||
assert sentinel not in _scrub(f"boom {sentinel} boom", "KEY123", sentinel)
|
||||
assert "KEY123" not in _scrub("boom KEY123", "KEY123", sentinel)
|
||||
assert _scrub("x" * 500, "KEY123") == "x" * 300 # bounded, so a huge body can't flood an event
|
||||
|
||||
# The channel that actually persists text: _http_error composes the message an order event and
|
||||
# letsencrypt_orders.error_detail will carry, so it must scrub its own inputs — a caller that
|
||||
# forgets to pre-scrub must not be able to leak. (Regression guard: scrubbing used to live at
|
||||
# the single call site in _request instead of here.)
|
||||
exc = p._http_error(403, f"DENIED_{sentinel}", f"token {sentinel} rejected", None)
|
||||
assert sentinel not in str(exc) and "***" in str(exc)
|
||||
|
||||
# This module must not log at all — logging is the one channel _scrub cannot reach, since the
|
||||
# arguments would be formatted by the logging framework rather than passed through it.
|
||||
import inspect
|
||||
import re as _re
|
||||
from services.dns_providers import godaddy as gd_mod
|
||||
|
||||
assert not _re.search(r"\blogger\.\w+\(", inspect.getsource(gd_mod)), \
|
||||
"godaddy.py must not log; surface everything through DnsProviderError so it is scrubbed"
|
||||
|
||||
|
||||
def test_godaddy_relative_record_name():
|
||||
# GoDaddy names are RELATIVE to the zone with no trailing dot; the apex is the literal "@".
|
||||
from services.dns_providers.godaddy import _relative_name
|
||||
|
||||
assert _relative_name("_acme-challenge.example.com", "example.com") == "_acme-challenge"
|
||||
assert _relative_name("_acme-challenge.foo.bar.example.com", "example.com") == "_acme-challenge.foo.bar"
|
||||
assert _relative_name("example.com", "example.com") == "@" # never "" — see _rrset_path
|
||||
assert _relative_name("_acme-challenge.example.com.", "example.com") == "_acme-challenge"
|
||||
assert _relative_name("_ACME-Challenge.Example.COM", "example.com") == "_acme-challenge"
|
||||
# Apex and wildcard produce the SAME relative name — which is exactly why the merge below
|
||||
# has to be additive.
|
||||
apex = ACMEService._challenge_dns_name("example.com")
|
||||
wild = ACMEService._challenge_dns_name("*.example.com")
|
||||
assert _relative_name(apex, "example.com") == _relative_name(wild, "example.com") == "_acme-challenge"
|
||||
|
||||
|
||||
def test_godaddy_rrset_merge_is_additive():
|
||||
# THE critical test: GoDaddy's PUT REPLACES an entire RRset, so the merge math is the only thing
|
||||
# keeping a wildcard+apex certificate's two coexisting TXT values alive.
|
||||
from services.dns_providers.godaddy import _live_values, _merge_add, _merge_remove
|
||||
|
||||
def vals(body):
|
||||
return sorted(r["data"] for r in body)
|
||||
|
||||
assert vals(_merge_add([{"data": "valueA", "ttl": 600}], "valueB")) == ["valueA", "valueB"]
|
||||
assert _merge_add([{"data": "valueA"}], "valueA") is None # idempotent; ACME retries land here
|
||||
# Total, not an all()-over-a-computed-list (which passes vacuously on an empty result): the
|
||||
# first publish at a fresh name must emit exactly one element, carrying the 600s TTL floor.
|
||||
assert _merge_add([], "v") == [{"data": "v", "ttl": 600}] # below 600 GoDaddy answers 422
|
||||
# Tombstone rows ({"data": ""}) must never be echoed back — GoDaddy answers 422 INVALID_BODY.
|
||||
assert _live_values([{"data": ""}, {"data": "x"}, {}]) == ["x"]
|
||||
assert vals(_merge_add([{"data": ""}, {"data": "valueA"}], "valueB")) == ["valueA", "valueB"]
|
||||
|
||||
assert vals(_merge_remove([{"data": "valueA"}, {"data": "valueB"}], "valueB")) == ["valueA"]
|
||||
assert _merge_remove([{"data": "valueA"}], "valueZ") is None # already gone — tolerate
|
||||
assert _merge_remove([], "valueZ") is None
|
||||
# [] means "use DELETE": PUT with an empty array is rejected (422, "Records must be specified").
|
||||
assert _merge_remove([{"data": "valueA"}], "valueA") == []
|
||||
assert _merge_remove([{"data": ""}, {"data": "valueA"}], "valueA") == []
|
||||
|
||||
|
||||
def test_godaddy_never_builds_a_zone_wide_txt_path():
|
||||
# A 3-segment path (.../records/TXT) is the endpoint that wipes EVERY TXT in the zone — SPF,
|
||||
# DKIM, DMARC, domain verifications. An empty relative name must never be able to produce it.
|
||||
from services.dns_providers.godaddy import _rrset_path
|
||||
|
||||
p = _rrset_path("example.com", "_acme-challenge")
|
||||
assert p == "/domains/example.com/records/TXT/_acme-challenge"
|
||||
assert p.count("/") == 5 and not p.endswith("/TXT")
|
||||
assert _rrset_path("example.com", "@").endswith("/%40") # apex percent-encoded for proxy safety
|
||||
# "." and ".." survive quote() and are then normalized away by yarl when the URL is built, so
|
||||
# ".../records/TXT/.." would resolve to the whole-zone endpoint. They must be refused too.
|
||||
for bad in [("example.com", ""), ("", "_acme-challenge"), ("example.com", "."),
|
||||
("example.com", ".."), ("example.com", "...")]:
|
||||
raised = False
|
||||
try:
|
||||
_rrset_path(*bad)
|
||||
except DnsProviderError:
|
||||
raised = True
|
||||
assert raised, f"_rrset_path{bad} must refuse to build a zone-wide TXT path"
|
||||
# And the only way to reach those inputs — a malformed domain — really does produce them.
|
||||
from services.dns_providers.godaddy import _relative_name as _rel
|
||||
assert _rel("..example.com", "example.com") == "."
|
||||
|
||||
|
||||
def test_godaddy_credential_encryption_roundtrip():
|
||||
# The two-field credential dict rides the same Fernet blob as Cloudflare's single token.
|
||||
reset_fernet_for_tests()
|
||||
creds = {"api_key": "gd-key-plaintext", "api_secret": "gd-secret-plaintext"}
|
||||
token = encrypt_dns_credentials(creds)
|
||||
assert "gd-key-plaintext" not in token and "gd-secret-plaintext" not in token # ciphertext
|
||||
assert decrypt_dns_credentials(token) == creds
|
||||
# This sorted key list is exactly what GET /dns-credentials exposes as credential_fields_present
|
||||
# — names only, never values.
|
||||
assert sorted(decrypt_dns_credentials(token).keys()) == ["api_key", "api_secret"]
|
||||
|
||||
|
||||
_GD_NS = "/domains/example.com/records/NS"
|
||||
_GD_TXT = "/domains/example.com/records/TXT/_acme-challenge"
|
||||
|
||||
|
||||
def _gd_provider(responses):
|
||||
"""A GoDaddy provider whose _request is replaced by a recorder.
|
||||
|
||||
The pure-merge tests above prove the MATH; this proves the WRITE PATH actually uses it. Without
|
||||
it, replacing the merge with a single-value PUT — the mutation that silently destroys the
|
||||
sibling value of every wildcard+apex certificate — leaves the whole suite green.
|
||||
|
||||
`responses` maps (method, path) -> value to return, or an Exception to raise. Unmapped calls
|
||||
return None, which is how the zone suffix-walk's failed probes are modelled.
|
||||
"""
|
||||
import types
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
calls = []
|
||||
|
||||
async def _fake_request(self, session, method, path, **kwargs):
|
||||
calls.append((method, path, kwargs.get("json")))
|
||||
result = responses.get((method, path))
|
||||
if isinstance(result, Exception):
|
||||
raise result
|
||||
return result
|
||||
|
||||
p = GoDaddyDNSProvider({"api_key": "k", "api_secret": "s"})
|
||||
p._request = types.MethodType(_fake_request, p)
|
||||
return p, calls
|
||||
|
||||
|
||||
def _assert_never_zone_wide(calls):
|
||||
# A write to .../records or .../records/TXT replaces every TXT (or every record) in the zone.
|
||||
for method, path, _json in calls:
|
||||
if method in ("PUT", "DELETE"):
|
||||
assert not path.endswith("/records"), f"zone-wide write: {method} {path}"
|
||||
assert not path.endswith("/records/TXT"), f"type-wide write: {method} {path}"
|
||||
|
||||
|
||||
def test_godaddy_add_write_path_merges_siblings():
|
||||
import asyncio
|
||||
|
||||
# An existing sibling value at the same name — the apex half of an apex+wildcard certificate.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA", "ttl": 600}],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
|
||||
writes = [c for c in calls if c[0] in ("PUT", "PATCH", "DELETE")]
|
||||
assert len(writes) == 1 and writes[0][0] == "PUT" and writes[0][1] == _GD_TXT
|
||||
# BOTH values must be in the body: GoDaddy's PUT replaces the whole RRset.
|
||||
assert sorted(r["data"] for r in writes[0][2]) == ["valueA", "valueB"]
|
||||
_assert_never_zone_wide(calls)
|
||||
|
||||
|
||||
def test_godaddy_add_write_path_is_idempotent_and_fails_closed():
|
||||
import asyncio
|
||||
|
||||
# Already published -> no write at all (this is where an ACME retry cycle lands).
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueB", "ttl": 600}],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
|
||||
# Unreadable RRset read (2xx whose body did not parse as a list) must FAIL, never be treated as
|
||||
# an empty RRset — the PUT that follows would replace the sibling values with only ours.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): None,
|
||||
})
|
||||
raised = False
|
||||
try:
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
except DnsProviderError:
|
||||
raised = True
|
||||
assert raised, "an unreadable RRset read must not be coerced into an empty RRset"
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
|
||||
|
||||
def test_godaddy_remove_write_path_uses_delete_for_the_last_value():
|
||||
import asyncio
|
||||
|
||||
# Two values -> PUT back the survivor only.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}, {"data": "valueB"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
writes = [c for c in calls if c[0] != "GET"]
|
||||
assert len(writes) == 1 and writes[0][0] == "PUT"
|
||||
assert [r["data"] for r in writes[0][2]] == ["valueA"]
|
||||
|
||||
# Last value -> DELETE. `PUT []` is rejected by GoDaddy (422 INVALID_BODY), so an empty PUT
|
||||
# body would make every cleanup fail forever.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueA"))
|
||||
writes = [c for c in calls if c[0] != "GET"]
|
||||
assert len(writes) == 1 and writes[0] == ("DELETE", _GD_TXT, None)
|
||||
assert not any(c[0] == "PUT" and c[2] == [] for c in calls)
|
||||
|
||||
# Value already gone -> no write, no error.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueZ"))
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
_assert_never_zone_wide(calls)
|
||||
|
||||
|
||||
class _FakeGDResponse:
|
||||
"""Minimal stand-in for aiohttp's ClientResponse: status, headers, and json()."""
|
||||
|
||||
_NO_BODY = object()
|
||||
|
||||
def __init__(self, status, body=_NO_BODY, headers=None):
|
||||
self.status = status
|
||||
self._body = body
|
||||
self.headers = headers or {}
|
||||
|
||||
async def json(self, content_type=None):
|
||||
if self._body is _FakeGDResponse._NO_BODY:
|
||||
raise ValueError("no body to decode") # what an empty 204 does
|
||||
return self._body
|
||||
|
||||
|
||||
class _FakeGDSession:
|
||||
def __init__(self, response):
|
||||
self._response = response
|
||||
self.calls = []
|
||||
|
||||
def request(self, method, url, **kwargs):
|
||||
self.calls.append((method, url, kwargs))
|
||||
response = self._response
|
||||
|
||||
class _Ctx:
|
||||
async def __aenter__(self_inner):
|
||||
return response
|
||||
|
||||
async def __aexit__(self_inner, *exc):
|
||||
return False
|
||||
|
||||
return _Ctx()
|
||||
|
||||
|
||||
def test_godaddy_request_status_handling():
|
||||
import asyncio
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
p = GoDaddyDNSProvider({"api_key": "KEY123", "api_secret": "SEC456"})
|
||||
|
||||
def call(response):
|
||||
session = _FakeGDSession(response)
|
||||
try:
|
||||
return asyncio.run(p._request(session, "PUT", "/domains/example.com/records/TXT/x",
|
||||
json=[{"data": "v", "ttl": 600}])), None, session
|
||||
except DnsProviderError as exc:
|
||||
return None, str(exc), session
|
||||
|
||||
# 204 with an EMPTY body is the normal answer to every GoDaddy write — it must not raise.
|
||||
body, err, session = call(_FakeGDResponse(204))
|
||||
assert body is None and err is None
|
||||
# Redirects are deliberately not followed (aiohttp would forward the Authorization header), so
|
||||
# a 3xx is a FAILED call. Treating it as success would report a redirected write as a no-op.
|
||||
_kw = session.calls[0][2]
|
||||
assert _kw["allow_redirects"] is False
|
||||
assert _kw["headers"]["Authorization"] == "sso-key KEY123:SEC456"
|
||||
assert _kw["headers"]["Accept"] == "application/json"
|
||||
for status in (301, 302, 307):
|
||||
body, err, _ = call(_FakeGDResponse(status))
|
||||
assert body is None and err and str(status) in err, f"HTTP {status} must not read as success"
|
||||
|
||||
# 200 with a list is passed through verbatim.
|
||||
body, err, _ = call(_FakeGDResponse(200, [{"data": "v"}]))
|
||||
assert err is None and body == [{"data": "v"}]
|
||||
|
||||
# Error mapping: each message must name what the operator has to fix.
|
||||
_, err, _ = call(_FakeGDResponse(401, {"code": "UNABLE_TO_AUTHENTICATE", "message": "nope"}))
|
||||
assert "PRODUCTION" in err and "UNABLE_TO_AUTHENTICATE" in err
|
||||
_, err, _ = call(_FakeGDResponse(403, {"code": "ACCESS_DENIED", "message": "not allowed"}))
|
||||
assert "domains.dns:update" in err
|
||||
# 429: Retry-After wins; the legacy body field is the fallback; absent both -> 60s default.
|
||||
_, err, _ = call(_FakeGDResponse(429, None, {"Retry-After": "17"}))
|
||||
assert "~17s" in err
|
||||
_, err, _ = call(_FakeGDResponse(429, {"retryAfterSec": 42}))
|
||||
assert "~42s" in err
|
||||
_, err, _ = call(_FakeGDResponse(429, {"Retry-After": "not-a-number"}))
|
||||
assert "~60s" in err
|
||||
# A non-dict error body must not crash the error mapper.
|
||||
_, err, _ = call(_FakeGDResponse(500, "<html>gateway</html>"))
|
||||
assert "500" in err
|
||||
|
||||
# A transport failure MID-READ must not be mistaken for "empty body". Only a decode error may
|
||||
# be swallowed: a caller reading an RRset would otherwise see None and could take it for an
|
||||
# empty record set, and the full-RRset PUT that follows would destroy the sibling values.
|
||||
import aiohttp
|
||||
|
||||
class _TruncatedResponse(_FakeGDResponse):
|
||||
async def json(self, content_type=None):
|
||||
raise aiohttp.ClientPayloadError("connection closed mid-body")
|
||||
|
||||
body, err, _ = call(_TruncatedResponse(200))
|
||||
assert body is None and err and "GoDaddy" in err
|
||||
|
||||
|
||||
def test_godaddy_zone_resolution_walks_suffixes_and_caches():
|
||||
import asyncio
|
||||
from services.dns_providers.godaddy import _GoDaddyHTTPError
|
||||
|
||||
# The deepest candidate is not a zone (404 = "not this zone"); the walk must continue to the
|
||||
# registrable domain and then reuse it, so the second challenge at the same name costs no probe.
|
||||
p, calls = _gd_provider({
|
||||
("GET", "/domains/_acme-challenge.example.com/records/NS"):
|
||||
_GoDaddyHTTPError("nope", status=404, code="UNKNOWN_DOMAIN"),
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueA"))
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
probes = [c for c in calls if c[1].endswith("/records/NS")]
|
||||
assert len(probes) == 2, "the resolved zone must be cached for the life of the provider"
|
||||
# Relative name derived from the RESOLVED zone, never from the deepest candidate.
|
||||
assert all(c[1] == _GD_TXT for c in calls if "/records/TXT/" in c[1])
|
||||
|
||||
# A credential/eligibility failure during the walk must surface, not be swallowed as
|
||||
# "no managed domain" — otherwise the operator chases a DNS problem that is really a bad key.
|
||||
p, calls = _gd_provider({
|
||||
("GET", "/domains/_acme-challenge.example.com/records/NS"):
|
||||
_GoDaddyHTTPError("denied", status=403, code="ACCESS_DENIED"),
|
||||
})
|
||||
raised = ""
|
||||
try:
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "v"))
|
||||
except DnsProviderError as exc:
|
||||
raised = str(exc)
|
||||
assert "denied" in raised and "No managed GoDaddy domain" not in raised
|
||||
|
||||
|
||||
def test_b64url_decode_padding_roundtrip():
|
||||
# Issue #35 v1.8.2: _b64url_decode must round-trip for EVERY length, including base64url strings
|
||||
# whose length is a multiple of 4 (the case the old padding formula '=' * (4 - len%4) over-padded).
|
||||
from services.acme_service import _b64url as enc_fn, _b64url_decode as dec_fn
|
||||
for n in range(0, 20):
|
||||
data = bytes(range(n))
|
||||
assert dec_fn(enc_fn(data)) == data, f"round-trip failed at byte length {n}"
|
||||
|
||||
|
||||
def test_nonce_scoped_per_directory():
|
||||
# Issue #35 v1.8.2: a nonce cached for one CA (directory_url) must never be returned for another,
|
||||
# and must be single-use. Both directories are pre-cached so _get_nonce returns without network.
|
||||
import asyncio
|
||||
svc = ACMEService()
|
||||
svc._nonce_by_dir = {"https://a.example/dir": "NONCE_A", "https://b.example/dir": "NONCE_B"}
|
||||
got = asyncio.run(svc._get_nonce("https://a.example/dir"))
|
||||
assert got == "NONCE_A" # returns THIS CA's nonce
|
||||
assert svc._nonce_by_dir.get("https://a.example/dir") is None # consumed (single-use)
|
||||
assert svc._nonce_by_dir.get("https://b.example/dir") == "NONCE_B" # the other CA is untouched
|
||||
|
||||
|
||||
def _sql_paren_depth(sql: str):
|
||||
"""Parenthesis depth of a SQL string, counting only OUTSIDE '...' literals (with ''
|
||||
escapes), `--` line comments and /* */ block comments. Single-pass state machine so a
|
||||
`--` inside a literal or a `'` inside a comment cannot corrupt the count. Dollar-quoted
|
||||
strings are out of scope (not used in this codebase). Returns (final_depth, min_depth).
|
||||
"""
|
||||
depth = 0
|
||||
min_depth = 0
|
||||
state = "normal"
|
||||
i, n = 0, len(sql)
|
||||
while i < n:
|
||||
ch = sql[i]
|
||||
nxt = sql[i + 1] if i + 1 < n else ""
|
||||
if state == "normal":
|
||||
if ch == "'":
|
||||
state = "string"
|
||||
elif ch == "-" and nxt == "-":
|
||||
state = "line_comment"
|
||||
i += 1
|
||||
elif ch == "/" and nxt == "*":
|
||||
state = "block_comment"
|
||||
i += 1
|
||||
elif ch == "(":
|
||||
depth += 1
|
||||
elif ch == ")":
|
||||
depth -= 1
|
||||
min_depth = min(min_depth, depth)
|
||||
elif state == "string":
|
||||
if ch == "'":
|
||||
if nxt == "'":
|
||||
i += 1 # escaped '' stays inside the literal
|
||||
else:
|
||||
state = "normal"
|
||||
elif state == "line_comment":
|
||||
if ch == "\n":
|
||||
state = "normal"
|
||||
else: # block_comment
|
||||
if ch == "*" and nxt == "/":
|
||||
state = "normal"
|
||||
i += 1
|
||||
i += 1
|
||||
return depth, min_depth
|
||||
|
||||
|
||||
def test_acme_sql_parentheses_balanced():
|
||||
"""Issue #35 v1.8.5: the completion task's order-claim query shipped (v1.8.0-v1.8.4) with an
|
||||
extra closing parenthesis, so EVERY 60s cycle died with `syntax error at or near ")"` and no
|
||||
background ACME work (claim/finalize/download, DNS-01 publish, wizard-staged promotion,
|
||||
retry, TXT cleanup) ever ran. The suite never caught it because the DB layer is mocked and
|
||||
raw SQL never reaches a real parser. This guard scans the ACME modules' SQL string literals
|
||||
for unbalanced parentheses.
|
||||
|
||||
Guard scope is deliberately conservative to avoid false positives on production changes:
|
||||
keyword matching is case-sensitive (SQL is uppercase in this codebase; prose in docstrings
|
||||
is not) and f-string fragments are excluded (they split at `{`, so a fragment may be
|
||||
legitimately unbalanced).
|
||||
"""
|
||||
import ast
|
||||
import re
|
||||
|
||||
backend_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
modules = [
|
||||
"main.py",
|
||||
os.path.join("services", "dns01_orchestrator.py"),
|
||||
os.path.join("services", "acme_service.py"),
|
||||
os.path.join("services", "letsencrypt_service.py"),
|
||||
os.path.join("routers", "letsencrypt.py"),
|
||||
os.path.join("routers", "acme_diagnostics.py"),
|
||||
]
|
||||
problems = []
|
||||
for rel in modules:
|
||||
with open(os.path.join(backend_dir, rel), encoding="utf-8") as fh:
|
||||
tree = ast.parse(fh.read())
|
||||
fstring_parts = {
|
||||
id(const)
|
||||
for joined in ast.walk(tree) if isinstance(joined, ast.JoinedStr)
|
||||
for const in ast.walk(joined) if isinstance(const, ast.Constant)
|
||||
}
|
||||
for node in ast.walk(tree):
|
||||
if not (isinstance(node, ast.Constant) and isinstance(node.value, str)):
|
||||
continue
|
||||
if id(node) in fstring_parts:
|
||||
continue
|
||||
sql = node.value
|
||||
if not re.search(r"\b(SELECT|INSERT|UPDATE|DELETE)\b", sql):
|
||||
continue
|
||||
if not re.search(r"\b(FROM|INTO|SET|WHERE)\b", sql):
|
||||
continue
|
||||
depth, min_depth = _sql_paren_depth(sql)
|
||||
if depth != 0 or min_depth < 0:
|
||||
problems.append(f"{rel}:{node.lineno} (paren depth {depth:+d}, min {min_depth})")
|
||||
assert not problems, f"Unbalanced parentheses in SQL literal(s): {problems}"
|
||||
|
||||
|
||||
def test_sql_paren_depth_scanner():
|
||||
# The guard's scanner itself: parens in literals/comments must not count; '' escapes and
|
||||
# block comments handled; an extra ')' is reported via min_depth even if a later '(' would
|
||||
# re-balance the total.
|
||||
assert _sql_paren_depth("SELECT (1)") == (0, 0)
|
||||
assert _sql_paren_depth("SELECT (1))") == (-1, -1) # the v1.8.0 bug shape
|
||||
assert _sql_paren_depth("SELECT ')' , '((' FROM t") == (0, 0) # literals ignored
|
||||
assert _sql_paren_depth("SELECT 'it''s ))' FROM t") == (0, 0) # '' escape stays inside
|
||||
assert _sql_paren_depth("SELECT 1 -- comment ) (\nFROM t") == (0, 0) # line comment ignored
|
||||
assert _sql_paren_depth("SELECT 1 /* ) */ FROM t") == (0, 0) # block comment ignored
|
||||
assert _sql_paren_depth("SELECT 'a--b' AND (x=1\n)") == (0, 0) # -- inside literal is data
|
||||
assert _sql_paren_depth("WHERE x) AND (y") == (0, -1) # net 0 but went negative
|
||||
@@ -206,41 +206,38 @@ def test_module_level_validate_haproxy_config_forwards_partial_fragment():
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# Wizard Pydantic gate: ACL `-f` flag must be REJECTED at submit.
|
||||
# Issue #38 follow-up: ACL `-f <file>` pattern-file references are
|
||||
# ACCEPTED (the Bulgu #12 hard reject was removed — pattern files are
|
||||
# operator-managed host files, the agent's pre-reload `haproxy -c`
|
||||
# makes a missing file fail safely, and bulk import always accepted
|
||||
# `-f`). These tests pin the ACCEPT behaviour.
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_wizard_pydantic_rejects_acl_with_file_flag():
|
||||
"""Pre-fix the wizard's ACL string validator passed
|
||||
`acl name path -i -m reg -f /path` straight through. Apply-time
|
||||
HAProxy `-c` then failed with "failed to open pattern file".
|
||||
Pin that the validator now rejects `-f` at submit.
|
||||
def test_wizard_pydantic_accepts_acl_with_file_flag():
|
||||
"""Issue #38 follow-up — the wizard's ACL string validator must
|
||||
ACCEPT `-f <file>` pattern-file references (Bulgu #12 reject
|
||||
removed). Operators with large host-managed IP blacklists rely
|
||||
on this in production.
|
||||
"""
|
||||
from models.site_wizard import FrontendStep
|
||||
|
||||
# Minimal valid wizard frontend kwargs — only the offending
|
||||
# acl_rules entry should trigger the failure.
|
||||
fe_kwargs = dict(
|
||||
fe = FrontendStep(
|
||||
name="fe1",
|
||||
mode="http",
|
||||
bind_address="*",
|
||||
bind_port=80,
|
||||
acl_rules=["acl1 path -i -m reg -f /path"],
|
||||
)
|
||||
from pydantic import ValidationError
|
||||
with pytest.raises(ValidationError) as exc_info:
|
||||
FrontendStep(**fe_kwargs)
|
||||
msg = str(exc_info.value)
|
||||
assert "-f" in msg or "pattern-file" in msg.lower(), (
|
||||
f"Bulgu #12 regression: ACL -f flag must be rejected with a "
|
||||
f"clear pattern-file error. Got: {msg}"
|
||||
assert fe.acl_rules == ["acl1 path -i -m reg -f /path"], (
|
||||
"ACL `-f` rule must round-trip verbatim through the wizard model"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"rule",
|
||||
[
|
||||
# Various spacing / position variants the regex must catch.
|
||||
# Various spacing / position variants must all be accepted.
|
||||
"acl1 path -f /etc/haproxy/list",
|
||||
"acl1 path -i -f /tmp/x.lst",
|
||||
"acl1 src -f /etc/haproxy/admins.lst",
|
||||
@@ -249,23 +246,19 @@ def test_wizard_pydantic_rejects_acl_with_file_flag():
|
||||
"acl1 path -f",
|
||||
],
|
||||
)
|
||||
def test_wizard_pydantic_rejects_acl_with_file_flag_variants(rule):
|
||||
"""Every spacing / position variant the operator might type must
|
||||
be rejected. Pinned defensively so the regex never accidentally
|
||||
relaxes to "only matches trailing -f".
|
||||
"""
|
||||
def test_wizard_pydantic_accepts_acl_with_file_flag_variants(rule):
|
||||
"""Every spacing / position variant must be accepted verbatim
|
||||
(Issue #38 follow-up — no `-f` shape may be rejected)."""
|
||||
from models.site_wizard import FrontendStep
|
||||
from pydantic import ValidationError
|
||||
|
||||
fe_kwargs = dict(
|
||||
fe = FrontendStep(
|
||||
name="fe1",
|
||||
mode="http",
|
||||
bind_address="*",
|
||||
bind_port=80,
|
||||
acl_rules=[rule],
|
||||
)
|
||||
with pytest.raises(ValidationError):
|
||||
FrontendStep(**fe_kwargs)
|
||||
assert fe.acl_rules == [rule]
|
||||
|
||||
|
||||
def test_wizard_pydantic_does_not_falsely_match_dash_f_inside_token():
|
||||
@@ -291,42 +284,29 @@ def test_wizard_pydantic_does_not_falsely_match_dash_f_inside_token():
|
||||
assert len(fe.acl_rules) == 3
|
||||
|
||||
|
||||
def test_manual_frontend_validator_rejects_acl_with_file_flag():
|
||||
"""Parity check: the manual Frontend API
|
||||
(`models/frontend.py::validate_acl_rules`) must apply the same
|
||||
`-f` rejection. Operators see consistent behaviour from both the
|
||||
wizard and the per-entity frontend page.
|
||||
def test_manual_frontend_validator_accepts_acl_with_file_flag():
|
||||
"""Parity check (Issue #38 follow-up): the manual Frontend API
|
||||
(`models/frontend.py::validate_acl_rules`) must ACCEPT `-f`
|
||||
pattern-file references, same as the wizard and bulk import.
|
||||
"""
|
||||
from models.frontend import FrontendConfig
|
||||
from pydantic import ValidationError
|
||||
|
||||
with pytest.raises(ValidationError) as exc_info:
|
||||
FrontendConfig(
|
||||
name="fe1",
|
||||
bind_port=80,
|
||||
mode="http",
|
||||
acl_rules=["acl1 path -i -m reg -f /path"],
|
||||
)
|
||||
msg = str(exc_info.value)
|
||||
assert "-f" in msg or "pattern-file" in msg.lower(), (
|
||||
f"Manual frontend API parity regression: ACL -f flag must be "
|
||||
f"rejected. Got: {msg}"
|
||||
fe = FrontendConfig(
|
||||
name="fe1",
|
||||
bind_port=80,
|
||||
mode="http",
|
||||
acl_rules=["acl1 path -i -m reg -f /path"],
|
||||
)
|
||||
assert fe.acl_rules == ["acl1 path -i -m reg -f /path"]
|
||||
|
||||
|
||||
def test_wizard_pydantic_rejects_structured_redirect_dict_with_file_flag():
|
||||
"""Round-3 audit extension — structured redirect dicts (the
|
||||
alternative shape that `models/site_wizard.py::_validate_redirect_rules`
|
||||
accepts alongside legacy strings) also flow through to
|
||||
`services/haproxy_config.py::_format_redirect_rule` and emit
|
||||
their `condition` / `target` verbatim into the rendered HAProxy
|
||||
directive. Without the dict-aware reject the visual builder's
|
||||
`-f` block could be bypassed by hand-crafting a dict payload
|
||||
against the API — recreating the same `failed to open pattern
|
||||
file` failure at apply time.
|
||||
def test_wizard_pydantic_accepts_structured_redirect_dict_with_file_flag():
|
||||
"""Issue #38 follow-up — structured redirect dicts carrying `-f`
|
||||
pattern-file references in `condition`/`target` are ACCEPTED
|
||||
(the Bulgu #12 dict-aware reject was removed together with the
|
||||
string-rule reject).
|
||||
"""
|
||||
from models.site_wizard import FrontendStep, BackendStep
|
||||
from pydantic import ValidationError
|
||||
from models.site_wizard import FrontendStep
|
||||
|
||||
fe_kwargs = dict(
|
||||
name="fe1",
|
||||
@@ -334,37 +314,32 @@ def test_wizard_pydantic_rejects_structured_redirect_dict_with_file_flag():
|
||||
mode="http",
|
||||
)
|
||||
|
||||
# `condition` carrying `-f` must be rejected.
|
||||
with pytest.raises(ValidationError) as exc_info:
|
||||
FrontendStep(
|
||||
**fe_kwargs,
|
||||
redirect_rules=[
|
||||
{
|
||||
"type": "scheme",
|
||||
"target": "https",
|
||||
"condition": "if { src -f /etc/haproxy/admins.lst }",
|
||||
}
|
||||
],
|
||||
)
|
||||
msg = str(exc_info.value)
|
||||
assert "pattern-file" in msg.lower() or "-f" in msg, msg
|
||||
# `condition` carrying `-f` is accepted.
|
||||
fe = FrontendStep(
|
||||
**fe_kwargs,
|
||||
redirect_rules=[
|
||||
{
|
||||
"type": "scheme",
|
||||
"target": "https",
|
||||
"condition": "if { src -f /etc/haproxy/admins.lst }",
|
||||
}
|
||||
],
|
||||
)
|
||||
assert fe.redirect_rules[0]["condition"] == "if { src -f /etc/haproxy/admins.lst }"
|
||||
|
||||
# `target` carrying `-f` must also be rejected (defence-in-depth
|
||||
# for hand-crafted payloads).
|
||||
with pytest.raises(ValidationError) as exc_info:
|
||||
FrontendStep(
|
||||
**fe_kwargs,
|
||||
redirect_rules=[
|
||||
{
|
||||
"type": "location",
|
||||
"target": "/foo -f /tmp/x.lst",
|
||||
}
|
||||
],
|
||||
)
|
||||
msg = str(exc_info.value)
|
||||
assert "pattern-file" in msg.lower() or "-f" in msg, msg
|
||||
# `target` carrying `-f` is accepted too.
|
||||
fe = FrontendStep(
|
||||
**fe_kwargs,
|
||||
redirect_rules=[
|
||||
{
|
||||
"type": "location",
|
||||
"target": "/foo -f /tmp/x.lst",
|
||||
}
|
||||
],
|
||||
)
|
||||
assert fe.redirect_rules[0]["target"] == "/foo -f /tmp/x.lst"
|
||||
|
||||
# Clean structured dict still passes — no false positive.
|
||||
# Clean structured dict still passes.
|
||||
FrontendStep(
|
||||
**fe_kwargs,
|
||||
redirect_rules=[
|
||||
@@ -667,8 +642,8 @@ def test_user_reported_wizard_config_emits_no_false_warnings():
|
||||
zero WARNINGs from the directives we expanded.
|
||||
"""
|
||||
# Distilled from the user's bulk-site-create snapshot, minus the
|
||||
# `-f` ACL (which the new Pydantic gate rejects before this
|
||||
# validator ever runs).
|
||||
# `-f` ACL (accepted since the Issue #38 follow-up, but irrelevant
|
||||
# to the directive-expansion warnings this test pins).
|
||||
config = """# ─── Wizard candidate fragment (dry-run preview) ───
|
||||
frontend fe-site1
|
||||
bind *:80
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
"""v1.7.6 — guard: the agent must surface keepalived FAULT state.
|
||||
|
||||
A VIP whose interface has no usable IPv4 (or whose track-script fails) puts keepalived into
|
||||
FAULT — the virtual IP is NOT held. Previously get_keepalive_state only grepped (MASTER|BACKUP),
|
||||
so a FAULT'd VIP reported as BACKUP — misleading (looks healthy-ish). It must report FAULT so the
|
||||
UI shows it red. Both get_keepalive_state copies (installer + SKIP_TO_DAEMON daemon) must include
|
||||
FAULT. Pure source guard."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def test_keepalive_state_detects_fault_in_both_copies():
|
||||
with open(os.path.join(ROOT, "utils", "agent_scripts", "linux_install.sh"), encoding="utf-8") as f:
|
||||
s = f.read()
|
||||
# FAULT added to the state grep in both copies (installer + daemon), both detection methods.
|
||||
assert s.count("MASTER|BACKUP|FAULT") >= 2
|
||||
# and the misleading MASTER|BACKUP-only grep is gone.
|
||||
assert 'grep -oE "(MASTER|BACKUP)"' not in s
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Issue #27 (v1.7.0) — unit tests for the keepalived config generator + secret crypto.
|
||||
|
||||
Pure-function tests; no DB. Validates the rendered keepalived.conf for MASTER/BACKUP,
|
||||
unicast peers, the failover weight arithmetic, the script-security requirements, the
|
||||
ownership marker, and Fernet round-trip.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from services import keepalived_config as kc # noqa: E402
|
||||
|
||||
|
||||
VIP = {
|
||||
"id": 3, "name": "web-vip", "virtual_ip": "10.0.0.100", "prefix_length": 24,
|
||||
"virtual_router_id": 51, "advert_int": 1, "use_unicast": True, "track_haproxy": True,
|
||||
}
|
||||
MEMBERS = [
|
||||
{"role": "MASTER", "priority": 150, "network_interface": "eth0", "agent_id": 1, "ip_address": "10.0.0.11"},
|
||||
{"role": "BACKUP", "priority": 100, "network_interface": "eth0", "agent_id": 2, "ip_address": "10.0.0.12"},
|
||||
]
|
||||
|
||||
|
||||
def _render(this_idx, auth="s3cr3t"):
|
||||
this_agent = MEMBERS[this_idx]
|
||||
peers = [m["ip_address"] for m in MEMBERS if m["agent_id"] != this_agent["agent_id"]]
|
||||
return kc.render_keepalived_conf(vip=VIP, members=MEMBERS, this_agent=this_agent,
|
||||
peer_ips=peers, auth_pass_plain=auth)
|
||||
|
||||
|
||||
class TestRender:
|
||||
def test_master_state_priority_iface_vrid(self):
|
||||
conf = _render(0)
|
||||
assert "state MASTER" in conf
|
||||
assert "priority 150" in conf
|
||||
assert "interface eth0" in conf
|
||||
assert "virtual_router_id 51" in conf
|
||||
assert "10.0.0.100/24 dev eth0" in conf
|
||||
|
||||
def test_backup_state(self):
|
||||
conf = _render(1)
|
||||
assert "state BACKUP" in conf
|
||||
assert "priority 100" in conf
|
||||
|
||||
def test_unicast_peers(self):
|
||||
# MASTER's config lists the BACKUP as its unicast peer (and its own src ip).
|
||||
conf = _render(0)
|
||||
assert "unicast_src_ip 10.0.0.11" in conf
|
||||
assert "unicast_peer" in conf
|
||||
assert "10.0.0.12" in conf
|
||||
|
||||
def test_script_security_block(self):
|
||||
conf = _render(0)
|
||||
assert "enable_script_security" in conf
|
||||
assert "script_user root" in conf
|
||||
|
||||
def test_ownership_marker(self):
|
||||
assert kc.OWNERSHIP_MARKER in _render(0)
|
||||
|
||||
def test_weight_makes_failed_master_lose(self):
|
||||
# On HAProxy failure the master's effective priority must drop below the backup.
|
||||
conf = _render(0)
|
||||
weight_line = [l for l in conf.splitlines() if l.strip().startswith("weight ")][0]
|
||||
weight = int(weight_line.strip().split()[1])
|
||||
assert 150 + weight < 100, "failed master must fall below every backup"
|
||||
assert "track_script" in conf and "chk_haproxy" in conf
|
||||
|
||||
def test_track_disabled_omits_script(self):
|
||||
vip = {**VIP, "track_haproxy": False}
|
||||
conf = kc.render_keepalived_conf(vip=vip, members=MEMBERS, this_agent=MEMBERS[0],
|
||||
peer_ips=["10.0.0.12"], auth_pass_plain=None)
|
||||
assert "vrrp_script" not in conf
|
||||
assert "track_script" not in conf
|
||||
|
||||
def test_no_auth_when_secret_absent(self):
|
||||
conf = _render(0, auth=None)
|
||||
assert "auth_pass" not in conf
|
||||
|
||||
def test_multicast_omits_unicast(self):
|
||||
vip = {**VIP, "use_unicast": False}
|
||||
conf = kc.render_keepalived_conf(vip=vip, members=MEMBERS, this_agent=MEMBERS[0],
|
||||
peer_ips=["10.0.0.12"], auth_pass_plain="x")
|
||||
assert "unicast_src_ip" not in conf
|
||||
assert "unicast_peer" not in conf
|
||||
|
||||
def test_single_node_omits_unicast_block(self):
|
||||
# Single-node VIP (no peers): even with use_unicast=True we must NOT emit a bare
|
||||
# `unicast_src_ip`/`unicast_peer` — keepalived treats a unicast keyword with no peers
|
||||
# as deprecated, warns, and falls back to multicast (and `keepalived -t` flags it).
|
||||
# Omitting the block yields a clean multicast config that holds the VIP solo.
|
||||
only = [{"role": "MASTER", "priority": 150, "network_interface": "eth0",
|
||||
"agent_id": 1, "ip_address": "10.0.0.11"}]
|
||||
conf = kc.render_keepalived_conf(vip=VIP, members=only, this_agent=only[0],
|
||||
peer_ips=[], auth_pass_plain=None)
|
||||
assert "unicast_src_ip" not in conf
|
||||
assert "unicast_peer" not in conf
|
||||
assert "state MASTER" in conf
|
||||
assert "10.0.0.100/24 dev eth0" in conf
|
||||
|
||||
|
||||
class TestCheckScript:
|
||||
def test_default_process_name(self):
|
||||
s = kc.build_haproxy_check_script()
|
||||
assert "pidof haproxy" in s
|
||||
assert kc.OWNERSHIP_MARKER in s
|
||||
|
||||
def test_process_name_from_bin_path(self):
|
||||
s = kc.build_haproxy_check_script(bin_path="/opt/hap/sbin/haproxy-ent")
|
||||
assert "pidof haproxy-ent" in s
|
||||
|
||||
def test_malicious_bin_path_falls_back(self):
|
||||
s = kc.build_haproxy_check_script(bin_path="/x/haproxy; rm -rf /")
|
||||
assert "rm -rf" not in s
|
||||
assert "pidof haproxy" in s
|
||||
|
||||
|
||||
class TestSecretCrypto:
|
||||
def test_roundtrip(self):
|
||||
kc.reset_fernet_for_tests()
|
||||
token = kc.encrypt_vrrp_secret("s3cr3t")
|
||||
assert token != "s3cr3t"
|
||||
assert kc.decrypt_vrrp_secret(token) == "s3cr3t"
|
||||
|
||||
def test_decrypt_garbage_returns_none(self):
|
||||
kc.reset_fernet_for_tests()
|
||||
assert kc.decrypt_vrrp_secret("not-a-fernet-token") is None
|
||||
@@ -0,0 +1,223 @@
|
||||
"""Regression tests for the 2026-07 security advisories.
|
||||
|
||||
Covers:
|
||||
- GHSA-7rhv-c5pc-69r8 (CRITICAL RCE): agent script-template management must
|
||||
require the agents.version permission, not merely authentication.
|
||||
- GHSA-3p5c-m5m4-mjpx (missing auth): agent data-plane endpoints must require a
|
||||
valid X-API-Key, and operator/UI endpoints must require a JWT. An anonymous
|
||||
caller must never get a 200 with sensitive data.
|
||||
|
||||
These are behavioral assertions via FastAPI's TestClient. The auth checks were
|
||||
deliberately moved ahead of any DB access, so an unauthenticated request is
|
||||
rejected without needing a database — the same approach as the existing
|
||||
test_ssl_list_endpoint_auth.py. Accepted rejection statuses are 401/403/422
|
||||
(never 200-with-data).
|
||||
"""
|
||||
import re
|
||||
import os
|
||||
import pytest
|
||||
|
||||
REJECT = (401, 403, 422)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-3p5c: agent data-plane endpoints must reject a missing X-API-Key
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_agent_config_requires_api_key(client):
|
||||
"""GET /api/agents/{name}/config leaked the full haproxy.cfg without a key."""
|
||||
res = client.get("/api/agents/prod-haproxy-1/config")
|
||||
assert res.status_code in REJECT, (
|
||||
f"GHSA-3p5c regression: agent config served without X-API-Key ({res.status_code})"
|
||||
)
|
||||
|
||||
|
||||
def test_agent_ssl_certificates_requires_api_key(client):
|
||||
"""GET /api/agents/{name}/ssl-certificates leaked SSL private keys without a key."""
|
||||
res = client.get("/api/agents/prod-haproxy-1/ssl-certificates")
|
||||
assert res.status_code in REJECT, (
|
||||
f"GHSA-3p5c regression: SSL certs (private keys!) served without X-API-Key ({res.status_code})"
|
||||
)
|
||||
if res.status_code == 200:
|
||||
assert "private_key_content" not in res.text
|
||||
|
||||
|
||||
def test_agent_upgrade_status_requires_api_key(client):
|
||||
res = client.get("/api/agents/prod-haproxy-1/upgrade-status")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_agent_pending_requests_requires_api_key(client):
|
||||
res = client.get("/api/configuration/agents/prod-haproxy-1/pending-requests")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_agent_heartbeat_by_name_requires_api_key(client):
|
||||
res = client.post("/api/agents/heartbeat", json={"name": "rogue-poc"})
|
||||
assert res.status_code in REJECT, (
|
||||
f"GHSA-3p5c regression: keyless heartbeat/auto-register accepted ({res.status_code})"
|
||||
)
|
||||
|
||||
|
||||
def test_agent_heartbeat_by_id_requires_api_key(client):
|
||||
res = client.post("/api/agents/1/heartbeat", json={"name": "spoofed"})
|
||||
assert res.status_code in REJECT, (
|
||||
f"GHSA-3p5c regression: keyless by-id heartbeat state-spoof accepted ({res.status_code})"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-3p5c: operator/UI endpoints must reject a missing JWT
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_agents_inventory_requires_jwt(client):
|
||||
"""GET /api/agents (RCE read-back channel) was served without a JWT."""
|
||||
res = client.get("/api/agents")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", ["/api/health/deep", "/api/health/agents", "/api/health/clusters"])
|
||||
def test_detailed_health_requires_jwt(client, path):
|
||||
res = client.get(path)
|
||||
assert res.status_code in REJECT, f"{path} served without a JWT ({res.status_code})"
|
||||
|
||||
|
||||
def test_simple_health_stays_public(client):
|
||||
"""The liveness probe endpoint (/api/health) must remain UNAUTHENTICATED.
|
||||
|
||||
It reports 200 when healthy and 503 when the DB is unreachable (as in this
|
||||
no-DB test env); what matters for the k8s probe is that it is never gated
|
||||
behind auth (401/403). We only added auth to /api/health/{deep,agents,clusters}.
|
||||
"""
|
||||
res = client.get("/api/health")
|
||||
assert res.status_code not in (401, 403), (
|
||||
f"Regression: /api/health liveness probe now requires auth ({res.status_code}) — "
|
||||
f"this breaks k8s liveness/readiness"
|
||||
)
|
||||
|
||||
|
||||
def test_dashboard_stats_requires_jwt(client):
|
||||
res = client.get("/api/dashboard-stats/stats?cluster_id=1")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_ssl_config_versions_requires_jwt(client):
|
||||
res = client.get("/api/ssl/certificates/1/config-versions")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-7rhv (CRITICAL RCE): script-template management
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_script_template_write_requires_auth(client):
|
||||
"""Anonymous POST must be rejected outright."""
|
||||
res = client.post("/api/agents/script-templates/linux",
|
||||
json={"script_content": "#!/bin/bash\nid", "version": "9.9.9"})
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_script_template_read_requires_auth(client):
|
||||
res = client.get("/api/agents/script-templates/linux")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-3p5c (round 2): sibling endpoints exposing the SAME class of data
|
||||
# (found during post-merge review — must also require a JWT)
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/haproxy-cluster-pools/1/agents", # full agent inventory — same class as GET /api/agents
|
||||
"/api/pools",
|
||||
"/api/haproxy-cluster-pools",
|
||||
"/api/dashboard/stats",
|
||||
"/api/dashboard/overview", # optional-auth pattern — leaked stats/names/alerts anonymously
|
||||
"/api/haproxy/stats",
|
||||
"/api/waf/rules",
|
||||
"/api/health/errors",
|
||||
])
|
||||
def test_sibling_inventory_endpoints_require_jwt(client, path):
|
||||
res = client.get(path)
|
||||
assert res.status_code in REJECT, (
|
||||
f"GHSA-3p5c (round 2) regression: {path} served without a JWT ({res.status_code}) — "
|
||||
f"anonymous access to inventory/topology/WAF/error data"
|
||||
)
|
||||
|
||||
|
||||
def test_pool_agents_no_anonymous_inventory_leak(client):
|
||||
"""The richest bypass: /api/haproxy-cluster-pools/{id}/agents must not leak inventory."""
|
||||
res = client.get("/api/haproxy-cluster-pools/1/agents")
|
||||
assert res.status_code in REJECT
|
||||
if res.status_code == 200:
|
||||
assert "ip_address" not in res.text and "hostname" not in res.text
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-3p5c (round 2): agent webhooks must return 401 (not a 200 error body)
|
||||
# for anonymous callers — the auth raise must propagate, not be swallowed.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/agents/some-agent/config-applied",
|
||||
"/api/agents/some-agent/config-validation-failed",
|
||||
"/api/agents/some-agent/config-sync",
|
||||
])
|
||||
def test_agent_webhooks_reject_anonymous_with_401(client, path):
|
||||
res = client.post(path, json={})
|
||||
assert res.status_code in REJECT, (
|
||||
f"{path} returned {res.status_code} for an anonymous caller — the auth "
|
||||
f"rejection must be a 401/403, not a swallowed 200 error body"
|
||||
)
|
||||
# Specifically must NOT be a 200 "status: error" body.
|
||||
assert res.status_code != 200
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# GHSA-3p5c (round 2): GET /api/agents must accept EITHER a JWT OR an agent
|
||||
# X-API-Key. Anonymous (neither) is still rejected — agents send a key, so a
|
||||
# JWT-only gate would break them (verified end-to-end in the localtest smoke).
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_agents_inventory_still_rejects_fully_anonymous(client):
|
||||
"""No JWT and no X-API-Key -> 401 (the agent-key accept path needs a valid key)."""
|
||||
res = client.get("/api/agents")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_generate_uninstall_script_requires_auth(client):
|
||||
"""Agent-management endpoint must not be anonymously reachable (JWT or agent key)."""
|
||||
res = client.get("/api/agents/generate-uninstall-script/linux")
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/config/validate",
|
||||
"/api/config/optimize",
|
||||
"/api/config/templates/default/generate",
|
||||
])
|
||||
def test_config_compute_endpoints_require_auth(client, path):
|
||||
"""Config compute endpoints (run a HAProxy validator on caller input) were
|
||||
optional-auth; now require a JWT. The dependency rejects before body parsing."""
|
||||
res = client.post(path, json={})
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
def test_script_template_write_enforces_agents_version_permission():
|
||||
"""Static guarantee: the write handler checks agents.version (not just authN).
|
||||
|
||||
A behavioral 403-for-viewer test would need a seeded DB + a minted viewer JWT;
|
||||
instead we assert the permission gate is present in source, mirroring the
|
||||
existing audit-style source tests. This is the core RCE fix (GHSA-7rhv).
|
||||
"""
|
||||
src_path = os.path.join(os.path.dirname(__file__), "..", "routers", "agent.py")
|
||||
with open(src_path, "r") as f:
|
||||
src = f.read()
|
||||
# Isolate the save_agent_script_template handler body.
|
||||
m = re.search(r"async def save_agent_script_template\(.*?\n(.*?)\n@router\.", src, re.DOTALL)
|
||||
assert m, "save_agent_script_template handler not found"
|
||||
body = m.group(1)
|
||||
assert 'check_user_permission' in body and '"agents", "version"' in body, (
|
||||
"GHSA-7rhv regression: script-template WRITE no longer enforces the "
|
||||
"agents.version permission — any JWT holder could poison the root install script"
|
||||
)
|
||||
@@ -276,14 +276,16 @@ def test_categorize_routes_directives_correctly():
|
||||
|
||||
def test_emit_buckets_flushed_in_canonical_order():
|
||||
"""The flush block at end of frontend processing must list buckets
|
||||
in: prelude → stick → tcp_req → acl → http_req → http_resp →
|
||||
in: prelude → filter → stick → tcp_req → acl → http_req → http_resp →
|
||||
redirect → use_be → default_be. Pre-fix `http-request` rules
|
||||
interleaved with `use_backend` rules in source order, producing
|
||||
HAProxy parser warnings."""
|
||||
HAProxy parser warnings. (Issue #38 added the `filter` bucket, flushed
|
||||
right after `prelude` so SPOE `filter` lines precede `send-spoe-group`.)"""
|
||||
src = _gen_src()
|
||||
flush_match = re.search(
|
||||
r'for\s+_bucket_key\s+in\s+\(\s*'
|
||||
r'"prelude"\s*,\s*'
|
||||
r'"filter"\s*,\s*'
|
||||
r'"stick"\s*,\s*'
|
||||
r'"tcp_req"\s*,\s*'
|
||||
r'"acl"\s*,\s*'
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
"""
|
||||
Issue #38 regression tests: HAProxy SPOE `filter` + frontend `log-format` support.
|
||||
|
||||
Bug: the bulk-config parser recognised only a fixed set of frontend directives,
|
||||
so `filter spoe engine coraza config ...` and `log-format ...` were silently
|
||||
dropped on import / manual edit. This regenerated a config missing the SPOE
|
||||
engine definition, so HAProxy failed with
|
||||
"unable to find SPOE engine 'coraza' used by the send-spoe-group 'coraza-req'".
|
||||
|
||||
These tests verify the end-to-end fix without requiring a database:
|
||||
1. parser captures `filter` + `log-format` into the new ParsedFrontend fields;
|
||||
2. `http-request send-spoe-group` is still preserved (regression guard);
|
||||
3. the generator's directive categoriser + bucket flush order emit `filter`
|
||||
BEFORE the `http-request send-spoe-group` rules and keep `log-format`;
|
||||
4. reject/rollback restores the new columns;
|
||||
5. a non-SPOE frontend is completely unaffected (zero-impact).
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from utils.haproxy_config_parser import parse_haproxy_config, ParsedFrontend
|
||||
from services.haproxy_config import _categorize_haproxy_directive
|
||||
from models.frontend import FrontendConfig
|
||||
|
||||
|
||||
# The exact frontend/backend config reported in Issue #38 (Coraza-SPOA).
|
||||
ISSUE_38_CONFIG = r"""
|
||||
frontend web-frontend
|
||||
bind *:8073
|
||||
mode http
|
||||
log-format "%ci:%cp\ [%t]\ %ft\ %b/%s\ %ST\ %B\ %{+Q}r\ %[var(txn.coraza.id)]\ waf-hit:\ %[var(txn.coraza.fail)]"
|
||||
filter spoe engine coraza config /etc/haproxy/coraza.cfg
|
||||
http-request set-var(txn.coraza.app) str(haproxy_waf)
|
||||
http-request send-spoe-group coraza coraza-req
|
||||
http-request deny if { var(txn.coraza.fail) -m int eq 1 }
|
||||
default_backend web-backend
|
||||
|
||||
backend web-backend
|
||||
balance roundrobin
|
||||
mode http
|
||||
server server1 192.168.1.10:443 weight 100 ssl verify none
|
||||
|
||||
backend coraza-spoa
|
||||
mode tcp
|
||||
option spop-check
|
||||
server coraza_spoa 192.168.12.21:9000
|
||||
"""
|
||||
|
||||
|
||||
def _get_frontend(parse_result, name):
|
||||
for fe in parse_result.frontends:
|
||||
if fe.name == name:
|
||||
return fe
|
||||
return None
|
||||
|
||||
|
||||
class TestParserCapturesSpoe:
|
||||
def test_filter_and_log_format_captured(self):
|
||||
result = parse_haproxy_config(ISSUE_38_CONFIG)
|
||||
fe = _get_frontend(result, "web-frontend")
|
||||
assert fe is not None, "web-frontend should be parsed and kept"
|
||||
assert fe.filters is not None
|
||||
assert "filter spoe engine coraza config /etc/haproxy/coraza.cfg" in fe.filters
|
||||
assert fe.log_format is not None
|
||||
assert fe.log_format.startswith("log-format")
|
||||
# the escaped/quoted format string must be preserved verbatim
|
||||
assert "%[var(txn.coraza.fail)]" in fe.log_format
|
||||
|
||||
def test_send_spoe_group_still_preserved(self):
|
||||
# Regression guard: http-request rules (incl. send-spoe-group) must
|
||||
# still be collected into request_headers as before.
|
||||
result = parse_haproxy_config(ISSUE_38_CONFIG)
|
||||
fe = _get_frontend(result, "web-frontend")
|
||||
assert fe.request_headers is not None
|
||||
assert "send-spoe-group coraza coraza-req" in fe.request_headers
|
||||
|
||||
def test_multiple_filters_preserved_in_order(self):
|
||||
cfg = """
|
||||
frontend f1
|
||||
bind *:80
|
||||
mode http
|
||||
filter compression
|
||||
filter spoe engine coraza config /etc/haproxy/coraza.cfg
|
||||
default_backend b1
|
||||
|
||||
backend b1
|
||||
mode http
|
||||
server s1 10.0.0.1:80
|
||||
"""
|
||||
fe = _get_frontend(parse_haproxy_config(cfg), "f1")
|
||||
lines = fe.filters.split("\n")
|
||||
assert lines == [
|
||||
"filter compression",
|
||||
"filter spoe engine coraza config /etc/haproxy/coraza.cfg",
|
||||
]
|
||||
|
||||
def test_log_format_sd_variant_captured(self):
|
||||
cfg = """
|
||||
frontend f1
|
||||
bind *:80
|
||||
mode http
|
||||
log-format-sd "[exampleSDID@1234 field=value]"
|
||||
default_backend b1
|
||||
|
||||
backend b1
|
||||
mode http
|
||||
server s1 10.0.0.1:80
|
||||
"""
|
||||
fe = _get_frontend(parse_haproxy_config(cfg), "f1")
|
||||
assert fe.log_format is not None
|
||||
assert fe.log_format.startswith("log-format-sd")
|
||||
|
||||
|
||||
class TestGeneratorOrderingContract:
|
||||
"""The generator routes directives into ordered buckets. Verify SPOE
|
||||
correctness at the (pure) categoriser + documented flush-order level."""
|
||||
|
||||
def test_filter_routes_to_filter_bucket(self):
|
||||
assert _categorize_haproxy_directive(" filter spoe engine coraza config /x.cfg") == "filter"
|
||||
|
||||
def test_send_spoe_group_routes_to_http_req(self):
|
||||
assert _categorize_haproxy_directive(" http-request send-spoe-group coraza coraza-req") == "http_req"
|
||||
|
||||
def test_log_format_routes_to_prelude(self):
|
||||
assert _categorize_haproxy_directive(' log-format "%ci:%cp"') == "prelude"
|
||||
assert _categorize_haproxy_directive(' log-format-sd "[x]"') == "prelude"
|
||||
|
||||
def test_flush_order_places_filter_before_http_req(self):
|
||||
# The bucket flush order is the single source of truth for emission
|
||||
# ordering. Assert `filter` is flushed before `http_req` (and after
|
||||
# `prelude`), guaranteeing `filter ...` renders before
|
||||
# `http-request send-spoe-group ...`.
|
||||
src = _read_source("services/haproxy_config.py")
|
||||
m = re.search(r"for _bucket_key in \((.*?)\):", src, re.DOTALL)
|
||||
assert m, "bucket flush loop not found"
|
||||
order = re.findall(r'"(\w+)"', m.group(1))
|
||||
assert "filter" in order, "new 'filter' bucket missing from flush order"
|
||||
assert order.index("prelude") < order.index("filter") < order.index("http_req")
|
||||
|
||||
|
||||
class TestModelAndRollback:
|
||||
def test_model_has_passthrough_fields(self):
|
||||
fc = FrontendConfig(
|
||||
name="f", bind_port=80,
|
||||
filters="filter spoe engine coraza config /etc/haproxy/coraza.cfg",
|
||||
log_format='log-format "%ci"',
|
||||
)
|
||||
assert fc.filters.startswith("filter spoe")
|
||||
assert fc.log_format.startswith("log-format")
|
||||
|
||||
def test_dataclass_defaults_none(self):
|
||||
fe = ParsedFrontend(name="f")
|
||||
assert fe.filters is None
|
||||
assert fe.log_format is None
|
||||
|
||||
def test_rollback_restores_new_columns(self):
|
||||
# Reject/rollback of a frontend UPDATE must restore the new columns,
|
||||
# otherwise the rejected (new) filters/log_format would persist.
|
||||
src = _read_source("utils/entity_snapshot.py")
|
||||
assert "log_format = $" in src
|
||||
assert "filters = $" in src
|
||||
assert "old_values.get('log_format')" in src
|
||||
assert "old_values.get('filters')" in src
|
||||
|
||||
|
||||
class TestZeroImpact:
|
||||
def test_non_spoe_frontend_unaffected(self):
|
||||
cfg = """
|
||||
frontend plain
|
||||
bind *:80
|
||||
mode http
|
||||
option httplog
|
||||
default_backend b1
|
||||
|
||||
backend b1
|
||||
mode http
|
||||
server s1 10.0.0.1:80
|
||||
"""
|
||||
fe = _get_frontend(parse_haproxy_config(cfg), "plain")
|
||||
# No filter / log-format present → new fields stay None (no behaviour change)
|
||||
assert fe.filters is None
|
||||
assert fe.log_format is None
|
||||
|
||||
def test_spop_check_backend_roundtrips_without_warning(self):
|
||||
result = parse_haproxy_config(ISSUE_38_CONFIG)
|
||||
be = next((b for b in result.backends if b.name == "coraza-spoa"), None)
|
||||
assert be is not None, "coraza-spoa backend should import"
|
||||
assert be.mode == "tcp"
|
||||
assert be.options and "option spop-check" in be.options
|
||||
# spop-check is now a known option → no spurious 'unknown option' warning
|
||||
assert not any(
|
||||
"coraza-spoa" in w and "spop-check" in w and "Unknown" in w
|
||||
for w in result.warnings
|
||||
)
|
||||
|
||||
|
||||
def _read_source(relpath):
|
||||
base = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
with open(os.path.join(base, relpath), "r", encoding="utf-8") as fh:
|
||||
return fh.read()
|
||||
@@ -0,0 +1,68 @@
|
||||
"""Unit tests for the SSRF guard (GHSA-3vh4-gvxx-wm2p).
|
||||
|
||||
The guard protects server-side fetches of ACME `directory_url` values. This
|
||||
project uses only public ACME CAs, so every non-public IP must be rejected.
|
||||
Tests avoid real network/DNS by using IP literals and scheme checks.
|
||||
"""
|
||||
import asyncio
|
||||
import pytest
|
||||
|
||||
from utils.ssrf_guard import is_public_ip, assert_public_url, SSRFValidationError
|
||||
|
||||
|
||||
# ---- is_public_ip -----------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("ip", [
|
||||
"8.8.8.8", "1.1.1.1", "93.184.216.34", # public
|
||||
])
|
||||
def test_public_ips_allowed(ip):
|
||||
assert is_public_ip(ip) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("ip", [
|
||||
"127.0.0.1", # loopback
|
||||
"10.0.0.5", # RFC1918
|
||||
"172.19.0.1", # RFC1918 (the SSRF PoC docker gateway)
|
||||
"192.168.1.1", # RFC1918
|
||||
"169.254.169.254", # link-local / cloud metadata
|
||||
"0.0.0.0", # unspecified
|
||||
"::1", # IPv6 loopback
|
||||
"fe80::1", # IPv6 link-local
|
||||
"::ffff:127.0.0.1", # IPv4-mapped IPv6 loopback (R18c bypass)
|
||||
"::ffff:169.254.169.254", # IPv4-mapped metadata
|
||||
"not-an-ip", # garbage
|
||||
])
|
||||
def test_non_public_ips_rejected(ip):
|
||||
assert is_public_ip(ip) is False
|
||||
|
||||
|
||||
# ---- assert_public_url ------------------------------------------------------
|
||||
|
||||
def _raises(url):
|
||||
with pytest.raises(SSRFValidationError):
|
||||
asyncio.run(assert_public_url(url))
|
||||
|
||||
|
||||
def test_rejects_non_https_scheme():
|
||||
# The SSRF PoC used http:// against an internal listener.
|
||||
_raises("http://172.19.0.1:2121/internal-secret")
|
||||
_raises("http://8.8.8.8/") # even a public IP over http is refused
|
||||
_raises("file:///etc/passwd")
|
||||
_raises("gopher://8.8.8.8/")
|
||||
|
||||
|
||||
def test_rejects_private_ip_literals():
|
||||
_raises("https://127.0.0.1/")
|
||||
_raises("https://10.0.0.5/")
|
||||
_raises("https://169.254.169.254/latest/meta-data/")
|
||||
_raises("https://[::1]/")
|
||||
|
||||
|
||||
def test_rejects_empty_or_hostless():
|
||||
_raises("")
|
||||
_raises("https://")
|
||||
|
||||
|
||||
def test_allows_public_ip_literal_https():
|
||||
# A public IP literal over https must pass (no DNS needed).
|
||||
asyncio.run(assert_public_url("https://8.8.8.8/directory"))
|
||||
@@ -0,0 +1,71 @@
|
||||
"""v1.8.7: app version is single-source and cannot silently drift.
|
||||
|
||||
The UI shows the version via GET /api/version, which returns main.py's `_version_info`. That MUST be
|
||||
sourced from the one canonical file backend/version.json (co-located with main.py so `COPY . .`
|
||||
bakes it into every image, regardless of pipeline). main.py's in-code fallback must NOT be a real
|
||||
version, otherwise it drifts when version.json is bumped but the constant is forgotten — exactly
|
||||
what left the UI reporting 1.8.4 after 1.8.5/1.8.6 shipped. These checks fail loudly on regression.
|
||||
"""
|
||||
import ast
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
_BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) # backend/
|
||||
_VERSION_JSON = os.path.join(_BACKEND, "version.json")
|
||||
_MAIN = os.path.join(_BACKEND, "main.py")
|
||||
_SEMVER = re.compile(r"^\d+\.\d+\.\d+$")
|
||||
|
||||
|
||||
def test_canonical_version_file_exists_and_valid():
|
||||
assert os.path.exists(_VERSION_JSON), "backend/version.json (single source of truth) is missing"
|
||||
with open(_VERSION_JSON) as f:
|
||||
data = json.load(f)
|
||||
assert _SEMVER.match(data.get("version", "")), \
|
||||
f"backend/version.json version is not semver: {data.get('version')!r}"
|
||||
assert data.get("releaseName"), "backend/version.json must have a releaseName"
|
||||
|
||||
|
||||
def _main_fallback_version_info():
|
||||
"""The literal dict assigned to _version_info in main.py (the in-code fallback)."""
|
||||
with open(_MAIN) as f:
|
||||
tree = ast.parse(f.read())
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Assign) and isinstance(node.value, ast.Dict):
|
||||
for t in node.targets:
|
||||
if isinstance(t, ast.Name) and t.id == "_version_info":
|
||||
return ast.literal_eval(node.value)
|
||||
return None
|
||||
|
||||
|
||||
def test_main_has_no_hardcoded_real_version():
|
||||
fb = _main_fallback_version_info()
|
||||
assert fb is not None, "could not find the _version_info fallback literal in main.py"
|
||||
# Must be a neutral marker, never a real version that can drift out of sync.
|
||||
assert not _SEMVER.match(str(fb.get("version", ""))), (
|
||||
f"main.py hardcodes a real version {fb.get('version')!r}; it must be a neutral marker "
|
||||
f"(e.g. 'unknown') so the version stays single-source in backend/version.json"
|
||||
)
|
||||
|
||||
|
||||
def test_main_loads_the_canonical_file_first():
|
||||
# The first candidate path main.py reads must resolve to the co-located backend/version.json,
|
||||
# so the correct version is available in every image (not dependent on CI staging).
|
||||
with open(_MAIN) as f:
|
||||
src = f.read()
|
||||
assert 'os.path.dirname(__file__), "version.json"' in src, \
|
||||
"main.py must read version.json co-located with the module (backend/version.json)"
|
||||
|
||||
|
||||
def test_frontend_package_json_matches_when_present():
|
||||
# Only meaningful in a full-repo checkout; the backend image build context (./backend) has no frontend/.
|
||||
pkg = os.path.join(os.path.dirname(_BACKEND), "frontend", "package.json")
|
||||
if not os.path.exists(pkg):
|
||||
pytest.skip("frontend/package.json not in this context (e.g. backend-only image build)")
|
||||
with open(_VERSION_JSON) as f:
|
||||
canonical = json.load(f)["version"]
|
||||
with open(pkg) as f:
|
||||
fe = json.load(f)["version"]
|
||||
assert fe == canonical, f"frontend/package.json {fe!r} != backend/version.json {canonical!r}"
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Issue #27 safety (v1.7.2) — source-level guards for APPROVAL-GATED VIP deletion.
|
||||
|
||||
The critical invariant: an agent must NEVER tear a VIP down without an explicit human approval.
|
||||
We enforce that by keeping the VIP is_active=TRUE (so the agent keeps serving it) when a delete
|
||||
is merely *requested*; only an APPROVE (apply) flips is_active=FALSE. These guards lock in that
|
||||
wiring so a future edit can't silently make delete immediate again. Pure (no DB)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _read(rel: str) -> str:
|
||||
with open(os.path.join(ROOT, rel), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_delete_stages_for_approval_keeps_vip_running():
|
||||
s = _read("routers/vip.py")
|
||||
# A running (applied) VIP's delete sets pending_delete=TRUE and stages a delete version —
|
||||
# it must NOT flip is_active=FALSE in delete_vip (that only happens on approval in apply_vip).
|
||||
assert "pending_delete=TRUE" in s
|
||||
assert '_stage_vip_version(conn, vip_id, "delete"' in s
|
||||
# The staged-delete branch (running VIP) must not contain an is_active=FALSE soft-delete;
|
||||
# only the "never applied" branch may remove immediately (guarded by applied_snapshot IS NULL).
|
||||
assert 'v["applied_snapshot"] is None' in s
|
||||
|
||||
|
||||
def test_apply_performs_delete_only_on_approval():
|
||||
s = _read("routers/vip.py")
|
||||
# apply_vip short-circuits on pending_delete and only THEN flips is_active=FALSE.
|
||||
assert 'if v["pending_delete"]:' in s
|
||||
assert "is_active=FALSE, pending_delete=FALSE" in s
|
||||
|
||||
|
||||
def test_reject_cancels_delete_as_noop():
|
||||
s = _read("routers/vip.py")
|
||||
# reject_vip clears the staged delete + purge intent; the VIP keeps running (is_active never
|
||||
# touched here) — the "nothing happened" path.
|
||||
assert "pending_delete=FALSE, purge_on_teardown=FALSE" in s
|
||||
assert "Deletion rejected" in s
|
||||
|
||||
|
||||
def test_agent_delivery_teardown_only_on_inactive():
|
||||
# The agent is told to tear down ONLY when is_active=FALSE; a pending-delete VIP is still
|
||||
# is_active=TRUE, so the delivery returns 'available' and the agent keeps serving it.
|
||||
s = _read("routers/agent.py")
|
||||
assert "if not row['is_active']:" in s
|
||||
assert '"status": "teardown"' in s
|
||||
|
||||
|
||||
def test_migration_adds_pending_delete_and_bumps_schema():
|
||||
import re
|
||||
s = _read("database/migrations.py")
|
||||
assert "pending_delete BOOLEAN NOT NULL DEFAULT FALSE" in s
|
||||
# Schema was bumped to accommodate this column. Assert >= 7 (the version it landed in)
|
||||
# rather than pinning an exact value, so later schema bumps don't re-break this test.
|
||||
m = re.search(r"^SCHEMA_VERSION\s*=\s*(\d+)", s, re.MULTILINE)
|
||||
assert m is not None and int(m.group(1)) >= 7
|
||||
|
||||
|
||||
def test_list_visibility_backward_compat_gates_on_applied():
|
||||
# A VIP soft-deleted under the OLD immediate-delete (pre-1.7.2: is_active=FALSE,
|
||||
# last_config_status='PENDING') must NOT reappear in the list as DELETING — the
|
||||
# teardown-tracking clause is gated on last_config_status='APPLIED' (new-flow approved
|
||||
# deletes only). Locks the backward-compat fix.
|
||||
s = _read("routers/vip.py")
|
||||
assert "v.last_config_status = 'APPLIED' AND EXISTS" in s
|
||||
@@ -0,0 +1,34 @@
|
||||
"""v1.7.5 — guard for the HA/VIP "View Change" diff.
|
||||
|
||||
Editing a VIP (e.g. priority + virtual IP) must show a REAL line diff (only the changed lines)
|
||||
against the previous applied vip-* config — not the whole keepalived.conf marked as "added", and
|
||||
without the doubled "+ +" prefix (the line must be stored WITHOUT a +/- prefix; the UI adds it
|
||||
from `type`, exactly like the standard haproxy diff). Pure source guard (the behavioural diff
|
||||
test needs a DB); locks the wiring so it can't regress."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _cluster_src() -> str:
|
||||
with open(os.path.join(ROOT, "routers", "cluster.py"), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_vip_diff_uses_real_difflib_against_previous_applied():
|
||||
s = _cluster_src()
|
||||
# The previous APPLIED vip-* config for this VIP is fetched as the diff baseline.
|
||||
assert "AND status='APPLIED' AND cluster_id=$2 AND id < $3 ORDER BY id DESC LIMIT 1" in s
|
||||
# And the diff is a real unified_diff of old vs new (not "everything added").
|
||||
assert "difflib.unified_diff(old_content.split('\\n'), new_content.split('\\n')" in s
|
||||
|
||||
|
||||
def test_vip_diff_stores_lines_without_prefix():
|
||||
s = _cluster_src()
|
||||
# The old vip bug prepended "+ {line}", doubling the UI prefix ("+ +"). The vip diff now
|
||||
# stores the stripped diff line (dl[1:]) — the UI adds the +/- from `type`. `dl` is unique
|
||||
# to the vip branch (the standard haproxy diff uses `line`), so this targets the vip fix
|
||||
# only and does not touch the (separate, out-of-scope) SSL diff branch.
|
||||
assert '"line": dl[1:], "line_number": line_number' in s
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Issue #27 (v1.7.0) — unit tests for VIP request-model validation (pure, no DB)."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from models.vip import VIPCreate, VIPMemberIn # noqa: E402
|
||||
|
||||
|
||||
def _members(master_prio=150, backup_prio=100, n_backup=1):
|
||||
members = [{"agent_id": 1, "network_interface": "eth0", "role": "MASTER", "priority": master_prio}]
|
||||
for i in range(n_backup):
|
||||
members.append({"agent_id": 2 + i, "network_interface": "eth0", "role": "BACKUP", "priority": backup_prio})
|
||||
return members
|
||||
|
||||
|
||||
def _create(**over):
|
||||
base = dict(name="web-vip", pool_id=1, virtual_ip="10.0.0.100", members=_members())
|
||||
base.update(over)
|
||||
return VIPCreate(**base)
|
||||
|
||||
|
||||
class TestField:
|
||||
def test_ok(self):
|
||||
v = _create()
|
||||
assert v.virtual_ip == "10.0.0.100"
|
||||
assert v.use_unicast is True # cloud-safe default
|
||||
|
||||
def test_ipv6_rejected(self):
|
||||
with pytest.raises(ValueError):
|
||||
_create(virtual_ip="fd00::1")
|
||||
|
||||
def test_bad_ip_rejected(self):
|
||||
with pytest.raises(ValueError):
|
||||
_create(virtual_ip="not-an-ip")
|
||||
|
||||
def test_auth_pass_max_8(self):
|
||||
_create(auth_pass="12345678")
|
||||
with pytest.raises(ValueError):
|
||||
_create(auth_pass="123456789")
|
||||
|
||||
def test_interface_forbidden_char(self):
|
||||
with pytest.raises(ValueError):
|
||||
VIPMemberIn(agent_id=1, network_interface="eth0; rm -rf /", role="BACKUP", priority=100)
|
||||
|
||||
def test_priority_range(self):
|
||||
with pytest.raises(ValueError):
|
||||
VIPMemberIn(agent_id=1, network_interface="eth0", role="BACKUP", priority=255)
|
||||
|
||||
def test_role_normalized(self):
|
||||
m = VIPMemberIn(agent_id=1, network_interface="eth0", role="master", priority=150)
|
||||
assert m.role == "MASTER"
|
||||
|
||||
def test_vrid_range(self):
|
||||
with pytest.raises(ValueError):
|
||||
_create(virtual_router_id=300)
|
||||
|
||||
|
||||
class TestMembers:
|
||||
def test_single_node_allowed(self):
|
||||
# A single-node VIP (one MASTER, no BACKUP) is valid: a keepalived-managed floating
|
||||
# IP without failover (Issue #27 follow-up — relaxed from >=2 members to >=1, e.g. a
|
||||
# one-box cluster that wants a stable VIP, or before a 2nd node is added for real HA).
|
||||
v = VIPCreate(name="x", pool_id=1, virtual_ip="10.0.0.5",
|
||||
members=[{"agent_id": 1, "network_interface": "eth0", "role": "MASTER", "priority": 150}])
|
||||
assert len(v.members) == 1 and v.members[0].role == "MASTER"
|
||||
|
||||
def test_needs_at_least_one(self):
|
||||
# An empty membership is still rejected — a VIP must have at least one node.
|
||||
with pytest.raises(ValueError):
|
||||
VIPCreate(name="x", pool_id=1, virtual_ip="10.0.0.5", members=[])
|
||||
|
||||
def test_exactly_one_master(self):
|
||||
bad = [{"agent_id": 1, "network_interface": "eth0", "role": "MASTER", "priority": 150},
|
||||
{"agent_id": 2, "network_interface": "eth0", "role": "MASTER", "priority": 140}]
|
||||
with pytest.raises(ValueError):
|
||||
VIPCreate(name="x", pool_id=1, virtual_ip="10.0.0.5", members=bad)
|
||||
|
||||
def test_master_must_be_highest(self):
|
||||
with pytest.raises(ValueError):
|
||||
_create(members=_members(master_prio=100, backup_prio=120))
|
||||
|
||||
def test_no_duplicate_agent(self):
|
||||
dup = [{"agent_id": 1, "network_interface": "eth0", "role": "MASTER", "priority": 150},
|
||||
{"agent_id": 1, "network_interface": "eth1", "role": "BACKUP", "priority": 100}]
|
||||
with pytest.raises(ValueError):
|
||||
VIPCreate(name="x", pool_id=1, virtual_ip="10.0.0.5", members=dup)
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Issue #27 follow-up (v1.7.2) — source-level guards for the opt-in keepalived package
|
||||
uninstall + the "we installed it" marker. Pure (no DB), mirroring the project's other
|
||||
source-assertion tests: they lock in the *safe defaults* so a future edit can't silently
|
||||
turn routine VIP deletion into a package purge, or purge a package we didn't install."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _read(rel: str) -> str:
|
||||
with open(os.path.join(ROOT, rel), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_agent_script_purge_is_optin_and_marker_guarded():
|
||||
s = _read("utils/agent_scripts/linux_install.sh")
|
||||
# The install marker is written only when WE install keepalived, and read by the purge
|
||||
# guard — present in BOTH function copies (installer-mode + SKIP_TO_DAEMON).
|
||||
assert s.count(".hom_installed") >= 4
|
||||
# Multi-distro purge cascade exists, but ONLY inside the opt-in branch.
|
||||
assert "apt-get purge -y -qq keepalived" in s
|
||||
assert "apk del keepalived" in s
|
||||
# Orphan self-heal (not_configured) must NEVER purge — graceful teardown only, both copies.
|
||||
assert s.count('_kp_teardown "false"') >= 2
|
||||
# Purge is honored only on an explicit teardown, parsed from the delivery response.
|
||||
assert s.count(".purge // false") >= 2
|
||||
|
||||
|
||||
def test_agent_delivery_signals_purge_on_teardown():
|
||||
s = _read("routers/agent.py")
|
||||
assert "v.purge_on_teardown" in s
|
||||
assert '"purge"' in s # teardown response carries the opt-in flag
|
||||
|
||||
|
||||
def test_delete_endpoint_accepts_purge_package_default_off():
|
||||
s = _read("routers/vip.py")
|
||||
# Query param defaults to False (safe), and it sets the persisted teardown flag.
|
||||
assert "purge_package: bool = False" in s
|
||||
assert "purge_on_teardown=$2" in s
|
||||
|
||||
|
||||
def test_migration_adds_purge_column_and_bumps_schema():
|
||||
import re
|
||||
s = _read("database/migrations.py")
|
||||
assert "purge_on_teardown BOOLEAN NOT NULL DEFAULT FALSE" in s
|
||||
# Schema was bumped to accommodate this column. Assert >= 7 (the version it landed in)
|
||||
# rather than pinning an exact value, so later schema bumps don't re-break this test.
|
||||
m = re.search(r"^SCHEMA_VERSION\s*=\s*(\d+)", s, re.MULTILINE)
|
||||
assert m is not None and int(m.group(1)) >= 7
|
||||
@@ -0,0 +1,34 @@
|
||||
"""v1.7.4 — guard for the stick-table auto-inject in the HAProxy config renderer.
|
||||
|
||||
A frontend that uses a stick counter (`track-sc<N>` or an `sc_*_rate(...)` fetch) but declares
|
||||
no `stick-table` makes HAProxy fatally reject the WHOLE cluster config with "table '<frontend>'
|
||||
used but not configured". This happens with rate-limit directives baked into a frontend's stored
|
||||
request_headers/options by an older version or a config import. The renderer now injects a default
|
||||
stick-table in that case. Pure source guard (the behavioural renderer test needs a DB and is
|
||||
disabled); this locks the wiring so it can't silently regress."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _src() -> str:
|
||||
with open(os.path.join(ROOT, "services", "haproxy_config.py"), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def test_renderer_detects_stick_counter_usage():
|
||||
s = _src()
|
||||
# Detection: any track-sc<N> write OR an sc_*_rate(...) fetch flips _sc_counter_used.
|
||||
assert "_sc_counter_used" in s
|
||||
assert '"track-sc" in stripped or ("sc_" in stripped and "_rate(" in stripped)' in s
|
||||
|
||||
|
||||
def test_renderer_injects_stick_table_when_missing():
|
||||
s = _src()
|
||||
# Inject ONLY when a counter is used AND no stick-table was emitted (purely additive).
|
||||
assert "if _sc_counter_used and not _stick_table_emitted:" in s
|
||||
assert 'stick-table type ip size 100k expire 30s store http_req_rate(10s)' in s
|
||||
# The inject must register the table so it is not added twice.
|
||||
assert "_stick_table_emitted = True" in s
|
||||
@@ -153,7 +153,7 @@ collect_system_info() {
|
||||
"disk_space": $disk_bytes,
|
||||
"ip_address": "$ip_address",
|
||||
"network_interfaces": ["${network_interfaces//,/\",\"}"],
|
||||
"capabilities": ["haproxy_management", "ssl_deployment", "config_reload", "systemd_service"]
|
||||
"capabilities": ["haproxy_management", "ssl_deployment", "config_reload", "systemd_service", "keepalived_management"]
|
||||
SYSTEM_INFO_EOF
|
||||
}
|
||||
|
||||
@@ -470,7 +470,12 @@ safe_remove() {
|
||||
[[ "$QUIET_MODE" != "true" ]] && echo "Terminating existing HAProxy Agent processes..."
|
||||
KILLED_COUNT=0
|
||||
INSTALLER_PID=$$
|
||||
for pattern in "haproxy-agent" "/usr/local/bin/haproxy-agent" "haproxy-agent.service"; do
|
||||
# issue #31: match ONLY the installed agent (binary path + service), never the bare string
|
||||
# "haproxy-agent". With pgrep -f, that bare string can also match the installer's OWN command line
|
||||
# or a sudo/PAM ancestor (which the $$/$PPID guard does not fully cover), making the cleanup kill
|
||||
# the installer itself ("Killed", install aborts). The systemd service is also stopped below; the
|
||||
# "$INSTALL_DIR/haproxy-agent" path still catches any running daemon.
|
||||
for pattern in "$INSTALL_DIR/haproxy-agent" "haproxy-agent.service"; do
|
||||
PIDS=$(pgrep -f "$pattern" 2>/dev/null || true)
|
||||
if [[ -n "$PIDS" ]]; then
|
||||
FILTERED=""
|
||||
@@ -635,7 +640,7 @@ if [[ "$SKIP_TO_DAEMON" != "true" ]]; then
|
||||
|
||||
# Validate cluster exists by checking management API
|
||||
log "DEBUG" "Validating HAProxy Cluster..."
|
||||
CLUSTER_CHECK=$("$CURL_BIN" -k -s -f "$MANAGEMENT_URL/api/clusters" -H "User-Agent: haproxy-agent-installer" || echo "FAILED")
|
||||
CLUSTER_CHECK=$("$CURL_BIN" -k -s -f "$MANAGEMENT_URL/api/clusters" -H "X-API-Key: $AGENT_TOKEN" -H "User-Agent: haproxy-agent-installer" || echo "FAILED")
|
||||
if [[ "$CLUSTER_CHECK" == "FAILED" ]]; then
|
||||
log "ERROR" "Failed to connect to management API!"
|
||||
echo " Please check your network connection and management URL."
|
||||
@@ -1055,7 +1060,12 @@ register_agent() {
|
||||
local arch=$(uname -m)
|
||||
platform=$(uname -s | tr '[:upper:]' '[:lower:]') # Remove local to make it global
|
||||
local system_info=$(collect_system_info)
|
||||
|
||||
# issue #31: if collect_system_info produced no JSON content (empty on an unusual host), the
|
||||
# '$system_info,' line below would collapse to a bare comma and break the heartbeat JSON. A
|
||||
# valid fragment always contains a quoted key; if none is present, fall back to one. The glob
|
||||
# '*"*' is the most portable bash test (no POSIX class / pattern-substitution), safe on bash 3.x+.
|
||||
[[ "$system_info" != *'"'* ]] && system_info='"operating_system": "unknown"'
|
||||
|
||||
local json_payload=$(cat <<SIMPLE_EOF
|
||||
{
|
||||
"name": "$AGENT_NAME",
|
||||
@@ -1289,7 +1299,7 @@ collect_system_info() {
|
||||
"disk_space": $disk_bytes,
|
||||
"ip_address": "$ip_address",
|
||||
"network_interfaces": ["${network_interfaces//,/\",\"}"],
|
||||
"capabilities": ["haproxy_management", "ssl_deployment", "config_reload", "systemd_service"]
|
||||
"capabilities": ["haproxy_management", "ssl_deployment", "config_reload", "systemd_service", "keepalived_management"]
|
||||
SYSTEM_INFO_EOF
|
||||
}
|
||||
|
||||
@@ -1309,7 +1319,7 @@ get_keepalive_state() {
|
||||
# Method 1: journalctl (most reliable on RHEL/CentOS/Ubuntu with systemd)
|
||||
if command -v journalctl &>/dev/null; then
|
||||
state=$(journalctl -u keepalived -n 50 --no-pager 2>/dev/null \
|
||||
| grep -oE "(MASTER|BACKUP)" | tail -1)
|
||||
| grep -oE "(MASTER|BACKUP|FAULT)" | tail -1)
|
||||
fi
|
||||
|
||||
# Method 2: Fallback to log files (for non-systemd or restricted journalctl)
|
||||
@@ -1317,22 +1327,27 @@ get_keepalive_state() {
|
||||
for logfile in /var/log/messages /var/log/syslog /var/log/keepalived.log; do
|
||||
if [[ -r "$logfile" ]]; then
|
||||
state=$(tail -200 "$logfile" 2>/dev/null \
|
||||
| grep -i keepalived | grep -oE "(MASTER|BACKUP)" | tail -1)
|
||||
| grep -i keepalived | grep -oE "(MASTER|BACKUP|FAULT)" | tail -1)
|
||||
[[ -n "$state" ]] && break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Method 3: Check VIP presence on network interfaces (confirms MASTER)
|
||||
# Method 3: VIP presence on local interfaces — the most portable signal, independent
|
||||
# of logging/journald (works on any distro / init system / keepalived install type).
|
||||
# MASTER iff ANY configured VIP is actually held locally; keepalived up but holding no
|
||||
# VIP => BACKUP. Exact whole-line IP match (grep -Fxq) so 10.0.0.1 can't falsely match
|
||||
# 10.0.0.10/100, and ALL configured VIPs are checked (not just the first).
|
||||
if [[ -z "$state" ]] && [[ -r /etc/keepalived/keepalived.conf ]]; then
|
||||
local conf_vip=$(grep -A10 'virtual_ipaddress' /etc/keepalived/keepalived.conf 2>/dev/null \
|
||||
| grep -oE '[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+' | head -1)
|
||||
if [[ -n "$conf_vip" ]]; then
|
||||
if ip addr show 2>/dev/null | grep -q "$conf_vip"; then
|
||||
state="MASTER"
|
||||
else
|
||||
state="BACKUP"
|
||||
fi
|
||||
local conf_vips=$(grep -A20 'virtual_ipaddress' /etc/keepalived/keepalived.conf 2>/dev/null \
|
||||
| grep -oE '([0-9]{1,3}\.){3}[0-9]{1,3}')
|
||||
if [[ -n "$conf_vips" ]]; then
|
||||
local local_ips=$(ip -4 -o addr show 2>/dev/null | grep -oE 'inet ([0-9]{1,3}\.){3}[0-9]{1,3}' | awk '{print $2}')
|
||||
state="BACKUP"
|
||||
local _cvip
|
||||
for _cvip in $conf_vips; do
|
||||
if printf '%s\n' "$local_ips" | grep -Fxq "$_cvip"; then state="MASTER"; break; fi
|
||||
done
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -1352,7 +1367,12 @@ send_heartbeat() {
|
||||
local server_statuses=$(get_server_statuses)
|
||||
local haproxy_stats_csv=$(get_haproxy_stats_csv)
|
||||
local system_info=$(collect_system_info)
|
||||
|
||||
# issue #31: if collect_system_info produced no JSON content (empty on an unusual host), the
|
||||
# '$system_info,' line below would collapse to a bare comma and break the heartbeat JSON. A
|
||||
# valid fragment always contains a quoted key; if none is present, fall back to one. The glob
|
||||
# '*"*' is the most portable bash test (no POSIX class / pattern-substitution), safe on bash 3.x+.
|
||||
[[ "$system_info" != *'"'* ]] && system_info='"operating_system": "unknown"'
|
||||
|
||||
# Get HAProxy version for heartbeat (safe extraction, fallback to "unknown")
|
||||
local haproxy_version="unknown"
|
||||
if command -v haproxy &> /dev/null; then
|
||||
@@ -1665,6 +1685,163 @@ check_ssl_updates() {
|
||||
}
|
||||
|
||||
# Reload HAProxy service (Linux specific)
|
||||
# Issue #27 (v1.7.0) — HA/VIP (Keepalived) convergence. OPT-IN and inert by default:
|
||||
# a node with no APPLIED VIP gets status=not_configured and (lacking our ownership
|
||||
# marker) does NOTHING. Never clobbers a hand-managed keepalived. Uses $AGENT_TOKEN
|
||||
# (kept in sync with the live token in both daemon loops).
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local marker="# Managed by HAProxy OpenManager"
|
||||
local resp status http_code we_own="false" conf chk
|
||||
|
||||
# Timeouts so a hung management server can never stall the daemon loop.
|
||||
resp=$(curl -k -s --connect-timeout 10 --max-time 30 -w '\n%{http_code}' -X GET \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-config" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" 2>/dev/null) || return 0
|
||||
http_code="${resp##*$'\n'}" # last line = HTTP status
|
||||
resp="${resp%$'\n'*}" # everything before = JSON body
|
||||
[[ -z "$resp" ]] && return 0
|
||||
status=$(echo "$resp" | jq -r '.status // "unknown"' 2>/dev/null)
|
||||
# A 404 means this agent was deleted server-side (node decommissioned / pool removed):
|
||||
# self-heal teardown OUR keepalived so a removed node stops advertising the VIP. The
|
||||
# teardown branch below is marker-guarded, so a node we don't manage stays untouched.
|
||||
[[ "$http_code" == "404" ]] && status="teardown"
|
||||
# Cluster-driven keepalived.conf path (delivered in every response); default is universal.
|
||||
conf=$(echo "$resp" | jq -r '.config_path // empty' 2>/dev/null)
|
||||
[[ -z "$conf" || "$conf" == "null" ]] && conf="/etc/keepalived/keepalived.conf"
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
|
||||
_kp_report() { # $1=state $2=vip_id(or empty) $3=hash $4=message
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"vip_id\":${vid},\"state\":\"$1\",\"config_hash\":\"${3:-}\",\"message\":\"${4:-}\"}" \
|
||||
>/dev/null 2>&1 || true
|
||||
}
|
||||
_kp_teardown() {
|
||||
# $1=purge ("true" only on an explicit operator opt-in delete). Default: stop+disable
|
||||
# keepalived and remove OUR config (the VIP is released) but KEEP the package — the safe
|
||||
# enterprise default. Purge only when opted in AND we are the ones who installed it
|
||||
# (the .hom_installed marker); never remove a package the admin pre-installed.
|
||||
local purge="${1:-false}" marker_inst; marker_inst="$(dirname "$conf")/.hom_installed"
|
||||
log "INFO" "KEEPALIVED: tearing down our managed VIP config (purge=$purge)"
|
||||
systemctl stop keepalived >/dev/null 2>&1
|
||||
systemctl disable keepalived >/dev/null 2>&1
|
||||
rm -f "$conf" "$chk"
|
||||
if [[ "$purge" == "true" && -f "$marker_inst" ]]; then
|
||||
log "INFO" "KEEPALIVED: uninstalling keepalived package (operator opt-in)"
|
||||
if command -v apt-get >/dev/null 2>&1; then timeout 300 apt-get purge -y -qq keepalived >/dev/null 2>&1
|
||||
elif command -v dnf >/dev/null 2>&1; then timeout 300 dnf remove -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v yum >/dev/null 2>&1; then timeout 300 yum remove -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v zypper >/dev/null 2>&1; then timeout 300 zypper --non-interactive remove -y keepalived >/dev/null 2>&1
|
||||
elif command -v apk >/dev/null 2>&1; then timeout 300 apk del keepalived >/dev/null 2>&1
|
||||
fi
|
||||
rm -f "$marker_inst"
|
||||
if command -v keepalived >/dev/null 2>&1; then
|
||||
_kp_report "disabled" "" "" "config removed; package uninstall attempted (still present)"
|
||||
else
|
||||
_kp_report "disabled" "" "" "torn down + keepalived package uninstalled"
|
||||
fi
|
||||
elif [[ "$purge" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: package was pre-existing (not installed by us) — left in place; our config removed"
|
||||
_kp_report "disabled" "" "" "torn down (package left: pre-existing, not installed by us)"
|
||||
else
|
||||
_kp_report "disabled" "" "" "torn down"
|
||||
fi
|
||||
}
|
||||
|
||||
case "$status" in
|
||||
available) : ;; # fall through to converge
|
||||
teardown)
|
||||
# Explicit server-side delete. Honor the operator's opt-in package purge (.purge);
|
||||
# marker-guarded inside _kp_teardown, and a node we don't own (no marker conf) is a no-op.
|
||||
local _kp_purge; _kp_purge=$(echo "$resp" | jq -r '.purge // false' 2>/dev/null)
|
||||
[[ "$we_own" == "true" ]] && _kp_teardown "$_kp_purge"
|
||||
return 0 ;;
|
||||
not_configured)
|
||||
[[ "$we_own" == "true" ]] && _kp_teardown "false" # T-2 orphan self-heal: graceful only, never purge
|
||||
return 0 ;;
|
||||
*) return 0 ;; # unknown / auth error → no-op
|
||||
esac
|
||||
|
||||
# --- status == available: converge ---
|
||||
local vip_id new_conf check_script install_if new_hash
|
||||
vip_id=$(echo "$resp" | jq -r '.keepalived.vip_id // empty' 2>/dev/null)
|
||||
new_conf=$(echo "$resp" | jq -r '.keepalived.config_content // empty' 2>/dev/null)
|
||||
new_hash=$(echo "$resp" | jq -r '.keepalived.config_hash // empty' 2>/dev/null)
|
||||
check_script=$(echo "$resp" | jq -r '.keepalived.check_script // empty' 2>/dev/null)
|
||||
install_if=$(echo "$resp" | jq -r '.keepalived.install_if_missing // false' 2>/dev/null)
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
# Ownership guard: never overwrite a keepalived.conf we don't own.
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Hybrid install: install keepalived only if missing.
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
if [[ "$install_if" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: installing package..."
|
||||
if command -v apt-get >/dev/null 2>&1; then timeout 300 apt-get install -y -qq keepalived >/dev/null 2>&1
|
||||
elif command -v dnf >/dev/null 2>&1; then timeout 300 dnf install -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v yum >/dev/null 2>&1; then timeout 300 yum install -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v zypper >/dev/null 2>&1; then timeout 300 zypper --non-interactive install -y keepalived >/dev/null 2>&1
|
||||
elif command -v apk >/dev/null 2>&1; then timeout 300 apk add --no-cache keepalived >/dev/null 2>&1
|
||||
fi
|
||||
fi
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
log "ERROR" "KEEPALIVED: not installed/available"
|
||||
_kp_report "error" "$vip_id" "" "keepalived not installed"
|
||||
return 0
|
||||
fi
|
||||
# We installed keepalived (it was missing) — drop a marker so an opt-in uninstall on
|
||||
# delete removes only OUR install, never an admin's pre-existing keepalived package.
|
||||
mkdir -p "$(dirname "$conf")" 2>/dev/null; : > "$(dirname "$conf")/.hom_installed" 2>/dev/null
|
||||
fi
|
||||
|
||||
# Idempotency: skip write+reload when the on-disk content already matches.
|
||||
if [[ -f "$conf" ]]; then
|
||||
local cur_hash would_hash
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
mkdir -p "$(dirname "$conf")"
|
||||
if [[ -n "$check_script" ]]; then
|
||||
printf '%s' "$check_script" > "$chk"
|
||||
chown root:root "$chk" 2>/dev/null
|
||||
chmod 0755 "$chk" # root-owned, not world-writable — required by enable_script_security
|
||||
fi
|
||||
# Validate on a TEMP file and swap in only on success: a bad render must never land on
|
||||
# $conf (it would also make the md5 idempotency guard above suppress retries forever).
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
chmod 0644 "$conf"
|
||||
systemctl enable keepalived >/dev/null 2>&1
|
||||
if systemctl reload keepalived >/dev/null 2>&1 || systemctl restart keepalived >/dev/null 2>&1; then
|
||||
log "INFO" "KEEPALIVED: applied config for VIP ${vip_id}"
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "applied"
|
||||
else
|
||||
log "ERROR" "KEEPALIVED: reload/restart failed"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "reload/restart failed"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
reload_haproxy_service() {
|
||||
log "INFO" "Performing zero-downtime HAProxy configuration reload..."
|
||||
|
||||
@@ -2356,8 +2533,12 @@ run_daemon() {
|
||||
_loop_count=$((_loop_count + 1))
|
||||
if (( _loop_count % 5 == 0 )); then
|
||||
check_ssl_updates
|
||||
# Issue #27 — HA/VIP convergence at the SSL cadence (inert for non-VIP nodes)
|
||||
if type fetch_and_deploy_keepalived_config &>/dev/null; then
|
||||
fetch_and_deploy_keepalived_config
|
||||
fi
|
||||
fi
|
||||
|
||||
|
||||
check_agent_upgrade
|
||||
done
|
||||
|
||||
@@ -2789,28 +2970,31 @@ SYSTEM_INFO_EOF
|
||||
|
||||
if command -v journalctl &>/dev/null; then
|
||||
state=$(journalctl -u keepalived -n 50 --no-pager 2>/dev/null \
|
||||
| grep -oE "(MASTER|BACKUP)" | tail -1)
|
||||
| grep -oE "(MASTER|BACKUP|FAULT)" | tail -1)
|
||||
fi
|
||||
|
||||
if [[ -z "$state" ]]; then
|
||||
for logfile in /var/log/messages /var/log/syslog /var/log/keepalived.log; do
|
||||
if [[ -r "$logfile" ]]; then
|
||||
state=$(tail -200 "$logfile" 2>/dev/null \
|
||||
| grep -i keepalived | grep -oE "(MASTER|BACKUP)" | tail -1)
|
||||
| grep -i keepalived | grep -oE "(MASTER|BACKUP|FAULT)" | tail -1)
|
||||
[[ -n "$state" ]] && break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Method 3: VIP presence on local interfaces — portable, logging-independent.
|
||||
# MASTER iff ANY configured VIP is held locally (exact whole-line match, all VIPs).
|
||||
if [[ -z "$state" ]] && [[ -r /etc/keepalived/keepalived.conf ]]; then
|
||||
local conf_vip=$(grep -A10 'virtual_ipaddress' /etc/keepalived/keepalived.conf 2>/dev/null \
|
||||
| grep -oE '[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+' | head -1)
|
||||
if [[ -n "$conf_vip" ]]; then
|
||||
if ip addr show 2>/dev/null | grep -q "$conf_vip"; then
|
||||
state="MASTER"
|
||||
else
|
||||
state="BACKUP"
|
||||
fi
|
||||
local conf_vips=$(grep -A20 'virtual_ipaddress' /etc/keepalived/keepalived.conf 2>/dev/null \
|
||||
| grep -oE '([0-9]{1,3}\.){3}[0-9]{1,3}')
|
||||
if [[ -n "$conf_vips" ]]; then
|
||||
local local_ips=$(ip -4 -o addr show 2>/dev/null | grep -oE 'inet ([0-9]{1,3}\.){3}[0-9]{1,3}' | awk '{print $2}')
|
||||
state="BACKUP"
|
||||
local _cvip
|
||||
for _cvip in $conf_vips; do
|
||||
if printf '%s\n' "$local_ips" | grep -Fxq "$_cvip"; then state="MASTER"; break; fi
|
||||
done
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -2849,7 +3033,15 @@ SYSTEM_INFO_EOF
|
||||
local keepalive_info=$(get_keepalive_state)
|
||||
local keepalive_state=$(echo "$keepalive_info" | cut -d'|' -f1)
|
||||
local keepalive_ip=$(echo "$keepalive_info" | cut -d'|' -f2)
|
||||
|
||||
|
||||
# Network interfaces (DAEMON) — reported so the HA/VIP create form can offer this
|
||||
# node's real NICs. /sys/class/net is universal across distros (no iproute2 needed);
|
||||
# exclude loopback here, and the UI further hides docker/veth/bridge interfaces. Empty
|
||||
# -> "[]" (never "[\"\"]") so a node with no NICs doesn't surface a blank option.
|
||||
local net_ifaces ni_json
|
||||
net_ifaces=$(ls /sys/class/net 2>/dev/null | grep -vE '^lo$' | tr '\n' ',' | sed 's/,$//')
|
||||
if [ -n "$net_ifaces" ]; then ni_json="[\"${net_ifaces//,/\",\"}\"]"; else ni_json="[]"; fi
|
||||
|
||||
# Prepare heartbeat payload with all data
|
||||
local heartbeat_payload="{
|
||||
\"name\": \"$AGENT_NAME\",
|
||||
@@ -2859,6 +3051,8 @@ SYSTEM_INFO_EOF
|
||||
\"platform\": \"$platform\",
|
||||
\"architecture\": \"$(uname -m)\",
|
||||
\"version\": \"{{AGENT_VERSION}}\",
|
||||
\"capabilities\": [\"haproxy_management\", \"ssl_deployment\", \"config_reload\", \"systemd_service\", \"keepalived_management\"],
|
||||
\"network_interfaces\": $ni_json,
|
||||
\"haproxy_status\": \"$haproxy_status\",
|
||||
\"haproxy_version\": \"$haproxy_version\",
|
||||
\"cluster_id\": $CLUSTER_ID,
|
||||
@@ -3041,9 +3235,159 @@ CONFIG_RESPONSE_EOF
|
||||
HAPROXY_BIN="${HAPROXY_BIN_PATH}"
|
||||
HAPROXY_CONFIG="${HAPROXY_CONFIG_PATH}"
|
||||
log "INFO" "DAEMON: Initialized HAProxy paths - bin: $HAPROXY_BIN, config: $HAPROXY_CONFIG"
|
||||
|
||||
|
||||
# Issue #27 (v1.7.0) — HA/VIP (Keepalived) convergence (DAEMON copy). Inert by
|
||||
# default: a node with no APPLIED VIP gets not_configured and (lacking our marker)
|
||||
# does nothing. Never clobbers a hand-managed keepalived. $AGENT_TOKEN is synced to
|
||||
# the live token at the top of each loop iteration below.
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local marker="# Managed by HAProxy OpenManager"
|
||||
local resp status http_code we_own="false" conf chk
|
||||
|
||||
# Timeouts so a hung management server can never stall the daemon loop.
|
||||
resp=$(curl -k -s --connect-timeout 10 --max-time 30 -w '\n%{http_code}' -X GET \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-config" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" 2>/dev/null) || return 0
|
||||
http_code="${resp##*$'\n'}" # last line = HTTP status
|
||||
resp="${resp%$'\n'*}" # everything before = JSON body
|
||||
[[ -z "$resp" ]] && return 0
|
||||
status=$(echo "$resp" | jq -r '.status // "unknown"' 2>/dev/null)
|
||||
# A 404 means this agent was deleted server-side (node decommissioned / pool removed):
|
||||
# self-heal teardown OUR keepalived so a removed node stops advertising the VIP. The
|
||||
# teardown branch below is marker-guarded, so a node we don't manage stays untouched.
|
||||
[[ "$http_code" == "404" ]] && status="teardown"
|
||||
# Cluster-driven keepalived.conf path (delivered in every response); default is universal.
|
||||
conf=$(echo "$resp" | jq -r '.config_path // empty' 2>/dev/null)
|
||||
[[ -z "$conf" || "$conf" == "null" ]] && conf="/etc/keepalived/keepalived.conf"
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
|
||||
_kp_report() {
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"vip_id\":${vid},\"state\":\"$1\",\"config_hash\":\"${3:-}\",\"message\":\"${4:-}\"}" \
|
||||
>/dev/null 2>&1 || true
|
||||
}
|
||||
_kp_teardown() {
|
||||
# $1=purge ("true" only on an explicit operator opt-in delete). Default: stop+disable
|
||||
# keepalived and remove OUR config (VIP released) but KEEP the package. Purge only when
|
||||
# opted in AND we installed it (.hom_installed marker) — never an admin's package.
|
||||
local purge="${1:-false}" marker_inst; marker_inst="$(dirname "$conf")/.hom_installed"
|
||||
log "INFO" "KEEPALIVED: tearing down our managed VIP config (purge=$purge)"
|
||||
systemctl stop keepalived >/dev/null 2>&1
|
||||
systemctl disable keepalived >/dev/null 2>&1
|
||||
rm -f "$conf" "$chk"
|
||||
if [[ "$purge" == "true" && -f "$marker_inst" ]]; then
|
||||
log "INFO" "KEEPALIVED: uninstalling keepalived package (operator opt-in)"
|
||||
if command -v apt-get >/dev/null 2>&1; then timeout 300 apt-get purge -y -qq keepalived >/dev/null 2>&1
|
||||
elif command -v dnf >/dev/null 2>&1; then timeout 300 dnf remove -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v yum >/dev/null 2>&1; then timeout 300 yum remove -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v zypper >/dev/null 2>&1; then timeout 300 zypper --non-interactive remove -y keepalived >/dev/null 2>&1
|
||||
elif command -v apk >/dev/null 2>&1; then timeout 300 apk del keepalived >/dev/null 2>&1
|
||||
fi
|
||||
rm -f "$marker_inst"
|
||||
if command -v keepalived >/dev/null 2>&1; then
|
||||
_kp_report "disabled" "" "" "config removed; package uninstall attempted (still present)"
|
||||
else
|
||||
_kp_report "disabled" "" "" "torn down + keepalived package uninstalled"
|
||||
fi
|
||||
elif [[ "$purge" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: package was pre-existing (not installed by us) — left in place; our config removed"
|
||||
_kp_report "disabled" "" "" "torn down (package left: pre-existing, not installed by us)"
|
||||
else
|
||||
_kp_report "disabled" "" "" "torn down"
|
||||
fi
|
||||
}
|
||||
|
||||
case "$status" in
|
||||
available) : ;;
|
||||
teardown)
|
||||
local _kp_purge; _kp_purge=$(echo "$resp" | jq -r '.purge // false' 2>/dev/null)
|
||||
[[ "$we_own" == "true" ]] && _kp_teardown "$_kp_purge" # explicit delete: honor opt-in purge
|
||||
return 0 ;;
|
||||
not_configured)
|
||||
[[ "$we_own" == "true" ]] && _kp_teardown "false" # T-2 orphan self-heal: graceful only
|
||||
return 0 ;;
|
||||
*) return 0 ;;
|
||||
esac
|
||||
|
||||
local vip_id new_conf check_script install_if new_hash
|
||||
vip_id=$(echo "$resp" | jq -r '.keepalived.vip_id // empty' 2>/dev/null)
|
||||
new_conf=$(echo "$resp" | jq -r '.keepalived.config_content // empty' 2>/dev/null)
|
||||
new_hash=$(echo "$resp" | jq -r '.keepalived.config_hash // empty' 2>/dev/null)
|
||||
check_script=$(echo "$resp" | jq -r '.keepalived.check_script // empty' 2>/dev/null)
|
||||
install_if=$(echo "$resp" | jq -r '.keepalived.install_if_missing // false' 2>/dev/null)
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
if [[ "$install_if" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: installing package..."
|
||||
if command -v apt-get >/dev/null 2>&1; then timeout 300 apt-get install -y -qq keepalived >/dev/null 2>&1
|
||||
elif command -v dnf >/dev/null 2>&1; then timeout 300 dnf install -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v yum >/dev/null 2>&1; then timeout 300 yum install -y -q keepalived >/dev/null 2>&1
|
||||
elif command -v zypper >/dev/null 2>&1; then timeout 300 zypper --non-interactive install -y keepalived >/dev/null 2>&1
|
||||
elif command -v apk >/dev/null 2>&1; then timeout 300 apk add --no-cache keepalived >/dev/null 2>&1
|
||||
fi
|
||||
fi
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
log "ERROR" "KEEPALIVED: not installed/available"
|
||||
_kp_report "error" "$vip_id" "" "keepalived not installed"
|
||||
return 0
|
||||
fi
|
||||
# We installed keepalived (it was missing) — drop a marker so an opt-in uninstall on
|
||||
# delete removes only OUR install, never an admin's pre-existing keepalived package.
|
||||
mkdir -p "$(dirname "$conf")" 2>/dev/null; : > "$(dirname "$conf")/.hom_installed" 2>/dev/null
|
||||
fi
|
||||
|
||||
if [[ -f "$conf" ]]; then
|
||||
local cur_hash would_hash
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
mkdir -p "$(dirname "$conf")"
|
||||
if [[ -n "$check_script" ]]; then
|
||||
printf '%s' "$check_script" > "$chk"
|
||||
chown root:root "$chk" 2>/dev/null
|
||||
chmod 0755 "$chk"
|
||||
fi
|
||||
# Validate on a TEMP file and swap in only on success: a bad render must never land on
|
||||
# $conf (it would also make the md5 idempotency guard above suppress retries forever).
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
chmod 0644 "$conf"
|
||||
systemctl enable keepalived >/dev/null 2>&1
|
||||
if systemctl reload keepalived >/dev/null 2>&1 || systemctl restart keepalived >/dev/null 2>&1; then
|
||||
log "INFO" "KEEPALIVED: applied config for VIP ${vip_id}"
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "applied"
|
||||
else
|
||||
log "ERROR" "KEEPALIVED: reload/restart failed"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "reload/restart failed"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
log "DEBUG" "DAEMON: Starting agent monitoring loop for $AGENT_NAME"
|
||||
|
||||
|
||||
# Enhanced daemon loop with upgrade capability - EXACT MacOS COPY
|
||||
while true; do
|
||||
sleep 30
|
||||
@@ -3208,6 +3552,13 @@ CONFIG_RESPONSE_EOF
|
||||
fi
|
||||
fi
|
||||
|
||||
# Issue #27 — HA/VIP (Keepalived) convergence at the SSL cadence (~2.5 min);
|
||||
# inert for non-VIP nodes (status=not_configured + no ownership marker).
|
||||
_kp_loop_count=$(( ${_kp_loop_count:-0} + 1 ))
|
||||
if (( _kp_loop_count % 5 == 0 )) && type fetch_and_deploy_keepalived_config &>/dev/null; then
|
||||
fetch_and_deploy_keepalived_config
|
||||
fi
|
||||
|
||||
# Check for configuration updates with proper error handling
|
||||
config_response=$(curl -k -s -X GET "$MANAGEMENT_URL/api/agents/$AGENT_NAME/config" \
|
||||
-H "X-API-Key: $CURRENT_AGENT_TOKEN" 2>/dev/null)
|
||||
|
||||
@@ -318,7 +318,11 @@ safe_remove() {
|
||||
[[ "$QUIET_MODE" != "true" ]] && echo "Terminating existing HAProxy Agent processes..."
|
||||
KILLED_COUNT=0
|
||||
INSTALLER_PID=$$
|
||||
for pattern in "haproxy-agent" "/usr/local/bin/haproxy-agent" "com.haproxy.agent"; do
|
||||
# issue #31: match ONLY the installed agent (binary path + LaunchDaemon label), never the bare
|
||||
# string "haproxy-agent". With pgrep -f, that bare string can also match the installer's OWN command
|
||||
# line or a sudo ancestor (which the $$/$PPID guard does not fully cover), making the cleanup kill
|
||||
# the installer itself. The "$INSTALL_DIR/haproxy-agent" path still catches any running daemon.
|
||||
for pattern in "$INSTALL_DIR/haproxy-agent" "com.haproxy.agent"; do
|
||||
PIDS=$(pgrep -f "$pattern" 2>/dev/null || true)
|
||||
if [[ -n "$PIDS" ]]; then
|
||||
FILTERED=""
|
||||
@@ -522,7 +526,7 @@ if [[ "$SKIP_TO_DAEMON" != "true" ]]; then
|
||||
|
||||
# Validate cluster exists by checking management API
|
||||
log "DEBUG" "Validating HAProxy Cluster..."
|
||||
CLUSTER_CHECK=$("$CURL_BIN" -k -s -f "$MANAGEMENT_URL/api/clusters" -H "User-Agent: haproxy-agent-installer" || echo "FAILED")
|
||||
CLUSTER_CHECK=$("$CURL_BIN" -k -s -f "$MANAGEMENT_URL/api/clusters" -H "X-API-Key: $AGENT_TOKEN" -H "User-Agent: haproxy-agent-installer" || echo "FAILED")
|
||||
if [[ "$CLUSTER_CHECK" == "FAILED" ]]; then
|
||||
log "ERROR" "Failed to connect to management API!"
|
||||
echo " Please check your network connection and management URL."
|
||||
@@ -867,7 +871,7 @@ SSL_SYNC_TIMESTAMP_FILE="/tmp/haproxy-agent-ssl-sync-${AGENT_NAME}"
|
||||
get_cluster_paths() {
|
||||
log "DEBUG" "Fetching cluster paths from management API"
|
||||
local cluster_response=$("$CURL_BIN" -k -s -X GET "$MANAGEMENT_URL/api/clusters" \
|
||||
-H "Authorization: Bearer $AGENT_TOKEN")
|
||||
-H "X-API-Key: $AGENT_TOKEN")
|
||||
|
||||
if [[ $? -eq 0 ]]; then
|
||||
local _cfg _bin _sock
|
||||
@@ -920,7 +924,12 @@ register_agent() {
|
||||
local arch=$(uname -m)
|
||||
platform=$(uname -s | tr '[:upper:]' '[:lower:]') # Remove local to make it global
|
||||
local system_info=$(collect_system_info)
|
||||
|
||||
# issue #31: if collect_system_info produced no JSON content (empty on an unusual host), the
|
||||
# '$system_info,' line below would collapse to a bare comma and break the heartbeat JSON. A
|
||||
# valid fragment always contains a quoted key; if none is present, fall back to one. The glob
|
||||
# '*"*' is the most portable bash test (no POSIX class / pattern-substitution), safe on bash 3.x+.
|
||||
[[ "$system_info" != *'"'* ]] && system_info='"operating_system": "unknown"'
|
||||
|
||||
local json_payload=$(cat <<SIMPLE_EOF
|
||||
{
|
||||
"name": "$AGENT_NAME",
|
||||
@@ -1164,13 +1173,39 @@ get_keepalive_state() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# Issue #27 — keepalived/VRRP is Linux-only. macOS agents cannot run keepalived, so they
|
||||
# can never be VIP members (the backend also withholds the keepalived_management capability,
|
||||
# so the UI warns at assign time). This is a safe no-op that, if a VIP is somehow mis-assigned
|
||||
# to a macOS node, reports a clear "unsupported on macOS" status instead of silently never
|
||||
# converging. It never installs/writes anything.
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local resp status vip_id
|
||||
resp=$(curl -k -s -X GET "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-config" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" 2>/dev/null) || return 0
|
||||
[[ -z "$resp" ]] && return 0
|
||||
status=$(echo "$resp" | jq -r '.status // "unknown"' 2>/dev/null)
|
||||
if [[ "$status" == "available" ]]; then
|
||||
vip_id=$(echo "$resp" | jq -r '.keepalived.vip_id // empty' 2>/dev/null)
|
||||
log "WARN" "KEEPALIVED: a VIP is assigned to this node, but keepalived/VRRP is not supported on macOS — ignoring"
|
||||
curl -k -s -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"vip_id\":${vip_id:-null},\"state\":\"error\",\"message\":\"keepalived/VRRP not supported on macOS\"}" >/dev/null 2>&1 || true
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# Send heartbeat
|
||||
send_heartbeat() {
|
||||
local haproxy_status=$(get_haproxy_status)
|
||||
local server_statuses=$(get_server_statuses)
|
||||
local haproxy_stats_csv=$(get_haproxy_stats_csv)
|
||||
local system_info=$(collect_system_info)
|
||||
|
||||
# issue #31: if collect_system_info produced no JSON content (empty on an unusual host), the
|
||||
# '$system_info,' line below would collapse to a bare comma and break the heartbeat JSON. A
|
||||
# valid fragment always contains a quoted key; if none is present, fall back to one. The glob
|
||||
# '*"*' is the most portable bash test (no POSIX class / pattern-substitution), safe on bash 3.x+.
|
||||
[[ "$system_info" != *'"'* ]] && system_info='"operating_system": "unknown"'
|
||||
|
||||
# Get HAProxy version for heartbeat (safe extraction, fallback to "unknown")
|
||||
local haproxy_version="unknown"
|
||||
if command -v haproxy &> /dev/null; then
|
||||
@@ -2197,8 +2232,12 @@ run_daemon() {
|
||||
_loop_count=$((_loop_count + 1))
|
||||
if (( _loop_count % 5 == 0 )); then
|
||||
check_ssl_updates
|
||||
# Issue #27 — HA/VIP convergence (no-op on macOS; reports unsupported if assigned)
|
||||
if type fetch_and_deploy_keepalived_config &>/dev/null; then
|
||||
fetch_and_deploy_keepalived_config
|
||||
fi
|
||||
fi
|
||||
|
||||
|
||||
check_agent_upgrade
|
||||
done
|
||||
|
||||
@@ -2629,7 +2668,25 @@ SYSTEM_INFO_EOF
|
||||
echo "|"
|
||||
return 0
|
||||
}
|
||||
|
||||
|
||||
# Issue #27 — keepalived/VRRP is Linux-only (DAEMON version). Safe no-op that reports a
|
||||
# clear "unsupported on macOS" status if a VIP is ever mis-assigned to a macOS node.
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local resp status vip_id
|
||||
resp=$(curl -k -s -X GET "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-config" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" 2>/dev/null) || return 0
|
||||
[[ -z "$resp" ]] && return 0
|
||||
status=$(echo "$resp" | jq -r '.status // "unknown"' 2>/dev/null)
|
||||
if [[ "$status" == "available" ]]; then
|
||||
vip_id=$(echo "$resp" | jq -r '.keepalived.vip_id // empty' 2>/dev/null)
|
||||
log "WARN" "KEEPALIVED: a VIP is assigned to this node, but keepalived/VRRP is not supported on macOS — ignoring"
|
||||
curl -k -s -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"vip_id\":${vip_id:-null},\"state\":\"error\",\"message\":\"keepalived/VRRP not supported on macOS\"}" >/dev/null 2>&1 || true
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# Send heartbeat (DAEMON version - includes all stats)
|
||||
send_heartbeat() {
|
||||
local haproxy_status=$(get_haproxy_status)
|
||||
@@ -3021,6 +3078,13 @@ CONFIG_RESPONSE_EOF
|
||||
fi
|
||||
fi
|
||||
|
||||
# Issue #27 — HA/VIP convergence at the SSL cadence (no-op on macOS; reports
|
||||
# unsupported if a VIP is mis-assigned to this node).
|
||||
_kp_loop_count=$(( ${_kp_loop_count:-0} + 1 ))
|
||||
if (( _kp_loop_count % 5 == 0 )) && type fetch_and_deploy_keepalived_config &>/dev/null; then
|
||||
fetch_and_deploy_keepalived_config
|
||||
fi
|
||||
|
||||
# Check for configuration updates with proper error handling
|
||||
config_response=$(curl -k -s -X GET "$MANAGEMENT_URL/api/agents/$AGENT_NAME/config" \
|
||||
-H "X-API-Key: $CURRENT_AGENT_TOKEN" 2>/dev/null)
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Issue #53 — at-rest encryption for the pending CSR private key (v1.10.1).
|
||||
|
||||
Mirrors the established Fernet + HKDF(SECRET_KEY) pattern already used for the VRRP secret
|
||||
(services/keepalived_config.py), TOTP secrets (services/mfa_service.py) and DNS provider
|
||||
credentials (utils/dns_credentials.py): prefer an explicit CSR_ENCRYPTION_KEY env var (enables
|
||||
key rotation), else derive a stable key from SECRET_KEY via HKDF with its own versioned info
|
||||
string, so a rotation of one secret class never affects another.
|
||||
|
||||
WHY this key and not every key in the system: the CSR private key is the one key that sits IDLE.
|
||||
It is generated at CSR creation, waits for an external CA to sign the request (days to weeks),
|
||||
and is destroyed the moment the signed certificate is imported — it is never transmitted to an
|
||||
agent and never leaves the server. `ssl_certificates.private_key_content` and the ACME order keys
|
||||
are different: agents must receive them in plaintext on every poll, so encrypting them at rest
|
||||
buys nothing without an end-to-end redesign.
|
||||
|
||||
STORAGE: the Fernet token replaces the PEM in the SAME `ssl_csrs.private_key_pem` TEXT column.
|
||||
No new column, no new table, and deliberately NO `SCHEMA_VERSION` bump — a bump would re-run the
|
||||
migration sequence and re-seed the four built-in roles to their defaults (see UPGRADE_GUIDE.md),
|
||||
which is a needless side effect for a storage-format change.
|
||||
|
||||
BACKWARD COMPATIBILITY: rows written before this release hold a raw PEM. `decrypt_csr_private_key`
|
||||
detects those by their `-----BEGIN` header and returns them unchanged. The discriminator is exact,
|
||||
not a heuristic: a Fernet token is base64url text and can never contain "-----". Legacy rows drain
|
||||
naturally, since a CSR's key copy is NULLed on import.
|
||||
|
||||
KEY ROTATION: if SECRET_KEY rotates while CSR_ENCRYPTION_KEY is unset, previously stored keys
|
||||
become undecryptable and `decrypt_csr_private_key` returns None. Callers MUST surface a clear
|
||||
"delete this CSR and create a new one" error — the CSR is unusable at that point, because the
|
||||
signed certificate can no longer be paired with its key.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from cryptography.fernet import Fernet, InvalidToken
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
from config import SECRET_KEY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A PEM private key always carries this header; a Fernet token is base64url and never can.
|
||||
_PEM_MARKER = "-----BEGIN"
|
||||
|
||||
_fernet_instance: Optional[Fernet] = None
|
||||
|
||||
|
||||
def _resolve_fernet_key() -> bytes:
|
||||
"""Prefer an explicit CSR_ENCRYPTION_KEY; else derive from SECRET_KEY via HKDF with a
|
||||
versioned info string (so stored keys survive restarts)."""
|
||||
explicit = os.getenv("CSR_ENCRYPTION_KEY", "").strip()
|
||||
if explicit:
|
||||
try:
|
||||
Fernet(explicit.encode())
|
||||
return explicit.encode()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("CSR_ENCRYPTION_KEY env var present but invalid: %s", exc)
|
||||
logger.warning(
|
||||
"CSR_ENCRYPTION_KEY not set; deriving the CSR private-key encryption key from SECRET_KEY. "
|
||||
"Set CSR_ENCRYPTION_KEY to a Fernet key to enable key rotation."
|
||||
)
|
||||
hkdf = HKDF(algorithm=hashes.SHA256(), length=32, salt=None, info=b"csr-private-key-v1")
|
||||
derived = hkdf.derive(SECRET_KEY.encode("utf-8"))
|
||||
return base64.urlsafe_b64encode(derived)
|
||||
|
||||
|
||||
def _get_fernet() -> Fernet:
|
||||
global _fernet_instance
|
||||
if _fernet_instance is None:
|
||||
_fernet_instance = Fernet(_resolve_fernet_key())
|
||||
return _fernet_instance
|
||||
|
||||
|
||||
def reset_fernet_for_tests() -> None:
|
||||
"""Test-only hook to force re-resolution after env mutation."""
|
||||
global _fernet_instance
|
||||
_fernet_instance = None
|
||||
|
||||
|
||||
def is_encrypted(stored: Optional[str]) -> bool:
|
||||
"""True when the stored value is a Fernet token rather than a legacy raw PEM.
|
||||
|
||||
Single source of the format discriminator: `decrypt_csr_private_key` branches on this, so
|
||||
the "what does a stored value look like" rule is stated exactly once.
|
||||
"""
|
||||
return bool(stored) and _PEM_MARKER not in stored
|
||||
|
||||
|
||||
def encrypt_csr_private_key(pem: str) -> str:
|
||||
"""Fernet-encrypt a PEM private key to a storable token string."""
|
||||
return _get_fernet().encrypt(pem.encode("utf-8")).decode("utf-8")
|
||||
|
||||
|
||||
def decrypt_csr_private_key(stored: Optional[str]) -> Optional[str]:
|
||||
"""Return the PEM private key for a stored value.
|
||||
|
||||
Accepts BOTH shapes so an upgrade needs no data migration:
|
||||
- a raw PEM written before v1.10.1 -> returned unchanged
|
||||
- a Fernet token -> decrypted
|
||||
|
||||
Returns None when the value is empty or cannot be decrypted (e.g. SECRET_KEY rotated without
|
||||
CSR_ENCRYPTION_KEY). Callers MUST treat None as "this CSR's key is unrecoverable" and tell the
|
||||
operator to delete it and create a new one; never fall through to a pairing attempt.
|
||||
"""
|
||||
if not stored:
|
||||
return None
|
||||
if not is_encrypted(stored):
|
||||
return stored # legacy plaintext row, pre-v1.10.1
|
||||
try:
|
||||
return _get_fernet().decrypt(stored.encode("utf-8")).decode("utf-8")
|
||||
except InvalidToken:
|
||||
logger.warning("Failed to decrypt a stored CSR private key (invalid Fernet token)")
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Unexpected error decrypting a stored CSR private key: %s", exc)
|
||||
return None
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Issue #35 — ACME DNS-01 (v1.8.0): at-rest encryption for per-account DNS provider credentials.
|
||||
|
||||
Mirrors the established Fernet + HKDF(SECRET_KEY) pattern used for the VRRP secret
|
||||
(backend/services/keepalived_config.py) and TOTP secrets (backend/services/mfa_service.py):
|
||||
prefer an explicit DNS_PROVIDER_ENCRYPTION_KEY env var (enables key rotation), else derive a
|
||||
stable key from SECRET_KEY via HKDF with a versioned info string.
|
||||
|
||||
DNS provider credentials are a small dict (e.g. {"api_token": "..."}). They are JSON-serialized,
|
||||
encrypted to a Fernet token string for storage, and only ever decrypted in-process when a DNS-01
|
||||
order needs to talk to the provider. Plaintext credentials are NEVER logged or returned by the API.
|
||||
|
||||
NOTE on key rotation: if SECRET_KEY rotates and DNS_PROVIDER_ENCRYPTION_KEY is not set, previously
|
||||
stored credentials become undecryptable (decrypt returns None). Callers MUST treat a None result as
|
||||
"credentials unavailable — re-enter in Settings" and surface a clear error, never a silent hang.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from typing import Dict, Optional
|
||||
|
||||
from cryptography.fernet import Fernet, InvalidToken
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
from config import SECRET_KEY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_fernet_instance: Optional[Fernet] = None
|
||||
|
||||
|
||||
def _resolve_fernet_key() -> bytes:
|
||||
"""Prefer an explicit DNS_PROVIDER_ENCRYPTION_KEY; else derive from SECRET_KEY via HKDF
|
||||
with a versioned info string (so credentials survive restarts)."""
|
||||
explicit = os.getenv("DNS_PROVIDER_ENCRYPTION_KEY", "").strip()
|
||||
if explicit:
|
||||
try:
|
||||
Fernet(explicit.encode())
|
||||
return explicit.encode()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("DNS_PROVIDER_ENCRYPTION_KEY env var present but invalid: %s", exc)
|
||||
logger.warning(
|
||||
"DNS_PROVIDER_ENCRYPTION_KEY not set; deriving the DNS-credentials encryption key from "
|
||||
"SECRET_KEY. Set DNS_PROVIDER_ENCRYPTION_KEY to a Fernet key to enable key rotation."
|
||||
)
|
||||
hkdf = HKDF(algorithm=hashes.SHA256(), length=32, salt=None, info=b"dns-provider-creds-v1")
|
||||
derived = hkdf.derive(SECRET_KEY.encode("utf-8"))
|
||||
return base64.urlsafe_b64encode(derived)
|
||||
|
||||
|
||||
def _get_fernet() -> Fernet:
|
||||
global _fernet_instance
|
||||
if _fernet_instance is None:
|
||||
_fernet_instance = Fernet(_resolve_fernet_key())
|
||||
return _fernet_instance
|
||||
|
||||
|
||||
def reset_fernet_for_tests() -> None:
|
||||
"""Test-only hook to force re-resolution after env mutation."""
|
||||
global _fernet_instance
|
||||
_fernet_instance = None
|
||||
|
||||
|
||||
def encrypt_dns_credentials(credentials: Dict[str, str]) -> str:
|
||||
"""JSON-serialize and Fernet-encrypt a credentials dict to a storable token string."""
|
||||
payload = json.dumps(credentials, separators=(",", ":")).encode("utf-8")
|
||||
return _get_fernet().encrypt(payload).decode("utf-8")
|
||||
|
||||
|
||||
def decrypt_dns_credentials(token: str) -> Optional[Dict[str, str]]:
|
||||
"""Decrypt a stored token back to the credentials dict. Returns None if the token can't be
|
||||
decrypted (e.g. key rotated) — callers must surface a clear 're-enter credentials' error."""
|
||||
try:
|
||||
plain = _get_fernet().decrypt(token.encode("utf-8")).decode("utf-8")
|
||||
data = json.loads(plain)
|
||||
if not isinstance(data, dict):
|
||||
logger.error("Decrypted DNS credentials are not a JSON object")
|
||||
return None
|
||||
return data
|
||||
except InvalidToken:
|
||||
logger.warning("Failed to decrypt DNS provider credentials (invalid Fernet token)")
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Unexpected error decrypting DNS provider credentials: %s", exc)
|
||||
return None
|
||||
@@ -329,9 +329,10 @@ async def _rollback_update(
|
||||
tcp_request_rules = $26, timeout_client = $27, timeout_http_request = $28,
|
||||
rate_limit = $29, compression = $30, log_separate = $31,
|
||||
monitor_uri = $32, maxconn = $33,
|
||||
cluster_id = $34, is_active = $35, last_config_status = $36,
|
||||
cluster_id = $34, is_active = $35, last_config_status = $36,
|
||||
log_format = $37, filters = $38,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
WHERE id = $37
|
||||
WHERE id = $39
|
||||
""",
|
||||
old_values.get('name'),
|
||||
old_values.get('bind_address'),
|
||||
@@ -369,6 +370,8 @@ async def _rollback_update(
|
||||
old_values.get('cluster_id'),
|
||||
old_values.get('is_active'),
|
||||
old_values.get('last_config_status'),
|
||||
old_values.get('log_format'), # Issue #38
|
||||
old_values.get('filters'), # Issue #38
|
||||
entity_id
|
||||
)
|
||||
|
||||
|
||||
@@ -105,6 +105,11 @@ class ParsedFrontend:
|
||||
response_headers: Optional[str] = None
|
||||
options: Optional[str] = None # HAProxy frontend options (option httplog, option forwardfor, etc.)
|
||||
tcp_request_rules: Optional[str] = None # TCP request directives (for TCP mode)
|
||||
# Issue #38: SPOE (and other) filter directives + frontend log-format.
|
||||
# Stored as full directive lines; `filters` is newline-joined to preserve
|
||||
# ordering when multiple `filter ...` lines exist.
|
||||
log_format: Optional[str] = None # `log-format` / `log-format-sd` line(s)
|
||||
filters: Optional[str] = None # `filter ...` line(s), e.g. `filter spoe engine coraza config ...`
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -256,8 +261,23 @@ class HAProxyConfigParser:
|
||||
acl_rules_list = []
|
||||
use_backend_rules_list = []
|
||||
tcp_request_rules_list = []
|
||||
filters_list = []
|
||||
log_format_list = []
|
||||
|
||||
for line in lines:
|
||||
# Issue #38: capture `filter ...` (SPOE/Coraza etc.) and
|
||||
# `log-format`/`log-format-sd` directives. Pre-fix these matched
|
||||
# no branch below and were silently dropped, so an imported SPOE
|
||||
# config lost `filter spoe engine coraza ...` (→ HAProxy fatal
|
||||
# "unable to find SPOE engine") and the frontend log-format.
|
||||
# `continue` isolates them from the header/option handling below.
|
||||
if line.startswith('filter '):
|
||||
filters_list.append(line.strip())
|
||||
continue
|
||||
if re.match(r'^log-format(-sd)?\s', line, re.IGNORECASE):
|
||||
log_format_list.append(line.strip())
|
||||
continue
|
||||
|
||||
# Parse bind directive
|
||||
# IMPORTANT: Handle multiple bind lines correctly
|
||||
# Example: bind *:1002 (HTTP) and bind *:443 ssl (HTTPS)
|
||||
@@ -540,6 +560,13 @@ class HAProxyConfigParser:
|
||||
if tcp_request_rules_list:
|
||||
frontend.tcp_request_rules = '\n'.join(tcp_request_rules_list)
|
||||
|
||||
# Issue #38: assign captured SPOE filters + log-format
|
||||
if filters_list:
|
||||
frontend.filters = '\n'.join(filters_list)
|
||||
|
||||
if log_format_list:
|
||||
frontend.log_format = '\n'.join(log_format_list)
|
||||
|
||||
self.frontends.append(frontend)
|
||||
logger.info(f"Parsed frontend: {name} -> {frontend.default_backend}")
|
||||
|
||||
@@ -666,9 +693,14 @@ class HAProxyConfigParser:
|
||||
'transparent', 'abortonclose', 'allbackups', 'checkcache', 'clitcpka',
|
||||
'srvtcpka', 'http-no-delay', 'socket-stats', 'tcp-smart-accept',
|
||||
'tcp-smart-connect', 'independant-streams', 'log-separate-errors',
|
||||
'log-health-checks', 'accept-invalid-http-request', 'accept-invalid-http-response'
|
||||
'log-health-checks', 'accept-invalid-http-request', 'accept-invalid-http-response',
|
||||
# Issue #38: SPOP health check for SPOE agent backends
|
||||
# (e.g. coraza-spoa). Already collected below regardless, but
|
||||
# listing it suppresses the spurious "unknown option" warning
|
||||
# for the exact SPOE use-case.
|
||||
'spop-check'
|
||||
]
|
||||
|
||||
|
||||
if option_name not in valid_options:
|
||||
# Unknown/invalid option - add warning but still collect it
|
||||
self.warnings.append(
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
"""
|
||||
SSRF guard for outbound HTTP fetches to user/DB-controlled URLs.
|
||||
|
||||
GHSA-3vh4-gvxx-wm2p: the ACME `directory_url` was fetched server-side with no
|
||||
validation, turning the backend into a request-forwarding primitive against
|
||||
loopback / RFC1918 / link-local / cloud-metadata IP space (and reflecting the
|
||||
upstream JSON keys back to the caller).
|
||||
|
||||
The classification logic mirrors the hardened ACME diagnostics probe
|
||||
(services/acme_diagnostics.py, R18b/R18c audits): unwrap IPv4-mapped IPv6, reject
|
||||
loopback/link-local/private/multicast/reserved/unspecified, resolve DNS off the
|
||||
event loop, and pin the aiohttp connector to IPv4 so the family the guard
|
||||
classifies equals the family the connector dials (no dual-stack AAAA bypass).
|
||||
|
||||
Deployment note: this project uses ONLY public ACME CAs (e.g. Let's Encrypt), so
|
||||
every non-public IP is rejected — there is no internal/private-IP CA to allow.
|
||||
Residual: DNS rebinding between validate-time and fetch-time is not fully closed
|
||||
(fetching by hostname keeps TLS cert validation working); the IPv4 pin +
|
||||
https-only + admin-gating + internal-only exposure keep this residual low.
|
||||
"""
|
||||
import asyncio
|
||||
import ipaddress
|
||||
import socket
|
||||
from typing import List
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import aiohttp
|
||||
|
||||
# Only https is legitimate for a public ACME directory URL.
|
||||
_ALLOWED_SCHEMES = {"https"}
|
||||
|
||||
|
||||
class SSRFValidationError(ValueError):
|
||||
"""Raised when a URL fails SSRF validation (bad scheme or non-public host)."""
|
||||
|
||||
|
||||
def is_public_ip(ip_str: str) -> bool:
|
||||
"""Return True only for globally-routable IPv4/IPv6 addresses.
|
||||
|
||||
Unwraps IPv4-mapped IPv6 (``::ffff:127.0.0.1``) before classification so an
|
||||
attacker-controlled AAAA record cannot smuggle loopback/metadata through the
|
||||
IPv6 checks.
|
||||
"""
|
||||
try:
|
||||
ip = ipaddress.ip_address(ip_str)
|
||||
except (ValueError, TypeError):
|
||||
return False
|
||||
if isinstance(ip, ipaddress.IPv6Address) and ip.ipv4_mapped is not None:
|
||||
ip = ip.ipv4_mapped
|
||||
if ip.is_loopback or ip.is_link_local or ip.is_private:
|
||||
return False
|
||||
if ip.is_multicast or ip.is_reserved or ip.is_unspecified:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
async def _resolve_ips(host: str, *, timeout: float = 5.0) -> List[str]:
|
||||
"""Resolve `host` to IPv4 addresses without blocking the event loop."""
|
||||
loop = asyncio.get_running_loop()
|
||||
_, _, ips = await asyncio.wait_for(
|
||||
loop.run_in_executor(None, socket.gethostbyname_ex, host),
|
||||
timeout=timeout,
|
||||
)
|
||||
return ips or []
|
||||
|
||||
|
||||
async def assert_public_url(url: str, *, timeout: float = 5.0) -> None:
|
||||
"""Validate that `url` is safe to fetch server-side.
|
||||
|
||||
Requirements: https scheme, and a host that either is a public IP literal or
|
||||
resolves entirely to public IPv4 addresses. Raises ``SSRFValidationError``
|
||||
otherwise. Intended to be called immediately before the outbound request,
|
||||
which MUST use ``safe_connector()`` and ``allow_redirects=False``.
|
||||
"""
|
||||
if not url or not isinstance(url, str):
|
||||
raise SSRFValidationError("A URL is required")
|
||||
parsed = urlparse(url.strip())
|
||||
if parsed.scheme.lower() not in _ALLOWED_SCHEMES:
|
||||
raise SSRFValidationError(f"URL scheme must be https (got '{parsed.scheme or 'none'}')")
|
||||
host = parsed.hostname
|
||||
if not host:
|
||||
raise SSRFValidationError("URL has no host")
|
||||
|
||||
# Literal IP host: classify directly, no DNS needed.
|
||||
try:
|
||||
ipaddress.ip_address(host)
|
||||
if not is_public_ip(host):
|
||||
raise SSRFValidationError(f"URL host {host} is not a public IP address")
|
||||
return
|
||||
except ValueError:
|
||||
pass # hostname, not an IP literal -> resolve below
|
||||
|
||||
try:
|
||||
ips = await _resolve_ips(host, timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
raise SSRFValidationError(f"DNS resolution timed out for {host}")
|
||||
except Exception as e: # socket.gaierror etc.
|
||||
raise SSRFValidationError(f"DNS resolution failed for {host}: {e}")
|
||||
|
||||
if not ips:
|
||||
raise SSRFValidationError(f"{host} did not resolve to any address")
|
||||
if not all(is_public_ip(ip) for ip in ips):
|
||||
raise SSRFValidationError(
|
||||
f"{host} resolves to a non-public IP {ips} — refusing to fetch (SSRF guard)"
|
||||
)
|
||||
|
||||
|
||||
def safe_connector() -> aiohttp.TCPConnector:
|
||||
"""IPv4-pinned aiohttp connector.
|
||||
|
||||
Forces the connect family to match what :func:`assert_public_url` classified
|
||||
(closes the dual-stack AAAA bypass). TLS verification stays ON (default), so
|
||||
the request must target the validated hostname. Always combine with
|
||||
``allow_redirects=False`` at the request call site.
|
||||
"""
|
||||
return aiohttp.TCPConnector(family=socket.AF_INET)
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"version": "1.10.2",
|
||||
"releaseName": "Dark mode fixes on Apply Management",
|
||||
"releaseDate": "2026-08-08"
|
||||
}
|
||||
@@ -10,7 +10,6 @@ services:
|
||||
dockerfile: Dockerfile
|
||||
volumes:
|
||||
- haproxy_configs:/etc/haproxy
|
||||
- ./version.json:/app/version.json:ro
|
||||
|
||||
frontend:
|
||||
image: haproxy-openmanager-frontend:localtest
|
||||
|
||||
+7
-2
@@ -22,7 +22,7 @@ services:
|
||||
|
||||
# Redis Cache
|
||||
redis:
|
||||
image: redis:7-alpine
|
||||
image: redis:8.8.0-alpine
|
||||
container_name: haproxy-openmanager-redis
|
||||
command: redis-server --maxmemory 2gb --maxmemory-policy volatile-lru --save ""
|
||||
ports:
|
||||
@@ -51,6 +51,9 @@ services:
|
||||
- LOG_LEVEL=INFO
|
||||
- PUBLIC_URL=http://localhost:8080
|
||||
- MANAGEMENT_BASE_URL=http://localhost:8080
|
||||
# Empty when unset on the host: the image CMD then falls back to
|
||||
# WEB_CONCURRENCY (uvicorn's native env) and finally to 1.
|
||||
- UVICORN_WORKERS=${UVICORN_WORKERS:-}
|
||||
volumes:
|
||||
- haproxy_configs:/etc/haproxy
|
||||
expose:
|
||||
@@ -96,7 +99,9 @@ services:
|
||||
|
||||
# Nginx Reverse Proxy
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
# Pinned to a patched release for the nginx "poolslip" advisory
|
||||
# (mainline <=1.31.0 affected; fixed in mainline 1.31.1+ / stable 1.30.2+).
|
||||
image: nginx:1.31.1-alpine
|
||||
container_name: haproxy-openmanager-nginx
|
||||
volumes:
|
||||
- ./nginx/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
|
||||
@@ -163,6 +163,11 @@ serve -s build -l 3000
|
||||
|
||||
Option B — copy to nginx (recommended, see next section).
|
||||
|
||||
> **Important: use the nginx URL on port 8080 as your entry point.** The `serve` option above (port 3000) hosts
|
||||
> only the static UI; there is no `/api` backend behind it, so the login page renders but cannot actually log you
|
||||
> in. nginx (next section) serves the UI *and* proxies `/api` to the backend on a single port, so do your login and
|
||||
> everyday use at `http://<server-ip>:8080`.
|
||||
|
||||
### Create a systemd service for the frontend (if using `serve`)
|
||||
|
||||
```bash
|
||||
@@ -268,10 +273,10 @@ curl -s -o /dev/null -w "%{http_code}" http://127.0.0.1:8080/
|
||||
# Login
|
||||
curl -s -X POST http://127.0.0.1:8080/api/auth/login \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"username": "admin", "password": "admin"}' | python3 -m json.tool
|
||||
-d '{"username": "admin", "password": "admin123"}' | python3 -m json.tool
|
||||
```
|
||||
|
||||
> **Default credentials**: `admin` / `admin` — change the password immediately after first login.
|
||||
> **Default credentials**: `admin` / `admin123` — change the password immediately after first login.
|
||||
|
||||
## 9. Firewall
|
||||
|
||||
@@ -290,8 +295,8 @@ sudo ufw allow 8080/tcp
|
||||
| Service | Port | URL |
|
||||
|---------|------|-----|
|
||||
| Backend API | 8000 | `http://127.0.0.1:8000/api/health` |
|
||||
| Frontend | 3000 | `http://127.0.0.1:3000` |
|
||||
| Nginx (unified) | 8080 | `http://your-server-ip:8080` |
|
||||
| Frontend (static only) | 3000 | `http://127.0.0.1:3000` (UI only, no API; not the login URL) |
|
||||
| Nginx (unified, entry point) | 8080 | `http://your-server-ip:8080` (use this) |
|
||||
| PostgreSQL | 5432 | local |
|
||||
| Redis | 6379 | local |
|
||||
|
||||
|
||||
Generated
+187
-172
@@ -1,26 +1,26 @@
|
||||
{
|
||||
"name": "haproxy-openmanager-frontend",
|
||||
"version": "1.6.0",
|
||||
"version": "1.9.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "haproxy-openmanager-frontend",
|
||||
"version": "1.6.0",
|
||||
"version": "1.9.0",
|
||||
"license": "AGPL-3.0-or-later",
|
||||
"dependencies": {
|
||||
"@ant-design/icons": "^5.0.0",
|
||||
"@monaco-editor/react": "^4.6.0",
|
||||
"ace-builds": "^1.23.4",
|
||||
"antd": "^5.2.0",
|
||||
"axios": "^1.3.0",
|
||||
"axios": "^1.16.0",
|
||||
"moment": "^2.29.0",
|
||||
"monaco-editor": "^0.36.0",
|
||||
"qrcode.react": "^4.0.0",
|
||||
"react": "^18.2.0",
|
||||
"react-ace": "^10.1.0",
|
||||
"react-dom": "^18.2.0",
|
||||
"react-router-dom": "^6.8.0",
|
||||
"react-router-dom": "^6.30.4",
|
||||
"react-window": "^1.8.10",
|
||||
"react18-json-view": "^0.2.9",
|
||||
"recharts": "^2.5.0"
|
||||
@@ -179,18 +179,18 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/core": {
|
||||
"version": "7.29.0",
|
||||
"resolved": "https://registry.npmjs.org/@babel/core/-/core-7.29.0.tgz",
|
||||
"integrity": "sha512-CGOfOJqWjg2qW/Mb6zNsDm+u5vFQ8DxXfbM09z69p5Z6+mE1ikP2jUXw+j42Pf1XTYED2Rni5f95npYeuwMDQA==",
|
||||
"version": "7.29.6",
|
||||
"resolved": "https://registry.npmjs.org/@babel/core/-/core-7.29.6.tgz",
|
||||
"integrity": "sha512-QdxmAo/ikZqqRGA8s43ww8lcql6naWRvEz0FFrl6MIlc7Gi6TroXnSdWa5U/kq6fzcpqpHesicQxFZIieZbyIA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/code-frame": "^7.29.0",
|
||||
"@babel/generator": "^7.29.0",
|
||||
"@babel/generator": "^7.29.6",
|
||||
"@babel/helper-compilation-targets": "^7.28.6",
|
||||
"@babel/helper-module-transforms": "^7.28.6",
|
||||
"@babel/helpers": "^7.28.6",
|
||||
"@babel/parser": "^7.29.0",
|
||||
"@babel/helpers": "^7.29.2",
|
||||
"@babel/parser": "^7.29.3",
|
||||
"@babel/template": "^7.28.6",
|
||||
"@babel/traverse": "^7.29.0",
|
||||
"@babel/types": "^7.29.0",
|
||||
@@ -239,14 +239,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/generator": {
|
||||
"version": "7.29.1",
|
||||
"resolved": "https://registry.npmjs.org/@babel/generator/-/generator-7.29.1.tgz",
|
||||
"integrity": "sha512-qsaF+9Qcm2Qv8SRIMMscAvG4O3lJ0F1GuMo5HR/Bp02LopNgnZBC/EkbevHFeGs4ls/oPz9v+Bsmzbkbe+0dUw==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/generator/-/generator-7.29.7.tgz",
|
||||
"integrity": "sha512-DkXD5OJQaAQIdZ1bt3UZdEnHAn9Imd3IVBdX03UFe+ony9Ojw5pzr9YVKGDY1jt+Gcn/FnGkNf8r+Vj5NOJWtQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/parser": "^7.29.0",
|
||||
"@babel/types": "^7.29.0",
|
||||
"@babel/parser": "^7.29.7",
|
||||
"@babel/types": "^7.29.7",
|
||||
"@jridgewell/gen-mapping": "^0.3.12",
|
||||
"@jridgewell/trace-mapping": "^0.3.28",
|
||||
"jsesc": "^3.0.2"
|
||||
@@ -472,9 +472,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/helper-string-parser": {
|
||||
"version": "7.27.1",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.27.1.tgz",
|
||||
"integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
|
||||
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -482,9 +482,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/helper-validator-identifier": {
|
||||
"version": "7.28.5",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.28.5.tgz",
|
||||
"integrity": "sha512-qSs4ifwzKJSV39ucNjsvc6WVHs6b7S03sOh2OcHF9UHfVPqWWALUsNUVzhSBiItjRZoLHx7nIarVjqKVusUZ1Q==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.29.7.tgz",
|
||||
"integrity": "sha512-qehxGkRj55h/ff8EMaJ+cYhyaKlHIxqYDn682wQD7RNp9UujOQsHog2uS0r2vzr4pW+sXf90NeeayjcNaX3fFg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -531,13 +531,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/parser": {
|
||||
"version": "7.29.2",
|
||||
"resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.29.2.tgz",
|
||||
"integrity": "sha512-4GgRzy/+fsBa72/RZVJmGKPmZu9Byn8o4MoLpmNe1m8ZfYnz5emHLQz3U4gLud6Zwl0RZIcgiLD7Uq7ySFuDLA==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.29.7.tgz",
|
||||
"integrity": "sha512-hnORnjP/1P/zFEndoeX+n+t1RwWRJiJpM/jO7FW32Kn9r5+sJB2JWOdYo4L6k78j15eCwY3Gm/7364B1EMwtNg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/types": "^7.29.0"
|
||||
"@babel/types": "^7.29.7"
|
||||
},
|
||||
"bin": {
|
||||
"parser": "bin/babel-parser.js"
|
||||
@@ -2274,14 +2274,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/types": {
|
||||
"version": "7.29.0",
|
||||
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.0.tgz",
|
||||
"integrity": "sha512-LwdZHpScM4Qz8Xw2iKSzS+cfglZzJGvofQICy7W7v4caru4EaAmyUuO6BGrbyQ2mYV11W0U8j5mBhd14dd3B0A==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.7.tgz",
|
||||
"integrity": "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/helper-string-parser": "^7.27.1",
|
||||
"@babel/helper-validator-identifier": "^7.28.5"
|
||||
"@babel/helper-string-parser": "^7.29.7",
|
||||
"@babel/helper-validator-identifier": "^7.29.7"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6.9.0"
|
||||
@@ -2669,10 +2669,20 @@
|
||||
"license": "Python-2.0"
|
||||
},
|
||||
"node_modules/@eslint/eslintrc/node_modules/js-yaml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz",
|
||||
"integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==",
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz",
|
||||
"integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/puzrin"
|
||||
},
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/nodeca"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
@@ -2756,6 +2766,20 @@
|
||||
"node": ">=6"
|
||||
}
|
||||
},
|
||||
"node_modules/@istanbuljs/load-nyc-config/node_modules/js-yaml": {
|
||||
"version": "3.15.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.15.0.tgz",
|
||||
"integrity": "sha512-ttBQIIQPDeLjpPOohtUdXuXUVoA2uIB6fEH9HyJ7234s5mBJ5wTx20njxplLZQgLaOfpmPQA7X2t5AX6tIPbog==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^1.0.7",
|
||||
"esprima": "^4.0.0"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/@istanbuljs/schema": {
|
||||
"version": "0.1.3",
|
||||
"resolved": "https://registry.npmjs.org/@istanbuljs/schema/-/schema-0.1.3.tgz",
|
||||
@@ -4260,9 +4284,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@remix-run/router": {
|
||||
"version": "1.23.2",
|
||||
"resolved": "https://registry.npmjs.org/@remix-run/router/-/router-1.23.2.tgz",
|
||||
"integrity": "sha512-Ic6m2U/rMjTkhERIa/0ZtXJP17QUi2CbWE7cqx4J58M8aA3QTfW+2UlQ4psvTX9IO1RfNVhK3pcpdjej7L+t2w==",
|
||||
"version": "1.23.3",
|
||||
"resolved": "https://registry.npmjs.org/@remix-run/router/-/router-1.23.3.tgz",
|
||||
"integrity": "sha512-4An71tdz9X8+3sI4Qqqd2LWd9vS39J7sqd9EU4Scw7TJE/qB10Flv/UuqbPVgfQV9XoK8Np6jNquZitnZq5i+Q==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=14.0.0"
|
||||
@@ -6545,16 +6569,32 @@
|
||||
}
|
||||
},
|
||||
"node_modules/axios": {
|
||||
"version": "1.14.0",
|
||||
"resolved": "https://registry.npmjs.org/axios/-/axios-1.14.0.tgz",
|
||||
"integrity": "sha512-3Y8yrqLSwjuzpXuZ0oIYZ/XGgLwUIBU3uLvbcpb0pidD9ctpShJd43KSlEEkVQg6DS0G9NKyzOvBfUtDKEyHvQ==",
|
||||
"version": "1.16.0",
|
||||
"resolved": "https://registry.npmjs.org/axios/-/axios-1.16.0.tgz",
|
||||
"integrity": "sha512-6hp5CwvTPlN2A31g5dxnwAX0orzM7pmCRDLnZSX772mv8WDqICwFjowHuPs04Mc8deIld1+ejhtaMn5vp6b+1w==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"follow-redirects": "^1.15.11",
|
||||
"follow-redirects": "^1.16.0",
|
||||
"form-data": "^4.0.5",
|
||||
"proxy-from-env": "^2.1.0"
|
||||
}
|
||||
},
|
||||
"node_modules/axios/node_modules/form-data": {
|
||||
"version": "4.0.6",
|
||||
"resolved": "https://registry.npmjs.org/form-data/-/form-data-4.0.6.tgz",
|
||||
"integrity": "sha512-vKatAh4SlVfgbv+YtmhiRjhEMJsYpsG1Y2rMQtR+SVSbytsSD1YGzDIcrAJmdFec88u/+VoGmxnl+80gL1tRCQ==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"asynckit": "^0.4.0",
|
||||
"combined-stream": "^1.0.8",
|
||||
"es-set-tostringtag": "^2.1.0",
|
||||
"hasown": "^2.0.4",
|
||||
"mime-types": "^2.1.35"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 6"
|
||||
}
|
||||
},
|
||||
"node_modules/axobject-query": {
|
||||
"version": "4.1.0",
|
||||
"resolved": "https://registry.npmjs.org/axobject-query/-/axobject-query-4.1.0.tgz",
|
||||
@@ -6924,9 +6964,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/body-parser": {
|
||||
"version": "1.20.4",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-1.20.4.tgz",
|
||||
"integrity": "sha512-ZTgYYLMOXY9qKU/57FAo8F+HA2dGX7bqGc71txDRC1rS4frdFI5R7NhluHxH6M0YItAP0sHB4uqAOcYKxO6uGA==",
|
||||
"version": "1.20.5",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-1.20.5.tgz",
|
||||
"integrity": "sha512-3grm+/2tUOvu2cjJkvsIxrv/wVpfXQW4PsQHYm7yk4vfpu7Ekl6nEsYBoJUL6qDwZUx8wUhQ8tR2qz+ad9c9OA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -6938,7 +6978,7 @@
|
||||
"http-errors": "~2.0.1",
|
||||
"iconv-lite": "~0.4.24",
|
||||
"on-finished": "~2.4.1",
|
||||
"qs": "~6.14.0",
|
||||
"qs": "~6.15.1",
|
||||
"raw-body": "~2.5.3",
|
||||
"type-is": "~1.6.18",
|
||||
"unpipe": "~1.0.0"
|
||||
@@ -9800,10 +9840,20 @@
|
||||
}
|
||||
},
|
||||
"node_modules/eslint/node_modules/js-yaml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz",
|
||||
"integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==",
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz",
|
||||
"integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/puzrin"
|
||||
},
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/nodeca"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
@@ -10023,15 +10073,15 @@
|
||||
}
|
||||
},
|
||||
"node_modules/express": {
|
||||
"version": "4.22.1",
|
||||
"resolved": "https://registry.npmjs.org/express/-/express-4.22.1.tgz",
|
||||
"integrity": "sha512-F2X8g9P1X7uCPZMA3MVf9wcTqlyNp7IhH5qPCI0izhaOIYXaW9L535tGA3qmjRzpH+bZczqq7hVKxTR4NWnu+g==",
|
||||
"version": "4.22.2",
|
||||
"resolved": "https://registry.npmjs.org/express/-/express-4.22.2.tgz",
|
||||
"integrity": "sha512-IuL+Elrou2ZvCFHs18/CIzy2Nzvo25nZ1/D2eIZlz7c+QUayAcYoiM2BthCjs+EBHVpjYjcuLDAiCWgeIX3X1Q==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"accepts": "~1.3.8",
|
||||
"array-flatten": "1.1.1",
|
||||
"body-parser": "~1.20.3",
|
||||
"body-parser": "~1.20.5",
|
||||
"content-disposition": "~0.5.4",
|
||||
"content-type": "~1.0.4",
|
||||
"cookie": "~0.7.1",
|
||||
@@ -10050,7 +10100,7 @@
|
||||
"parseurl": "~1.3.3",
|
||||
"path-to-regexp": "~0.1.12",
|
||||
"proxy-addr": "~2.0.7",
|
||||
"qs": "~6.14.0",
|
||||
"qs": "~6.15.1",
|
||||
"range-parser": "~1.2.1",
|
||||
"safe-buffer": "5.2.1",
|
||||
"send": "~0.19.0",
|
||||
@@ -10147,9 +10197,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/fast-uri": {
|
||||
"version": "3.1.0",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.0.tgz",
|
||||
"integrity": "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA==",
|
||||
"version": "3.1.2",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz",
|
||||
"integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -10414,9 +10464,9 @@
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/follow-redirects": {
|
||||
"version": "1.15.11",
|
||||
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.11.tgz",
|
||||
"integrity": "sha512-deG2P0JfjrTxl50XGCDyfI97ZGVCxIpfKYmfyrQ54n5FO/0gfIES8C/Psl6kWVDolizcaaxZJnTS0QSMxvnsBQ==",
|
||||
"version": "1.16.0",
|
||||
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.16.0.tgz",
|
||||
"integrity": "sha512-y5rN/uOsadFT/JfYwhxRS5R7Qce+g3zG97+JrtFZlC9klX/W5hD7iiLzScI4nZqUS7DNUdhPgw4xI8W2LuXlUw==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "individual",
|
||||
@@ -10581,22 +10631,6 @@
|
||||
"node": ">=6"
|
||||
}
|
||||
},
|
||||
"node_modules/form-data": {
|
||||
"version": "4.0.5",
|
||||
"resolved": "https://registry.npmjs.org/form-data/-/form-data-4.0.5.tgz",
|
||||
"integrity": "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"asynckit": "^0.4.0",
|
||||
"combined-stream": "^1.0.8",
|
||||
"es-set-tostringtag": "^2.1.0",
|
||||
"hasown": "^2.0.2",
|
||||
"mime-types": "^2.1.12"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 6"
|
||||
}
|
||||
},
|
||||
"node_modules/forwarded": {
|
||||
"version": "0.2.0",
|
||||
"resolved": "https://registry.npmjs.org/forwarded/-/forwarded-0.2.0.tgz",
|
||||
@@ -11103,9 +11137,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/hasown": {
|
||||
"version": "2.0.2",
|
||||
"resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.2.tgz",
|
||||
"integrity": "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ==",
|
||||
"version": "2.0.4",
|
||||
"resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.4.tgz",
|
||||
"integrity": "sha512-T2UbfbBEF32wiepXIsMlTW9+dDYC6wMh/t/vYA4tuOMKqWz/n3vr1NFSxQiyP+zk2mXsoMA/i/7qV6LKut1t1A==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"function-bind": "^1.1.2"
|
||||
@@ -11365,9 +11399,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/http-proxy-middleware": {
|
||||
"version": "2.0.9",
|
||||
"resolved": "https://registry.npmjs.org/http-proxy-middleware/-/http-proxy-middleware-2.0.9.tgz",
|
||||
"integrity": "sha512-c1IyJYLYppU574+YI7R4QyX2ystMtVXZwIdzazUIPIJsHuWNd+mho2j+bKoHftndicGj9yh+xjd+l0yj7VeT1Q==",
|
||||
"version": "2.0.10",
|
||||
"resolved": "https://registry.npmjs.org/http-proxy-middleware/-/http-proxy-middleware-2.0.10.tgz",
|
||||
"integrity": "sha512-RKzRWNPxUZqbuk3BC5mGVJbBnWgr+diEnjJexIOytFbBzDy88Fbh/YvBr3DsNrl1jYAfjWfpATEv0NO35FDuPQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -15186,20 +15220,6 @@
|
||||
"integrity": "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "3.14.2",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.14.2.tgz",
|
||||
"integrity": "sha512-PMSmkqxr106Xa156c2M265Z+FTrPl+oxd/rgOQy2tijQeK5TxQ43psO1ZCwhVOSdnn+RzkzlRz/eY4BgJBYVpg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^1.0.7",
|
||||
"esprima": "^4.0.0"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/jsdom": {
|
||||
"version": "16.7.0",
|
||||
"resolved": "https://registry.npmjs.org/jsdom/-/jsdom-16.7.0.tgz",
|
||||
@@ -15248,16 +15268,16 @@
|
||||
}
|
||||
},
|
||||
"node_modules/jsdom/node_modules/form-data": {
|
||||
"version": "3.0.4",
|
||||
"resolved": "https://registry.npmjs.org/form-data/-/form-data-3.0.4.tgz",
|
||||
"integrity": "sha512-f0cRzm6dkyVYV3nPoooP8XlccPQukegwhAnpoLcXy+X+A8KfpGOoXwDr9FLZd3wzgLaBGQBE3lY93Zm/i1JvIQ==",
|
||||
"version": "3.0.5",
|
||||
"resolved": "https://registry.npmjs.org/form-data/-/form-data-3.0.5.tgz",
|
||||
"integrity": "sha512-j23EibVLnp4zNXGW7LjryXYa2X6U/M96yoOX+ybZxwkYajdxRNEqYY3zhh7y0i6kfISKS2jr+EJq1YTUDEv5+w==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"asynckit": "^0.4.0",
|
||||
"combined-stream": "^1.0.8",
|
||||
"es-set-tostringtag": "^2.1.0",
|
||||
"hasown": "^2.0.2",
|
||||
"hasown": "^2.0.4",
|
||||
"mime-types": "^2.1.35"
|
||||
},
|
||||
"engines": {
|
||||
@@ -15452,14 +15472,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/launch-editor": {
|
||||
"version": "2.13.2",
|
||||
"resolved": "https://registry.npmjs.org/launch-editor/-/launch-editor-2.13.2.tgz",
|
||||
"integrity": "sha512-4VVDnbOpLXy/s8rdRCSXb+zfMeFR0WlJWpET1iA9CQdlZDfwyLjUuGQzXU4VeOoey6AicSAluWan7Etga6Kcmg==",
|
||||
"version": "2.14.1",
|
||||
"resolved": "https://registry.npmjs.org/launch-editor/-/launch-editor-2.14.1.tgz",
|
||||
"integrity": "sha512-QWBrQsMpH7gPr965dsKD/3cKWiNoTjpATQf++Xq63N6sKRGMwlVXz41O1IZTMfZQgBctD/K5Zt06+/I6pP6+HA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"picocolors": "^1.1.1",
|
||||
"shell-quote": "^1.8.3"
|
||||
"shell-quote": "^1.8.4"
|
||||
}
|
||||
},
|
||||
"node_modules/leven": {
|
||||
@@ -16706,9 +16726,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/postcss": {
|
||||
"version": "8.5.8",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.8.tgz",
|
||||
"integrity": "sha512-OW/rX8O/jXnm82Ey1k44pObPtdblfiuWnrd8X7GJ7emImCOstunGbXUpp7HdBrFQX6rJzn3sPT397Wp5aCwCHg==",
|
||||
"version": "8.5.10",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.10.tgz",
|
||||
"integrity": "sha512-pMMHxBOZKFU6HgAZ4eyGnwXF/EvPGGqUr0MnZ5+99485wwW41kW91A4LOGxSHhgugZmSChL5AlElNdwlNgcnLQ==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -18249,9 +18269,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/qs": {
|
||||
"version": "6.14.2",
|
||||
"resolved": "https://registry.npmjs.org/qs/-/qs-6.14.2.tgz",
|
||||
"integrity": "sha512-V/yCWTTF7VJ9hIh18Ugr2zhJMP01MY7c5kh4J870L7imm6/DIzBsNLTXzMwUA3yZ5b/KBqLx8Kp3uRvd7xSe3Q==",
|
||||
"version": "6.15.2",
|
||||
"resolved": "https://registry.npmjs.org/qs/-/qs-6.15.2.tgz",
|
||||
"integrity": "sha512-Rzq0KEyX/w/tEybncDgdkZrJgVUsUMk3xjh3t5bv3S1HTAtg+uOYt72+ZfwiQwKdysThkTBdL/rTi6HDmX9Ddw==",
|
||||
"dev": true,
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
@@ -19176,12 +19196,12 @@
|
||||
}
|
||||
},
|
||||
"node_modules/react-router": {
|
||||
"version": "6.30.3",
|
||||
"resolved": "https://registry.npmjs.org/react-router/-/react-router-6.30.3.tgz",
|
||||
"integrity": "sha512-XRnlbKMTmktBkjCLE8/XcZFlnHvr2Ltdr1eJX4idL55/9BbORzyZEaIkBFDhFGCEWBBItsVrDxwx3gnisMitdw==",
|
||||
"version": "6.30.4",
|
||||
"resolved": "https://registry.npmjs.org/react-router/-/react-router-6.30.4.tgz",
|
||||
"integrity": "sha512-SVUsDe+DybHM/WmYKIVYhZh1o5Dcuf16yM6WjG02Q9XVFMZIJyHYhwrr6bFBXZkVP6z69kNkMyBCujt8FaFLJA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@remix-run/router": "1.23.2"
|
||||
"@remix-run/router": "1.23.3"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=14.0.0"
|
||||
@@ -19191,13 +19211,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/react-router-dom": {
|
||||
"version": "6.30.3",
|
||||
"resolved": "https://registry.npmjs.org/react-router-dom/-/react-router-dom-6.30.3.tgz",
|
||||
"integrity": "sha512-pxPcv1AczD4vso7G4Z3TKcvlxK7g7TNt3/FNGMhfqyntocvYKj+GCatfigGDjbLozC4baguJ0ReCigoDJXb0ag==",
|
||||
"version": "6.30.4",
|
||||
"resolved": "https://registry.npmjs.org/react-router-dom/-/react-router-dom-6.30.4.tgz",
|
||||
"integrity": "sha512-q4HvNl+mmDdkS0g+MqiBZNteQJCuimWoOyHMy4T/RQLAn9Z29+E91QXRaxOujeMl2HTzRSS0KFPd7lxX3PjV0Q==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@remix-run/router": "1.23.2",
|
||||
"react-router": "6.30.3"
|
||||
"@remix-run/router": "1.23.3",
|
||||
"react-router": "6.30.4"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=14.0.0"
|
||||
@@ -19688,32 +19708,20 @@
|
||||
}
|
||||
},
|
||||
"node_modules/resolve-url-loader": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/resolve-url-loader/-/resolve-url-loader-4.0.0.tgz",
|
||||
"integrity": "sha512-05VEMczVREcbtT7Bz+C+96eUO5HDNvdthIiMB34t7FcF8ehcu4wC0sSgPUubs3XW2Q3CNLJk/BJrCU9wVRymiA==",
|
||||
"version": "5.0.0",
|
||||
"resolved": "https://registry.npmjs.org/resolve-url-loader/-/resolve-url-loader-5.0.0.tgz",
|
||||
"integrity": "sha512-uZtduh8/8srhBoMx//5bwqjQ+rfYOUq8zC9NrMUGtjBiGTtFJM42s58/36+hTqeqINcnYe08Nj3LkK9lW4N8Xg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"adjust-sourcemap-loader": "^4.0.0",
|
||||
"convert-source-map": "^1.7.0",
|
||||
"loader-utils": "^2.0.0",
|
||||
"postcss": "^7.0.35",
|
||||
"postcss": "^8.2.14",
|
||||
"source-map": "0.6.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=8.9"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"rework": "1.0.1",
|
||||
"rework-visit": "1.0.0"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"rework": {
|
||||
"optional": true
|
||||
},
|
||||
"rework-visit": {
|
||||
"optional": true
|
||||
}
|
||||
"node": ">=12"
|
||||
}
|
||||
},
|
||||
"node_modules/resolve-url-loader/node_modules/convert-source-map": {
|
||||
@@ -19723,31 +19731,6 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/resolve-url-loader/node_modules/picocolors": {
|
||||
"version": "0.2.1",
|
||||
"resolved": "https://registry.npmjs.org/picocolors/-/picocolors-0.2.1.tgz",
|
||||
"integrity": "sha512-cMlDqaLEqfSaW8Z7N5Jw+lyIW869EzT73/F5lhtY9cLGoVxSXznfgfXMO0Z5K0o0Q2TkTXq+0KFsdnSe3jDViA==",
|
||||
"dev": true,
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/resolve-url-loader/node_modules/postcss": {
|
||||
"version": "7.0.39",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-7.0.39.tgz",
|
||||
"integrity": "sha512-yioayjNbHn6z1/Bywyb2Y4s3yvDAeXGOyxqD+LnVOinq6Mdmd++SW2wUNVzavyyHxd6+DxzWGIuosg6P1Rj8uA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"picocolors": "^0.2.1",
|
||||
"source-map": "^0.6.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6.0.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/postcss/"
|
||||
}
|
||||
},
|
||||
"node_modules/resolve-url-loader/node_modules/source-map": {
|
||||
"version": "0.6.1",
|
||||
"resolved": "https://registry.npmjs.org/source-map/-/source-map-0.6.1.tgz",
|
||||
@@ -20368,9 +20351,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/shell-quote": {
|
||||
"version": "1.8.3",
|
||||
"resolved": "https://registry.npmjs.org/shell-quote/-/shell-quote-1.8.3.tgz",
|
||||
"integrity": "sha512-ObmnIF4hXNg1BqhnHmgbDETF8dLPCggZWBjkQfhZpbszZnYur5DUljTcCHii5LC3J5E0yeO/1LIMyH+UvHQgyw==",
|
||||
"version": "1.8.4",
|
||||
"resolved": "https://registry.npmjs.org/shell-quote/-/shell-quote-1.8.4.tgz",
|
||||
"integrity": "sha512-VsC6n6vz1ihYYyZZwX7YZSF5l5x36ca17OC+a69h94YqB7X6XLwf+5MOgynYir2SLFUbl8gIYvBo8K8RoNQ6bQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -21212,6 +21195,20 @@
|
||||
"node": ">=4"
|
||||
}
|
||||
},
|
||||
"node_modules/svgo/node_modules/js-yaml": {
|
||||
"version": "3.15.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.15.0.tgz",
|
||||
"integrity": "sha512-ttBQIIQPDeLjpPOohtUdXuXUVoA2uIB6fEH9HyJ7234s5mBJ5wTx20njxplLZQgLaOfpmPQA7X2t5AX6tIPbog==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^1.0.7",
|
||||
"esprima": "^4.0.0"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/svgo/node_modules/nth-check": {
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/nth-check/-/nth-check-1.0.2.tgz",
|
||||
@@ -21336,6 +21333,24 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tailwindcss/node_modules/yaml": {
|
||||
"version": "2.9.0",
|
||||
"resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.0.tgz",
|
||||
"integrity": "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==",
|
||||
"dev": true,
|
||||
"license": "ISC",
|
||||
"optional": true,
|
||||
"peer": true,
|
||||
"bin": {
|
||||
"yaml": "bin.mjs"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 14.6"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/eemeli"
|
||||
}
|
||||
},
|
||||
"node_modules/tapable": {
|
||||
"version": "2.3.2",
|
||||
"resolved": "https://registry.npmjs.org/tapable/-/tapable-2.3.2.tgz",
|
||||
@@ -22200,7 +22215,7 @@
|
||||
}
|
||||
},
|
||||
"node_modules/wbuf": {
|
||||
"version": "1.7.3",
|
||||
"version": "1.7.8",
|
||||
"resolved": "https://registry.npmjs.org/wbuf/-/wbuf-1.7.3.tgz",
|
||||
"integrity": "sha512-O84QOnr0icsbFGLS0O3bI5FswxzRr8/gHwWkDlQFskhSPryQXvrTMxjxGP4+iWYoauLoBvfDpkrOauZ+0iZpDA==",
|
||||
"dev": true,
|
||||
@@ -22353,9 +22368,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/webpack-dev-server/node_modules/ws": {
|
||||
"version": "8.20.0",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-8.20.0.tgz",
|
||||
"integrity": "sha512-sAt8BhgNbzCtgGbt2OxmpuryO63ZoDk/sqaB/znQm94T4fCEsy/yV+7CdC1kJhOU9lboAEU7R3kquuycDoibVA==",
|
||||
"version": "8.21.0",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-8.21.0.tgz",
|
||||
"integrity": "sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
@@ -22450,9 +22465,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/websocket-driver": {
|
||||
"version": "0.7.4",
|
||||
"resolved": "https://registry.npmjs.org/websocket-driver/-/websocket-driver-0.7.4.tgz",
|
||||
"integrity": "sha512-b17KeDIQVjvb0ssuSDF2cYXSg2iztliJ4B9WdsuB6J952qCPKmnVq4DyW5motImXHDC1cBT/1UezrJVsKw5zjg==",
|
||||
"version": "0.7.5",
|
||||
"resolved": "https://registry.npmjs.org/websocket-driver/-/websocket-driver-0.7.5.tgz",
|
||||
"integrity": "sha512-ZL2+3c7kMBdIRCMz6l8jQMHyGVxj+UL+xVk74Ombiciboca8rHa15L86B19E5oh1pL9Ii/uj54gtsIrZGMo6zA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
@@ -23031,9 +23046,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/ws": {
|
||||
"version": "7.5.10",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-7.5.10.tgz",
|
||||
"integrity": "sha512-+dbF1tHwZpXcbOJdVOkzLDxZP1ailvSxM6ZweXTegylPny803bFhA+vqBYw4s31NSAk4S2Qz+AKXK9a4wkdjcQ==",
|
||||
"version": "7.5.11",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-7.5.11.tgz",
|
||||
"integrity": "sha512-zS54Oen9bITtp7kp2XM3AydrCIq1D+HwJOuH+c+e4LfpL/lotP5osijd+UoMnxwAam1GN8R4KtLAyIrIcBNpiA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
|
||||
+18
-3
@@ -1,13 +1,13 @@
|
||||
{
|
||||
"name": "haproxy-openmanager-frontend",
|
||||
"version": "1.6.0",
|
||||
"version": "1.10.2",
|
||||
"description": "HAProxy Load Balancer Management UI",
|
||||
"license": "AGPL-3.0-or-later",
|
||||
"dependencies": {
|
||||
"react": "^18.2.0",
|
||||
"react-dom": "^18.2.0",
|
||||
"react-router-dom": "^6.8.0",
|
||||
"axios": "^1.3.0",
|
||||
"react-router-dom": "^6.30.4",
|
||||
"axios": "^1.16.0",
|
||||
"antd": "^5.2.0",
|
||||
"@ant-design/icons": "^5.0.0",
|
||||
"recharts": "^2.5.0",
|
||||
@@ -30,6 +30,21 @@
|
||||
"@testing-library/user-event": "^14.4.3",
|
||||
"@babel/plugin-proposal-private-property-in-object": "^7.21.0"
|
||||
},
|
||||
"overrides": {
|
||||
"ws": "7.5.11",
|
||||
"webpack-dev-server": { "ws": "8.21.0" },
|
||||
"form-data": "4.0.6",
|
||||
"jsdom": { "form-data": "3.0.5" },
|
||||
"js-yaml": "3.15.0",
|
||||
"eslint": { "js-yaml": "4.2.0" },
|
||||
"@eslint/eslintrc": { "js-yaml": "4.2.0" },
|
||||
"http-proxy-middleware": "2.0.10",
|
||||
"launch-editor": "2.14.1",
|
||||
"postcss": "8.5.10",
|
||||
"resolve-url-loader": "5.0.0",
|
||||
"@babel/core": "7.29.6",
|
||||
"websocket-driver": "0.7.5"
|
||||
},
|
||||
"scripts": {
|
||||
"start": "react-scripts start",
|
||||
"build": "react-scripts build",
|
||||
|
||||
+31
-5
@@ -5,6 +5,7 @@ import {
|
||||
DashboardOutlined,
|
||||
SettingOutlined,
|
||||
CloudServerOutlined,
|
||||
DeploymentUnitOutlined,
|
||||
FileTextOutlined,
|
||||
GlobalOutlined,
|
||||
SafetyCertificateOutlined,
|
||||
@@ -42,6 +43,7 @@ import PoolManagement from './components/PoolManagement';
|
||||
import Login from './components/Login';
|
||||
import ClusterSelector from './components/ClusterSelector';
|
||||
import ClusterManagement from './components/ClusterManagement';
|
||||
import VIPManagement from './components/VIPManagement';
|
||||
import Configuration from './components/Configuration';
|
||||
import APIDocumentation from './components/APIDocumentation';
|
||||
import IPInventory from './components/IPInventory';
|
||||
@@ -156,6 +158,11 @@ const { Text } = Typography;
|
||||
icon: <CloudServerOutlined />,
|
||||
label: <Link to="/clusters">Clusters</Link>,
|
||||
},
|
||||
{
|
||||
key: '/ha-vip',
|
||||
icon: <DeploymentUnitOutlined />,
|
||||
label: <Link to="/ha-vip">HA / VIP</Link>,
|
||||
},
|
||||
{
|
||||
key: '/pools',
|
||||
icon: <GlobalOutlined />,
|
||||
@@ -466,6 +473,7 @@ function AppContent() {
|
||||
<Route path="/pools" element={<PoolManagement />} />
|
||||
<Route path="/settings" element={<Settings />} />
|
||||
<Route path="/clusters" element={<ClusterManagement />} />
|
||||
<Route path="/ha-vip" element={<VIPManagement />} />
|
||||
<Route path="/ip-inventory" element={<IPInventory />} />
|
||||
</Routes>
|
||||
</Content>
|
||||
@@ -476,12 +484,30 @@ function AppContent() {
|
||||
|
||||
function ThemedApp() {
|
||||
const { isDarkMode } = useTheme();
|
||||
const antdTheme = {
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
};
|
||||
|
||||
// Ant Design 5: the STATIC message/notification/Modal.confirm APIs render into their own
|
||||
// detached root, so they do not see this ConfigProvider and always fall back to the light
|
||||
// algorithm — a confirm dialog came up white while the app was in dark mode. `holderRender`
|
||||
// wraps that detached root in the same ConfigProvider, which fixes every static call in the
|
||||
// app at once (12 components use Modal.confirm) instead of migrating each one to
|
||||
// App.useApp(). In an effect rather than during render: ConfigProvider.config() mutates
|
||||
// antd module state, and effects still run long before a user can click anything that opens
|
||||
// a static modal. Re-registered on theme change so the toggle takes effect immediately.
|
||||
React.useEffect(() => {
|
||||
ConfigProvider.config({
|
||||
holderRender: (children) => (
|
||||
<ConfigProvider theme={{ algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm }}>
|
||||
{children}
|
||||
</ConfigProvider>
|
||||
),
|
||||
});
|
||||
}, [isDarkMode]);
|
||||
|
||||
return (
|
||||
<ConfigProvider
|
||||
theme={{
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
}}
|
||||
>
|
||||
<ConfigProvider theme={antdTheme}>
|
||||
<AuthProvider>
|
||||
<ClusterProvider>
|
||||
<ProgressProvider>
|
||||
|
||||
@@ -53,19 +53,19 @@ const MATCH_TYPE_GROUPS = [
|
||||
{ label: 'Advanced', options: MATCH_TYPES.filter(m => m.category === 'Advanced') },
|
||||
];
|
||||
|
||||
// Phase K Phase D follow-up (Bulgu #12 round 3) — the `-f <file>`
|
||||
// flag was removed from the visual builder because HAProxy OpenManager
|
||||
// does not provision pattern files onto the HAProxy node filesystem.
|
||||
// Allowing `-f` in the visual builder produced ACL rules that passed
|
||||
// every UI / Pydantic / heuristic check but ALWAYS failed HAProxy's
|
||||
// real `-c` parse at apply time with "failed to open pattern file".
|
||||
// Operators reported a multi-page wizard run ending at the Apply
|
||||
// Management red-badge for a footgun the UI made trivial to step on.
|
||||
// The Pydantic validators on the manual API + wizard reject `-f`
|
||||
// universally; the visual builder simply removes the option from the
|
||||
// dropdown so operators cannot author the unsupported state.
|
||||
// Issue #38 follow-up — `-f <file>` is back in the visual builder:
|
||||
// the Bulgu #12 removal (and the matching Pydantic rejects) assumed a
|
||||
// missing pattern file would surprise the operator at apply time, but
|
||||
// the agent runs `haproxy -c` before every reload so a missing file
|
||||
// fails safely (previous config keeps running), and bulk import plus
|
||||
// the free-form fields always accepted `-f`. Pattern files are
|
||||
// operator-managed host files, same policy as SPOE filter configs
|
||||
// (v1.8.8). The value field carries the file path (e.g. flag `-f`
|
||||
// + value `/etc/haproxy/blacklist.lst`); an informational note is
|
||||
// rendered on rules that use it.
|
||||
const FLAGS = [
|
||||
{ value: '-i', label: '-i (case insensitive)' },
|
||||
{ value: '-f', label: '-f (pattern file on host)' },
|
||||
{ value: '-m beg', label: '-m beg (begins with)' },
|
||||
{ value: '-m end', label: '-m end (ends with)' },
|
||||
{ value: '-m sub', label: '-m sub (contains)' },
|
||||
@@ -396,25 +396,23 @@ function ACLDefinitionCard({ rule, index, onChange, onDelete }) {
|
||||
};
|
||||
const isRaw = rule.raw !== undefined;
|
||||
|
||||
// Phase K Phase D follow-up (Bulgu #12 round 3) — surface `-f` flag
|
||||
// usage inline. The Pydantic validator rejects the rule server-side,
|
||||
// but operators benefit from seeing the error AS they type / when
|
||||
// they re-open a draft that carries a `-f`-flagged rule (e.g. from
|
||||
// a pre-fix draft). The error message matches the Pydantic error
|
||||
// verbatim so support flows are consistent.
|
||||
// Issue #38 follow-up — `-f <file>` pattern-file references are
|
||||
// ACCEPTED now (the Bulgu #12 reject was removed server-side too).
|
||||
// We still detect them, but only to render an informational note:
|
||||
// the referenced file is operator-managed and must exist on every
|
||||
// HAProxy host; a missing file fails safely at the agent's
|
||||
// pre-reload `haproxy -c`.
|
||||
const rawHasFileFlag = isRaw && typeof rule.raw === 'string' && ACL_FILE_FLAG_PATTERN.test(rule.raw);
|
||||
const structuredHasFileFlag =
|
||||
!isRaw && Array.isArray(rule.flags) && rule.flags.includes('-f');
|
||||
const hasFileFlag = rawHasFileFlag || structuredHasFileFlag;
|
||||
const cardStyleWithError = hasFileFlag
|
||||
? { ...ruleCardStyle, border: `1px solid ${token.colorError}` }
|
||||
: ruleCardStyle;
|
||||
const cardStyleWithError = ruleCardStyle;
|
||||
const FILE_FLAG_TOOLTIP =
|
||||
"ACL pattern-file references (-f <file>) are not supported by "
|
||||
+ "HAProxy OpenManager: the product does not provision pattern "
|
||||
+ "files onto the HAProxy node filesystem, so the reference "
|
||||
+ "would fail at HAProxy reload time. Remove '-f' and use inline "
|
||||
+ "values instead.";
|
||||
"This rule references a pattern file (-f <file>). The file must "
|
||||
+ "exist at that exact path on every HAProxy host in the cluster — "
|
||||
+ "HAProxy OpenManager does not create or distribute pattern files. "
|
||||
+ "A missing file fails safely at 'haproxy -c' (the previous config "
|
||||
+ "keeps running).";
|
||||
|
||||
if (isRaw) {
|
||||
return (
|
||||
@@ -427,11 +425,10 @@ function ACLDefinitionCard({ rule, index, onChange, onDelete }) {
|
||||
onChange={(e) => onChange(index, { raw: e.target.value })}
|
||||
placeholder="Raw ACL rule (e.g. my_acl path_beg /api)"
|
||||
prefix={<Tag color="default" style={{ marginRight: 4 }}>RAW</Tag>}
|
||||
status={hasFileFlag ? 'error' : undefined}
|
||||
/>
|
||||
</Tooltip>
|
||||
{hasFileFlag && (
|
||||
<Text type="danger" style={{ fontSize: 11, display: 'block', marginTop: 2 }}>
|
||||
<Text type="secondary" style={{ fontSize: 11, display: 'block', marginTop: 2 }}>
|
||||
{FILE_FLAG_TOOLTIP}
|
||||
</Text>
|
||||
)}
|
||||
@@ -549,12 +546,11 @@ function ACLDefinitionCard({ rule, index, onChange, onDelete }) {
|
||||
onChange={(e) => onChange(index, { ...rule, value: e.target.value })}
|
||||
placeholder={matchDef?.placeholder || 'Value'}
|
||||
size="small"
|
||||
status={structuredHasFileFlag ? 'error' : undefined}
|
||||
/>
|
||||
);
|
||||
})()}
|
||||
{structuredHasFileFlag && (
|
||||
<Text type="danger" style={{ fontSize: 11, display: 'block', marginTop: 2 }}>
|
||||
<Text type="secondary" style={{ fontSize: 11, display: 'block', marginTop: 2 }}>
|
||||
{FILE_FLAG_TOOLTIP}
|
||||
</Text>
|
||||
)}
|
||||
@@ -891,11 +887,11 @@ export default function ACLRuleBuilder({ aclRules = [], useBackendRules = [], re
|
||||
.map(d => d.name);
|
||||
}, [aclDefs]);
|
||||
|
||||
// Phase K Phase D follow-up (Bulgu #12 round 3) — count rules that
|
||||
// still carry the unsupported `-f <file>` flag. Surfaced as a
|
||||
// section-level Alert so operators know the section as a whole
|
||||
// has invalid rules even if individual cards / raw text would
|
||||
// otherwise need scrolling to find them.
|
||||
// Issue #38 follow-up — count rules that reference a `-f <file>`
|
||||
// pattern file. Surfaced as a section-level informational Alert
|
||||
// (non-blocking): the file is operator-managed and must exist on
|
||||
// every HAProxy host; a missing file fails safely at the agent's
|
||||
// pre-reload `haproxy -c`.
|
||||
const fileFlagRuleCount = useMemo(() => {
|
||||
let count = 0;
|
||||
for (const d of aclDefs) {
|
||||
@@ -1153,19 +1149,19 @@ export default function ACLRuleBuilder({ aclRules = [], useBackendRules = [], re
|
||||
Define named conditions to match incoming requests by path, header, source IP, and more.
|
||||
</Text>
|
||||
|
||||
{/* Phase K Phase D follow-up (Bulgu #12 round 3) — section-
|
||||
level warning when one or more rules still carry the
|
||||
unsupported `-f <file>` pattern-file flag. Render as a
|
||||
blocking-style Alert so the operator notices BEFORE
|
||||
Submit. The Pydantic validator rejects the same shape
|
||||
server-side; this is the up-front authoring guardrail. */}
|
||||
{/* Issue #38 follow-up — section-level informational note
|
||||
when one or more rules reference `-f <file>` pattern
|
||||
files. Non-blocking: pattern files are operator-managed
|
||||
host files (the Bulgu #12 reject was removed) and a
|
||||
missing file fails safely at the agent's pre-reload
|
||||
`haproxy -c`. */}
|
||||
{fileFlagRuleCount > 0 && (
|
||||
<Alert
|
||||
type="error"
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 8 }}
|
||||
message={`${fileFlagRuleCount} ACL rule${fileFlagRuleCount === 1 ? '' : 's'} use the unsupported \`-f <file>\` flag`}
|
||||
description="HAProxy OpenManager does not provision pattern files onto the HAProxy node filesystem, so any `-f /path/...` reference would fail HAProxy reload at apply time with 'failed to open pattern file'. Remove the `-f` flag and switch to inline values (e.g. `src 10.0.0.0/24` instead of `src -f /etc/haproxy/admins.lst`)."
|
||||
message={`${fileFlagRuleCount} ACL rule${fileFlagRuleCount === 1 ? '' : 's'} reference a \`-f <file>\` pattern file`}
|
||||
description="The referenced file must exist at that exact path on every HAProxy host in the cluster — HAProxy OpenManager does not create or distribute pattern files. A missing file fails safely at 'haproxy -c' (the previous config keeps running)."
|
||||
/>
|
||||
)}
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import React, { useState, useEffect, useCallback, useRef } from 'react';
|
||||
import {
|
||||
Card, Table, Button, Tag, Space, Modal, Form, Input, Select, Steps,
|
||||
message, Row, Col, Statistic, Alert, Tooltip, Switch, theme, Segmented,
|
||||
Tabs, Timeline, Spin, Empty
|
||||
message, Row, Col, Statistic, Alert, Tooltip, Switch, theme, Segmented, Collapse,
|
||||
Tabs, Timeline, Spin, Empty, Typography, Divider
|
||||
} from 'antd';
|
||||
import {
|
||||
SafetyCertificateOutlined, PlusOutlined, ReloadOutlined,
|
||||
@@ -10,7 +10,7 @@ import {
|
||||
SyncOutlined, CloseCircleOutlined,
|
||||
DeleteOutlined, EyeOutlined,
|
||||
CloudDownloadOutlined, UserOutlined, InfoCircleOutlined,
|
||||
RocketOutlined, ExperimentOutlined
|
||||
RocketOutlined, ExperimentOutlined, KeyOutlined
|
||||
} from '@ant-design/icons';
|
||||
import { useNavigate } from 'react-router-dom';
|
||||
import { useCluster } from '../contexts/ClusterContext';
|
||||
@@ -18,8 +18,26 @@ import axios from 'axios';
|
||||
|
||||
const { Option } = Select;
|
||||
|
||||
const getErrorMsg = (err, fallback) =>
|
||||
err?.response?.data?.error?.message || err?.response?.data?.detail || fallback;
|
||||
const getErrorMsg = (err, fallback) => {
|
||||
const data = err?.response?.data;
|
||||
// FastAPI/Pydantic 422s wrap the specific field message in error.details.validation_errors[];
|
||||
// the top-level error.message is generic ("Validation error in request data"), so prefer the
|
||||
// field-level message (e.g. the EAB base64 hint) when present.
|
||||
const fieldMsg = data?.error?.details?.validation_errors?.[0]?.message;
|
||||
return fieldMsg || data?.error?.message || data?.detail || fallback;
|
||||
};
|
||||
|
||||
// Issue #35: humanize the dotted event_type tokens emitted for DNS-01 orders so the diagnostics
|
||||
// timeline reads as a step-by-step progress log rather than raw machine strings. Unknown types
|
||||
// fall back to the raw token.
|
||||
const EVENT_LABELS = {
|
||||
'acme.dns01.publish': 'Published DNS TXT record',
|
||||
'acme.dns01.responded': 'Asked the CA to validate',
|
||||
'acme.dns01.validation': 'DNS-01 validation result',
|
||||
'acme.dns01.cleanup': 'Removed DNS TXT record',
|
||||
'acme.order.requested': 'Certificate requested',
|
||||
};
|
||||
const humanizeEventType = (t) => EVENT_LABELS[t] || t;
|
||||
|
||||
// Render letsencrypt_orders.error_detail (TEXT column). Backend now writes
|
||||
// structured JSON-strings (stage / http_status / ca_response / timestamp) for
|
||||
@@ -83,6 +101,25 @@ const ACMEAutomation = () => {
|
||||
const [orderFilter, setOrderFilter] = useState('active');
|
||||
const { token } = theme.useToken();
|
||||
|
||||
// Issue #35: DNS-01 provider catalog + global enable flag (from GET /dns-providers).
|
||||
const [dnsProviders, setDnsProviders] = useState([]);
|
||||
const [dns01Enabled, setDns01Enabled] = useState(false);
|
||||
const [confirming, setConfirming] = useState(false);
|
||||
const regChallengeType = Form.useWatch('challenge_type', registerForm);
|
||||
const regDnsProvider = Form.useWatch('dns_provider', registerForm);
|
||||
const wizardAccountId = Form.useWatch('account_id', wizardForm);
|
||||
const wizardDomains = Form.useWatch('domains', wizardForm);
|
||||
const selectedDnsProvider = dnsProviders.find(p => p.name === regDnsProvider) || null;
|
||||
|
||||
// Issue #35: per-account DNS credential management (view/replace/clear after creation).
|
||||
const [credModalVisible, setCredModalVisible] = useState(false);
|
||||
const [credAccount, setCredAccount] = useState(null);
|
||||
const [credMeta, setCredMeta] = useState(null);
|
||||
const [credLoading, setCredLoading] = useState(false);
|
||||
const [credSaving, setCredSaving] = useState(false);
|
||||
const [credForm] = Form.useForm();
|
||||
const credProvider = credAccount ? (dnsProviders.find(p => p.name === credAccount.dns_provider) || null) : null;
|
||||
|
||||
// v1.5.0 Issue #13: ACME Diagnostic Panel state
|
||||
const [diagVisible, setDiagVisible] = useState(false);
|
||||
const [diagOrderId, setDiagOrderId] = useState(null);
|
||||
@@ -105,18 +142,23 @@ const ACMEAutomation = () => {
|
||||
const fetchData = useCallback(async () => {
|
||||
setLoading(true);
|
||||
try {
|
||||
const [ordersRes, accountsRes, renewalRes, clustersRes, prereqRes] = await Promise.allSettled([
|
||||
const [ordersRes, accountsRes, renewalRes, clustersRes, prereqRes, dnsRes] = await Promise.allSettled([
|
||||
axios.get('/api/letsencrypt/orders'),
|
||||
axios.get('/api/letsencrypt/accounts'),
|
||||
axios.get('/api/letsencrypt/renewal-schedule'),
|
||||
axios.get('/api/clusters'),
|
||||
axios.get('/api/letsencrypt/prerequisites'),
|
||||
axios.get('/api/letsencrypt/dns-providers'),
|
||||
]);
|
||||
if (ordersRes.status === 'fulfilled') setOrders(ordersRes.value.data || []);
|
||||
if (accountsRes.status === 'fulfilled') setAccounts(accountsRes.value.data || []);
|
||||
if (prereqRes.status === 'fulfilled') setPrerequisites(prereqRes.value.data);
|
||||
if (renewalRes.status === 'fulfilled') setRenewalSchedule(renewalRes.value.data || []);
|
||||
if (clustersRes.status === 'fulfilled') setClusters(clustersRes.value.data?.clusters || []);
|
||||
if (dnsRes.status === 'fulfilled') {
|
||||
setDnsProviders(dnsRes.value.data?.providers || []);
|
||||
setDns01Enabled(!!dnsRes.value.data?.dns01_enabled);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Error loading ACME data:', err);
|
||||
} finally {
|
||||
@@ -133,9 +175,29 @@ const ACMEAutomation = () => {
|
||||
o.status === 'pending' || o.status === 'processing' || o.status === 'ready' ||
|
||||
(o.status === 'valid' && !o.ssl_certificate_id)
|
||||
);
|
||||
// Track the order whose detail modal is open so the poll can refresh it WITHOUT
|
||||
// making the interval depend on orderDetail (which would recreate it every poll).
|
||||
const openDetailIdRef = useRef(null);
|
||||
const detailRefetchInFlightRef = useRef(false);
|
||||
useEffect(() => {
|
||||
openDetailIdRef.current = (detailVisible && orderDetail?.id) ? orderDetail.id : null;
|
||||
}, [detailVisible, orderDetail]);
|
||||
useEffect(() => {
|
||||
if (!hasInProgress) return undefined;
|
||||
const interval = setInterval(() => { fetchData(); }, 30000);
|
||||
const interval = setInterval(() => {
|
||||
fetchData();
|
||||
// Keep an open order-detail modal (e.g. a manual DNS-01 order awaiting validation)
|
||||
// in sync so its status / TXT block / Verify button cannot go stale. Skip if a prior
|
||||
// refetch is still in flight so slow backends don't pile up overlapping requests.
|
||||
const oid = openDetailIdRef.current;
|
||||
if (oid && !detailRefetchInFlightRef.current) {
|
||||
detailRefetchInFlightRef.current = true;
|
||||
axios.get(`/api/letsencrypt/orders/${oid}`)
|
||||
.then((r) => setOrderDetail((prev) => (prev && prev.id === oid ? r.data : prev)))
|
||||
.catch(() => { /* transient; the next poll retries */ })
|
||||
.finally(() => { detailRefetchInFlightRef.current = false; });
|
||||
}
|
||||
}, 30000);
|
||||
return () => clearInterval(interval);
|
||||
}, [hasInProgress, fetchData]);
|
||||
|
||||
@@ -144,7 +206,10 @@ const ACMEAutomation = () => {
|
||||
if (!expiryDate) return null;
|
||||
return Math.ceil((new Date(expiryDate) - new Date()) / (1000 * 60 * 60 * 24));
|
||||
};
|
||||
const nextRenewal = renewalSchedule.find(c => c.auto_renew && calcDaysLeft(c.expiry_date) > 0);
|
||||
// Manual DNS-01 certs are not auto-renewed (even a legacy row left at auto_renew=TRUE), so they
|
||||
// must not drive the "Next Renewal" countdown, which implies an automated event.
|
||||
const isManualDnsCert = (c) => c.challenge_type === 'dns-01' && (c.dns_provider || 'manual') === 'manual';
|
||||
const nextRenewal = renewalSchedule.find(c => c.auto_renew && !isManualDnsCert(c) && calcDaysLeft(c.expiry_date) > 0);
|
||||
const nextRenewalDays = nextRenewal ? calcDaysLeft(nextRenewal.expiry_date) : null;
|
||||
// Issue #11/#12: an order is "in progress" if it's pre-valid OR valid-but-not-downloaded (stuck).
|
||||
// Including 'ready' here ensures the dashboard counter & UI auto-refresh react to all in-flight states.
|
||||
@@ -155,6 +220,21 @@ const ACMEAutomation = () => {
|
||||
const activeAccount = accounts.find(a => a.status === 'valid') || null;
|
||||
const acmeAccount = activeAccount || (accounts.length > 0 ? accounts[accounts.length - 1] : null);
|
||||
const acmeEnabledClusters = clusters.filter(c => c.acme_enabled && c.is_active);
|
||||
// Issue #35: the cert wizard adapts to the selected account's challenge method.
|
||||
const wizardAccount = accounts.find(a => a.id === wizardAccountId) || activeAccount || acmeAccount;
|
||||
const wizardIsDns01 = (wizardAccount?.challenge_type === 'dns-01');
|
||||
const wizardDnsManual = wizardIsDns01 && (wizardAccount?.dns_provider === 'manual');
|
||||
// Wildcard certificates can only be issued over DNS-01. Catch this client-side so the user is
|
||||
// told their mistake up front instead of waiting for a CA-side rejection.
|
||||
const wizardHasWildcard = (wizardDomains || []).some(d => typeof d === 'string' && d.trim().startsWith('*.'));
|
||||
const wizardWildcardBlocked = wizardHasWildcard && !wizardIsDns01;
|
||||
// A DNS-01 account can exist while the global kill-switch is off (e.g. an admin disabled it later).
|
||||
// Issuing would be rejected by the backend, so block it in the wizard with a clear reason.
|
||||
const wizardDns01Disabled = wizardIsDns01 && !dns01Enabled;
|
||||
// Surface a direct credentials shortcut on the dashboard card when the primary account uses an
|
||||
// automated DNS provider (manual providers need no credentials).
|
||||
const acmeAccountProvider = acmeAccount ? dnsProviders.find(p => p.name === acmeAccount.dns_provider) : null;
|
||||
const acmeAccountNeedsCreds = acmeAccount?.challenge_type === 'dns-01' && (acmeAccountProvider?.credential_fields || []).length > 0;
|
||||
|
||||
const filteredOrders = orders.filter(o => {
|
||||
if (orderFilter === 'active') return !['cancelled', 'invalid', 'valid'].includes(o.status);
|
||||
@@ -170,12 +250,29 @@ const ACMEAutomation = () => {
|
||||
message.error('At least one domain is required');
|
||||
return;
|
||||
}
|
||||
// Defense-in-depth: the Submit button is already disabled for these, but guard here too.
|
||||
if (wizardWildcardBlocked) {
|
||||
message.error('Wildcard certificates require a DNS-01 account. Select a DNS-01 account or remove the wildcard domain.');
|
||||
return;
|
||||
}
|
||||
if (wizardDns01Disabled) {
|
||||
message.error('DNS-01 is disabled by an administrator. Enable it in Settings > ACME to issue this certificate.');
|
||||
return;
|
||||
}
|
||||
setSubmitting(true);
|
||||
// Resolve the chosen account's challenge method so DNS-01/wildcard requests are explicit.
|
||||
// Use the same resolution as the wizard description (wizardAccount) so what the user reviewed
|
||||
// matches what is sent.
|
||||
const challengeType = wizardAccount?.challenge_type; // 'http-01' | 'dns-01' | undefined
|
||||
// Manual DNS-01 can't auto-renew (the wizard shows the switch off+disabled). Send false to
|
||||
// match the displayed state rather than relying only on the backend to override it.
|
||||
const autoRenew = wizardDnsManual ? false : (values.auto_renew !== false);
|
||||
const res = await axios.post('/api/letsencrypt/certificates', {
|
||||
domains: values.domains,
|
||||
cluster_ids: values.cluster_ids || [],
|
||||
auto_renew: values.auto_renew !== false,
|
||||
auto_renew: autoRenew,
|
||||
account_id: values.account_id || null,
|
||||
challenge_type: challengeType || undefined,
|
||||
});
|
||||
message.success(res.data?.message || 'Certificate request submitted');
|
||||
if (res.data?.warnings?.length > 0) {
|
||||
@@ -185,6 +282,16 @@ const ACMEAutomation = () => {
|
||||
setWizardStep(0);
|
||||
wizardForm.resetFields();
|
||||
fetchData();
|
||||
// For a manual DNS-01 order the user must publish the TXT record(s) next, so open the
|
||||
// order detail straight away instead of leaving them to hunt for it.
|
||||
if (res.data?.challenge_type === 'dns-01'
|
||||
&& (res.data?.dns_provider || 'manual') === 'manual'
|
||||
&& res.data?.order_id) {
|
||||
handleViewOrder(res.data.order_id);
|
||||
} else if (res.data?.challenge_type === 'dns-01') {
|
||||
// Automated DNS-01 (e.g. Cloudflare): reassure the user it is hands-off.
|
||||
message.info('Automated DNS-01: the TXT records will be published and validated automatically. No action needed.', 6);
|
||||
}
|
||||
} catch (err) {
|
||||
message.error(getErrorMsg(err, 'Failed to request certificate'));
|
||||
} finally {
|
||||
@@ -234,7 +341,7 @@ const ACMEAutomation = () => {
|
||||
setOrderDetail(res.data);
|
||||
setDetailVisible(true);
|
||||
} catch (err) {
|
||||
message.error('Failed to load order details');
|
||||
message.error(getErrorMsg(err, 'Failed to load order details'));
|
||||
}
|
||||
};
|
||||
|
||||
@@ -453,21 +560,144 @@ const ACMEAutomation = () => {
|
||||
try {
|
||||
const values = await registerForm.validateFields();
|
||||
setRegistering(true);
|
||||
const challengeType = dns01Enabled ? (values.challenge_type || 'http-01') : 'http-01';
|
||||
const dnsProvider = challengeType === 'dns-01' ? (values.dns_provider || null) : null;
|
||||
const res = await axios.post('/api/letsencrypt/accounts', {
|
||||
email: values.email,
|
||||
tos_agreed: values.tos_agreed,
|
||||
challenge_type: challengeType,
|
||||
dns_provider: dnsProvider,
|
||||
// EAB for CAs that require it (ZeroSSL/Google). Empty → backend falls back to global Settings.
|
||||
eab_kid: (values.eab_kid || '').trim() || undefined,
|
||||
eab_hmac_key: (values.eab_hmac_key || '').trim() || undefined,
|
||||
});
|
||||
message.success(`ACME account registered: ${res.data?.email || values.email}`);
|
||||
setRegisterVisible(false);
|
||||
registerForm.resetFields();
|
||||
const accountId = res.data?.id;
|
||||
// For an automated DNS-01 provider, store the entered credentials (verified server-side).
|
||||
const provider = dnsProviders.find(p => p.name === dnsProvider);
|
||||
let credFailed = false;
|
||||
if (challengeType === 'dns-01' && accountId && provider && (provider.credential_fields || []).length > 0) {
|
||||
const creds = {};
|
||||
(provider.credential_fields || []).forEach(f => {
|
||||
const v = values[`cred_${f.key}`];
|
||||
if (v != null && v !== '') creds[f.key] = v;
|
||||
});
|
||||
try {
|
||||
const r2 = await axios.put(`/api/letsencrypt/accounts/${accountId}/dns-credentials`, {
|
||||
dns_provider: dnsProvider,
|
||||
credentials: creds,
|
||||
});
|
||||
message.success(r2.data?.detail || 'DNS provider credentials saved');
|
||||
} catch (credErr) {
|
||||
credFailed = true;
|
||||
message.warning(getErrorMsg(credErr, `Account "${res.data?.email || values.email}" registered, but the DNS credentials could not be saved. You can fix them from the account's DNS credentials action.`), 8);
|
||||
}
|
||||
}
|
||||
// Only close + reset on full success. On a credential-save failure, keep the modal open with
|
||||
// the entered values so the user can correct the token and re-submit (the account already
|
||||
// exists and the PUT re-verifies) — avoids discarding input, a dead-end, and a misleading
|
||||
// success toast (Finding 7). The warning toast above explains what to fix.
|
||||
if (!credFailed) {
|
||||
message.success(`ACME account registered: ${res.data?.email || values.email}`);
|
||||
setRegisterVisible(false);
|
||||
registerForm.resetFields();
|
||||
}
|
||||
fetchData();
|
||||
} catch (err) {
|
||||
// Inline field-validation rejections already render under each field; don't also
|
||||
// fire a generic error toast (mirrors handleSaveDnsCreds).
|
||||
if (err?.errorFields) return;
|
||||
message.error(getErrorMsg(err, 'Account registration failed'));
|
||||
} finally {
|
||||
setRegistering(false);
|
||||
}
|
||||
};
|
||||
|
||||
// Issue #35: manual DNS-01 — user asserts the TXT records are published; tell the CA to validate.
|
||||
const handleDnsConfirm = async (orderId) => {
|
||||
if (confirming) return; // guard the leading-edge double-click (loading alone does not block a synchronous re-fire)
|
||||
try {
|
||||
setConfirming(true);
|
||||
const res = await axios.post(`/api/letsencrypt/orders/${orderId}/dns-confirm`);
|
||||
message.success(res.data?.message || 'DNS-01 confirmation submitted; the CA will validate shortly.');
|
||||
// Refresh the orders list AND the open detail modal so the user sees the new state
|
||||
// (otherwise the modal shows a stale TXT block + an active Verify button).
|
||||
fetchData();
|
||||
try {
|
||||
const fresh = await axios.get(`/api/letsencrypt/orders/${orderId}`);
|
||||
// Only repopulate if the same order's detail is still open (the user may have
|
||||
// closed it or navigated to another order while the request was in flight).
|
||||
setOrderDetail((prev) => (prev && prev.id === orderId ? fresh.data : prev));
|
||||
} catch (_e) { /* list refresh already happened; modal stays as-is */ }
|
||||
} catch (err) {
|
||||
message.error(getErrorMsg(err, 'Failed to confirm DNS-01'));
|
||||
} finally {
|
||||
setConfirming(false);
|
||||
}
|
||||
};
|
||||
|
||||
// Issue #35: view / replace / clear an account's DNS provider credentials after creation.
|
||||
const openDnsCredsModal = async (account) => {
|
||||
setCredAccount(account);
|
||||
setCredMeta(null);
|
||||
credForm.resetFields();
|
||||
setCredModalVisible(true);
|
||||
setCredLoading(true);
|
||||
try {
|
||||
const res = await axios.get(`/api/letsencrypt/accounts/${account.id}/dns-credentials`);
|
||||
setCredMeta(res.data);
|
||||
} catch (err) {
|
||||
message.error(getErrorMsg(err, 'Failed to load DNS credentials'));
|
||||
} finally {
|
||||
setCredLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const handleSaveDnsCreds = async () => {
|
||||
if (!credAccount || !credProvider) return;
|
||||
try {
|
||||
const values = await credForm.validateFields();
|
||||
setCredSaving(true);
|
||||
const creds = {};
|
||||
(credProvider.credential_fields || []).forEach(f => {
|
||||
const v = values[`cred_${f.key}`];
|
||||
if (v != null && v !== '') creds[f.key] = v;
|
||||
});
|
||||
const res = await axios.put(`/api/letsencrypt/accounts/${credAccount.id}/dns-credentials`, {
|
||||
dns_provider: credAccount.dns_provider,
|
||||
credentials: creds,
|
||||
});
|
||||
message.success(res.data?.detail || 'DNS provider credentials saved and verified');
|
||||
setCredModalVisible(false);
|
||||
credForm.resetFields();
|
||||
fetchData();
|
||||
} catch (err) {
|
||||
if (err?.errorFields) return; // antd validation errors shown inline
|
||||
message.error(getErrorMsg(err, 'Failed to save DNS credentials'));
|
||||
} finally {
|
||||
setCredSaving(false);
|
||||
}
|
||||
};
|
||||
|
||||
const handleClearDnsCreds = () => {
|
||||
if (!credAccount) return;
|
||||
Modal.confirm({
|
||||
title: 'Clear DNS credentials',
|
||||
content: `Remove the stored DNS provider credentials for ${credAccount.email}? Automated DNS-01 issuance/renewal will stop working until you re-enter them.`,
|
||||
okText: 'Clear',
|
||||
okButtonProps: { danger: true },
|
||||
onOk: async () => {
|
||||
try {
|
||||
await axios.delete(`/api/letsencrypt/accounts/${credAccount.id}/dns-credentials`);
|
||||
message.success('DNS credentials cleared');
|
||||
setCredModalVisible(false);
|
||||
fetchData();
|
||||
} catch (err) {
|
||||
message.error(getErrorMsg(err, 'Failed to clear DNS credentials'));
|
||||
}
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
const handleDeactivateAccount = (accountId, email) => {
|
||||
Modal.confirm({
|
||||
title: 'Deactivate ACME Account',
|
||||
@@ -564,7 +794,19 @@ const ACMEAutomation = () => {
|
||||
render: (status, record) => statusTag(status, record),
|
||||
},
|
||||
{
|
||||
title: 'Account', dataIndex: 'account_email', key: 'account_email',
|
||||
title: 'Method', key: 'method', width: 170,
|
||||
render: (_, record) => {
|
||||
const ct = record.challenge_type || 'http-01';
|
||||
if (ct !== 'dns-01') return <Tag>HTTP-01</Tag>;
|
||||
const prov = record.dns_provider || 'manual';
|
||||
const needsAction = prov === 'manual' && (record.status === 'pending' || record.status === 'processing');
|
||||
return needsAction
|
||||
? <Tag color="warning" icon={<ExclamationCircleOutlined />}>DNS-01 (manual): action needed</Tag>
|
||||
: <Tag>DNS-01 ({prov})</Tag>;
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Account', dataIndex: 'account_email', key: 'account_email', ellipsis: true,
|
||||
render: (e) => e || '-',
|
||||
},
|
||||
{
|
||||
@@ -630,9 +872,27 @@ const ACMEAutomation = () => {
|
||||
return <Tag color={color}>{d} days</Tag>;
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Method', key: 'method', width: 130,
|
||||
render: (_, record) => {
|
||||
const ct = record.challenge_type || 'http-01';
|
||||
if (ct !== 'dns-01') return <Tag>HTTP-01</Tag>;
|
||||
return <Tag>DNS-01 ({record.dns_provider || 'manual'})</Tag>;
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Auto-Renew', dataIndex: 'auto_renew', key: 'auto_renew',
|
||||
render: (v) => v ? <Tag color="green">Enabled</Tag> : <Tag>Disabled</Tag>,
|
||||
render: (v, record) => {
|
||||
const isManualDns = record.challenge_type === 'dns-01' && (record.dns_provider || 'manual') === 'manual';
|
||||
if (isManualDns) {
|
||||
return (
|
||||
<Tooltip title="Manual DNS-01 cannot auto-renew unattended. Re-publish the TXT record and request renewal before expiry.">
|
||||
<Tag color="warning" icon={<ExclamationCircleOutlined />}>Manual (re-publish TXT)</Tag>
|
||||
</Tooltip>
|
||||
);
|
||||
}
|
||||
return v ? <Tag color="green">Enabled</Tag> : <Tag>Disabled</Tag>;
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
@@ -655,8 +915,19 @@ const ACMEAutomation = () => {
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
message="Each domain must resolve to an HAProxy node with ACME challenge routing enabled."
|
||||
message={wizardIsDns01
|
||||
? "DNS-01: validated via a DNS TXT record, so no public port 80 is needed. Wildcards (*.example.com) are supported. Note that a wildcard does not cover the bare apex (example.com); add it as a separate domain if you need both."
|
||||
: "Each domain must resolve to an HAProxy node with ACME challenge routing enabled (HTTP-01)."}
|
||||
/>
|
||||
{wizardWildcardBlocked && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginTop: 12 }}
|
||||
message="Wildcard requires a DNS-01 account"
|
||||
description="A wildcard domain (*.example.com) can only be validated over DNS-01. The currently selected account uses HTTP-01. Choose a DNS-01 account in the next step, or remove the wildcard domain."
|
||||
/>
|
||||
)}
|
||||
</>
|
||||
),
|
||||
},
|
||||
@@ -668,20 +939,63 @@ const ACMEAutomation = () => {
|
||||
<Select mode="multiple" placeholder="Leave empty for global certificate" allowClear>
|
||||
{clusters.map(c => (
|
||||
<Option key={c.id} value={c.id}>
|
||||
{c.name} {c.acme_enabled ? '' : '(ACME not enabled)'}
|
||||
{c.name} {wizardIsDns01 ? '' : (c.acme_enabled ? '' : '(ACME not enabled)')}
|
||||
</Option>
|
||||
))}
|
||||
</Select>
|
||||
</Form.Item>
|
||||
<Form.Item name="auto_renew" label="Auto-Renew" valuePropName="checked" initialValue={true}>
|
||||
<Switch defaultChecked />
|
||||
</Form.Item>
|
||||
{wizardIsDns01 && (
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="DNS-01 needs no ACME Challenge Routing. Any active cluster works."
|
||||
description={wizardDnsManual
|
||||
? "This account uses a manual DNS provider: after submitting, open the order and publish the shown TXT record, then confirm."
|
||||
: "The DNS TXT record(s) will be published automatically."}
|
||||
/>
|
||||
)}
|
||||
{wizardDnsManual ? (
|
||||
// Manual DNS-01 cannot auto-renew (the backend forces it off); show the control off and
|
||||
// disabled so it matches the outcome rather than implying an automated renewal.
|
||||
<Form.Item label="Auto-Renew">
|
||||
<Switch checked={false} disabled />
|
||||
</Form.Item>
|
||||
) : (
|
||||
<Form.Item name="auto_renew" label="Auto-Renew" valuePropName="checked" initialValue={true}>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
)}
|
||||
{wizardDnsManual && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginTop: -8, marginBottom: 16 }}
|
||||
message="Manual DNS-01 cannot auto-renew unattended. You will need to re-publish the TXT record at renewal time."
|
||||
/>
|
||||
)}
|
||||
{accounts.length > 1 && (
|
||||
<Form.Item name="account_id" label="ACME Account">
|
||||
<Select placeholder="Use default account">
|
||||
{accounts.map(a => (
|
||||
<Option key={a.id} value={a.id}>{a.email} ({a.directory_url})</Option>
|
||||
))}
|
||||
<Form.Item
|
||||
name="account_id"
|
||||
label="ACME Account"
|
||||
extra={<span style={{ fontSize: 12, color: token.colorTextSecondary }}>The validation method (HTTP-01 or DNS-01) is set by the chosen account. To use DNS-01, pick a DNS-01 account.</span>}
|
||||
>
|
||||
<Select placeholder="Use default account" optionLabelProp="label">
|
||||
{accounts.map(a => {
|
||||
// Match the parenthesized "DNS-01 (provider)" form used in the tables and order detail.
|
||||
const methodLabel = a.challenge_type === 'dns-01'
|
||||
? `DNS-01 (${a.dns_provider || 'manual'})`
|
||||
: 'HTTP-01';
|
||||
return (
|
||||
<Option key={a.id} value={a.id} label={`${a.email} · ${methodLabel}`}>
|
||||
<span>{a.email}{' '}
|
||||
<Typography.Text type="secondary" style={{ fontSize: 12 }}>
|
||||
{methodLabel} · {a.directory_url}
|
||||
</Typography.Text>
|
||||
</span>
|
||||
</Option>
|
||||
);
|
||||
})}
|
||||
</Select>
|
||||
</Form.Item>
|
||||
)}
|
||||
@@ -707,7 +1021,31 @@ const ACMEAutomation = () => {
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
)}
|
||||
{acmeEnabledClusters.length === 0 && (
|
||||
{wizardDns01Disabled && (
|
||||
<Alert
|
||||
type="error"
|
||||
showIcon
|
||||
message="DNS-01 is disabled"
|
||||
description={
|
||||
<span>
|
||||
This account uses DNS-01, but DNS-01 is currently disabled by an administrator.{' '}
|
||||
<Button type="link" size="small" style={{ padding: 0 }} onClick={() => navigate('/settings?tab=acme')}>Enable it in Settings > ACME</Button>
|
||||
{' '}to issue this certificate.
|
||||
</span>
|
||||
}
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
)}
|
||||
{wizardWildcardBlocked && (
|
||||
<Alert
|
||||
type="error"
|
||||
showIcon
|
||||
message="Wildcard requires a DNS-01 account"
|
||||
description="Remove the wildcard domain or select a DNS-01 account in the Configuration step."
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
)}
|
||||
{!wizardIsDns01 && acmeEnabledClusters.length === 0 && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
@@ -723,7 +1061,7 @@ const ACMEAutomation = () => {
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
)}
|
||||
{prerequisites?.steps?.find(s => s.key === 'config_applied' && s.ok === false) && (() => {
|
||||
{!wizardIsDns01 && prerequisites?.steps?.find(s => s.key === 'config_applied' && s.ok === false) && (() => {
|
||||
const configStep = prerequisites.steps.find(s => s.key === 'config_applied');
|
||||
const pendingNames = (configStep?.pending_clusters || []).map(c => c.name).join(', ');
|
||||
return (
|
||||
@@ -755,8 +1093,17 @@ const ACMEAutomation = () => {
|
||||
description={
|
||||
<ul style={{ margin: 0, paddingLeft: 20 }}>
|
||||
<li>ACME Account: {activeAccount ? <Tag color="success">Active ({activeAccount.email})</Tag> : <Tag color="error">No active account</Tag>}</li>
|
||||
<li>ACME-enabled Clusters: {acmeEnabledClusters.length > 0 ? <Tag color="success">{acmeEnabledClusters.map(c => c.name).join(', ')}</Tag> : <Tag color="warning">None</Tag>}</li>
|
||||
<li>Domains must resolve to HAProxy node IPs for HTTP-01 validation</li>
|
||||
{wizardIsDns01 ? (
|
||||
<>
|
||||
<li>Challenge Method: <Tag>DNS-01</Tag> (TXT record; no port 80 / ACME routing needed)</li>
|
||||
<li>DNS provider: <Tag>{wizardAccount?.dns_provider || 'manual'}</Tag></li>
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<li>ACME-enabled Clusters: {acmeEnabledClusters.length > 0 ? <Tag color="success">{acmeEnabledClusters.map(c => c.name).join(', ')}</Tag> : <Tag color="warning">None</Tag>}</li>
|
||||
<li>Domains must resolve to HAProxy node IPs for HTTP-01 validation</li>
|
||||
</>
|
||||
)}
|
||||
</ul>
|
||||
}
|
||||
/>
|
||||
@@ -807,9 +1154,15 @@ const ACMEAutomation = () => {
|
||||
|
||||
{pendingOrders.length > 0 && (() => {
|
||||
const stuckCount = pendingOrders.filter(isOrderStuck).length;
|
||||
const inFlightCount = pendingOrders.length - stuckCount;
|
||||
// Manual DNS-01 orders are waiting on the USER to publish a TXT record, not on the CA —
|
||||
// call them out separately so the action item is visible without scanning the table.
|
||||
const manualDnsAwaiting = pendingOrders.filter(o =>
|
||||
o.challenge_type === 'dns-01' && (o.dns_provider || 'manual') === 'manual'
|
||||
&& (o.status === 'pending' || o.status === 'processing')).length;
|
||||
const inFlightCount = pendingOrders.length - stuckCount - manualDnsAwaiting;
|
||||
const parts = [];
|
||||
if (inFlightCount > 0) parts.push(`${inFlightCount} awaiting validation`);
|
||||
if (manualDnsAwaiting > 0) parts.push(`${manualDnsAwaiting} manual DNS-01 awaiting your TXT record${manualDnsAwaiting > 1 ? 's' : ''}`);
|
||||
if (stuckCount > 0) parts.push(`${stuckCount} pending download (auto-retrying every 60s)`);
|
||||
return (
|
||||
<Alert
|
||||
@@ -817,9 +1170,11 @@ const ACMEAutomation = () => {
|
||||
showIcon
|
||||
icon={<ExclamationCircleOutlined />}
|
||||
message={`${pendingOrders.length} certificate order(s) in progress: ${parts.join(', ')}`}
|
||||
description={stuckCount > 0
|
||||
? "Stuck orders will auto-complete via the background task. You can also click \"Complete\" to retry immediately."
|
||||
: undefined}
|
||||
description={manualDnsAwaiting > 0
|
||||
? "Open a manual DNS-01 order to see the TXT record to publish, then confirm it."
|
||||
: stuckCount > 0
|
||||
? "Stuck orders will auto-complete via the background task. You can also click \"Complete\" to retry immediately."
|
||||
: undefined}
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
);
|
||||
@@ -866,6 +1221,11 @@ const ACMEAutomation = () => {
|
||||
<InfoCircleOutlined /> Manage
|
||||
</Button>
|
||||
)}
|
||||
{acmeAccountNeedsCreds && (
|
||||
<Button type="link" size="small" style={{ padding: 0 }} onClick={() => openDnsCredsModal(acmeAccount)}>
|
||||
<KeyOutlined /> DNS credentials
|
||||
</Button>
|
||||
)}
|
||||
</div>
|
||||
</Card>
|
||||
</Col>
|
||||
@@ -965,7 +1325,7 @@ const ACMEAutomation = () => {
|
||||
</Button>
|
||||
)}
|
||||
{wizardStep === wizardSteps.length - 1 && (
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={!activeAccount || acmeEnabledClusters.length === 0}>
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={!activeAccount || wizardWildcardBlocked || wizardDns01Disabled || (!wizardIsDns01 && acmeEnabledClusters.length === 0)}>
|
||||
Submit Request
|
||||
</Button>
|
||||
)}
|
||||
@@ -994,7 +1354,7 @@ const ACMEAutomation = () => {
|
||||
</Tag>
|
||||
) : orderDetail.status === 'valid' ? (
|
||||
<Tag color="warning" icon={<ExclamationCircleOutlined />}>
|
||||
Pending download — auto-completion task will retry every 60s
|
||||
Pending download. Auto-completion task will retry every 60s
|
||||
</Tag>
|
||||
) : orderDetail.status === 'invalid' || orderDetail.status === 'cancelled' ? (
|
||||
<Tag color="default">Not issued</Tag>
|
||||
@@ -1010,6 +1370,71 @@ const ACMEAutomation = () => {
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
)}
|
||||
{orderDetail.challenge_type === 'dns-01'
|
||||
&& !(orderDetail.challenges || []).some(c => c.challenge_type === 'dns-01')
|
||||
&& (orderDetail.status === 'pending' || orderDetail.status === 'processing') && (
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Preparing DNS-01 challenge"
|
||||
description="The TXT record(s) for this order are being provisioned. They will appear here shortly; this view refreshes automatically."
|
||||
/>
|
||||
)}
|
||||
{orderDetail.challenge_type === 'dns-01' && (orderDetail.challenges || []).some(c => c.challenge_type === 'dns-01') && (() => {
|
||||
const dnsChallenges = (orderDetail.challenges || []).filter(c => c.challenge_type === 'dns-01');
|
||||
const isManual = (orderDetail.dns_provider || 'manual') === 'manual';
|
||||
const canConfirm = isManual && (orderDetail.status === 'pending' || orderDetail.status === 'processing');
|
||||
return (
|
||||
<Alert
|
||||
type={isManual ? 'warning' : 'info'}
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message={isManual
|
||||
? 'DNS-01 (manual): publish these TXT record(s), then verify'
|
||||
: 'DNS-01 (automated): the TXT record(s) are published for you'}
|
||||
description={
|
||||
<div>
|
||||
<div style={{ marginBottom: 8 }}>
|
||||
Add the following DNS TXT record{dnsChallenges.length > 1 ? 's' : ''} at your DNS provider:
|
||||
</div>
|
||||
<Table
|
||||
size="small"
|
||||
pagination={false}
|
||||
dataSource={dnsChallenges}
|
||||
rowKey="id"
|
||||
scroll={{ x: 'max-content' }}
|
||||
columns={[
|
||||
{ title: 'Record name', dataIndex: 'dns_record_name', key: 'name',
|
||||
render: (v) => v ? <Typography.Text code copyable style={{ wordBreak: 'break-all' }}>{v}</Typography.Text> : <Typography.Text type="secondary">-</Typography.Text> },
|
||||
{ title: 'Type', key: 'type', width: 60, render: () => 'TXT' },
|
||||
{ title: 'Value', dataIndex: 'dns_txt_value', key: 'val',
|
||||
render: (v) => v ? <Typography.Text code copyable style={{ wordBreak: 'break-all' }}>{v}</Typography.Text> : <Typography.Text type="secondary">-</Typography.Text> },
|
||||
]}
|
||||
/>
|
||||
{canConfirm && (
|
||||
<>
|
||||
<div style={{ marginTop: 12, fontSize: 12, color: token.colorTextSecondary }}>
|
||||
DNS changes can take a few minutes to propagate. If verification fails,
|
||||
wait a short while and try again. The order keeps retrying in the background.
|
||||
</div>
|
||||
<Button
|
||||
type="primary"
|
||||
size="small"
|
||||
loading={confirming}
|
||||
disabled={confirming}
|
||||
style={{ marginTop: 8 }}
|
||||
onClick={() => handleDnsConfirm(orderDetail.id)}
|
||||
>
|
||||
I've added the record(s), verify now
|
||||
</Button>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
}
|
||||
/>
|
||||
);
|
||||
})()}
|
||||
{orderDetail.challenges?.length > 0 && (
|
||||
<>
|
||||
<h4>Challenges</h4>
|
||||
@@ -1020,7 +1445,10 @@ const ACMEAutomation = () => {
|
||||
rowKey="id"
|
||||
columns={[
|
||||
{ title: 'Domain', dataIndex: 'domain', key: 'domain' },
|
||||
{ title: 'Token', dataIndex: 'token', key: 'token', ellipsis: true },
|
||||
{ title: 'Method', dataIndex: 'challenge_type', key: 'method', width: 90,
|
||||
render: (ct) => <Tag>{(ct || 'http-01') === 'dns-01' ? 'DNS-01' : 'HTTP-01'}</Tag> },
|
||||
{ title: 'Token', dataIndex: 'token', key: 'token', ellipsis: true,
|
||||
render: (t, r) => (r.challenge_type === 'dns-01') ? <Typography.Text type="secondary">TXT-based (see above)</Typography.Text> : t },
|
||||
{ title: 'Status', dataIndex: 'status', key: 'status', render: (s) => statusTag(s) },
|
||||
]}
|
||||
/>
|
||||
@@ -1047,7 +1475,7 @@ const ACMEAutomation = () => {
|
||||
rowKey="id"
|
||||
columns={[
|
||||
{
|
||||
title: 'Email', dataIndex: 'email', key: 'email',
|
||||
title: 'Email', dataIndex: 'email', key: 'email', ellipsis: true,
|
||||
render: (email) => <strong>{email}</strong>,
|
||||
},
|
||||
{
|
||||
@@ -1098,6 +1526,10 @@ const ACMEAutomation = () => {
|
||||
</p>
|
||||
)}
|
||||
{record.eab_kid && <p><strong>EAB Key ID:</strong> {record.eab_kid}</p>}
|
||||
<p><strong>Challenge Method:</strong> {(record.challenge_type || 'http-01') === 'dns-01' ? 'DNS-01' : 'HTTP-01'}</p>
|
||||
{(record.challenge_type === 'dns-01') && (
|
||||
<p><strong>DNS Provider:</strong> {record.dns_provider || 'manual'}</p>
|
||||
)}
|
||||
<p><strong>ToS Accepted:</strong> {record.tos_agreed ? 'Yes' : 'No'}</p>
|
||||
<p><strong>Registered:</strong> {record.created_at ? new Date(record.created_at).toLocaleString() : '-'}</p>
|
||||
</div>
|
||||
@@ -1106,6 +1538,16 @@ const ACMEAutomation = () => {
|
||||
}}
|
||||
/>
|
||||
</Tooltip>
|
||||
{record.challenge_type === 'dns-01'
|
||||
&& (dnsProviders.find(p => p.name === record.dns_provider)?.credential_fields || []).length > 0 && (
|
||||
<Tooltip title="DNS Provider Credentials">
|
||||
<Button
|
||||
icon={<KeyOutlined />}
|
||||
size="small"
|
||||
onClick={() => openDnsCredsModal(record)}
|
||||
/>
|
||||
</Tooltip>
|
||||
)}
|
||||
{record.status !== 'deactivated' ? (
|
||||
<Tooltip title="Deactivate">
|
||||
<Button
|
||||
@@ -1154,7 +1596,7 @@ const ACMEAutomation = () => {
|
||||
message="ACME account will be registered with the directory URL configured in Settings > ACME."
|
||||
style={{ marginBottom: 16 }}
|
||||
/>
|
||||
<Form form={registerForm} layout="vertical">
|
||||
<Form form={registerForm} layout="vertical" preserve={false}>
|
||||
<Form.Item
|
||||
name="email"
|
||||
label="Contact Email"
|
||||
@@ -1167,11 +1609,12 @@ const ACMEAutomation = () => {
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="tos_agreed"
|
||||
label="Terms of Service"
|
||||
valuePropName="checked"
|
||||
initialValue={false}
|
||||
rules={[{ validator: (_, v) => v ? Promise.resolve() : Promise.reject('You must accept the Terms of Service') }]}
|
||||
>
|
||||
<Switch checkedChildren="Accepted" unCheckedChildren="Not Accepted" />
|
||||
<Switch checkedChildren="Accepted" unCheckedChildren="Not Accepted" aria-label="Accept Terms of Service" />
|
||||
</Form.Item>
|
||||
<div style={{ fontSize: 12, color: token.colorTextSecondary }}>
|
||||
By accepting, you agree to the ACME CA's Terms of Service (e.g.{' '}
|
||||
@@ -1179,12 +1622,194 @@ const ACMEAutomation = () => {
|
||||
Let's Encrypt Subscriber Agreement
|
||||
</a>).
|
||||
</div>
|
||||
{/* Issue #35: External Account Binding — required by ZeroSSL / Google Trust Services. */}
|
||||
<Collapse
|
||||
ghost
|
||||
style={{ marginTop: 12 }}
|
||||
items={[{
|
||||
key: 'eab',
|
||||
label: 'External Account Binding (EAB)',
|
||||
children: (
|
||||
<>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 12 }}
|
||||
message="Required by some CAs (ZeroSSL, Google Trust Services). Copy the Key ID and HMAC Key from your CA account. Re-enter them on each registration; the HMAC key is not stored."
|
||||
/>
|
||||
<Form.Item
|
||||
name="eab_kid"
|
||||
label="EAB Key ID"
|
||||
dependencies={['eab_hmac_key']}
|
||||
rules={[({ getFieldValue }) => ({
|
||||
validator(_, value) {
|
||||
const kid = (value || '').trim();
|
||||
const hmac = (getFieldValue('eab_hmac_key') || '').trim();
|
||||
if (!kid && hmac) {
|
||||
return Promise.reject(new Error('EAB Key ID is required when an HMAC Key is entered.'));
|
||||
}
|
||||
return Promise.resolve();
|
||||
},
|
||||
})]}
|
||||
>
|
||||
<Input placeholder="EAB Key Identifier" autoComplete="off" />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="eab_hmac_key"
|
||||
label="EAB HMAC Key"
|
||||
dependencies={['eab_kid']}
|
||||
rules={[({ getFieldValue }) => ({
|
||||
validator(_, value) {
|
||||
const kid = (getFieldValue('eab_kid') || '').trim();
|
||||
const hmac = (value || '').trim();
|
||||
if (kid && !hmac) {
|
||||
return Promise.reject(new Error('EAB HMAC Key is required when a Key ID is entered.'));
|
||||
}
|
||||
return Promise.resolve();
|
||||
},
|
||||
})]}
|
||||
>
|
||||
<Input.Password placeholder="EAB HMAC Key (base64url)" autoComplete="new-password" />
|
||||
</Form.Item>
|
||||
</>
|
||||
),
|
||||
}]}
|
||||
/>
|
||||
{dns01Enabled && (
|
||||
<>
|
||||
<Divider style={{ margin: '16px 0 12px' }} />
|
||||
<Form.Item
|
||||
name="challenge_type"
|
||||
label="Challenge Method"
|
||||
initialValue="http-01"
|
||||
tooltip="HTTP-01 validates over port 80. DNS-01 validates via a DNS TXT record. It works for internal/isolated clusters (no public port 80) and supports wildcards."
|
||||
>
|
||||
<Select>
|
||||
<Option value="http-01">HTTP-01 (default)</Option>
|
||||
<Option value="dns-01">DNS-01 (TXT record)</Option>
|
||||
</Select>
|
||||
</Form.Item>
|
||||
{regChallengeType === 'dns-01' && (
|
||||
<>
|
||||
{dnsProviders.length === 0 ? (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
message="No DNS providers are available"
|
||||
description="DNS-01 appears enabled, but no provider catalog was returned. Confirm DNS-01 is enabled in Settings and that the backend is reachable, then reopen this dialog."
|
||||
/>
|
||||
) : (
|
||||
<Form.Item
|
||||
name="dns_provider"
|
||||
label="DNS Provider"
|
||||
rules={[{ required: true, message: 'Select a DNS provider' }]}
|
||||
>
|
||||
<Select placeholder="Select a DNS provider">
|
||||
{dnsProviders.map(p => (
|
||||
<Option key={p.name} value={p.name}>{p.label}</Option>
|
||||
))}
|
||||
</Select>
|
||||
</Form.Item>
|
||||
)}
|
||||
{selectedDnsProvider && (selectedDnsProvider.credential_fields || []).length > 0 && (
|
||||
<>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 12 }}
|
||||
message="Provider credentials are encrypted at rest and verified before they are saved. The token needs permission to create and delete TXT records in your domain's DNS zone."
|
||||
/>
|
||||
{(selectedDnsProvider.credential_fields || []).map(f => (
|
||||
<Form.Item
|
||||
key={f.key}
|
||||
name={`cred_${f.key}`}
|
||||
label={f.label}
|
||||
extra={f.help ? <span style={{ fontSize: 12, color: token.colorTextSecondary }}>{f.help}</span> : null}
|
||||
rules={f.required ? [{ required: true, message: `${f.label} is required` }] : []}
|
||||
>
|
||||
{f.type === 'password'
|
||||
? <Input.Password placeholder={f.label} maxLength={f.max_length || undefined} />
|
||||
: <Input placeholder={f.label} maxLength={f.max_length || undefined} />}
|
||||
</Form.Item>
|
||||
))}
|
||||
</>
|
||||
)}
|
||||
{selectedDnsProvider && !selectedDnsProvider.automated && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
message="Manual DNS provider"
|
||||
description="You will publish the DNS TXT record yourself and confirm it. Manual DNS-01 certificates cannot auto-renew unattended."
|
||||
/>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
</Form>
|
||||
</Modal>
|
||||
|
||||
{/* Issue #35: DNS provider credentials for an existing DNS-01 account (view / replace / clear) */}
|
||||
<Modal
|
||||
title={<span><KeyOutlined /> DNS Provider Credentials{credAccount ? `: ${credAccount.email}` : ''}</span>}
|
||||
open={credModalVisible}
|
||||
onCancel={() => { setCredModalVisible(false); credForm.resetFields(); }}
|
||||
width={560}
|
||||
footer={[
|
||||
<Button key="clear" danger onClick={handleClearDnsCreds} disabled={!credMeta?.configured || credSaving}>
|
||||
Clear credentials
|
||||
</Button>,
|
||||
<Button key="cancel" onClick={() => { setCredModalVisible(false); credForm.resetFields(); }}>
|
||||
Cancel
|
||||
</Button>,
|
||||
<Button key="save" type="primary" loading={credSaving} onClick={handleSaveDnsCreds}>
|
||||
Save & verify
|
||||
</Button>,
|
||||
]}
|
||||
>
|
||||
{credLoading ? (
|
||||
<div style={{ textAlign: 'center', padding: 24 }}><Spin /></div>
|
||||
) : (
|
||||
<>
|
||||
<p style={{ marginBottom: 4 }}>
|
||||
<strong>Provider:</strong> {credAccount?.dns_provider || '-'}
|
||||
</p>
|
||||
<p style={{ marginTop: 0, fontSize: 12, color: token.colorTextSecondary }}>
|
||||
{credMeta?.configured
|
||||
? `Configured: ${(credMeta.credential_fields_present || []).join(', ') || '(none)'}${credMeta.updated_at ? ` · updated ${new Date(credMeta.updated_at).toLocaleString()}` : ''}`
|
||||
: 'No credentials are stored yet for this account.'}
|
||||
</p>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Existing values are never shown. Enter the values to (re)store them; they are verified against the provider and encrypted at rest. The token needs permission to create and delete TXT records in your domain's DNS zone."
|
||||
/>
|
||||
<Form form={credForm} layout="vertical">
|
||||
{(credProvider?.credential_fields || []).map(f => (
|
||||
<Form.Item
|
||||
key={f.key}
|
||||
name={`cred_${f.key}`}
|
||||
label={f.label}
|
||||
rules={f.required ? [{ required: true, message: `${f.label} is required` }] : []}
|
||||
extra={f.help ? <span style={{ fontSize: 12, color: token.colorTextSecondary }}>{f.help}</span> : null}
|
||||
>
|
||||
{f.type === 'password'
|
||||
? <Input.Password placeholder={f.label} maxLength={f.max_length || undefined} />
|
||||
: <Input placeholder={f.label} maxLength={f.max_length || undefined} />}
|
||||
</Form.Item>
|
||||
))}
|
||||
{(credProvider?.credential_fields || []).length === 0 && (
|
||||
<Alert type="warning" showIcon message="This provider needs no credentials (manual mode)." />
|
||||
)}
|
||||
</Form>
|
||||
</>
|
||||
)}
|
||||
</Modal>
|
||||
|
||||
{/* v1.5.0 Issue #13: ACME Diagnostic Panel modal */}
|
||||
<Modal
|
||||
title={diagOrderId ? `Diagnostics — Order #${diagOrderId}` : 'Diagnostics'}
|
||||
title={diagOrderId ? `Diagnostics: Order #${diagOrderId}` : 'Diagnostics'}
|
||||
open={diagVisible}
|
||||
onCancel={closeDiagModal}
|
||||
width={920}
|
||||
@@ -1266,12 +1891,21 @@ const ACMEAutomation = () => {
|
||||
{
|
||||
title: 'Re-run', key: 'rerun', width: 90,
|
||||
render: (_, record) => (
|
||||
<Button
|
||||
size="small"
|
||||
icon={<ReloadOutlined />}
|
||||
loading={diagRunningCheckId === record.id}
|
||||
onClick={() => handleRerunCheck(record.id)}
|
||||
/>
|
||||
// A skipped check (e.g. port 80 / routing for a DNS-01 order) has no
|
||||
// transient cause to re-test, so re-running just reproduces "skipped".
|
||||
record.status === 'skipped'
|
||||
? <Typography.Text type="secondary">-</Typography.Text>
|
||||
: (
|
||||
<Tooltip title="Re-run this check">
|
||||
<Button
|
||||
size="small"
|
||||
aria-label="Re-run this check"
|
||||
icon={<ReloadOutlined />}
|
||||
loading={diagRunningCheckId === record.id}
|
||||
onClick={() => handleRerunCheck(record.id)}
|
||||
/>
|
||||
</Tooltip>
|
||||
)
|
||||
),
|
||||
},
|
||||
]}
|
||||
@@ -1327,13 +1961,13 @@ const ACMEAutomation = () => {
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginBottom: 12 }}
|
||||
message="Event log partial — one or more sources failed"
|
||||
message="Event log partial: one or more sources failed"
|
||||
description={
|
||||
<div>
|
||||
<ul style={{ margin: '4px 0 4px 16px' }}>
|
||||
{diagEventsError.errors.map((err, i) => (
|
||||
<li key={i}>
|
||||
<strong>{err.section}</strong>: {err.exception_type} — {err.message}
|
||||
<strong>{err.section}</strong>: {err.exception_type}: {err.message}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
@@ -1359,7 +1993,7 @@ const ACMEAutomation = () => {
|
||||
children: (
|
||||
<div>
|
||||
<div style={{ fontSize: 12, color: '#888' }}>{ev.created_at} · {ev.source}</div>
|
||||
<div><strong>{ev.event_type}</strong></div>
|
||||
<div><strong>{humanizeEventType(ev.event_type)}</strong></div>
|
||||
{ev.message && <div>{ev.message}</div>}
|
||||
</div>
|
||||
),
|
||||
|
||||
@@ -2378,7 +2378,7 @@ const AgentManagement = () => {
|
||||
const element = document.createElement('a');
|
||||
const file = new Blob([installScript], { type: 'text/plain' });
|
||||
element.href = URL.createObjectURL(file);
|
||||
element.download = `install-haproxy-agent-${selectedPlatform}.sh`;
|
||||
element.download = `install-agent-${selectedPlatform}.sh`;
|
||||
document.body.appendChild(element);
|
||||
element.click();
|
||||
document.body.removeChild(element);
|
||||
@@ -2516,7 +2516,7 @@ const AgentManagement = () => {
|
||||
const element = document.createElement('a');
|
||||
const file = new Blob([uninstallScript], { type: 'text/plain' });
|
||||
element.href = URL.createObjectURL(file);
|
||||
element.download = `uninstall-haproxy-agent-${selectedPlatform}.sh`;
|
||||
element.download = `uninstall-agent-${selectedPlatform}.sh`;
|
||||
document.body.appendChild(element);
|
||||
element.click();
|
||||
document.body.removeChild(element);
|
||||
@@ -3147,7 +3147,7 @@ const AgentManagement = () => {
|
||||
const element = document.createElement('a');
|
||||
const file = new Blob([deleteUninstallScript], { type: 'text/plain' });
|
||||
element.href = URL.createObjectURL(file);
|
||||
element.download = `uninstall-haproxy-agent-${agentToDelete.platform || 'linux'}.sh`;
|
||||
element.download = `uninstall-agent-${agentToDelete.platform || 'linux'}.sh`;
|
||||
document.body.appendChild(element);
|
||||
element.click();
|
||||
document.body.removeChild(element);
|
||||
|
||||
@@ -37,6 +37,7 @@ const ApplyManagement = () => {
|
||||
backends: [],
|
||||
waf_rules: [],
|
||||
ssl_certificates: [],
|
||||
vips: [],
|
||||
total_count: 0
|
||||
});
|
||||
const [configVersions, setConfigVersions] = useState([]);
|
||||
@@ -53,6 +54,8 @@ const ApplyManagement = () => {
|
||||
// Validation Error Modal state
|
||||
const [validationErrorModalVisible, setValidationErrorModalVisible] = useState(false);
|
||||
const [selectedValidationError, setSelectedValidationError] = useState(null);
|
||||
// HA/VIP (Issue #27): VIP changes appear in the standard right-panel "Pending Versions"
|
||||
// (vip-* config_versions) and use the standard "View Change" diff — no bespoke VIP modal.
|
||||
|
||||
// Initial load on component mount
|
||||
useEffect(() => {
|
||||
@@ -67,7 +70,7 @@ const ApplyManagement = () => {
|
||||
useEffect(() => {
|
||||
if (selectedCluster) {
|
||||
// CRITICAL: Clear all state immediately when cluster changes to prevent cross-cluster contamination
|
||||
setPendingChanges({ frontends: [], backends: [], waf_rules: [], ssl_certificates: [], total_count: 0 });
|
||||
setPendingChanges({ frontends: [], backends: [], waf_rules: [], ssl_certificates: [], vips: [], total_count: 0 });
|
||||
setConfigVersions([]);
|
||||
setAgentSync(null);
|
||||
setEntitySyncStates({});
|
||||
@@ -77,7 +80,7 @@ const ApplyManagement = () => {
|
||||
fetchConfigVersions();
|
||||
fetchAgentSync();
|
||||
} else {
|
||||
setPendingChanges({ frontends: [], backends: [], waf_rules: [], ssl_certificates: [], total_count: 0 });
|
||||
setPendingChanges({ frontends: [], backends: [], waf_rules: [], ssl_certificates: [], vips: [], total_count: 0 });
|
||||
setConfigVersions([]);
|
||||
setAgentSync(null);
|
||||
setEntitySyncStates({});
|
||||
@@ -125,26 +128,32 @@ const ApplyManagement = () => {
|
||||
'Pragma': 'no-cache'
|
||||
};
|
||||
|
||||
const [frontendsRes, backendsRes, wafRes, sslRes] = await Promise.all([
|
||||
axios.get('/api/frontends', {
|
||||
const [frontendsRes, backendsRes, wafRes, sslRes, vipsRes] = await Promise.all([
|
||||
axios.get('/api/frontends', {
|
||||
params: { cluster_id: selectedCluster.id, include_inactive: true },
|
||||
headers: cacheHeaders
|
||||
}).catch(() => ({ data: { frontends: [] } })),
|
||||
|
||||
axios.get('/api/backends', {
|
||||
|
||||
axios.get('/api/backends', {
|
||||
params: { cluster_id: selectedCluster.id, include_inactive: true },
|
||||
headers: cacheHeaders
|
||||
}).catch(() => ({ data: { backends: [] } })),
|
||||
|
||||
axios.get('/api/waf/rules', {
|
||||
|
||||
axios.get('/api/waf/rules', {
|
||||
params: { cluster_id: selectedCluster.id },
|
||||
headers: cacheHeaders
|
||||
}).catch(() => ({ data: { rules: [] } })),
|
||||
|
||||
axios.get('/api/ssl/certificates', {
|
||||
|
||||
axios.get('/api/ssl/certificates', {
|
||||
params: { cluster_id: selectedCluster.id },
|
||||
headers: cacheHeaders
|
||||
}).catch(() => ({ data: [] }))
|
||||
}).catch(() => ({ data: [] })),
|
||||
|
||||
// HA/VIP (Issue #27): pool-scoped, fetched cluster-scoped so it lists like other entities
|
||||
axios.get('/api/vip', {
|
||||
params: { cluster_id: selectedCluster.id },
|
||||
headers: cacheHeaders
|
||||
}).catch(() => ({ data: { vips: [] } }))
|
||||
]);
|
||||
|
||||
// DEBUG: Log raw backend data
|
||||
@@ -180,14 +189,17 @@ const ApplyManagement = () => {
|
||||
|
||||
const waf_rules = (wafRes.data.rules || []).filter(w => w.has_pending_config);
|
||||
const ssl_certificates = (sslRes.data.ssl_certificates || sslRes.data || []).filter(s => s.has_pending_config);
|
||||
// VIP "pending" is its last_config_status (no has_pending_config flag).
|
||||
const vips = (vipsRes.data.vips || []).filter(v => v.last_config_status === 'PENDING');
|
||||
|
||||
const total_count = frontends.length + backends.length + waf_rules.length + ssl_certificates.length;
|
||||
const total_count = frontends.length + backends.length + waf_rules.length + ssl_certificates.length + vips.length;
|
||||
|
||||
setPendingChanges({
|
||||
frontends,
|
||||
backends,
|
||||
waf_rules,
|
||||
ssl_certificates,
|
||||
vips,
|
||||
total_count
|
||||
});
|
||||
|
||||
@@ -356,7 +368,7 @@ const ApplyManagement = () => {
|
||||
title: 'Apply All Configuration Changes',
|
||||
content: (
|
||||
<div>
|
||||
<p>You are about to apply <strong>{effectiveTotal}</strong> pending changes:</p>
|
||||
<p>You are about to apply <strong>{modalChangeCount}</strong> pending changes:</p>
|
||||
<ul style={{ marginTop: 10, marginBottom: 10 }}>
|
||||
{pendingChanges.frontends.length > 0 && (
|
||||
<li><strong>{pendingChanges.frontends.length}</strong> Frontend changes</li>
|
||||
@@ -370,6 +382,15 @@ const ApplyManagement = () => {
|
||||
{pendingChanges.ssl_certificates.length > 0 && (
|
||||
<li><strong>{pendingChanges.ssl_certificates.length}</strong> SSL certificate changes</li>
|
||||
)}
|
||||
{(pendingChanges.vips || []).length > 0 && (
|
||||
<li><strong>{pendingChanges.vips.length}</strong> HA/VIP changes</li>
|
||||
)}
|
||||
{acmeVersions.length > 0 && (
|
||||
<li><strong>{acmeVersions.length}</strong> ACME Challenge Routing changes</li>
|
||||
)}
|
||||
{otherConfigVersions.length > 0 && (
|
||||
<li><strong>{otherConfigVersions.length}</strong> Other configuration changes</li>
|
||||
)}
|
||||
</ul>
|
||||
<Alert
|
||||
message="All changes will be applied together and sent to agents"
|
||||
@@ -387,6 +408,74 @@ const ApplyManagement = () => {
|
||||
});
|
||||
};
|
||||
|
||||
// HA/VIP convergence tracking (Issue #27 follow-up). VIP teardown/deploy is asynchronous —
|
||||
// member agents converge keepalived on their next poll. Mirror the HAProxy agent-sync widget
|
||||
// so the progress popup keeps showing "Syncing HA/VIP... X/Y" until the nodes report the VIP
|
||||
// ACTIVE, instead of flashing green while the HA/VIP page still shows SYNCING. Fire-and-forget
|
||||
// recursive poll exactly like checkAgentSync (applyLoading is released immediately; this runs
|
||||
// in the background and updates the floating widget). Bounded so an offline node can't poll
|
||||
// forever — it then completes with an informational "still converging" note.
|
||||
const trackVipConvergence = (clusterId, vipIds, startedAt) => {
|
||||
const token = localStorage.getItem('token');
|
||||
const total = vipIds.length;
|
||||
const poll = async () => {
|
||||
let vips = [];
|
||||
try {
|
||||
const r = await axios.get(`/api/vip?cluster_id=${clusterId}`, { headers: { Authorization: `Bearer ${token}` } });
|
||||
vips = r.data.vips || [];
|
||||
} catch (e) { /* transient — keep polling */ }
|
||||
// A change has converged when the node reports the VIP ACTIVE (create/edit) OR the VIP is
|
||||
// fully torn down (delete approval): deploy_status DELETED, or it has dropped off the list
|
||||
// entirely once every member acked the teardown.
|
||||
const isConverged = (vid) => {
|
||||
const v = vips.find(x => x.id === vid);
|
||||
if (!v) return true; // gone from the list → torn down / deleted
|
||||
return v.deploy_status === 'ACTIVE' || v.deploy_status === 'DELETED';
|
||||
};
|
||||
const synced = vipIds.filter(isConverged).length; // VIP-level (drives completion)
|
||||
const errored = vips.filter(v => vipIds.includes(v.id)
|
||||
&& (v.deploy_status === 'ERROR' || v.deploy_status === 'ATTENTION')).length;
|
||||
// Per-NODE progress: sum member acks across the tracked VIPs still present, so a multi-node
|
||||
// VIP shows "1/2 node(s)" like the HA/VIP table — not just "1 change". Gone (torn-down) VIPs
|
||||
// are already counted converged via isConverged.
|
||||
let nodeTotal = 0, nodeSynced = 0;
|
||||
for (const vid of vipIds) {
|
||||
const v = vips.find(x => x.id === vid);
|
||||
if (v) { nodeTotal += (v.deploy_total || 0); nodeSynced += (v.deploy_synced || 0); }
|
||||
}
|
||||
const label = nodeTotal > 0 ? `${nodeSynced}/${nodeTotal} node(s)` : `${synced}/${total} change(s)`;
|
||||
const prog = Math.min(95, 60 + Math.round(35 * (nodeTotal > 0 ? nodeSynced / nodeTotal : synced / Math.max(1, total))));
|
||||
updateEntityCounts(nodeTotal > 0 ? nodeSynced : synced, nodeTotal > 0 ? nodeTotal : total, 0, 0, 0);
|
||||
|
||||
if (synced === total) {
|
||||
const msg = `${total} HA/VIP change(s) fully converged on all member node(s).`;
|
||||
setSyncProgress({ visible: true, step: msg, progress: 100 });
|
||||
completeProgress(msg);
|
||||
setTimeout(() => { setSyncProgress({ visible: false, step: '', progress: 0 }); message.success(msg); }, 1500);
|
||||
return;
|
||||
}
|
||||
if (errored > 0) {
|
||||
const msg = 'HA/VIP applied, but a member node reported an issue — open Diagnostics on the HA / VIP page.';
|
||||
setSyncProgress({ visible: true, step: msg, progress: 100 });
|
||||
completeProgress(msg);
|
||||
setTimeout(() => { setSyncProgress({ visible: false, step: '', progress: 0 }); message.warning(msg); }, 1800);
|
||||
return;
|
||||
}
|
||||
if (Date.now() - startedAt > 300000) { // ~5 min cap (e.g. an offline member node)
|
||||
const msg = `HA/VIP applied — ${label} converged; the rest are still converging (or a node's agent is offline). Track live status on the HA / VIP page.`;
|
||||
setSyncProgress({ visible: true, step: msg, progress: 100 });
|
||||
completeProgress(msg);
|
||||
setTimeout(() => { setSyncProgress({ visible: false, step: '', progress: 0 }); message.info(msg); }, 2000);
|
||||
return;
|
||||
}
|
||||
const step = `Syncing HA/VIP... ${label} converged (keepalived deploy can take a couple of minutes)`;
|
||||
setSyncProgress({ visible: true, step, progress: prog });
|
||||
updateProgress(step, prog, { 'Synced HA/VIP': label });
|
||||
setTimeout(poll, 5000);
|
||||
};
|
||||
setTimeout(poll, 2000);
|
||||
};
|
||||
|
||||
const executeApplyAll = async () => {
|
||||
setApplyLoading(true);
|
||||
|
||||
@@ -429,13 +518,38 @@ const ApplyManagement = () => {
|
||||
|
||||
// Detect if this is a restore operation (safe display-only check)
|
||||
const pendingVersions = configVersions.filter(v => v.status === 'PENDING');
|
||||
const isRestoreOperation = totalEntities === 0 && pendingVersions.some(v => v.version_name.startsWith('restore-'));
|
||||
|
||||
// HA/VIP (Issue #27): vip-* versions are VIP-owned and excluded from the HAProxy apply
|
||||
// (cluster.py). Ignore them when deciding whether this apply has any HAProxy/cluster
|
||||
// work to track — otherwise a VIP-only apply (which now stages a vip-* version) would
|
||||
// look like it has a pending version and the agent-sync tracker would hang on 0/0.
|
||||
const nonVipPendingVersions = pendingVersions.filter(v => !(v.version_name || '').startsWith('vip-'));
|
||||
const isRestoreOperation = totalEntities === 0 && nonVipPendingVersions.some(v => v.version_name.startsWith('restore-'));
|
||||
// HA/VIP-only apply: no HAProxy entity or config-version goes through the cluster-sync
|
||||
// pipeline, so complete promptly after the isolated VIP apply (member nodes converge
|
||||
// async on their next agent poll) instead of looping on "Entities: 0/0".
|
||||
const vipCount = (pendingChanges.vips || []).length;
|
||||
const isVipOnly = totalEntities === 0 && !isRestoreOperation && nonVipPendingVersions.length === 0 && vipCount > 0;
|
||||
// Config-version-only apply (e.g. cluster ACME enable/disable): no entity rows, not a restore,
|
||||
// but there ARE non-vip config versions to push. Without this branch it falls to the "else" and
|
||||
// shows a misleading "Entities: 0/0" while still syncing agents.
|
||||
const isConfigVersionOnly = totalEntities === 0 && !isRestoreOperation && vipCount === 0 && nonVipPendingVersions.length > 0;
|
||||
|
||||
if (isRestoreOperation) {
|
||||
// Restore operation: Show "Configuration" instead of "Entities"
|
||||
setSyncProgress({ visible: true, step: `Applying configuration restore... Configuration: 0/1, Agents: ⏳`, progress: 20 });
|
||||
startProgress('apply', `Applying configuration restore... Configuration: 0/1, Agents: ⏳`);
|
||||
updateEntityCounts(0, 1, 0, totalAgents, disabledAgents); // Show 1 configuration item
|
||||
} else if (isVipOnly) {
|
||||
// HA/VIP-only: avoid the "Entities: 0/0" / agent-sync widget entirely.
|
||||
setSyncProgress({ visible: true, step: `Applying ${vipCount} HA/VIP change(s)...`, progress: 20 });
|
||||
startProgress('apply', `Applying ${vipCount} HA/VIP change(s)...`);
|
||||
updateEntityCounts(0, vipCount, 0, 0, 0);
|
||||
} else if (isConfigVersionOnly) {
|
||||
// Config-version-only (ACME toggle, etc.): show "Configuration" instead of "Entities: 0/0".
|
||||
const cfgCount = nonVipPendingVersions.length;
|
||||
setSyncProgress({ visible: true, step: `Applying configuration change... Configuration: 0/${cfgCount}, Agents: ⏳`, progress: 20 });
|
||||
startProgress('apply', `Applying configuration change... Configuration: 0/${cfgCount}, Agents: ⏳`);
|
||||
updateEntityCounts(0, cfgCount, 0, totalAgents, disabledAgents);
|
||||
} else {
|
||||
// Normal operation: Show "Entities" as usual
|
||||
setSyncProgress({ visible: true, step: `Applying configuration changes... Entities: 0/${totalEntities}, Agents: ⏳`, progress: 20 });
|
||||
@@ -445,16 +559,61 @@ const ApplyManagement = () => {
|
||||
|
||||
try {
|
||||
const token = localStorage.getItem('token');
|
||||
|
||||
const response = await axios.post(
|
||||
`/api/clusters/${selectedCluster.id}/apply-changes`,
|
||||
{},
|
||||
{ headers: { Authorization: `Bearer ${token}` } }
|
||||
);
|
||||
|
||||
|
||||
// HA/VIP (Issue #27): apply pending VIPs first (isolated endpoint; safe no-op for
|
||||
// HAProxy config). Done here so VIPs apply even when there are no HAProxy changes.
|
||||
const pendingVips = pendingChanges.vips || [];
|
||||
for (const vip of pendingVips) {
|
||||
try {
|
||||
await axios.post(`/api/vip/${vip.id}/apply`, {}, { headers: { Authorization: `Bearer ${token}` } });
|
||||
} catch (vipErr) {
|
||||
console.error(`[APPLY VIP ${vip.id}] failed:`, vipErr);
|
||||
message.warning(`VIP "${vip.name}" apply failed: ${extractApiError(vipErr, 'error')}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Apply HAProxy changes if there are entity-level changes OR any non-vip config version
|
||||
// (cluster ACME enable/disable, restore, bulk-import). Gating only on haproxyPending used to
|
||||
// skip the call for config-version-only states, leaving those versions stuck PENDING.
|
||||
const haproxyPending = pendingChanges.frontends.length + pendingChanges.backends.length
|
||||
+ pendingChanges.waf_rules.length + pendingChanges.ssl_certificates.length;
|
||||
const response = (haproxyPending > 0 || nonVipPendingVersions.length > 0)
|
||||
? await axios.post(
|
||||
`/api/clusters/${selectedCluster.id}/apply-changes`,
|
||||
{},
|
||||
{ headers: { Authorization: `Bearer ${token}` } }
|
||||
)
|
||||
: { data: { message: `Applied ${pendingVips.length} HA/VIP change(s)`, applied_count: pendingVips.length } };
|
||||
|
||||
// CRITICAL DEBUG: Log apply response
|
||||
console.log('[APPLY ALL RESPONSE]:', response.data);
|
||||
|
||||
|
||||
// HA/VIP-only apply (Issue #27): nothing goes through the HAProxy config-version /
|
||||
// cluster-sync pipeline, so the agent-sync tracker below would loop forever on
|
||||
// "Entities: 0/0". The isolated VIP apply already staged each node's snapshot; member
|
||||
// agents install/configure keepalived and converge on their next poll — live status
|
||||
// shows on the HA / VIP page (PENDING → SYNCING → ACTIVE). Complete now (no flash).
|
||||
if (haproxyPending === 0 && nonVipPendingVersions.length === 0) {
|
||||
await fetchPendingChanges();
|
||||
await fetchConfigVersions();
|
||||
if (pendingVips.length > 0) {
|
||||
// Keep the popup in a "Syncing HA/VIP... X/Y" state (like every other entity) until the
|
||||
// member nodes report the VIP ACTIVE — instead of flashing green while the HA/VIP page
|
||||
// still shows SYNCING. The poll runs in the background (finally{} releases applyLoading).
|
||||
const step = `Applied — syncing HA/VIP... 0/${pendingVips.length} node(s) converged`;
|
||||
updateEntityCounts(0, pendingVips.length, 0, 0, 0);
|
||||
setSyncProgress({ visible: true, step, progress: 60 });
|
||||
updateProgress(step, 60);
|
||||
trackVipConvergence(selectedCluster.id, pendingVips.map(v => v.id), Date.now());
|
||||
} else {
|
||||
const vmsg = 'Changes applied.';
|
||||
setSyncProgress({ visible: true, step: vmsg, progress: 100 });
|
||||
completeProgress(vmsg);
|
||||
setTimeout(() => { setSyncProgress({ visible: false, step: '', progress: 0 }); message.success(vmsg); }, 1500);
|
||||
}
|
||||
return; // finally{} resets applyLoading; the VIP poll runs in the background (like checkAgentSync)
|
||||
}
|
||||
|
||||
setSyncProgress({ visible: true, step: `Configuration applied, syncing agents... Entities: ${totalEntities}/${totalEntities}, Agents: 0/${totalAgents}`, progress: 60 });
|
||||
updateProgress(`Configuration applied, syncing agents... Entities: ${totalEntities}/${totalEntities}, Agents: 0/${totalAgents}`, 60);
|
||||
updateEntityCounts(totalEntities, totalEntities, 0, totalAgents, disabledAgents);
|
||||
@@ -664,7 +823,7 @@ const ApplyManagement = () => {
|
||||
title: 'Reject All Configuration Changes',
|
||||
content: (
|
||||
<div>
|
||||
<p>You are about to reject <strong>{effectiveTotal}</strong> pending changes:</p>
|
||||
<p>You are about to reject <strong>{modalChangeCount}</strong> pending changes:</p>
|
||||
<ul style={{ marginTop: 10, marginBottom: 10 }}>
|
||||
{pendingChanges.frontends.length > 0 && (
|
||||
<li><strong>{pendingChanges.frontends.length}</strong> Frontend changes</li>
|
||||
@@ -678,6 +837,15 @@ const ApplyManagement = () => {
|
||||
{pendingChanges.ssl_certificates.length > 0 && (
|
||||
<li><strong>{pendingChanges.ssl_certificates.length}</strong> SSL certificate changes</li>
|
||||
)}
|
||||
{(pendingChanges.vips || []).length > 0 && (
|
||||
<li><strong>{pendingChanges.vips.length}</strong> HA/VIP changes</li>
|
||||
)}
|
||||
{acmeVersions.length > 0 && (
|
||||
<li><strong>{acmeVersions.length}</strong> ACME Challenge Routing changes</li>
|
||||
)}
|
||||
{otherConfigVersions.length > 0 && (
|
||||
<li><strong>{otherConfigVersions.length}</strong> Other configuration changes</li>
|
||||
)}
|
||||
</ul>
|
||||
<Alert
|
||||
message="All pending changes will be permanently discarded"
|
||||
@@ -718,12 +886,34 @@ const ApplyManagement = () => {
|
||||
|
||||
try {
|
||||
const token = localStorage.getItem('token');
|
||||
|
||||
const response = await axios.delete(
|
||||
`/api/clusters/${selectedCluster.id}/pending-changes`,
|
||||
{ headers: { Authorization: `Bearer ${token}` } }
|
||||
|
||||
// HA/VIP (Issue #27): reject pending VIPs (isolated endpoint; restores last applied state).
|
||||
const pendingVips = pendingChanges.vips || [];
|
||||
for (const vip of pendingVips) {
|
||||
try {
|
||||
await axios.post(`/api/vip/${vip.id}/reject`, {}, { headers: { Authorization: `Bearer ${token}` } });
|
||||
} catch (vipErr) {
|
||||
console.error(`[REJECT VIP ${vip.id}] failed:`, vipErr);
|
||||
message.warning(`VIP "${vip.name}" reject failed: ${extractApiError(vipErr, 'error')}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Reject HAProxy changes if there are entity-level changes OR any non-vip config version
|
||||
// (e.g. cluster ACME enable/disable, restore, bulk-import) — the backend DELETE rejects all
|
||||
// non-vip PENDING versions and rolls back their snapshots. Gating only on haproxyPending used
|
||||
// to skip the call for config-version-only states, returning the misleading "Rejected 0 HA/VIP".
|
||||
const haproxyPending = pendingChanges.frontends.length + pendingChanges.backends.length
|
||||
+ pendingChanges.waf_rules.length + pendingChanges.ssl_certificates.length;
|
||||
const nonVipPendingVersions = configVersions.filter(
|
||||
v => v.status === 'PENDING' && !(v.version_name || '').startsWith('vip-')
|
||||
);
|
||||
|
||||
const response = (haproxyPending > 0 || nonVipPendingVersions.length > 0)
|
||||
? await axios.delete(
|
||||
`/api/clusters/${selectedCluster.id}/pending-changes`,
|
||||
{ headers: { Authorization: `Bearer ${token}` } }
|
||||
)
|
||||
: { data: { message: `Rejected ${pendingVips.length} HA/VIP change(s)` } };
|
||||
|
||||
// CRITICAL DEBUG: Log reject response
|
||||
console.log('[REJECT ALL RESPONSE]:', response.data);
|
||||
|
||||
@@ -797,6 +987,23 @@ const ApplyManagement = () => {
|
||||
const appliedVersions = configVersions.filter(v => v.status === 'APPLIED');
|
||||
const rejectedVersions = configVersions.filter(v => v.status === 'REJECTED');
|
||||
const effectiveTotal = pendingChanges.total_count > 0 ? pendingChanges.total_count : pendingVersions.length;
|
||||
// Issue #35: cluster ACME enable/disable produce `cluster-<id>-acme-<enable|disable>-<ts>` config
|
||||
// versions that have NO entity-level pending flag, so they were neither categorized nor counted.
|
||||
const ACME_VERSION_RE = /^cluster-\d+-acme-(enable|disable)-/;
|
||||
const ENTITY_VERSION_PREFIXES = ['frontend-', 'backend-', 'server-', 'ssl-', 'waf-'];
|
||||
const acmeVersions = pendingVersions.filter(v => ACME_VERSION_RE.test(v.version_name || ''));
|
||||
// "Other" config versions for the confirm modal = non-vip, non-acme, non-entity-backed (i.e.
|
||||
// restore-*/bulk-import-*/other cluster-level) — entity-backed versions are already counted via
|
||||
// total_count, and vips are listed separately, so excluding them avoids double-counting.
|
||||
const otherConfigVersions = pendingVersions.filter(v => {
|
||||
const n = v.version_name || '';
|
||||
if (n.startsWith('vip-') || ACME_VERSION_RE.test(n)) return false;
|
||||
return !ENTITY_VERSION_PREFIXES.some(p => n.startsWith(p));
|
||||
});
|
||||
// Confirm-modal header count: entities+VIPs (total_count) + ACME + other config versions, so the
|
||||
// header equals the sum of the listed <li> items in every state. The button-enable gate keeps
|
||||
// using effectiveTotal (unchanged) so entity-only button/Alert behavior is byte-identical.
|
||||
const modalChangeCount = (pendingChanges.total_count || 0) + acmeVersions.length + otherConfigVersions.length;
|
||||
|
||||
const renderPendingItem = (item, type, icon) => {
|
||||
// For PENDING items, don't show sync status since they haven't been applied yet
|
||||
@@ -885,7 +1092,7 @@ const ApplyManagement = () => {
|
||||
style={{
|
||||
marginBottom: 24,
|
||||
borderRadius: 8,
|
||||
border: '1px solid #ffccc7',
|
||||
border: `1px solid ${token.colorErrorBorder}`,
|
||||
boxShadow: '0 2px 8px rgba(255, 77, 79, 0.15)'
|
||||
}}
|
||||
message={
|
||||
@@ -1133,12 +1340,67 @@ const ApplyManagement = () => {
|
||||
</div>
|
||||
)}
|
||||
|
||||
{pendingChanges.frontends.length === 0 && pendingChanges.backends.length === 0 && pendingChanges.waf_rules.length === 0 && pendingChanges.ssl_certificates.length === 0 && pendingVersions.length > 0 && (
|
||||
{/* HA/VIP Changes (Issue #27) */}
|
||||
{(pendingChanges.vips || []).length > 0 && (
|
||||
<div style={{ marginBottom: 16 }}>
|
||||
<Title level={5}>
|
||||
<CloudServerOutlined style={{ marginRight: 8, color: '#13c2c2' }} />
|
||||
HA / VIP Changes ({pendingChanges.vips.length})
|
||||
</Title>
|
||||
{pendingChanges.vips.map(item => (
|
||||
<div key={`vip-${item.id}`} style={{ display: 'flex', alignItems: 'center', gap: 8, padding: '8px 12px', marginBottom: 6, border: `1px solid ${item.pending_delete ? token.colorErrorBorder : token.colorBorderSecondary}`, borderRadius: 6, background: item.pending_delete ? token.colorErrorBg : undefined }}>
|
||||
<CloudServerOutlined style={{ color: item.pending_delete ? '#cf1322' : '#13c2c2' }} />
|
||||
<span style={{ fontWeight: 500 }}>{item.name}</span>
|
||||
<Tag>{item.virtual_ip}/{item.prefix_length}</Tag>
|
||||
{item.pool_name && <Tag color="blue">{item.pool_name}</Tag>}
|
||||
{item.pending_delete
|
||||
? <Tag color="red">DELETION — approve to tear down, reject to keep</Tag>
|
||||
: <Tag color="orange">PENDING</Tag>}
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Issue #35: ACME Challenge Routing (cluster-<id>-acme-*) versions have no entity
|
||||
flag. Render them in their own section REGARDLESS of whether entity sections are
|
||||
present, so a co-pending ACME toggle is never hidden in the left panel. */}
|
||||
{acmeVersions.length > 0 && (
|
||||
<div style={{ marginTop: 8, marginBottom: 8 }}>
|
||||
<Title level={5}>
|
||||
<SafetyCertificateOutlined style={{ marginRight: 8, color: '#1890ff' }} />
|
||||
ACME Challenge Routing ({acmeVersions.length})
|
||||
</Title>
|
||||
{acmeVersions.map(v => {
|
||||
const isEnable = /^cluster-\d+-acme-enable-/.test(v.version_name);
|
||||
return (
|
||||
<div key={v.id} style={{
|
||||
padding: 10, border: `1px dashed ${token.colorPrimary}`, borderRadius: 6, marginBottom: 8,
|
||||
display: 'flex', alignItems: 'center', justifyContent: 'space-between', backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<span>
|
||||
<Tag color={isEnable ? 'green' : 'default'}>{isEnable ? 'ENABLE' : 'DISABLE'}</Tag>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
</span>
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
<div style={{ fontSize: 12, color: token.colorTextSecondary, marginTop: 4 }}>
|
||||
ACME challenge routing change. Apply to push the updated HAProxy config to the agents.
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{pendingChanges.frontends.length === 0 && pendingChanges.backends.length === 0 && pendingChanges.waf_rules.length === 0 && pendingChanges.ssl_certificates.length === 0 && (pendingChanges.vips || []).length === 0 && pendingVersions.length > 0 && (
|
||||
<div style={{ marginTop: 8 }}>
|
||||
{(() => {
|
||||
const restoreVersions = pendingVersions.filter(v => v.version_name.startsWith('restore-'));
|
||||
const bulkImportVersions = pendingVersions.filter(v => v.version_name.startsWith('bulk-import-'));
|
||||
const otherVersions = pendingVersions.filter(v => !v.version_name.startsWith('restore-') && !v.version_name.startsWith('bulk-import-'));
|
||||
// Exclude restore-/bulk-import- (own sections), vip-* (VIP section), and acme-*
|
||||
// (the dedicated ACME section above) so they aren't duplicated in "Other".
|
||||
const otherVersions = pendingVersions.filter(v =>
|
||||
!v.version_name.startsWith('restore-') && !v.version_name.startsWith('bulk-import-')
|
||||
&& !v.version_name.startsWith('vip-') && !ACME_VERSION_RE.test(v.version_name));
|
||||
|
||||
return (
|
||||
<>
|
||||
@@ -1157,7 +1419,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f0f8ff'
|
||||
backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1181,7 +1443,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f6ffed'
|
||||
backgroundColor: token.colorSuccessBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1256,9 +1518,13 @@ const ApplyManagement = () => {
|
||||
</Descriptions>
|
||||
</div>
|
||||
|
||||
{/* Pending Versions Section */}
|
||||
{/* Pending Versions Section.
|
||||
Theme tokens, not the light-mode literals #fffbe6/#ffe58f: in dark mode those
|
||||
produced a cream panel with light text on it, so the version name, timestamp
|
||||
and "View Change" link were unreadable. colorWarningBg/Border track the
|
||||
algorithm, so the "pending" tint survives in both themes. */}
|
||||
{pendingVersions.length > 0 && (
|
||||
<div style={{ marginBottom: 24, background: '#fffbe6', borderRadius: 8, border: '1px solid #ffe58f', padding: '16px 16px 8px' }}>
|
||||
<div style={{ marginBottom: 24, background: token.colorWarningBg, borderRadius: 8, border: `1px solid ${token.colorWarningBorder}`, padding: '16px 16px 8px' }}>
|
||||
<Title level={5} style={{ marginTop: 0 }}>
|
||||
<ClockCircleOutlined style={{ marginRight: 8, color: '#faad14' }} />
|
||||
Pending Changes ({pendingVersions.length})
|
||||
@@ -1401,6 +1667,10 @@ const ApplyManagement = () => {
|
||||
confusing error post-click. */}
|
||||
{(() => {
|
||||
const vn = version?.version_name || '';
|
||||
// vip-* versions ARE undoable (Issue #27): the VIP router
|
||||
// re-stages the rejected change as PENDING (reactivating a
|
||||
// rejected create, or re-applying a rejected edit), so they
|
||||
// use the normal Undo button below — not this disabled case.
|
||||
const destructive = (
|
||||
vn.startsWith('bulk-site-create-')
|
||||
|| vn.startsWith('bulk-import-')
|
||||
@@ -1499,8 +1769,8 @@ const ApplyManagement = () => {
|
||||
<div>
|
||||
{/* Parsed suggestion */}
|
||||
{agentSync.parsed_error?.suggestion && (
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: '#fff7e6', borderRadius: 4, border: '1px solid #ffd591' }}>
|
||||
<Text strong style={{ color: '#ad4e00' }}>Recommendation: </Text>
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: token.colorWarningBg, borderRadius: 4, border: `1px solid ${token.colorWarningBorder}` }}>
|
||||
<Text strong style={{ color: token.colorWarningText }}>Recommendation: </Text>
|
||||
<Text>{agentSync.parsed_error.suggestion}</Text>
|
||||
</div>
|
||||
)}
|
||||
@@ -1685,13 +1955,17 @@ const ApplyManagement = () => {
|
||||
{diffData.changes && diffData.changes.length > 0 ? (
|
||||
diffData.changes.map((change, index) => (
|
||||
<div key={index} style={{ marginBottom: '4px' }}>
|
||||
{/* Token-based, not the light literals #f6ffed/#fff2f0: those stayed
|
||||
near-white in dark mode, so the added/removed rows glared against
|
||||
the dark diff panel around them. colorSuccessBg/colorErrorBg darken
|
||||
with the algorithm while keeping the green/red semantics. */}
|
||||
{change.type === 'added' && (
|
||||
<div style={{ backgroundColor: '#f6ffed', color: '#52c41a', padding: '2px 8px', borderLeft: '3px solid #52c41a' }}>
|
||||
<div style={{ backgroundColor: token.colorSuccessBg, color: token.colorSuccessText, padding: '2px 8px', borderLeft: `3px solid ${token.colorSuccess}` }}>
|
||||
+ {change.line}
|
||||
</div>
|
||||
)}
|
||||
{change.type === 'removed' && (
|
||||
<div style={{ backgroundColor: '#fff2f0', color: '#ff4d4f', padding: '2px 8px', borderLeft: '3px solid #ff4d4f' }}>
|
||||
<div style={{ backgroundColor: token.colorErrorBg, color: token.colorErrorText, padding: '2px 8px', borderLeft: `3px solid ${token.colorError}` }}>
|
||||
- {change.line}
|
||||
</div>
|
||||
)}
|
||||
|
||||
@@ -172,11 +172,13 @@ const BackendServers = () => {
|
||||
|
||||
setLoading(true);
|
||||
try {
|
||||
const params = { cluster_id: selectedCluster.id };
|
||||
|
||||
// Issue #24: request inactive entities so DISABLED (toggled-OFF) servers
|
||||
// are returned and can be reactivated from the UI (mirrors ApplyManagement).
|
||||
const params = { cluster_id: selectedCluster.id, include_inactive: true };
|
||||
|
||||
// CRITICAL FIX: Add cache busting to prevent stale data from appearing
|
||||
// Browser/axios may cache GET requests, causing deleted backends to reappear
|
||||
const response = await axios.get('/api/backends', {
|
||||
const response = await axios.get('/api/backends', {
|
||||
params,
|
||||
headers: {
|
||||
'Cache-Control': 'no-cache, no-store, must-revalidate',
|
||||
@@ -184,7 +186,21 @@ const BackendServers = () => {
|
||||
'Expires': '0'
|
||||
}
|
||||
});
|
||||
const fetchedBackends = response.data.backends || [];
|
||||
// Issue #24: we now request include_inactive=true so DISABLED servers come
|
||||
// back. That ALSO returns soft-deleted (is_active=false) BACKENDS, which the
|
||||
// pre-change default (include_inactive=false -> WHERE is_active=TRUE) hid.
|
||||
// Restore that filter here so soft-deleted backends don't reappear in the
|
||||
// normal view (preserve commit f34a6ee), while still keeping inactive
|
||||
// SERVERS inside active backends. Then hide ONLY soft-deleted-pending
|
||||
// servers (last_config_status==='DELETION'); DISABLED servers (toggled
|
||||
// OFF — status PENDING/APPLIED) stay visible so operators can re-enable
|
||||
// them. Single chokepoint covers counts, expanded rows, and All Servers tab.
|
||||
const fetchedBackends = (response.data.backends || [])
|
||||
.filter(b => b.is_active !== false)
|
||||
.map(b => ({
|
||||
...b,
|
||||
servers: (b.servers || []).filter(s => s.last_config_status !== 'DELETION'),
|
||||
}));
|
||||
setBackends(fetchedBackends);
|
||||
// CRITICAL FIX: Apply status filters after fetching to maintain filter state
|
||||
// This prevents backends from disappearing when updated (e.g., APPLIED → PENDING)
|
||||
@@ -1289,6 +1305,7 @@ const BackendServers = () => {
|
||||
/>
|
||||
<strong>{text}</strong>
|
||||
{record.backup_server && <Tag color="orange">Backup</Tag>}
|
||||
{!record.is_active && <Tag color="red">Inactive</Tag>}
|
||||
</Space>
|
||||
),
|
||||
},
|
||||
|
||||
@@ -458,6 +458,8 @@ backend web-backend
|
||||
{record.request_headers && <Tag color="blue">Req Headers</Tag>}
|
||||
{record.response_headers && <Tag color="green">Resp Headers</Tag>}
|
||||
{record.tcp_request_rules && <Tag color="purple">TCP Rules</Tag>}
|
||||
{record.filters && <Tag color="magenta">Filters</Tag>}
|
||||
{record.log_format && <Tag color="geekblue">Log Format</Tag>}
|
||||
{record.acl_rules && record.acl_rules.length > 0 && <Tag color="orange">{record.acl_rules.length} ACLs</Tag>}
|
||||
{record.use_backend_rules && record.use_backend_rules.length > 0 && <Tag color="cyan">{record.use_backend_rules.length} Routes</Tag>}
|
||||
</Space>
|
||||
@@ -1076,8 +1078,9 @@ backend web-backend
|
||||
size="small"
|
||||
expandable={{
|
||||
expandedRowRender: (frontend) => {
|
||||
const hasDetails = frontend.request_headers || frontend.response_headers ||
|
||||
frontend.options || frontend.tcp_request_rules ||
|
||||
const hasDetails = frontend.request_headers || frontend.response_headers ||
|
||||
frontend.options || frontend.tcp_request_rules ||
|
||||
frontend.filters || frontend.log_format ||
|
||||
(frontend.acl_rules && frontend.acl_rules.length > 0) ||
|
||||
(frontend.use_backend_rules && frontend.use_backend_rules.length > 0);
|
||||
|
||||
@@ -1157,12 +1160,52 @@ backend web-backend
|
||||
</span>
|
||||
}
|
||||
>
|
||||
<MultiLineDiffRenderer
|
||||
<MultiLineDiffRenderer
|
||||
value={frontend.tcp_request_rules}
|
||||
changeInfo={frontend._changes?.tcp_request_rules}
|
||||
/>
|
||||
</Descriptions.Item>
|
||||
)}
|
||||
{/* Issue #38: SPOE filters */}
|
||||
{frontend.filters && (
|
||||
<Descriptions.Item
|
||||
label={
|
||||
<span>
|
||||
Filters (SPOE/WAF)
|
||||
{frontend._changes?.filters && (
|
||||
<Tag color="green" style={{ marginLeft: 8, fontSize: '10px' }}>
|
||||
{frontend._changes.filters.old ? 'CHANGED' : 'NEW'}
|
||||
</Tag>
|
||||
)}
|
||||
</span>
|
||||
}
|
||||
>
|
||||
<MultiLineDiffRenderer
|
||||
value={frontend.filters}
|
||||
changeInfo={frontend._changes?.filters}
|
||||
/>
|
||||
</Descriptions.Item>
|
||||
)}
|
||||
{/* Issue #38: frontend log-format */}
|
||||
{frontend.log_format && (
|
||||
<Descriptions.Item
|
||||
label={
|
||||
<span>
|
||||
Log Format
|
||||
{frontend._changes?.log_format && (
|
||||
<Tag color="green" style={{ marginLeft: 8, fontSize: '10px' }}>
|
||||
{frontend._changes.log_format.old ? 'CHANGED' : 'NEW'}
|
||||
</Tag>
|
||||
)}
|
||||
</span>
|
||||
}
|
||||
>
|
||||
<MultiLineDiffRenderer
|
||||
value={frontend.log_format}
|
||||
changeInfo={frontend._changes?.log_format}
|
||||
/>
|
||||
</Descriptions.Item>
|
||||
)}
|
||||
{frontend.acl_rules && frontend.acl_rules.length > 0 && (
|
||||
<Descriptions.Item label={`ACL Rules (${frontend.acl_rules.length})`}>
|
||||
{frontend.acl_rules.map((acl, idx) => (
|
||||
|
||||
@@ -0,0 +1,886 @@
|
||||
import React, { useState, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Card, Table, Button, Modal, Form, Input, Space, message,
|
||||
Popconfirm, Tag, Tooltip, Row, Col, Typography, Alert, Select,
|
||||
Collapse, Descriptions, Badge
|
||||
} from 'antd';
|
||||
import {
|
||||
PlusOutlined, ReloadOutlined, DeleteOutlined, EyeOutlined,
|
||||
DownloadOutlined, CopyOutlined, FileProtectOutlined, ImportOutlined,
|
||||
KeyOutlined
|
||||
} from '@ant-design/icons';
|
||||
import axios from 'axios';
|
||||
import { useCluster } from '../contexts/ClusterContext';
|
||||
import { extractApiError } from '../utils/apiError';
|
||||
import { antdDomainRule, antdDomainsListRule } from '../utils/validation';
|
||||
|
||||
const { Text } = Typography;
|
||||
const { TextArea } = Input;
|
||||
|
||||
const KEY_ALGORITHM_OPTIONS = [
|
||||
{ value: 'rsa-2048', label: 'RSA 2048 (recommended)' },
|
||||
{ value: 'rsa-4096', label: 'RSA 4096' },
|
||||
{ value: 'ecdsa-p256', label: 'ECDSA P-256' },
|
||||
{ value: 'ecdsa-p384', label: 'ECDSA P-384' },
|
||||
];
|
||||
|
||||
const KEY_ALGORITHM_LABELS = {
|
||||
'rsa-2048': 'RSA 2048',
|
||||
'rsa-4096': 'RSA 4096',
|
||||
'ecdsa-p256': 'ECDSA P-256',
|
||||
'ecdsa-p384': 'ECDSA P-384',
|
||||
};
|
||||
|
||||
// CSR creation (v1.9.0): generate the private key + CSR server-side, submit
|
||||
// the CSR to an external CA, then import the signed certificate. The private
|
||||
// key never leaves the backend — this component only ever handles the CSR
|
||||
// PEM and the CA's certificate response.
|
||||
const CSRManagement = ({ onCertificateImported }) => {
|
||||
const { clusters } = useCluster();
|
||||
const [csrs, setCsrs] = useState([]);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [createModalOpen, setCreateModalOpen] = useState(false);
|
||||
const [creating, setCreating] = useState(false);
|
||||
const [viewCsr, setViewCsr] = useState(null);
|
||||
const [importCsr, setImportCsr] = useState(null);
|
||||
const [importing, setImporting] = useState(false);
|
||||
const [createForm] = Form.useForm();
|
||||
const [importForm] = Form.useForm();
|
||||
|
||||
const fetchCsrs = useCallback(async () => {
|
||||
setLoading(true);
|
||||
try {
|
||||
const response = await axios.get('/api/ssl/csrs', {
|
||||
headers: {
|
||||
'Cache-Control': 'no-cache, no-store, must-revalidate',
|
||||
'Pragma': 'no-cache',
|
||||
},
|
||||
});
|
||||
setCsrs(Array.isArray(response.data) ? response.data : []);
|
||||
} catch (error) {
|
||||
console.error('Error fetching CSRs:', error);
|
||||
message.error(extractApiError(error, 'Failed to fetch CSRs'));
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
fetchCsrs();
|
||||
}, [fetchCsrs]);
|
||||
|
||||
const handleCreate = async (values) => {
|
||||
setCreating(true);
|
||||
try {
|
||||
const payload = {
|
||||
name: values.name,
|
||||
common_name: values.common_name,
|
||||
sans: values.sans || [],
|
||||
key_algorithm: values.key_algorithm || 'rsa-2048',
|
||||
organization: values.organization || null,
|
||||
organizational_unit: values.organizational_unit || null,
|
||||
locality: values.locality || null,
|
||||
state: values.state || null,
|
||||
country: values.country || null,
|
||||
email: values.email || null,
|
||||
};
|
||||
const response = await axios.post('/api/ssl/csrs', payload);
|
||||
message.success(
|
||||
<div>
|
||||
<strong>CSR '{values.name}' created</strong>
|
||||
<br />
|
||||
<small>Submit the CSR to your Certificate Authority for signing.</small>
|
||||
</div>,
|
||||
5
|
||||
);
|
||||
setCreateModalOpen(false);
|
||||
createForm.resetFields();
|
||||
fetchCsrs();
|
||||
// Open the view modal immediately so the operator can copy/download
|
||||
// the CSR PEM in one round trip.
|
||||
if (response.data?.csr) {
|
||||
setViewCsr(response.data.csr);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Error creating CSR:', error);
|
||||
message.error(extractApiError(error, 'Failed to create CSR'));
|
||||
} finally {
|
||||
setCreating(false);
|
||||
}
|
||||
};
|
||||
|
||||
const handleView = async (record) => {
|
||||
try {
|
||||
const response = await axios.get(`/api/ssl/csrs/${record.id}`);
|
||||
setViewCsr(response.data);
|
||||
} catch (error) {
|
||||
console.error('Error fetching CSR details:', error);
|
||||
message.error(extractApiError(error, 'Failed to fetch CSR details'));
|
||||
}
|
||||
};
|
||||
|
||||
const handleDuplicate = (record) => {
|
||||
const subject = record.subject || {};
|
||||
createForm.setFieldsValue({
|
||||
name: `${record.name}-new`,
|
||||
common_name: record.common_name,
|
||||
sans: (record.sans || []).filter((s) => s !== record.common_name),
|
||||
key_algorithm: record.key_algorithm || 'rsa-2048',
|
||||
organization: subject.O || undefined,
|
||||
organizational_unit: subject.OU || undefined,
|
||||
locality: subject.L || undefined,
|
||||
state: subject.ST || undefined,
|
||||
country: subject.C || undefined,
|
||||
email: subject.emailAddress || undefined,
|
||||
});
|
||||
setCreateModalOpen(true);
|
||||
};
|
||||
|
||||
const handleDelete = async (record) => {
|
||||
try {
|
||||
const response = await axios.delete(`/api/ssl/csrs/${record.id}`);
|
||||
message.success(response.data?.message || `CSR '${record.name}' deleted`);
|
||||
fetchCsrs();
|
||||
} catch (error) {
|
||||
console.error('Error deleting CSR:', error);
|
||||
message.error(extractApiError(error, 'Failed to delete CSR'));
|
||||
}
|
||||
};
|
||||
|
||||
const handleImport = async (values) => {
|
||||
if (!importCsr) return;
|
||||
setImporting(true);
|
||||
try {
|
||||
const isGlobal = values.ssl_type === 'global';
|
||||
const payload = {
|
||||
certificate_content: values.certificate_content,
|
||||
chain_content: values.chain_content || null,
|
||||
usage_type: values.usage_type || 'frontend',
|
||||
is_global: isGlobal,
|
||||
cluster_ids: isGlobal ? null : values.cluster_ids,
|
||||
name: values.name_override ? values.name_override.trim() : null,
|
||||
};
|
||||
const response = await axios.post(
|
||||
`/api/ssl/csrs/${importCsr.id}/import`,
|
||||
payload
|
||||
);
|
||||
const warnings = response.data?.warnings || [];
|
||||
if (warnings.length > 0) {
|
||||
Modal.warning({
|
||||
title: 'Certificate imported with warnings',
|
||||
width: 560,
|
||||
content: (
|
||||
<ul style={{ paddingLeft: 18, marginTop: 8 }}>
|
||||
{warnings.map((w, i) => (
|
||||
<li key={i}>{w}</li>
|
||||
))}
|
||||
</ul>
|
||||
),
|
||||
});
|
||||
}
|
||||
message.success(
|
||||
<div>
|
||||
<strong>Certificate imported successfully</strong>
|
||||
<br />
|
||||
<small>Go to Apply Management to deploy it to the cluster(s).</small>
|
||||
</div>,
|
||||
6
|
||||
);
|
||||
setImportCsr(null);
|
||||
importForm.resetFields();
|
||||
fetchCsrs();
|
||||
if (onCertificateImported) {
|
||||
onCertificateImported();
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Error importing signed certificate:', error);
|
||||
message.error(extractApiError(error, 'Failed to import certificate'));
|
||||
} finally {
|
||||
setImporting(false);
|
||||
}
|
||||
};
|
||||
|
||||
const downloadCsrPem = (csr) => {
|
||||
if (!csr?.csr_pem) return;
|
||||
const blob = new Blob([csr.csr_pem], { type: 'application/pkcs10;charset=utf-8' });
|
||||
const url = URL.createObjectURL(blob);
|
||||
const link = document.createElement('a');
|
||||
link.href = url;
|
||||
link.download = `${csr.name}.csr`;
|
||||
document.body.appendChild(link);
|
||||
link.click();
|
||||
document.body.removeChild(link);
|
||||
URL.revokeObjectURL(url);
|
||||
};
|
||||
|
||||
const copyCsrPem = (csr) => {
|
||||
if (!csr?.csr_pem) return;
|
||||
if (navigator.clipboard && navigator.clipboard.writeText) {
|
||||
navigator.clipboard
|
||||
.writeText(csr.csr_pem)
|
||||
.then(() => message.success('CSR PEM copied to clipboard'))
|
||||
.catch(() => message.error('Failed to copy CSR PEM'));
|
||||
} else {
|
||||
message.warning('Clipboard is not available in this browser');
|
||||
}
|
||||
};
|
||||
|
||||
const columns = [
|
||||
{
|
||||
title: 'CSR',
|
||||
dataIndex: 'name',
|
||||
key: 'name',
|
||||
render: (text, record) => (
|
||||
<Space>
|
||||
<FileProtectOutlined style={{ color: '#1677ff' }} />
|
||||
<div>
|
||||
<strong>{text}</strong>
|
||||
<br />
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
{record.common_name}
|
||||
</Text>
|
||||
</div>
|
||||
</Space>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'SANs',
|
||||
dataIndex: 'sans',
|
||||
key: 'sans',
|
||||
render: (sans) => {
|
||||
const list = Array.isArray(sans) ? sans : [];
|
||||
if (list.length === 0) return <Text type="secondary">-</Text>;
|
||||
const visible = list.slice(0, 2);
|
||||
const rest = list.slice(2);
|
||||
return (
|
||||
<Space size={4} wrap>
|
||||
{visible.map((d) => (
|
||||
<Tag key={d}>{d}</Tag>
|
||||
))}
|
||||
{rest.length > 0 && (
|
||||
<Tooltip title={rest.join(', ')}>
|
||||
<Tag>+{rest.length}</Tag>
|
||||
</Tooltip>
|
||||
)}
|
||||
</Space>
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Key',
|
||||
dataIndex: 'key_algorithm',
|
||||
key: 'key_algorithm',
|
||||
render: (algo) => (
|
||||
<Tag icon={<KeyOutlined />} color="geekblue">
|
||||
{KEY_ALGORITHM_LABELS[algo] || algo}
|
||||
</Tag>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Status',
|
||||
dataIndex: 'status',
|
||||
key: 'status',
|
||||
render: (status, record) => {
|
||||
if (status === 'completed') {
|
||||
return (
|
||||
<div>
|
||||
<Badge status="success" text="Imported" />
|
||||
{record.certificate_name && (
|
||||
<>
|
||||
<br />
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
→ {record.certificate_name}
|
||||
</Text>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
return <Badge status="processing" text="Awaiting certificate" />;
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Created',
|
||||
dataIndex: 'created_at',
|
||||
key: 'created_at',
|
||||
render: (date, record) => (
|
||||
<div>
|
||||
{date
|
||||
? new Date(date).toLocaleString(undefined, {
|
||||
year: 'numeric',
|
||||
month: 'short',
|
||||
day: 'numeric',
|
||||
hour: '2-digit',
|
||||
minute: '2-digit',
|
||||
})
|
||||
: '-'}
|
||||
{record.created_by_username && (
|
||||
<>
|
||||
<br />
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
by {record.created_by_username}
|
||||
</Text>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Actions',
|
||||
key: 'actions',
|
||||
render: (_, record) => (
|
||||
<Space size="small">
|
||||
<Tooltip title="View / download CSR">
|
||||
<Button
|
||||
type="text"
|
||||
size="small"
|
||||
icon={<EyeOutlined />}
|
||||
onClick={() => handleView(record)}
|
||||
/>
|
||||
</Tooltip>
|
||||
{record.status === 'pending' && (
|
||||
<Tooltip title="Import signed certificate">
|
||||
<Button
|
||||
type="primary"
|
||||
size="small"
|
||||
icon={<ImportOutlined />}
|
||||
onClick={() => {
|
||||
importForm.resetFields();
|
||||
setImportCsr(record);
|
||||
}}
|
||||
>
|
||||
Import
|
||||
</Button>
|
||||
</Tooltip>
|
||||
)}
|
||||
<Tooltip title="Duplicate (pre-fill a new CSR)">
|
||||
<Button
|
||||
type="text"
|
||||
size="small"
|
||||
icon={<CopyOutlined />}
|
||||
onClick={() => handleDuplicate(record)}
|
||||
/>
|
||||
</Tooltip>
|
||||
<Popconfirm
|
||||
title="Delete this CSR?"
|
||||
description={
|
||||
record.status === 'pending'
|
||||
? 'The private key will be permanently destroyed — any certificate later signed from this CSR becomes unusable.'
|
||||
: 'Only the CSR history entry is removed — the imported certificate is not affected.'
|
||||
}
|
||||
onConfirm={() => handleDelete(record)}
|
||||
okText="Delete"
|
||||
okType="danger"
|
||||
cancelText="Cancel"
|
||||
>
|
||||
<Tooltip title="Delete CSR">
|
||||
<Button type="text" size="small" danger icon={<DeleteOutlined />} />
|
||||
</Tooltip>
|
||||
</Popconfirm>
|
||||
</Space>
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
const pendingCount = csrs.filter((c) => c.status === 'pending').length;
|
||||
|
||||
return (
|
||||
<div>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Certificate Signing Requests for external CAs"
|
||||
description="Generate a private key and CSR here, submit the CSR to your Certificate Authority, then import the signed certificate. The private key never leaves the server; the imported certificate goes through the normal Apply Management deployment flow."
|
||||
/>
|
||||
<Row gutter={16} style={{ marginBottom: 16 }}>
|
||||
<Col flex="auto">
|
||||
{pendingCount > 0 && (
|
||||
<Text type="secondary">
|
||||
{pendingCount} CSR{pendingCount > 1 ? 's' : ''} awaiting a signed
|
||||
certificate
|
||||
</Text>
|
||||
)}
|
||||
</Col>
|
||||
<Col>
|
||||
<Space>
|
||||
<Button icon={<ReloadOutlined />} onClick={fetchCsrs} loading={loading}>
|
||||
Refresh
|
||||
</Button>
|
||||
<Button
|
||||
type="primary"
|
||||
icon={<PlusOutlined />}
|
||||
onClick={() => {
|
||||
createForm.resetFields();
|
||||
setCreateModalOpen(true);
|
||||
}}
|
||||
>
|
||||
Create CSR
|
||||
</Button>
|
||||
</Space>
|
||||
</Col>
|
||||
</Row>
|
||||
|
||||
<Card>
|
||||
<Table
|
||||
columns={columns}
|
||||
dataSource={csrs}
|
||||
rowKey="id"
|
||||
loading={loading}
|
||||
pagination={{
|
||||
showSizeChanger: true,
|
||||
showQuickJumper: true,
|
||||
showTotal: (total) => `Total ${total} CSRs`,
|
||||
}}
|
||||
/>
|
||||
</Card>
|
||||
|
||||
{/* Create CSR Modal */}
|
||||
<Modal
|
||||
title="Create Certificate Signing Request"
|
||||
open={createModalOpen}
|
||||
onCancel={() => {
|
||||
setCreateModalOpen(false);
|
||||
createForm.resetFields();
|
||||
}}
|
||||
footer={null}
|
||||
width={700}
|
||||
forceRender
|
||||
>
|
||||
<Form form={createForm} layout="vertical" onFinish={handleCreate}>
|
||||
<Form.Item
|
||||
name="name"
|
||||
label="Name"
|
||||
rules={[
|
||||
{ required: true, message: 'Please enter a CSR name' },
|
||||
{
|
||||
pattern: /^[a-zA-Z0-9_.-]+$/,
|
||||
message:
|
||||
'Only letters, digits, underscore, hyphen and dot are allowed',
|
||||
},
|
||||
{ max: 100, message: 'Name must be 100 characters or fewer' },
|
||||
{
|
||||
validator: (_, value) => {
|
||||
if (!value) return Promise.resolve();
|
||||
if (value.includes('..')) {
|
||||
return Promise.reject(new Error('Name must not contain ".."'));
|
||||
}
|
||||
if (value.startsWith('.') || value.startsWith('-')) {
|
||||
return Promise.reject(
|
||||
new Error('Name must not start with "." or "-"')
|
||||
);
|
||||
}
|
||||
return Promise.resolve();
|
||||
},
|
||||
},
|
||||
]}
|
||||
extra="Becomes the certificate name and file path at import: /etc/ssl/haproxy/{name}.pem"
|
||||
>
|
||||
<Input placeholder="e.g. www-example-com" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="common_name"
|
||||
label="Common Name (CN)"
|
||||
rules={[
|
||||
{ required: true, message: 'Please enter the Common Name' },
|
||||
antdDomainRule,
|
||||
{ max: 64, message: 'Common Name must be 64 characters or fewer' },
|
||||
]}
|
||||
extra="The primary domain, e.g. www.example.com or *.example.com"
|
||||
>
|
||||
<Input placeholder="www.example.com" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="sans"
|
||||
label="Subject Alternative Names (SANs)"
|
||||
rules={[antdDomainsListRule]}
|
||||
extra="Additional DNS names — the Common Name is included automatically"
|
||||
>
|
||||
<Select
|
||||
mode="tags"
|
||||
tokenSeparators={[',', ' ']}
|
||||
placeholder="api.example.com, cdn.example.com"
|
||||
open={false}
|
||||
suffixIcon={null}
|
||||
/>
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="key_algorithm"
|
||||
label="Key Algorithm"
|
||||
initialValue="rsa-2048"
|
||||
rules={[{ required: true }]}
|
||||
>
|
||||
<Select options={KEY_ALGORITHM_OPTIONS} />
|
||||
</Form.Item>
|
||||
|
||||
<Collapse
|
||||
style={{ marginBottom: 16 }}
|
||||
items={[
|
||||
{
|
||||
key: 'subject',
|
||||
label: 'Subject details (optional)',
|
||||
children: (
|
||||
<>
|
||||
<Row gutter={12}>
|
||||
<Col span={12}>
|
||||
<Form.Item
|
||||
name="organization"
|
||||
label="Organization (O)"
|
||||
rules={[{ max: 64 }]}
|
||||
>
|
||||
<Input placeholder="Example Corp" />
|
||||
</Form.Item>
|
||||
</Col>
|
||||
<Col span={12}>
|
||||
<Form.Item
|
||||
name="organizational_unit"
|
||||
label="Organizational Unit (OU)"
|
||||
rules={[{ max: 64 }]}
|
||||
>
|
||||
<Input placeholder="IT Department" />
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
<Row gutter={12}>
|
||||
<Col span={8}>
|
||||
<Form.Item name="locality" label="Locality (L)" rules={[{ max: 64 }]}>
|
||||
<Input placeholder="Istanbul" />
|
||||
</Form.Item>
|
||||
</Col>
|
||||
<Col span={8}>
|
||||
<Form.Item name="state" label="State / Province (ST)" rules={[{ max: 64 }]}>
|
||||
<Input placeholder="Marmara" />
|
||||
</Form.Item>
|
||||
</Col>
|
||||
<Col span={8}>
|
||||
<Form.Item
|
||||
name="country"
|
||||
label="Country (C)"
|
||||
rules={[
|
||||
{
|
||||
pattern: /^[A-Za-z]{2}$/,
|
||||
message: 'Exactly 2 letters (e.g. TR, US)',
|
||||
},
|
||||
]}
|
||||
>
|
||||
<Input placeholder="TR" maxLength={2} />
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
<Form.Item
|
||||
name="email"
|
||||
label="Email"
|
||||
rules={[{ type: 'email', message: 'Invalid email address' }]}
|
||||
>
|
||||
<Input placeholder="ops@example.com" />
|
||||
</Form.Item>
|
||||
</>
|
||||
),
|
||||
},
|
||||
]}
|
||||
/>
|
||||
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="🔐 The private key is generated and stored server-side"
|
||||
description="You will only receive the CSR to hand to your CA. After the signed certificate is imported, the key is available on the certificate itself."
|
||||
/>
|
||||
|
||||
<Form.Item style={{ textAlign: 'right', marginBottom: 0 }}>
|
||||
<Space>
|
||||
<Button
|
||||
onClick={() => {
|
||||
setCreateModalOpen(false);
|
||||
createForm.resetFields();
|
||||
}}
|
||||
>
|
||||
Cancel
|
||||
</Button>
|
||||
<Button type="primary" htmlType="submit" loading={creating}>
|
||||
Generate CSR
|
||||
</Button>
|
||||
</Space>
|
||||
</Form.Item>
|
||||
</Form>
|
||||
</Modal>
|
||||
|
||||
{/* View CSR Modal */}
|
||||
<Modal
|
||||
title={
|
||||
<Space>
|
||||
<FileProtectOutlined />
|
||||
{viewCsr ? `CSR: ${viewCsr.name}` : 'CSR'}
|
||||
</Space>
|
||||
}
|
||||
open={!!viewCsr}
|
||||
onCancel={() => setViewCsr(null)}
|
||||
width={760}
|
||||
footer={[
|
||||
<Button key="close" onClick={() => setViewCsr(null)}>
|
||||
Close
|
||||
</Button>,
|
||||
]}
|
||||
>
|
||||
{viewCsr && (
|
||||
<div>
|
||||
<Descriptions size="small" column={2} bordered style={{ marginBottom: 12 }}>
|
||||
<Descriptions.Item label="Common Name" span={2}>
|
||||
{viewCsr.common_name}
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="Key">
|
||||
{KEY_ALGORITHM_LABELS[viewCsr.key_algorithm] || viewCsr.key_algorithm}
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="Status">
|
||||
{viewCsr.status === 'completed' ? (
|
||||
<Badge status="success" text="Imported" />
|
||||
) : (
|
||||
<Badge status="processing" text="Awaiting certificate" />
|
||||
)}
|
||||
</Descriptions.Item>
|
||||
{viewCsr.subject && Object.keys(viewCsr.subject).length > 0 && (
|
||||
<Descriptions.Item label="Subject" span={2}>
|
||||
{Object.entries(viewCsr.subject)
|
||||
.map(([k, v]) => `${k}=${v}`)
|
||||
.join(', ')}
|
||||
</Descriptions.Item>
|
||||
)}
|
||||
</Descriptions>
|
||||
|
||||
{Array.isArray(viewCsr.sans) && viewCsr.sans.length > 0 && (
|
||||
<div style={{ marginBottom: 12 }}>
|
||||
<Text strong>SANs: </Text>
|
||||
<Space size={4} wrap>
|
||||
{viewCsr.sans.map((d) => (
|
||||
<Tag key={d}>{d}</Tag>
|
||||
))}
|
||||
</Space>
|
||||
</div>
|
||||
)}
|
||||
|
||||
<Row justify="space-between" align="middle" style={{ marginBottom: 8 }}>
|
||||
<Col>
|
||||
<Text strong>CSR (PEM)</Text>
|
||||
</Col>
|
||||
<Col>
|
||||
<Space>
|
||||
<Button
|
||||
size="small"
|
||||
icon={<CopyOutlined />}
|
||||
onClick={() => copyCsrPem(viewCsr)}
|
||||
>
|
||||
Copy
|
||||
</Button>
|
||||
<Button
|
||||
size="small"
|
||||
type="primary"
|
||||
icon={<DownloadOutlined />}
|
||||
onClick={() => downloadCsrPem(viewCsr)}
|
||||
>
|
||||
Download .csr
|
||||
</Button>
|
||||
</Space>
|
||||
</Col>
|
||||
</Row>
|
||||
<TextArea
|
||||
value={viewCsr.csr_pem}
|
||||
rows={12}
|
||||
readOnly
|
||||
style={{ fontFamily: 'monospace', fontSize: 12 }}
|
||||
/>
|
||||
{viewCsr.status !== 'completed' && (
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginTop: 12 }}
|
||||
message="Next step"
|
||||
description="Submit this CSR to your Certificate Authority. When you receive the signed certificate, come back and click Import on this CSR."
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</Modal>
|
||||
|
||||
{/* Import Signed Certificate Modal */}
|
||||
<Modal
|
||||
title={
|
||||
<Space>
|
||||
<ImportOutlined />
|
||||
{importCsr ? `Import Signed Certificate — ${importCsr.name}` : 'Import'}
|
||||
</Space>
|
||||
}
|
||||
open={!!importCsr}
|
||||
onCancel={() => {
|
||||
setImportCsr(null);
|
||||
importForm.resetFields();
|
||||
}}
|
||||
footer={null}
|
||||
width={800}
|
||||
>
|
||||
{importCsr && (
|
||||
<div>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message={`CSR: ${importCsr.name} (CN: ${importCsr.common_name})`}
|
||||
description="Paste the certificate your CA issued for this CSR. It will be verified against the stored private key before anything is saved."
|
||||
/>
|
||||
<Form
|
||||
form={importForm}
|
||||
layout="vertical"
|
||||
onFinish={handleImport}
|
||||
initialValues={{ ssl_type: 'cluster', usage_type: 'frontend' }}
|
||||
>
|
||||
<Row gutter={12}>
|
||||
<Col span={12}>
|
||||
<Form.Item
|
||||
name="usage_type"
|
||||
label="Usage Type"
|
||||
rules={[{ required: true }]}
|
||||
>
|
||||
<Select
|
||||
options={[
|
||||
{ value: 'frontend', label: 'Frontend SSL (HTTPS termination)' },
|
||||
{ value: 'server', label: 'Server SSL (backend verification)' },
|
||||
]}
|
||||
/>
|
||||
</Form.Item>
|
||||
</Col>
|
||||
<Col span={12}>
|
||||
<Form.Item name="ssl_type" label="Scope" rules={[{ required: true }]}>
|
||||
<Select
|
||||
options={[
|
||||
{ value: 'global', label: 'Global (all clusters)' },
|
||||
{ value: 'cluster', label: 'Cluster-specific' },
|
||||
]}
|
||||
/>
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
|
||||
<Form.Item
|
||||
noStyle
|
||||
shouldUpdate={(prev, cur) => prev.ssl_type !== cur.ssl_type}
|
||||
>
|
||||
{({ getFieldValue }) =>
|
||||
getFieldValue('ssl_type') === 'cluster' && (
|
||||
<Form.Item
|
||||
name="cluster_ids"
|
||||
label="Clusters"
|
||||
rules={[
|
||||
{ required: true, message: 'Select at least one cluster' },
|
||||
]}
|
||||
>
|
||||
<Select
|
||||
mode="multiple"
|
||||
placeholder="Select cluster(s)"
|
||||
options={(clusters || []).map((c) => ({
|
||||
value: c.id,
|
||||
label: c.name,
|
||||
}))}
|
||||
/>
|
||||
</Form.Item>
|
||||
)
|
||||
}
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="certificate_content"
|
||||
label="Signed Certificate (PEM)"
|
||||
rules={[
|
||||
{ required: true, message: 'Please paste the signed certificate' },
|
||||
{
|
||||
validator: (_, value) => {
|
||||
if (!value) return Promise.resolve();
|
||||
if (
|
||||
value.includes('-----BEGIN CERTIFICATE-----') &&
|
||||
value.includes('-----END CERTIFICATE-----')
|
||||
) {
|
||||
return Promise.resolve();
|
||||
}
|
||||
return Promise.reject(
|
||||
new Error('Certificate must be in PEM format')
|
||||
);
|
||||
},
|
||||
},
|
||||
]}
|
||||
>
|
||||
<TextArea
|
||||
rows={8}
|
||||
placeholder={'-----BEGIN CERTIFICATE-----\n...\n-----END CERTIFICATE-----'}
|
||||
style={{ fontFamily: 'monospace', fontSize: 12 }}
|
||||
/>
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="name_override"
|
||||
label="Certificate name override (optional)"
|
||||
rules={[
|
||||
{
|
||||
pattern: /^[a-zA-Z0-9_.-]+$/,
|
||||
message:
|
||||
'Only letters, digits, underscore, hyphen and dot are allowed',
|
||||
},
|
||||
{ max: 100, message: 'Name must be 100 characters or fewer' },
|
||||
]}
|
||||
extra={`Leave empty to use the CSR name ('${importCsr.name}'). Use this only if that name is now taken by another certificate.`}
|
||||
>
|
||||
<Input placeholder={importCsr.name} />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
name="chain_content"
|
||||
label="Certificate Chain (PEM, optional)"
|
||||
rules={[
|
||||
{
|
||||
validator: (_, value) => {
|
||||
if (!value || !value.trim()) return Promise.resolve();
|
||||
if (
|
||||
value.includes('-----BEGIN CERTIFICATE-----') &&
|
||||
value.includes('-----END CERTIFICATE-----')
|
||||
) {
|
||||
return Promise.resolve();
|
||||
}
|
||||
return Promise.reject(
|
||||
new Error('Certificate chain must be in PEM format')
|
||||
);
|
||||
},
|
||||
},
|
||||
]}
|
||||
>
|
||||
<TextArea
|
||||
rows={4}
|
||||
placeholder={'-----BEGIN CERTIFICATE-----\n(intermediate CA)\n-----END CERTIFICATE-----'}
|
||||
style={{ fontFamily: 'monospace', fontSize: 12 }}
|
||||
/>
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item style={{ textAlign: 'right', marginBottom: 0 }}>
|
||||
<Space>
|
||||
<Button
|
||||
onClick={() => {
|
||||
setImportCsr(null);
|
||||
importForm.resetFields();
|
||||
}}
|
||||
>
|
||||
Cancel
|
||||
</Button>
|
||||
<Button type="primary" htmlType="submit" loading={importing}>
|
||||
Import Certificate
|
||||
</Button>
|
||||
</Space>
|
||||
</Form.Item>
|
||||
</Form>
|
||||
</div>
|
||||
)}
|
||||
</Modal>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
export default CSRManagement;
|
||||
@@ -156,6 +156,7 @@ const ClusterManagement = () => {
|
||||
stats_socket_path: cluster.stats_socket_path || '/run/haproxy/admin.sock',
|
||||
haproxy_config_path: cluster.haproxy_config_path || '/etc/haproxy/haproxy.cfg',
|
||||
haproxy_bin_path: cluster.haproxy_bin_path,
|
||||
keepalived_config_path: cluster.keepalived_config_path || '/etc/keepalived/keepalived.conf',
|
||||
agent_pool_id: cluster.pool_id || undefined,
|
||||
haproxy_user: cluster.haproxy_user || '',
|
||||
haproxy_group: cluster.haproxy_group || '',
|
||||
@@ -197,6 +198,7 @@ const ClusterManagement = () => {
|
||||
stats_socket_path: values.stats_socket_path,
|
||||
haproxy_config_path: values.haproxy_config_path,
|
||||
haproxy_bin_path: values.haproxy_bin_path,
|
||||
keepalived_config_path: values.keepalived_config_path,
|
||||
pool_id: values.agent_pool_id || null,
|
||||
haproxy_user: values.haproxy_user || null,
|
||||
haproxy_group: values.haproxy_group || null,
|
||||
@@ -748,6 +750,14 @@ const ClusterManagement = () => {
|
||||
<Input placeholder="/usr/sbin/haproxy" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
label="Keepalived Config Path"
|
||||
name="keepalived_config_path"
|
||||
tooltip="Where the agent writes keepalived.conf for this cluster's HA/VIPs. The default (/etc/keepalived/keepalived.conf) is what the keepalived service loads on every distro — leave it unless you run a non-standard install AND have configured the keepalived service/unit to load this exact path."
|
||||
>
|
||||
<Input placeholder="/etc/keepalived/keepalived.conf" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
label="ACME Challenge Routing"
|
||||
name="acme_enabled"
|
||||
|
||||
@@ -642,7 +642,11 @@ const FrontendManagement = () => {
|
||||
// Explicitly set options field to handle null/undefined case (NEW field)
|
||||
options: frontend.options || '',
|
||||
// BUGFIX: Explicitly set tcp_request_rules field to handle null/undefined case
|
||||
tcp_request_rules: frontend.tcp_request_rules || ''
|
||||
tcp_request_rules: frontend.tcp_request_rules || '',
|
||||
// Issue #38: SPOE filters + frontend log-format (null → '' so the
|
||||
// TextAreas populate on edit and round-trip on save, preventing null-wipe)
|
||||
log_format: frontend.log_format || '',
|
||||
filters: frontend.filters || ''
|
||||
});
|
||||
|
||||
// Update SSL field visibility after setting values
|
||||
@@ -862,31 +866,13 @@ const FrontendManagement = () => {
|
||||
return;
|
||||
}
|
||||
|
||||
// Phase K Phase D follow-up (Bulgu #12 round 3) — hard-gate any
|
||||
// ACL / use_backend / redirect rule that carries the unsupported
|
||||
// HAProxy `-f <file>` pattern-file flag. The Pydantic validator
|
||||
// on the backend (`models/frontend.py::validate_acl_rules`)
|
||||
// rejects the same shape; blocking here surfaces the error
|
||||
// immediately at the manual frontend form and matches the wizard
|
||||
// gate so operators see consistent behaviour between the two
|
||||
// entry points.
|
||||
const FILE_FLAG_RE = /(?:^|\s)-f(?:\s|$)/;
|
||||
const aclRulesAll = [
|
||||
...(aclBuilderData.aclRules || []),
|
||||
...(aclBuilderData.useBackendRules || []),
|
||||
...(aclBuilderData.redirectRules || []).map(
|
||||
(r) => (typeof r === 'string' ? r : ''),
|
||||
),
|
||||
];
|
||||
if (aclRulesAll.some((r) => typeof r === 'string' && FILE_FLAG_RE.test(r))) {
|
||||
message.error(
|
||||
'One or more ACL / routing / redirect rules use the unsupported HAProxy ' +
|
||||
'`-f <file>` pattern-file flag. HAProxy OpenManager does not provision ' +
|
||||
'pattern files onto the HAProxy node filesystem, so the reference would ' +
|
||||
'fail at reload time. Remove the `-f` flag and use inline values instead.'
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Issue #38 follow-up — the Bulgu #12 client-side hard gate for
|
||||
// the ACL `-f <file>` pattern-file flag was removed together with
|
||||
// the server-side Pydantic rejects: pattern files are operator-
|
||||
// managed host files (same policy as SPOE filter configs since
|
||||
// v1.8.8) and the agent's pre-reload `haproxy -c` makes a missing
|
||||
// file fail safely. The server response now carries a non-blocking
|
||||
// warning listing the referenced files (rendered below).
|
||||
|
||||
// Phase K Phase D follow-up (Bulgu #13) — gate for
|
||||
// self-contradictory routing / redirect conditions (`X !X`).
|
||||
@@ -1178,8 +1164,31 @@ const FrontendManagement = () => {
|
||||
} else {
|
||||
message.success('Frontend created successfully');
|
||||
}
|
||||
|
||||
// Issue #38 follow-up — surface server-emitted warnings on
|
||||
// CREATE too (e.g. the `-f <file>` pattern-file advisory).
|
||||
// Mirrors the update-branch rendering above.
|
||||
const createWarnings = Array.isArray(response.data?.warnings)
|
||||
? response.data.warnings
|
||||
: [];
|
||||
if (createWarnings.length > 0) {
|
||||
message.warning(
|
||||
<div>
|
||||
<div><strong>Frontend saved, but the server flagged {createWarnings.length} rule warning(s):</strong></div>
|
||||
<div style={{ marginTop: 6, fontSize: '12px', fontFamily: 'monospace' }}>
|
||||
{createWarnings.slice(0, 5).map((w, i) => (
|
||||
<div key={i}>• {w.length > 240 ? `${w.slice(0, 237)}...` : w}</div>
|
||||
))}
|
||||
{createWarnings.length > 5 && (
|
||||
<div>(+{createWarnings.length - 5} more)</div>
|
||||
)}
|
||||
</div>
|
||||
</div>,
|
||||
10,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
setModalVisible(false);
|
||||
fetchFrontends();
|
||||
fetchSSLCertificates(); // Refresh SSL certificates after frontend update
|
||||
@@ -2250,6 +2259,40 @@ tcp-request connection reject if { src -f /etc/haproxy/blacklist.lst }`}
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
|
||||
{/* Issue #38: SPOE filters + frontend log-format */}
|
||||
<Row gutter={16}>
|
||||
<Col span={24}>
|
||||
<Form.Item
|
||||
name="filters"
|
||||
label="Filters (SPOE / WAF)"
|
||||
extra="HAProxy filter directives (one per line). Emitted before send-spoe-group rules."
|
||||
tooltip="e.g. Coraza WAF via SPOE. The referenced engine config file and its SPOA backend must exist on the HAProxy host."
|
||||
>
|
||||
<TextArea
|
||||
rows={3}
|
||||
placeholder={`Examples:
|
||||
filter spoe engine coraza config /etc/haproxy/coraza.cfg`}
|
||||
/>
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
|
||||
<Row gutter={16}>
|
||||
<Col span={24}>
|
||||
<Form.Item
|
||||
name="log_format"
|
||||
label="Custom Log Format"
|
||||
extra="HAProxy log-format / log-format-sd directive (kept verbatim)"
|
||||
tooltip="Overrides option httplog/tcplog. Use the full directive including the quoted format string."
|
||||
>
|
||||
<TextArea
|
||||
rows={3}
|
||||
placeholder={'log-format "%ci:%cp [%t] %ft %b/%s %ST %B %{+Q}r"'}
|
||||
/>
|
||||
</Form.Item>
|
||||
</Col>
|
||||
</Row>
|
||||
</Panel>
|
||||
</Collapse>
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ import {
|
||||
PlayCircleOutlined, EditOutlined,
|
||||
CloudServerOutlined, CheckCircleOutlined, SyncOutlined,
|
||||
ExclamationCircleOutlined, CloseCircleOutlined, ClockCircleOutlined,
|
||||
ThunderboltOutlined
|
||||
ThunderboltOutlined, FileProtectOutlined
|
||||
} from '@ant-design/icons';
|
||||
import axios from 'axios';
|
||||
import { useSearchParams } from 'react-router-dom';
|
||||
@@ -22,6 +22,7 @@ import { useProgress } from '../contexts/ProgressContext';
|
||||
import { formatEntityForSync } from '../utils/agentSync';
|
||||
import { extractApiError } from '../utils/apiError';
|
||||
import ACMEAutomation from './ACMEAutomation';
|
||||
import CSRManagement from './CSRManagement';
|
||||
|
||||
const { Title, Text } = Typography;
|
||||
const { TextArea } = Input;
|
||||
@@ -643,12 +644,21 @@ const SSLManagement = () => {
|
||||
title: 'Source',
|
||||
dataIndex: 'source',
|
||||
key: 'source',
|
||||
render: (source) => (
|
||||
<Tag color={source === 'letsencrypt' ? 'green' : 'default'}
|
||||
icon={source === 'letsencrypt' ? <SafetyCertificateOutlined /> : null}>
|
||||
{source === 'letsencrypt' ? 'Auto (ACME)' : 'Manual'}
|
||||
</Tag>
|
||||
),
|
||||
render: (source) => {
|
||||
if (source === 'csr') {
|
||||
return (
|
||||
<Tag color="blue" icon={<FileProtectOutlined />}>
|
||||
CSR
|
||||
</Tag>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<Tag color={source === 'letsencrypt' ? 'green' : 'default'}
|
||||
icon={source === 'letsencrypt' ? <SafetyCertificateOutlined /> : null}>
|
||||
{source === 'letsencrypt' ? 'Auto (ACME)' : 'Manual'}
|
||||
</Tag>
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Sync Status',
|
||||
@@ -1020,6 +1030,11 @@ const SSLManagement = () => {
|
||||
label: <span><ThunderboltOutlined /> ACME Automation</span>,
|
||||
children: <ACMEAutomation />,
|
||||
},
|
||||
{
|
||||
key: 'csr',
|
||||
label: <span><FileProtectOutlined /> CSR</span>,
|
||||
children: <CSRManagement onCertificateImported={fetchCertificates} />,
|
||||
},
|
||||
]}
|
||||
/>
|
||||
|
||||
|
||||
@@ -212,6 +212,7 @@ const Settings = () => {
|
||||
eab_kid: '',
|
||||
eab_hmac_key: '',
|
||||
challenge_backend_url: '',
|
||||
dns01_enabled: false,
|
||||
}}
|
||||
>
|
||||
<Form.Item name="provider" label="ACME Provider">
|
||||
@@ -308,6 +309,36 @@ const Settings = () => {
|
||||
}]}
|
||||
/>
|
||||
|
||||
{/* Issue #35: DNS-01 challenge support (global kill-switch). Per-account DNS provider
|
||||
credentials are configured on each ACME account in ACME Automation. */}
|
||||
<Collapse
|
||||
ghost
|
||||
style={{ marginTop: 16 }}
|
||||
items={[{
|
||||
key: 'dns01',
|
||||
label: 'DNS-01 Challenge (Advanced)',
|
||||
children: (
|
||||
<>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="DNS-01 validates certificates via a DNS TXT record instead of HTTP on port 80."
|
||||
description="Use it for internal/isolated clusters with no public inbound port 80, or for wildcard certificates. When enabled, choose DNS-01 and a DNS provider per ACME account in ACME Automation. Leaving this off keeps the default HTTP-01 behavior unchanged."
|
||||
/>
|
||||
<Form.Item
|
||||
name="dns01_enabled"
|
||||
label="Enable DNS-01 Challenge"
|
||||
valuePropName="checked"
|
||||
tooltip="Master switch. While off, DNS-01 options are hidden and no DNS-01 orders can be created."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
</>
|
||||
),
|
||||
}]}
|
||||
/>
|
||||
|
||||
<div style={{ marginTop: 24, display: 'flex', gap: 12 }}>
|
||||
<Button type="primary" htmlType="submit" loading={acmeSaving}>
|
||||
Save ACME Settings
|
||||
|
||||
@@ -3177,36 +3177,15 @@ const SiteWizard = () => {
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Phase K Phase D follow-up (Bulgu #12 round 3) —
|
||||
// hard-gate the Step 2 → Step 3 advance on any ACL
|
||||
// rule that carries the unsupported `-f <file>`
|
||||
// pattern-file flag. The Pydantic validator rejects
|
||||
// the same shape at submit, but blocking the Next
|
||||
// button here surfaces the error immediately at
|
||||
// its source step (the ACL builder is right above)
|
||||
// instead of bouncing the operator from Step 4's
|
||||
// dry-run card back to Step 2 with a less-specific
|
||||
// jumpback button. The ACLRuleBuilder ALSO renders
|
||||
// a section-level red Alert when this state is
|
||||
// active so the operator already sees what to fix.
|
||||
const FILE_FLAG_RE = /(?:^|\s)-f(?:\s|$)/;
|
||||
const aclRulesAll = [
|
||||
...(aclBuilderData.aclRules || []),
|
||||
...(aclBuilderData.useBackendRules || []),
|
||||
...(aclBuilderData.redirectRules || []).map(
|
||||
(r) => (typeof r === 'string' ? r : ''),
|
||||
),
|
||||
];
|
||||
if (aclRulesAll.some((r) => typeof r === 'string' && FILE_FLAG_RE.test(r))) {
|
||||
message.error(
|
||||
'One or more rules use the unsupported HAProxy `-f <file>` ' +
|
||||
'pattern-file flag. HAProxy OpenManager does not provision ' +
|
||||
'pattern files onto the HAProxy node filesystem, so the ' +
|
||||
'reference would fail at reload time. Remove the `-f` flag ' +
|
||||
'and use inline values instead before continuing.'
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Issue #38 follow-up — the Bulgu #12 Step 2 → 3
|
||||
// hard gate for the ACL `-f <file>` pattern-file
|
||||
// flag was removed together with the server-side
|
||||
// Pydantic rejects: pattern files are operator-
|
||||
// managed host files (same policy as SPOE filter
|
||||
// configs since v1.8.8) and the agent's pre-reload
|
||||
// `haproxy -c` makes a missing file fail safely.
|
||||
// The ACLRuleBuilder renders an informational note
|
||||
// on `-f` rules instead of a blocking error.
|
||||
// Phase K Phase D follow-up (Bulgu #13) — block
|
||||
// advance when any routing / redirect rule has a
|
||||
// self-contradictory condition (`acl1 !acl1`).
|
||||
|
||||
@@ -146,6 +146,17 @@ const PERMISSION_TREE = [
|
||||
{ title: 'View Cluster Config', key: 'clusters.config' }
|
||||
]
|
||||
},
|
||||
{
|
||||
title: '🛰️ HA / VIP Management',
|
||||
key: 'vip',
|
||||
children: [
|
||||
{ title: 'View VIPs', key: 'vip.read' },
|
||||
{ title: 'Create VIP', key: 'vip.create' },
|
||||
{ title: 'Edit VIP', key: 'vip.update' },
|
||||
{ title: 'Delete VIP', key: 'vip.delete' },
|
||||
{ title: 'Apply / Deploy VIP', key: 'vip.apply' }
|
||||
]
|
||||
},
|
||||
{
|
||||
title: '📝 Configuration',
|
||||
key: 'config',
|
||||
|
||||
@@ -0,0 +1,583 @@
|
||||
import React, { useState, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Table, Button, Space, Modal, Form, Input, InputNumber, Select, Tag, message,
|
||||
Switch, Typography, Card, Alert, Tooltip, Spin
|
||||
} from 'antd';
|
||||
import {
|
||||
PlusOutlined, EditOutlined, DeleteOutlined, ReloadOutlined, WarningOutlined,
|
||||
PlayCircleOutlined, SyncOutlined, CrownOutlined, ClockCircleOutlined, FileSearchOutlined,
|
||||
InfoCircleOutlined
|
||||
} from '@ant-design/icons';
|
||||
import { useCluster } from '../contexts/ClusterContext';
|
||||
import { extractApiError } from '../utils/apiError';
|
||||
|
||||
const { Option } = Select;
|
||||
const { Title, Text } = Typography;
|
||||
|
||||
// Interfaces that are never sensible VIP carriers — hidden from the dropdown.
|
||||
const IFACE_HIDE = /^(lo|docker|veth|br-|cni|flannel|kube|virbr)/i;
|
||||
|
||||
const authHeaders = () => ({
|
||||
'Content-Type': 'application/json',
|
||||
'Authorization': `Bearer ${localStorage.getItem('authToken') || ''}`,
|
||||
});
|
||||
|
||||
// Convergence-aware status (issue #27 follow-up): like other entities, a VIP only reads
|
||||
// "live" once its member agents have actually deployed & acked — never the instant Apply
|
||||
// is clicked. Backend returns deploy_status; we fall back to last_config_status.
|
||||
const DEPLOY_STATUS = {
|
||||
PENDING: { color: 'orange', label: 'PENDING', tip: 'Staged change — review and Apply (or Reject) it from the Apply Management page.' },
|
||||
PENDING_DELETE: { color: 'volcano', label: 'PENDING DELETE', tip: 'Deletion staged for approval — the VIP keeps running untouched until you APPROVE it in Apply Management. Reject to keep it. The node is not changed until approval.' },
|
||||
DELETING: { color: 'processing', label: 'DELETING', tip: 'Deletion approved — member node(s) are stopping keepalived and releasing the VIP. Disappears once every node has torn down.' },
|
||||
DELETED: { color: 'default', label: 'DELETED', tip: 'All member nodes have torn keepalived down.' },
|
||||
SYNCING: { color: 'processing', label: 'SYNCING', tip: 'Applied — an online member node is installing/configuring keepalived and will acknowledge on its next poll (~2–3 min).' },
|
||||
AWAITING: { color: 'gold', label: 'AWAITING AGENT', tip: 'Applied, but the member node(s) that still need it are OFFLINE, so nothing can deploy yet. Bring the node\'s agent online — it converges on its next poll. (Not a hang.)' },
|
||||
ACTIVE: { color: 'green', label: 'ACTIVE', tip: 'Applied and every member node has deployed keepalived and acknowledged the current config.' },
|
||||
ERROR: { color: 'red', label: 'ERROR', tip: 'A member node failed to deploy keepalived — see Members / Live state for the node, then check that agent.' },
|
||||
ATTENTION: { color: 'gold', label: 'ATTENTION',tip: 'A member already runs a hand-managed keepalived; the agent left it untouched (externally managed). Resolve it on that node or remove it from the VIP.' },
|
||||
APPLIED: { color: 'green', label: 'APPLIED', tip: 'Applied.' },
|
||||
};
|
||||
|
||||
// agents.capabilities / network_interfaces come from the API as JSONB → a JSON string
|
||||
// (asyncpg has no jsonb codec). Mirror AgentManagement.js and JSON.parse when needed.
|
||||
const parseArr = (v) => {
|
||||
if (Array.isArray(v)) return v;
|
||||
if (typeof v === 'string') { try { const p = JSON.parse(v); return Array.isArray(p) ? p : []; } catch { return []; } }
|
||||
return [];
|
||||
};
|
||||
|
||||
// This component uses raw fetch(), but extractApiError expects an axios-shaped error
|
||||
// (err.response.data). Read the fetch Response body and reuse the envelope-aware extractor
|
||||
// so backend messages — e.g. the 409 "node already in VIP X" — actually reach the user.
|
||||
const fetchApiError = async (res, fallback) => {
|
||||
try { const data = await res.json(); return extractApiError({ response: { data } }, fallback); }
|
||||
catch { return fallback; }
|
||||
};
|
||||
|
||||
const VIPManagement = () => {
|
||||
const { clusters } = useCluster();
|
||||
const [vips, setVips] = useState([]);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [modalVisible, setModalVisible] = useState(false);
|
||||
const [editing, setEditing] = useState(null);
|
||||
const [selectedPoolId, setSelectedPoolId] = useState(null);
|
||||
// One row per agent in the selected pool — the user toggles which participate.
|
||||
const [memberRows, setMemberRows] = useState([]);
|
||||
const [form] = Form.useForm();
|
||||
// Delete confirmation (with opt-in package uninstall) + diagnostics modal state.
|
||||
const [deleteTarget, setDeleteTarget] = useState(null);
|
||||
const [showL2Note, setShowL2Note] = useState(false);
|
||||
const [diagVip, setDiagVip] = useState(null);
|
||||
const [diagData, setDiagData] = useState(null);
|
||||
const [diagLoading, setDiagLoading] = useState(false);
|
||||
|
||||
// Distinct pools derived from the cluster list (cluster -> pool_id).
|
||||
const pools = React.useMemo(() => {
|
||||
const seen = new Map();
|
||||
(clusters || []).forEach((c) => {
|
||||
if (c.pool_id && !seen.has(c.pool_id)) seen.set(c.pool_id, c.name || `pool ${c.pool_id}`);
|
||||
});
|
||||
return Array.from(seen, ([id, name]) => ({ id, name }));
|
||||
}, [clusters]);
|
||||
|
||||
const fetchVips = useCallback(async () => {
|
||||
setLoading(true);
|
||||
try {
|
||||
const res = await fetch('/api/vip', { headers: authHeaders() });
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
setVips(data.vips || []);
|
||||
} else if (res.status === 403) {
|
||||
message.warning('You do not have permission to view VIPs (vip.read).');
|
||||
setVips([]);
|
||||
}
|
||||
} catch (e) {
|
||||
console.error('fetchVips failed', e);
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
fetchVips();
|
||||
const t = setInterval(fetchVips, 30000); // live MASTER/BACKUP via existing detection pipeline
|
||||
return () => clearInterval(t);
|
||||
}, [fetchVips]);
|
||||
|
||||
// Build the participating-nodes table from the pool's EXISTING agents (installed via the
|
||||
// standard Agent Management process). On edit, pre-select the VIP's current members.
|
||||
const buildMemberRows = (agents, existing) => {
|
||||
const ex = {};
|
||||
(existing || []).forEach((m) => { ex[m.agent_id] = m; });
|
||||
return (agents || [])
|
||||
.filter((a) => !String(a.name).startsWith('token_'))
|
||||
.map((a) => {
|
||||
const interfaces = parseArr(a.network_interfaces).filter((n) => !IFACE_HIDE.test(n));
|
||||
const e = ex[a.id];
|
||||
return {
|
||||
agent_id: a.id,
|
||||
agent_name: a.name,
|
||||
ip_address: a.ip_address,
|
||||
capable: parseArr(a.capabilities).includes('keepalived_management'),
|
||||
interfaces,
|
||||
participate: !!e,
|
||||
role: e ? e.role : 'BACKUP',
|
||||
priority: e ? e.priority : 100,
|
||||
network_interface: e ? e.network_interface : (interfaces[0] || ''),
|
||||
};
|
||||
});
|
||||
};
|
||||
|
||||
const loadPoolMembers = useCallback(async (poolId, existing) => {
|
||||
if (!poolId) { setMemberRows([]); return; }
|
||||
try {
|
||||
const res = await fetch(`/api/agents?pool_id=${poolId}`, { headers: authHeaders() });
|
||||
const data = res.ok ? await res.json() : { agents: [] };
|
||||
setMemberRows(buildMemberRows(data.agents || [], existing));
|
||||
} catch (e) {
|
||||
console.error('loadPoolMembers failed', e);
|
||||
setMemberRows([]);
|
||||
}
|
||||
}, []);
|
||||
|
||||
const setRow = (agentId, patch) =>
|
||||
setMemberRows((rows) => rows.map((r) => (r.agent_id === agentId ? { ...r, ...patch } : r)));
|
||||
|
||||
// Toggling a node into the VIP: if no other participating node is MASTER yet, make this
|
||||
// one the MASTER. This makes the single-node case work without the operator having to flip
|
||||
// the role by hand (a one-node VIP's only node IS the master), and gives a sensible default
|
||||
// for multi-node (first picked = master, the rest backup). Editing keeps stored roles.
|
||||
const toggleParticipate = (agentId, on) =>
|
||||
setMemberRows((rows) => {
|
||||
const otherMaster = rows.some((r) => r.agent_id !== agentId && r.participate && r.role === 'MASTER');
|
||||
return rows.map((r) => {
|
||||
if (r.agent_id !== agentId) return r;
|
||||
if (on && !otherMaster) return { ...r, participate: true, role: 'MASTER', priority: 150 };
|
||||
return { ...r, participate: on };
|
||||
});
|
||||
});
|
||||
|
||||
const openCreate = () => {
|
||||
setEditing(null);
|
||||
setSelectedPoolId(null);
|
||||
setMemberRows([]);
|
||||
form.resetFields();
|
||||
form.setFieldsValue({ prefix_length: 24, advert_int: 1, use_unicast: true, track_haproxy: true });
|
||||
setModalVisible(true);
|
||||
};
|
||||
|
||||
const openEdit = (vip) => {
|
||||
setEditing(vip);
|
||||
setSelectedPoolId(vip.pool_id);
|
||||
loadPoolMembers(vip.pool_id, vip.members);
|
||||
form.resetFields();
|
||||
form.setFieldsValue({
|
||||
name: vip.name, description: vip.description, pool_id: vip.pool_id,
|
||||
virtual_ip: vip.virtual_ip, prefix_length: vip.prefix_length,
|
||||
virtual_router_id: vip.virtual_router_id, advert_int: vip.advert_int,
|
||||
use_unicast: vip.use_unicast, track_haproxy: vip.track_haproxy,
|
||||
});
|
||||
setModalVisible(true);
|
||||
};
|
||||
|
||||
const submit = async () => {
|
||||
let values;
|
||||
try { values = await form.validateFields(); }
|
||||
catch { return; }
|
||||
|
||||
const chosen = memberRows.filter((r) => r.participate);
|
||||
if (chosen.length < 1) { message.error('Select at least 1 participating node.'); return; }
|
||||
const masters = chosen.filter((r) => r.role === 'MASTER');
|
||||
if (masters.length !== 1) { message.error('Exactly one participating node must be MASTER.'); return; }
|
||||
if (chosen.some((r) => !r.network_interface)) { message.error('Pick a network interface for every participating node.'); return; }
|
||||
const maxBackup = Math.max(...chosen.filter((r) => r.role === 'BACKUP').map((r) => r.priority));
|
||||
if (masters[0].priority <= maxBackup) { message.error('The MASTER must have a higher priority than every BACKUP.'); return; }
|
||||
|
||||
const body = {
|
||||
...values,
|
||||
members: chosen.map((r) => ({
|
||||
agent_id: r.agent_id, network_interface: r.network_interface, role: r.role, priority: r.priority,
|
||||
})),
|
||||
};
|
||||
if (!body.auth_pass) delete body.auth_pass; // omit to keep existing on edit
|
||||
try {
|
||||
const url = editing ? `/api/vip/${editing.id}` : '/api/vip';
|
||||
const res = await fetch(url, { method: editing ? 'PUT' : 'POST', headers: authHeaders(), body: JSON.stringify(body) });
|
||||
if (res.ok) {
|
||||
message.success(editing
|
||||
? 'VIP updated (PENDING) — apply it from the Apply Management page'
|
||||
: 'VIP created (PENDING) — apply it from the Apply Management page');
|
||||
setModalVisible(false);
|
||||
fetchVips();
|
||||
} else {
|
||||
message.error(await fetchApiError(res, 'Failed to save VIP'));
|
||||
}
|
||||
} catch (e) {
|
||||
message.error('Failed to save VIP: ' + e.message);
|
||||
}
|
||||
};
|
||||
|
||||
const deleteVip = async (vip, purge) => {
|
||||
try {
|
||||
const res = await fetch(`/api/vip/${vip.id}${purge ? '?purge_package=true' : ''}`,
|
||||
{ method: 'DELETE', headers: authHeaders() });
|
||||
if (res.ok) {
|
||||
let body = {};
|
||||
try { body = await res.json(); } catch (_) { /* ignore */ }
|
||||
// Backend returns staged=true (approval required, VIP still running) or staged=false
|
||||
// (never-applied VIP removed at once). Surface its exact message either way.
|
||||
(body.staged ? message.info : message.success)(
|
||||
body.message || 'Deletion requested.');
|
||||
setDeleteTarget(null); fetchVips();
|
||||
} else message.error(await fetchApiError(res, 'Delete failed'));
|
||||
} catch (e) { message.error('Delete failed: ' + e.message); }
|
||||
};
|
||||
|
||||
// Diagnostics: per-member deploy state/ack from GET /api/vip/{id}/status — the live view
|
||||
// of what each node reported (installing/applied/error/externally-managed), most useful
|
||||
// while a freshly-applied VIP is SYNCING (keepalived install can take ~30s).
|
||||
const openDiagnostics = async (vip) => {
|
||||
setDiagVip(vip); setDiagData(null); setDiagLoading(true);
|
||||
try {
|
||||
const res = await fetch(`/api/vip/${vip.id}/status`, { headers: authHeaders() });
|
||||
if (res.ok) setDiagData(await res.json());
|
||||
else message.error(await fetchApiError(res, 'Failed to load diagnostics'));
|
||||
} catch (e) { message.error('Diagnostics failed: ' + e.message); }
|
||||
finally { setDiagLoading(false); }
|
||||
};
|
||||
|
||||
// Colorful, IP-Inventory/Agent-consistent state: live VRRP MASTER (green, crowned) /
|
||||
// BACKUP (orange) / FAULT (red); when the agent hasn't reported a live state yet, show
|
||||
// the configured role as a dashed outline tag (same color) so it's clearly "intended,
|
||||
// not yet observed". A node whose agent is too old gets an "awaiting agent" flag.
|
||||
const renderMembers = (_, vip) => (
|
||||
<Space direction="vertical" size={4}>
|
||||
{(vip.members || []).map((m) => {
|
||||
const live = m.keepalive_state && m.keepalive_state !== 'NONE' ? m.keepalive_state : null;
|
||||
const roleColor = m.role === 'MASTER' ? 'green' : 'orange';
|
||||
const awaiting = !live && !m.keepalived_capable;
|
||||
return (
|
||||
<Space key={m.agent_id} size={6}>
|
||||
<Text style={{ fontSize: 12 }}>{m.agent_name || `agent ${m.agent_id}`}</Text>
|
||||
{live ? (
|
||||
<Tooltip title={`Live VRRP state: ${live}`}>
|
||||
<Tag
|
||||
color={live === 'MASTER' ? 'green' : live === 'BACKUP' ? 'orange' : 'red'}
|
||||
icon={live === 'MASTER' ? <CrownOutlined /> : undefined}
|
||||
style={{ marginInlineEnd: 0, fontWeight: 600 }}
|
||||
>
|
||||
{live}
|
||||
</Tag>
|
||||
</Tooltip>
|
||||
) : (
|
||||
<Tooltip title="Configured role — live VRRP state not observed yet (agent offline or still converging).">
|
||||
<Tag color={roleColor} style={{ marginInlineEnd: 0, borderStyle: 'dashed', opacity: 0.85 }}>
|
||||
{m.role}
|
||||
</Tag>
|
||||
</Tooltip>
|
||||
)}
|
||||
{awaiting && (
|
||||
<Tooltip title="This node's agent does not advertise keepalived_management — upgrade the agent.">
|
||||
<Tag color="gold" icon={<WarningOutlined />} style={{ marginInlineEnd: 0 }}>awaiting agent</Tag>
|
||||
</Tooltip>
|
||||
)}
|
||||
</Space>
|
||||
);
|
||||
})}
|
||||
</Space>
|
||||
);
|
||||
|
||||
const columns = [
|
||||
{ title: 'Name', dataIndex: 'name', key: 'name' },
|
||||
{ title: 'Virtual IP', key: 'vip', render: (_, v) => <Text code>{v.virtual_ip}/{v.prefix_length}</Text> },
|
||||
{ title: 'Pool', dataIndex: 'pool_name', key: 'pool' },
|
||||
{ title: 'VRID', dataIndex: 'virtual_router_id', key: 'vrid' },
|
||||
{ title: 'Members / Live state', key: 'members', render: renderMembers },
|
||||
{
|
||||
title: 'Status', key: 'status', render: (_, v) => {
|
||||
const s = v.deploy_status || v.last_config_status;
|
||||
const d = DEPLOY_STATUS[s] || { color: 'default', label: s, tip: '' };
|
||||
const count = (s === 'SYNCING' || s === 'AWAITING' || s === 'ACTIVE' || s === 'DELETING' || s === 'DELETED') && v.deploy_total
|
||||
? ` (${v.deploy_synced}/${v.deploy_total})` : '';
|
||||
return (
|
||||
<Tooltip title={d.tip}>
|
||||
<Tag color={d.color} icon={(s === 'SYNCING' || s === 'DELETING') ? <SyncOutlined spin /> : s === 'AWAITING' ? <ClockCircleOutlined /> : undefined}>
|
||||
{(d.label || s)}{count}
|
||||
</Tag>
|
||||
</Tooltip>
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Actions', key: 'actions', render: (_, v) => (
|
||||
<Space>
|
||||
{v.last_config_status === 'PENDING' && (
|
||||
<Tooltip title="Apply pending configuration changes">
|
||||
<Button type="primary" size="small" icon={<PlayCircleOutlined />}
|
||||
onClick={() => { window.location.href = '/apply-management'; }}
|
||||
style={{ backgroundColor: '#1890ff', borderColor: '#1890ff' }}>
|
||||
Apply
|
||||
</Button>
|
||||
</Tooltip>
|
||||
)}
|
||||
<Tooltip title="Edit VIP (changes become PENDING; apply from Apply Management)">
|
||||
<Button size="small" icon={<EditOutlined />} onClick={() => openEdit(v)} />
|
||||
</Tooltip>
|
||||
<Tooltip title="Diagnostics — per-node keepalived deploy status & logs">
|
||||
<Button size="small" icon={<FileSearchOutlined />} onClick={() => openDiagnostics(v)} />
|
||||
</Tooltip>
|
||||
<Tooltip title="Delete VIP">
|
||||
<Button size="small" danger icon={<DeleteOutlined />}
|
||||
onClick={() => setDeleteTarget(v)} />
|
||||
</Tooltip>
|
||||
</Space>
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
// Member-selection table inside the modal — the pool's installed agents (nodes).
|
||||
const memberColumns = [
|
||||
{
|
||||
title: 'Node (agent)', key: 'node', render: (_, r) => (
|
||||
<span>
|
||||
<Text strong>{r.agent_name}</Text>{' '}
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>{r.ip_address ? `(${r.ip_address})` : '(no IP yet)'}</Text>
|
||||
{!r.capable && (
|
||||
<Tooltip title="This agent doesn't advertise keepalived_management — upgrade it or this node won't deploy.">
|
||||
{' '}<Tag color="gold" icon={<WarningOutlined />}>agent too old</Tag>
|
||||
</Tooltip>
|
||||
)}
|
||||
</span>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Participate', key: 'participate', width: 100, render: (_, r) => (
|
||||
<Switch checked={r.participate} onChange={(c) => toggleParticipate(r.agent_id, c)} />
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Role', key: 'role', width: 130, render: (_, r) => (
|
||||
<Select size="small" style={{ width: 110 }} value={r.role} disabled={!r.participate}
|
||||
onChange={(val) => setRow(r.agent_id, { role: val, priority: val === 'MASTER' ? 150 : 100 })}>
|
||||
<Option value="MASTER">MASTER</Option>
|
||||
<Option value="BACKUP">BACKUP</Option>
|
||||
</Select>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Priority', key: 'priority', width: 110, render: (_, r) => (
|
||||
<InputNumber size="small" min={1} max={254} value={r.priority} disabled={!r.participate}
|
||||
onChange={(val) => setRow(r.agent_id, { priority: val })} />
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Interface', key: 'iface', width: 160, render: (_, r) => (
|
||||
<Select size="small" style={{ width: 140 }} value={r.network_interface || undefined}
|
||||
placeholder="interface" disabled={!r.participate} showSearch
|
||||
onChange={(val) => setRow(r.agent_id, { network_interface: val })}
|
||||
notFoundContent="no interfaces reported"
|
||||
options={(r.interfaces || []).map((n) => ({ label: n, value: n }))}
|
||||
{...((r.interfaces || []).length === 0 ? { mode: 'tags' } : {})} />
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
return (
|
||||
<div>
|
||||
<Card>
|
||||
<div style={{ display: 'flex', justifyContent: 'space-between', alignItems: 'center', marginBottom: 16 }}>
|
||||
<Title level={2} style={{ margin: 0 }}>HA / VIP (Keepalived)</Title>
|
||||
<Space>
|
||||
<Button icon={<ReloadOutlined />} onClick={fetchVips}>Refresh</Button>
|
||||
<Button type="primary" icon={<PlusOutlined />} onClick={openCreate}>Create VIP</Button>
|
||||
</Space>
|
||||
</div>
|
||||
{/* The cloud caveat is rarely relevant for the on-prem target audience, so it's a
|
||||
subtle, collapsed-by-default info note (not a prominent yellow warning). */}
|
||||
<div style={{ marginBottom: 12 }}>
|
||||
<Button type="link" size="small" icon={<InfoCircleOutlined />} style={{ paddingLeft: 0 }}
|
||||
onClick={() => setShowL2Note((v) => !v)}>
|
||||
Network requirements (on-prem / L2)
|
||||
</Button>
|
||||
{showL2Note && (
|
||||
<Alert
|
||||
type="info" showIcon style={{ marginTop: 4 }}
|
||||
message="On-prem / L2 networks"
|
||||
description="VRRP-based VIP failover targets bare-metal / VMware / on-prem L2 segments. On AWS/Azure/GCP, cloud fabrics don't honor VRRP/gratuitous-ARP, so VIPs won't move. Ensure VRRP (IP protocol 112) is permitted by host firewalls."
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
<Table rowKey="id" columns={columns} dataSource={vips} loading={loading} pagination={{ pageSize: 10 }} />
|
||||
</Card>
|
||||
|
||||
<Modal
|
||||
title={editing ? `Edit VIP — ${editing.name}` : 'Create VIP'}
|
||||
open={modalVisible}
|
||||
onCancel={() => setModalVisible(false)}
|
||||
onOk={submit}
|
||||
okText={editing ? 'Save (PENDING)' : 'Create (PENDING)'}
|
||||
width={880}
|
||||
destroyOnClose
|
||||
>
|
||||
<Form form={form} layout="vertical">
|
||||
<Form.Item name="name" label="Name" rules={[{ required: true }]}>
|
||||
<Input placeholder="web-vip" disabled={!!editing} />
|
||||
</Form.Item>
|
||||
<Form.Item name="description" label="Description">
|
||||
<Input placeholder="optional" />
|
||||
</Form.Item>
|
||||
{!editing && (
|
||||
<Form.Item name="pool_id" label="Pool" rules={[{ required: true }]}
|
||||
tooltip="The VIP's nodes are the HAProxy servers (agents) already enrolled in this pool.">
|
||||
<Select placeholder="Select a pool" onChange={(pid) => { setSelectedPoolId(pid); loadPoolMembers(pid); }}>
|
||||
{pools.map((p) => <Option key={p.id} value={p.id}>{p.name}</Option>)}
|
||||
</Select>
|
||||
</Form.Item>
|
||||
)}
|
||||
<Space size="large" style={{ display: 'flex' }}>
|
||||
<Form.Item name="virtual_ip" label="Virtual IP (IPv4)" rules={[{ required: true }]}>
|
||||
<Input placeholder="10.0.0.100" />
|
||||
</Form.Item>
|
||||
<Form.Item name="prefix_length" label="Prefix" rules={[{ required: true }]}>
|
||||
<InputNumber min={1} max={32} />
|
||||
</Form.Item>
|
||||
<Form.Item name="virtual_router_id" label="VRID (blank = auto)">
|
||||
<InputNumber min={1} max={255} placeholder="auto" />
|
||||
</Form.Item>
|
||||
<Form.Item name="advert_int" label="Advert int (s)">
|
||||
<InputNumber min={1} max={255} />
|
||||
</Form.Item>
|
||||
</Space>
|
||||
<Space size="large">
|
||||
<Form.Item name="use_unicast" label="Unicast VRRP" valuePropName="checked" tooltip="Recommended; works where multicast is blocked.">
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item name="track_haproxy" label="Fail over when HAProxy drops" valuePropName="checked">
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item name="auth_pass" label="VRRP secret (≤8 chars)">
|
||||
<Input.Password placeholder={editing ? '•••• (unchanged)' : 'optional'} maxLength={8} />
|
||||
</Form.Item>
|
||||
</Space>
|
||||
|
||||
<Text strong>Participating nodes</Text>
|
||||
<div style={{ color: '#888', fontSize: 12, marginBottom: 8 }}>
|
||||
These are the HAProxy servers (agents) already enrolled in this pool — toggle which join the VIP.
|
||||
Pick <b>exactly one MASTER</b> (highest priority); the rest are BACKUP. A <b>single node</b> is allowed
|
||||
(a keepalived-managed VIP <i>without</i> failover) — add a second node for real HA. Add new servers from
|
||||
the standard Agent Management install flow.
|
||||
</div>
|
||||
<Alert
|
||||
type="info" showIcon style={{ marginBottom: 8 }}
|
||||
message="keepalived is installed automatically on Apply"
|
||||
description={<>On Apply, any participating node that doesn’t already run keepalived will <b>install it from the node’s OS package repositories</b> (apt/dnf/yum/zypper/apk) — make sure the node can reach its repos (internet or an internal mirror). A node already running a <b>hand-managed</b> keepalived is left untouched (reported as “externally managed”).</>}
|
||||
/>
|
||||
<Table
|
||||
rowKey="agent_id"
|
||||
size="small"
|
||||
columns={memberColumns}
|
||||
dataSource={memberRows}
|
||||
pagination={false}
|
||||
locale={{ emptyText: selectedPoolId ? 'No agents in this pool — install agents from Agent Management first.' : 'Select a pool to list its nodes.' }}
|
||||
/>
|
||||
</Form>
|
||||
</Modal>
|
||||
|
||||
{/* Delete confirmation with opt-in package uninstall. Enterprise-safe DEFAULT keeps the
|
||||
package (just stop/disable + remove our config + release the VIP). */}
|
||||
<Modal
|
||||
title="Delete VIP — requires approval"
|
||||
open={!!deleteTarget}
|
||||
onCancel={() => setDeleteTarget(null)}
|
||||
onOk={() => deleteVip(deleteTarget, false)}
|
||||
okText="Stage deletion for approval"
|
||||
okButtonProps={{ danger: true }}
|
||||
>
|
||||
{deleteTarget && (
|
||||
<Space direction="vertical" size={12} style={{ width: '100%' }}>
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
message="This does NOT delete the VIP immediately"
|
||||
description={<>It stages the deletion for approval. The VIP <b>keeps running, untouched</b>, on its
|
||||
member node(s) until you <b>Approve</b> it on the <b>Apply Management</b> page — and you can
|
||||
<b> Reject</b> it there to keep it. The node is changed <b>only after approval</b>, so an
|
||||
accidental click can't tear down a production VIP.</>}
|
||||
/>
|
||||
<Text>
|
||||
Stage deletion of <Text strong>{deleteTarget.name}</Text> ({deleteTarget.virtual_ip}/{deleteTarget.prefix_length})?
|
||||
When approved, the member node(s) <b>stop & disable keepalived, remove the config we manage, and release the VIP</b>. The keepalived package itself is left installed, so re-adding a VIP later is instant.
|
||||
</Text>
|
||||
</Space>
|
||||
)}
|
||||
</Modal>
|
||||
|
||||
{/* Per-node keepalived deploy diagnostics + node-side log commands (esp. during SYNCING). */}
|
||||
<Modal
|
||||
title={diagVip ? `Diagnostics — ${diagVip.name}` : 'Diagnostics'}
|
||||
open={!!diagVip}
|
||||
onCancel={() => { setDiagVip(null); setDiagData(null); }}
|
||||
width={780}
|
||||
footer={[
|
||||
<Button key="refresh" icon={<ReloadOutlined />} onClick={() => diagVip && openDiagnostics(diagVip)}>Refresh</Button>,
|
||||
<Button key="close" type="primary" onClick={() => { setDiagVip(null); setDiagData(null); }}>Close</Button>,
|
||||
]}
|
||||
>
|
||||
{diagLoading && <div style={{ textAlign: 'center', padding: 24 }}><Spin /></div>}
|
||||
{!diagLoading && diagData && (
|
||||
<Space direction="vertical" size={12} style={{ width: '100%' }}>
|
||||
<Text type="secondary">
|
||||
Staging <Tag>{diagData.last_config_status}</Tag> — each node reports its deploy state after every poll; a fresh keepalived install can take ~30s.
|
||||
</Text>
|
||||
<Table
|
||||
size="small" rowKey={(m) => m.agent_name} pagination={false}
|
||||
dataSource={diagData.members || []}
|
||||
columns={[
|
||||
{ title: 'Node', key: 'node', render: (_, m) => (
|
||||
<Space size={4}>
|
||||
<Text style={{ fontSize: 12 }}>{m.agent_name}</Text>
|
||||
{m.role === 'MASTER'
|
||||
? <Tag color="green" icon={<CrownOutlined />} style={{ marginInlineEnd: 0 }}>MASTER</Tag>
|
||||
: <Tag color="orange" style={{ marginInlineEnd: 0 }}>BACKUP</Tag>}
|
||||
</Space>) },
|
||||
{ title: 'Agent', dataIndex: 'agent_status', key: 'agent',
|
||||
render: (s) => <Tag color={s === 'online' ? 'green' : 'red'}>{s || 'offline'}</Tag> },
|
||||
{ title: 'Live VRRP', dataIndex: 'keepalive_state', key: 'live',
|
||||
render: (s) => (s && s !== 'NONE')
|
||||
? <Tag color={s === 'MASTER' ? 'green' : s === 'BACKUP' ? 'orange' : 'red'}>{s}</Tag>
|
||||
: <Text type="secondary">—</Text> },
|
||||
{ title: 'Deploy', key: 'deploy', render: (_, m) => {
|
||||
const st = m.deploy_state;
|
||||
const color = st === 'enabled' ? 'green' : st === 'error' ? 'red'
|
||||
: st === 'externally_managed' ? 'gold' : 'blue';
|
||||
return <Tooltip title={m.deploy_message || ''}><Tag color={color}>{m.convergence || st || 'pending'}</Tag></Tooltip>;
|
||||
} },
|
||||
{ title: 'Last ack', dataIndex: 'deploy_at', key: 'ack',
|
||||
render: (t) => t ? <Text style={{ fontSize: 11 }}>{new Date(t).toLocaleString()}</Text> : <Text type="secondary">—</Text> },
|
||||
]}
|
||||
/>
|
||||
{(diagData.members || []).some((m) => m.deploy_message) && (
|
||||
<Card size="small" title="Latest node messages" bodyStyle={{ padding: 8 }}>
|
||||
{(diagData.members || []).filter((m) => m.deploy_message).map((m) => (
|
||||
<div key={m.agent_name} style={{ fontSize: 12 }}><Text strong>{m.agent_name}:</Text> {m.deploy_message}</div>
|
||||
))}
|
||||
</Card>
|
||||
)}
|
||||
<Alert
|
||||
type="info" showIcon
|
||||
message="See the live install / VRRP logs on the node"
|
||||
description={
|
||||
<pre style={{ margin: 0, fontSize: 11, whiteSpace: 'pre-wrap' }}>{`systemctl status keepalived --no-pager
|
||||
journalctl -u keepalived --no-pager -n 50
|
||||
tail -n 100 /var/log/haproxy-agent/agent.log | grep -i keepalived`}</pre>
|
||||
}
|
||||
/>
|
||||
</Space>
|
||||
)}
|
||||
{!diagLoading && !diagData && <Text type="secondary">No diagnostics available.</Text>}
|
||||
</Modal>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
export default VIPManagement;
|
||||
@@ -22,7 +22,7 @@ spec:
|
||||
serviceAccountName: haproxy-openmanager-redis
|
||||
containers:
|
||||
- name: redis
|
||||
image: redis:7-alpine
|
||||
image: redis:8.8.0-alpine
|
||||
command:
|
||||
- redis-server
|
||||
- /usr/local/etc/redis/redis.conf
|
||||
|
||||
@@ -22,7 +22,9 @@ spec:
|
||||
serviceAccountName: haproxy-openmanager-nginx
|
||||
containers:
|
||||
- name: nginx
|
||||
image: nginx:alpine
|
||||
# Pinned to a patched release for the nginx "poolslip" advisory
|
||||
# (mainline <=1.31.0 affected; fixed in mainline 1.31.1+ / stable 1.30.2+).
|
||||
image: nginx:1.31.1-alpine
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
name: http
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"version": "1.6.0",
|
||||
"releaseName": "Multi-Factor Authentication (MFA)",
|
||||
"releaseDate": "2026-05-18"
|
||||
}
|
||||
Reference in New Issue
Block a user