mirror of
https://github.com/taylanbakircioglu/haproxy-openmanager.git
synced 2026-10-02 15:08:13 +00:00
Compare commits
36 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 1d4e4286af | |||
| 0eb587dfa8 | |||
| 9e64002f1c | |||
| 4f24d5bdd9 | |||
| a87994e06a | |||
| 7d95c737f0 | |||
| eda7f36c93 | |||
| 11e5bf57d9 | |||
| a4c74f2a52 | |||
| a36dd87a74 | |||
| 2f125da043 | |||
| 9e6e4dd03b | |||
| 47cc79dcf7 | |||
| bb774141d4 | |||
| acfd32dd63 | |||
| 78af849fdc | |||
| 709817fec3 | |||
| 822c441d34 | |||
| 1ca811e211 | |||
| 164841219a | |||
| d92a7e9660 | |||
| 7dfd31832a | |||
| 8ac567dfe0 | |||
| 02667bbda4 | |||
| c97df53da8 | |||
| be01ddd119 | |||
| bbd8359f50 | |||
| dd7b7822bf | |||
| c4139bb11a | |||
| dbb9189f16 | |||
| eee0a4716a | |||
| 3c8832330a | |||
| 81ab674072 | |||
| 6bf6d016f5 | |||
| 0a0226c758 | |||
| 8e534ef170 |
+33
-3
@@ -17,11 +17,34 @@ REDIS_URL=redis://redis:6379
|
||||
# Change this to a strong random string in production
|
||||
SECRET_KEY=your-secret-key-change-this-in-production
|
||||
|
||||
# Optional: dedicated Fernet key for encrypting VRRP secrets of HA/VIP (Issue #27).
|
||||
# If unset, it is derived from SECRET_KEY (HKDF), exactly like MFA. Set an explicit
|
||||
# key (urlsafe-base64, 32 bytes) in production if you want independent key rotation.
|
||||
# ----------------------------------------------------------------------------
|
||||
# Optional per-purpose encryption keys.
|
||||
#
|
||||
# Every secret the application stores is encrypted at rest with Fernet. Each class
|
||||
# derives its own key, so rotating one never affects another. If a variable below is
|
||||
# unset, that class's key is derived from SECRET_KEY via HKDF — which works, but means
|
||||
# rotating SECRET_KEY makes the existing values of that class UNDECRYPTABLE. Set an
|
||||
# explicit key (urlsafe-base64, 32 bytes) in production if you want independent
|
||||
# rotation. Generate one with:
|
||||
# python -c "from cryptography.fernet import Fernet; print(Fernet.generate_key().decode())"
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
# VRRP secrets for HA/VIP (Issue #27).
|
||||
# VIP_ENCRYPTION_KEY=
|
||||
|
||||
# TOTP secrets for multi-factor authentication (Issue #18).
|
||||
# MFA_ENCRYPTION_KEY=
|
||||
|
||||
# Per-account DNS provider credentials for ACME DNS-01 (Issue #35).
|
||||
# Rotating this without re-entering credentials makes DNS-01 renewals fail until the
|
||||
# affected accounts' credentials are re-saved in ACME Automation.
|
||||
# DNS_PROVIDER_ENCRYPTION_KEY=
|
||||
|
||||
# Private keys of PENDING CSRs, held only until the signed certificate is imported
|
||||
# (Issue #53). Rotating this while CSRs are out for signature makes those CSRs
|
||||
# unusable — they must be deleted and re-created.
|
||||
# CSR_ENCRYPTION_KEY=
|
||||
|
||||
# ============================================================================
|
||||
# PUBLIC URL CONFIGURATION
|
||||
# ============================================================================
|
||||
@@ -36,6 +59,13 @@ PUBLIC_URL=http://localhost:8000
|
||||
|
||||
# Management base URL (defaults to PUBLIC_URL if not set)
|
||||
# Override this if your management interface is on a different URL
|
||||
#
|
||||
# This is also the last-resort fallback for the ACME HTTP-01 challenge backend,
|
||||
# i.e. the address written into haproxy.cfg as `server _acme_mgmt <host>:<port>`.
|
||||
# That address is resolved BY HAPROXY, ON THE HAPROXY NODE. If HAProxy runs
|
||||
# anywhere other than this machine, localhost points at the wrong box and HTTP-01
|
||||
# validation fails while DNS-01 keeps working. Set a routable address, with the
|
||||
# port, e.g. MANAGEMENT_BASE_URL=http://10.90.1.4:8080
|
||||
MANAGEMENT_BASE_URL=http://localhost:8000
|
||||
|
||||
# ============================================================================
|
||||
|
||||
@@ -107,7 +107,7 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
✅ **SSL Certificate Management** - Centralized SSL with expiration tracking
|
||||
✅ **CSR Creation** *(v1.9.0)* - Generate a private key + CSR in-app (RSA 2048/4096, ECDSA P-256/P-384, full subject + SANs), have it signed by any external CA, then import the signed certificate — the key never leaves the server
|
||||
✅ **ACME Auto SSL (Let's Encrypt)** - Automated certificate issuance, renewal, and deployment via ACME protocol
|
||||
✅ **ACME DNS-01 Challenge** *(v1.8.0)* - TXT-record validation for internal/isolated clusters (no public port 80) and wildcard certificates; pluggable DNS providers (Manual + Cloudflare), opt-in, HTTP-01 unchanged
|
||||
✅ **ACME DNS-01 Challenge** *(v1.8.0)* - TXT-record validation for internal/isolated clusters (no public port 80) and wildcard certificates; pluggable DNS providers (Manual + Cloudflare + GoDaddy *(v1.10.0)*), opt-in, HTTP-01 unchanged
|
||||
✅ **ACME Certificate Diagnostic Panel** - Automated preflight that checks agent readiness, DNS resolution, port 80 reachability, and ACME challenge ACL before issuing certificates
|
||||
✅ **WAF Rules** - Web Application Firewall management and deployment
|
||||
✅ **Agent Script Versioning** - Update agents via UI (Monaco editor) with auto-upgrade
|
||||
@@ -256,7 +256,7 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
- **Stuck Order Detection** *(v1.4.0)*: Setup wizard surfaces orders that the CA has validated but not yet downloaded, with one-click `Complete` action and automatic 60-second retry
|
||||
- **Multi-Provider Support**: Configurable ACME directory URL supports Let's Encrypt, ZeroSSL, Google Trust Services, Buypass, and custom CAs
|
||||
- **HTTP-01 Challenge**: Built-in challenge responder with automatic HAProxy routing injection; reserved backend name `_acme_challenge_backend` is auto-managed and protected from manual edits / agent sync collisions
|
||||
- **DNS-01 Challenge** *(v1.8.0 — Issue #35)*: Validate via a DNS TXT record instead of HTTP on port 80, for **internal/isolated clusters with no public ingress** and for **wildcard** certificates (`*.example.com`). Pluggable per-account DNS provider (Manual + Cloudflare to start; credentials encrypted at rest and verified on save), same PENDING → APPLIED pipeline, bounded automatic retry on propagation lag, and a DNS-01 event timeline. Opt-in via a global setting; HTTP-01 behaviour is unchanged. (See the *DNS-01 Challenge* subsection under ACME Auto SSL below.)
|
||||
- **DNS-01 Challenge** *(v1.8.0 — Issue #35)*: Validate via a DNS TXT record instead of HTTP on port 80, for **internal/isolated clusters with no public ingress** and for **wildcard** certificates (`*.example.com`). Pluggable per-account DNS provider (Manual + Cloudflare + GoDaddy *(v1.10.0)*; credentials encrypted at rest and verified on save), same PENDING → APPLIED pipeline, bounded automatic retry on propagation lag, and a DNS-01 event timeline. Opt-in via a global setting; HTTP-01 behaviour is unchanged. (See the *DNS-01 Challenge* subsection under ACME Auto SSL below.)
|
||||
- **ACME Account Management**: Register, view, and deactivate ACME accounts from the UI
|
||||
- **Staging Mode**: Test certificate issuance with Let's Encrypt staging environment before production
|
||||
- **Custom Staging Endpoint** *(v1.4.0)*: Optional `staging_url_override` setting lets you point staging mode at a private ACME test CA (e.g. Pebble) without touching the production directory URL
|
||||
@@ -989,9 +989,9 @@ DNS-01 is **opt-in** and fully backward compatible: it is disabled until an admi
|
||||
|
||||
- **Enable it**: Settings → ACME / SSL Automation → **DNS-01 Challenge (advanced)** → turn on *Enable DNS-01 Challenge* and Save. While off, DNS-01 options are hidden and no DNS-01 orders can be created.
|
||||
- **Per-account provider**: in ACME Automation, create (or reconfigure) an ACME account with **Challenge Method = DNS-01** and a **DNS Provider**. Provider credentials are **verified before saving** and **encrypted at rest** (Fernet, mirroring the VRRP/MFA secret pattern); they are never returned by the API or written to logs.
|
||||
- **Supported providers**: **Manual** (publish the TXT record yourself in any DNS — including fully internal DNS — then click *Verify*; works everywhere but cannot auto-renew unattended) and **Cloudflare** (API token with `Zone:DNS:Edit` + `Zone:Read`; the TXT record is created and cleaned up automatically and renews unattended). The provider interface is pluggable — more providers can be added without changing the issuance flow.
|
||||
- **Supported providers**: **Manual** (publish the TXT record yourself in any DNS — including fully internal DNS — then click *Verify*; works everywhere but cannot auto-renew unattended), **Cloudflare** (API token with `Zone:DNS:Edit` + `Zone:Read`; the TXT record is created and cleaned up automatically and renews unattended), and **GoDaddy** *(v1.10.0)* (a **Production** API Key + Secret pair from `developer.godaddy.com/keys` — the first key that dashboard issues is an OTE/test key and is rejected; the zone must be in the same GoDaddy account, which needs at least one registered domain before GoDaddy allows DNS API access at all. A **Personal Access Token** works too: paste it as the API Key and leave the Secret blank — that is the forward path as GoDaddy retires the `sso-key` scheme. TXT records are created and cleaned up automatically and renew unattended). The provider interface is pluggable — more providers can be added without changing the issuance flow.
|
||||
- **Same pipeline**: after validation the certificate follows the normal PENDING → APPLIED flow (assign to clusters / Apply Management) and the agent serves it — identical to HTTP-01 from finalize onward, with **zero agent or rendered-config changes** for DNS-01.
|
||||
- **Manual flow**: the order detail shows the exact `_acme-challenge.<domain>` record name + TXT value (copyable); publish it and click *I've added the records — Verify*. For Cloudflare it is automatic.
|
||||
- **Manual flow**: the order detail shows the exact `_acme-challenge.<domain>` record name + TXT value (copyable); publish it and click *I've added the records — Verify*. For Cloudflare and GoDaddy it is automatic.
|
||||
- **Resilience**: a propagation-lag failure is recovered by a **bounded fresh-order retry chain** (1 original + 3 retries with increasing backoff, kept under Let's Encrypt's rate limits); any orphaned TXT record is cleaned up by a reconcile sweep. The order detail shows a DNS-01 event timeline (publish → validation → cleanup).
|
||||
- **Wildcards**: `*.example.com` is validated at `_acme-challenge.example.com`; it does **not** cover the apex — add `example.com` as a separate name if you need both (the providers handle the two coexisting TXT values automatically).
|
||||
- **Scope (this release)**: the Site Wizard remains HTTP-01-only; issue DNS-01 / wildcard certificates from **ACME Automation**.
|
||||
@@ -2474,6 +2474,21 @@ Developed with ❤️ for the HAProxy community
|
||||
|
||||
## Release Notes
|
||||
|
||||
- **v1.10.14** (2026-08-14) — **A converged node keeps acknowledging**: the deploy report is the server's only evidence that a member node applied its `keepalived.conf`, and it was sent on the write path alone. Once the rendered config was on disk the agent took the idempotency early return on every cycle and never reported again, so a **single lost report** — a backend restart, a 5xx, a network blip — left the VIP reading `SYNCING (0/n)` with an empty *Last ack* forever, while the node was demonstrably running the right config. Nothing would ever reconcile the two: the node was correct, the page was not, and the only way out was to change the rendered config so the agent wrote it again. The agent now re-asserts its state on the idempotent path too, which costs one request per node per ~2.5 minutes and touches nothing on the node — keepalived is not reloaded and the file is not rewritten. This is a long-standing gap from the original HA/VIP work, surfaced when acknowledgements were dropped for an unrelated reason in v1.10.12. Agent-script change: sync the script from Agent Management and let the agents upgrade. No schema or API change.
|
||||
- **v1.10.13** (2026-08-14) — **Agent deploy acknowledgements were silently dropped** (regression in v1.10.12, fix it before or with that release): the takeover-retirement clause added to `POST /agents/{name}/keepalived-status` in v1.10.12 reused one query placeholder for both the assignment `last_deploy_hash=$n` and the comparison inside its `CASE`. PostgreSQL deduces a type per **use**, so the same placeholder came out as `text` in one and `character varying` in the other, and asyncpg rejected the statement with `AmbiguousParameterError`. The failure was not partial: the whole UPDATE never ran, so **no member ever recorded an acknowledgement**. Every VIP sat at `SYNCING (0/n)` with an empty *Last ack*, even after the nodes had deployed the config successfully, and teardown acknowledgements were lost the same way. The hash is now bound to its own placeholder, which is only ever compared against the column and therefore unambiguous. Verified against a real PostgreSQL: both statements execute, a matching hash retires the takeover authorisation, a non-matching hash and a NULL `applied_config_hash` both leave it in place, and every case records the acknowledgement. A test now asserts every `$n` in these statements is bound exactly once and that the count matches the arguments passed. Backend only: no schema, agent or API-shape change.
|
||||
- **v1.10.12** (2026-08-14) — **A valid keepalived config is no longer rejected by its own warning**: before writing a rendered `keepalived.conf` the agent validates it with `keepalived -t` and, on failure, keeps the running config and does not restart keepalived. That fail-safe is right, but it treated **any** non-zero exit as invalid, and keepalived's config-test exit code does not separate fatal from benign. Measured on 2.2.8: a clean config exits 0, but `Truncating auth_pass to 8 characters` exits **5** and so does a missing `}` or an `Unknown keyword`. A VRRP password longer than eight characters was therefore enough to make every apply fail, including on nodes whose own running config produces the same warning and has been serving the VIP for weeks. The gate now judges the **output**: messages known to be benign are dropped and anything that remains still fails, so it fails **closed** and an unrecognised message is treated as fatal. Verified against real keepalived: a truncation warning passes while a missing brace, an unknown keyword and a `SECURITY VIOLATION` are all still refused. The agent also **reports what keepalived said** now, in the log and in the status the HA/VIP page shows; discarding it left a correct refusal with no way to act on it. Agent-script change: sync the script from Agent Management and let the agents upgrade for it to take effect. No schema or API change.
|
||||
- **v1.10.11** (2026-08-14) — **The *Adoptable* tag names the problem that actually blocks adoption**: the tag and the disabled *Adopt* button were computed separately and could disagree. A pair blocked because its peer's `keepalived.conf` could not be parsed was labelled **MASTER missing** — technically true, since the unreadable node's `state MASTER` had not been counted, but it pointed the operator at the wrong node while the real reason sat in the button's own tooltip. Both now come from one ordered decision, so the label, its colour and the tooltip always describe the condition that stops the adoption; a group held up by an unreadable or unreachable peer reads **blocked by peer**, and two MASTERs is now distinct from none. Display only: what the endpoint accepts or refuses is unchanged. On the public repo this is the first artifact carrying v1.10.4 through v1.10.10: none was released separately, because VIP adoption did not work end to end until these fixes landed.
|
||||
- **v1.10.10** (2026-08-14) — **Adoption blockers are listed once per instance**: with the instance-based panel a two-node pair reported the *same* problems about the *same* shared config twice, once per member, and the line numbers differ between the two files so plain de-duplication did not collapse them. Four issues on a pair read as eight, in both the *Adoptable* tooltip and the adopt dialog. They are now merged on the message text with the leading `line N:` ignored, so each distinct problem appears once. Display only: the endpoint already evaluated the combined set and its refusals are unchanged.
|
||||
- **v1.10.9** (2026-08-14) — **Adoption refuses to strand a node or silently normalise a peer's settings**: v1.10.8 adopted the whole VRRP instance, but it could only *match* a node it was able to read, that was enabled, and that sat in the same pool. Each of those was a door a real member of the group left through silently — the nodes that remained were rewritten while the one that left kept serving the same address from an unmanaged config. Seen on a live pool: one node of a pair had an unclosed `vrrp_instance` block, so it parsed to nothing while its partner parsed cleanly. Instead of guarding each door, adoption now asks the question directly — is there **any** reported `keepalived.conf` that mentions this virtual address and is not among the nodes being taken over — and refuses naming the node and the reason (unparseable, agent disabled, different pool). Separately, four VIP-level fields (`prefix_length`, unicast/multicast mode, HAProxy tracking and the VRRP password) are stored once and re-rendered onto **every** member, so taking them from whichever node was clicked imposed its settings on the others; the prefix length is the sharpest, because the design refuses to *guess* a netmask for a live VIP and copying one node's netmask onto another is that same change by another name. Adoption now requires the nodes to agree on all four, and compares the VRRP secret by decrypting each node's token (Fernet is non-deterministic, so the ciphertexts cannot be compared). Finally, the takeover authorisation is genuinely one-shot: `takeover_expected_hash` was written at adoption and never cleared, so it stayed valid for that file content indefinitely — it is now retired the moment a member acknowledges our rendered config, gated on the acked hash matching so a failed deploy never drops it. No schema change, no agent change.
|
||||
- **v1.10.8** (2026-08-13) — **VIP adoption takes the whole VRRP instance**: adoption used to take only the node whose row was clicked, which broke the exact case the feature exists for, a running HA pair. Adopting the **BACKUP** alone produced a VIP that could never be applied (*exactly one member must be MASTER*); adopting the **MASTER** alone left the peer unmanaged, and adopting it afterwards hit the VRID-collision guard with 409, so the pair could not be completed from the panel at all. Most serious, on a **unicast** instance the single-member render dropped the unicast block entirely — the renderer emits it only when it has peer addresses — so keepalived fell back to **multicast** on the adopted node while its peer stayed unicast: they stop seeing each other and **both** claim the VIP. Adoption now resolves the whole instance, keyed on `(virtual_router_id, virtual address)` exactly as keepalived groups nodes, and every participating node becomes a member with the role, priority and interface its own file declares and its own one-shot takeover hash. It refuses, with the reason, when the group does not have exactly one MASTER, when the nodes disagree on `advert_int`, when a declared unicast peer is not among the nodes being adopted, or when a node already belongs to a live VIP — a rule create/edit enforced and adoption did not. The panel now lists one row per **instance** instead of per node. Two further fixes: the Apply Management **View Change** diff did not recognise the `adopt` action, so it fell through to the generic HAProxy diff and rendered the cluster's entire `haproxy.cfg` as removed; and **rejecting** an adoption hid the node from the panel permanently, because `adopted_vip_id` is write-once, a VIP is only ever soft-deleted (so the column's `ON DELETE SET NULL` never fires) and the agent does not re-report an unchanged file — adoptability is now derived from whether the linked VIP is still active, which self-heals reject, undo-reject and approved teardown alike. No schema change, no agent change.
|
||||
- **v1.10.7** (2026-08-13) — **HA / VIP follows the selected cluster**: the page ignored the cluster picker in the header. On a multi-cluster install both the VIP table and the new *Unmanaged keepalived detected* panel listed every cluster's nodes at once and did not change when the selection did, so the panel appeared to be stuck on one cluster's keepalived. Both lists now send `cluster_id`, resolved to that cluster's pool exactly as the Apply Management view already did. The API parameter is **optional**: a caller that omits it still receives the whole fleet, so nothing outside the page changes. This is a deliberate behaviour change for the VIP table, which was fleet-wide before. Backend and frontend only: no schema, no agent change.
|
||||
- **v1.10.6** (2026-08-13) — **VIP adoption panel was unreachable**: v1.10.4's *Unmanaged keepalived detected* panel never appeared, even on a fleet where the agents had reported their configs correctly. `GET /discoveries` was declared **after** `GET /{vip_id}` in `routers/vip.py`, and FastAPI matches routes in declaration order, so every request for the discovery list was answered by the get-one-VIP handler, which takes `vip_id: int` and rejected `"discoveries"` with **422** before the real handler ran. Nothing surfaced the failure: the agents reported normally, the rows landed in `vip_discoveries`, and the HA/VIP page treats any non-OK response as "nothing to show" — so the whole feature was invisible with no error anywhere. The route is moved above the parameterised ones, and a static source scan now asserts that **no** literal path in **any** router is shadowed by an earlier parameterised one, so the class of bug cannot come back silently. Data reported under v1.10.4 is not lost: existing `vip_discoveries` rows appear as soon as the fixed backend is deployed, with no agent action needed. Backend-only fix. No schema, API-shape or agent change.
|
||||
- **v1.10.5** (2026-08-09) — **HTTP-01 challenge backend on split deployments**: on a deployment where the HAProxy nodes and the management stack are on different hosts, HTTP-01 issuance could fail silently for weeks while DNS-01 kept working — the rendered config pointed `server _acme_mgmt` at an address that resolves **on the HAProxy node**, defaulting to loopback, and every diagnostic still reported success. The per-cluster `acme_backend_url` now has a UI field, changing it actually mints a config version, and the value is validated where it is written. Three adjacent bugs are fixed with it: a config-generation failure was returned as `# Error ...` text and then stored as an APPLIED version and pushed to agents as the cluster's whole `haproxy.cfg` (both call sites now refuse with 422); a nullable `frontends.mode` was interpolated raw and emitted `mode None`, which HAProxy rejects and which takes down the entire cluster config; and cluster creation silently dropped the ACME fields. `docker-compose.yml` now interpolates `PUBLIC_URL` / `MANAGEMENT_BASE_URL` instead of hardcoding them, with the old literals as defaults. Diagnostics read the response body so an SPA answering 200 is no longer counted as healthy, and every new condition is a warning rather than a failure so no install is locked on upgrade. No schema, API-shape or agent change.
|
||||
- **v1.10.4** (2026-08-08) — **Adopt an existing keepalived VIP** (Issue #27 follow-up): on a fleet that already runs keepalived, the **HA / VIP** page came up empty, because the flow was one-way — VIPs were declared in OpenManager and pushed to the node, and nothing ever read what was already there. Agents now **report the `keepalived.conf` they find and do not own** (strictly read-only; the node is never touched), the page lists those nodes under *Unmanaged keepalived detected*, and **Adopt** turns one `vrrp_instance` into a managed VIP with the values from the file instead of retyping them. The heartbeat could not drive this: it carries the VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs **eleven** fields, and guessing them is not cosmetic — a wrong `virtual_router_id` puts the nodes in separate VRRP domains and a wrong `auth_pass` makes them reject each other, so both would claim the VIP. Because adoption **replaces** the operator's file with OpenManager's render, the parser reports every directive it cannot reproduce — a `notify_master` hook, an LVS `virtual_server` section, a `vrrp_sync_group`, a second address in one instance, a custom `track_script` — and **refuses** while any remain; the operator can waive that class explicitly, but a value that is simply *unknown* (an absent VRID or prefix length) can never be waived, only supplied. keepalived's own documented defaults (`state BACKUP`, `priority 100`, `advert_int 1`) are applied and shown as assumed. The agent's ownership guard is **not** weakened: adoption authorises exactly **one** takeover of exactly the file that was analysed, pinned to its hash, so a config edited between adoption and Apply is still refused. The adopted VIP is created **PENDING** like any other, so nothing reaches the node until it is applied from Apply Management. VRRP passwords are Fernet-encrypted at ingest and masked in the stored copy and the preview. Schema change: one new table `vip_discoveries` plus two additive columns (SCHEMA_VERSION 10 → 11, auto-migrated, no existing table altered) — **see the upgrade notes: this bump re-seeds the four built-in roles, and the Linux agent script must reach the nodes before discovery starts**.
|
||||
- **v1.10.3** (2026-08-08) — **Multi-account ACME: the certificate wizard honours the account you pick**: with more than one ACME account registered, picking an **HTTP-01** account in *Request ACME Certificate* still produced a **DNS-01** request. Three faults compounded. (1) `Form.useWatch` reports only fields that are currently **rendered**, and the account `Select` lives on the *Configuration* step — so as soon as the wizard advanced to *Review* the watch read `undefined` and the wizard silently reverted to the default account, even though the value was still in the form store; the watches now pass `preserve: true`. The same fault disabled the **wildcard guard** on *Review*, the one step where Submit lives. (2) The UI and the backend disagreed on which account is the *default*: the backend takes the **newest** valid account (`ORDER BY created_at DESC`), the UI took the **oldest** entry of a list ordered by id — the opposite account whenever the two differ. The wizard now resolves the same one, and sends `account_id` **explicitly** so there is no guess left to disagree about. (3) `account_id` was read from the form store while `challenge_type` came from the reverted account object, so the request asked for DNS-01 validation on an HTTP-01 account and the API answered `The selected ACME account has no DNS provider configured for DNS-01.` — both are now derived from one resolved account. The *Review* step also showed the default account's address instead of the chosen one, and Submit stayed enabled for a deactivated account; both fixed. Frontend only — no schema, API-shape, agent or rendered-config changes, and single-account installations behave exactly as before.
|
||||
- **v1.10.2** (2026-08-08) — **Dark mode fixes on Apply Management**: several panels on the Apply Management page were painted with light-mode colour literals, so in dark mode the **Pending Changes** box rendered as a cream panel with light text on it — measured contrast **1.03:1**, effectively unreadable, now **11.50:1**. The same bug affected the added/removed rows in the *View Change* diff (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME and pending-version panels, the VIP pending-delete row, and the agent-error recommendation box; all now derive from theme tokens. Separately, **static confirm dialogs came up white in dark mode**: in Ant Design 5 the static `Modal.confirm` / `message` / `notification` APIs render into their own detached root and never see the app's `ConfigProvider`, so they always used the light algorithm. Registering `ConfigProvider.config({ holderRender })` once at the app root fixes **every** static dialog in the application (12 components use them), not only this page. Light mode is byte-identical — each token resolves under the default algorithm to exactly the literal it replaced. Frontend only: no schema, API, environment or agent change.
|
||||
- **v1.10.1** (2026-08-08) — **CSR private key encrypted at rest** (Issue #53): the private key of a **pending** CSR is now Fernet-encrypted in the database instead of stored as PEM. It is the one key in the system worth protecting this way — it sits idle for the entire signing window (days to weeks), is never transmitted to an agent, and is destroyed the moment the signed certificate is imported; `ssl_certificates.private_key_content` and the ACME order keys are unchanged, because agents must receive those in plaintext on every poll. The token replaces the PEM in the **same column**, so there is **no schema change and no `SCHEMA_VERSION` bump** (and therefore no re-seed of the built-in roles). CSRs created before this release keep a raw PEM and are still read transparently, so anything already out for signature imports normally with no data migration. The key derives from `SECRET_KEY` via HKDF with its own info string, independent of the VIP/MFA/DNS keys, and an optional `CSR_ENCRYPTION_KEY` enables independent rotation — rotating `SECRET_KEY` without it makes pending CSR keys unrecoverable, which now fails with an explicit "delete and re-create this CSR" error rather than a misleading key-mismatch. `.env.template` now documents all four per-purpose encryption keys. No API, UI or agent change.
|
||||
- **v1.10.0** (2026-08-07) — **GoDaddy DNS provider for DNS-01** (Issue #35 follow-up): DNS-01 challenges can now be published and cleaned up automatically through **GoDaddy**, alongside the existing Manual and Cloudflare providers, so wildcard and internal-cluster certificates on GoDaddy-hosted zones **renew unattended**. Credentials are a **Production API Key + Secret** pair from `developer.godaddy.com/keys` (a **Personal Access Token** also works — paste it as the Key and leave the Secret blank, which is the forward path as GoDaddy retires `sso-key`); they are **verified against the GoDaddy API before being saved** and **encrypted at rest** (Fernet, the same path as Cloudflare), and are never returned by the API, logged, or written to an order event. GoDaddy's v1 API has **no per-value TXT write** — `PUT` replaces an entire RRset — so add/remove are read-modify-write with sibling values merged back, empty-`data` tombstones filtered out, and `DELETE` used for the last value (`PUT []` is rejected); this is what keeps the **apex + wildcard** case (two TXT values at one `_acme-challenge` name) working, and the record path is hard-gated so it can never collapse onto the zone-wide endpoint that would wipe SPF/DKIM/DMARC. Zone lookup probes the records API rather than the domain listing, so **delegated sub-zones** resolve and small accounts are not falsely rejected. Registry-only addition: one new provider module plus one registry line — no frontend change (the credential form is schema-driven). No schema, API-shape, agent, or rendered-config changes; Manual, Cloudflare and HTTP-01 are unaffected.
|
||||
- **v1.9.0** (2026-08-04) — **CSR creation** (in-app key + CSR generation and signed-certificate import): a new **CSR tab** on the SSL Certificates page generates a private key and Certificate Signing Request server-side (RSA 2048/4096 or ECDSA P-256/P-384; full subject — O/OU/L/ST/C/email — plus DNS SANs with wildcard support), for certificates signed by an **external or corporate CA**. The operator downloads/copies the CSR PEM, has it signed, then imports the signed certificate (+ optional chain): the backend verifies the certificate against the stored key (hard gate), rejects expired certs, warns on SAN drift, and creates a normal SSL certificate entry (source `CSR`) that flows through the standard **PENDING → Apply Management → agent pull** pipeline. The private key **never leaves the server** — no CSR endpoint returns it, and after import the CSR row's key copy is destroyed (the key then lives only on the certificate, like every other key). Additive schema change: one new table `ssl_csrs` (SCHEMA_VERSION 9 → 10, auto-migrated, no existing table altered); key generation runs off the event loop and is rate-limited per user; existing `ssl.*` permissions govern all new endpoints. No agent or rendered-config changes.
|
||||
- **v1.8.10** (2026-07-20) — **Security hardening** (GHSA-7rhv-c5pc-69r8, GHSA-3p5c-m5m4-mjpx, GHSA-3vh4): three advisory classes remediated, backend-only, no agent changes. (1) **RCE**: the agent script-template read/write endpoints now require the `agents.version` permission on top of authentication — a poisoned template is executed as root on every HAProxy node, so authentication alone was insufficient. (2) **Missing authentication**: operator/UI endpoints that were served without a JWT (dashboard stats, pool/cluster listings, agent inventory, WAF rules, config validate/optimize, SSL config-versions, health deep/agents/clusters) are now gated by a `require_authenticated_user` dependency, and agent data-plane endpoints that treated the `X-API-Key` header as *optional* (heartbeat, config, ssl-certificates, upgrade-status, pending-requests) now hard-reject a missing key. In every case the auth check was moved **ahead of** the handler's `try:` block so a 401 can no longer be rewritten into a 500 by the generic exception handler. (3) **SSRF**: a new `utils/ssrf_guard.py` (https-only, IPv4-pinned connector, all resolved addresses must be public, no redirects) protects the ACME directory fetch, the signed-request target and the ACME connection test, which accept operator- or DB-supplied URLs; the connection test also stopped reflecting arbitrary upstream JSON. Frontend dependency advisories patched in the same release. No schema, API-shape or rendered-config changes.
|
||||
- **v1.8.9** (2026-07-13) — **ACL `-f` pattern-file support** (Issue #38 follow-up): ACL definitions that reference a host-side pattern file (`acl … -f /etc/haproxy/lists/blocked.lst`) are accepted on import and edit instead of being rejected. The referenced file lives on the HAProxy node and cannot be validated from the manager, so the manager emits an **advisory warning** rather than a hard rejection and lets the agent's `haproxy -c` check be the fail-safe gate (a broken reference fails validation on the node and the previous config is restored). Consistent with the SPOE handling introduced in v1.8.8.
|
||||
|
||||
@@ -1,3 +1,426 @@
|
||||
# Upgrade Notes — v1.10.14 (a converged node keeps acknowledging)
|
||||
|
||||
**Agent-script change, no schema change.** No `SCHEMA_VERSION` bump. After deploying, sync the
|
||||
Linux agent script from **Agent Management** and let the agents upgrade, or the fix does not
|
||||
reach the nodes.
|
||||
|
||||
- **Symptom:** a VIP shows `SYNCING (0/n)` with an empty *Last ack* even though every member node
|
||||
has the rendered `keepalived.conf` on disk, keepalived is running and the VIP is held.
|
||||
- **Cause:** the deploy report was sent only when the agent actually wrote the config. Once the
|
||||
node matched, it took the idempotency early return every cycle and never reported again, so any
|
||||
report lost in transit was never retried and the server's view stayed stale permanently.
|
||||
- **Fix:** the agent re-asserts its state on the idempotent path as well. One request per node
|
||||
per poll cycle (~2.5 min); nothing is written and keepalived is not reloaded.
|
||||
- **Recovery is automatic.** A VIP stuck at SYNCING converges on the first poll after the agents
|
||||
pick up the new script. No action on the nodes, no re-apply, no edit to force a rewrite.
|
||||
- **This is not new in 1.10.12.** The gap dates from the original HA/VIP work; it only became
|
||||
visible when acknowledgements were dropped for an unrelated reason.
|
||||
|
||||
**Rollback:** safe. Reverting restores the previous behaviour, in which a lost acknowledgement is
|
||||
never recovered.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.13 (deploy acknowledgements were dropped)
|
||||
|
||||
**Backend only, no schema change.** No `SCHEMA_VERSION` bump, no agent change. If you deployed
|
||||
v1.10.12, deploy this one too.
|
||||
|
||||
- **Regression in v1.10.12.** The takeover-retirement clause added to the keepalived status
|
||||
endpoint reused a query placeholder for both an assignment and a comparison. PostgreSQL types a
|
||||
placeholder per use, so it was deduced as `text` in one place and `character varying` in the
|
||||
other, and asyncpg refused the statement outright.
|
||||
- **Symptom:** a VIP stayed at `SYNCING (0/n)` with an empty *Last ack* even though the agent log
|
||||
showed `applied config for VIP <id>` on every member. Teardown acknowledgements were lost the
|
||||
same way, so a deletion never showed as complete.
|
||||
- **Nothing was damaged.** The failure was on the write of the acknowledgement, not on the node.
|
||||
Configs were deployed correctly throughout; only the reporting was lost. Existing VIPs converge
|
||||
on the next poll once this is deployed, with no action on the nodes.
|
||||
- **Verified against a real PostgreSQL**, not by inspection: both statements execute, a matching
|
||||
hash retires the takeover authorisation, a non-matching hash and a NULL `applied_config_hash`
|
||||
leave it in place, and the acknowledgement is recorded in every case.
|
||||
|
||||
**Rollback:** do not roll back to v1.10.12; roll back to v1.10.11 instead, which predates the
|
||||
clause entirely.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.12 (valid config rejected by its own warning)
|
||||
|
||||
**Agent-script change, no schema change.** No `SCHEMA_VERSION` bump. After deploying, sync the
|
||||
Linux agent script from **Agent Management** and let the agents upgrade, or the fix does not
|
||||
reach the nodes.
|
||||
|
||||
- **Symptom:** applying a VIP left it stuck at `SYNCING`, the node kept its previous config and
|
||||
the agent logged only `config validation failed (keepalived -t)`.
|
||||
- **Cause:** the agent treated any non-zero exit from `keepalived -t` as invalid. keepalived's
|
||||
config-test exit code does not separate fatal from benign: on 2.2.8 a clean config exits 0,
|
||||
while `Truncating auth_pass to 8 characters` exits 5 and so do a missing `}` and an unknown
|
||||
keyword. A VRRP password longer than eight characters was enough to block every apply, even on
|
||||
a node whose own running config emits the same warning.
|
||||
- **Fix:** the gate judges the output instead. Known-benign messages are dropped and anything
|
||||
left still fails, so it fails closed. Verified against real keepalived: the truncation warning
|
||||
passes; a missing brace, an unknown keyword and a `SECURITY VIOLATION` are refused.
|
||||
- **Also:** the agent now reports what keepalived actually said, in its log and in the status the
|
||||
HA/VIP page shows. The refusal was correct but unactionable without reproducing it by hand.
|
||||
- **The fail-safe itself is unchanged:** a config that genuinely fails validation is never
|
||||
written and keepalived is never restarted.
|
||||
|
||||
**Rollback:** safe. Reverting restores the stricter gate, which rejects valid configs whose
|
||||
password exceeds eight characters.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.11 (Adoptable tag names the real blocker)
|
||||
|
||||
**Frontend only, no schema change.** No `SCHEMA_VERSION` bump, no API change, no agent change.
|
||||
|
||||
- The *Adoptable* tag and the disabled *Adopt* button were derived separately, so they could
|
||||
name different problems. A pair blocked by a peer whose config could not be parsed showed
|
||||
**MASTER missing**, because the unreadable node's `state MASTER` had not been counted — true,
|
||||
but it sent the operator to the wrong node. Both now come from one ordered decision.
|
||||
- New label **blocked by peer** for a group held up by a node that references the same address
|
||||
but cannot be taken over with it. Two MASTERs is now distinct from none.
|
||||
- Display only. The endpoint's checks and refusals are unchanged.
|
||||
|
||||
**Rollback:** safe; purely presentational.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.10 (Adoption blockers listed once per instance)
|
||||
|
||||
**Frontend only, no schema change.** No `SCHEMA_VERSION` bump, no API change, no agent change.
|
||||
|
||||
- The adoption panel merged every member's blocker list, so a two-node pair showed each shared
|
||||
problem twice. The two files report different line numbers for the same directive, so exact
|
||||
de-duplication did not collapse them. Blockers are now merged on the message with the leading
|
||||
`line N:` ignored.
|
||||
- Display only. The endpoint already evaluated the combined set across all nodes, and what it
|
||||
accepts or refuses is unchanged.
|
||||
|
||||
**Rollback:** safe; purely presentational.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.9 (Adoption refuses to strand a node)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent change.
|
||||
|
||||
- **Adoption will not leave a node behind.** v1.10.8 resolved the whole VRRP instance, but only
|
||||
from nodes it could parse, that were enabled and that were in the same pool. Anything else fell
|
||||
out of the set silently while its peers were rewritten. Adoption now refuses if any reported
|
||||
`keepalived.conf` mentions the virtual address and is not among the nodes being taken over, and
|
||||
says which node and why.
|
||||
- **The nodes must agree on the shared fields.** `prefix_length`, unicast/multicast mode, HAProxy
|
||||
tracking and the VRRP password live on the VIP and are re-rendered onto every member, so one
|
||||
node's value used to be imposed on the rest. A disagreement is now refused with both values
|
||||
shown.
|
||||
- **The takeover authorisation is now retired on acknowledgement.** It is the permission to
|
||||
overwrite a `keepalived.conf` that does not carry our marker. It was never cleared, so it stayed
|
||||
valid for that exact file content indefinitely; restoring the pre-adoption file would have been
|
||||
overwritten again without fresh approval. It is now dropped once the member acks our rendered
|
||||
config, gated on the acked hash matching `applied_config_hash` so a failed deploy cannot strand
|
||||
the VIP.
|
||||
- **Nothing to do on upgrade.** Existing adopted VIPs keep working; their authorisation is retired
|
||||
on the next successful acknowledgement.
|
||||
|
||||
**Rollback:** safe. No schema or data migration; reverting restores the previous (more permissive)
|
||||
adoption checks.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.8 (VIP adoption takes the whole VRRP instance)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent impact: nothing about what the agent reports or how it takes a
|
||||
config over changes.
|
||||
|
||||
- **Adoption is now per VRRP instance, not per node.** Every node in the pool reporting the same
|
||||
`virtual_router_id` and virtual address becomes a member of one VIP, each with the role,
|
||||
priority and interface its own `keepalived.conf` declares, and each with its own one-shot
|
||||
takeover hash. The panel lists one row per instance.
|
||||
- **Why this mattered:** single-node adoption could not produce a working pair. The BACKUP alone
|
||||
failed apply, the MASTER alone left the peer unmanaged and the peer could not then be adopted
|
||||
(VRID collision). On a **unicast** instance it was worse than inconvenient: the render drops
|
||||
the unicast block when there are no peers, so the adopted node fell back to multicast while its
|
||||
peer stayed unicast and both could hold the address.
|
||||
- **New refusals, each with the reason in the message:** the group does not have exactly one
|
||||
MASTER; the nodes disagree on `advert_int`; a declared unicast peer is not among the nodes being
|
||||
adopted; a node is already a member of a live VIP.
|
||||
- **Apply Management "View Change" now renders the adopt diff correctly.** It did not recognise
|
||||
the `adopt` action and fell through to the generic HAProxy diff, which compared the staged
|
||||
`keepalived.conf` against the cluster's previous `haproxy.cfg` and showed the whole HAProxy
|
||||
config as removed. Alarming, but display-only — nothing was ever applied from that view.
|
||||
- **Rejecting an adoption is recoverable again.** It used to hide the node from the panel
|
||||
permanently. Nothing clears `vip_discoveries.adopted_vip_id`, a VIP is only soft-deleted so the
|
||||
column's `ON DELETE SET NULL` never fires, and the agent does not re-report a file whose hash
|
||||
has not changed. Adoptability is now derived from whether the linked VIP is still active.
|
||||
- **If you adopted a VIP on 1.10.4-1.10.7**, check it before applying: it may have only one
|
||||
member. Add the peer from the VIP's edit form, or reject the pending adoption and adopt again —
|
||||
the node reappears in the panel under this release.
|
||||
|
||||
**Rollback:** safe. No schema or data change; reverting restores the previous single-node
|
||||
adoption behaviour.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.7 (HA / VIP follows the selected cluster)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent impact.
|
||||
|
||||
- **The HA / VIP page ignored the cluster picker.** Both the VIP table and the *Unmanaged
|
||||
keepalived detected* panel queried the whole fleet, so on an install with more than one
|
||||
cluster the lists never changed when the selection did. Both now pass `cluster_id`, mapped to
|
||||
the cluster's pool the same way `GET /api/vip?cluster_id=` already worked for Apply
|
||||
Management.
|
||||
- **Behaviour change worth knowing:** the VIP table is now scoped to the selected cluster. It
|
||||
used to show every VIP in the fleet. If you relied on the fleet-wide view, the API still
|
||||
supports it — `GET /api/vip` and `GET /api/vip/discoveries` without `cluster_id` return
|
||||
everything, unchanged.
|
||||
- **API compatibility:** `cluster_id` is optional on both endpoints. Existing integrations that
|
||||
do not send it behave exactly as before.
|
||||
|
||||
**Rollback:** safe. The change is a query parameter plus the page that sends it; reverting
|
||||
restores the fleet-wide lists and touches no data.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.6 (VIP adoption panel was unreachable)
|
||||
|
||||
**One backend fix, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No API-shape change, no frontend change and zero agent impact.
|
||||
|
||||
- **v1.10.4's adoption panel never appeared.** `GET /discoveries` was declared after
|
||||
`GET /{vip_id}` in `routers/vip.py`. FastAPI matches routes in declaration order, so the
|
||||
discovery list was routed into the get-one-VIP handler, which declares `vip_id: int` and
|
||||
answered **422** before the real handler ran. The HA/VIP page treats any non-OK response as
|
||||
"nothing to show", so the feature was invisible with no error in any log.
|
||||
- **Nothing was lost.** The agent side always worked: discoveries were reported and stored in
|
||||
`vip_discoveries`. Deploy this backend and the rows appear immediately — no agent upgrade, no
|
||||
re-sync of the agent script, no re-report needed.
|
||||
- **If you are upgrading straight from 1.10.3 or earlier**, follow the v1.10.4 notes below as
|
||||
well: that release does bump `SCHEMA_VERSION` (10 → 11), which re-seeds the four built-in
|
||||
roles, and its agent script has to reach the nodes before discovery starts.
|
||||
- **Regression guard.** A static source scan now fails the build if any literal API path in any
|
||||
router is declared after a parameterised route that would swallow it. The whole router tree is
|
||||
clean as of this release.
|
||||
|
||||
**Rollback:** safe and immediate. The change is a route declaration order plus a test; reverting
|
||||
to 1.10.5 restores the previous (broken-panel) behaviour and touches no data.
|
||||
|
||||
---
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.5 (HTTP-01 challenge backend on split deployments)
|
||||
|
||||
**Bug fixes, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are **not**
|
||||
re-seeded. No API-shape change and zero agent impact.
|
||||
|
||||
- **HTTP-01 could fail silently when HAProxy runs on different hosts than the management stack.**
|
||||
The rendered config wrote `server _acme_mgmt <mgmt>:8080` from a value that defaults to
|
||||
loopback — and HAProxy resolves that address **on the HAProxy node**, so it pointed at the wrong
|
||||
box. Every check still reported success. The per-cluster `acme_backend_url` now has a UI field
|
||||
(Cluster Management), changing it actually mints a config version, and the value is validated at
|
||||
the write boundary.
|
||||
- **A config-generation failure could be pushed to agents as the cluster's whole `haproxy.cfg`.**
|
||||
The generator reported failure by *returning* `# Error ...` instead of raising, and the apply
|
||||
path hashed that comment and stored it as an APPLIED version. Both persisting call sites now
|
||||
refuse with 422 and leave the running config in force. **This is worth knowing even if you never
|
||||
touch ACME**, since any exception in the generator could trigger it.
|
||||
- **`frontends.mode` is nullable and was interpolated raw**, emitting a literal `mode None` that
|
||||
HAProxy rejects — which fails the whole cluster config, not just that frontend. Normalised now.
|
||||
- **Cluster creation ignored the ACME fields**: a cluster created with ACME switched on came back
|
||||
switched off, with no error.
|
||||
- **`docker-compose.yml` hardcoded `PUBLIC_URL` / `MANAGEMENT_BASE_URL`**, so a value in your
|
||||
`.env` or host environment was silently ignored. They are interpolated now, with the previous
|
||||
literals as defaults, so behaviour is unchanged unless you actually set them.
|
||||
- **Diagnostics stop over-reporting health.** The port-80 check now reads the body, so a reverse
|
||||
proxy answering 200 with a web page is no longer counted as a working challenge endpoint. Every
|
||||
new condition is a **warning, never a failure** — the Site Wizard blocks submit on a failing
|
||||
check, so a new failing condition would have locked installs on upgrade day.
|
||||
- **Rollback:** downgrade freely. No schema or data change.
|
||||
|
||||
---
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.4 (Adopt an existing keepalived VIP)
|
||||
|
||||
**Additive, but this release DOES bump the schema — read the role warning below.** Nothing on any
|
||||
node changes until you adopt a VIP and apply it.
|
||||
|
||||
- **Schema:** `SCHEMA_VERSION` bumps to `11`, so on first start the (idempotent) migration
|
||||
sequence re-runs once and adds **one new table** (`vip_discoveries`) plus two additive columns
|
||||
(`vip_instances.adopted_at`, `vip_members.takeover_expected_hash`). **No existing table is
|
||||
altered**, no existing row changes, and the admin password is not reset.
|
||||
- **⚠️ Built-in roles are re-seeded to their defaults** — the pre-existing behaviour of every
|
||||
`SCHEMA_VERSION` bump. If you customised `super_admin` / `operator` / `security_admin` /
|
||||
`viewer`, **re-apply those changes after upgrading**. (The three previous releases did not bump
|
||||
the version, so this is the first re-seed since v1.9.0.) No new permission strings are
|
||||
introduced: discovery and adoption are governed by the existing `vip.read` / `vip.create`.
|
||||
- **⚠️ The Linux agent script changed, and discovery does not start until nodes run it.** The
|
||||
fallback latest Linux agent version moves `2.0.0` → `2.1.0`, so nodes will pull the new script
|
||||
through the normal agent-upgrade path. The addition is **read-only**: the agent reads the
|
||||
`keepalived.conf` it does not own and reports it, rate-limited to once per content change. It
|
||||
writes nothing new to the node. Until a node has upgraded, it simply never appears under
|
||||
*Unmanaged keepalived detected*.
|
||||
- **Nothing is taken over implicitly.** The agent still refuses to overwrite a `keepalived.conf`
|
||||
that lacks OpenManager's ownership marker. Adoption authorises exactly **one** takeover of
|
||||
exactly the file that was analysed, pinned to its md5: if the file changes between adoption and
|
||||
Apply, the agent refuses again and reports `externally_managed` rather than clobbering your
|
||||
edit. Re-adopt to pick up the current file.
|
||||
- **Adoption can refuse, on purpose.** It replaces the file with OpenManager's render, so anything
|
||||
the renderer cannot reproduce would be destroyed. Those directives are listed as blockers —
|
||||
`notify_*` failover hooks, `vrrp_sync_group`, LVS `virtual_server` sections, a second address in
|
||||
one instance, a custom `track_script`, extra `global_defs`. You can accept that loss explicitly
|
||||
with a tick, but a value that is *unknown* rather than lost (an absent `virtual_router_id`, or
|
||||
an address with no prefix length) cannot be waived — the VRID is fatal to guess and the prefix
|
||||
has to be supplied, because picking a netmask for a live VIP would change its routing.
|
||||
- **Multi-node VIPs need every node.** Adoption covers the node that reported. Its unicast peers
|
||||
hold their own `keepalived.conf`, so adopt or add them as members before applying — otherwise
|
||||
the render has no peers. The UI says so after a successful adopt.
|
||||
- **Secrets:** the reported config may contain the VRRP `auth_pass`. It is split at ingest — the
|
||||
password is Fernet-encrypted into its own column (same key path as `vip_instances`,
|
||||
`VIP_ENCRYPTION_KEY` falling back to a key derived from `SECRET_KEY`) and the stored copy of the
|
||||
file has it masked, so nothing readable through the API, the UI preview or a DB dump carries it
|
||||
in cleartext.
|
||||
- **Rollback:** downgrading to 1.10.3 leaves `vip_discoveries` as an unused table and the two new
|
||||
columns unread; managed VIPs keep working. One caveat: a VIP adopted on 1.10.4 but **not yet
|
||||
applied** loses its takeover authorisation on downgrade, so the node's original config stays in
|
||||
place and the VIP sits PENDING — harmless, but re-adopt after upgrading again. Agents already on
|
||||
script 2.1.0 keep reporting discoveries to an endpoint that no longer exists; the report fails
|
||||
quietly and nothing on the node is affected.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.3 (Multi-account ACME wizard fix)
|
||||
|
||||
**Frontend only. Nothing to do on upgrade.** No schema, no `SCHEMA_VERSION` bump, no API change, no
|
||||
environment variable, zero agent impact. Installations with a single ACME account behave exactly as
|
||||
before.
|
||||
|
||||
- **What was broken:** with **more than one** ACME account registered, the *Request ACME
|
||||
Certificate* wizard did not honour the account you selected. Choosing an HTTP-01 account still
|
||||
submitted a DNS-01 request, which the API rejected with
|
||||
`The selected ACME account has no DNS provider configured for DNS-01.` The *Review* step also
|
||||
named the default account rather than the chosen one, so the mismatch was invisible before
|
||||
submitting.
|
||||
- **Default account:** the wizard previously previewed the **oldest** valid account while the
|
||||
backend uses the **newest** (`ORDER BY created_at DESC`). If you never picked an account
|
||||
explicitly and have several, requests were already going to the newest one — only the preview was
|
||||
wrong. The wizard now previews that same account, marks it `(default)`, and sends `account_id`
|
||||
explicitly so the two can no longer diverge.
|
||||
- **Wildcard guard:** the client-side "wildcard requires a DNS-01 account" block silently stopped
|
||||
applying on the *Review* step. Requests were still rejected by the backend, so nothing incorrect
|
||||
was ever issued — you now get the warning before submitting instead of an error after.
|
||||
- **No action needed on existing certificates or orders.** Nothing about issuance, renewal or the
|
||||
stored accounts changes; only how the wizard resolves which account a new request uses.
|
||||
- **Rollback:** downgrade freely. This release changes frontend behaviour only.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.2 (Dark mode fixes on Apply Management)
|
||||
|
||||
**Frontend only. Nothing to do on upgrade.** No schema, no `SCHEMA_VERSION` bump, no API change,
|
||||
no environment variable, zero agent impact. Light mode is byte-identical: every colour swapped in
|
||||
this release resolves, under the default algorithm, to exactly the literal it replaced
|
||||
(`colorWarningBg` → `#fffbe6`, `colorSuccessBg` → `#f6ffed`, `colorErrorBg` → `#fff2f0`, …), so
|
||||
only dark mode changes.
|
||||
|
||||
- **Apply Management panels** were painted with light-mode colour literals, so in dark mode the
|
||||
"Pending Changes" box rendered as a cream panel with light text on it. Measured contrast was
|
||||
**1.03:1** — effectively invisible. It is now **11.50:1**. The same class of bug affected the
|
||||
diff rows in *View Change* (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME/pending version
|
||||
panels, the VIP pending-delete row and the agent-error recommendation box.
|
||||
- **Static confirm dialogs came up white in dark mode.** In Ant Design 5 the static
|
||||
`Modal.confirm` / `message` / `notification` APIs render into their own detached root and never
|
||||
see the app's `ConfigProvider`, so they always used the light algorithm. This release registers
|
||||
`ConfigProvider.config({ holderRender })` once at the app root, which fixes **every** static
|
||||
dialog in the app (12 components use them), not just Apply Management.
|
||||
- **Rollback:** downgrade freely. This release changes rendering only.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.1 (CSR private key encrypted at rest)
|
||||
|
||||
**Backward compatible.** Nothing to do on upgrade, and nothing changes for existing clusters,
|
||||
agents or certificates:
|
||||
|
||||
- **Schema:** **no `SCHEMA_VERSION` bump and no migration.** The Fernet token replaces the PEM
|
||||
inside the *existing* `ssl_csrs.private_key_pem` TEXT column. As in v1.10.0, this means the
|
||||
four built-in roles are **not** re-seeded, so any customization of `super_admin` / `operator` /
|
||||
`security_admin` / `viewer` survives.
|
||||
- **Existing pending CSRs keep working.** Rows written before this release hold a raw PEM and are
|
||||
still read transparently, so a CSR that is already out for signature can be imported normally
|
||||
after the upgrade. There is no data migration and no downtime step. Those rows stay plaintext
|
||||
until they are imported (which NULLs the key) — if you want everything encrypted immediately,
|
||||
delete and re-create any long-pending CSRs.
|
||||
- **Scope:** this covers the PENDING CSR key only. It is the one key in the system that sits idle
|
||||
for the whole signing window and is never transmitted. `ssl_certificates.private_key_content`
|
||||
and the ACME order keys are unchanged, because agents must receive those in plaintext on every
|
||||
poll.
|
||||
- **Optional env:** `CSR_ENCRYPTION_KEY` (see `.env.template`). If unset, the key is derived from
|
||||
`SECRET_KEY` via HKDF with its own info string, so it is independent of the VIP, MFA and DNS
|
||||
provider keys.
|
||||
- **⚠️ Rotating `SECRET_KEY` while `CSR_ENCRYPTION_KEY` is unset makes pending CSR keys
|
||||
unrecoverable.** Import then fails with an explicit "delete this CSR and create a new one"
|
||||
error rather than a misleading key-mismatch. Set an explicit `CSR_ENCRYPTION_KEY` if you
|
||||
rotate `SECRET_KEY`. Certificates already imported are unaffected — their key lives on the
|
||||
certificate row.
|
||||
- **API / UI / agents:** unchanged. No CSR endpoint ever returned the private key before or now,
|
||||
and nothing about the CSR tab changes.
|
||||
- **Rollback:** the application downgrades cleanly — 1.10.0 starts normally against the same
|
||||
database and every other feature is unaffected. The one casualty is a CSR **created on 1.10.1
|
||||
and still pending**: 1.10.0 has no decrypt step, so it hands the Fernet token straight to the
|
||||
key-pairing check. Measured on a real downgrade, the import then fails with
|
||||
`HTTP 500 — Could not verify the certificate/key pair: key parse failed (encrypted?)`; it does
|
||||
**not** silently pair the wrong key, and it does not corrupt anything. Import or delete CSRs
|
||||
created on 1.10.1 before downgrading. Certificates already imported are unaffected, since their
|
||||
key lives on the certificate row, and CSRs created before 1.10.1 are plaintext and still work.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.0 (GoDaddy DNS-01 provider)
|
||||
|
||||
**Backward compatible & additive.** Nothing changes unless you select **GoDaddy** as an ACME
|
||||
account's DNS provider:
|
||||
|
||||
- **Schema:** **no `SCHEMA_VERSION` bump.** The GoDaddy credentials (API Key + Secret) are stored
|
||||
as two keys inside the *existing* encrypted
|
||||
`letsencrypt_account_dns_credentials.credentials_encrypted` blob — no new table, no new column,
|
||||
no migration.
|
||||
- **✅ Built-in roles are NOT re-seeded.** The re-seed warning in the v1.9.0 notes below is
|
||||
triggered by a `SCHEMA_VERSION` bump. This release does not bump it, so any customization you
|
||||
made to `super_admin` / `operator` / `security_admin` / `viewer` survives untouched.
|
||||
- **Permissions / API shape:** unchanged. `GET /api/letsencrypt/dns-providers` simply returns one
|
||||
extra entry in its `providers` array; every request and response shape is identical, and the
|
||||
credential form is rendered from that schema, so there is no frontend behaviour change either.
|
||||
- **Environment:** no new variable. GoDaddy credentials use the same Fernet-at-rest path as
|
||||
Cloudflare (`DNS_PROVIDER_ENCRYPTION_KEY`, falling back to a key derived from `SECRET_KEY`).
|
||||
- **Agents:** zero agent changes. DNS-01 is invisible to agents; an issued certificate follows the
|
||||
normal PENDING → Apply Management → agent pull pipeline exactly as before.
|
||||
- **Using it:** the API Key must be a **Production** key from `developer.godaddy.com/keys` (the
|
||||
first key that dashboard issues is an OTE/test key and is rejected), the zone must be in the same
|
||||
GoDaddy account, and that account needs at least one registered domain before GoDaddy permits DNS
|
||||
API access. A Personal Access Token also works — paste it as the Key and leave the Secret blank.
|
||||
Credentials are checked against the GoDaddy API before they are stored, so an invalid, OTE or
|
||||
ineligible key fails at save time. Note the check is a **read**: a Personal Access Token that has
|
||||
`domains.domain:read` but not `domains.dns:update` saves successfully and only fails at the first
|
||||
publish, with a 403 in the order timeline.
|
||||
- **Rollback:** don't select GoDaddy. Existing Manual and Cloudflare accounts and all HTTP-01
|
||||
issuance are untouched. **Downgrading after adopting GoDaddy is not a no-op**: on 1.9.0
|
||||
`godaddy` is not a known provider, so any account still set to it degrades to the manual-confirm
|
||||
path (in-flight DNS-01 orders wait for a confirmation nobody can give and expire after 48h, and
|
||||
renewals stop), and the cleanup sweep marks published TXT records cleaned without removing them.
|
||||
Before downgrading, switch affected accounts back to Manual or Cloudflare and let the reconcile
|
||||
sweep remove outstanding `_acme-challenge` records first. The stored credential row itself is
|
||||
inert — an encrypted blob for an unknown provider.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.9.0 (CSR creation)
|
||||
|
||||
**Backward compatible & additive.** Upgrading to v1.9.0 changes nothing for existing
|
||||
|
||||
@@ -1758,7 +1758,13 @@ async def ensure_agent_activity_logs_table():
|
||||
# until the operator imports the CA-signed certificate; the import creates a
|
||||
# normal ssl_certificates row and NULLs the key copy here. Additive + idempotent;
|
||||
# no existing table is altered, agents never read this table.
|
||||
SCHEMA_VERSION = 10
|
||||
# v1.10.4 (VIP adoption): bumped 10 -> 11 for the new `vip_discoveries` table plus two
|
||||
# additive columns (`vip_instances.adopted_at`, `vip_members.takeover_expected_hash`).
|
||||
# Holds the keepalived.conf an agent found already on a node so an existing VIP can be
|
||||
# adopted instead of retyped. Additive + idempotent; no existing table is altered and no
|
||||
# existing row changes. NOTE for the upgrade notes: a SCHEMA_VERSION bump re-seeds the four
|
||||
# built-in roles to their defaults, so role customizations are lost on this upgrade.
|
||||
SCHEMA_VERSION = 11
|
||||
|
||||
|
||||
async def run_all_migrations():
|
||||
@@ -2161,6 +2167,45 @@ async def ensure_vip_tables():
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_members_agent ON vip_members(agent_id);"
|
||||
)
|
||||
|
||||
# ── v1.10.4 — VIP adoption: what the agent found already on the node ──────────
|
||||
# A node with a hand-maintained keepalived.conf reports it here so an existing VIP can
|
||||
# be adopted instead of retyped. One row per agent (the file is per-node); the agent
|
||||
# only reports a config it does NOT own, and only when the content changed.
|
||||
#
|
||||
# SECRETS: `raw_config` is stored MASKED (auth_pass replaced) because it is served to
|
||||
# the UI. The real VRRP password is Fernet-encrypted in auth_pass_encrypted, mirroring
|
||||
# vip_instances, so adoption can carry it into the managed VIP without it ever being
|
||||
# readable through the API or a DB dump. `analysis` is the parser output with auth_pass
|
||||
# stripped out.
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS vip_discoveries (
|
||||
id SERIAL PRIMARY KEY,
|
||||
agent_id INTEGER NOT NULL REFERENCES agents(id) ON DELETE CASCADE,
|
||||
config_path VARCHAR(500) NOT NULL,
|
||||
config_hash VARCHAR(64) NOT NULL,
|
||||
is_managed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
raw_config_masked TEXT,
|
||||
auth_pass_encrypted TEXT,
|
||||
analysis JSONB,
|
||||
parse_error TEXT,
|
||||
adopted_vip_id INTEGER REFERENCES vip_instances(id) ON DELETE SET NULL,
|
||||
reported_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT vip_discovery_agent_unique UNIQUE (agent_id)
|
||||
);
|
||||
""")
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_discoveries_agent ON vip_discoveries(agent_id);"
|
||||
)
|
||||
# Adoption provenance + the one-shot takeover authorisation. The agent refuses to
|
||||
# overwrite a keepalived.conf that lacks our ownership marker, which is exactly the
|
||||
# guard adoption has to pass. Rather than weaken it, an adopted VIP carries the hash of
|
||||
# the file we analysed: the agent takes over ONLY if the file on disk still hashes to
|
||||
# that value, so a config that changed after adoption is never clobbered.
|
||||
await conn.execute(
|
||||
"ALTER TABLE vip_instances ADD COLUMN IF NOT EXISTS adopted_at TIMESTAMP;")
|
||||
await conn.execute(
|
||||
"ALTER TABLE vip_members ADD COLUMN IF NOT EXISTS takeover_expected_hash VARCHAR(64);")
|
||||
|
||||
logger.info("✅ VIP tables ensured (Issue #27 — HA/VIP Keepalived management)")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to ensure VIP tables: {e}")
|
||||
|
||||
+12
-2
@@ -1111,9 +1111,19 @@ async def get_version():
|
||||
return _version_info
|
||||
|
||||
@app.get("/.well-known/acme-challenge/{token}")
|
||||
async def serve_acme_challenge(token: str):
|
||||
async def serve_acme_challenge(token: str, request: Request):
|
||||
"""Serve ACME HTTP-01 challenge token. Public endpoint, no auth required."""
|
||||
logger.info(f"ACME-CHALLENGE: Incoming request for token={token[:32]}...")
|
||||
# Log who reached us. When HTTP-01 fails, the first question is always "did the
|
||||
# request get here at all?" — and the answer separates a broken challenge-backend
|
||||
# address (nothing arrives) from a wrong response (arrives, wrong body). The peer
|
||||
# is normally the HAProxy node; X-Forwarded-For carries the CA when the frontend
|
||||
# sets `option forwardfor`.
|
||||
_peer = request.client.host if request.client else 'unknown'
|
||||
_xff = request.headers.get('x-forwarded-for') or '-'
|
||||
logger.info(
|
||||
f"ACME-CHALLENGE: Incoming request for token={token[:32]}... "
|
||||
f"peer={_peer} xff={_xff} host={request.headers.get('host') or '-'}"
|
||||
)
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
@@ -1,6 +1,22 @@
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, field_validator
|
||||
from typing import Optional, List
|
||||
|
||||
from utils.acme_backend_url import AcmeBackendUrlError, validate_acme_backend_url
|
||||
|
||||
|
||||
def _validated_acme_backend_url(value: Optional[str]) -> Optional[str]:
|
||||
"""Shared field validator body for `acme_backend_url`.
|
||||
|
||||
Pydantic turns the raised ValueError into a 422 with this message attached, so
|
||||
the operator sees why the value was refused instead of discovering months later
|
||||
that HTTP-01 never worked. Returns the normalised value — callers persist THIS,
|
||||
not the raw input, so surrounding whitespace never reaches haproxy.cfg.
|
||||
"""
|
||||
try:
|
||||
return validate_acme_backend_url(value)
|
||||
except AcmeBackendUrlError as exc:
|
||||
raise ValueError(str(exc)) from None
|
||||
|
||||
class HAProxyClusterCreate(BaseModel):
|
||||
name: str
|
||||
description: Optional[str] = None
|
||||
@@ -10,6 +26,16 @@ class HAProxyClusterCreate(BaseModel):
|
||||
haproxy_bin_path: str = "/usr/sbin/haproxy" # HAProxy binary path
|
||||
keepalived_config_path: str = "/etc/keepalived/keepalived.conf" # HA/VIP: keepalived.conf path (Issue #27)
|
||||
pool_id: Optional[int] = None # Which pool this cluster belongs to
|
||||
# The create form submits both of these. Until they were declared here pydantic
|
||||
# dropped them and the INSERT never carried them, so a cluster created with ACME
|
||||
# switched on came back switched off with no error shown — the same silent-success
|
||||
# failure this work exists to remove.
|
||||
acme_enabled: Optional[bool] = None
|
||||
acme_backend_url: Optional[str] = None
|
||||
|
||||
_validate_acme_backend_url = field_validator("acme_backend_url")(
|
||||
_validated_acme_backend_url
|
||||
)
|
||||
|
||||
class HAProxyClusterUpdate(BaseModel):
|
||||
name: Optional[str] = None
|
||||
@@ -24,6 +50,10 @@ class HAProxyClusterUpdate(BaseModel):
|
||||
acme_enabled: Optional[bool] = None
|
||||
acme_backend_url: Optional[str] = None
|
||||
|
||||
_validate_acme_backend_url = field_validator("acme_backend_url")(
|
||||
_validated_acme_backend_url
|
||||
)
|
||||
|
||||
class HAProxyClusterResponse(BaseModel):
|
||||
id: int
|
||||
name: str
|
||||
|
||||
+138
-8
@@ -9,8 +9,15 @@ import os
|
||||
import json
|
||||
import ipaddress
|
||||
import hashlib
|
||||
import re
|
||||
# Pipeline trigger - force backend redeploy v2
|
||||
|
||||
# v1.10.4 — a discovered keepalived.conf is stored and served to the UI, so the VRRP password is
|
||||
# masked out of the stored copy (the real value lives Fernet-encrypted in its own column). Mask
|
||||
# the WHOLE remainder of the line, mirroring vip.py's version-diff masking, so a password
|
||||
# containing whitespace cannot partially leak.
|
||||
_AUTH_PASS_MASK_RE = re.compile(r"(auth_pass\s+).*")
|
||||
|
||||
from models import AgentCreate
|
||||
from models.agent import AgentToggle, AgentHeartbeat, AgentScriptRequest, AgentUpgradeRequest
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
@@ -26,7 +33,7 @@ logger = logging.getLogger(__name__)
|
||||
# Global version storage (acts as in-memory database)
|
||||
AGENT_VERSIONS = {
|
||||
"macos": "2.1.0", # Updated via endpoint
|
||||
"linux": "2.0.0"
|
||||
"linux": "2.1.0"
|
||||
}
|
||||
|
||||
|
||||
@@ -2332,7 +2339,7 @@ async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str]
|
||||
row = await conn.fetchrow("""
|
||||
SELECT v.id AS vip_id, v.name AS vip_name, v.is_active, v.track_haproxy,
|
||||
v.purge_on_teardown,
|
||||
m.applied_config_content, m.applied_config_hash
|
||||
m.applied_config_content, m.applied_config_hash, m.takeover_expected_hash
|
||||
FROM vip_members m JOIN vip_instances v ON v.id = m.vip_id
|
||||
WHERE m.agent_id = $1
|
||||
-- Active VIP first (an agent has at most one). With NO active VIP, pick the most
|
||||
@@ -2367,6 +2374,14 @@ async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str]
|
||||
"config_content": row['applied_config_content'],
|
||||
"config_hash": row['applied_config_hash'],
|
||||
"check_script": check_script,
|
||||
# v1.10.4 adoption handoff. The agent refuses to overwrite a keepalived.conf
|
||||
# without our ownership marker — the guard that protects a hand-maintained
|
||||
# setup. Adoption does not weaken it: it authorises exactly ONE takeover, of
|
||||
# exactly the file we analysed, by pinning its hash. If the file changed since
|
||||
# adoption the hashes differ and the agent keeps refusing, so an edit made
|
||||
# between adoption and Apply can never be silently overwritten.
|
||||
"allow_takeover": bool(row['takeover_expected_hash']),
|
||||
"takeover_expected_hash": row['takeover_expected_hash'],
|
||||
},
|
||||
}
|
||||
except HTTPException:
|
||||
@@ -2408,19 +2423,40 @@ async def agent_keepalived_status(agent_name: str, status_data: dict, x_api_key:
|
||||
state = (status_data.get("state") or "").strip()[:24]
|
||||
config_hash = (status_data.get("config_hash") or "")[:64]
|
||||
message = status_data.get("message")
|
||||
# v1.10.9 — retire the adoption takeover authorisation once the node CONFIRMS it is
|
||||
# running our rendered config. `takeover_expected_hash` is the permission to overwrite a
|
||||
# keepalived.conf that lacks our ownership marker; it was written at adoption and never
|
||||
# cleared, so it stayed valid indefinitely and "one-shot" was only true in the sense of
|
||||
# "for exactly that file content". Clearing it the moment the member acks OUR hash makes
|
||||
# the claim real: if the file is replaced by hand afterwards the agent refuses and reports
|
||||
# "externally managed", which is the visible behaviour an operator should get.
|
||||
#
|
||||
# Gated on the acked hash MATCHING applied_config_hash, so a partial or failed deploy
|
||||
# never drops the authorisation and leaves the VIP unable to converge.
|
||||
# The hash is passed TWICE on purpose. Reusing one placeholder for both the assignment
|
||||
# (`last_deploy_hash=$n`, a VARCHAR column) and the comparison inside the CASE made
|
||||
# PostgreSQL deduce two different types for it and asyncpg refused the whole statement
|
||||
# with AmbiguousParameterError ("text versus character varying"). Because the failure is
|
||||
# in the UPDATE itself, not in one column, EVERY status ack was lost and every VIP sat at
|
||||
# SYNCING forever — including teardown acks. A separate placeholder is only ever compared
|
||||
# against the column, so its type is unambiguous.
|
||||
_retire_takeover = ("takeover_expected_hash = CASE WHEN applied_config_hash IS NOT NULL "
|
||||
"AND applied_config_hash = {p} THEN NULL ELSE takeover_expected_hash END")
|
||||
if vip_id is None:
|
||||
# No specific VIP (e.g. a teardown ack) — update all this agent's memberships.
|
||||
await conn.execute("""
|
||||
await conn.execute(f"""
|
||||
UPDATE vip_members SET last_deploy_state=$2, last_deploy_message=$3,
|
||||
last_deploy_hash=$4, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
last_deploy_hash=$4, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP,
|
||||
{_retire_takeover.format(p="$5")}
|
||||
WHERE agent_id=$1
|
||||
""", agent['id'], state, message, config_hash)
|
||||
""", agent['id'], state, message, config_hash, config_hash)
|
||||
else:
|
||||
await conn.execute("""
|
||||
await conn.execute(f"""
|
||||
UPDATE vip_members SET last_deploy_state=$3, last_deploy_message=$4,
|
||||
last_deploy_hash=$5, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
last_deploy_hash=$5, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP,
|
||||
{_retire_takeover.format(p="$6")}
|
||||
WHERE agent_id=$1 AND vip_id=$2
|
||||
""", agent['id'], int(vip_id), state, message, config_hash)
|
||||
""", agent['id'], int(vip_id), state, message, config_hash, config_hash)
|
||||
return {"status": "ok"}
|
||||
except HTTPException:
|
||||
raise
|
||||
@@ -2431,6 +2467,100 @@ async def agent_keepalived_status(agent_name: str, status_data: dict, x_api_key:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/{agent_name}/keepalived-discovery")
|
||||
async def agent_keepalived_discovery(agent_name: str, payload: dict, x_api_key: Optional[str] = Header(None)):
|
||||
"""v1.10.4 — the agent reports a keepalived.conf it found on the node but does NOT own.
|
||||
|
||||
This is what makes adopting a hand-maintained VIP possible: the heartbeat only carries the
|
||||
VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs eleven
|
||||
fields, so the file itself has to be read. Read-only on the agent side — reporting never
|
||||
changes anything on the node.
|
||||
|
||||
Auth mirrors /keepalived-status: a MISSING key is rejected outright, and because the token
|
||||
is a shared install token a name mismatch is an advisory audit log rather than a 403.
|
||||
|
||||
SECRETS: the reported content may contain the VRRP `auth_pass`. It is split immediately —
|
||||
the password is Fernet-encrypted into its own column and the stored copy of the file has it
|
||||
masked, so nothing readable through the API or a DB dump carries it in cleartext. The
|
||||
parse result is never logged.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not x_api_key or not agent_auth:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
if agent_auth['name'] != agent_name:
|
||||
logger.info(f"Agent '{agent_name}' reporting keepalived discovery using API key "
|
||||
f"from agent '{agent_auth['name']}'")
|
||||
|
||||
conn = await get_database_connection()
|
||||
agent = await conn.fetchrow("SELECT id FROM agents WHERE name = $1", agent_name)
|
||||
if not agent:
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
|
||||
config_path = (payload.get("config_path") or "/etc/keepalived/keepalived.conf")[:500]
|
||||
exists = bool(payload.get("exists"))
|
||||
if not exists:
|
||||
# The file is gone (keepalived removed, or we adopted and now own it) — drop the row
|
||||
# so the UI stops offering a stale candidate.
|
||||
await conn.execute("DELETE FROM vip_discoveries WHERE agent_id = $1", agent['id'])
|
||||
return {"status": "cleared"}
|
||||
|
||||
content = payload.get("config_content") or ""
|
||||
if len(content) > 256_000:
|
||||
raise HTTPException(status_code=413, detail="keepalived.conf too large to analyse")
|
||||
is_managed = bool(payload.get("is_managed"))
|
||||
config_hash = hashlib.md5(content.encode("utf-8", "replace")).hexdigest()
|
||||
|
||||
from services.keepalived_parser import analyse_keepalived_conf, KeepalivedParseError
|
||||
from services.keepalived_config import encrypt_vrrp_secret
|
||||
|
||||
parse_error = None
|
||||
analysis = None
|
||||
auth_enc = None
|
||||
try:
|
||||
analysis = analyse_keepalived_conf(content)
|
||||
# Split the secret out of everything we persist or serve.
|
||||
for cand in analysis.get("candidates", []):
|
||||
secret = (cand.get("vip") or {}).pop("auth_pass", None)
|
||||
cand["vip"]["has_auth_pass"] = bool(secret)
|
||||
if secret and auth_enc is None:
|
||||
auth_enc = encrypt_vrrp_secret(secret)
|
||||
except KeepalivedParseError as exc:
|
||||
parse_error = str(exc)[:500]
|
||||
except Exception as exc: # noqa: BLE001 — a malformed file must not 500 the agent loop
|
||||
parse_error = f"could not analyse the config ({type(exc).__name__})"
|
||||
|
||||
masked = _AUTH_PASS_MASK_RE.sub(r"\1********", content)
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_discoveries
|
||||
(agent_id, config_path, config_hash, is_managed, raw_config_masked,
|
||||
auth_pass_encrypted, analysis, parse_error, reported_at)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,CURRENT_TIMESTAMP)
|
||||
ON CONFLICT (agent_id) DO UPDATE SET
|
||||
config_path = EXCLUDED.config_path,
|
||||
config_hash = EXCLUDED.config_hash,
|
||||
is_managed = EXCLUDED.is_managed,
|
||||
raw_config_masked = EXCLUDED.raw_config_masked,
|
||||
auth_pass_encrypted = EXCLUDED.auth_pass_encrypted,
|
||||
analysis = EXCLUDED.analysis,
|
||||
parse_error = EXCLUDED.parse_error,
|
||||
reported_at = CURRENT_TIMESTAMP
|
||||
""", agent['id'], config_path, config_hash, is_managed, masked, auth_enc,
|
||||
json.dumps(analysis) if analysis is not None else None, parse_error)
|
||||
return {"status": "recorded", "config_hash": config_hash}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"keepalived-discovery failed for '{agent_name}': {e}")
|
||||
raise HTTPException(status_code=500, detail="keepalived-discovery failed")
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/script-version")
|
||||
async def get_latest_script_version(platform: str = "macos"):
|
||||
"""Get the latest available agent script version for specified platform"""
|
||||
|
||||
+134
-12
@@ -111,6 +111,14 @@ router = APIRouter(prefix="/api/clusters", tags=["clusters"])
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _AcmeNoConfigChange(Exception):
|
||||
"""Internal signal: an ACME edit renders the same config, so mint nothing.
|
||||
|
||||
Control flow, not an error — it unwinds out of the version-minting block without
|
||||
tripping the generic `except Exception` handler that would log it as a failure.
|
||||
"""
|
||||
|
||||
|
||||
class _ConcurrentlyDrained(Exception):
|
||||
"""Sentinel raised inside ``apply_pending_changes`` when the
|
||||
advisory-lock-protected re-fetch shows that another caller
|
||||
@@ -294,12 +302,14 @@ async def create_cluster(cluster: HAProxyClusterCreate, authorization: str = Hea
|
||||
cluster_id = await conn.fetchval("""
|
||||
INSERT INTO haproxy_clusters (name, description, connection_type, is_active,
|
||||
stats_socket_path, haproxy_config_path, haproxy_bin_path,
|
||||
keepalived_config_path, pool_id)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7, $8)
|
||||
keepalived_config_path, pool_id,
|
||||
acme_enabled, acme_backend_url)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7, $8, COALESCE($9, FALSE), $10)
|
||||
RETURNING id
|
||||
""", cluster.name, cluster.description, cluster.connection_type,
|
||||
cluster.stats_socket_path, cluster.haproxy_config_path, cluster.haproxy_bin_path,
|
||||
cluster.keepalived_config_path, cluster.pool_id)
|
||||
cluster.keepalived_config_path, cluster.pool_id,
|
||||
cluster.acme_enabled, cluster.acme_backend_url)
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -389,7 +399,8 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
# Check if cluster exists and get current values
|
||||
existing_cluster = await conn.fetchrow("""
|
||||
SELECT name, description, connection_type, is_active, stats_socket_path,
|
||||
haproxy_config_path, haproxy_bin_path, pool_id, acme_enabled
|
||||
haproxy_config_path, haproxy_bin_path, pool_id, acme_enabled,
|
||||
acme_backend_url
|
||||
FROM haproxy_clusters WHERE id = $1
|
||||
""", cluster_id)
|
||||
if not existing_cluster:
|
||||
@@ -453,7 +464,13 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
update_fields.append(f"acme_enabled = ${param_counter}")
|
||||
update_values.append(cluster.acme_enabled)
|
||||
param_counter += 1
|
||||
if cluster.acme_backend_url is not None:
|
||||
# Keyed on "was the field submitted?", not "is it non-None". With a plain
|
||||
# `is not None` test there is no way to CLEAR the value: the validator maps an
|
||||
# empty box to None, which is indistinguishable from "not supplied", so once an
|
||||
# operator set a per-cluster URL they could never revert to the global setting —
|
||||
# the field would accept the edit and silently keep the old value.
|
||||
_acme_url_submitted = 'acme_backend_url' in cluster.model_fields_set
|
||||
if _acme_url_submitted:
|
||||
update_fields.append(f"acme_backend_url = ${param_counter}")
|
||||
update_values.append(cluster.acme_backend_url)
|
||||
param_counter += 1
|
||||
@@ -463,14 +480,75 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
update_query = f"UPDATE haproxy_clusters SET {', '.join(update_fields)} WHERE id = $1"
|
||||
await conn.execute(update_query, *update_values)
|
||||
|
||||
# If acme_enabled actually changed, create a PENDING config version with entity snapshot
|
||||
if cluster.acme_enabled is not None and cluster.acme_enabled != existing_cluster.get('acme_enabled', False):
|
||||
# Create a PENDING config version when an ACME edit would change what the
|
||||
# HAProxy nodes actually run.
|
||||
#
|
||||
# This used to trigger only on `acme_enabled` flipping. `acme_backend_url` is
|
||||
# written to the DB a few lines above but minted nothing, so correcting a wrong
|
||||
# challenge backend from the panel was a silent no-op: the value changed, no
|
||||
# pending version existed, Apply answered "No pending changes to apply", and the
|
||||
# nodes kept the old address indefinitely. That made the one field an operator
|
||||
# needs to fix HTTP-01 impossible to actually apply.
|
||||
_acme_toggled = (
|
||||
cluster.acme_enabled is not None
|
||||
and cluster.acme_enabled != existing_cluster.get('acme_enabled', False)
|
||||
)
|
||||
if _acme_toggled or _acme_url_submitted:
|
||||
try:
|
||||
from services.haproxy_config import generate_haproxy_config_for_cluster
|
||||
from services.haproxy_config import (
|
||||
generate_haproxy_config_for_cluster,
|
||||
is_config_generation_error,
|
||||
extract_acme_backend_target,
|
||||
)
|
||||
config_content = await generate_haproxy_config_for_cluster(cluster_id)
|
||||
if is_config_generation_error(config_content):
|
||||
# Same sentinel-instead-of-exception contract as the apply path. A
|
||||
# PENDING version holding the sentinel is a landmine: the operator
|
||||
# sees a pending change and applies it, replacing the whole config.
|
||||
logger.error(
|
||||
f"ACME TOGGLE: config generation for cluster {cluster_id} returned an "
|
||||
f"error sentinel; no PENDING version created: {config_content!r}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=(
|
||||
"ACME setting was saved, but a configuration could not be generated "
|
||||
"for this cluster, so no pending change was created. "
|
||||
f"Generator reported: {config_content.strip()[:300]}"
|
||||
),
|
||||
)
|
||||
# A URL edit that renders the same `server _acme_mgmt` line changes
|
||||
# nothing on the nodes, so minting a version would put a no-op pending
|
||||
# change in front of the operator. Compare that one line rather than the
|
||||
# whole config: the generator also renders unrelated PENDING entities,
|
||||
# so a full-text diff reports a change on every edit anyone has queued.
|
||||
_active_config = await conn.fetchval("""
|
||||
SELECT config_content FROM config_versions
|
||||
WHERE cluster_id = $1 AND is_active = TRUE
|
||||
ORDER BY created_at DESC LIMIT 1
|
||||
""", cluster_id)
|
||||
_new_target = extract_acme_backend_target(config_content)
|
||||
_old_target = extract_acme_backend_target(_active_config)
|
||||
if not _acme_toggled and _new_target == _old_target:
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} URL updated but the rendered "
|
||||
f"challenge backend is unchanged ({_new_target!r}); no config "
|
||||
f"version created."
|
||||
)
|
||||
raise _AcmeNoConfigChange()
|
||||
|
||||
import time as _time
|
||||
import json as _json
|
||||
version_name = f"cluster-{cluster_id}-acme-{'enable' if cluster.acme_enabled else 'disable'}-{int(_time.time())}"
|
||||
if _acme_toggled:
|
||||
_acme_kind = 'enable' if cluster.acme_enabled else 'disable'
|
||||
else:
|
||||
_acme_kind = 'backend'
|
||||
version_name = f"cluster-{cluster_id}-acme-{_acme_kind}-{int(_time.time())}"
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} pending config version "
|
||||
f"'{version_name}' created — challenge backend {_old_target!r} -> "
|
||||
f"{_new_target!r}. Apply the cluster for the nodes to pick it up."
|
||||
)
|
||||
|
||||
from utils.entity_snapshot import save_entity_snapshot
|
||||
snapshot_metadata = await save_entity_snapshot(
|
||||
@@ -482,7 +560,14 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
"acme_backend_url": existing_cluster.get('acme_backend_url'),
|
||||
},
|
||||
new_values={
|
||||
"acme_enabled": cluster.acme_enabled,
|
||||
# `acme_enabled` is None when only the URL was submitted; the
|
||||
# snapshot must record the value that is actually in force, or a
|
||||
# rollback would write NULL over a working flag.
|
||||
"acme_enabled": (
|
||||
cluster.acme_enabled
|
||||
if cluster.acme_enabled is not None
|
||||
else existing_cluster.get('acme_enabled', False)
|
||||
),
|
||||
"acme_backend_url": getattr(cluster, 'acme_backend_url', None) or existing_cluster.get('acme_backend_url'),
|
||||
},
|
||||
operation="UPDATE"
|
||||
@@ -497,9 +582,20 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
""", cluster_id, version_name, config_content, current_user.get('id', 1), metadata_json)
|
||||
finally:
|
||||
await close_database_connection(conn2)
|
||||
except _AcmeNoConfigChange:
|
||||
# Not an error: the edit was accepted and simply renders the same
|
||||
# address, so there is nothing for the operator to apply.
|
||||
pass
|
||||
except HTTPException:
|
||||
# The sentinel guard above deliberately fails the request. Without this
|
||||
# clause the generic handler below would swallow it and report success
|
||||
# while no PENDING version exists — the exact silent-success failure
|
||||
# mode this change exists to remove.
|
||||
await close_database_connection(conn)
|
||||
raise
|
||||
except Exception as acme_err:
|
||||
logger.error(f"Failed to create ACME config version for cluster {cluster_id}: {acme_err}")
|
||||
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
# Log activity
|
||||
@@ -1914,6 +2010,28 @@ defaults
|
||||
logger.info(f"🧩 APPLY: Generating fresh configuration from database for cluster {cluster_id}")
|
||||
fresh_config_content = await generate_haproxy_config_for_cluster(cluster_id, conn)
|
||||
|
||||
# `generate_haproxy_config_for_cluster` reports failure by RETURNING a
|
||||
# one-line comment instead of raising (haproxy_config.py outer `except`).
|
||||
# Without this guard that sentinel is hashed, stored as an APPLIED version
|
||||
# and shipped to every agent — silently replacing the cluster's entire
|
||||
# configuration with a comment. Any generator exception (an out-of-range
|
||||
# port in acme_backend_url is enough) triggers it. Refuse the apply instead;
|
||||
# the previous APPLIED version stays in force.
|
||||
from services.haproxy_config import is_config_generation_error
|
||||
if is_config_generation_error(fresh_config_content):
|
||||
logger.error(
|
||||
f"APPLY ABORTED: config generation for cluster {cluster_id} returned an "
|
||||
f"error sentinel instead of a configuration: {fresh_config_content!r}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=(
|
||||
"Configuration could not be generated for this cluster, so nothing "
|
||||
"was applied and the running configuration is unchanged. "
|
||||
f"Generator reported: {fresh_config_content.strip()[:300]}"
|
||||
),
|
||||
)
|
||||
|
||||
# Create a new consolidated config version with fresh content
|
||||
import hashlib
|
||||
import time
|
||||
@@ -2586,7 +2704,11 @@ async def get_config_version_diff(cluster_id: int, version_id: int, authorizatio
|
||||
# HA/VIP (Issue #27): vip-{id}-{action} versions show the generated keepalived.conf
|
||||
# each member node will deploy as the change content (VRRP secret masked). Mirrors
|
||||
# the ssl-* special case above so VIP uses the STANDARD View Change diff modal.
|
||||
vip_match = re.search(r'vip-(\d+)-(create|update|delete)', current_version['version_name'])
|
||||
# `adopt` MUST stay in this alternation. A vip-* action missing here does not degrade
|
||||
# gracefully: the version falls through to the generic HAProxy diff, which compares this
|
||||
# row's keepalived.conf against the cluster's previous haproxy.cfg and shows the whole
|
||||
# HAProxy config as removed. v1.10.4 added `adopt` without it (fixed in v1.10.8).
|
||||
vip_match = re.search(r'vip-(\d+)-(create|update|delete|adopt)', current_version['version_name'])
|
||||
if vip_match:
|
||||
vip_id = int(vip_match.group(1))
|
||||
vip_action = vip_match.group(2)
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
from fastapi import APIRouter, HTTPException, Header
|
||||
from pydantic import BaseModel
|
||||
from typing import Dict, Any
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.acme_backend_url import AcmeBackendUrlError, validate_acme_backend_url
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -50,6 +52,39 @@ async def get_settings_by_category(category: str, authorization: str = Header(No
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
def _validate_acme_challenge_backend_url(value):
|
||||
"""Validate `acme.challenge_backend_url` exactly as the config renderer reads it.
|
||||
|
||||
Settings values are stored as jsonb, so the renderer json.loads them before use
|
||||
(services/haproxy_config.py). Validating the raw column text instead of the
|
||||
decoded string would check the quoting rather than the URL.
|
||||
"""
|
||||
decoded = value
|
||||
if isinstance(decoded, str):
|
||||
try:
|
||||
decoded = json.loads(decoded)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if decoded is None:
|
||||
return
|
||||
try:
|
||||
validate_acme_backend_url(str(decoded))
|
||||
except AcmeBackendUrlError as exc:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=f"acme.challenge_backend_url: {exc}",
|
||||
) from None
|
||||
|
||||
|
||||
# Per-key validators for settings that end up in generated configuration or in
|
||||
# outbound requests. Everything else is still written through unchecked; this is a
|
||||
# deliberate allow-list of the keys where a bad value causes silent breakage rather
|
||||
# than an obvious one.
|
||||
_SETTING_VALIDATORS = {
|
||||
"acme.challenge_backend_url": _validate_acme_challenge_backend_url,
|
||||
}
|
||||
|
||||
|
||||
@router.put("/{category}")
|
||||
async def update_settings_by_category(
|
||||
category: str,
|
||||
@@ -59,6 +94,13 @@ async def update_settings_by_category(
|
||||
current_user = await _get_admin_user(authorization)
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
# Validate the whole batch before writing any of it, so a rejected key cannot
|
||||
# leave the category half-applied.
|
||||
for key_suffix, value in body.settings.items():
|
||||
validator = _SETTING_VALIDATORS.get(f"{category}.{key_suffix}")
|
||||
if validator is not None:
|
||||
validator(value)
|
||||
|
||||
updated = []
|
||||
for key_suffix, value in body.settings.items():
|
||||
full_key = f"{category}.{key_suffix}"
|
||||
|
||||
@@ -458,6 +458,46 @@ async def list_vips(cluster_id: Optional[int] = None, authorization: str = Heade
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# ORDER MATTERS: every literal path under this router MUST be declared before the
|
||||
# `/{vip_id}` routes below. FastAPI matches in declaration order, so a literal placed after
|
||||
# `/{vip_id}` is swallowed by it and answered with 422 ("discoveries" is not an int) — the
|
||||
# handler never runs. That is not a visible failure either: the HA/VIP page treats any
|
||||
# non-OK response as "nothing to show", so the whole adoption feature silently disappears.
|
||||
# See the v1.10.4 adoption section further down for the endpoint's own documentation.
|
||||
@router.get("/discoveries")
|
||||
async def list_vip_discoveries(cluster_id: Optional[int] = None, authorization: str = Header(None)):
|
||||
"""Unmanaged keepalived configs the agents found on their nodes.
|
||||
|
||||
Read-only and safe to poll: this is what the HA/VIP page shows so an existing VIP is
|
||||
visible before anyone adopts it.
|
||||
|
||||
`cluster_id` scopes the result to the agents in that cluster's pool, resolved exactly like
|
||||
the VIP list above. Without it every discovery in the fleet is returned, which is what a
|
||||
caller that does not know about the parameter still gets.
|
||||
"""
|
||||
await _require(authorization, "read")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
try:
|
||||
rows = await conn.fetch("""
|
||||
SELECT d.*, a.name AS agent_name, a.pool_id, p.name AS pool_name,
|
||||
av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN haproxy_cluster_pools p ON p.id = a.pool_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE $1::int IS NULL
|
||||
OR a.pool_id = (SELECT pool_id FROM haproxy_clusters WHERE id = $1::int)
|
||||
ORDER BY d.reported_at DESC, d.id DESC
|
||||
""", cluster_id)
|
||||
except Exception as exc: # noqa: BLE001 — a missing relation degrades to empty (B-7)
|
||||
logger.debug(f"vip_discoveries unavailable: {exc}")
|
||||
return {"discoveries": []}
|
||||
return {"discoveries": [_discovery_row_to_api(r) for r in rows]}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{vip_id}")
|
||||
async def get_vip(vip_id: int, authorization: str = Header(None)):
|
||||
await _require(authorization, "read")
|
||||
@@ -985,3 +1025,425 @@ async def vip_status(vip_id: int, authorization: str = Header(None)):
|
||||
# endpoint (cluster.py get_config_version_diff, vip-* branch) via render_vip_config_masked
|
||||
# above — there is no bespoke VIP preview endpoint, so VIP changes use the product's
|
||||
# standard "View Change" like every other entity (issue #27 follow-up).
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# v1.10.4 — Adoption of a keepalived setup that already exists on the nodes
|
||||
# ---------------------------------------------------------------------------
|
||||
# The HA/VIP page starts empty on a fleet that already runs keepalived, because the flow is
|
||||
# one-way: VIPs are declared here and pushed to the node, and nothing read what was already
|
||||
# there. The agent now reports the keepalived.conf it found (read-only) into vip_discoveries;
|
||||
# these two endpoints list those findings and turn one into a managed VIP.
|
||||
#
|
||||
# Adoption REPLACES the operator's file with our render, so it is gated hard: the parser
|
||||
# reports every directive we cannot reproduce and every value we cannot know, and adoption
|
||||
# refuses while any remain. See services/keepalived_parser.py for the reasoning.
|
||||
|
||||
|
||||
def _discovery_row_to_api(row) -> dict:
|
||||
"""Shape a vip_discoveries row for the UI. Never includes the VRRP password: the stored
|
||||
config copy is masked and the analysis has the secret replaced by a boolean."""
|
||||
analysis = row["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
try:
|
||||
analysis = json.loads(analysis)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
analysis = None
|
||||
return {
|
||||
"agent_id": row["agent_id"],
|
||||
"agent_name": row["agent_name"],
|
||||
"pool_id": row["pool_id"],
|
||||
"pool_name": row["pool_name"],
|
||||
"config_path": row["config_path"],
|
||||
"config_hash": row["config_hash"],
|
||||
"is_managed": row["is_managed"],
|
||||
"parse_error": row["parse_error"],
|
||||
"adopted_vip_id": row["adopted_vip_id"],
|
||||
# v1.10.8 — adoptability is derived from whether the linked VIP is STILL active, not
|
||||
# from the link existing. `adopted_vip_id` is write-once and nothing clears it, and a
|
||||
# VIP is only ever soft-deleted (is_active=FALSE), so the column's ON DELETE SET NULL
|
||||
# never fires. Keying the UI on the link alone made a rejected adoption hide the node
|
||||
# from the panel forever: the VIP was gone from the VIP list too, and the agent does not
|
||||
# re-report an unchanged file. Deriving it here self-heals reject, undo-reject, approved
|
||||
# teardown and anything added later, without a write on each path.
|
||||
"adopted_vip_active": bool(row["adopted_vip_active"]) if row["adopted_vip_id"] else False,
|
||||
"reported_at": row["reported_at"].isoformat() if row["reported_at"] else None,
|
||||
"config_preview": row["raw_config_masked"],
|
||||
"analysis": analysis,
|
||||
}
|
||||
|
||||
|
||||
# NOTE: list_vip_discoveries lives above the `/{vip_id}` routes — see the ordering comment
|
||||
# there. `/adopt` below is a POST and no `POST /{vip_id}` exists, so it is not shadowed.
|
||||
|
||||
|
||||
def _find_candidate(analysis: Optional[dict], instance_name: str) -> Optional[dict]:
|
||||
for cand in ((analysis or {}).get("candidates") or []):
|
||||
if cand.get("instance_name") == instance_name:
|
||||
return cand
|
||||
return None
|
||||
|
||||
|
||||
async def _collect_instance_participants(conn, *, pool_id: int, vrid: int, virtual_ip: str):
|
||||
"""Every discovered node in the pool that reports the SAME vrrp_instance.
|
||||
|
||||
v1.10.8. Adoption used to take only the node whose row was clicked, which broke the exact
|
||||
case the feature exists for — a running HA pair:
|
||||
|
||||
* adopting the BACKUP alone produced a VIP whose apply fails outright, because apply
|
||||
requires exactly one MASTER member;
|
||||
* adopting the MASTER alone left the peer unmanaged, and adopting it afterwards hit the
|
||||
VRID-collision guard with a 409, so the pair could never be completed from the panel;
|
||||
* worst, on a UNICAST pair the single-member render silently drops the unicast block —
|
||||
render_keepalived_conf only emits it when peer_ips is non-empty — so keepalived falls
|
||||
back to multicast on the adopted node while its peer stays unicast. They stop seeing
|
||||
each other and BOTH claim the VIP.
|
||||
|
||||
Identity is (virtual_router_id, virtual address), which is what keepalived itself uses to
|
||||
decide two nodes belong to one VRRP group, so it is the correct key. A node whose config
|
||||
failed to parse cannot be matched and is therefore skipped — the unicast peer check in the
|
||||
caller is what stops that turning into a silent half-adoption.
|
||||
"""
|
||||
rows = await conn.fetch("""
|
||||
SELECT d.agent_id, d.config_hash, d.analysis, d.parse_error, d.adopted_vip_id,
|
||||
d.auth_pass_encrypted,
|
||||
a.name AS agent_name, a.ip_address, av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE a.pool_id = $1 AND COALESCE(a.enabled, TRUE) = TRUE
|
||||
""", pool_id)
|
||||
|
||||
participants = []
|
||||
for r in rows:
|
||||
if r["parse_error"]:
|
||||
continue
|
||||
if r["adopted_vip_id"] and r["adopted_vip_active"]:
|
||||
continue # already under management by a VIP that still stands
|
||||
analysis = r["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
try:
|
||||
analysis = json.loads(analysis)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
for cand in ((analysis or {}).get("candidates") or []):
|
||||
v = cand.get("vip") or {}
|
||||
if v.get("virtual_router_id") == vrid and v.get("virtual_ip") == virtual_ip:
|
||||
participants.append({
|
||||
"agent_id": r["agent_id"], "agent_name": r["agent_name"],
|
||||
"ip_address": r["ip_address"], "config_hash": r["config_hash"],
|
||||
"auth_pass_encrypted": r["auth_pass_encrypted"],
|
||||
"candidate": cand,
|
||||
})
|
||||
break
|
||||
return participants
|
||||
|
||||
|
||||
@router.post("/adopt")
|
||||
async def adopt_vip(payload: dict, request: Request, authorization: str = Header(None)):
|
||||
"""Turn one discovered vrrp_instance into a managed VIP.
|
||||
|
||||
Body: agent_id, instance_name, name, [description], [prefix_length], [accept_data_loss].
|
||||
|
||||
Refuses while the parser reports blockers. Two of them are resolvable by the operator
|
||||
rather than fatal:
|
||||
* a missing prefix length can be supplied as `prefix_length` (we never guess a netmask
|
||||
for a live VIP);
|
||||
* "we would delete this directive" can be accepted with `accept_data_loss: true`, which
|
||||
is an explicit choice to lose e.g. a notify hook. Everything else — an unknown VRID, a
|
||||
fractional advert_int, an unsupported auth_type — is not a loss but an impossibility,
|
||||
and no flag overrides it.
|
||||
|
||||
The VIP is created PENDING like any other, so nothing reaches the node until the operator
|
||||
applies it from Apply Management.
|
||||
"""
|
||||
current_user = await _require(authorization, "create")
|
||||
agent_id = payload.get("agent_id")
|
||||
instance_name = (payload.get("instance_name") or "").strip()
|
||||
name = (payload.get("name") or "").strip()
|
||||
if not agent_id or not instance_name or not name:
|
||||
raise HTTPException(status_code=400, detail="agent_id, instance_name and name are required")
|
||||
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
disc = await conn.fetchrow("""
|
||||
SELECT d.*, a.name AS agent_name, a.pool_id, a.ip_address,
|
||||
av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE d.agent_id = $1
|
||||
""", int(agent_id))
|
||||
if not disc:
|
||||
raise HTTPException(status_code=404, detail="No discovered keepalived config for that agent")
|
||||
# Only an adoption that is still STANDING blocks a new one. A rejected adoption leaves
|
||||
# adopted_vip_id pointing at a soft-deleted VIP, and nothing clears it, so keying on the
|
||||
# link alone made the node permanently unadoptable (v1.10.8).
|
||||
if disc["adopted_vip_id"] and disc["adopted_vip_active"]:
|
||||
raise HTTPException(status_code=409, detail="This discovery has already been adopted")
|
||||
if not disc["pool_id"]:
|
||||
raise HTTPException(status_code=400, detail="The agent is not in a pool; assign it first")
|
||||
if disc["parse_error"]:
|
||||
raise HTTPException(status_code=422,
|
||||
detail=f"Config could not be parsed: {disc['parse_error']}")
|
||||
|
||||
analysis = disc["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
analysis = json.loads(analysis)
|
||||
cand = _find_candidate(analysis, instance_name)
|
||||
if not cand:
|
||||
raise HTTPException(status_code=404,
|
||||
detail=f"No vrrp_instance '{instance_name}' in the report")
|
||||
|
||||
vip_fields = dict(cand.get("vip") or {})
|
||||
member = dict(cand.get("member") or {})
|
||||
|
||||
# The operator may supply the one value we refuse to guess.
|
||||
supplied_prefix = payload.get("prefix_length")
|
||||
if vip_fields.get("prefix_length") is None and supplied_prefix is not None:
|
||||
try:
|
||||
vip_fields["prefix_length"] = int(supplied_prefix)
|
||||
except (TypeError, ValueError):
|
||||
raise HTTPException(status_code=400, detail="prefix_length must be an integer")
|
||||
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
vrid = vip_fields.get("virtual_router_id")
|
||||
virtual_ip = vip_fields.get("virtual_ip")
|
||||
if vrid is None or not virtual_ip or not member.get("network_interface"):
|
||||
raise HTTPException(status_code=422,
|
||||
detail="Incomplete candidate (vrid/address/interface)")
|
||||
|
||||
# v1.10.8 — adopt the whole VRRP instance. Every node in the pool reporting this same
|
||||
# VRID + address becomes a member, each with the role, priority and interface ITS OWN
|
||||
# file declares. See _collect_instance_participants for why single-node adoption was
|
||||
# unsafe on a unicast pair.
|
||||
participants = await _collect_instance_participants(
|
||||
conn, pool_id=disc["pool_id"], vrid=int(vrid), virtual_ip=virtual_ip)
|
||||
if not any(p["agent_id"] == int(agent_id) for p in participants):
|
||||
# The reporting node must be in its own instance; if it is not, something changed
|
||||
# underneath us (re-report, concurrent adoption) — refuse rather than guess.
|
||||
raise HTTPException(status_code=409,
|
||||
detail="The discovery changed while adopting; refresh and try again")
|
||||
|
||||
# One source of truth for which blockers an operator may resolve (see the docstring on
|
||||
# remaining_blockers): a supplied prefix, and an explicit acceptance of directives our
|
||||
# renderer would delete. Nothing else is waivable. Checked for EVERY node we are about
|
||||
# to overwrite, not only the one that was clicked.
|
||||
for p in participants:
|
||||
rem = remaining_blockers(
|
||||
list(p["candidate"].get("blockers") or []),
|
||||
prefix_supplied=vip_fields.get("prefix_length") is not None,
|
||||
accept_data_loss=bool(payload.get("accept_data_loss")),
|
||||
)
|
||||
if rem:
|
||||
raise HTTPException(status_code=422, detail={
|
||||
"message": f"This keepalived config cannot be adopted as-is ({p['agent_name']})",
|
||||
"node": p["agent_name"],
|
||||
"blockers": rem,
|
||||
})
|
||||
if not (p["candidate"].get("member") or {}).get("network_interface"):
|
||||
raise HTTPException(status_code=422,
|
||||
detail=f"{p['agent_name']} does not declare an interface for this instance")
|
||||
|
||||
# STRANDING GUARD. Participant resolution can only match a node it can READ, that is
|
||||
# ENABLED, and that is in THIS pool. Every one of those is a door through which a real
|
||||
# member of the VRRP group leaves the set silently — and a silent exit is the whole
|
||||
# failure mode this release exists to close, because the nodes we do adopt get rewritten
|
||||
# while the one that left keeps running an unmanaged config on the same address.
|
||||
#
|
||||
# So rather than guard each door, ask the question directly: is there any reported
|
||||
# keepalived.conf that mentions THIS virtual address and is not among the nodes we are
|
||||
# about to adopt? Scoping on the address keeps an unrelated file elsewhere in the fleet
|
||||
# from blocking every adoption.
|
||||
#
|
||||
# Two exclusions are legitimate rather than stranding:
|
||||
# * a node already held by a VIP that still STANDS is under management already — and it
|
||||
# cannot be this instance, because uq_vip_vrid_active forbids a second active VIP with
|
||||
# this VRID in the pool (the clash check above would have refused first);
|
||||
# * a node whose file carries our ownership marker (is_managed) is ours already.
|
||||
adopted_ids = [p["agent_id"] for p in participants]
|
||||
stranded = await conn.fetch("""
|
||||
SELECT a.name AS agent_name, a.pool_id, COALESCE(a.enabled, TRUE) AS enabled,
|
||||
d.parse_error
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE d.raw_config_masked LIKE '%' || $1 || '%'
|
||||
AND NOT (a.id = ANY($2::int[]))
|
||||
AND COALESCE(av.is_active, FALSE) = FALSE
|
||||
AND d.is_managed = FALSE
|
||||
ORDER BY a.name
|
||||
""", virtual_ip, adopted_ids)
|
||||
if stranded:
|
||||
def _why(r):
|
||||
if r["parse_error"]:
|
||||
return f"its config could not be parsed: {r['parse_error']}"
|
||||
if not r["enabled"]:
|
||||
return "the agent is disabled, so it can never receive a config"
|
||||
if r["pool_id"] != disc["pool_id"]:
|
||||
return "it is in a different agent pool, and a VIP's members must share one pool"
|
||||
return "it does not report this vrrp_instance"
|
||||
named = "; ".join(f"{r['agent_name']} — {_why(r)}" for r in stranded)
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"These node(s) also reference {virtual_ip} but cannot be taken over with the rest "
|
||||
f"of the instance, so adopting now would rewrite the others and leave them running "
|
||||
f"an unmanaged config on the same address: {named}. Resolve that first, then adopt."))
|
||||
|
||||
# AGREEMENT ON THE SHARED VIP FIELDS. Everything below is stored once on the VIP row and
|
||||
# re-rendered onto EVERY member, so a value taken from the node that happened to be
|
||||
# clicked would be imposed on nodes whose own file said something else. prefix_length is
|
||||
# the sharpest: the design refuses to GUESS a netmask for a live VIP, and quietly copying
|
||||
# one node's netmask onto another is the same change by another name.
|
||||
def _vfield(p, key):
|
||||
return (p["candidate"].get("vip") or {}).get(key)
|
||||
|
||||
for key, label in (("prefix_length", "prefix length"),
|
||||
("use_unicast", "unicast/multicast mode"),
|
||||
("track_haproxy", "HAProxy tracking")):
|
||||
seen = {}
|
||||
for p in participants:
|
||||
seen.setdefault(_vfield(p, key), []).append(p["agent_name"])
|
||||
if len(seen) > 1:
|
||||
spread = "; ".join(f"{v!r}: {', '.join(names)}" for v, names in seen.items())
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes of this instance disagree on {label} ({spread}). One value is "
|
||||
f"stored on the VIP and re-rendered onto every member, so adopting would "
|
||||
f"impose one node's setting on the others. Align the files first."))
|
||||
|
||||
# The VRRP secret is stored per discovery as its own Fernet token, and Fernet is
|
||||
# non-deterministic, so the ciphertexts cannot be compared — decrypt and compare the
|
||||
# plaintexts. Nothing is logged. A node we cannot decrypt is treated as a mismatch rather
|
||||
# than assumed equal.
|
||||
secrets = {}
|
||||
for p in participants:
|
||||
enc = p.get("auth_pass_encrypted")
|
||||
plain = decrypt_vrrp_secret(enc) if enc else None
|
||||
if enc and not plain:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"The VRRP secret reported by {p['agent_name']} cannot be decrypted "
|
||||
f"(encryption key changed?). Re-report or fix that node before adopting."))
|
||||
secrets.setdefault(plain, []).append(p["agent_name"])
|
||||
if len(secrets) > 1:
|
||||
groups = "; ".join(
|
||||
("no password: " if k is None else "one password: ") + ", ".join(v)
|
||||
for k, v in secrets.items())
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes of this instance do not share one VRRP password ({groups}). "
|
||||
f"Adoption stores a single secret and renders it onto every member, so it would "
|
||||
f"silently change authentication on the others. Align the files first."))
|
||||
|
||||
# keepalived requires one MASTER and a matching advertisement interval across the group.
|
||||
# Both are checked here so the operator learns at adoption time instead of at apply.
|
||||
roles = [(p["candidate"].get("member") or {}).get("role") or "BACKUP" for p in participants]
|
||||
if roles.count("MASTER") != 1:
|
||||
listed = ", ".join(f"{p['agent_name']}={r}" for p, r in zip(participants, roles))
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"VRID {vrid} is reported by {len(participants)} node(s) with "
|
||||
f"{roles.count('MASTER')} MASTER ({listed}); exactly one must be MASTER. If a "
|
||||
f"member is missing, install or enable its agent so it reports its keepalived.conf, "
|
||||
f"then adopt again."))
|
||||
adverts = {int((p["candidate"].get("vip") or {}).get("advert_int") or 1) for p in participants}
|
||||
if len(adverts) > 1:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes disagree on advert_int ({sorted(adverts)}). keepalived needs the same "
|
||||
f"advertisement interval across a VRRP group, so align the files first."))
|
||||
|
||||
# UNICAST SAFETY. Our renderer emits the unicast block only when it has peer addresses,
|
||||
# so a peer that is not a member would be dropped and keepalived would fall back to
|
||||
# multicast on this node while the real peer stays unicast — both would then claim the
|
||||
# VIP. Refuse instead, naming the address that is unaccounted for.
|
||||
declared_peers = {str(x) for p in participants for x in (p["candidate"].get("peers") or [])}
|
||||
if declared_peers:
|
||||
missing_ip = [p["agent_name"] for p in participants if not p["ip_address"]]
|
||||
if missing_ip:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"These nodes have not reported an IP address yet: {', '.join(missing_ip)}. "
|
||||
f"The unicast peer list cannot be verified until they do."))
|
||||
member_ips = {str(p["ip_address"]) for p in participants}
|
||||
unknown = sorted(declared_peers - member_ips)
|
||||
if unknown:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"This instance is unicast and lists peer(s) {', '.join(unknown)} that are not "
|
||||
f"among the nodes being adopted. Adopting would drop them from the peer list and "
|
||||
f"keepalived would silently fall back to multicast, so both sides could end up "
|
||||
f"holding the VIP. Register those nodes as agents so they report their config, "
|
||||
f"then adopt again."))
|
||||
|
||||
# One active VIP per agent — the delivery endpoint serves a single keepalived.conf per
|
||||
# node, so a second active membership never converges. create/update enforce this via
|
||||
# _validate_members_against_pool; adoption did not call it at all.
|
||||
busy = await conn.fetchrow("""
|
||||
SELECT a.name AS agent_name, v.name AS vip_name
|
||||
FROM vip_members vm JOIN vip_instances v ON v.id = vm.vip_id
|
||||
JOIN agents a ON a.id = vm.agent_id
|
||||
WHERE vm.agent_id = ANY($1::int[]) AND v.is_active = TRUE LIMIT 1
|
||||
""", [p["agent_id"] for p in participants])
|
||||
if busy:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"node '{busy['agent_name']}' is already a member of VIP '{busy['vip_name']}'"))
|
||||
|
||||
# Adoption must keep the VRRP identity it found. A different VRID would create a second
|
||||
# VRRP domain on the wire, so a collision inside the pool is a hard conflict, never an
|
||||
# auto-reallocation the way create_vip does it.
|
||||
clash = await conn.fetchrow("""
|
||||
SELECT id, name FROM vip_instances
|
||||
WHERE pool_id = $1 AND is_active = TRUE AND virtual_router_id = $2
|
||||
""", disc["pool_id"], int(vrid))
|
||||
if clash:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"VRID {vrid} is already used by VIP '{clash['name']}' in this pool; the adopted "
|
||||
f"config must keep its VRID, so resolve the collision first"))
|
||||
|
||||
async with conn.transaction():
|
||||
vip_id = await conn.fetchval("""
|
||||
INSERT INTO vip_instances
|
||||
(name, description, pool_id, virtual_ip, prefix_length, virtual_router_id,
|
||||
advert_int, auth_pass_encrypted, use_unicast, track_haproxy,
|
||||
is_active, last_config_status, adopted_at, created_by)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,TRUE,'PENDING',CURRENT_TIMESTAMP,$11)
|
||||
RETURNING id
|
||||
""", name, payload.get("description") or f"Adopted from {disc['agent_name']}",
|
||||
disc["pool_id"], virtual_ip, int(vip_fields["prefix_length"]), int(vrid),
|
||||
int(vip_fields.get("advert_int") or 1),
|
||||
disc["auth_pass_encrypted"], # already Fernet-encrypted at ingest
|
||||
bool(vip_fields.get("use_unicast")), bool(vip_fields.get("track_haproxy")),
|
||||
current_user["id"])
|
||||
# Every node of the instance becomes a member with the role/priority/interface ITS
|
||||
# OWN file declares, and each carries its own one-shot takeover authorisation pinned
|
||||
# to the hash of the file we analysed on THAT node. The guard stays per-node: a file
|
||||
# edited on one member between adoption and Apply is still refused there.
|
||||
for p in participants:
|
||||
pm = p["candidate"].get("member") or {}
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members
|
||||
(vip_id, agent_id, network_interface, role, priority, takeover_expected_hash)
|
||||
VALUES ($1,$2,$3,$4,$5,$6)
|
||||
""", vip_id, p["agent_id"], pm["network_interface"],
|
||||
pm.get("role") or "BACKUP", int(pm.get("priority") or 100),
|
||||
p["config_hash"])
|
||||
await conn.execute(
|
||||
"UPDATE vip_discoveries SET adopted_vip_id = $2 WHERE agent_id = $1",
|
||||
p["agent_id"], vip_id)
|
||||
|
||||
await _stage_vip_version(conn, vip_id, "adopt", current_user["id"])
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="adopt", resource_type="vip",
|
||||
resource_id=str(vip_id),
|
||||
details={"name": name, "virtual_ip": virtual_ip, "vrid": vrid,
|
||||
"adopted_from_agent": disc["agent_name"],
|
||||
"members": [p["agent_name"] for p in participants],
|
||||
"accepted_data_loss": bool(payload.get("accept_data_loss"))},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
|
||||
node_names = [p["agent_name"] for p in participants]
|
||||
return {
|
||||
"id": vip_id,
|
||||
"members": node_names,
|
||||
"message": (
|
||||
f"VIP adopted with {len(node_names)} member node(s): {', '.join(node_names)} "
|
||||
f"(PENDING — review it in Apply Management, then apply to hand their "
|
||||
f"keepalived.conf over to OpenManager)"),
|
||||
}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -8,9 +8,13 @@ suitable for an Antd Tabs/Steps display.
|
||||
Key constraints (Section 3.3 of the v1.5.0 plan):
|
||||
- DNS resolution uses stdlib socket.gethostbyname_ex via run_in_executor (we
|
||||
intentionally avoid pulling aiodns as a runtime dep for v1.5.0).
|
||||
- Port-80 probe is HEAD-only, target locked to the order's domains, success on
|
||||
HTTP 200 OR 404, warns on egress timeout (don't fail-hard — corp egress
|
||||
policies often blackhole outbound 80).
|
||||
- Port-80 probe is a GET (not HEAD) locked to the order's domains, because the
|
||||
status code alone cannot tell a working challenge endpoint from a web UI: a
|
||||
reverse proxy that has lost its /.well-known/acme-challenge/ location serves
|
||||
its SPA with HTTP 200. The body's shape decides. Warns rather than fails on
|
||||
egress timeout (corp egress policies often blackhole outbound 80) and on a
|
||||
wrong responder (this probe sees the PUBLIC domain, not the challenge backend,
|
||||
so it is evidence rather than a verdict).
|
||||
- All checks have hard wall-clock timeouts (asyncio.wait_for) to bound impact
|
||||
on the API event loop.
|
||||
- humanize_error_detail covers >= 11 RFC8555 problem types and is backwards
|
||||
@@ -330,10 +334,41 @@ async def check_dns(domains: List[str]) -> Dict[str, Any]:
|
||||
)
|
||||
|
||||
|
||||
# Enough to classify a response without turning the diagnostic into a way to pull
|
||||
# arbitrary amounts of a third party's content into our JSON.
|
||||
_PROBE_BODY_LIMIT = 65536
|
||||
|
||||
|
||||
def _classify_probe_body(body: bytes, content_type: str) -> str:
|
||||
"""Coarse shape of a probe response: html | json | text | empty | binary.
|
||||
|
||||
The body itself is deliberately NOT retained anywhere — the shape is all that is
|
||||
needed to tell "served me a web page" from "served me a token", and keeping the
|
||||
bytes would open a new read surface onto whatever is behind the address.
|
||||
"""
|
||||
if not body:
|
||||
return "empty"
|
||||
ct = (content_type or "").lower()
|
||||
head = body[:512].lstrip().lower()
|
||||
if ct.startswith("text/html") or head.startswith((b"<!doctype", b"<html")):
|
||||
return "html"
|
||||
if ct.startswith("application/json") or head[:1] in (b"{", b"["):
|
||||
return "json"
|
||||
try:
|
||||
body.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return "binary"
|
||||
return "text"
|
||||
|
||||
|
||||
async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict[str, Any]:
|
||||
"""Probe HTTP-01 readiness on port 80 with a HEAD request to a synthetic
|
||||
challenge URL. Success on 200 OR 404 (404 means the well-known path is
|
||||
served but no challenge yet — fine).
|
||||
"""Probe HTTP-01 readiness on port 80 with a GET to a synthetic challenge URL.
|
||||
|
||||
404 means the path is served but no challenge is outstanding, which is fine. A
|
||||
200 is only fine if the body is NOT a web page: a proxy that has lost its
|
||||
/.well-known/acme-challenge/ route falls through to its catch-all and answers
|
||||
200 with index.html, which a status-code-only check accepts as healthy while
|
||||
every real validation fails.
|
||||
|
||||
On egress timeout we WARN rather than FAIL because many corporate egress
|
||||
policies blackhole port 80 outbound; that does not impair LE's ingress
|
||||
@@ -398,12 +433,53 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
continue
|
||||
url = f"http://{d}/.well-known/acme-challenge/diagnostic-probe"
|
||||
try:
|
||||
async with session.head(url, allow_redirects=False) as resp:
|
||||
targets.append({
|
||||
# GET, not HEAD: the status code alone cannot tell a working challenge
|
||||
# endpoint from a SPA. A reverse proxy that has lost its
|
||||
# /.well-known/acme-challenge/ location falls through to its catch-all
|
||||
# and serves index.html with HTTP 200 — which the old
|
||||
# `status in (200, 404)` rule accepted as healthy while every real
|
||||
# validation failed. Only the body distinguishes them.
|
||||
async with session.get(url, allow_redirects=False) as resp:
|
||||
# The body is EVIDENCE, not a precondition. If it cannot be read —
|
||||
# connection reset mid-response, a server that hangs after headers —
|
||||
# fall back to the status-only semantics this check has always had
|
||||
# rather than turning a healthy 404 into a hard failure. The stricter
|
||||
# rule below applies only when there is something to judge.
|
||||
try:
|
||||
body = await resp.content.read(_PROBE_BODY_LIMIT)
|
||||
except Exception:
|
||||
body = None
|
||||
content_type = (resp.headers.get("content-type") or "").split(";")[0].strip()
|
||||
body_class = (
|
||||
_classify_probe_body(body, content_type) if body is not None else "unread"
|
||||
)
|
||||
target = {
|
||||
"domain": d,
|
||||
"status": resp.status,
|
||||
"ok": resp.status in (200, 404),
|
||||
})
|
||||
"content_type": content_type or None,
|
||||
"body_len": len(body) if body is not None else None,
|
||||
"body_class": body_class,
|
||||
}
|
||||
if resp.status == 200 and body_class == "html":
|
||||
# Reachable, wrong responder. Reported as a warning rather than
|
||||
# a failure: this check probes the PUBLIC domain and cannot see
|
||||
# the challenge backend, so it is evidence, not a verdict — and
|
||||
# a new `fail` here would block the site wizard on upgrade day
|
||||
# for every install.
|
||||
target["warn"] = True
|
||||
target["diagnosis"] = (
|
||||
"responded 200 with an HTML page, not a challenge token — "
|
||||
"the request is reaching a web UI instead of the ACME endpoint"
|
||||
)
|
||||
elif resp.status in (301, 302, 303, 307, 308):
|
||||
target["warn"] = True
|
||||
target["redirect_location"] = resp.headers.get("location")
|
||||
target["diagnosis"] = (
|
||||
"redirected instead of serving the challenge path"
|
||||
)
|
||||
else:
|
||||
target["ok"] = resp.status in (200, 404)
|
||||
targets.append(target)
|
||||
except asyncio.TimeoutError:
|
||||
targets.append({"domain": d, "error": "egress timeout", "warn": True})
|
||||
skip_reason = "egress timeout"
|
||||
@@ -425,7 +501,11 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
details={"targets": targets},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
if warns and not [t for t in targets if t.get("ok")]:
|
||||
# Surface warnings even when OTHER domains answered correctly. The old condition
|
||||
# ("warn only if nothing succeeded") hid the single most diagnostic outcome there
|
||||
# is: a multi-domain certificate where one name reaches a web UI instead of the
|
||||
# challenge endpoint reported a clean pass.
|
||||
if warns:
|
||||
# R18b audit fix (round 7): branch the rollup message on the
|
||||
# actual cause. Pre-fix the message was always "Egress to
|
||||
# port 80 appears blocked" — even when every target was
|
||||
@@ -435,6 +515,36 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
# corporate firewall logs while the real cause was an
|
||||
# internal-only DNS A record. Also harden against
|
||||
# `skip_reason=None` so the message never reads "(None)".
|
||||
# Wrong-responder warnings take priority over every other cause: they are the
|
||||
# only ones that mean "your server answered, and answered wrong", which is a
|
||||
# different problem from "we could not test".
|
||||
wrong_responder = [t for t in targets if t.get("diagnosis")]
|
||||
if wrong_responder:
|
||||
first = wrong_responder[0]
|
||||
if first.get("body_class") == "html":
|
||||
human = (
|
||||
f"{first['domain']} answered HTTP {first.get('status')} with an HTML "
|
||||
f"page ({first.get('content_type') or 'unknown type'}, "
|
||||
f"{first.get('body_len')} bytes) instead of a challenge token. The "
|
||||
"path is reaching a web interface, not the ACME endpoint — check "
|
||||
"that the reverse proxy in front of OpenManager routes "
|
||||
"/.well-known/acme-challenge/ to the API."
|
||||
)
|
||||
else:
|
||||
human = (
|
||||
f"{first['domain']} answered HTTP {first.get('status')} "
|
||||
f"({first.get('diagnosis')})"
|
||||
)
|
||||
return _check_result(
|
||||
"port80",
|
||||
"Port 80 reachability",
|
||||
"warn",
|
||||
human,
|
||||
severity="warn",
|
||||
details={"targets": targets},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
ssrf_skip = any(
|
||||
"non-public" in (t.get("skip") or "")
|
||||
or "SSRF" in (t.get("skip") or "")
|
||||
@@ -491,11 +601,20 @@ async def check_routing(conn, domains: List[str], cluster_ids: List[int]) -> Dic
|
||||
duration_ms=int((time.time() - started) * 1000),
|
||||
)
|
||||
|
||||
# The WHERE clause is deliberately identical to the pre-existing one, so `not rows`
|
||||
# still means exactly what it meant before and the `fail` branch below cannot fire
|
||||
# in any situation where it previously passed. Narrowing it here (e.g. by adding a
|
||||
# mode filter) would turn a tcp-only port-80 cluster from "ok" into "fail", and the
|
||||
# site wizard blocks submit on any failing check — locking those installs the day
|
||||
# this ships. Mode is examined afterwards, in Python, and only ever downgrades to
|
||||
# `warn`.
|
||||
rows = await conn.fetch(
|
||||
"""
|
||||
SELECT id, name, bind_address, bind_port, mode, default_backend
|
||||
FROM frontends
|
||||
WHERE cluster_id = ANY($1::int[]) AND is_active = TRUE AND bind_port = 80
|
||||
SELECT f.id, f.name, f.bind_address, f.bind_port, f.mode, f.default_backend,
|
||||
f.cluster_id, c.acme_enabled
|
||||
FROM frontends f
|
||||
JOIN haproxy_clusters c ON c.id = f.cluster_id
|
||||
WHERE f.cluster_id = ANY($1::int[]) AND f.is_active = TRUE AND f.bind_port = 80
|
||||
""",
|
||||
cluster_ids,
|
||||
)
|
||||
@@ -510,13 +629,144 @@ async def check_routing(conn, domains: List[str], cluster_ids: List[int]) -> Dic
|
||||
details={"cluster_ids": cluster_ids},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
# Character-for-character the renderer's normalisation (services/haproxy_config.py),
|
||||
# so this can never disagree with what actually gets emitted.
|
||||
http_rows = [r for r in rows if (r["mode"] or "http").strip().lower() == "http"]
|
||||
if not http_rows:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
"The only port-80 frontend(s) in the target cluster(s) are in tcp mode. "
|
||||
"A tcp-mode frontend cannot carry the /.well-known/acme-challenge/ ACL, "
|
||||
"so HTTP-01 cannot be served — use DNS-01, or add an http-mode frontend "
|
||||
"on port 80."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows]},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
rows = http_rows
|
||||
|
||||
# A frontend row proves only that the DATABASE describes port-80 routing. The
|
||||
# renderer gates the challenge ACL on `acme_enabled`, and the nodes run whatever
|
||||
# config was last APPLIED — so the row said "ok" during an incident where the
|
||||
# live config had no usable challenge route at all. Check the two things the row
|
||||
# cannot tell us. Both report `warn`, never `fail`: the site wizard blocks submit
|
||||
# on any `fail`, so a new failing condition would lock every install on the day
|
||||
# it ships.
|
||||
# `.get()` rather than `[]`: the column gates a WARNING, so a row shape without
|
||||
# it should not blow up the whole diagnostic. Absent means "assume enabled" —
|
||||
# the applied-config check below is the authoritative one either way.
|
||||
acme_off = sorted({r["cluster_id"] for r in rows if not r.get("acme_enabled", True)})
|
||||
if acme_off:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {acme_off} have ACME Challenge Routing disabled, so the "
|
||||
"generated config contains no /.well-known/acme-challenge/ route. "
|
||||
"Enable it in Cluster Management, then apply the cluster."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "acme_disabled_clusters": acme_off},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
missing_in_applied = []
|
||||
challenge_backends = {}
|
||||
for cluster_id in sorted({r["cluster_id"] for r in rows}):
|
||||
# Selector matched to the one the AGENT uses to fetch its config
|
||||
# (routers/agent.py: status='APPLIED' AND is_active=TRUE), because the question
|
||||
# here is "what are the nodes running right now?". Without `is_active` this can
|
||||
# read a superseded row and report on a config that was never delivered.
|
||||
# Extracting in SQL rather than pulling whole configs back per cluster: these
|
||||
# files run to hundreds of KB on real installs.
|
||||
applied = await conn.fetchrow(
|
||||
"""
|
||||
SELECT position('use_backend _acme_challenge_backend' in config_content) > 0
|
||||
AS has_route,
|
||||
substring(config_content from 'server _acme_mgmt [^\\n]*') AS server_line
|
||||
FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'APPLIED' AND is_active = TRUE
|
||||
AND config_content IS NOT NULL
|
||||
ORDER BY created_at DESC LIMIT 1
|
||||
""",
|
||||
cluster_id,
|
||||
)
|
||||
if not applied or not applied["has_route"]:
|
||||
missing_in_applied.append(cluster_id)
|
||||
continue
|
||||
server_line = (applied["server_line"] or "").strip()
|
||||
challenge_backends[cluster_id] = (
|
||||
server_line[len("server _acme_mgmt "):].strip() if server_line else None
|
||||
)
|
||||
|
||||
if missing_in_applied:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {missing_in_applied} have no applied configuration carrying "
|
||||
"the challenge route. The change exists in the database but the nodes are "
|
||||
"still running an older config — apply the cluster."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "clusters_not_applied": missing_in_applied},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
# A backend section with no `server` line: the route exists, `haproxy -c` passes,
|
||||
# and every challenge request gets a 503 from an empty backend. Without this branch
|
||||
# the falsy target slips past the loopback filter below and the check reports "ok".
|
||||
serverless = sorted(cid for cid, target in challenge_backends.items() if not target)
|
||||
if serverless:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {serverless} route the challenge path to a backend that has "
|
||||
"no server line, so every request returns 503. The configured ACME "
|
||||
"Challenge Backend URL could not be resolved into an address — check it "
|
||||
"in Cluster Management, or Settings > ACME for the global value."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "clusters_without_server": serverless},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
loopback = {
|
||||
cid: target for cid, target in challenge_backends.items()
|
||||
if target and target.split(":")[0].strip("[]").lower()
|
||||
in ("localhost", "127.0.0.1", "::1", "0.0.0.0")
|
||||
}
|
||||
if loopback:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"The applied config points the challenge backend at {sorted(loopback.values())}. "
|
||||
"HAProxy resolves that on the HAProxy node, so it means the node itself, not "
|
||||
"this management server. Set ACME Challenge Backend URL to a routable address."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "challenge_backends": challenge_backends},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"ok",
|
||||
f"Found {len(rows)} HTTP frontend(s) on port 80",
|
||||
f"Found {len(rows)} HTTP frontend(s) on port 80; challenge route present in applied config",
|
||||
severity="info",
|
||||
details={"frontends": [dict(r) for r in rows]},
|
||||
details={"frontends": [dict(r) for r in rows], "challenge_backends": challenge_backends},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
|
||||
@@ -19,8 +19,14 @@ subject instead of CN-only, ECDSA support, same PKCS8/NoEncryption key
|
||||
serialisation (the agent concatenates cert+key+chain into one PEM and HAProxy
|
||||
cannot read passphrase-protected keys).
|
||||
|
||||
Private keys are stored PLAINTEXT, consistent with every other key in the
|
||||
system (ssl_certificates.private_key_content, letsencrypt_orders.cert_private_key).
|
||||
Private keys are ENCRYPTED AT REST from v1.10.1 (Issue #53): the Fernet token
|
||||
replaces the PEM in the same `ssl_csrs.private_key_pem` column, so there is no
|
||||
schema change and no SCHEMA_VERSION bump. Rows written earlier hold a raw PEM
|
||||
and are still read transparently — see utils/csr_key_crypto.py for the format
|
||||
discriminator and the key-rotation caveat. The pending CSR key is the one key
|
||||
in the system worth encrypting: it sits idle for the whole signing window and
|
||||
is never transmitted, unlike ssl_certificates.private_key_content and the ACME
|
||||
order keys, which agents must receive in plaintext on every poll.
|
||||
The key is NEVER returned by any CSR API response — `csr_row_to_dict` strips
|
||||
it unconditionally.
|
||||
"""
|
||||
@@ -39,6 +45,7 @@ from cryptography.hazmat.primitives.asymmetric import ec, rsa
|
||||
from cryptography.x509.oid import NameOID
|
||||
|
||||
from services import ssl_service
|
||||
from utils.csr_key_crypto import decrypt_csr_private_key, encrypt_csr_private_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -186,8 +193,14 @@ async def assert_csr_name_available(conn, name: str) -> None:
|
||||
|
||||
|
||||
async def insert_csr_row(conn, payload: Any, bundle: Dict[str, Any], user_id: Optional[int]) -> int:
|
||||
"""Persist a freshly generated CSR bundle. Returns the new csr id."""
|
||||
"""Persist a freshly generated CSR bundle. Returns the new csr id.
|
||||
|
||||
Issue #53 (v1.10.1): the private key is Fernet-encrypted before it is stored. The token goes
|
||||
into the SAME private_key_pem TEXT column — no schema change — and is only ever decrypted
|
||||
in-process by import_signed_certificate. No CSR endpoint returns the column either way.
|
||||
"""
|
||||
await assert_csr_name_available(conn, payload.name)
|
||||
stored_key = encrypt_csr_private_key(bundle['private_key_pem'])
|
||||
try:
|
||||
csr_id = await conn.fetchval(
|
||||
"""
|
||||
@@ -203,7 +216,7 @@ async def insert_csr_row(conn, payload: Any, bundle: Dict[str, Any], user_id: Op
|
||||
json.dumps(bundle['sans']),
|
||||
payload.key_algorithm,
|
||||
bundle['csr_pem'],
|
||||
bundle['private_key_pem'],
|
||||
stored_key,
|
||||
user_id,
|
||||
)
|
||||
except asyncpg.exceptions.UniqueViolationError:
|
||||
@@ -243,8 +256,7 @@ async def import_signed_certificate(conn, csr_id: int, imp: Any, user_id: Option
|
||||
"Create a new CSR to reissue."
|
||||
),
|
||||
)
|
||||
stored_key = row['private_key_pem']
|
||||
if not stored_key:
|
||||
if not row['private_key_pem']:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
@@ -252,6 +264,23 @@ async def import_signed_certificate(conn, csr_id: int, imp: Any, user_id: Option
|
||||
"corrupt. Delete it and create a new CSR."
|
||||
),
|
||||
)
|
||||
# Issue #53: the column holds a Fernet token from v1.10.1 on, and a raw PEM for rows
|
||||
# written before it. decrypt_csr_private_key accepts both, so no data migration is
|
||||
# needed. A None here means the token cannot be decrypted — SECRET_KEY was rotated
|
||||
# without CSR_ENCRYPTION_KEY set. Fail loudly: the key is gone, so the CA's certificate
|
||||
# can never be paired with it, and silently falling through would surface as the far
|
||||
# more confusing "certificate does not match this CSR's private key".
|
||||
stored_key = decrypt_csr_private_key(row['private_key_pem'])
|
||||
if not stored_key:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
f"The stored private key for CSR '{row['name']}' cannot be decrypted. This "
|
||||
"happens when SECRET_KEY was rotated while CSR_ENCRYPTION_KEY was not set. "
|
||||
"The key is unrecoverable, so this CSR can no longer be completed — delete "
|
||||
"it and create a new one (then have the new CSR signed)."
|
||||
),
|
||||
)
|
||||
|
||||
effective_name = getattr(imp, 'name', None) or row['name']
|
||||
|
||||
|
||||
@@ -5,8 +5,8 @@ A small adapter layer so DNS-01 challenges can publish/clean up the
|
||||
additive at the RRset level (add/remove a single value by name+content, never
|
||||
overwrite-by-name) so multiple coexisting values at one name (wildcard + apex) work.
|
||||
|
||||
MVP providers: manual (user publishes the TXT themselves) and Cloudflare. New providers
|
||||
plug in via the registry without touching the orchestration.
|
||||
Providers: manual (user publishes the TXT themselves), Cloudflare, and GoDaddy (v1.10.0). New
|
||||
providers plug in via the registry without touching the orchestration.
|
||||
"""
|
||||
from .base import DnsProvider, DnsProviderError
|
||||
from .registry import get_provider, list_providers, is_supported
|
||||
|
||||
@@ -0,0 +1,512 @@
|
||||
"""GoDaddy DNS provider for ACME DNS-01 (Issue #35 follow-up, v1.10.0).
|
||||
|
||||
Uses the GoDaddy Domains API v1 over aiohttp (no new dependency). The base URL is a hardcoded
|
||||
constant and redirects are not followed (no user-controlled URL — only the already-validated
|
||||
domain name selects which zone is touched), which is the same reason cloudflare.py is exempt from
|
||||
utils/ssrf_guard.py. Every failure is wrapped in DnsProviderError with a SANITIZED message: the
|
||||
API Key and Secret are scrubbed out of any text that could reach a log, an order event, or
|
||||
letsencrypt_orders.error_detail.
|
||||
|
||||
Two GoDaddy-specific hazards drive the shape of this module — neither exists on Cloudflare:
|
||||
|
||||
1. NO PER-VALUE WRITE. `PUT /v1/domains/{d}/records/TXT/{name}` REPLACES the entire RRset at that
|
||||
type+name; it does not merge. A certificate for `example.com` + `*.example.com` publishes two
|
||||
DIFFERENT TXT values at the SAME name `_acme-challenge.example.com` (base.py's additive
|
||||
contract), so a naive single-value PUT would silently destroy the sibling and fail the wildcard
|
||||
authorization. Every mutation here is therefore read-modify-write: GET the current RRset, merge,
|
||||
PUT the whole list back. An EMPTY array is rejected (422 INVALID_BODY, "Records must be
|
||||
specified"), so removing the LAST value must use DELETE — never `PUT []`.
|
||||
|
||||
2. ZONE-DESTRUCTIVE SIBLING PATHS. `PUT /v1/domains/{d}/records/TXT` (three segments, no name)
|
||||
wipes EVERY TXT in the zone — SPF, DKIM, DMARC, Microsoft/Google verification — and
|
||||
`PUT /v1/domains/{d}/records` wipes the whole zone (this is dehydrated issue #430 verbatim).
|
||||
The record path is built only by _rrset_path(), which refuses an empty zone or relative name so
|
||||
a URL can never collapse onto one of those endpoints.
|
||||
|
||||
Concurrency: v1 has no ETag, no If-Match and no per-record id, so read-modify-write can lose an
|
||||
update if two mutations at one name overlap. Today they cannot: orders are advanced sequentially
|
||||
(`for oid in claimed_ids: await advance_dns01_order(oid)` in main.py) and an order's challenges are
|
||||
published sequentially (`for ch in challenges: await provider.add_txt_record(...)` in
|
||||
dns01_orchestrator.py), so the apex+wildcard pair is strictly ordered and the second publish sees
|
||||
the first. _rrset_lock() makes that safety structural rather than incidental. Across REPLICAS the
|
||||
window is real but narrow (two orders publishing at the same record name in overlapping cycles) and
|
||||
self-healing: a lost publish ends `invalid` and the bounded retry chain mints a fresh order, a lost
|
||||
cleanup is retried by the reconcile sweep, and an orphaned `_acme-challenge` TXT is inert. The real
|
||||
fix is the v3 API (POST + DELETE by recordId, natively per-value), which is PAT-only and a
|
||||
follow-up; it is deliberately not used here because v1 + sso-key is what operators can use today.
|
||||
|
||||
Credentials: an API Key + Secret pair from https://developer.godaddy.com/keys. It must be a
|
||||
PRODUCTION key — the first key the dashboard issues is an OTE (test) key and an OTE credential
|
||||
against api.godaddy.com returns 401. A Personal Access Token also works: paste it as the API Key
|
||||
and leave the Secret blank, and the Authorization header becomes `Bearer <token>`. That path is not
|
||||
cosmetic — GoDaddy marks sso-key "deprecated, supported through 2026" and the current v1 OpenAPI
|
||||
advertises only bearer auth, so the PAT is the migration target, not an alternative.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from urllib.parse import quote
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .base import DnsProvider, DnsProviderError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
GODADDY_API_BASE = "https://api.godaddy.com/v1"
|
||||
_TIMEOUT = aiohttp.ClientTimeout(total=20)
|
||||
|
||||
# GoDaddy enforces a 600s (10 min) TTL floor at request time. The published v1 OpenAPI declares no
|
||||
# minimum, so a smaller value is not caught by the schema — it fails with
|
||||
# 422 {"code":"INVALID_BODY","fields":[{"message":"must have a minimum value of 600", ...}]}.
|
||||
# Pin the floor; DNS-01 has no reason to want anything longer.
|
||||
_TXT_TTL = 600
|
||||
|
||||
# Read-modify-write serialization, keyed by the RRset (record name), not the zone — the RRset is the
|
||||
# actual unit of contention, and keying on it avoids serializing unrelated subdomains of one zone.
|
||||
# The orchestrator is sequential today (see the module docstring), so this is defence in depth: it
|
||||
# is what stops a future `asyncio.gather()` over the publish loop from silently breaking every
|
||||
# wildcard+apex certificate. Bounded in practice by the certificate inventory of one process, so
|
||||
# there is no eviction; the entries are empty Lock objects.
|
||||
_RRSET_LOCKS: Dict[str, asyncio.Lock] = {}
|
||||
|
||||
|
||||
def _rrset_lock(record_name: str) -> asyncio.Lock:
|
||||
key = (record_name or "").rstrip(".").lower()
|
||||
lock = _RRSET_LOCKS.get(key)
|
||||
if lock is None:
|
||||
# Safe without a guard: a single event loop never preempts between the get and the assign.
|
||||
lock = _RRSET_LOCKS[key] = asyncio.Lock()
|
||||
return lock
|
||||
|
||||
|
||||
def _scrub(text: str, *secrets: str) -> str:
|
||||
"""Remove credential substrings from a message before it can reach a log or an order event.
|
||||
|
||||
GoDaddy error bodies do not echo the Authorization header, so this is belt-and-braces — but it
|
||||
makes base.py's "never leak a secret" invariant structural instead of a matter of care. Short
|
||||
strings are skipped so a 1-2 char credential fragment cannot blank out ordinary prose.
|
||||
"""
|
||||
out = text or ""
|
||||
for secret in secrets:
|
||||
if secret and len(secret) >= 4:
|
||||
out = out.replace(secret, "***")
|
||||
return out[:300]
|
||||
|
||||
|
||||
def _relative_name(fqdn: str, zone: str) -> str:
|
||||
"""Convert an absolute record name to the zone-relative form GoDaddy's API requires.
|
||||
|
||||
GoDaddy record names are RELATIVE to the zone with NO trailing dot, and the zone apex is the
|
||||
literal "@" — never an empty string (which would collapse the URL onto the zone-wide TXT
|
||||
endpoint) and never the domain name itself.
|
||||
|
||||
("_acme-challenge.example.com", "example.com") -> "_acme-challenge"
|
||||
("_acme-challenge.foo.bar.example.com", "example.com") -> "_acme-challenge.foo.bar"
|
||||
("example.com", "example.com") -> "@"
|
||||
"""
|
||||
f = (fqdn or "").rstrip(".").lower()
|
||||
z = (zone or "").rstrip(".").lower()
|
||||
if z and f == z:
|
||||
return "@"
|
||||
if z and f.endswith("." + z):
|
||||
return f[: -(len(z) + 1)]
|
||||
# Defensive: callers always pass a zone that _resolve_domain derived from this very name.
|
||||
return f or "@"
|
||||
|
||||
|
||||
def _rrset_path(zone: str, rel_name: str) -> str:
|
||||
"""Build the 4-segment record path `/domains/{zone}/records/TXT/{name}`.
|
||||
|
||||
SAFETY GATE: an empty rel_name would collapse the URL to `/domains/{zone}/records/TXT` — the
|
||||
endpoint that replaces EVERY TXT record in the zone (SPF, DKIM, DMARC, domain verifications).
|
||||
A "." or ".." segment does the same thing one step later: `quote()` leaves both untouched
|
||||
(they are unreserved) and yarl normalizes dot segments away when it builds the URL, so
|
||||
".../records/TXT/.." would resolve to ".../records" — the whole-zone endpoint. Refuse both
|
||||
rather than build them. `safe=''` percent-encodes the apex "@" as "%40" (accepted bare too,
|
||||
but safer through proxies); "_", "-" and "." are unreserved and pass through unchanged, so a
|
||||
multi-label relative name stays one readable path segment.
|
||||
"""
|
||||
if not zone or not rel_name:
|
||||
raise DnsProviderError("Internal error: refusing to build a zone-wide GoDaddy TXT record path.")
|
||||
if rel_name.strip(".") == "" or any(part in (".", "..") for part in rel_name.split("/")):
|
||||
raise DnsProviderError("Internal error: refusing to build a GoDaddy TXT path from a dot segment.")
|
||||
return f"/domains/{quote(zone, safe='')}/records/TXT/{quote(rel_name, safe='')}"
|
||||
|
||||
|
||||
def _live_values(records: List[Dict]) -> List[str]:
|
||||
"""The non-empty `data` values in an RRset read.
|
||||
|
||||
GoDaddy leaves tombstone rows with `"data": ""` behind at a name after some removals. Echoing
|
||||
one back in a PUT body is rejected with 422 INVALID_BODY, so every field implementation
|
||||
(lego, acme.sh, Posh-ACME) filters them independently — so do we.
|
||||
"""
|
||||
out: List[str] = []
|
||||
for rec in records or []:
|
||||
data = (rec or {}).get("data") or ""
|
||||
if data:
|
||||
out.append(data)
|
||||
return out
|
||||
|
||||
|
||||
def _merge_add(existing: List[Dict], value: str) -> Optional[List[Dict]]:
|
||||
"""PUT body that adds `value` while preserving every coexisting sibling value.
|
||||
|
||||
Returns None when `value` is already present — an idempotent no-op, which is where an ACME
|
||||
retry cycle lands.
|
||||
"""
|
||||
live = _live_values(existing)
|
||||
if value in live:
|
||||
return None
|
||||
return [{"data": d, "ttl": _TXT_TTL} for d in live] + [{"data": value, "ttl": _TXT_TTL}]
|
||||
|
||||
|
||||
def _merge_remove(existing: List[Dict], value: str) -> Optional[List[Dict]]:
|
||||
"""PUT body that removes ONLY `value`, keeping every sibling.
|
||||
|
||||
Three-state result, because GoDaddy needs three different calls:
|
||||
None -> `value` is not there; already gone, tolerate (base.py's remove contract).
|
||||
[] -> it was the last value; the caller must DELETE, since `PUT []` is rejected.
|
||||
list -> PUT this body.
|
||||
"""
|
||||
live = _live_values(existing)
|
||||
if value not in live:
|
||||
return None
|
||||
return [{"data": d, "ttl": _TXT_TTL} for d in live if d != value]
|
||||
|
||||
|
||||
def _require_rrset(body: Any) -> List[Dict]:
|
||||
"""The RRset read, or a refusal.
|
||||
|
||||
FAIL CLOSED. A read that did not come back as a JSON array must never be treated as "the RRset
|
||||
is empty" — the very next call is a full-RRset PUT, so coercing an unreadable read to [] would
|
||||
replace every coexisting sibling value with just ours. Failing instead is free: the orchestrator
|
||||
reverts the publish flag and retries next cycle, while a destructive PUT is unrecoverable.
|
||||
"""
|
||||
if not isinstance(body, list):
|
||||
raise DnsProviderError(
|
||||
"GoDaddy returned an unreadable TXT record list; refusing to replace the record set."
|
||||
)
|
||||
return body
|
||||
|
||||
|
||||
def _error_fields(body: Any) -> Tuple[str, str]:
|
||||
"""The whitelisted (code, message) pair from a GoDaddy error body.
|
||||
|
||||
Only these two string fields are ever read; the raw body is never interpolated into a
|
||||
user-facing message.
|
||||
"""
|
||||
if not isinstance(body, dict):
|
||||
return "", ""
|
||||
code = body.get("code")
|
||||
message = body.get("message")
|
||||
return (code if isinstance(code, str) else ""), (message if isinstance(message, str) else "")
|
||||
|
||||
|
||||
def _retry_after_seconds(headers, body: Any) -> int:
|
||||
"""Seconds to wait after a 429.
|
||||
|
||||
The current platform sends `Retry-After` and `ratelimit-reset` headers with no body, while the
|
||||
legacy v1 OpenAPI documents an `ErrorLimit` body carrying `retryAfterSec`. All three shapes are
|
||||
live in the wild — and so is none of them, hence the 60s default.
|
||||
"""
|
||||
for key in ("Retry-After", "ratelimit-reset"):
|
||||
raw = (headers or {}).get(key)
|
||||
if raw:
|
||||
try:
|
||||
return max(1, int(str(raw).strip()))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if isinstance(body, dict):
|
||||
raw = body.get("retryAfterSec")
|
||||
if isinstance(raw, int) and raw > 0:
|
||||
return raw
|
||||
return 60
|
||||
|
||||
|
||||
class _GoDaddyHTTPError(DnsProviderError):
|
||||
"""A DnsProviderError that also carries the HTTP status and GoDaddy `code`.
|
||||
|
||||
Callers INSIDE this module branch on the status (tolerate a 404 read-back, fall through a
|
||||
zone probe), while everything outside — dns01_orchestrator, letsencrypt.py — still sees a
|
||||
plain sanitized DnsProviderError and needs no change.
|
||||
"""
|
||||
|
||||
def __init__(self, message: str, status: int, code: str = ""):
|
||||
super().__init__(message)
|
||||
self.status = status
|
||||
self.code = code
|
||||
|
||||
|
||||
class GoDaddyDNSProvider(DnsProvider):
|
||||
name = "godaddy"
|
||||
label = "GoDaddy"
|
||||
automated = True
|
||||
credential_fields: List[Dict] = [
|
||||
{
|
||||
"key": "api_key",
|
||||
"label": "API Key",
|
||||
"type": "password",
|
||||
"required": True,
|
||||
"max_length": 200,
|
||||
"help": ("Production API Key from developer.godaddy.com/keys — the first key the dashboard "
|
||||
"issues is an OTE (test) key and will be rejected. A Personal Access Token also "
|
||||
"works: paste it here and leave the Secret blank."),
|
||||
},
|
||||
{
|
||||
"key": "api_secret",
|
||||
"label": "API Secret",
|
||||
"type": "password",
|
||||
"required": False,
|
||||
"max_length": 200,
|
||||
"help": ("The Secret half of the same API Key pair. Leave blank ONLY if the field above "
|
||||
"holds a Personal Access Token. The account also needs at least one registered "
|
||||
"domain for GoDaddy to allow DNS API access at all."),
|
||||
},
|
||||
]
|
||||
|
||||
def __init__(self, credentials: Dict[str, str] | None = None):
|
||||
super().__init__(credentials)
|
||||
# Normalize, never validate: dns01_orchestrator.py calls get_provider() OUTSIDE any
|
||||
# DnsProviderError guard, so a constructor that raised on malformed credentials would escape
|
||||
# as an unhandled exception in the 60s background cycle. The UI drops blank fields before
|
||||
# submitting, so a left-blank field arrives as a MISSING key rather than "" — `.get() or ""`
|
||||
# covers both.
|
||||
self._api_key = (self.credentials.get("api_key") or "").strip()
|
||||
self._api_secret = (self.credentials.get("api_secret") or "").strip()
|
||||
# Per-INSTANCE zone cache. A module-level cache would leak one ACME account's zone visibility
|
||||
# into another's; an instance lives for exactly one orchestrator step, which is precisely the
|
||||
# scope where caching pays off (apex + wildcard resolve the same zone from the same name).
|
||||
self._zone_cache: Dict[str, str] = {}
|
||||
|
||||
def _auth_header(self) -> str:
|
||||
"""`sso-key <key>:<secret>` when a Secret is present, else `Bearer <token>` for a PAT.
|
||||
|
||||
Literal prefix, one space, a single colon — no base64, no URL-encoding, no quotes. Keeping
|
||||
this as one swappable string is what makes GoDaddy's sso-key sunset a credential change
|
||||
rather than a code change.
|
||||
"""
|
||||
if self._api_secret:
|
||||
return f"sso-key {self._api_key}:{self._api_secret}"
|
||||
return f"Bearer {self._api_key}"
|
||||
|
||||
def _headers(self) -> Dict[str, str]:
|
||||
# Accept is not optional: these endpoints content-negotiate application/xml and
|
||||
# text/javascript. Content-Type is required on every write or GoDaddy answers 400/415.
|
||||
return {
|
||||
"Authorization": self._auth_header(),
|
||||
"Accept": "application/json",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
|
||||
def _http_error(self, status: int, code: str, message: str, retry_after: Optional[int]) -> _GoDaddyHTTPError:
|
||||
"""Map an HTTP status to a sanitized, operator-actionable DnsProviderError.
|
||||
|
||||
These strings land in acme_order_events and letsencrypt_orders.error_detail and are shown
|
||||
in the order timeline, so each one names what to fix. GoDaddy's own `code`/`message` is
|
||||
appended when present because the two 403 causes — account not eligible for the DNS API vs.
|
||||
a PAT missing `domains.dns:update` — are indistinguishable by status alone. Scrubbing
|
||||
happens HERE, at the single point where provider-supplied text enters a message, so a new
|
||||
caller cannot forget it.
|
||||
"""
|
||||
code = _scrub(code, self._api_key, self._api_secret)
|
||||
message = _scrub(message, self._api_key, self._api_secret)
|
||||
if status == 401:
|
||||
detail = ("GoDaddy rejected the API credentials. Check they are a PRODUCTION Key/Secret pair "
|
||||
"from developer.godaddy.com/keys — the first key the dashboard issues is an OTE "
|
||||
"(test) key and is not valid here.")
|
||||
elif status == 403:
|
||||
detail = ("GoDaddy denied access to the DNS API. The account needs at least one registered "
|
||||
"domain, and a Personal Access Token needs the domains.domain:read and "
|
||||
"domains.dns:update scopes.")
|
||||
elif status == 404:
|
||||
detail = ("GoDaddy has no zone for this domain (check it is registered in this account and "
|
||||
"uses GoDaddy nameservers).")
|
||||
elif status == 409:
|
||||
detail = "GoDaddy reports this domain is not eligible to have its DNS records changed."
|
||||
elif status == 422:
|
||||
detail = "GoDaddy rejected the record change as invalid (HTTP 422)."
|
||||
elif status == 429:
|
||||
detail = f"GoDaddy rate limit reached; retry in ~{retry_after or 60}s."
|
||||
else:
|
||||
detail = f"GoDaddy API error (HTTP {status})."
|
||||
if code or message:
|
||||
detail += f" (GoDaddy: {code}{': ' + message if message else ''})"
|
||||
return _GoDaddyHTTPError(detail, status=status, code=code)
|
||||
|
||||
async def _request(self, session: aiohttp.ClientSession, method: str, path: str, **kwargs) -> Any:
|
||||
"""One GoDaddy API call. Returns the parsed JSON body, or None for the empty-bodied writes.
|
||||
|
||||
Raises a SANITIZED _GoDaddyHTTPError / DnsProviderError — never the credentials, never the
|
||||
request, never a response body verbatim.
|
||||
"""
|
||||
url = f"{GODADDY_API_BASE}{path}"
|
||||
try:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
# content_type=None: every GoDaddy write answers 200/204 with an EMPTY body, and
|
||||
# aiohttp would otherwise raise on the missing/other content type before parsing.
|
||||
body = await resp.json(content_type=None)
|
||||
except ValueError:
|
||||
# ONLY a decode failure (JSONDecodeError subclasses ValueError) is swallowed —
|
||||
# an empty write body, or an HTML error page on a >=400. A transport failure
|
||||
# mid-read (ClientPayloadError, TimeoutError) must NOT land here: it would look
|
||||
# identical to "empty body", and a caller that reads an RRset would then see
|
||||
# None and could mistake it for an empty RRset. Those propagate to the handlers
|
||||
# below and become a real DnsProviderError.
|
||||
body = None
|
||||
# 2xx only. Redirects are deliberately not followed (aiohttp would forward the
|
||||
# Authorization header), so a 3xx is a failed call — treating `< 400` as success
|
||||
# would report a redirected write as a silent no-op.
|
||||
if 200 <= resp.status < 300:
|
||||
return body
|
||||
code, message = _error_fields(body)
|
||||
retry_after = _retry_after_seconds(resp.headers, body) if resp.status == 429 else None
|
||||
raise self._http_error(resp.status, code, message, retry_after)
|
||||
except DnsProviderError:
|
||||
raise
|
||||
except aiohttp.ClientError as exc:
|
||||
# Only the exception TYPE is interpolated: an aiohttp client error's str() can carry the
|
||||
# request URL, and the message is persisted to the order timeline.
|
||||
raise DnsProviderError(f"Could not reach the GoDaddy API ({type(exc).__name__}).")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise DnsProviderError(f"Unexpected GoDaddy API failure ({type(exc).__name__}).")
|
||||
|
||||
async def verify_credentials(self) -> Dict:
|
||||
if not self._api_key:
|
||||
return {"ok": False, "detail": "No GoDaddy API Key provided."}
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
# Cheapest read-only check: one request, no zone needed. Deliberately NOT
|
||||
# GET /v1/domains/{domain} — GoDaddy has rejected that details call for small
|
||||
# accounts since 2024-05 while record-level calls keep working, so verifying with it
|
||||
# produces false negatives on accounts where DNS-01 would succeed.
|
||||
body = await self._request(session, "GET", "/domains?limit=1")
|
||||
if not isinstance(body, list):
|
||||
return {"ok": False, "detail": "GoDaddy returned an unexpected response to the credential check."}
|
||||
if not body:
|
||||
# An empty list is NOT a failure: sub-zones delegated to GoDaddy nameservers are
|
||||
# manageable via the records API but never appear in the domain listing.
|
||||
return {"ok": True, "detail": ("GoDaddy credentials valid, but no domains are visible in this "
|
||||
"account — the domain you validate must be registered here, or "
|
||||
"be a zone delegated to GoDaddy nameservers.")}
|
||||
return {"ok": True, "detail": "GoDaddy credentials valid."}
|
||||
except DnsProviderError as exc:
|
||||
detail = str(exc)
|
||||
if not self._api_secret:
|
||||
# The Bearer path is silent otherwise, and a half-filled form is the likeliest cause.
|
||||
detail += (" Note: no API Secret was entered, so the API Key was sent as a Personal Access "
|
||||
"Token (Bearer). If you have a Key + Secret pair, enter both halves.")
|
||||
return {"ok": False, "detail": detail}
|
||||
except Exception: # noqa: BLE001 — never leak an internal/transport error verbatim
|
||||
return {"ok": False, "detail": "Could not verify the GoDaddy credentials."}
|
||||
|
||||
async def _resolve_domain(self, session: aiohttp.ClientSession, record_name: str) -> str:
|
||||
"""Find the most-specific (longest-suffix) GoDaddy-managed zone for an absolute record name.
|
||||
|
||||
GoDaddy has no `/zones?name=` equivalent, so this walks suffixes longest-to-shortest and
|
||||
probes `GET /v1/domains/{candidate}/records/NS`. That probe (rather than the domain listing
|
||||
or the domain-details call) is deliberate: it finds sub-zones delegated to GoDaddy
|
||||
nameservers, which never appear in `GET /v1/domains` at all, and it does not depend on the
|
||||
details endpoint that small accounts are rejected from.
|
||||
"""
|
||||
cached = self._zone_cache.get(record_name)
|
||||
if cached:
|
||||
return cached
|
||||
labels = record_name.rstrip(".").lower().split(".")
|
||||
for i in range(len(labels) - 1):
|
||||
candidate = ".".join(labels[i:])
|
||||
if candidate.count(".") < 1:
|
||||
break # a zone needs at least two labels
|
||||
try:
|
||||
body = await self._request(
|
||||
session, "GET", f"/domains/{quote(candidate, safe='')}/records/NS"
|
||||
)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status in (404, 422):
|
||||
continue # not a zone in this account — keep walking
|
||||
# 401/403/409/429/5xx are credential, eligibility or platform failures, not
|
||||
# "wrong zone". Continuing would burn the rate-limit budget re-failing on every
|
||||
# remaining suffix and would bury the real cause under "no managed domain".
|
||||
raise
|
||||
if isinstance(body, list) and body:
|
||||
self._zone_cache[record_name] = candidate
|
||||
return candidate
|
||||
raise DnsProviderError(f"No managed GoDaddy domain found for {record_name}.")
|
||||
|
||||
async def add_txt_record(self, name: str, value: str) -> None:
|
||||
async with _rrset_lock(name):
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
zone = await self._resolve_domain(session, name)
|
||||
path = _rrset_path(zone, _relative_name(name, zone))
|
||||
try:
|
||||
existing = await self._request(session, "GET", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status != 404:
|
||||
raise
|
||||
# Some accounts 404 reading back a record set in a zone whose WRITES succeed
|
||||
# (acme.sh #6517). Reachable only when the NS probe resolved the zone but the
|
||||
# TXT read 404s — if the NS probe itself 404s we never get here and the caller
|
||||
# sees "No managed GoDaddy domain found", which is the honest answer. We cannot
|
||||
# merge what we cannot read, and a single-value PUT would destroy any coexisting
|
||||
# sibling, so PATCH is the only correct recovery: it is the one genuinely
|
||||
# ADDITIVE primitive in v1 ("Appends DNS records ... Existing records with the
|
||||
# same type and name are preserved"). It cannot dedupe, but a duplicate
|
||||
# identical TXT is harmless for validation and cleanup removes the whole RRset.
|
||||
await self._request(
|
||||
session, "PATCH", f"/domains/{quote(zone, safe='')}/records",
|
||||
json=[{"type": "TXT", "name": _relative_name(name, zone),
|
||||
"data": value, "ttl": _TXT_TTL}],
|
||||
)
|
||||
return
|
||||
body = _merge_add(_require_rrset(existing), value)
|
||||
if body is None:
|
||||
return # already published — idempotent, this is where ACME retries land
|
||||
await self._request(session, "PUT", path, json=body)
|
||||
|
||||
async def remove_txt_record(self, name: str, value: str) -> None:
|
||||
async with _rrset_lock(name):
|
||||
async with aiohttp.ClientSession(timeout=_TIMEOUT) as session:
|
||||
try:
|
||||
zone = await self._resolve_domain(session, name)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
# Raise only what a later sweep could plausibly succeed at. reconcile_dns01_cleanup
|
||||
# swallows the error and leaves dns_record_cleaned FALSE, so the row is re-selected
|
||||
# every cycle — and its query takes a bare LIMIT 50, so rows that can NEVER succeed
|
||||
# (revoked key, account lost DNS-API eligibility) would monopolise the whole
|
||||
# cleanup budget and starve every other account. For those terminal statuses we
|
||||
# give up quietly: the orphaned `_acme-challenge` TXT is inert, and the same
|
||||
# credential failure is already loud on the publish path, where it is actionable.
|
||||
if exc.status == 429 or exc.status >= 500:
|
||||
raise
|
||||
return
|
||||
except DnsProviderError:
|
||||
return # zone genuinely not resolvable — nothing we could clean up
|
||||
path = _rrset_path(zone, _relative_name(name, zone))
|
||||
try:
|
||||
existing = await self._request(session, "GET", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status == 404:
|
||||
return # RRset (or the read) is gone — tolerate
|
||||
raise
|
||||
body = _merge_remove(_require_rrset(existing), value)
|
||||
if body is None:
|
||||
return # our value is not there — already gone, tolerate
|
||||
if not body:
|
||||
# The LAST value at this name. `PUT []` is rejected (422 INVALID_BODY, "Records
|
||||
# must be specified"), so emptying an RRset REQUIRES DELETE. This removes only
|
||||
# TXT at this exact name; other names and other record types are preserved.
|
||||
# Do NOT fall back to the "write an empty string to delete" folklore — that hack
|
||||
# is what creates the tombstone rows _live_values has to filter.
|
||||
try:
|
||||
await self._request(session, "DELETE", path)
|
||||
except _GoDaddyHTTPError as exc:
|
||||
if exc.status == 404:
|
||||
return # raced with another cleanup — tolerate
|
||||
raise
|
||||
return
|
||||
await self._request(session, "PUT", path, json=body)
|
||||
@@ -10,11 +10,13 @@ from typing import Dict, List, Type
|
||||
|
||||
from .base import DnsProvider
|
||||
from .cloudflare import CloudflareDNSProvider
|
||||
from .godaddy import GoDaddyDNSProvider
|
||||
from .manual import ManualDNSProvider
|
||||
|
||||
_PROVIDERS: Dict[str, Type[DnsProvider]] = {
|
||||
ManualDNSProvider.name: ManualDNSProvider,
|
||||
CloudflareDNSProvider.name: CloudflareDNSProvider,
|
||||
GoDaddyDNSProvider.name: GoDaddyDNSProvider,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -6,9 +6,98 @@ import json
|
||||
import urllib.parse
|
||||
from typing import Optional, List, Dict, Any
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.acme_backend_url import resolve_acme_backend_target
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Sentinel prefixes returned by `generate_haproxy_config_for_cluster` INSTEAD of a
|
||||
# configuration when generation fails. They are plain strings (not exceptions) for
|
||||
# historical reasons: the function's outer `except` swallows everything and returns
|
||||
# a one-line comment as the "config".
|
||||
#
|
||||
# That is a data-loss primitive on its own: an exception anywhere in the generator —
|
||||
# e.g. `urllib.parse.urlparse('http://host:99999').port` raising ValueError for an
|
||||
# out-of-range port in `acme_backend_url` — collapses a whole cluster's haproxy.cfg
|
||||
# into a single comment line, which the apply path then stores as APPLIED and pushes
|
||||
# to every agent. Callers that PERSIST the returned text MUST reject it first; use
|
||||
# `is_config_generation_error()` rather than matching the string by hand.
|
||||
CONFIG_GENERATION_ERROR_PREFIXES = (
|
||||
"# Error generating configuration:",
|
||||
"# Error: Cluster not found",
|
||||
)
|
||||
|
||||
|
||||
def is_config_generation_error(config_content: Optional[str]) -> bool:
|
||||
"""True when `config_content` is a generator failure sentinel, not a configuration.
|
||||
|
||||
Persisting or shipping a sentinel silently destroys a cluster's configuration, so
|
||||
every call site that writes the generator's output to `config_versions` (or hands
|
||||
it to an agent) must guard with this.
|
||||
"""
|
||||
if not config_content:
|
||||
return True
|
||||
return config_content.lstrip().startswith(CONFIG_GENERATION_ERROR_PREFIXES)
|
||||
|
||||
|
||||
def select_acme_backend_source(candidates: List[tuple]):
|
||||
"""Pick the first candidate that resolves into a usable address.
|
||||
|
||||
`candidates` is an ordered list of ``(source_name, url)`` from most to least
|
||||
specific. Returns ``(source_name, url, target, skipped)`` where ``skipped`` lists
|
||||
the ``(source_name, url, target)`` of candidates that were rejected.
|
||||
|
||||
Falling through on UNUSABLE values, not just empty ones, is the point. Values
|
||||
predating validation are common — the settings field was free text — and a
|
||||
scheme-less ``10.90.1.4:8080`` cannot be resolved. Stopping at the first non-empty
|
||||
candidate would emit a backend section with no ``server`` line: ``haproxy -c``
|
||||
still passes because the section exists, Apply succeeds, and every challenge
|
||||
request then 503s from an empty backend with nothing to show for it.
|
||||
"""
|
||||
skipped = []
|
||||
for source, url in candidates:
|
||||
if not url:
|
||||
continue
|
||||
target = resolve_acme_backend_target(url)
|
||||
if target.error_code:
|
||||
skipped.append((source, url, target))
|
||||
continue
|
||||
return source, url, target, skipped
|
||||
|
||||
# Nothing usable. Report against the last non-empty candidate so the rendered
|
||||
# comment and the log name a concrete value rather than an empty one.
|
||||
if skipped:
|
||||
source, url, target = skipped[-1]
|
||||
return source, url, target, skipped[:-1]
|
||||
source, url = candidates[0] if candidates else ('none', '')
|
||||
return source, url, resolve_acme_backend_target(url), skipped
|
||||
|
||||
|
||||
def extract_acme_backend_target(config_content: Optional[str]) -> Optional[str]:
|
||||
"""Return the `server _acme_mgmt` argument string from a rendered config.
|
||||
|
||||
e.g. ``"10.90.1.4:80"`` or ``"mgmt.internal:443 ssl verify none"``; ``None`` when
|
||||
the cluster renders no ACME challenge backend at all.
|
||||
|
||||
Used to decide whether an ACME-related edit actually CHANGES the shipped
|
||||
configuration. Comparing whole config texts would report a difference on every
|
||||
unrelated pending edit; comparing this one line answers the only question that
|
||||
matters here — "would the HAProxy nodes start talking to a different address?"
|
||||
"""
|
||||
if not config_content:
|
||||
return None
|
||||
in_section = False
|
||||
for line in config_content.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("backend "):
|
||||
in_section = stripped == "backend _acme_challenge_backend"
|
||||
continue
|
||||
if stripped.startswith(("frontend ", "listen ", "defaults", "global")):
|
||||
in_section = False
|
||||
continue
|
||||
if in_section and stripped.startswith("server _acme_mgmt "):
|
||||
return stripped[len("server _acme_mgmt "):].strip()
|
||||
return None
|
||||
|
||||
|
||||
def _format_redirect_rule(rule: Any) -> Optional[str]:
|
||||
"""Render a single redirect rule into a HAProxy `redirect ...` line.
|
||||
@@ -878,7 +967,15 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
f" bind {frontend['bind_address']}:{frontend['bind_port']}"
|
||||
)
|
||||
|
||||
config_lines.append(f" mode {frontend['mode']}")
|
||||
# `frontends.mode` is `VARCHAR(10) DEFAULT 'http'` but NULLABLE, and rows
|
||||
# can arrive with it unset via agent sync or config import. Interpolating
|
||||
# the raw value then emits a literal `mode None`, which HAProxy rejects —
|
||||
# taking down the whole cluster config, not just this frontend. Normalise
|
||||
# once here and use the result everywhere below, so the ACME gate, the
|
||||
# backend-mode check and the rendered line can never disagree with each
|
||||
# other about what mode this frontend is in.
|
||||
frontend_mode = (frontend.get('mode') or 'http').strip().lower()
|
||||
config_lines.append(f" mode {frontend_mode}")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# R2.3 / R3.3 (PR-1 hotfix): emit ordering buckets.
|
||||
@@ -990,7 +1087,7 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
_fe_buckets[cat].append(line)
|
||||
|
||||
# ACME HTTP-01 Challenge routing (auto-managed)
|
||||
if frontend['mode'] == 'http' and cluster_info.get('acme_enabled', False):
|
||||
if frontend_mode == 'http' and cluster_info.get('acme_enabled', False):
|
||||
_emit_fe(" acl is_acme_challenge path_beg /.well-known/acme-challenge/")
|
||||
_emit_fe(" http-request allow if is_acme_challenge")
|
||||
_emit_fe(" use_backend _acme_challenge_backend if is_acme_challenge")
|
||||
@@ -1020,14 +1117,14 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
if default_backend_name and default_backend_name not in ('[]', '{}', 'null', 'None'):
|
||||
backend_mode = backend_modes.get(default_backend_name)
|
||||
|
||||
if backend_mode and backend_mode != frontend['mode']:
|
||||
logger.error(f"CONFIG ERROR: Frontend '{frontend['name']}' mode '{frontend['mode']}' does not match backend '{default_backend_name}' mode '{backend_mode}'")
|
||||
if backend_mode and backend_mode != frontend_mode:
|
||||
logger.error(f"CONFIG ERROR: Frontend '{frontend['name']}' mode '{frontend_mode}' does not match backend '{default_backend_name}' mode '{backend_mode}'")
|
||||
# FIX-10 marker: 'BACKEND-MODE-WARNING' keyword in
|
||||
# the comment body routes it to the 'default_be'
|
||||
# bucket via _categorize_haproxy_directive, so the
|
||||
# warning emits next to the actual default_backend
|
||||
# directive instead of at the top of the block.
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: Backend '{default_backend_name}' has mode '{backend_mode}' but frontend has mode '{frontend['mode']}'")
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: Backend '{default_backend_name}' has mode '{backend_mode}' but frontend has mode '{frontend_mode}'")
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: HAProxy will reject this configuration! Please fix the mode mismatch in UI.")
|
||||
|
||||
_emit_fe(f" default_backend {default_backend_name}")
|
||||
@@ -1589,36 +1686,82 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
f"frontend found (would create orphan backend section)."
|
||||
)
|
||||
else:
|
||||
acme_url = cluster_info.get('acme_backend_url') or ''
|
||||
if not acme_url:
|
||||
try:
|
||||
acme_settings = await db_conn.fetchrow(
|
||||
"SELECT value FROM system_settings WHERE key = 'acme.challenge_backend_url'"
|
||||
)
|
||||
if acme_settings and acme_settings['value']:
|
||||
val = acme_settings['value']
|
||||
if isinstance(val, str):
|
||||
try:
|
||||
val = json.loads(val)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if val:
|
||||
acme_url = str(val)
|
||||
except Exception:
|
||||
pass
|
||||
if not acme_url:
|
||||
from config import MANAGEMENT_BASE_URL
|
||||
acme_url = MANAGEMENT_BASE_URL
|
||||
# Track WHERE the effective URL came from. Support has no way today to
|
||||
# tell an operator-set value from the shipped `localhost` default, and
|
||||
# this whole block emits no log line at all (contrast the skip branches
|
||||
# above), so a wrong challenge backend is invisible until Let's Encrypt
|
||||
# fails. `acme_source` is logged with the rendered host:port below.
|
||||
acme_candidates = [
|
||||
('cluster.acme_backend_url', cluster_info.get('acme_backend_url') or '')
|
||||
]
|
||||
_settings_url = ''
|
||||
try:
|
||||
acme_settings = await db_conn.fetchrow(
|
||||
"SELECT value FROM system_settings WHERE key = 'acme.challenge_backend_url'"
|
||||
)
|
||||
if acme_settings and acme_settings['value']:
|
||||
val = acme_settings['value']
|
||||
if isinstance(val, str):
|
||||
try:
|
||||
val = json.loads(val)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if val:
|
||||
_settings_url = str(val)
|
||||
except Exception:
|
||||
pass
|
||||
acme_candidates.append(
|
||||
('system_settings.acme.challenge_backend_url', _settings_url)
|
||||
)
|
||||
from config import MANAGEMENT_BASE_URL
|
||||
acme_candidates.append(('config.MANAGEMENT_BASE_URL', MANAGEMENT_BASE_URL))
|
||||
|
||||
parsed = urllib.parse.urlparse(acme_url)
|
||||
host = parsed.hostname or 'localhost'
|
||||
port = parsed.port or (443 if parsed.scheme == 'https' else 8080)
|
||||
ssl_flag = ' ssl verify none' if parsed.scheme == 'https' else ''
|
||||
acme_source, acme_url, target, _skipped = select_acme_backend_source(
|
||||
acme_candidates
|
||||
)
|
||||
for _s_source, _s_url, _s_target in _skipped:
|
||||
logger.warning(
|
||||
f"ACME-BACKEND: cluster {cluster_id} skipping unusable value from "
|
||||
f"{_s_source} (reason={_s_target.error_code}): "
|
||||
f"{_s_target.error_message} value={_s_url!r}"
|
||||
)
|
||||
|
||||
config_lines.append("# ACME Challenge Backend (auto-managed by HAProxy OpenManager)")
|
||||
config_lines.append("backend _acme_challenge_backend")
|
||||
config_lines.append(" mode http")
|
||||
config_lines.append(f" server _acme_mgmt {host}:{port}{ssl_flag}")
|
||||
if target.error_code:
|
||||
# The value cannot produce an address. Emit the section without a
|
||||
# `server` line rather than guessing: every HTTP frontend already
|
||||
# carries `use_backend _acme_challenge_backend`, and a use_backend
|
||||
# with no matching backend is fatal to `haproxy -c`. The operator's
|
||||
# raw value is deliberately NOT echoed into the file — a value
|
||||
# containing a newline would inject directives into a config pushed
|
||||
# to every node. It goes to the log instead.
|
||||
config_lines.append(
|
||||
f" # ACME challenge backend unavailable ({target.error_code}) — "
|
||||
f"see Cluster Management > ACME Challenge Backend URL"
|
||||
)
|
||||
logger.error(
|
||||
f"ACME-BACKEND: cluster {cluster_id} has an unusable challenge backend "
|
||||
f"URL (source={acme_source}, reason={target.error_code}): "
|
||||
f"{target.error_message} value={acme_url!r}"
|
||||
)
|
||||
else:
|
||||
config_lines.append(
|
||||
f" server _acme_mgmt {target.host}:{target.port}{target.ssl_flag}"
|
||||
)
|
||||
# One greppable line per render. `ACME-BACKEND` is the support
|
||||
# keyword: it answers "what address did we actually ship, and who
|
||||
# chose it?" without shell access to the node.
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} challenge backend rendered as "
|
||||
f"{target.host}:{target.port}{target.ssl_flag} "
|
||||
f"(source={acme_source}, url={acme_url!r})"
|
||||
)
|
||||
for _warning in target.warnings:
|
||||
logger.warning(
|
||||
f"ACME-BACKEND: cluster {cluster_id} (source={acme_source}): {_warning}"
|
||||
)
|
||||
config_lines.append("")
|
||||
|
||||
# Only close the connection if it was created within this function
|
||||
|
||||
@@ -0,0 +1,533 @@
|
||||
"""Issue #27 follow-up — parse an EXISTING keepalived.conf so a hand-maintained VIP can be
|
||||
adopted into OpenManager's model (v1.10.4).
|
||||
|
||||
Standalone and DB-free, like keepalived_config.py: the agent reports the file it found on a
|
||||
node, this module turns it into the fields `vip_instances` / `vip_members` need, and the
|
||||
adoption endpoint decides whether taking ownership is safe.
|
||||
|
||||
WHY A PARSER AND NOT THE HEARTBEAT. The heartbeat carries two keepalived facts —
|
||||
`keepalive_state` (MASTER/BACKUP, best-effort from logs) and `keepalive_ip` (the first
|
||||
address grepped out of `virtual_ipaddress`). Rendering a node's config needs eleven:
|
||||
virtual_router_id, auth_pass, interface, priority, prefix_length, advert_int, unicast
|
||||
peers, track_haproxy, role and the address itself. Guessing the missing ones is not a
|
||||
cosmetic risk — a wrong VRID puts the nodes in two separate VRRP domains and a wrong
|
||||
auth_pass makes them reject each other, and either way both nodes claim the VIP.
|
||||
|
||||
THE SAFETY CONTRACT. Adoption REPLACES the operator's file with our render, so anything in
|
||||
their file that `render_keepalived_conf` cannot reproduce would be silently destroyed on
|
||||
takeover — a `notify_master` failover hook, an LVS `virtual_server` section, a second
|
||||
address in one instance, a sync group. Extracting the fields is the easy half; the half
|
||||
that matters is `unsupported`, the list of directives we would drop. The caller must treat
|
||||
a non-empty `unsupported` as a refusal to adopt, not a warning to log.
|
||||
|
||||
Secrets: a parsed instance carries `auth_pass` in cleartext because that is the only way to
|
||||
re-render an identical config. NEVER log a parse result. Callers persist it through
|
||||
`encrypt_vrrp_secret` and mask it in anything UI-facing, exactly as the VIP router already
|
||||
does for `auth_pass` in version diffs.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
# Directives `render_keepalived_conf` emits, and therefore the only ones a takeover can
|
||||
# reproduce. Anything else found in a vrrp_instance is reported in `unsupported`.
|
||||
_SUPPORTED_INSTANCE_KEYS = {
|
||||
"state", "interface", "virtual_router_id", "priority", "advert_int",
|
||||
"authentication", "unicast_src_ip", "unicast_peer", "virtual_ipaddress", "track_script",
|
||||
}
|
||||
# Top-level blocks we can account for. `vrrp_script` is reproduced only when it is the
|
||||
# check script we generate ourselves (see _classify_script).
|
||||
_SUPPORTED_TOP_KEYS = {"global_defs", "vrrp_script", "vrrp_instance"}
|
||||
|
||||
# global_defs entries our render emits. An operator's file usually carries more (notification
|
||||
# email, router_id, ...) and losing those is a real change, so they are reported too.
|
||||
_SUPPORTED_GLOBAL_KEYS = {"enable_script_security", "script_user"}
|
||||
|
||||
_IDENT_RE = re.compile(r"^[A-Za-z0-9._:-]+$")
|
||||
|
||||
|
||||
class KeepalivedParseError(ValueError):
|
||||
"""The text is not a keepalived.conf we can reason about (unbalanced braces etc.)."""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tokenizer / block reader
|
||||
# ---------------------------------------------------------------------------
|
||||
def _strip_comment(line: str) -> str:
|
||||
"""Drop a trailing comment. keepalived treats BOTH `#` and `!` as comment starters, and
|
||||
neither is meaningful inside the quoted script paths we care about, so a quote-aware
|
||||
scan is enough (a `#` inside quotes stays)."""
|
||||
out: List[str] = []
|
||||
quote: Optional[str] = None
|
||||
for ch in line:
|
||||
if quote:
|
||||
out.append(ch)
|
||||
if ch == quote:
|
||||
quote = None
|
||||
continue
|
||||
if ch in ('"', "'"):
|
||||
quote = ch
|
||||
out.append(ch)
|
||||
continue
|
||||
if ch in ("#", "!"):
|
||||
break
|
||||
out.append(ch)
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def _split_tokens(line: str) -> List[str]:
|
||||
"""Whitespace split that keeps quoted strings whole and isolates braces, so
|
||||
`virtual_ipaddress { 10.0.0.1/24 dev eth0 }` tokenizes the same as its multi-line form."""
|
||||
tokens: List[str] = []
|
||||
buf: List[str] = []
|
||||
quote: Optional[str] = None
|
||||
|
||||
def flush() -> None:
|
||||
if buf:
|
||||
tokens.append("".join(buf))
|
||||
buf.clear()
|
||||
|
||||
for ch in line:
|
||||
if quote:
|
||||
if ch == quote:
|
||||
quote = None
|
||||
else:
|
||||
buf.append(ch)
|
||||
continue
|
||||
if ch in ('"', "'"):
|
||||
quote = ch
|
||||
continue
|
||||
if ch.isspace():
|
||||
flush()
|
||||
elif ch in ("{", "}"):
|
||||
flush()
|
||||
tokens.append(ch)
|
||||
else:
|
||||
buf.append(ch)
|
||||
flush()
|
||||
return tokens
|
||||
|
||||
|
||||
def _read_blocks(text: str) -> List[Dict[str, Any]]:
|
||||
"""Parse the file into nested entries.
|
||||
|
||||
Each entry is either
|
||||
{"kind": "block", "name": str, "args": [str], "body": [entries], "line": int}
|
||||
{"kind": "line", "tokens": [str], "line": int}
|
||||
|
||||
Line boundaries matter: inside `virtual_ipaddress` and `unicast_peer` each line is one
|
||||
bare value, so a flat token stream could not tell two addresses apart.
|
||||
"""
|
||||
root: List[Dict[str, Any]] = []
|
||||
stack: List[List[Dict[str, Any]]] = [root]
|
||||
# Blocks whose opening `{` we have seen, so a stray `}` can be reported with context.
|
||||
open_blocks: List[str] = []
|
||||
|
||||
for lineno, raw in enumerate(text.splitlines(), start=1):
|
||||
pending: List[str] = []
|
||||
for tok in _split_tokens(_strip_comment(raw)):
|
||||
if tok == "{":
|
||||
name = pending[0] if pending else ""
|
||||
args = pending[1:]
|
||||
block = {"kind": "block", "name": name, "args": args, "body": [], "line": lineno}
|
||||
stack[-1].append(block)
|
||||
stack.append(block["body"])
|
||||
open_blocks.append(name)
|
||||
pending = []
|
||||
elif tok == "}":
|
||||
if pending:
|
||||
stack[-1].append({"kind": "line", "tokens": pending, "line": lineno})
|
||||
pending = []
|
||||
if len(stack) == 1:
|
||||
raise KeepalivedParseError(f"unbalanced '}}' on line {lineno}")
|
||||
stack.pop()
|
||||
open_blocks.pop()
|
||||
else:
|
||||
pending.append(tok)
|
||||
if pending:
|
||||
stack[-1].append({"kind": "line", "tokens": pending, "line": lineno})
|
||||
|
||||
if len(stack) != 1:
|
||||
raise KeepalivedParseError(f"unclosed block '{open_blocks[-1] or '?'}' at end of file")
|
||||
return root
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Interpretation
|
||||
# ---------------------------------------------------------------------------
|
||||
def _as_int(tokens: List[str]) -> Optional[int]:
|
||||
if len(tokens) < 2:
|
||||
return None
|
||||
try:
|
||||
return int(tokens[1])
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _parse_vip_entry(tokens: List[str]) -> Optional[Dict[str, Any]]:
|
||||
"""One `virtual_ipaddress` line: `<addr>[/<prefix>] [dev <iface>] [label ...]`.
|
||||
|
||||
Returns None when the first token is not an address — a shape we do not understand must
|
||||
surface as unsupported rather than be silently dropped.
|
||||
"""
|
||||
spec = tokens[0]
|
||||
addr, _, prefix = spec.partition("/")
|
||||
try:
|
||||
ip = ipaddress.ip_address(addr)
|
||||
except ValueError:
|
||||
return None
|
||||
entry: Dict[str, Any] = {
|
||||
"address": str(ip),
|
||||
"prefix_length": None,
|
||||
"dev": None,
|
||||
"extra": [],
|
||||
}
|
||||
if prefix:
|
||||
try:
|
||||
entry["prefix_length"] = int(prefix)
|
||||
except ValueError:
|
||||
return None
|
||||
rest = tokens[1:]
|
||||
i = 0
|
||||
while i < len(rest):
|
||||
if rest[i] == "dev" and i + 1 < len(rest):
|
||||
entry["dev"] = rest[i + 1]
|
||||
i += 2
|
||||
continue
|
||||
# `label`, `scope`, `brd`, ... — all real directives we do not render.
|
||||
entry["extra"].append(rest[i])
|
||||
i += 1
|
||||
return entry
|
||||
|
||||
|
||||
def _classify_script(block: Dict[str, Any]) -> Tuple[str, Optional[str]]:
|
||||
"""Return (name, script_path) for a vrrp_script block."""
|
||||
name = block["args"][0] if block["args"] else (block["name"] or "")
|
||||
path = None
|
||||
for entry in block["body"]:
|
||||
if entry["kind"] == "line" and entry["tokens"] and entry["tokens"][0] == "script":
|
||||
path = " ".join(entry["tokens"][1:]) or None
|
||||
return name, path
|
||||
|
||||
|
||||
def _parse_instance(block: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Interpret one `vrrp_instance` block into VIP-model fields plus its own unsupported list."""
|
||||
inst: Dict[str, Any] = {
|
||||
"instance_name": block["args"][0] if block["args"] else "",
|
||||
"state": None,
|
||||
"interface": None,
|
||||
"virtual_router_id": None,
|
||||
"priority": None,
|
||||
"advert_int": None,
|
||||
"auth_type": None,
|
||||
"auth_pass": None,
|
||||
"unicast_src_ip": None,
|
||||
"unicast_peers": [],
|
||||
"virtual_ips": [],
|
||||
"track_scripts": [],
|
||||
"unsupported": [],
|
||||
"line": block["line"],
|
||||
}
|
||||
|
||||
def unsupported(what: str, lineno: int) -> None:
|
||||
inst["unsupported"].append({"directive": what, "line": lineno})
|
||||
|
||||
for entry in block["body"]:
|
||||
if entry["kind"] == "line":
|
||||
tokens = entry["tokens"]
|
||||
key = tokens[0]
|
||||
if key == "state":
|
||||
inst["state"] = (tokens[1].upper() if len(tokens) > 1 else None)
|
||||
elif key == "interface":
|
||||
inst["interface"] = tokens[1] if len(tokens) > 1 else None
|
||||
elif key == "virtual_router_id":
|
||||
inst["virtual_router_id"] = _as_int(tokens)
|
||||
elif key == "priority":
|
||||
inst["priority"] = _as_int(tokens)
|
||||
elif key == "advert_int":
|
||||
# keepalived accepts sub-second floats; our model column is an integer.
|
||||
raw = tokens[1] if len(tokens) > 1 else ""
|
||||
try:
|
||||
val = float(raw)
|
||||
except (TypeError, ValueError):
|
||||
val = None
|
||||
if val is None:
|
||||
unsupported(f"advert_int {raw}", entry["line"])
|
||||
elif val != int(val):
|
||||
# Rounding would change VRRP timing, so refuse rather than adopt-and-alter.
|
||||
unsupported(f"advert_int {raw} (fractional; model stores whole seconds)",
|
||||
entry["line"])
|
||||
else:
|
||||
inst["advert_int"] = int(val)
|
||||
elif key == "unicast_src_ip":
|
||||
inst["unicast_src_ip"] = tokens[1] if len(tokens) > 1 else None
|
||||
else:
|
||||
unsupported(" ".join(tokens), entry["line"])
|
||||
continue
|
||||
|
||||
name = entry["name"]
|
||||
if name == "authentication":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] != "line" or not sub["tokens"]:
|
||||
continue
|
||||
k = sub["tokens"][0]
|
||||
if k == "auth_type":
|
||||
inst["auth_type"] = (sub["tokens"][1].upper() if len(sub["tokens"]) > 1 else None)
|
||||
elif k == "auth_pass":
|
||||
# Everything after the keyword: a VRRP password may contain spaces.
|
||||
inst["auth_pass"] = " ".join(sub["tokens"][1:]) or None
|
||||
else:
|
||||
unsupported(f"authentication/{' '.join(sub['tokens'])}", sub["line"])
|
||||
elif name == "unicast_peer":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
inst["unicast_peers"].append(sub["tokens"][0])
|
||||
else:
|
||||
unsupported("unicast_peer/<block>", entry["line"])
|
||||
elif name == "virtual_ipaddress":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] != "line" or not sub["tokens"]:
|
||||
unsupported("virtual_ipaddress/<block>", entry["line"])
|
||||
continue
|
||||
parsed = _parse_vip_entry(sub["tokens"])
|
||||
if parsed is None:
|
||||
unsupported(f"virtual_ipaddress/{' '.join(sub['tokens'])}", sub["line"])
|
||||
else:
|
||||
if parsed["extra"]:
|
||||
unsupported(
|
||||
f"virtual_ipaddress/{parsed['address']} "
|
||||
f"({' '.join(parsed['extra'])})", sub["line"])
|
||||
inst["virtual_ips"].append(parsed)
|
||||
elif name == "track_script":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
inst["track_scripts"].append(sub["tokens"][0])
|
||||
else:
|
||||
unsupported(f"{name} {{...}}", entry["line"])
|
||||
|
||||
return inst
|
||||
|
||||
|
||||
def parse_keepalived_conf(text: str) -> Dict[str, Any]:
|
||||
"""Parse a keepalived.conf into VIP-model fields plus everything we could not model.
|
||||
|
||||
Raises KeepalivedParseError on structurally broken input. Never log the result: parsed
|
||||
instances carry `auth_pass` in cleartext.
|
||||
"""
|
||||
root = _read_blocks(text or "")
|
||||
result: Dict[str, Any] = {
|
||||
"instances": [],
|
||||
"scripts": {},
|
||||
"global_defs": {},
|
||||
"unsupported": [], # top-level directives our render would drop
|
||||
"sync_groups": [],
|
||||
}
|
||||
|
||||
for entry in root:
|
||||
if entry["kind"] == "line":
|
||||
# A bare top-level directive (e.g. `include /etc/keepalived/conf.d/*.conf`).
|
||||
result["unsupported"].append(
|
||||
{"directive": " ".join(entry["tokens"]), "line": entry["line"]})
|
||||
continue
|
||||
name = entry["name"]
|
||||
if name == "global_defs":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
key = sub["tokens"][0]
|
||||
result["global_defs"][key] = " ".join(sub["tokens"][1:])
|
||||
if key not in _SUPPORTED_GLOBAL_KEYS:
|
||||
result["unsupported"].append(
|
||||
{"directive": f"global_defs/{' '.join(sub['tokens'])}",
|
||||
"line": sub["line"]})
|
||||
else:
|
||||
result["unsupported"].append(
|
||||
{"directive": f"global_defs/{sub.get('name', '?')} {{...}}",
|
||||
"line": sub["line"]})
|
||||
elif name == "vrrp_script":
|
||||
script_name, path = _classify_script(entry)
|
||||
result["scripts"][script_name] = {"script": path, "line": entry["line"]}
|
||||
elif name == "vrrp_instance":
|
||||
result["instances"].append(_parse_instance(entry))
|
||||
elif name == "vrrp_sync_group":
|
||||
# A sync group ties instances together so they fail over as a unit. Our render has
|
||||
# no equivalent, and dropping it changes failover semantics — never adopt silently.
|
||||
group = entry["args"][0] if entry["args"] else ""
|
||||
result["sync_groups"].append({"name": group, "line": entry["line"]})
|
||||
result["unsupported"].append(
|
||||
{"directive": f"vrrp_sync_group {group}", "line": entry["line"]})
|
||||
else:
|
||||
# virtual_server (LVS), static_routes, bfd_instance, ...
|
||||
args = " ".join(entry["args"])
|
||||
result["unsupported"].append(
|
||||
{"directive": f"{name} {args} {{...}}".replace(" ", " "), "line": entry["line"]})
|
||||
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mapping to the VIP model + the adoption gate
|
||||
# ---------------------------------------------------------------------------
|
||||
# keepalived defaults we are willing to apply when a directive is absent, because the value
|
||||
# is unambiguous and re-rendering it changes nothing on the wire.
|
||||
_DEFAULT_ADVERT_INT = 1
|
||||
_DEFAULT_PRIORITY = 100
|
||||
_DEFAULT_STATE = "BACKUP"
|
||||
|
||||
# The only track_script our renderer emits (keepalived_config.build_haproxy_check_script).
|
||||
OUR_CHECK_SCRIPT_NAME = "chk_haproxy"
|
||||
|
||||
|
||||
def build_adoption_candidate(parsed: Dict[str, Any], instance: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Map one parsed `vrrp_instance` onto vip_instances / vip_members fields.
|
||||
|
||||
Returns `adoptable` plus `blockers`. A blocker means taking ownership would change what
|
||||
is running — either because our render cannot reproduce something in the file, or because
|
||||
a value we must write is not knowable from the file. Adoption REPLACES the operator's
|
||||
config, so "we could not read it" and "we would change it" are the same hazard, and both
|
||||
have to stop the flow rather than be logged.
|
||||
|
||||
Never log the return value: `vip.auth_pass` is cleartext.
|
||||
"""
|
||||
blockers: List[str] = []
|
||||
|
||||
# Directives we would drop. Report the file's own line numbers so the operator can look.
|
||||
dropped = list(parsed.get("unsupported") or []) + list(instance.get("unsupported") or [])
|
||||
for d in dropped:
|
||||
blockers.append(
|
||||
f"line {d['line']}: `{d['directive']}` — OpenManager's renderer cannot reproduce "
|
||||
f"this, so adopting would delete it")
|
||||
|
||||
vips = instance.get("virtual_ips") or []
|
||||
if len(vips) == 0:
|
||||
blockers.append("the instance declares no virtual_ipaddress — nothing to adopt")
|
||||
elif len(vips) > 1:
|
||||
addrs = ", ".join(v["address"] for v in vips)
|
||||
blockers.append(
|
||||
f"the instance carries {len(vips)} addresses ({addrs}); a managed VIP holds exactly "
|
||||
f"one, so adopting would drop all but the first")
|
||||
|
||||
vip_entry = vips[0] if vips else None
|
||||
|
||||
if instance.get("virtual_router_id") is None:
|
||||
blockers.append("no virtual_router_id — it cannot be guessed: a wrong VRID puts the "
|
||||
"nodes in separate VRRP domains and both would claim the VIP")
|
||||
if not instance.get("interface"):
|
||||
blockers.append("no interface — required to render the instance and the address")
|
||||
|
||||
# An explicit prefix is required. Our renderer ALWAYS writes `<addr>/<prefix>`, the model
|
||||
# column defaults to 24, and keepalived's own default for a bare address is a host route.
|
||||
# Picking either one for the operator would silently change the VIP's netmask, so ask.
|
||||
if vip_entry is not None and vip_entry.get("prefix_length") is None:
|
||||
blockers.append(
|
||||
f"`{vip_entry['address']}` has no explicit prefix length; state it during adoption "
|
||||
f"so the netmask cannot change on takeover")
|
||||
|
||||
# The address must live on the instance's interface — that is the only `dev` we can render.
|
||||
if vip_entry is not None and vip_entry.get("dev") and instance.get("interface") \
|
||||
and vip_entry["dev"] != instance["interface"]:
|
||||
blockers.append(
|
||||
f"the address is bound to `dev {vip_entry['dev']}` but the instance uses "
|
||||
f"`interface {instance['interface']}`; the render always uses the instance interface")
|
||||
|
||||
auth_type = instance.get("auth_type")
|
||||
if auth_type not in (None, "PASS"):
|
||||
blockers.append(f"auth_type {auth_type} is not supported (only PASS is rendered)")
|
||||
|
||||
# A tracked script that is not ours would be replaced by our HAProxy check.
|
||||
tracked = [t for t in (instance.get("track_scripts") or [])]
|
||||
foreign = [t for t in tracked if t != OUR_CHECK_SCRIPT_NAME]
|
||||
if foreign:
|
||||
blockers.append(
|
||||
f"track_script {', '.join(foreign)} would be replaced by OpenManager's HAProxy "
|
||||
f"health check")
|
||||
|
||||
state = instance.get("state") or _DEFAULT_STATE
|
||||
if state not in ("MASTER", "BACKUP"):
|
||||
blockers.append(f"state {state} is not MASTER or BACKUP")
|
||||
|
||||
peers = list(instance.get("unicast_peers") or [])
|
||||
src = instance.get("unicast_src_ip")
|
||||
# Our renderer emits unicast_src_ip and unicast_peer together, or neither.
|
||||
if bool(src) != bool(peers):
|
||||
which = "unicast_src_ip without unicast_peer" if src else "unicast_peer without unicast_src_ip"
|
||||
blockers.append(f"{which} — the render emits both or neither")
|
||||
|
||||
candidate: Dict[str, Any] = {
|
||||
"instance_name": instance.get("instance_name") or "",
|
||||
"adoptable": not blockers,
|
||||
"blockers": blockers,
|
||||
"dropped_directives": dropped,
|
||||
"vip": {
|
||||
"virtual_ip": vip_entry["address"] if vip_entry else None,
|
||||
"prefix_length": vip_entry.get("prefix_length") if vip_entry else None,
|
||||
"virtual_router_id": instance.get("virtual_router_id"),
|
||||
"advert_int": instance.get("advert_int") if instance.get("advert_int") is not None
|
||||
else _DEFAULT_ADVERT_INT,
|
||||
"use_unicast": bool(peers),
|
||||
"track_haproxy": OUR_CHECK_SCRIPT_NAME in tracked,
|
||||
"auth_pass": instance.get("auth_pass"),
|
||||
},
|
||||
"member": {
|
||||
"network_interface": instance.get("interface"),
|
||||
"role": state,
|
||||
"priority": instance.get("priority") if instance.get("priority") is not None
|
||||
else _DEFAULT_PRIORITY,
|
||||
},
|
||||
"peers": peers,
|
||||
"unicast_src_ip": src,
|
||||
# Which values came from a keepalived default rather than the file, so the UI can say so.
|
||||
"defaulted": [
|
||||
k for k, present in (
|
||||
("advert_int", instance.get("advert_int") is not None),
|
||||
("priority", instance.get("priority") is not None),
|
||||
("state", instance.get("state") is not None),
|
||||
) if not present
|
||||
],
|
||||
}
|
||||
return candidate
|
||||
|
||||
|
||||
# Substrings that identify the two blocker classes an operator is allowed to resolve. They are
|
||||
# matched rather than typed because the blocker text is what the UI shows; keeping the marker in
|
||||
# the sentence means the message and the rule cannot drift apart.
|
||||
_LOSS_MARKER = "would delete it"
|
||||
_PREFIX_MARKER = "no explicit prefix length"
|
||||
|
||||
|
||||
def remaining_blockers(blockers: List[str], *, prefix_supplied: bool = False,
|
||||
accept_data_loss: bool = False) -> List[str]:
|
||||
"""Blockers that survive what the operator is permitted to resolve.
|
||||
|
||||
Exactly two classes are resolvable, and the distinction is the whole safety argument:
|
||||
|
||||
* a missing prefix length is *unknown*, and the operator can supply it — we refuse to pick
|
||||
a netmask for a live VIP ourselves;
|
||||
* "our renderer cannot reproduce this, so adopting would delete it" is a *loss*, and losing
|
||||
it can be an informed choice.
|
||||
|
||||
Everything else — an unknown virtual_router_id, a fractional advert_int, an unsupported
|
||||
auth_type, an address on a different interface — is neither unknown nor a loss but an
|
||||
impossibility, and no flag may wave it through. This is the single source of truth for that
|
||||
rule; the endpoint and the UI both derive from it.
|
||||
"""
|
||||
out: List[str] = []
|
||||
for b in blockers or []:
|
||||
if prefix_supplied and _PREFIX_MARKER in b:
|
||||
continue
|
||||
if accept_data_loss and _LOSS_MARKER in b:
|
||||
continue
|
||||
out.append(b)
|
||||
return out
|
||||
|
||||
|
||||
def analyse_keepalived_conf(text: str) -> Dict[str, Any]:
|
||||
"""Parse + map in one call: the shape the discovery endpoint stores and the UI renders."""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
return {
|
||||
"instance_count": len(parsed["instances"]),
|
||||
"sync_groups": parsed["sync_groups"],
|
||||
"global_defs": parsed["global_defs"],
|
||||
"candidates": [build_adoption_candidate(parsed, inst) for inst in parsed["instances"]],
|
||||
}
|
||||
@@ -0,0 +1,291 @@
|
||||
"""ACME challenge backend URL validation, resolution and change detection.
|
||||
|
||||
These pin the behaviour behind a real incident: a split deployment rendered
|
||||
`server _acme_mgmt <mgmt>:8080` against a port with no listener, HTTP-01 failed for
|
||||
weeks while DNS-01 kept working, and every existing check reported success. The
|
||||
three mechanisms below are what make that impossible to repeat silently.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from services.haproxy_config import (
|
||||
extract_acme_backend_target,
|
||||
is_config_generation_error,
|
||||
select_acme_backend_source,
|
||||
)
|
||||
from utils.acme_backend_url import (
|
||||
AcmeBackendUrlError,
|
||||
resolve_acme_backend_target,
|
||||
validate_acme_backend_url,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Boundary validation — what an operator may type.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"value,expected",
|
||||
[
|
||||
("http://10.90.1.4:80", "http://10.90.1.4:80"),
|
||||
("http://10.90.1.4", "http://10.90.1.4"),
|
||||
("https://mgmt.internal:8443", "https://mgmt.internal:8443"),
|
||||
# RFC1918 is the NORMAL answer here, unlike utils/ssrf_guard's policy: the
|
||||
# operator is naming their own management host, which on a split deployment
|
||||
# is private by definition.
|
||||
("http://192.168.1.5:8080", "http://192.168.1.5:8080"),
|
||||
# Empty means "inherit from the next level of the resolution chain".
|
||||
(None, None),
|
||||
("", None),
|
||||
(" ", None),
|
||||
# Surrounding whitespace is normalised, not rejected — and the NORMALISED
|
||||
# value is what callers persist, so it can never reach haproxy.cfg.
|
||||
(" http://10.0.0.5:80 ", "http://10.0.0.5:80"),
|
||||
],
|
||||
)
|
||||
def test_accepts_and_normalises_usable_values(value, expected):
|
||||
assert validate_acme_backend_url(value) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"value,code",
|
||||
[
|
||||
# Scheme-less values used to be accepted and then silently became `localhost`
|
||||
# in the renderer — the trap that makes a correct diagnosis un-actionable.
|
||||
("10.90.1.4:8080", "no_scheme"),
|
||||
("localhost:8080", "no_scheme"),
|
||||
("ftp://10.0.0.5", "bad_scheme"),
|
||||
# A newline would inject directives into a file pushed to every node.
|
||||
("http://10.90.1.4\nbind :9", "whitespace"),
|
||||
("http://10.90.1.4 x", "whitespace"),
|
||||
# Loopback by number AND by name: `localhost` is what both shipped defaults
|
||||
# contain, so catching only the numeric form would miss the common case.
|
||||
("http://localhost:8080", "loopback"),
|
||||
("http://LOCALHOST", "loopback"),
|
||||
("http://127.0.0.1", "loopback"),
|
||||
("http://[::1]:80", "loopback"),
|
||||
("http://0.0.0.0:80", "unspecified"),
|
||||
("http://169.254.169.254", "link_local"),
|
||||
("http://u:p@10.0.0.5", "userinfo"),
|
||||
("http://10.0.0.5/api", "has_path"),
|
||||
("http://10.0.0.5?x=1", "has_path"),
|
||||
# urlparse defers port parsing to attribute access; unguarded this raises
|
||||
# inside the config generator and destroys the cluster's whole config.
|
||||
("http://10.0.0.5:99999", "bad_port"),
|
||||
("http://10.0.0.5:abc", "bad_port"),
|
||||
("http://-bad-.com", "invalid_host"),
|
||||
],
|
||||
)
|
||||
def test_rejects_unusable_values_with_stable_codes(value, code):
|
||||
with pytest.raises(AcmeBackendUrlError) as exc:
|
||||
validate_acme_backend_url(value)
|
||||
assert exc.value.code == code
|
||||
assert str(exc.value), "every rejection must carry operator-facing prose"
|
||||
|
||||
|
||||
def test_rejects_values_longer_than_the_column():
|
||||
# VARCHAR(500); without this the write fails as an opaque asyncpg 22001 -> 500.
|
||||
with pytest.raises(AcmeBackendUrlError) as exc:
|
||||
validate_acme_backend_url("http://" + "a" * 600 + ".com")
|
||||
assert exc.value.code == "too_long"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Render-time resolution — never rejects, never raises.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"url,host,port,ssl_flag",
|
||||
[
|
||||
("http://10.90.1.4:80", "10.90.1.4", 80, ""),
|
||||
# Port-less http stays 8080, NOT the scheme default 80: the bundled compose
|
||||
# publishes nginx on 8080, so installs relying on this have a working path
|
||||
# today and changing it would break them silently in the renewal loop.
|
||||
("http://10.90.1.4", "10.90.1.4", 8080, ""),
|
||||
("https://m.io", "m.io", 443, " ssl verify none"),
|
||||
("http://localhost:8080", "localhost", 8080, ""),
|
||||
],
|
||||
)
|
||||
def test_resolution_preserves_existing_rendering(url, host, port, ssl_flag):
|
||||
target = resolve_acme_backend_target(url)
|
||||
assert (target.host, target.port, target.ssl_flag) == (host, port, ssl_flag)
|
||||
assert target.error_code is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"url",
|
||||
["", None, " ", "10.0.0.5:80", "http://h:99999", "http://10.0.0.5\nx", "http://ba d"],
|
||||
)
|
||||
def test_resolution_reports_instead_of_raising(url):
|
||||
target = resolve_acme_backend_target(url)
|
||||
assert target.error_code, "unusable values must be reported, not raised"
|
||||
assert target.error_message
|
||||
|
||||
|
||||
def test_resolution_warns_on_loopback_rather_than_refusing():
|
||||
# Refusing here would make every existing install unappliable: the shipped
|
||||
# defaults ARE loopback, and the failure would block changes unrelated to ACME.
|
||||
target = resolve_acme_backend_target("http://localhost:8080")
|
||||
assert target.error_code is None
|
||||
assert target.warnings and "Loopback" in target.warnings[0]
|
||||
|
||||
|
||||
def test_resolution_warns_when_the_port_is_omitted():
|
||||
target = resolve_acme_backend_target("http://10.0.0.5")
|
||||
assert target.port == 8080
|
||||
assert any("port" in w.lower() for w in target.warnings)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Change detection — what makes a panel edit actually reach the nodes.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
_CONFIG = """global
|
||||
daemon
|
||||
|
||||
frontend fe_http
|
||||
bind 10.90.1.100:80
|
||||
mode http
|
||||
acl is_acme_challenge path_beg /.well-known/acme-challenge/
|
||||
use_backend _acme_challenge_backend if is_acme_challenge
|
||||
default_backend app
|
||||
|
||||
backend app
|
||||
server s1 10.0.0.9:8080
|
||||
|
||||
# ACME Challenge Backend (auto-managed by HAProxy OpenManager)
|
||||
backend _acme_challenge_backend
|
||||
mode http
|
||||
server _acme_mgmt 10.90.1.4:80
|
||||
"""
|
||||
|
||||
|
||||
def test_extracts_the_challenge_backend_target():
|
||||
assert extract_acme_backend_target(_CONFIG) == "10.90.1.4:80"
|
||||
|
||||
|
||||
def test_extracts_target_with_ssl_flag():
|
||||
cfg = _CONFIG.replace("10.90.1.4:80", "m.io:443 ssl verify none")
|
||||
assert extract_acme_backend_target(cfg) == "m.io:443 ssl verify none"
|
||||
|
||||
|
||||
def test_ignores_server_lines_in_other_backends():
|
||||
# Comparing the whole config would flag every unrelated pending edit as a change;
|
||||
# this must key on the ACME section alone.
|
||||
cfg = _CONFIG.replace("backend _acme_challenge_backend", "backend something_else")
|
||||
assert extract_acme_backend_target(cfg) is None
|
||||
|
||||
|
||||
def test_returns_none_when_the_section_has_no_server_line():
|
||||
cfg = (
|
||||
"backend _acme_challenge_backend\n"
|
||||
" mode http\n"
|
||||
" # ACME challenge backend unavailable (loopback)\n"
|
||||
)
|
||||
assert extract_acme_backend_target(cfg) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", ["", None])
|
||||
def test_extraction_tolerates_empty_input(value):
|
||||
assert extract_acme_backend_target(value) is None
|
||||
|
||||
|
||||
def test_url_change_that_renders_the_same_target_is_not_a_change():
|
||||
# `http://10.90.1.4` and `http://10.90.1.4:8080` are different strings but the
|
||||
# same shipped address; minting a config version for that would put a no-op
|
||||
# pending change in front of the operator.
|
||||
a = resolve_acme_backend_target("http://10.90.1.4")
|
||||
b = resolve_acme_backend_target("http://10.90.1.4:8080")
|
||||
assert (a.host, a.port, a.ssl_flag) == (b.host, b.port, b.ssl_flag)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Source selection — an unusable value must not shadow a usable one.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_prefers_the_most_specific_usable_source():
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "http://10.0.0.1:80"),
|
||||
("settings", "http://10.0.0.2:80"),
|
||||
("env", "http://10.0.0.3:80"),
|
||||
])
|
||||
assert (source, url, target.host) == ("cluster", "http://10.0.0.1:80", "10.0.0.1")
|
||||
assert skipped == []
|
||||
|
||||
|
||||
def test_skips_an_unusable_value_and_uses_the_next_source():
|
||||
# The regression this guards: a scheme-less value left over from the era when the
|
||||
# settings field was free text resolves to nothing. Stopping there would emit a
|
||||
# backend section with no `server` line — `haproxy -c` passes, Apply succeeds, and
|
||||
# every challenge request 503s with no visible cause.
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "10.90.1.4:8080"),
|
||||
("settings", ""),
|
||||
("env", "http://10.90.1.4:8080"),
|
||||
])
|
||||
assert source == "env"
|
||||
assert target.error_code is None and target.host == "10.90.1.4"
|
||||
assert [s[0] for s in skipped] == ["cluster"]
|
||||
|
||||
|
||||
def test_empty_sources_are_skipped_without_being_reported():
|
||||
source, _url, target, skipped = select_acme_backend_source([
|
||||
("cluster", ""),
|
||||
("settings", None),
|
||||
("env", "http://10.0.0.9:80"),
|
||||
])
|
||||
assert source == "env" and target.error_code is None
|
||||
assert skipped == []
|
||||
|
||||
|
||||
def test_reports_the_last_attempted_value_when_nothing_resolves():
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "10.0.0.1:80"),
|
||||
("env", "not a url"),
|
||||
])
|
||||
assert (source, url) == ("env", "not a url")
|
||||
assert target.error_code, "the caller needs an error to render and log"
|
||||
assert [s[0] for s in skipped] == ["cluster"]
|
||||
|
||||
|
||||
def test_all_sources_empty_yields_an_error_not_a_crash():
|
||||
source, _url, target, skipped = select_acme_backend_source([
|
||||
("cluster", ""), ("settings", ""), ("env", ""),
|
||||
])
|
||||
assert source == "cluster" and target.error_code == "empty" and skipped == []
|
||||
|
||||
|
||||
def test_loopback_is_usable_enough_to_render():
|
||||
# Warned about, never skipped: the shipped defaults are loopback, so treating it as
|
||||
# unusable would make the fallback chain fall off its own end on a stock install.
|
||||
source, _url, target, _skipped = select_acme_backend_source([
|
||||
("env", "http://localhost:8080"),
|
||||
])
|
||||
assert source == "env" and target.error_code is None
|
||||
assert target.warnings
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The generator's failure sentinel must never be mistaken for a config.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"content",
|
||||
[
|
||||
"# Error generating configuration: boom",
|
||||
"# Error: Cluster not found",
|
||||
"",
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_detects_generation_failure_sentinels(content):
|
||||
assert is_config_generation_error(content) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("content", [_CONFIG, "global\n daemon\n"])
|
||||
def test_real_configs_are_not_flagged(content):
|
||||
assert is_config_generation_error(content) is False
|
||||
@@ -104,13 +104,30 @@ async def test_check_dns_empty_ips_marks_failure(monkeypatch):
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Port-80 check (HEAD probe)
|
||||
# Port-80 check (GET probe)
|
||||
#
|
||||
# The probe is a GET, not a HEAD: a reverse proxy that has lost its
|
||||
# /.well-known/acme-challenge/ location falls through to its catch-all and serves
|
||||
# an SPA with HTTP 200, which a status-code-only check accepts as healthy while
|
||||
# every real validation fails. The fakes below therefore carry a body.
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
class _FakeHEADResp:
|
||||
def __init__(self, status):
|
||||
class _FakeContent:
|
||||
def __init__(self, body):
|
||||
self._body = body
|
||||
|
||||
async def read(self, n=-1):
|
||||
if self._body is None:
|
||||
raise ConnectionResetError("reset mid-body")
|
||||
return self._body if n is None or n < 0 else self._body[:n]
|
||||
|
||||
|
||||
class _FakeGETResp:
|
||||
def __init__(self, status, body=b"", content_type="text/plain"):
|
||||
self.status = status
|
||||
self.headers = {"content-type": content_type}
|
||||
self.content = _FakeContent(body)
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
@@ -120,10 +137,13 @@ class _FakeHEADResp:
|
||||
|
||||
|
||||
class _FakeSession:
|
||||
def __init__(self, *, statuses=None, raise_timeout=False, raise_client_error=False):
|
||||
def __init__(self, *, statuses=None, raise_timeout=False, raise_client_error=False,
|
||||
bodies=None, content_types=None):
|
||||
self._statuses = list(statuses or [])
|
||||
self._raise_timeout = raise_timeout
|
||||
self._raise_client_error = raise_client_error
|
||||
self._bodies = list(bodies or [])
|
||||
self._content_types = list(content_types or [])
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
@@ -131,14 +151,16 @@ class _FakeSession:
|
||||
async def __aexit__(self, *args):
|
||||
return False
|
||||
|
||||
def head(self, url, allow_redirects=False):
|
||||
def get(self, url, allow_redirects=False):
|
||||
if self._raise_timeout:
|
||||
raise asyncio.TimeoutError()
|
||||
if self._raise_client_error:
|
||||
import aiohttp
|
||||
raise aiohttp.ClientError("connection refused")
|
||||
status = self._statuses.pop(0) if self._statuses else 200
|
||||
return _FakeHEADResp(status)
|
||||
body = self._bodies.pop(0) if self._bodies else b""
|
||||
ctype = self._content_types.pop(0) if self._content_types else "text/plain"
|
||||
return _FakeGETResp(status, body, ctype)
|
||||
|
||||
|
||||
def _mock_public_dns(monkeypatch, ip="93.184.216.34"):
|
||||
@@ -178,6 +200,65 @@ async def test_check_port80_ok_on_404(monkeypatch):
|
||||
assert out["status"] == "ok"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warns_when_200_carries_a_web_page(monkeypatch):
|
||||
"""The failure that motivated the GET probe.
|
||||
|
||||
A reverse proxy whose /.well-known/acme-challenge/ location has drifted away
|
||||
falls through to its catch-all and serves the SPA. The status is 200, so the old
|
||||
`status in (200, 404)` rule called the install healthy while every validation
|
||||
failed. Only the body distinguishes them.
|
||||
"""
|
||||
_mock_public_dns(monkeypatch)
|
||||
spa = b'<!doctype html><html><head><title>HAProxy OpenManager</title></head>'
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[200], bodies=[spa], content_types=["text/html"])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "warn"
|
||||
target = out["details"]["targets"][0]
|
||||
assert target["body_class"] == "html"
|
||||
assert not target.get("ok")
|
||||
assert "HTML" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_falls_back_to_status_when_the_body_cannot_be_read(monkeypatch):
|
||||
"""A body that cannot be read is missing evidence, not a verdict.
|
||||
|
||||
Turning a connection reset mid-response into a hard failure would make a healthy
|
||||
404 fail intermittently, so the check keeps its original status-only semantics
|
||||
whenever there is nothing to judge.
|
||||
"""
|
||||
_mock_public_dns(monkeypatch)
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[404], bodies=[None])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "ok"
|
||||
assert out["details"]["targets"][0]["body_class"] == "unread"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warns_on_a_redirect(monkeypatch):
|
||||
_mock_public_dns(monkeypatch)
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[301])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "warn"
|
||||
assert out["details"]["targets"][0]["diagnosis"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warn_on_egress_timeout(monkeypatch):
|
||||
"""Corporate egress blocks port 80 outbound — warn, don't fail."""
|
||||
@@ -337,16 +418,92 @@ async def test_check_routing_fail_when_no_port80_frontend():
|
||||
assert "No HTTP frontend" in out["message"]
|
||||
|
||||
|
||||
def _routing_row(**over):
|
||||
row = {"id": 1, "name": "fe-http", "bind_address": "0.0.0.0", "bind_port": 80,
|
||||
"mode": "http", "default_backend": "be", "cluster_id": 1, "acme_enabled": True}
|
||||
row.update(over)
|
||||
return row
|
||||
|
||||
|
||||
def _applied(has_route=True, server_line="server _acme_mgmt 10.90.1.4:80"):
|
||||
return {"has_route": has_route, "server_line": server_line}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_ok_when_port80_frontend_present():
|
||||
async def test_check_routing_ok_when_challenge_route_is_in_the_applied_config():
|
||||
# A port-80 frontend row alone is NOT enough. It describes what the database
|
||||
# wants; the nodes run whatever was last applied. During the incident this
|
||||
# function reported "ok" from the row count while the live config had no usable
|
||||
# challenge route at all.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [
|
||||
{"id": 1, "name": "fe-http", "bind_address": "0.0.0.0", "bind_port": 80,
|
||||
"mode": "http", "default_backend": "be"},
|
||||
]
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied()
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "ok"
|
||||
assert len(out["details"]["frontends"]) == 1
|
||||
assert out["details"]["challenge_backends"] == {1: "10.90.1.4:80"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_acme_is_disabled_on_the_cluster():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(acme_enabled=False)]
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "disabled" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_route_is_not_in_the_applied_config():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(has_route=False)
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "apply the cluster" in out["message"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_challenge_backend_has_no_server_line():
|
||||
# `haproxy -c` passes because the section exists, so nothing else catches this;
|
||||
# every challenge request 503s from an empty backend.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(server_line=None)
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "503" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_challenge_backend_is_loopback():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(server_line="server _acme_mgmt 127.0.0.1:8080")
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "HAProxy node" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_rather_than_fails_on_a_tcp_only_port80_cluster():
|
||||
# The `fail` branch must stay reachable only when NO port-80 frontend exists at
|
||||
# all: SiteWizard blocks submit on any failing check, so turning this into a
|
||||
# failure would lock tcp-only installs the day it ships.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(mode="tcp")]
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "tcp mode" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_treats_null_mode_as_http_like_the_renderer():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(mode=None)]
|
||||
conn.fetchrow.return_value = _applied()
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "ok"
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Issue #53 (v1.10.1) — at-rest encryption for the pending CSR private key.
|
||||
|
||||
Pure logic: no DB, no network. Covers the round-trip, the backward-compatible read of rows
|
||||
written before this release, the unrecoverable-key path after a key rotation, and a static
|
||||
assertion that the write path can no longer store a raw PEM.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-csr-encryption-unit-tests")
|
||||
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
from utils.csr_key_crypto import (
|
||||
decrypt_csr_private_key,
|
||||
encrypt_csr_private_key,
|
||||
is_encrypted,
|
||||
reset_fernet_for_tests,
|
||||
)
|
||||
|
||||
_SAMPLE_PEM = (
|
||||
"-----BEGIN PRIVATE KEY-----\n"
|
||||
"MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj\n"
|
||||
"-----END PRIVATE KEY-----\n"
|
||||
)
|
||||
|
||||
|
||||
def test_roundtrip_and_ciphertext_does_not_contain_the_key():
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# The stored form must not be the PEM, and must not leak any recognisable fragment of it.
|
||||
assert token != _SAMPLE_PEM
|
||||
assert "-----BEGIN" not in token
|
||||
assert "MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj" not in token
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_is_encrypted_discriminates_token_from_legacy_pem():
|
||||
reset_fernet_for_tests()
|
||||
assert is_encrypted(encrypt_csr_private_key(_SAMPLE_PEM)) is True
|
||||
assert is_encrypted(_SAMPLE_PEM) is False
|
||||
assert is_encrypted("") is False
|
||||
assert is_encrypted(None) is False
|
||||
|
||||
|
||||
def test_legacy_plaintext_row_is_read_unchanged():
|
||||
# Rows written before v1.10.1 hold a raw PEM. They must keep working with NO data migration,
|
||||
# otherwise upgrading would strand every CSR that is out for signature.
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(_SAMPLE_PEM) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_empty_or_missing_value_returns_none():
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(None) is None
|
||||
assert decrypt_csr_private_key("") is None
|
||||
|
||||
|
||||
def test_key_rotation_makes_the_stored_key_unrecoverable_rather_than_wrong():
|
||||
"""After a rotation the caller must get None, never a silently wrong key."""
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
|
||||
# Rotate: an explicit, different CSR_ENCRYPTION_KEY takes precedence over the derived one.
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = Fernet.generate_key().decode()
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) is None
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_explicit_env_key_is_used_and_survives_reset():
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
key = Fernet.generate_key().decode()
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = key
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# Decryptable with the same explicit key from a fresh instance...
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
# ...and independently verifiable with the raw Fernet key.
|
||||
assert Fernet(key.encode()).decrypt(token.encode()).decode() == _SAMPLE_PEM
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_derivation_uses_its_own_hkdf_info_string():
|
||||
"""Each secret class derives an independent key, so rotating one never affects another."""
|
||||
src = (Path(__file__).resolve().parent.parent / "utils" / "csr_key_crypto.py").read_text()
|
||||
assert b"csr-private-key-v1".decode() in src
|
||||
# Must NOT reuse another class's info string.
|
||||
for foreign in ("dns-provider-creds-v1", "vip-vrrp-secret-v1", "mfa-totp-secret-v1"):
|
||||
assert foreign not in src, f"CSR key derivation must not reuse the {foreign} info string"
|
||||
|
||||
|
||||
def test_write_path_stores_the_encrypted_form_not_the_pem():
|
||||
"""Static pin: insert_csr_row must encrypt before the INSERT.
|
||||
|
||||
A future refactor that passed bundle['private_key_pem'] straight through would silently
|
||||
reintroduce plaintext storage, and no unit test with a mocked connection would notice.
|
||||
"""
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
insert_fn = src[src.index("async def insert_csr_row("):]
|
||||
insert_fn = insert_fn[: insert_fn.index("\nasync def ")]
|
||||
assert "encrypt_csr_private_key(bundle['private_key_pem'])" in insert_fn
|
||||
# The raw PEM must not be a bind parameter of the INSERT itself.
|
||||
assert not re.search(r"^\s*bundle\['private_key_pem'\],\s*$", insert_fn, re.M)
|
||||
|
||||
|
||||
def test_import_path_decrypts_and_fails_closed_on_unrecoverable_key():
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
fn = src[src.index("async def import_signed_certificate("):]
|
||||
assert "decrypt_csr_private_key(row['private_key_pem'])" in fn
|
||||
# A None decrypt must raise rather than fall through to the key-match comparison.
|
||||
assert "cannot be decrypted" in fn
|
||||
+399
-4
@@ -2,7 +2,8 @@
|
||||
|
||||
Covers the TXT-value math (RFC 8555 §8.4 — raw SHA-256 digest, base64url, NOT hex),
|
||||
the _acme-challenge record-name derivation (wildcard stripping), credential encryption
|
||||
round-trip + tamper handling, and the DNS provider registry/allow-list.
|
||||
round-trip + tamper handling, the DNS provider registry/allow-list, and (v1.10.0) the
|
||||
GoDaddy provider's zone-relative name derivation and additive RRset merge math.
|
||||
"""
|
||||
import base64
|
||||
import hashlib
|
||||
@@ -53,9 +54,9 @@ def test_decrypt_invalid_token_returns_none():
|
||||
|
||||
def test_provider_registry_and_allow_list():
|
||||
names = {p["name"] for p in list_providers()}
|
||||
assert {"manual", "cloudflare"} <= names
|
||||
assert is_supported("manual") and is_supported("cloudflare")
|
||||
assert not is_supported("route53") # not in MVP allow-list
|
||||
assert {"manual", "cloudflare", "godaddy"} <= names
|
||||
assert is_supported("manual") and is_supported("cloudflare") and is_supported("godaddy")
|
||||
assert not is_supported("route53") # not in the allow-list
|
||||
|
||||
assert get_provider("manual").automated is False
|
||||
cf = get_provider("cloudflare", {"api_token": "x"})
|
||||
@@ -91,6 +92,400 @@ def test_cloudflare_token_sanitize():
|
||||
assert p._raw_token == '"my-token_123"'
|
||||
|
||||
|
||||
# --- v1.10.0: GoDaddy provider (pure logic only — no network, no DB) ---
|
||||
|
||||
|
||||
def test_godaddy_credential_fields_schema():
|
||||
# Re-assert DnsCredentialsUpsert's validator rules directly against the declared schema, so the
|
||||
# UI can never render a field whose submission the API would reject with a 422.
|
||||
import re
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
fields = GoDaddyDNSProvider.credential_fields
|
||||
assert [f["key"] for f in fields] == ["api_key", "api_secret"]
|
||||
for f in fields:
|
||||
assert re.match(r"^[a-zA-Z0-9_]{1,50}$", f["key"]) # DnsCredentialsUpsert key regex
|
||||
assert f["type"] == "password" # renders Input.Password, not Input
|
||||
assert isinstance(f["max_length"], int) and 0 < f["max_length"] <= 4000 # validator value cap
|
||||
assert f["help"] and isinstance(f["help"], str) # shown in the Form.Item `extra` slot
|
||||
# api_secret is optional on purpose: leaving it blank is how a Personal Access Token is used
|
||||
# (Bearer), which is the migration path off the sso-key scheme GoDaddy is retiring.
|
||||
assert fields[0]["required"] is True and fields[1]["required"] is False
|
||||
# Must not reuse Cloudflare's field name: the register modal's credential Form.Items are named
|
||||
# cred_<key> in a SHARED form and are not cleared when the provider dropdown changes.
|
||||
assert "api_token" not in {f["key"] for f in fields}
|
||||
|
||||
|
||||
def test_godaddy_provider_is_automated():
|
||||
p = get_provider("godaddy", {"api_key": "k", "api_secret": "s"})
|
||||
assert p.automated is True # else the orchestrator takes the manual-confirm branch
|
||||
assert p.name == "godaddy" and 1 <= len(p.name) <= 50 # dns_provider Field(min_length=1, max_length=50)
|
||||
assert p.label == "GoDaddy"
|
||||
|
||||
|
||||
def test_godaddy_missing_credentials_returns_not_ok():
|
||||
# verify_credentials must RETURN {"ok": False}, never raise: the router turns any non-
|
||||
# DnsProviderError into the information-free generic 422 and the user never sees the reason.
|
||||
import asyncio
|
||||
|
||||
for creds in ({}, {"api_secret": "s"}): # blank UI fields arrive as MISSING keys, not ""
|
||||
r = asyncio.run(get_provider("godaddy", creds).verify_credentials())
|
||||
assert r["ok"] is False and r["detail"]
|
||||
# Short-circuits before any request, so this touches no network.
|
||||
|
||||
|
||||
def test_godaddy_auth_header_formats_and_secret_never_leaks():
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider, _scrub
|
||||
|
||||
sentinel = "SENTINEL-SECRET-DO-NOT-LEAK"
|
||||
p = GoDaddyDNSProvider({"api_key": "KEY123", "api_secret": sentinel})
|
||||
# Literal prefix, one space, a single colon — no base64, no quoting.
|
||||
assert p._auth_header() == f"sso-key KEY123:{sentinel}"
|
||||
# No secret -> Personal Access Token. This one branch is the whole sso-key-sunset migration.
|
||||
assert GoDaddyDNSProvider({"api_key": "PAT"})._auth_header() == "Bearer PAT"
|
||||
# _scrub removes credential substrings from anything bound for a log or an order event.
|
||||
assert sentinel not in _scrub(f"boom {sentinel} boom", "KEY123", sentinel)
|
||||
assert "KEY123" not in _scrub("boom KEY123", "KEY123", sentinel)
|
||||
assert _scrub("x" * 500, "KEY123") == "x" * 300 # bounded, so a huge body can't flood an event
|
||||
|
||||
# The channel that actually persists text: _http_error composes the message an order event and
|
||||
# letsencrypt_orders.error_detail will carry, so it must scrub its own inputs — a caller that
|
||||
# forgets to pre-scrub must not be able to leak. (Regression guard: scrubbing used to live at
|
||||
# the single call site in _request instead of here.)
|
||||
exc = p._http_error(403, f"DENIED_{sentinel}", f"token {sentinel} rejected", None)
|
||||
assert sentinel not in str(exc) and "***" in str(exc)
|
||||
|
||||
# This module must not log at all — logging is the one channel _scrub cannot reach, since the
|
||||
# arguments would be formatted by the logging framework rather than passed through it.
|
||||
import inspect
|
||||
import re as _re
|
||||
from services.dns_providers import godaddy as gd_mod
|
||||
|
||||
assert not _re.search(r"\blogger\.\w+\(", inspect.getsource(gd_mod)), \
|
||||
"godaddy.py must not log; surface everything through DnsProviderError so it is scrubbed"
|
||||
|
||||
|
||||
def test_godaddy_relative_record_name():
|
||||
# GoDaddy names are RELATIVE to the zone with no trailing dot; the apex is the literal "@".
|
||||
from services.dns_providers.godaddy import _relative_name
|
||||
|
||||
assert _relative_name("_acme-challenge.example.com", "example.com") == "_acme-challenge"
|
||||
assert _relative_name("_acme-challenge.foo.bar.example.com", "example.com") == "_acme-challenge.foo.bar"
|
||||
assert _relative_name("example.com", "example.com") == "@" # never "" — see _rrset_path
|
||||
assert _relative_name("_acme-challenge.example.com.", "example.com") == "_acme-challenge"
|
||||
assert _relative_name("_ACME-Challenge.Example.COM", "example.com") == "_acme-challenge"
|
||||
# Apex and wildcard produce the SAME relative name — which is exactly why the merge below
|
||||
# has to be additive.
|
||||
apex = ACMEService._challenge_dns_name("example.com")
|
||||
wild = ACMEService._challenge_dns_name("*.example.com")
|
||||
assert _relative_name(apex, "example.com") == _relative_name(wild, "example.com") == "_acme-challenge"
|
||||
|
||||
|
||||
def test_godaddy_rrset_merge_is_additive():
|
||||
# THE critical test: GoDaddy's PUT REPLACES an entire RRset, so the merge math is the only thing
|
||||
# keeping a wildcard+apex certificate's two coexisting TXT values alive.
|
||||
from services.dns_providers.godaddy import _live_values, _merge_add, _merge_remove
|
||||
|
||||
def vals(body):
|
||||
return sorted(r["data"] for r in body)
|
||||
|
||||
assert vals(_merge_add([{"data": "valueA", "ttl": 600}], "valueB")) == ["valueA", "valueB"]
|
||||
assert _merge_add([{"data": "valueA"}], "valueA") is None # idempotent; ACME retries land here
|
||||
# Total, not an all()-over-a-computed-list (which passes vacuously on an empty result): the
|
||||
# first publish at a fresh name must emit exactly one element, carrying the 600s TTL floor.
|
||||
assert _merge_add([], "v") == [{"data": "v", "ttl": 600}] # below 600 GoDaddy answers 422
|
||||
# Tombstone rows ({"data": ""}) must never be echoed back — GoDaddy answers 422 INVALID_BODY.
|
||||
assert _live_values([{"data": ""}, {"data": "x"}, {}]) == ["x"]
|
||||
assert vals(_merge_add([{"data": ""}, {"data": "valueA"}], "valueB")) == ["valueA", "valueB"]
|
||||
|
||||
assert vals(_merge_remove([{"data": "valueA"}, {"data": "valueB"}], "valueB")) == ["valueA"]
|
||||
assert _merge_remove([{"data": "valueA"}], "valueZ") is None # already gone — tolerate
|
||||
assert _merge_remove([], "valueZ") is None
|
||||
# [] means "use DELETE": PUT with an empty array is rejected (422, "Records must be specified").
|
||||
assert _merge_remove([{"data": "valueA"}], "valueA") == []
|
||||
assert _merge_remove([{"data": ""}, {"data": "valueA"}], "valueA") == []
|
||||
|
||||
|
||||
def test_godaddy_never_builds_a_zone_wide_txt_path():
|
||||
# A 3-segment path (.../records/TXT) is the endpoint that wipes EVERY TXT in the zone — SPF,
|
||||
# DKIM, DMARC, domain verifications. An empty relative name must never be able to produce it.
|
||||
from services.dns_providers.godaddy import _rrset_path
|
||||
|
||||
p = _rrset_path("example.com", "_acme-challenge")
|
||||
assert p == "/domains/example.com/records/TXT/_acme-challenge"
|
||||
assert p.count("/") == 5 and not p.endswith("/TXT")
|
||||
assert _rrset_path("example.com", "@").endswith("/%40") # apex percent-encoded for proxy safety
|
||||
# "." and ".." survive quote() and are then normalized away by yarl when the URL is built, so
|
||||
# ".../records/TXT/.." would resolve to the whole-zone endpoint. They must be refused too.
|
||||
for bad in [("example.com", ""), ("", "_acme-challenge"), ("example.com", "."),
|
||||
("example.com", ".."), ("example.com", "...")]:
|
||||
raised = False
|
||||
try:
|
||||
_rrset_path(*bad)
|
||||
except DnsProviderError:
|
||||
raised = True
|
||||
assert raised, f"_rrset_path{bad} must refuse to build a zone-wide TXT path"
|
||||
# And the only way to reach those inputs — a malformed domain — really does produce them.
|
||||
from services.dns_providers.godaddy import _relative_name as _rel
|
||||
assert _rel("..example.com", "example.com") == "."
|
||||
|
||||
|
||||
def test_godaddy_credential_encryption_roundtrip():
|
||||
# The two-field credential dict rides the same Fernet blob as Cloudflare's single token.
|
||||
reset_fernet_for_tests()
|
||||
creds = {"api_key": "gd-key-plaintext", "api_secret": "gd-secret-plaintext"}
|
||||
token = encrypt_dns_credentials(creds)
|
||||
assert "gd-key-plaintext" not in token and "gd-secret-plaintext" not in token # ciphertext
|
||||
assert decrypt_dns_credentials(token) == creds
|
||||
# This sorted key list is exactly what GET /dns-credentials exposes as credential_fields_present
|
||||
# — names only, never values.
|
||||
assert sorted(decrypt_dns_credentials(token).keys()) == ["api_key", "api_secret"]
|
||||
|
||||
|
||||
_GD_NS = "/domains/example.com/records/NS"
|
||||
_GD_TXT = "/domains/example.com/records/TXT/_acme-challenge"
|
||||
|
||||
|
||||
def _gd_provider(responses):
|
||||
"""A GoDaddy provider whose _request is replaced by a recorder.
|
||||
|
||||
The pure-merge tests above prove the MATH; this proves the WRITE PATH actually uses it. Without
|
||||
it, replacing the merge with a single-value PUT — the mutation that silently destroys the
|
||||
sibling value of every wildcard+apex certificate — leaves the whole suite green.
|
||||
|
||||
`responses` maps (method, path) -> value to return, or an Exception to raise. Unmapped calls
|
||||
return None, which is how the zone suffix-walk's failed probes are modelled.
|
||||
"""
|
||||
import types
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
calls = []
|
||||
|
||||
async def _fake_request(self, session, method, path, **kwargs):
|
||||
calls.append((method, path, kwargs.get("json")))
|
||||
result = responses.get((method, path))
|
||||
if isinstance(result, Exception):
|
||||
raise result
|
||||
return result
|
||||
|
||||
p = GoDaddyDNSProvider({"api_key": "k", "api_secret": "s"})
|
||||
p._request = types.MethodType(_fake_request, p)
|
||||
return p, calls
|
||||
|
||||
|
||||
def _assert_never_zone_wide(calls):
|
||||
# A write to .../records or .../records/TXT replaces every TXT (or every record) in the zone.
|
||||
for method, path, _json in calls:
|
||||
if method in ("PUT", "DELETE"):
|
||||
assert not path.endswith("/records"), f"zone-wide write: {method} {path}"
|
||||
assert not path.endswith("/records/TXT"), f"type-wide write: {method} {path}"
|
||||
|
||||
|
||||
def test_godaddy_add_write_path_merges_siblings():
|
||||
import asyncio
|
||||
|
||||
# An existing sibling value at the same name — the apex half of an apex+wildcard certificate.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA", "ttl": 600}],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
|
||||
writes = [c for c in calls if c[0] in ("PUT", "PATCH", "DELETE")]
|
||||
assert len(writes) == 1 and writes[0][0] == "PUT" and writes[0][1] == _GD_TXT
|
||||
# BOTH values must be in the body: GoDaddy's PUT replaces the whole RRset.
|
||||
assert sorted(r["data"] for r in writes[0][2]) == ["valueA", "valueB"]
|
||||
_assert_never_zone_wide(calls)
|
||||
|
||||
|
||||
def test_godaddy_add_write_path_is_idempotent_and_fails_closed():
|
||||
import asyncio
|
||||
|
||||
# Already published -> no write at all (this is where an ACME retry cycle lands).
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueB", "ttl": 600}],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
|
||||
# Unreadable RRset read (2xx whose body did not parse as a list) must FAIL, never be treated as
|
||||
# an empty RRset — the PUT that follows would replace the sibling values with only ours.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): None,
|
||||
})
|
||||
raised = False
|
||||
try:
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
except DnsProviderError:
|
||||
raised = True
|
||||
assert raised, "an unreadable RRset read must not be coerced into an empty RRset"
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
|
||||
|
||||
def test_godaddy_remove_write_path_uses_delete_for_the_last_value():
|
||||
import asyncio
|
||||
|
||||
# Two values -> PUT back the survivor only.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}, {"data": "valueB"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
writes = [c for c in calls if c[0] != "GET"]
|
||||
assert len(writes) == 1 and writes[0][0] == "PUT"
|
||||
assert [r["data"] for r in writes[0][2]] == ["valueA"]
|
||||
|
||||
# Last value -> DELETE. `PUT []` is rejected by GoDaddy (422 INVALID_BODY), so an empty PUT
|
||||
# body would make every cleanup fail forever.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueA"))
|
||||
writes = [c for c in calls if c[0] != "GET"]
|
||||
assert len(writes) == 1 and writes[0] == ("DELETE", _GD_TXT, None)
|
||||
assert not any(c[0] == "PUT" and c[2] == [] for c in calls)
|
||||
|
||||
# Value already gone -> no write, no error.
|
||||
p, calls = _gd_provider({
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [{"data": "valueA"}],
|
||||
})
|
||||
asyncio.run(p.remove_txt_record("_acme-challenge.example.com", "valueZ"))
|
||||
assert [c for c in calls if c[0] != "GET"] == []
|
||||
_assert_never_zone_wide(calls)
|
||||
|
||||
|
||||
class _FakeGDResponse:
|
||||
"""Minimal stand-in for aiohttp's ClientResponse: status, headers, and json()."""
|
||||
|
||||
_NO_BODY = object()
|
||||
|
||||
def __init__(self, status, body=_NO_BODY, headers=None):
|
||||
self.status = status
|
||||
self._body = body
|
||||
self.headers = headers or {}
|
||||
|
||||
async def json(self, content_type=None):
|
||||
if self._body is _FakeGDResponse._NO_BODY:
|
||||
raise ValueError("no body to decode") # what an empty 204 does
|
||||
return self._body
|
||||
|
||||
|
||||
class _FakeGDSession:
|
||||
def __init__(self, response):
|
||||
self._response = response
|
||||
self.calls = []
|
||||
|
||||
def request(self, method, url, **kwargs):
|
||||
self.calls.append((method, url, kwargs))
|
||||
response = self._response
|
||||
|
||||
class _Ctx:
|
||||
async def __aenter__(self_inner):
|
||||
return response
|
||||
|
||||
async def __aexit__(self_inner, *exc):
|
||||
return False
|
||||
|
||||
return _Ctx()
|
||||
|
||||
|
||||
def test_godaddy_request_status_handling():
|
||||
import asyncio
|
||||
from services.dns_providers.godaddy import GoDaddyDNSProvider
|
||||
|
||||
p = GoDaddyDNSProvider({"api_key": "KEY123", "api_secret": "SEC456"})
|
||||
|
||||
def call(response):
|
||||
session = _FakeGDSession(response)
|
||||
try:
|
||||
return asyncio.run(p._request(session, "PUT", "/domains/example.com/records/TXT/x",
|
||||
json=[{"data": "v", "ttl": 600}])), None, session
|
||||
except DnsProviderError as exc:
|
||||
return None, str(exc), session
|
||||
|
||||
# 204 with an EMPTY body is the normal answer to every GoDaddy write — it must not raise.
|
||||
body, err, session = call(_FakeGDResponse(204))
|
||||
assert body is None and err is None
|
||||
# Redirects are deliberately not followed (aiohttp would forward the Authorization header), so
|
||||
# a 3xx is a FAILED call. Treating it as success would report a redirected write as a no-op.
|
||||
_kw = session.calls[0][2]
|
||||
assert _kw["allow_redirects"] is False
|
||||
assert _kw["headers"]["Authorization"] == "sso-key KEY123:SEC456"
|
||||
assert _kw["headers"]["Accept"] == "application/json"
|
||||
for status in (301, 302, 307):
|
||||
body, err, _ = call(_FakeGDResponse(status))
|
||||
assert body is None and err and str(status) in err, f"HTTP {status} must not read as success"
|
||||
|
||||
# 200 with a list is passed through verbatim.
|
||||
body, err, _ = call(_FakeGDResponse(200, [{"data": "v"}]))
|
||||
assert err is None and body == [{"data": "v"}]
|
||||
|
||||
# Error mapping: each message must name what the operator has to fix.
|
||||
_, err, _ = call(_FakeGDResponse(401, {"code": "UNABLE_TO_AUTHENTICATE", "message": "nope"}))
|
||||
assert "PRODUCTION" in err and "UNABLE_TO_AUTHENTICATE" in err
|
||||
_, err, _ = call(_FakeGDResponse(403, {"code": "ACCESS_DENIED", "message": "not allowed"}))
|
||||
assert "domains.dns:update" in err
|
||||
# 429: Retry-After wins; the legacy body field is the fallback; absent both -> 60s default.
|
||||
_, err, _ = call(_FakeGDResponse(429, None, {"Retry-After": "17"}))
|
||||
assert "~17s" in err
|
||||
_, err, _ = call(_FakeGDResponse(429, {"retryAfterSec": 42}))
|
||||
assert "~42s" in err
|
||||
_, err, _ = call(_FakeGDResponse(429, {"Retry-After": "not-a-number"}))
|
||||
assert "~60s" in err
|
||||
# A non-dict error body must not crash the error mapper.
|
||||
_, err, _ = call(_FakeGDResponse(500, "<html>gateway</html>"))
|
||||
assert "500" in err
|
||||
|
||||
# A transport failure MID-READ must not be mistaken for "empty body". Only a decode error may
|
||||
# be swallowed: a caller reading an RRset would otherwise see None and could take it for an
|
||||
# empty record set, and the full-RRset PUT that follows would destroy the sibling values.
|
||||
import aiohttp
|
||||
|
||||
class _TruncatedResponse(_FakeGDResponse):
|
||||
async def json(self, content_type=None):
|
||||
raise aiohttp.ClientPayloadError("connection closed mid-body")
|
||||
|
||||
body, err, _ = call(_TruncatedResponse(200))
|
||||
assert body is None and err and "GoDaddy" in err
|
||||
|
||||
|
||||
def test_godaddy_zone_resolution_walks_suffixes_and_caches():
|
||||
import asyncio
|
||||
from services.dns_providers.godaddy import _GoDaddyHTTPError
|
||||
|
||||
# The deepest candidate is not a zone (404 = "not this zone"); the walk must continue to the
|
||||
# registrable domain and then reuse it, so the second challenge at the same name costs no probe.
|
||||
p, calls = _gd_provider({
|
||||
("GET", "/domains/_acme-challenge.example.com/records/NS"):
|
||||
_GoDaddyHTTPError("nope", status=404, code="UNKNOWN_DOMAIN"),
|
||||
("GET", _GD_NS): [{"data": "ns1.domaincontrol.com"}],
|
||||
("GET", _GD_TXT): [],
|
||||
})
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueA"))
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "valueB"))
|
||||
probes = [c for c in calls if c[1].endswith("/records/NS")]
|
||||
assert len(probes) == 2, "the resolved zone must be cached for the life of the provider"
|
||||
# Relative name derived from the RESOLVED zone, never from the deepest candidate.
|
||||
assert all(c[1] == _GD_TXT for c in calls if "/records/TXT/" in c[1])
|
||||
|
||||
# A credential/eligibility failure during the walk must surface, not be swallowed as
|
||||
# "no managed domain" — otherwise the operator chases a DNS problem that is really a bad key.
|
||||
p, calls = _gd_provider({
|
||||
("GET", "/domains/_acme-challenge.example.com/records/NS"):
|
||||
_GoDaddyHTTPError("denied", status=403, code="ACCESS_DENIED"),
|
||||
})
|
||||
raised = ""
|
||||
try:
|
||||
asyncio.run(p.add_txt_record("_acme-challenge.example.com", "v"))
|
||||
except DnsProviderError as exc:
|
||||
raised = str(exc)
|
||||
assert "denied" in raised and "No managed GoDaddy domain" not in raised
|
||||
|
||||
|
||||
def test_b64url_decode_padding_roundtrip():
|
||||
# Issue #35 v1.8.2: _b64url_decode must round-trip for EVERY length, including base64url strings
|
||||
# whose length is a multiple of 4 (the case the old padding formula '=' * (4 - len%4) over-padded).
|
||||
|
||||
@@ -0,0 +1,341 @@
|
||||
"""Issue #27 follow-up (v1.10.4) — unit tests for parsing an EXISTING keepalived.conf so a
|
||||
hand-maintained VIP can be adopted.
|
||||
|
||||
Pure-function tests; no DB, no network. The parser exists because the heartbeat carries only
|
||||
the VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs eleven
|
||||
fields — and because adoption REPLACES the operator's file, so anything our renderer cannot
|
||||
reproduce has to be reported as a blocker rather than silently dropped.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-keepalived-parser-tests")
|
||||
|
||||
from services import keepalived_config as kc # noqa: E402
|
||||
from services.keepalived_parser import ( # noqa: E402
|
||||
KeepalivedParseError, analyse_keepalived_conf, build_adoption_candidate,
|
||||
parse_keepalived_conf,
|
||||
)
|
||||
|
||||
|
||||
# A realistic hand-maintained config: two nodes, unicast VRRP, password auth, HAProxy check.
|
||||
HANDWRITTEN = """\
|
||||
! Configuration File for keepalived
|
||||
global_defs {
|
||||
enable_script_security
|
||||
script_user root
|
||||
}
|
||||
|
||||
vrrp_script chk_haproxy {
|
||||
script "/etc/keepalived/check_haproxy.sh"
|
||||
interval 2
|
||||
weight -21
|
||||
}
|
||||
|
||||
vrrp_instance VI_1 {
|
||||
state MASTER
|
||||
interface eth0 # public leg
|
||||
virtual_router_id 51
|
||||
priority 150
|
||||
advert_int 1
|
||||
authentication {
|
||||
auth_type PASS
|
||||
auth_pass s3cr3t
|
||||
}
|
||||
unicast_src_ip 10.0.0.11
|
||||
unicast_peer {
|
||||
10.0.0.12
|
||||
}
|
||||
virtual_ipaddress {
|
||||
10.0.0.100/24 dev eth0
|
||||
}
|
||||
track_script {
|
||||
chk_haproxy
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def _only_candidate(text):
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert len(parsed["instances"]) == 1
|
||||
return build_adoption_candidate(parsed, parsed["instances"][0])
|
||||
|
||||
|
||||
def test_parses_a_handwritten_config_into_model_fields():
|
||||
cand = _only_candidate(HANDWRITTEN)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["blockers"] == []
|
||||
assert cand["vip"] == {
|
||||
"virtual_ip": "10.0.0.100",
|
||||
"prefix_length": 24,
|
||||
"virtual_router_id": 51,
|
||||
"advert_int": 1,
|
||||
"use_unicast": True,
|
||||
"track_haproxy": True,
|
||||
"auth_pass": "s3cr3t",
|
||||
}
|
||||
assert cand["member"] == {"network_interface": "eth0", "role": "MASTER", "priority": 150}
|
||||
assert cand["peers"] == ["10.0.0.12"] and cand["unicast_src_ip"] == "10.0.0.11"
|
||||
assert cand["defaulted"] == [] # every value came from the file, nothing assumed
|
||||
|
||||
|
||||
def test_comment_and_layout_variants():
|
||||
# `!` and `#` both start comments; a block may open and close on one line; a quoted
|
||||
# script path keeps its spaces. None of this may change the parse.
|
||||
text = """\
|
||||
#!/not/a/shebang — this whole line is a comment
|
||||
vrrp_script chk { script "/opt/my scripts/chk.sh" }
|
||||
vrrp_instance VI_1 { state BACKUP
|
||||
interface eth1 ! trailing bang comment
|
||||
virtual_router_id 7
|
||||
priority 90
|
||||
virtual_ipaddress { 192.168.5.9/32 dev eth1 }
|
||||
}
|
||||
"""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert parsed["scripts"]["chk"]["script"] == "/opt/my scripts/chk.sh"
|
||||
inst = parsed["instances"][0]
|
||||
assert inst["state"] == "BACKUP" and inst["interface"] == "eth1"
|
||||
assert inst["virtual_router_id"] == 7 and inst["priority"] == 90
|
||||
assert inst["virtual_ips"] == [
|
||||
{"address": "192.168.5.9", "prefix_length": 32, "dev": "eth1", "extra": []}
|
||||
]
|
||||
|
||||
|
||||
def test_documented_defaults_are_applied_and_flagged():
|
||||
# keepalived's own defaults for absent directives. Applying them re-renders the same
|
||||
# behaviour, so they are allowed — but the UI must be able to say they were assumed.
|
||||
text = """\
|
||||
vrrp_instance VI_1 {
|
||||
interface eth0
|
||||
virtual_router_id 12
|
||||
virtual_ipaddress { 10.1.1.5/24 dev eth0 }
|
||||
}
|
||||
"""
|
||||
cand = _only_candidate(text)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["member"]["role"] == "BACKUP" and cand["member"]["priority"] == 100
|
||||
assert cand["vip"]["advert_int"] == 1
|
||||
assert sorted(cand["defaulted"]) == ["advert_int", "priority", "state"]
|
||||
# No authentication block and no track_script — both legal, both faithfully represented.
|
||||
assert cand["vip"]["auth_pass"] is None and cand["vip"]["track_haproxy"] is False
|
||||
assert cand["vip"]["use_unicast"] is False
|
||||
|
||||
|
||||
def _blockers_for(text):
|
||||
return " | ".join(_only_candidate(text)["blockers"])
|
||||
|
||||
|
||||
def test_directives_we_cannot_render_block_adoption():
|
||||
# THE central safety property: adoption overwrites the file, so a failover hook we do not
|
||||
# render would be destroyed. It must stop the flow, not warn.
|
||||
text = HANDWRITTEN.replace(
|
||||
" track_script {", ' notify_master "/usr/local/bin/promote.sh"\n track_script {')
|
||||
blockers = _blockers_for(text)
|
||||
assert "notify_master" in blockers and "would delete it" in blockers
|
||||
assert _only_candidate(text)["adoptable"] is False
|
||||
|
||||
|
||||
def test_multiple_addresses_in_one_instance_block_adoption():
|
||||
text = HANDWRITTEN.replace(" 10.0.0.100/24 dev eth0",
|
||||
" 10.0.0.100/24 dev eth0\n 10.0.0.101/24 dev eth0")
|
||||
blockers = _blockers_for(text)
|
||||
assert "2 addresses" in blockers and "10.0.0.101" in blockers
|
||||
|
||||
|
||||
def test_missing_vrid_blocks_adoption_with_the_split_brain_reason():
|
||||
text = HANDWRITTEN.replace(" virtual_router_id 51\n", "")
|
||||
blockers = _blockers_for(text)
|
||||
assert "no virtual_router_id" in blockers and "separate VRRP domains" in blockers
|
||||
|
||||
|
||||
def test_missing_prefix_blocks_adoption():
|
||||
# Our renderer always writes an explicit prefix; guessing one would change the netmask of a
|
||||
# live VIP, so the operator has to state it.
|
||||
text = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100 dev eth0")
|
||||
blockers = _blockers_for(text)
|
||||
assert "no explicit prefix length" in blockers
|
||||
|
||||
|
||||
def test_address_on_a_different_dev_blocks_adoption():
|
||||
text = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100/24 dev eth1")
|
||||
blockers = _blockers_for(text)
|
||||
assert "dev eth1" in blockers and "interface eth0" in blockers
|
||||
|
||||
|
||||
def test_foreign_track_script_blocks_adoption():
|
||||
text = HANDWRITTEN.replace(" chk_haproxy", " chk_custom")
|
||||
blockers = _blockers_for(text)
|
||||
assert "chk_custom" in blockers and "replaced by OpenManager" in blockers
|
||||
|
||||
|
||||
def test_unsupported_auth_type_blocks_adoption():
|
||||
text = HANDWRITTEN.replace("auth_type PASS", "auth_type AH")
|
||||
assert "auth_type AH" in _blockers_for(text)
|
||||
|
||||
|
||||
def test_fractional_advert_int_blocks_adoption():
|
||||
# Rounding 0.5s to 1s changes VRRP timing, so adopt-and-alter is not acceptable.
|
||||
text = HANDWRITTEN.replace("advert_int 1", "advert_int 0.5")
|
||||
blockers = _blockers_for(text)
|
||||
assert "advert_int 0.5" in blockers and "fractional" in blockers
|
||||
|
||||
|
||||
def test_half_configured_unicast_blocks_adoption():
|
||||
text = HANDWRITTEN.replace(" unicast_peer {\n 10.0.0.12\n }\n", "")
|
||||
assert "unicast_src_ip without unicast_peer" in _blockers_for(text)
|
||||
|
||||
|
||||
def test_sync_group_and_lvs_sections_block_adoption():
|
||||
text = HANDWRITTEN + """
|
||||
vrrp_sync_group VG1 {
|
||||
group {
|
||||
VI_1
|
||||
}
|
||||
}
|
||||
virtual_server 10.0.0.100 80 {
|
||||
lb_algo rr
|
||||
}
|
||||
"""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert [g["name"] for g in parsed["sync_groups"]] == ["VG1"]
|
||||
directives = " ".join(d["directive"] for d in parsed["unsupported"])
|
||||
assert "vrrp_sync_group VG1" in directives and "virtual_server" in directives
|
||||
# Both are top-level, so EVERY candidate in the file is blocked — a sync group changes
|
||||
# failover semantics for the instances it groups.
|
||||
cand = build_adoption_candidate(parsed, parsed["instances"][0])
|
||||
assert cand["adoptable"] is False
|
||||
|
||||
|
||||
def test_extra_global_defs_are_reported_as_losses():
|
||||
text = HANDWRITTEN.replace(" script_user root",
|
||||
" script_user root\n router_id LVS_DEVEL")
|
||||
parsed = parse_keepalived_conf(text)
|
||||
directives = " ".join(d["directive"] for d in parsed["unsupported"])
|
||||
assert "global_defs/router_id LVS_DEVEL" in directives
|
||||
assert parsed["global_defs"]["router_id"] == "LVS_DEVEL"
|
||||
|
||||
|
||||
def test_multiple_instances_yield_one_candidate_each():
|
||||
text = HANDWRITTEN + """
|
||||
vrrp_instance VI_2 {
|
||||
state BACKUP
|
||||
interface eth0
|
||||
virtual_router_id 52
|
||||
priority 100
|
||||
advert_int 1
|
||||
virtual_ipaddress { 10.0.0.200/24 dev eth0 }
|
||||
}
|
||||
"""
|
||||
analysed = analyse_keepalived_conf(text)
|
||||
assert analysed["instance_count"] == 2
|
||||
names = [c["instance_name"] for c in analysed["candidates"]]
|
||||
assert names == ["VI_1", "VI_2"]
|
||||
assert [c["vip"]["virtual_ip"] for c in analysed["candidates"]] == ["10.0.0.100", "10.0.0.200"]
|
||||
assert all(c["adoptable"] for c in analysed["candidates"])
|
||||
|
||||
|
||||
def test_unbalanced_braces_raise():
|
||||
for bad in ("vrrp_instance VI_1 {\n state MASTER\n", "}\n"):
|
||||
raised = False
|
||||
try:
|
||||
parse_keepalived_conf(bad)
|
||||
except KeepalivedParseError:
|
||||
raised = True
|
||||
assert raised, f"should have raised for {bad!r}"
|
||||
|
||||
|
||||
def test_our_own_render_round_trips_with_zero_blockers():
|
||||
"""The invariant that keeps the parser honest: a config WE generated must parse back into
|
||||
the same model with nothing unsupported. If a future change to render_keepalived_conf emits
|
||||
a directive the parser does not know, this fails — instead of adoption silently reporting
|
||||
that OpenManager's own output is unadoptable."""
|
||||
vip = {"id": 3, "name": "web-vip", "virtual_ip": "10.0.0.100", "prefix_length": 24,
|
||||
"virtual_router_id": 51, "advert_int": 1, "use_unicast": True, "track_haproxy": True}
|
||||
members = [{"role": "MASTER", "priority": 150, "network_interface": "eth0",
|
||||
"agent_id": 1, "ip_address": "10.0.0.11"},
|
||||
{"role": "BACKUP", "priority": 100, "network_interface": "eth0",
|
||||
"agent_id": 2, "ip_address": "10.0.0.12"}]
|
||||
rendered = kc.render_keepalived_conf(
|
||||
vip=vip, members=members, this_agent=members[0],
|
||||
peer_ips=["10.0.0.12"], auth_pass_plain="s3cr3t")
|
||||
|
||||
cand = _only_candidate(rendered)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["vip"]["virtual_ip"] == vip["virtual_ip"]
|
||||
assert cand["vip"]["prefix_length"] == vip["prefix_length"]
|
||||
assert cand["vip"]["virtual_router_id"] == vip["virtual_router_id"]
|
||||
assert cand["vip"]["track_haproxy"] is True and cand["vip"]["use_unicast"] is True
|
||||
assert cand["member"] == {"network_interface": "eth0", "role": "MASTER", "priority": 150}
|
||||
assert cand["vip"]["auth_pass"] == "s3cr3t"
|
||||
|
||||
# And the same for the no-auth / multicast / untracked shape, which renders fewer blocks.
|
||||
plain = kc.render_keepalived_conf(
|
||||
vip={**vip, "use_unicast": False, "track_haproxy": False},
|
||||
members=members, this_agent=members[1], peer_ips=[], auth_pass_plain=None)
|
||||
cand2 = _only_candidate(plain)
|
||||
assert cand2["adoptable"] is True, cand2["blockers"]
|
||||
assert cand2["vip"]["use_unicast"] is False and cand2["vip"]["track_haproxy"] is False
|
||||
assert cand2["vip"]["auth_pass"] is None
|
||||
|
||||
|
||||
# --- v1.10.4 adoption gate: which blockers an operator may resolve --------------------------
|
||||
|
||||
|
||||
def test_only_prefix_and_data_loss_are_waivable():
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
loss = "line 9: `notify_master \"/x.sh\"` — OpenManager's renderer cannot reproduce this, so adopting would delete it"
|
||||
prefix = "`10.0.0.5` has no explicit prefix length; state it during adoption so the netmask cannot change on takeover"
|
||||
hard_vrid = "no virtual_router_id — it cannot be guessed: a wrong VRID puts the nodes in separate VRRP domains"
|
||||
hard_auth = "auth_type AH is not supported (only PASS is rendered)"
|
||||
all_four = [loss, prefix, hard_vrid, hard_auth]
|
||||
|
||||
# Nothing waived: everything survives.
|
||||
assert remaining_blockers(all_four) == all_four
|
||||
# A supplied prefix resolves ONLY the prefix blocker.
|
||||
assert remaining_blockers(all_four, prefix_supplied=True) == [loss, hard_vrid, hard_auth]
|
||||
# Accepting data loss resolves ONLY the loss blocker.
|
||||
assert remaining_blockers(all_four, accept_data_loss=True) == [prefix, hard_vrid, hard_auth]
|
||||
# Both together still cannot wave through an impossibility — this is the property that stops
|
||||
# a UI flag from destroying a VIP whose VRID or auth_type we could not reproduce.
|
||||
assert remaining_blockers(all_four, prefix_supplied=True, accept_data_loss=True) == \
|
||||
[hard_vrid, hard_auth]
|
||||
# And an adoptable candidate stays adoptable.
|
||||
assert remaining_blockers([]) == []
|
||||
|
||||
|
||||
def test_waiver_markers_match_the_messages_the_parser_actually_emits():
|
||||
# The gate matches on substrings of the blocker prose, so a reworded message would silently
|
||||
# stop being waivable. Pin both directions against real parser output.
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
no_prefix = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100 dev eth0")
|
||||
blockers = _only_candidate(no_prefix)["blockers"]
|
||||
assert blockers, "expected a prefix blocker"
|
||||
assert remaining_blockers(blockers, prefix_supplied=True) == []
|
||||
|
||||
with_hook = HANDWRITTEN.replace(
|
||||
" track_script {", ' notify_master "/usr/local/bin/promote.sh"\n track_script {')
|
||||
blockers = _only_candidate(with_hook)["blockers"]
|
||||
assert blockers, "expected a data-loss blocker"
|
||||
assert remaining_blockers(blockers, accept_data_loss=True) == []
|
||||
|
||||
|
||||
def test_auth_pass_masking_leaves_no_trace_of_the_secret():
|
||||
# The discovered config is stored and served to the UI, so the ingest endpoint masks the VRRP
|
||||
# password. Reuse the router's own regex so the test breaks if it is loosened.
|
||||
from routers.agent import _AUTH_PASS_MASK_RE
|
||||
|
||||
secret = "s3cr3t with spaces"
|
||||
text = HANDWRITTEN.replace("auth_pass s3cr3t", f"auth_pass {secret}")
|
||||
masked = _AUTH_PASS_MASK_RE.sub(r"\1********", text)
|
||||
assert secret not in masked and "s3cr3t" not in masked
|
||||
assert "auth_pass ********" in masked
|
||||
# Everything else survives, so the preview is still useful.
|
||||
assert "virtual_router_id 51" in masked and "10.0.0.100/24 dev eth0" in masked
|
||||
@@ -0,0 +1,141 @@
|
||||
"""
|
||||
v1.10.6 — literal API paths must never be declared after a parameterised one.
|
||||
|
||||
Found in production on the v1.10.4 VIP-adoption feature: `@router.get("/discoveries")`
|
||||
sat at the bottom of routers/vip.py, below `@router.get("/{vip_id}")`. FastAPI matches
|
||||
routes in DECLARATION order, so every `GET /api/vip/discoveries` was answered by the
|
||||
`/{vip_id}` handler, which declares `vip_id: int` and therefore rejected the request with
|
||||
422 before `list_vip_discoveries` ever ran.
|
||||
|
||||
Nothing about that failure was visible. The agents reported their discoveries correctly,
|
||||
the rows landed in `vip_discoveries`, and the HA/VIP page treats any non-OK response as
|
||||
"nothing to show" — so the adoption feature simply did not exist as far as the UI was
|
||||
concerned, with no error anywhere.
|
||||
|
||||
These tests are a STATIC source scan on purpose: no imports, no app construction, no DB.
|
||||
They therefore also cover routers that cannot be imported in a bare test environment, and
|
||||
they keep covering routes added in the future.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
ROUTERS_DIR = pathlib.Path(__file__).resolve().parents[1] / "routers"
|
||||
|
||||
# `@router.get("/x")`, `@some_router.post("/x", ...)` — the path is the first string arg.
|
||||
_DECORATOR = re.compile(r'^@(?:\w+)\.(get|post|put|delete|patch)\(\s*[\'"]([^\'"]*)[\'"]')
|
||||
|
||||
|
||||
def _routes(source: str):
|
||||
"""[(line_no, verb, path)] in declaration order."""
|
||||
out = []
|
||||
for line_no, line in enumerate(source.splitlines(), 1):
|
||||
match = _DECORATOR.match(line)
|
||||
if match:
|
||||
out.append((line_no, match.group(1), match.group(2)))
|
||||
return out
|
||||
|
||||
|
||||
def _shadows(earlier: str, later: str) -> bool:
|
||||
"""True if `earlier` (declared first) swallows the literal path `later`.
|
||||
|
||||
Only literal paths can be silently swallowed, and only by a path that has the same
|
||||
number of segments where every non-placeholder segment matches. The collection route
|
||||
("" or "/") is its own path and never collides.
|
||||
"""
|
||||
if not later.strip("/") or not earlier.strip("/"):
|
||||
return False
|
||||
if "{" in later:
|
||||
return False
|
||||
if "{" not in earlier:
|
||||
return False
|
||||
earlier_segments = earlier.strip("/").split("/")
|
||||
later_segments = later.strip("/").split("/")
|
||||
if len(earlier_segments) != len(later_segments):
|
||||
return False
|
||||
return all(
|
||||
e.startswith("{") or e == l
|
||||
for e, l in zip(earlier_segments, later_segments)
|
||||
)
|
||||
|
||||
|
||||
def _shadowed_routes(path: pathlib.Path):
|
||||
routes = _routes(path.read_text())
|
||||
found = []
|
||||
for index, (line_no, verb, route_path) in enumerate(routes):
|
||||
for prior_line, prior_verb, prior_path in routes[:index]:
|
||||
if prior_verb == verb and _shadows(prior_path, route_path):
|
||||
found.append(
|
||||
f"{path.name}:{line_no} {verb.upper()} {route_path} is swallowed by "
|
||||
f"{prior_path} declared at line {prior_line}"
|
||||
)
|
||||
return found
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 1. The specific regression: /api/vip/discoveries must outrank /{vip_id}
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_vip_discoveries_declared_before_vip_id():
|
||||
routes = _routes((ROUTERS_DIR / "vip.py").read_text())
|
||||
get_paths = [path for _line, verb, path in routes if verb == "get"]
|
||||
|
||||
assert "/discoveries" in get_paths, "the discoveries endpoint disappeared"
|
||||
assert "/{vip_id}" in get_paths, "the get-one endpoint disappeared"
|
||||
assert get_paths.index("/discoveries") < get_paths.index("/{vip_id}"), (
|
||||
"GET /discoveries is declared after GET /{vip_id}; FastAPI will route "
|
||||
"/api/vip/discoveries into get_vip and answer 422, silently emptying the "
|
||||
"adoption panel"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 2. The general guard: no literal path anywhere is shadowed
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_no_literal_route_is_shadowed_in_any_router():
|
||||
problems = []
|
||||
for router_file in sorted(ROUTERS_DIR.glob("*.py")):
|
||||
problems.extend(_shadowed_routes(router_file))
|
||||
|
||||
assert not problems, (
|
||||
"literal route(s) declared after a parameterised route that swallows them:\n "
|
||||
+ "\n ".join(problems)
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 3. The detector itself must actually detect (guards against a vacuous pass)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"earlier,later,expected",
|
||||
[
|
||||
("/{vip_id}", "/discoveries", True), # the v1.10.4 bug
|
||||
("/{vip_id}", "/{other}", False), # two placeholders never shadow
|
||||
("/{vip_id}", "", False), # collection route is its own path
|
||||
("/{vip_id}/apply", "/adopt", False), # different segment counts
|
||||
("/{vip_id}/apply", "/adopt/now", False), # literal mismatch in segment 2
|
||||
("/{vip_id}/{action}", "/adopt/now", True), # both segments placeheld
|
||||
("/vips", "/discoveries", False), # literal never shadows a literal
|
||||
],
|
||||
)
|
||||
def test_shadow_detector_semantics(earlier, later, expected):
|
||||
assert _shadows(earlier, later) is expected
|
||||
|
||||
|
||||
def test_detector_flags_the_original_declaration_order():
|
||||
"""A synthetic file in the pre-fix order must be reported, so a future refactor that
|
||||
breaks the detector cannot make the guard above pass vacuously."""
|
||||
source = (
|
||||
'@router.get("")\n'
|
||||
"async def list_vips(): ...\n"
|
||||
'@router.get("/{vip_id}")\n'
|
||||
"async def get_vip(vip_id: int): ...\n"
|
||||
'@router.get("/discoveries")\n'
|
||||
"async def list_vip_discoveries(): ...\n"
|
||||
)
|
||||
routes = _routes(source)
|
||||
assert [verb for _l, verb, _p in routes] == ["get", "get", "get"]
|
||||
assert _shadows(routes[1][2], routes[2][2]) is True
|
||||
@@ -0,0 +1,431 @@
|
||||
"""
|
||||
v1.10.8 — the four adoption defects found while testing v1.10.4 on a live HA pair.
|
||||
|
||||
B1 The Apply Management "View Change" diff did not recognise the `adopt` action, so the
|
||||
version fell through to the generic HAProxy diff and rendered the cluster's whole
|
||||
haproxy.cfg as removed.
|
||||
|
||||
B2 `vip_discoveries.adopted_vip_id` is write-once and nothing clears it, while a VIP is only
|
||||
ever SOFT-deleted — so `ON DELETE SET NULL` never fires. Rejecting an adoption therefore
|
||||
hid the node from the panel permanently: the VIP was gone from the VIP list too, and the
|
||||
agent does not re-report a file whose hash has not changed. Adoptability is now derived
|
||||
from whether the linked VIP is still active.
|
||||
|
||||
B3 Adoption took only the node that was clicked. On a two-node pair that meant: adopting the
|
||||
BACKUP produced a VIP whose apply fails ("exactly one member must be MASTER"), adopting the
|
||||
MASTER left the peer unmanaged, and adopting the peer afterwards hit the VRID-collision
|
||||
guard with 409. The pair could never be completed from the panel.
|
||||
|
||||
B4 Worst of the four. `render_keepalived_conf` emits the unicast block only when it has peer
|
||||
addresses, so a single-member adoption of a UNICAST instance silently dropped it and
|
||||
keepalived fell back to multicast on that node while its peer stayed unicast — they stop
|
||||
seeing each other and BOTH claim the VIP.
|
||||
|
||||
B3 and B4 share one root and one fix: adoption now resolves the whole VRRP instance, keyed on
|
||||
(virtual_router_id, virtual address) exactly as keepalived groups nodes.
|
||||
|
||||
These are source-level and unit tests: the adoption endpoint needs a live database, so the
|
||||
behaviour that can be exercised without one is pinned here, and the SQL/flow invariants are
|
||||
pinned by reading the module.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
BACKEND = pathlib.Path(__file__).resolve().parents[1]
|
||||
VIP_ROUTER = (BACKEND / "routers" / "vip.py").read_text()
|
||||
CLUSTER_ROUTER = (BACKEND / "routers" / "cluster.py").read_text()
|
||||
RENDERER = (BACKEND / "services" / "keepalived_config.py").read_text()
|
||||
|
||||
|
||||
def _adopt_body() -> str:
|
||||
start = VIP_ROUTER.index("async def adopt_vip")
|
||||
return VIP_ROUTER[start:]
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B1 — the diff must recognise `adopt`
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_view_change_diff_recognises_the_adopt_action():
|
||||
match = re.search(r"vip_match = re\.search\(r'vip-\(\\d\+\)-\(([^)]+)\)'", CLUSTER_ROUTER)
|
||||
assert match, "the vip version regex moved; re-point this test"
|
||||
actions = set(match.group(1).split("|"))
|
||||
assert actions == {"create", "update", "delete", "adopt"}, (
|
||||
f"the View Change diff recognises {sorted(actions)}. An action missing here does not "
|
||||
f"degrade gracefully: the version falls through to the generic HAProxy diff and shows "
|
||||
f"the cluster's whole haproxy.cfg as removed."
|
||||
)
|
||||
|
||||
|
||||
def test_every_staged_vip_action_is_covered_by_the_diff_regex():
|
||||
"""Whatever _stage_vip_version can be called with must be in that alternation."""
|
||||
staged = set(re.findall(r'_stage_vip_version\(conn, vip_id, "(\w+)"', VIP_ROUTER))
|
||||
match = re.search(r"vip_match = re\.search\(r'vip-\(\\d\+\)-\(([^)]+)\)'", CLUSTER_ROUTER)
|
||||
recognised = set(match.group(1).split("|"))
|
||||
assert staged, "no _stage_vip_version call sites found; re-point this test"
|
||||
assert staged <= recognised, (
|
||||
f"staged action(s) {sorted(staged - recognised)} are not recognised by the View Change "
|
||||
f"diff regex {sorted(recognised)}"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B2 — adoptability follows the VIP's liveness, not the bare link
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_discovery_api_exposes_whether_the_adopted_vip_still_stands():
|
||||
assert "adopted_vip_active" in VIP_ROUTER, (
|
||||
"the discovery payload no longer reports whether the adopted VIP is still active; the "
|
||||
"UI would go back to hiding a rejected adoption forever"
|
||||
)
|
||||
assert re.search(r"LEFT JOIN vip_instances av ON av\.id = d\.adopted_vip_id", VIP_ROUTER), (
|
||||
"the discovery query must join the linked VIP to report its is_active"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_only_while_the_previous_adoption_still_stands():
|
||||
body = _adopt_body()
|
||||
assert re.search(r'if disc\["adopted_vip_id"\] and disc\["adopted_vip_active"\]', body), (
|
||||
"adopt must refuse only when the linked VIP is still ACTIVE — refusing on the bare link "
|
||||
"makes a rejected adoption impossible to retry, because nothing ever clears the column"
|
||||
)
|
||||
|
||||
|
||||
def test_nothing_clears_adopted_vip_id_so_the_derivation_is_load_bearing():
|
||||
"""If a future change starts clearing the column, this test should be revisited rather than
|
||||
silently left in place — the derived flag is what makes reject recoverable today."""
|
||||
writes = re.findall(r"UPDATE vip_discoveries SET adopted_vip_id = (\S+)", VIP_ROUTER)
|
||||
assert writes, "no adopted_vip_id write found; re-point this test"
|
||||
assert all(w != "NULL" for w in writes), (
|
||||
"adopted_vip_id is now cleared somewhere — re-check that the adopted_vip_active "
|
||||
"derivation and this test still describe reality"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B3 — the whole VRRP instance is adopted, not one node
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_participants_are_resolved_by_vrid_and_address():
|
||||
assert "async def _collect_instance_participants" in VIP_ROUTER
|
||||
start = VIP_ROUTER.index("async def _collect_instance_participants")
|
||||
end = VIP_ROUTER.index("@router.post(\"/adopt\")", start)
|
||||
body = VIP_ROUTER[start:end]
|
||||
assert 'v.get("virtual_router_id") == vrid' in body and 'v.get("virtual_ip") == virtual_ip' in body, (
|
||||
"instance identity must be (VRID, address) — the same key keepalived uses to decide two "
|
||||
"nodes are one VRRP group"
|
||||
)
|
||||
assert 'if r["parse_error"]' in body, "a node whose config failed to parse must not become a member"
|
||||
assert 'r["adopted_vip_id"] and r["adopted_vip_active"]' in body, (
|
||||
"a node already held by a STANDING VIP must not be pulled into a second one"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_inserts_one_member_per_participant():
|
||||
body = _adopt_body()
|
||||
insert = body.index("INSERT INTO vip_members")
|
||||
preceding = body[:insert]
|
||||
assert "for p in participants:" in preceding, (
|
||||
"members must be inserted in a loop over the resolved participants; a single insert is "
|
||||
"the half-adoption bug"
|
||||
)
|
||||
assert 'p["config_hash"]' in body[insert:insert + 800], (
|
||||
"each member must carry ITS OWN takeover hash — the one-shot takeover guard is per node"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_exactly_one_master_across_the_instance():
|
||||
body = _adopt_body()
|
||||
assert 'roles.count("MASTER") != 1' in body, (
|
||||
"adoption must reject an instance that does not have exactly one MASTER, instead of "
|
||||
"letting apply fail later with 'exactly one member must be MASTER'"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_enforces_one_active_vip_per_agent():
|
||||
body = _adopt_body()
|
||||
assert "already a member of VIP" in body, (
|
||||
"adoption must enforce the one-active-VIP-per-agent rule that create/update enforce via "
|
||||
"_validate_members_against_pool; a second membership never converges"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B4 — a unicast instance can never be half-adopted
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_renderer_only_emits_unicast_when_it_has_peers():
|
||||
"""The property that makes B4 dangerous. Pinned so the guard below keeps its reason."""
|
||||
assert re.search(r"if use_unicast and peer_ips:", RENDERER), (
|
||||
"the renderer no longer gates the unicast block on having peers; re-derive whether the "
|
||||
"adoption guard is still needed"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_a_unicast_peer_that_is_not_being_adopted():
|
||||
body = _adopt_body()
|
||||
assert "declared_peers" in body and "member_ips" in body, (
|
||||
"adoption must verify every declared unicast peer is among the nodes being adopted"
|
||||
)
|
||||
assert "fall back to multicast" in body, (
|
||||
"the refusal must explain the consequence — silently dropping a peer puts both nodes in "
|
||||
"MASTER state on the same address"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_to_strand_any_node_that_references_the_address():
|
||||
"""The half-adoption hole that instance resolution alone does not close.
|
||||
|
||||
Participant resolution can only match a node it can READ, that is ENABLED and that is in the
|
||||
SAME pool. Each of those is a door a real member leaves through silently, and the nodes that
|
||||
remain get rewritten while it keeps serving the same address unmanaged. Seen for real: one
|
||||
node of a pair had a missing closing brace, so it parsed to nothing while its partner parsed
|
||||
cleanly. Rather than guard each door, the endpoint asks whether ANY reported config mentions
|
||||
this address and is not among the nodes being adopted.
|
||||
"""
|
||||
body = _adopt_body()
|
||||
assert "raw_config_masked LIKE" in body, (
|
||||
"the check must be scoped to configs that reference THIS virtual address, so an unrelated "
|
||||
"file elsewhere in the fleet does not block every adoption"
|
||||
)
|
||||
assert "NOT (a.id = ANY($2::int[]))" in body, (
|
||||
"the guard must catch every node that is NOT a participant, not just the unparseable "
|
||||
"ones — a disabled agent and a peer in another pool are stranded exactly the same way"
|
||||
)
|
||||
assert "COALESCE(av.is_active, FALSE) = FALSE" in body and "d.is_managed = FALSE" in body, (
|
||||
"a node already under management is not stranded and must not block adoption"
|
||||
)
|
||||
for reason in ("could not be parsed", "agent is disabled", "different agent pool"):
|
||||
assert reason in body, f"the refusal must be able to explain '{reason}'"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field,label", [
|
||||
("prefix_length", "prefix length"),
|
||||
("use_unicast", "unicast/multicast mode"),
|
||||
("track_haproxy", "HAProxy tracking"),
|
||||
])
|
||||
def test_adopt_requires_agreement_on_shared_vip_fields(field, label):
|
||||
"""These live on the VIP row and are re-rendered onto EVERY member, so taking them from the
|
||||
node that happened to be clicked imposes its settings on the others. prefix_length is the
|
||||
sharpest: the design refuses to guess a netmask for a live VIP, and copying one node's
|
||||
netmask onto another is that same change by another name."""
|
||||
body = _adopt_body()
|
||||
assert f'("{field}", "{label}")' in body, (
|
||||
f"{field} is written to the VIP row from one node's report; adoption must refuse when the "
|
||||
f"nodes disagree about it"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_one_shared_vrrp_password():
|
||||
body = _adopt_body()
|
||||
assert "decrypt_vrrp_secret(enc)" in body, (
|
||||
"Fernet is non-deterministic, so the per-node tokens cannot be compared as ciphertext — "
|
||||
"they must be decrypted and compared as plaintext"
|
||||
)
|
||||
assert "do not share one VRRP password" in body, (
|
||||
"adoption stores ONE secret and renders it onto every member, so a mismatch must be "
|
||||
"refused rather than silently normalised"
|
||||
)
|
||||
assert "cannot be decrypted" in body, (
|
||||
"a token we cannot decrypt must be an error, not silently treated as equal to another"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_reported_ips_before_trusting_the_peer_check():
|
||||
body = _adopt_body()
|
||||
assert "have not reported an IP address yet" in body, (
|
||||
"the unicast peer check compares against member IPs, so a member without a reported IP "
|
||||
"must block the check rather than silently pass it"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# The takeover authorisation is genuinely one-shot
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_takeover_authorisation_is_retired_once_the_node_acks_our_config():
|
||||
"""`takeover_expected_hash` is the permission to overwrite a keepalived.conf that does NOT
|
||||
carry our ownership marker. It was written at adoption and never cleared, so the release
|
||||
notes' "authorises exactly ONE takeover" was only true as "for exactly that file content,
|
||||
indefinitely" — restoring the pre-adoption file would have been silently overwritten again
|
||||
with no fresh human approval."""
|
||||
agent_router = (BACKEND / "routers" / "agent.py").read_text()
|
||||
assert "takeover_expected_hash = CASE WHEN applied_config_hash IS NOT NULL" in agent_router, (
|
||||
"the status ack must retire the takeover authorisation once the node confirms our config"
|
||||
)
|
||||
assert "ELSE takeover_expected_hash END" in agent_router, (
|
||||
"a non-matching ack must LEAVE the authorisation in place — dropping it on a partial or "
|
||||
"failed deploy would leave the VIP unable to converge"
|
||||
)
|
||||
|
||||
|
||||
def test_validation_judges_the_output_not_the_exit_code():
|
||||
"""keepalived's config-test exit code cannot separate fatal from benign.
|
||||
|
||||
Measured on keepalived 2.2.8 rather than assumed:
|
||||
|
||||
clean config ..................... 0
|
||||
auth_pass longer than 8 chars .... 5 "Truncating auth_pass to 8 characters"
|
||||
missing '}' ...................... 5 "There are 1 missing '}'s"
|
||||
unknown keyword .................. 5 "Unknown keyword '...'"
|
||||
script without script_security ... 6 "SECURITY VIOLATION ..."
|
||||
|
||||
So 5 covers both a harmless truncation and a broken file. Treating any non-zero exit as
|
||||
invalid rejected VALID configs: a VRRP password over 8 characters is enough, and keepalived
|
||||
truncates it to 8 anyway, exactly as it does for the file the operator already runs. Found
|
||||
on the first live adoption, where the node's OWN running config also exited non-zero.
|
||||
|
||||
The gate must therefore judge the output, and it must fail CLOSED: anything not on the
|
||||
benign list still counts as fatal.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count("grep -v 'Truncating auth_pass to 8 characters'") == 2, (
|
||||
"both daemon copies must drop the known-benign truncation warning before judging"
|
||||
)
|
||||
assert script.count('if [[ -n "$kp_fatal" ]]; then') == 2, (
|
||||
"the decision must be made on what REMAINS after the benign lines are dropped"
|
||||
)
|
||||
# Fail-closed: the benign list is an allowlist, never a denylist of fatal messages. The
|
||||
# measured table above is quoted in a comment inside the script, so assert on what is used
|
||||
# as a grep PATTERN rather than on the text appearing anywhere.
|
||||
patterns = re.findall(r"grep -v '([^']*)'", script)
|
||||
assert set(patterns) == {"Truncating auth_pass to 8 characters", "^[[:space:]]*$"}, (
|
||||
f"the validation filter greps for {sorted(set(patterns))}. It must drop only messages "
|
||||
f"known to be benign; matching on fatal messages instead would let an unrecognised "
|
||||
f"error through."
|
||||
)
|
||||
|
||||
|
||||
def test_validation_fails_closed_when_keepalived_says_nothing():
|
||||
"""A non-zero exit with no output must stay fatal.
|
||||
|
||||
Filtering the output introduces a way to reach the accept path with an EMPTY filter result,
|
||||
and some keepalived builds log to syslog rather than stderr — on such a host every config
|
||||
would then be accepted regardless of what is wrong with it. Verified against a stub that
|
||||
exits non-zero silently, and against one that emits only whitespace.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count('if [[ -z "${kp_out//[[:space:]]/}" ]]; then') == 2, (
|
||||
"both daemon copies must treat a non-zero exit with no readable output as fatal"
|
||||
)
|
||||
assert script.count('kp_fatal="keepalived -t exited non-zero without output"') == 2
|
||||
|
||||
|
||||
def test_validation_failure_reports_what_keepalived_said():
|
||||
"""The fail-safe protected the node correctly on a live adoption but logged only
|
||||
"config validation failed", with keepalived's own output sent to /dev/null. The operator
|
||||
had no way to act on it without reproducing the check by hand on the node."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert 'keepalived -t -f "$tmp_conf" >/dev/null 2>&1' not in script, (
|
||||
"keepalived's output must not be discarded — a fail-safe that cannot say why it fired "
|
||||
"is only half a safety feature"
|
||||
)
|
||||
assert script.count('kp_out=$(keepalived -t -f "$tmp_conf" 2>&1)') == 2, (
|
||||
"both daemon copies must capture the validation output"
|
||||
)
|
||||
# The text is interpolated into JSON by both `log` and _kp_report, so it must be sanitised.
|
||||
assert script.count(r"""tr -d '"\\'""") == 2, (
|
||||
"captured output must have quotes and backslashes stripped before it reaches the JSON "
|
||||
"log line and the status report"
|
||||
)
|
||||
assert script.count('keepalived -t failed: ${kp_err}') == 2, (
|
||||
"the reason must also travel to the server so the UI can show it"
|
||||
)
|
||||
|
||||
|
||||
def test_converged_node_keeps_acknowledging():
|
||||
"""The idempotent path must still report, or a lost ack is never recovered.
|
||||
|
||||
The status report is the server's ONLY evidence that a member converged, and it used to be
|
||||
sent solely on the write path. Once the rendered config was on disk the agent took the
|
||||
idempotency early return every cycle and never spoke again, so a single lost report - a
|
||||
backend restart, a 5xx, a network blip - left the VIP reading SYNCING forever with an empty
|
||||
"Last ack" while the node was demonstrably running the right config. Seen in the field after
|
||||
acks were dropped for an unrelated reason: the node was correct, the page was not, and
|
||||
nothing would ever reconcile them.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count('_kp_report "enabled" "$vip_id" "$new_hash" "already converged"') == 2, (
|
||||
"both daemon copies must re-assert the deploy state on the idempotent path; without it "
|
||||
"the server can never recover a lost acknowledgement"
|
||||
)
|
||||
# The report has to come BEFORE the early return in both copies.
|
||||
for m in re.finditer(r'if \[\[ -n "\$cur_hash" && "\$cur_hash" == "\$would_hash" \]\]; then(.*?)fi',
|
||||
script, re.S):
|
||||
body = m.group(1)
|
||||
assert body.index("_kp_report") < body.index("return 0"), (
|
||||
"the acknowledgement must be sent before returning, or the early return skips it"
|
||||
)
|
||||
|
||||
|
||||
def test_status_ack_statements_bind_each_placeholder_once():
|
||||
"""Every `$n` in the keepalived-status UPDATEs must be used exactly once, and the count must
|
||||
match the arguments passed.
|
||||
|
||||
Reusing one placeholder for both the assignment (`last_deploy_hash=$n`, a VARCHAR column)
|
||||
and the comparison inside the takeover-retirement CASE made PostgreSQL deduce two types for
|
||||
it, and asyncpg rejected the whole statement with AmbiguousParameterError. The failure was
|
||||
not partial: no ack was written at all, so every VIP sat at SYNCING forever and teardown acks
|
||||
were lost too. Shipped in v1.10.12 and caught in the field.
|
||||
|
||||
The suite has no database, so this pins the shape that made it possible rather than the SQL
|
||||
behaviour: one placeholder, one binding site.
|
||||
"""
|
||||
src = (BACKEND / "routers" / "agent.py").read_text()
|
||||
start = src.index("async def agent_keepalived_status")
|
||||
seg = src[start:src.index('return {"status": "ok"}', start)]
|
||||
|
||||
retire = "".join(re.findall(r'"([^"]*)"',
|
||||
re.search(r"_retire_takeover = \((.*?)\)\n", seg, re.S).group(1)))
|
||||
calls = re.findall(r'await conn\.execute\(f"""(.*?)""",\s*(.*?)\)\n', seg, re.S)
|
||||
assert len(calls) == 2, f"expected the two ack UPDATEs, found {len(calls)}"
|
||||
|
||||
for sql, args in calls:
|
||||
placeholder = re.search(r'_retire_takeover\.format\(p="(\$\d+)"\)', sql).group(1)
|
||||
rendered = re.sub(r"\{_retire_takeover\.format\(p=\"\$\d+\"\)\}",
|
||||
retire.replace("{p}", placeholder), sql)
|
||||
used = re.findall(r"\$(\d+)", rendered)
|
||||
dupes = {n for n in used if used.count(n) > 1}
|
||||
assert not dupes, (
|
||||
f"placeholder(s) {sorted('$'+d for d in dupes)} are bound more than once. PostgreSQL "
|
||||
f"deduces a type per USE, so a placeholder that is both assigned to a column and "
|
||||
f"compared against one is ambiguous and the whole UPDATE is rejected."
|
||||
)
|
||||
n_args = len([a for a in args.split(",") if a.strip()])
|
||||
assert max(int(n) for n in used) == n_args, (
|
||||
f"the statement uses ${max(int(n) for n in used)} but {n_args} arguments are passed"
|
||||
)
|
||||
|
||||
|
||||
def test_takeover_still_requires_the_pinned_hash_to_match_on_disk():
|
||||
"""The guard that stops an edit between adoption and Apply from being overwritten."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
guard = '[[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]'
|
||||
assert script.count(guard) == 2, (
|
||||
f"the takeover guard must be present in BOTH daemon copies (found {script.count(guard)}); "
|
||||
f"a self-upgraded agent runs the in-script copy, a freshly installed one the heredoc"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Backward compatibility
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("guard", [
|
||||
"version_name NOT LIKE 'vip-%'", # bulk apply/reject still skip VIP versions
|
||||
])
|
||||
def test_vip_versions_stay_excluded_from_the_haproxy_apply_flow(guard):
|
||||
assert guard in CLUSTER_ROUTER, (
|
||||
"vip-* versions must stay out of the HAProxy apply/reject sweep; they are owned by the "
|
||||
"VIP endpoints and are never served as haproxy.cfg"
|
||||
)
|
||||
|
||||
|
||||
def test_vip_version_transition_matches_any_action():
|
||||
"""_transition_vip_versions must key on the VIP id alone, or a new action's PENDING row
|
||||
would be stranded in Apply Management after apply/reject."""
|
||||
assert 'f"vip-{vip_id}-%"' in VIP_ROUTER, (
|
||||
"the PENDING -> APPLIED/REJECTED transition must match every action for the VIP"
|
||||
)
|
||||
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
v1.10.6 — GET /api/vip/discoveries resolves to its own handler and honours cluster_id.
|
||||
|
||||
Two defects, one endpoint, found in that order on a live fleet:
|
||||
|
||||
1. The route was declared after `GET /{vip_id}`, so FastAPI matched it there and answered
|
||||
422 ("discoveries" is not an int) before the handler ran. The static declaration-order
|
||||
guard lives in test_router_path_shadowing.py; this file pins the observable behaviour,
|
||||
because a 422 is what the browser actually saw.
|
||||
|
||||
2. Once reachable, it returned every discovery in the fleet regardless of the cluster
|
||||
selected in the header, so a multi-cluster install saw one undifferentiated list. The
|
||||
endpoint now takes the same optional `cluster_id` the VIP list takes.
|
||||
|
||||
The auth tests here deliberately assert `!= 422`: the repo's generic endpoint-auth tests
|
||||
accept 401/403/422 together, which is precisely why defect 1 slipped through them.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
VIP_ROUTER = pathlib.Path(__file__).resolve().parents[1] / "routers" / "vip.py"
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 1. The route reaches its own handler (defect 1)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/vip/discoveries",
|
||||
"/api/vip/discoveries?cluster_id=7",
|
||||
])
|
||||
def test_discoveries_route_is_not_captured_by_the_vip_id_route(client, path):
|
||||
res = client.get(path)
|
||||
assert res.status_code != 422, (
|
||||
f"GET {path} returned 422 — the request was routed into the get-one-VIP handler, "
|
||||
f"which parses the path segment as an int. Declaration order regressed. "
|
||||
f"Body: {res.text[:200]}"
|
||||
)
|
||||
assert res.status_code in (401, 403), (
|
||||
f"GET {path} without a token should be refused by the vip.read gate, got "
|
||||
f"{res.status_code}. Body: {res.text[:200]}"
|
||||
)
|
||||
|
||||
|
||||
def test_get_one_vip_still_parses_a_numeric_id(client):
|
||||
"""Moving /discoveries above /{vip_id} must not shadow the numeric route itself."""
|
||||
res = client.get("/api/vip/12")
|
||||
assert res.status_code in (401, 403), res.text[:200]
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 2. cluster_id is accepted and actually scopes the query (defect 2)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_handler_accepts_cluster_id():
|
||||
from routers.vip import list_vip_discoveries
|
||||
import inspect
|
||||
|
||||
params = inspect.signature(list_vip_discoveries).parameters
|
||||
assert "cluster_id" in params, (
|
||||
"list_vip_discoveries no longer takes cluster_id; the HA/VIP page would show every "
|
||||
"cluster's nodes at once again"
|
||||
)
|
||||
assert params["cluster_id"].default is None, (
|
||||
"cluster_id must stay optional — omitting it returns the whole fleet, which is what "
|
||||
"a caller that predates the parameter expects"
|
||||
)
|
||||
|
||||
|
||||
def test_discovery_query_scopes_by_the_cluster_pool():
|
||||
"""The filter must resolve cluster -> pool the same way the VIP list does, and must be a
|
||||
no-op when the parameter is absent."""
|
||||
source = VIP_ROUTER.read_text()
|
||||
start = source.index("async def list_vip_discoveries")
|
||||
end = source.index("def _find_candidate", start)
|
||||
body = source[start:end]
|
||||
|
||||
assert "FROM vip_discoveries" in body, "the discovery query moved; re-point this test"
|
||||
assert re.search(r"a\.pool_id\s*=\s*\(\s*SELECT\s+pool_id\s+FROM\s+haproxy_clusters", body), (
|
||||
"the cluster filter must map cluster -> pool via haproxy_clusters, matching list_vips"
|
||||
)
|
||||
assert "IS NULL" in body, (
|
||||
"the filter must short-circuit when cluster_id is absent, so an unscoped call still "
|
||||
"returns the whole fleet"
|
||||
)
|
||||
@@ -0,0 +1,266 @@
|
||||
"""Validation and resolution for the ACME HTTP-01 challenge backend URL.
|
||||
|
||||
This URL tells HAProxy where to proxy ``/.well-known/acme-challenge/*``. It is
|
||||
rendered into ``backend _acme_challenge_backend`` as ``server _acme_mgmt host:port``
|
||||
and — this is the part that makes it unlike every other URL in the product —
|
||||
**resolved on the HAProxy node, not on the management host**. A value that works
|
||||
when pasted into the management server's own browser can be completely dead from
|
||||
the data plane.
|
||||
|
||||
Two entry points, deliberately asymmetric:
|
||||
|
||||
``validate_acme_backend_url``
|
||||
Called at the WRITE BOUNDARY (cluster PUT, settings PUT). Rejects values that
|
||||
cannot express a reachable target. Strict here is safe: it only ever affects a
|
||||
value an operator is typing right now, and the error text can teach.
|
||||
|
||||
``resolve_acme_backend_target``
|
||||
Called at RENDER TIME. Never raises, never rejects. Strictness here would be a
|
||||
catastrophe: the shipped defaults (``config.py`` ``http://localhost:8000``,
|
||||
``docker-compose.yml`` ``http://localhost:8080``) mean essentially every
|
||||
existing install resolves to loopback today, and refusing to render would make
|
||||
every ``acme_enabled`` cluster unappliable — including for urgent changes that
|
||||
have nothing to do with ACME. It reports problems instead of enforcing them.
|
||||
|
||||
Two conscious departures from ``utils/ssrf_guard.py``, whose policy is the exact
|
||||
opposite of what is needed here:
|
||||
|
||||
* **RFC1918 is allowed, and is usually the correct answer.** The guard exists to
|
||||
stop the server being tricked into dialling internal space. Here the operator is
|
||||
deliberately naming their own management host, which on a split deployment is
|
||||
private by definition.
|
||||
* **No DNS resolution.** Resolving from the management host would re-introduce the
|
||||
very wrong-vantage-point mistake this work exists to remove: what this box can
|
||||
resolve says nothing about what the HAProxy node can reach.
|
||||
"""
|
||||
import ipaddress
|
||||
import re
|
||||
from typing import List, NamedTuple, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
# `haproxy_clusters.acme_backend_url` / `system_settings.value` are VARCHAR(500).
|
||||
# Without this check asyncpg raises 22001 and the operator gets an opaque 500.
|
||||
MAX_URL_LENGTH = 500
|
||||
|
||||
ALLOWED_SCHEMES = ("http", "https")
|
||||
|
||||
# Port assumed when the URL omits one. NOT the scheme's default: the bundled
|
||||
# docker-compose publishes nginx on 8080 (`nginx/nginx.conf` listens 8080,
|
||||
# `docker-compose.yml` maps 8080:8080), so an operator who wrote a bare
|
||||
# `http://10.0.0.5` has a WORKING path today that resolves to :8080. Changing this
|
||||
# to 80 would break those installs silently — the first symptom would be the
|
||||
# unattended renewal loop failing months later. The value is kept and the omission
|
||||
# is surfaced as a warning instead.
|
||||
DEFAULT_HTTP_PORT = 8080
|
||||
DEFAULT_HTTPS_PORT = 443
|
||||
|
||||
# RFC 1123 host label set. Deliberately not a full IDN implementation: an operator
|
||||
# naming their management host in a config file pushed to HAProxy nodes should use
|
||||
# ASCII, and HAProxy itself would not accept anything else on a `server` line.
|
||||
_HOSTNAME_RE = re.compile(
|
||||
r"^(?=.{1,253}$)[A-Za-z0-9]([A-Za-z0-9-]{0,61}[A-Za-z0-9])?"
|
||||
r"(\.[A-Za-z0-9]([A-Za-z0-9-]{0,61}[A-Za-z0-9])?)*\.?$"
|
||||
)
|
||||
|
||||
_CONTROL_CHARS = frozenset("\t\n\r\v\f\x00")
|
||||
|
||||
|
||||
class AcmeBackendUrlError(ValueError):
|
||||
"""A value that cannot express a usable challenge backend target.
|
||||
|
||||
``code`` is stable and machine-readable so the UI can map it to help text;
|
||||
``args[0]`` is operator-facing prose.
|
||||
"""
|
||||
|
||||
def __init__(self, code: str, message: str):
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
|
||||
|
||||
class AcmeBackendTarget(NamedTuple):
|
||||
"""What the renderer should emit, plus everything worth telling the operator."""
|
||||
|
||||
host: str
|
||||
port: int
|
||||
ssl_flag: str
|
||||
#: Non-fatal observations. Rendered anyway; surfaced in logs and the panel.
|
||||
warnings: List[str]
|
||||
#: Set when the value could not be parsed at all and the caller must not emit
|
||||
#: a `server` line. None on success.
|
||||
error_code: Optional[str]
|
||||
error_message: Optional[str]
|
||||
|
||||
|
||||
def _classify_host(host: str) -> Optional[str]:
|
||||
"""Return a rejection code for hosts that cannot be a management address."""
|
||||
try:
|
||||
ip = ipaddress.ip_address(host)
|
||||
except ValueError:
|
||||
lowered = host.rstrip(".").lower()
|
||||
# `localhost` is loopback by name and is the single most likely wrong value
|
||||
# here — it is what both shipped defaults contain. Catching only the numeric
|
||||
# form would let the exact failure this module exists to prevent straight
|
||||
# through. RFC 6761 also reserves the whole `.localhost` tree.
|
||||
if lowered == "localhost" or lowered.endswith(".localhost"):
|
||||
return "loopback"
|
||||
return None if _HOSTNAME_RE.match(host) else "invalid_host"
|
||||
|
||||
if isinstance(ip, ipaddress.IPv6Address) and ip.ipv4_mapped is not None:
|
||||
ip = ip.ipv4_mapped
|
||||
if ip.is_loopback:
|
||||
return "loopback"
|
||||
if ip.is_unspecified:
|
||||
return "unspecified"
|
||||
# Includes 169.254.169.254, the cloud metadata endpoint.
|
||||
if ip.is_link_local:
|
||||
return "link_local"
|
||||
if ip.is_multicast:
|
||||
return "multicast"
|
||||
# NOTE: private (RFC1918) addresses fall through on purpose — see module docstring.
|
||||
return None
|
||||
|
||||
|
||||
_REJECTION_PROSE = {
|
||||
"too_long": f"URL must be at most {MAX_URL_LENGTH} characters.",
|
||||
"whitespace": (
|
||||
"URL must not contain spaces or line breaks. A trailing space survives parsing "
|
||||
"and would be written into haproxy.cfg as part of the address."
|
||||
),
|
||||
"no_scheme": (
|
||||
"URL must start with http:// or https://. Without a scheme the value cannot be "
|
||||
"parsed as an address and silently falls back to localhost, which on a HAProxy "
|
||||
"node means the node itself."
|
||||
),
|
||||
"bad_scheme": "URL scheme must be http or https.",
|
||||
"userinfo": "URL must not contain credentials.",
|
||||
"has_path": (
|
||||
"Enter only the scheme, host and port — no path, query or fragment. The "
|
||||
"challenge path is appended by HAProxy."
|
||||
),
|
||||
"bad_port": "Port must be a number between 1 and 65535.",
|
||||
"no_host": "URL must contain a host.",
|
||||
"invalid_host": "Host is not a valid IP address or hostname.",
|
||||
"loopback": (
|
||||
"Loopback addresses cannot work here. HAProxy resolves this address on the "
|
||||
"HAProxy node, so 127.0.0.1 means the node itself, not the management server. "
|
||||
"Use the management server's routable address."
|
||||
),
|
||||
"unspecified": (
|
||||
"0.0.0.0 is a listen address, not a destination. Use the management server's "
|
||||
"routable address."
|
||||
),
|
||||
"link_local": "Link-local addresses cannot be used as a management address.",
|
||||
"multicast": "Multicast addresses cannot be used as a management address.",
|
||||
}
|
||||
|
||||
|
||||
def validate_acme_backend_url(value: Optional[str]) -> Optional[str]:
|
||||
"""Validate an operator-supplied URL at the write boundary.
|
||||
|
||||
Returns the normalised value (stripped), or ``None`` for empty input, which
|
||||
legitimately means "inherit from the next level of the resolution chain".
|
||||
Raises :class:`AcmeBackendUrlError` otherwise.
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
if not isinstance(value, str):
|
||||
raise AcmeBackendUrlError("invalid_host", _REJECTION_PROSE["invalid_host"])
|
||||
|
||||
stripped = value.strip()
|
||||
if not stripped:
|
||||
return None
|
||||
|
||||
if len(stripped) > MAX_URL_LENGTH:
|
||||
raise AcmeBackendUrlError("too_long", _REJECTION_PROSE["too_long"])
|
||||
if any(c in _CONTROL_CHARS for c in stripped) or " " in stripped:
|
||||
raise AcmeBackendUrlError("whitespace", _REJECTION_PROSE["whitespace"])
|
||||
|
||||
parsed = urlparse(stripped)
|
||||
|
||||
if not parsed.scheme:
|
||||
raise AcmeBackendUrlError("no_scheme", _REJECTION_PROSE["no_scheme"])
|
||||
if parsed.scheme.lower() not in ALLOWED_SCHEMES:
|
||||
# `10.0.0.5:8080` parses as scheme='10.0.0.5' with no netloc, and
|
||||
# `localhost:8080` as scheme='localhost'. Both are the same operator mistake,
|
||||
# so point at the missing scheme rather than the nonsense one.
|
||||
if not parsed.netloc:
|
||||
raise AcmeBackendUrlError("no_scheme", _REJECTION_PROSE["no_scheme"])
|
||||
raise AcmeBackendUrlError("bad_scheme", _REJECTION_PROSE["bad_scheme"])
|
||||
|
||||
if parsed.username is not None or parsed.password is not None:
|
||||
raise AcmeBackendUrlError("userinfo", _REJECTION_PROSE["userinfo"])
|
||||
if parsed.path not in ("", "/") or parsed.query or parsed.fragment:
|
||||
raise AcmeBackendUrlError("has_path", _REJECTION_PROSE["has_path"])
|
||||
|
||||
try:
|
||||
port = parsed.port
|
||||
except ValueError:
|
||||
# urlparse defers port parsing to attribute access; an out-of-range or
|
||||
# non-numeric port raises here. Unguarded, this exception reaches the config
|
||||
# generator's blanket `except` and collapses the cluster's whole config.
|
||||
raise AcmeBackendUrlError("bad_port", _REJECTION_PROSE["bad_port"]) from None
|
||||
if port is not None and not (1 <= port <= 65535):
|
||||
raise AcmeBackendUrlError("bad_port", _REJECTION_PROSE["bad_port"])
|
||||
|
||||
host = parsed.hostname
|
||||
if not host:
|
||||
raise AcmeBackendUrlError("no_host", _REJECTION_PROSE["no_host"])
|
||||
|
||||
code = _classify_host(host)
|
||||
if code is not None:
|
||||
raise AcmeBackendUrlError(code, _REJECTION_PROSE[code])
|
||||
|
||||
return stripped
|
||||
|
||||
|
||||
def resolve_acme_backend_target(url: Optional[str]) -> AcmeBackendTarget:
|
||||
"""Resolve a stored URL into what the renderer emits. Never raises.
|
||||
|
||||
Values already in the database predate validation (and the shipped defaults are
|
||||
themselves loopback), so anything unparseable or discouraged is reported through
|
||||
``warnings`` / ``error_code`` rather than refused.
|
||||
"""
|
||||
warnings: List[str] = []
|
||||
raw = (url or "").strip()
|
||||
|
||||
if not raw:
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "empty", "No challenge backend URL configured."
|
||||
)
|
||||
|
||||
if any(c in _CONTROL_CHARS for c in raw) or " " in raw:
|
||||
# Must never reach haproxy.cfg: a newline here writes attacker- or
|
||||
# accident-chosen directives into a file pushed to every node.
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "whitespace", _REJECTION_PROSE["whitespace"]
|
||||
)
|
||||
|
||||
parsed = urlparse(raw)
|
||||
scheme = (parsed.scheme or "").lower()
|
||||
|
||||
try:
|
||||
port = parsed.port
|
||||
except ValueError:
|
||||
return AcmeBackendTarget("", 0, "", warnings, "bad_port", _REJECTION_PROSE["bad_port"])
|
||||
|
||||
host = parsed.hostname
|
||||
if not host or scheme not in ALLOWED_SCHEMES:
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "no_scheme", _REJECTION_PROSE["no_scheme"]
|
||||
)
|
||||
|
||||
if port is None:
|
||||
port = DEFAULT_HTTPS_PORT if scheme == "https" else DEFAULT_HTTP_PORT
|
||||
warnings.append(
|
||||
f"No port given, assuming {port}. State the port explicitly — the assumed "
|
||||
f"value is the bundled reverse proxy's port, not the scheme's default."
|
||||
)
|
||||
|
||||
code = _classify_host(host)
|
||||
if code == "invalid_host":
|
||||
return AcmeBackendTarget("", 0, "", warnings, code, _REJECTION_PROSE[code])
|
||||
if code is not None:
|
||||
warnings.append(_REJECTION_PROSE[code])
|
||||
|
||||
ssl_flag = " ssl verify none" if scheme == "https" else ""
|
||||
return AcmeBackendTarget(host, port, ssl_flag, warnings, None, None)
|
||||
@@ -1711,6 +1711,45 @@ fetch_and_deploy_keepalived_config() {
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
|
||||
# v1.10.4 — VIP adoption discovery. Report a keepalived.conf we do NOT own so an existing
|
||||
# VIP can be adopted from the UI instead of retyped. STRICTLY READ-ONLY: this never writes
|
||||
# to the node. The heartbeat cannot carry this — it has the VIP address and a best-effort
|
||||
# MASTER/BACKUP, while rendering a node's config needs eleven fields.
|
||||
#
|
||||
# Rate limited by content: the hash of the last report is cached next to the config, so the
|
||||
# file (which may contain the VRRP password) is posted only when it actually changes, not on
|
||||
# every cycle. Once we own the file there is nothing to adopt, so the record is cleared once.
|
||||
_kp_discover() {
|
||||
local cache="$(dirname "$conf")/.hom_discovery_hash" cur_hash="" body content_json
|
||||
if [[ "$we_own" == "true" || ! -f "$conf" ]]; then
|
||||
# Nothing adoptable here. Clear a previous report exactly once.
|
||||
[[ -f "$cache" ]] || return 0
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"config_path\":\"$conf\",\"exists\":false}" >/dev/null 2>&1 || return 0
|
||||
rm -f "$cache"
|
||||
return 0
|
||||
fi
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
[[ -z "$cur_hash" ]] && return 0
|
||||
[[ -f "$cache" && "$(cat "$cache" 2>/dev/null)" == "$cur_hash" ]] && return 0
|
||||
# jq -Rs makes the file a single JSON string with its newlines intact, so the content the
|
||||
# server hashes is byte-identical to what is on disk — the takeover authorisation is
|
||||
# pinned to that hash.
|
||||
content_json=$(jq -Rs . < "$conf" 2>/dev/null) || return 0
|
||||
body=$(jq -n --arg p "$conf" --argjson c "$content_json" \
|
||||
'{config_path:$p, exists:true, is_managed:false, config_content:$c}' 2>/dev/null) || return 0
|
||||
if curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -X POST \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
--data-binary "$body" 2>/dev/null; then
|
||||
printf '%s' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "INFO" "KEEPALIVED: reported an unmanaged keepalived.conf for adoption"
|
||||
fi
|
||||
}
|
||||
_kp_discover
|
||||
|
||||
_kp_report() { # $1=state $2=vip_id(or empty) $3=hash $4=message
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
@@ -1774,10 +1813,29 @@ fetch_and_deploy_keepalived_config() {
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
# Ownership guard: never overwrite a keepalived.conf we don't own.
|
||||
#
|
||||
# v1.10.4 adoption is the ONE exception, and it does not weaken the guard: the server
|
||||
# authorises a single takeover of a specific file by pinning the md5 the operator adopted
|
||||
# from. We overwrite only when that hash still matches what is on disk, so a config edited
|
||||
# between adoption and Apply is still refused — the operator's later edit wins over a stale
|
||||
# adoption rather than being silently destroyed.
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
local allow_takeover expected_hash disk_hash
|
||||
allow_takeover=$(echo "$resp" | jq -r '.keepalived.allow_takeover // false' 2>/dev/null)
|
||||
expected_hash=$(echo "$resp" | jq -r '.keepalived.takeover_expected_hash // empty' 2>/dev/null)
|
||||
disk_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
if [[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]; then
|
||||
log "INFO" "KEEPALIVED: adopting $conf (one-shot takeover authorised; on-disk hash matches)"
|
||||
elif [[ "$allow_takeover" == "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: adoption authorised but $conf changed since it was adopted — refusing"
|
||||
_kp_report "externally_managed" "$vip_id" "$disk_hash" \
|
||||
"config changed after adoption; re-adopt to pick up the current file"
|
||||
return 0
|
||||
else
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# Hybrid install: install keepalived only if missing.
|
||||
@@ -1807,6 +1865,13 @@ fetch_and_deploy_keepalived_config() {
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
# STILL ACK. This report is the server's only evidence that the node converged, and
|
||||
# it used to be sent on the write path alone — so a single lost ack (a backend
|
||||
# restart, a 5xx, a network blip) left the VIP reading SYNCING forever: the node was
|
||||
# already correct on disk, took this early return every cycle, and never spoke again.
|
||||
# Re-asserting the state makes the loop self-healing, costs one request per ~2.5
|
||||
# minutes, and touches nothing on the node.
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "already converged"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
@@ -1822,11 +1887,43 @@ fetch_and_deploy_keepalived_config() {
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
# Capture what keepalived actually said. Discarding it made the fail-safe useless in
|
||||
# practice: the node was correctly protected, but neither the log nor the UI could say WHY,
|
||||
# so the only way forward was to reproduce the check by hand on the node. Sanitised hard
|
||||
# (quotes, backslashes and newlines removed, tail kept) because both `log` and _kp_report
|
||||
# embed the text in JSON built by string interpolation.
|
||||
local kp_out kp_err kp_fatal
|
||||
if ! kp_out=$(keepalived -t -f "$tmp_conf" 2>&1); then
|
||||
# keepalived's config-test EXIT CODE does not separate a fatal config error from a
|
||||
# harmless warning. Measured on 2.2.8, not assumed:
|
||||
# clean config .................. 0
|
||||
# auth_pass longer than 8 chars . 5 "Truncating auth_pass to 8 characters"
|
||||
# missing '}' ................... 5 "There are 1 missing '}'s"
|
||||
# unknown keyword ............... 5 "Unknown keyword '...'"
|
||||
# script without script_security 6 "SECURITY VIOLATION ..."
|
||||
# So 5 covers both a benign truncation and a broken file, and treating any non-zero
|
||||
# exit as invalid rejected VALID configs: a VRRP password over 8 characters is enough,
|
||||
# and keepalived truncates it to 8 regardless, exactly as it does for the file the
|
||||
# operator already runs. Judge on the OUTPUT instead, dropping only messages known to
|
||||
# be benign; anything unrecognised is still fatal, so this fails CLOSED.
|
||||
if [[ -z "${kp_out//[[:space:]]/}" ]]; then
|
||||
# Non-zero with NOTHING to read. We cannot confirm the reason is benign, and some
|
||||
# builds log to syslog rather than stderr, so proceeding here would silently accept
|
||||
# every config on such a host. Treat as fatal — the gate must fail closed.
|
||||
kp_fatal="keepalived -t exited non-zero without output"
|
||||
else
|
||||
kp_fatal=$(printf '%s\n' "$kp_out" \
|
||||
| grep -v 'Truncating auth_pass to 8 characters' \
|
||||
| grep -v '^[[:space:]]*$' || true)
|
||||
fi
|
||||
if [[ -n "$kp_fatal" ]]; then
|
||||
rm -f "$tmp_conf"
|
||||
kp_err=$(printf '%s' "$kp_fatal" | tr '\n\r\t' ' ' | tr -d '"\\' | tail -c 300)
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t): ${kp_err} — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed: ${kp_err}"
|
||||
return 0
|
||||
fi
|
||||
log "WARN" "KEEPALIVED: keepalived -t exited non-zero with only known-benign warnings; proceeding"
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
@@ -3262,6 +3359,45 @@ CONFIG_RESPONSE_EOF
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
|
||||
# v1.10.4 — VIP adoption discovery. Report a keepalived.conf we do NOT own so an existing
|
||||
# VIP can be adopted from the UI instead of retyped. STRICTLY READ-ONLY: this never writes
|
||||
# to the node. The heartbeat cannot carry this — it has the VIP address and a best-effort
|
||||
# MASTER/BACKUP, while rendering a node's config needs eleven fields.
|
||||
#
|
||||
# Rate limited by content: the hash of the last report is cached next to the config, so the
|
||||
# file (which may contain the VRRP password) is posted only when it actually changes, not on
|
||||
# every cycle. Once we own the file there is nothing to adopt, so the record is cleared once.
|
||||
_kp_discover() {
|
||||
local cache="$(dirname "$conf")/.hom_discovery_hash" cur_hash="" body content_json
|
||||
if [[ "$we_own" == "true" || ! -f "$conf" ]]; then
|
||||
# Nothing adoptable here. Clear a previous report exactly once.
|
||||
[[ -f "$cache" ]] || return 0
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"config_path\":\"$conf\",\"exists\":false}" >/dev/null 2>&1 || return 0
|
||||
rm -f "$cache"
|
||||
return 0
|
||||
fi
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
[[ -z "$cur_hash" ]] && return 0
|
||||
[[ -f "$cache" && "$(cat "$cache" 2>/dev/null)" == "$cur_hash" ]] && return 0
|
||||
# jq -Rs makes the file a single JSON string with its newlines intact, so the content the
|
||||
# server hashes is byte-identical to what is on disk — the takeover authorisation is
|
||||
# pinned to that hash.
|
||||
content_json=$(jq -Rs . < "$conf" 2>/dev/null) || return 0
|
||||
body=$(jq -n --arg p "$conf" --argjson c "$content_json" \
|
||||
'{config_path:$p, exists:true, is_managed:false, config_content:$c}' 2>/dev/null) || return 0
|
||||
if curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -X POST \
|
||||
"$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
--data-binary "$body" 2>/dev/null; then
|
||||
printf '%s' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "INFO" "KEEPALIVED: reported an unmanaged keepalived.conf for adoption"
|
||||
fi
|
||||
}
|
||||
_kp_discover
|
||||
|
||||
_kp_report() {
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
curl -k -s --connect-timeout 10 --max-time 30 -X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-status" \
|
||||
@@ -3321,11 +3457,26 @@ CONFIG_RESPONSE_EOF
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
local allow_takeover expected_hash disk_hash
|
||||
allow_takeover=$(echo "$resp" | jq -r '.keepalived.allow_takeover // false' 2>/dev/null)
|
||||
expected_hash=$(echo "$resp" | jq -r '.keepalived.takeover_expected_hash // empty' 2>/dev/null)
|
||||
disk_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
if [[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]; then
|
||||
log "INFO" "KEEPALIVED: adopting $conf (one-shot takeover authorised; on-disk hash matches)"
|
||||
elif [[ "$allow_takeover" == "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: adoption authorised but $conf changed since it was adopted — refusing"
|
||||
_kp_report "externally_managed" "$vip_id" "$disk_hash" \
|
||||
"config changed after adoption; re-adopt to pick up the current file"
|
||||
return 0
|
||||
else
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# Hybrid install: install keepalived only if missing.
|
||||
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
if [[ "$install_if" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: installing package..."
|
||||
@@ -3351,6 +3502,9 @@ CONFIG_RESPONSE_EOF
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
# See the heredoc copy: the ack must be re-asserted here or a single lost report
|
||||
# leaves the VIP reading SYNCING forever.
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "already converged"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
@@ -3366,11 +3520,26 @@ CONFIG_RESPONSE_EOF
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
# See the heredoc copy for why the output is captured rather than discarded.
|
||||
# See the heredoc copy for the measured exit-code table and why the OUTPUT, not the
|
||||
# exit code, decides. Fails closed on anything not known to be benign.
|
||||
local kp_out kp_err kp_fatal
|
||||
if ! kp_out=$(keepalived -t -f "$tmp_conf" 2>&1); then
|
||||
if [[ -z "${kp_out//[[:space:]]/}" ]]; then
|
||||
kp_fatal="keepalived -t exited non-zero without output"
|
||||
else
|
||||
kp_fatal=$(printf '%s\n' "$kp_out" \
|
||||
| grep -v 'Truncating auth_pass to 8 characters' \
|
||||
| grep -v '^[[:space:]]*$' || true)
|
||||
fi
|
||||
if [[ -n "$kp_fatal" ]]; then
|
||||
rm -f "$tmp_conf"
|
||||
kp_err=$(printf '%s' "$kp_fatal" | tr '\n\r\t' ' ' | tr -d '"\\' | tail -c 300)
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t): ${kp_err} — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed: ${kp_err}"
|
||||
return 0
|
||||
fi
|
||||
log "WARN" "KEEPALIVED: keepalived -t exited non-zero with only known-benign warnings; proceeding"
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Issue #53 — at-rest encryption for the pending CSR private key (v1.10.1).
|
||||
|
||||
Mirrors the established Fernet + HKDF(SECRET_KEY) pattern already used for the VRRP secret
|
||||
(services/keepalived_config.py), TOTP secrets (services/mfa_service.py) and DNS provider
|
||||
credentials (utils/dns_credentials.py): prefer an explicit CSR_ENCRYPTION_KEY env var (enables
|
||||
key rotation), else derive a stable key from SECRET_KEY via HKDF with its own versioned info
|
||||
string, so a rotation of one secret class never affects another.
|
||||
|
||||
WHY this key and not every key in the system: the CSR private key is the one key that sits IDLE.
|
||||
It is generated at CSR creation, waits for an external CA to sign the request (days to weeks),
|
||||
and is destroyed the moment the signed certificate is imported — it is never transmitted to an
|
||||
agent and never leaves the server. `ssl_certificates.private_key_content` and the ACME order keys
|
||||
are different: agents must receive them in plaintext on every poll, so encrypting them at rest
|
||||
buys nothing without an end-to-end redesign.
|
||||
|
||||
STORAGE: the Fernet token replaces the PEM in the SAME `ssl_csrs.private_key_pem` TEXT column.
|
||||
No new column, no new table, and deliberately NO `SCHEMA_VERSION` bump — a bump would re-run the
|
||||
migration sequence and re-seed the four built-in roles to their defaults (see UPGRADE_GUIDE.md),
|
||||
which is a needless side effect for a storage-format change.
|
||||
|
||||
BACKWARD COMPATIBILITY: rows written before this release hold a raw PEM. `decrypt_csr_private_key`
|
||||
detects those by their `-----BEGIN` header and returns them unchanged. The discriminator is exact,
|
||||
not a heuristic: a Fernet token is base64url text and can never contain "-----". Legacy rows drain
|
||||
naturally, since a CSR's key copy is NULLed on import.
|
||||
|
||||
KEY ROTATION: if SECRET_KEY rotates while CSR_ENCRYPTION_KEY is unset, previously stored keys
|
||||
become undecryptable and `decrypt_csr_private_key` returns None. Callers MUST surface a clear
|
||||
"delete this CSR and create a new one" error — the CSR is unusable at that point, because the
|
||||
signed certificate can no longer be paired with its key.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from cryptography.fernet import Fernet, InvalidToken
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
from config import SECRET_KEY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A PEM private key always carries this header; a Fernet token is base64url and never can.
|
||||
_PEM_MARKER = "-----BEGIN"
|
||||
|
||||
_fernet_instance: Optional[Fernet] = None
|
||||
|
||||
|
||||
def _resolve_fernet_key() -> bytes:
|
||||
"""Prefer an explicit CSR_ENCRYPTION_KEY; else derive from SECRET_KEY via HKDF with a
|
||||
versioned info string (so stored keys survive restarts)."""
|
||||
explicit = os.getenv("CSR_ENCRYPTION_KEY", "").strip()
|
||||
if explicit:
|
||||
try:
|
||||
Fernet(explicit.encode())
|
||||
return explicit.encode()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("CSR_ENCRYPTION_KEY env var present but invalid: %s", exc)
|
||||
logger.warning(
|
||||
"CSR_ENCRYPTION_KEY not set; deriving the CSR private-key encryption key from SECRET_KEY. "
|
||||
"Set CSR_ENCRYPTION_KEY to a Fernet key to enable key rotation."
|
||||
)
|
||||
hkdf = HKDF(algorithm=hashes.SHA256(), length=32, salt=None, info=b"csr-private-key-v1")
|
||||
derived = hkdf.derive(SECRET_KEY.encode("utf-8"))
|
||||
return base64.urlsafe_b64encode(derived)
|
||||
|
||||
|
||||
def _get_fernet() -> Fernet:
|
||||
global _fernet_instance
|
||||
if _fernet_instance is None:
|
||||
_fernet_instance = Fernet(_resolve_fernet_key())
|
||||
return _fernet_instance
|
||||
|
||||
|
||||
def reset_fernet_for_tests() -> None:
|
||||
"""Test-only hook to force re-resolution after env mutation."""
|
||||
global _fernet_instance
|
||||
_fernet_instance = None
|
||||
|
||||
|
||||
def is_encrypted(stored: Optional[str]) -> bool:
|
||||
"""True when the stored value is a Fernet token rather than a legacy raw PEM.
|
||||
|
||||
Single source of the format discriminator: `decrypt_csr_private_key` branches on this, so
|
||||
the "what does a stored value look like" rule is stated exactly once.
|
||||
"""
|
||||
return bool(stored) and _PEM_MARKER not in stored
|
||||
|
||||
|
||||
def encrypt_csr_private_key(pem: str) -> str:
|
||||
"""Fernet-encrypt a PEM private key to a storable token string."""
|
||||
return _get_fernet().encrypt(pem.encode("utf-8")).decode("utf-8")
|
||||
|
||||
|
||||
def decrypt_csr_private_key(stored: Optional[str]) -> Optional[str]:
|
||||
"""Return the PEM private key for a stored value.
|
||||
|
||||
Accepts BOTH shapes so an upgrade needs no data migration:
|
||||
- a raw PEM written before v1.10.1 -> returned unchanged
|
||||
- a Fernet token -> decrypted
|
||||
|
||||
Returns None when the value is empty or cannot be decrypted (e.g. SECRET_KEY rotated without
|
||||
CSR_ENCRYPTION_KEY). Callers MUST treat None as "this CSR's key is unrecoverable" and tell the
|
||||
operator to delete it and create a new one; never fall through to a pairing attempt.
|
||||
"""
|
||||
if not stored:
|
||||
return None
|
||||
if not is_encrypted(stored):
|
||||
return stored # legacy plaintext row, pre-v1.10.1
|
||||
try:
|
||||
return _get_fernet().decrypt(stored.encode("utf-8")).decode("utf-8")
|
||||
except InvalidToken:
|
||||
logger.warning("Failed to decrypt a stored CSR private key (invalid Fernet token)")
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Unexpected error decrypting a stored CSR private key: %s", exc)
|
||||
return None
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": "1.9.0",
|
||||
"releaseName": "CSR creation — in-app key+CSR generation and signed-certificate import",
|
||||
"releaseDate": "2026-08-04"
|
||||
"version": "1.10.14",
|
||||
"releaseName": "A converged node keeps acknowledging, so a lost report self-heals",
|
||||
"releaseDate": "2026-08-14"
|
||||
}
|
||||
|
||||
+11
-2
@@ -49,13 +49,22 @@ services:
|
||||
- SECRET_KEY=your-secret-key-change-this-in-production
|
||||
- DEBUG=False
|
||||
- LOG_LEVEL=INFO
|
||||
- PUBLIC_URL=http://localhost:8080
|
||||
- MANAGEMENT_BASE_URL=http://localhost:8080
|
||||
# Interpolated, not hardcoded: these were literals, so a value set in the
|
||||
# host environment or .env was silently ignored and every install kept the
|
||||
# localhost default. That default is also what the ACME challenge backend
|
||||
# falls back to, and HAProxy resolves it ON THE HAPROXY NODE — so on any
|
||||
# deployment where HAProxy is not this machine, it points at the wrong box.
|
||||
- PUBLIC_URL=${PUBLIC_URL:-http://localhost:8080}
|
||||
- MANAGEMENT_BASE_URL=${MANAGEMENT_BASE_URL:-http://localhost:8080}
|
||||
# Empty when unset on the host: the image CMD then falls back to
|
||||
# WEB_CONCURRENCY (uvicorn's native env) and finally to 1.
|
||||
- UVICORN_WORKERS=${UVICORN_WORKERS:-}
|
||||
volumes:
|
||||
- haproxy_configs:/etc/haproxy
|
||||
# NOT published to the host: the API is reachable only through the nginx
|
||||
# service (host :8080). Host port 8000 is therefore NOT this API — on a box
|
||||
# running Portainer it is Portainer's edge tunnel, which answers 404 and looks
|
||||
# deceptively like a working challenge endpoint.
|
||||
expose:
|
||||
- "8000"
|
||||
depends_on:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "haproxy-openmanager-frontend",
|
||||
"version": "1.9.0",
|
||||
"version": "1.10.14",
|
||||
"description": "HAProxy Load Balancer Management UI",
|
||||
"license": "AGPL-3.0-or-later",
|
||||
"dependencies": {
|
||||
|
||||
+23
-5
@@ -484,12 +484,30 @@ function AppContent() {
|
||||
|
||||
function ThemedApp() {
|
||||
const { isDarkMode } = useTheme();
|
||||
const antdTheme = {
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
};
|
||||
|
||||
// Ant Design 5: the STATIC message/notification/Modal.confirm APIs render into their own
|
||||
// detached root, so they do not see this ConfigProvider and always fall back to the light
|
||||
// algorithm — a confirm dialog came up white while the app was in dark mode. `holderRender`
|
||||
// wraps that detached root in the same ConfigProvider, which fixes every static call in the
|
||||
// app at once (12 components use Modal.confirm) instead of migrating each one to
|
||||
// App.useApp(). In an effect rather than during render: ConfigProvider.config() mutates
|
||||
// antd module state, and effects still run long before a user can click anything that opens
|
||||
// a static modal. Re-registered on theme change so the toggle takes effect immediately.
|
||||
React.useEffect(() => {
|
||||
ConfigProvider.config({
|
||||
holderRender: (children) => (
|
||||
<ConfigProvider theme={{ algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm }}>
|
||||
{children}
|
||||
</ConfigProvider>
|
||||
),
|
||||
});
|
||||
}, [isDarkMode]);
|
||||
|
||||
return (
|
||||
<ConfigProvider
|
||||
theme={{
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
}}
|
||||
>
|
||||
<ConfigProvider theme={antdTheme}>
|
||||
<AuthProvider>
|
||||
<ClusterProvider>
|
||||
<ProgressProvider>
|
||||
|
||||
@@ -107,8 +107,14 @@ const ACMEAutomation = () => {
|
||||
const [confirming, setConfirming] = useState(false);
|
||||
const regChallengeType = Form.useWatch('challenge_type', registerForm);
|
||||
const regDnsProvider = Form.useWatch('dns_provider', registerForm);
|
||||
const wizardAccountId = Form.useWatch('account_id', wizardForm);
|
||||
const wizardDomains = Form.useWatch('domains', wizardForm);
|
||||
// `preserve: true` is load-bearing, not a nicety. Each wizard step renders only its own fields —
|
||||
// the domain list on Domains, the account Select on Configuration — and a plain useWatch reports
|
||||
// only fields that are currently REGISTERED, so every one of these read `undefined` from the
|
||||
// Review step onward even though the values were still in the form store. That is what made
|
||||
// Review (and the request it submits) silently fall back to the default ACME account, and it
|
||||
// disabled the wildcard guard at exactly the step where Submit lives.
|
||||
const wizardAccountId = Form.useWatch('account_id', { form: wizardForm, preserve: true });
|
||||
const wizardDomains = Form.useWatch('domains', { form: wizardForm, preserve: true });
|
||||
const selectedDnsProvider = dnsProviders.find(p => p.name === regDnsProvider) || null;
|
||||
|
||||
// Issue #35: per-account DNS credential management (view/replace/clear after creation).
|
||||
@@ -217,7 +223,16 @@ const ACMEAutomation = () => {
|
||||
const pendingOrders = orders.filter(o =>
|
||||
o.status === 'pending' || o.status === 'processing' || o.status === 'ready' || isOrderStuck(o)
|
||||
);
|
||||
const activeAccount = accounts.find(a => a.status === 'valid') || null;
|
||||
// The account a request lands on when it carries no explicit account_id. This MUST match the
|
||||
// backend, which takes `ORDER BY created_at DESC LIMIT 1` (routers/letsencrypt.py). The list
|
||||
// arrives ORDER BY id, so picking the first valid entry would preview the OLDEST account — the
|
||||
// opposite one. With two accounts of different challenge methods that made the wizard describe
|
||||
// DNS-01 while the request would actually have gone to an HTTP-01 account.
|
||||
const activeAccount = accounts.filter(a => a.status === 'valid').reduce((best, a) => {
|
||||
if (!best) return a;
|
||||
const delta = new Date(a.created_at) - new Date(best.created_at);
|
||||
return delta > 0 || (delta === 0 && a.id > best.id) ? a : best;
|
||||
}, null);
|
||||
const acmeAccount = activeAccount || (accounts.length > 0 ? accounts[accounts.length - 1] : null);
|
||||
const acmeEnabledClusters = clusters.filter(c => c.acme_enabled && c.is_active);
|
||||
// Issue #35: the cert wizard adapts to the selected account's challenge method.
|
||||
@@ -260,18 +275,26 @@ const ACMEAutomation = () => {
|
||||
return;
|
||||
}
|
||||
setSubmitting(true);
|
||||
// Resolve the chosen account's challenge method so DNS-01/wildcard requests are explicit.
|
||||
// Use the same resolution as the wizard description (wizardAccount) so what the user reviewed
|
||||
// matches what is sent.
|
||||
const challengeType = wizardAccount?.challenge_type; // 'http-01' | 'dns-01' | undefined
|
||||
// Resolve the account ONCE and derive everything else from that single object. The id and the
|
||||
// challenge method used to come from different places — account_id from the form store,
|
||||
// challenge_type from wizardAccount — so whenever those two disagreed the request asked for
|
||||
// DNS-01 validation on an HTTP-01 account and the backend answered "The selected ACME account
|
||||
// has no DNS provider configured for DNS-01."
|
||||
const selectedAccountId = values.account_id ?? wizardAccount?.id ?? null;
|
||||
const account = accounts.find(a => a.id === selectedAccountId) || wizardAccount || null;
|
||||
const challengeType = account?.challenge_type; // 'http-01' | 'dns-01' | undefined
|
||||
// Manual DNS-01 can't auto-renew (the wizard shows the switch off+disabled). Send false to
|
||||
// match the displayed state rather than relying only on the backend to override it.
|
||||
const autoRenew = wizardDnsManual ? false : (values.auto_renew !== false);
|
||||
const isManualDns01 = challengeType === 'dns-01' && (account?.dns_provider || 'manual') === 'manual';
|
||||
const autoRenew = isManualDns01 ? false : (values.auto_renew !== false);
|
||||
const res = await axios.post('/api/letsencrypt/certificates', {
|
||||
domains: values.domains,
|
||||
cluster_ids: values.cluster_ids || [],
|
||||
auto_renew: autoRenew,
|
||||
account_id: values.account_id || null,
|
||||
// Always explicit: sending the resolved id removes the frontend/backend "default account"
|
||||
// guess, which disagreed (the UI previewed the oldest valid account, the backend used the
|
||||
// newest) and made the Review step describe an account the request never went to.
|
||||
account_id: account?.id ?? null,
|
||||
challenge_type: challengeType || undefined,
|
||||
});
|
||||
message.success(res.data?.message || 'Certificate request submitted');
|
||||
@@ -1092,7 +1115,17 @@ const ACMEAutomation = () => {
|
||||
message="Prerequisite Check"
|
||||
description={
|
||||
<ul style={{ margin: 0, paddingLeft: 20 }}>
|
||||
<li>ACME Account: {activeAccount ? <Tag color="success">Active ({activeAccount.email})</Tag> : <Tag color="error">No active account</Tag>}</li>
|
||||
{/* The account the request will actually use — NOT `activeAccount`, which is only
|
||||
the default and would name a different account whenever the user picked one. */}
|
||||
<li>ACME Account: {wizardAccount
|
||||
? <Tag color={wizardAccount.status === 'valid' ? 'success' : 'error'}>
|
||||
{wizardAccount.email}{wizardAccount.status !== 'valid' ? ` (${wizardAccount.status})` : ''}
|
||||
</Tag>
|
||||
: <Tag color="error">No active account</Tag>}
|
||||
{wizardAccount && accounts.length > 1 && !wizardAccountId && (
|
||||
<Typography.Text type="secondary" style={{ fontSize: 12 }}> (default)</Typography.Text>
|
||||
)}
|
||||
</li>
|
||||
{wizardIsDns01 ? (
|
||||
<>
|
||||
<li>Challenge Method: <Tag>DNS-01</Tag> (TXT record; no port 80 / ACME routing needed)</li>
|
||||
@@ -1324,8 +1357,11 @@ const ACMEAutomation = () => {
|
||||
Next
|
||||
</Button>
|
||||
)}
|
||||
{/* Gate on the account the request will actually use, and on its status: with no valid
|
||||
account wizardAccount falls back to the newest (deactivated) one, and the Select lists
|
||||
deactivated accounts too, so a bare null-check would leave Submit enabled. */}
|
||||
{wizardStep === wizardSteps.length - 1 && (
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={!activeAccount || wizardWildcardBlocked || wizardDns01Disabled || (!wizardIsDns01 && acmeEnabledClusters.length === 0)}>
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={wizardAccount?.status !== 'valid' || wizardWildcardBlocked || wizardDns01Disabled || (!wizardIsDns01 && acmeEnabledClusters.length === 0)}>
|
||||
Submit Request
|
||||
</Button>
|
||||
)}
|
||||
|
||||
@@ -1092,7 +1092,7 @@ const ApplyManagement = () => {
|
||||
style={{
|
||||
marginBottom: 24,
|
||||
borderRadius: 8,
|
||||
border: '1px solid #ffccc7',
|
||||
border: `1px solid ${token.colorErrorBorder}`,
|
||||
boxShadow: '0 2px 8px rgba(255, 77, 79, 0.15)'
|
||||
}}
|
||||
message={
|
||||
@@ -1348,7 +1348,7 @@ const ApplyManagement = () => {
|
||||
HA / VIP Changes ({pendingChanges.vips.length})
|
||||
</Title>
|
||||
{pendingChanges.vips.map(item => (
|
||||
<div key={`vip-${item.id}`} style={{ display: 'flex', alignItems: 'center', gap: 8, padding: '8px 12px', marginBottom: 6, border: item.pending_delete ? '1px solid #ffccc7' : '1px solid #f0f0f0', borderRadius: 6, background: item.pending_delete ? '#fff1f0' : undefined }}>
|
||||
<div key={`vip-${item.id}`} style={{ display: 'flex', alignItems: 'center', gap: 8, padding: '8px 12px', marginBottom: 6, border: `1px solid ${item.pending_delete ? token.colorErrorBorder : token.colorBorderSecondary}`, borderRadius: 6, background: item.pending_delete ? token.colorErrorBg : undefined }}>
|
||||
<CloudServerOutlined style={{ color: item.pending_delete ? '#cf1322' : '#13c2c2' }} />
|
||||
<span style={{ fontWeight: 500 }}>{item.name}</span>
|
||||
<Tag>{item.virtual_ip}/{item.prefix_length}</Tag>
|
||||
@@ -1374,8 +1374,8 @@ const ApplyManagement = () => {
|
||||
const isEnable = /^cluster-\d+-acme-enable-/.test(v.version_name);
|
||||
return (
|
||||
<div key={v.id} style={{
|
||||
padding: 10, border: '1px dashed #1890ff', borderRadius: 6, marginBottom: 8,
|
||||
display: 'flex', alignItems: 'center', justifyContent: 'space-between', backgroundColor: '#f0f8ff'
|
||||
padding: 10, border: `1px dashed ${token.colorPrimary}`, borderRadius: 6, marginBottom: 8,
|
||||
display: 'flex', alignItems: 'center', justifyContent: 'space-between', backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<span>
|
||||
@@ -1419,7 +1419,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f0f8ff'
|
||||
backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1443,7 +1443,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f6ffed'
|
||||
backgroundColor: token.colorSuccessBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1518,9 +1518,13 @@ const ApplyManagement = () => {
|
||||
</Descriptions>
|
||||
</div>
|
||||
|
||||
{/* Pending Versions Section */}
|
||||
{/* Pending Versions Section.
|
||||
Theme tokens, not the light-mode literals #fffbe6/#ffe58f: in dark mode those
|
||||
produced a cream panel with light text on it, so the version name, timestamp
|
||||
and "View Change" link were unreadable. colorWarningBg/Border track the
|
||||
algorithm, so the "pending" tint survives in both themes. */}
|
||||
{pendingVersions.length > 0 && (
|
||||
<div style={{ marginBottom: 24, background: '#fffbe6', borderRadius: 8, border: '1px solid #ffe58f', padding: '16px 16px 8px' }}>
|
||||
<div style={{ marginBottom: 24, background: token.colorWarningBg, borderRadius: 8, border: `1px solid ${token.colorWarningBorder}`, padding: '16px 16px 8px' }}>
|
||||
<Title level={5} style={{ marginTop: 0 }}>
|
||||
<ClockCircleOutlined style={{ marginRight: 8, color: '#faad14' }} />
|
||||
Pending Changes ({pendingVersions.length})
|
||||
@@ -1765,8 +1769,8 @@ const ApplyManagement = () => {
|
||||
<div>
|
||||
{/* Parsed suggestion */}
|
||||
{agentSync.parsed_error?.suggestion && (
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: '#fff7e6', borderRadius: 4, border: '1px solid #ffd591' }}>
|
||||
<Text strong style={{ color: '#ad4e00' }}>Recommendation: </Text>
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: token.colorWarningBg, borderRadius: 4, border: `1px solid ${token.colorWarningBorder}` }}>
|
||||
<Text strong style={{ color: token.colorWarningText }}>Recommendation: </Text>
|
||||
<Text>{agentSync.parsed_error.suggestion}</Text>
|
||||
</div>
|
||||
)}
|
||||
@@ -1951,13 +1955,17 @@ const ApplyManagement = () => {
|
||||
{diffData.changes && diffData.changes.length > 0 ? (
|
||||
diffData.changes.map((change, index) => (
|
||||
<div key={index} style={{ marginBottom: '4px' }}>
|
||||
{/* Token-based, not the light literals #f6ffed/#fff2f0: those stayed
|
||||
near-white in dark mode, so the added/removed rows glared against
|
||||
the dark diff panel around them. colorSuccessBg/colorErrorBg darken
|
||||
with the algorithm while keeping the green/red semantics. */}
|
||||
{change.type === 'added' && (
|
||||
<div style={{ backgroundColor: '#f6ffed', color: '#52c41a', padding: '2px 8px', borderLeft: '3px solid #52c41a' }}>
|
||||
<div style={{ backgroundColor: token.colorSuccessBg, color: token.colorSuccessText, padding: '2px 8px', borderLeft: `3px solid ${token.colorSuccess}` }}>
|
||||
+ {change.line}
|
||||
</div>
|
||||
)}
|
||||
{change.type === 'removed' && (
|
||||
<div style={{ backgroundColor: '#fff2f0', color: '#ff4d4f', padding: '2px 8px', borderLeft: '3px solid #ff4d4f' }}>
|
||||
<div style={{ backgroundColor: token.colorErrorBg, color: token.colorErrorText, padding: '2px 8px', borderLeft: `3px solid ${token.colorError}` }}>
|
||||
- {change.line}
|
||||
</div>
|
||||
)}
|
||||
|
||||
@@ -160,7 +160,8 @@ const ClusterManagement = () => {
|
||||
agent_pool_id: cluster.pool_id || undefined,
|
||||
haproxy_user: cluster.haproxy_user || '',
|
||||
haproxy_group: cluster.haproxy_group || '',
|
||||
acme_enabled: cluster.acme_enabled || false
|
||||
acme_enabled: cluster.acme_enabled || false,
|
||||
acme_backend_url: cluster.acme_backend_url || ''
|
||||
};
|
||||
|
||||
console.log('🔍 CLUSTER EDIT DEBUG - Form values being set:', formValues);
|
||||
@@ -202,7 +203,11 @@ const ClusterManagement = () => {
|
||||
pool_id: values.agent_pool_id || null,
|
||||
haproxy_user: values.haproxy_user || null,
|
||||
haproxy_group: values.haproxy_group || null,
|
||||
acme_enabled: values.acme_enabled || false
|
||||
acme_enabled: values.acme_enabled || false,
|
||||
// Send null, not '', when cleared: null means "inherit the global setting",
|
||||
// and the backend validator treats empty as inherit too. Omitting the key
|
||||
// entirely would make the field look saved while silently discarding it.
|
||||
acme_backend_url: (values.acme_backend_url || '').trim() || null
|
||||
};
|
||||
|
||||
// For new clusters, explicitly set is_active to true
|
||||
@@ -767,6 +772,47 @@ const ClusterManagement = () => {
|
||||
<Switch checkedChildren="Enabled" unCheckedChildren="Disabled" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
label="ACME Challenge Backend URL"
|
||||
name="acme_backend_url"
|
||||
tooltip="Where this cluster's HAProxy nodes reach OpenManager to fetch HTTP-01 challenge tokens. Leave empty to use the global setting under Settings > ACME."
|
||||
extra="This address is resolved on the HAProxy node, not here — localhost would mean the HAProxy box itself. Include the port: without one, 8080 is assumed. Example: http://10.90.1.4:80"
|
||||
rules={[
|
||||
{
|
||||
validator: (_, value) => {
|
||||
const v = (value || '').trim();
|
||||
if (!v) return Promise.resolve();
|
||||
if (!/^https?:\/\//i.test(v)) {
|
||||
return Promise.reject(new Error(
|
||||
'Start with http:// or https:// — without a scheme the value cannot be parsed as an address.'
|
||||
));
|
||||
}
|
||||
if (/\s/.test(v)) {
|
||||
return Promise.reject(new Error('Must not contain spaces or line breaks.'));
|
||||
}
|
||||
let parsed;
|
||||
try {
|
||||
parsed = new URL(v);
|
||||
} catch (e) {
|
||||
return Promise.reject(new Error('Not a valid URL.'));
|
||||
}
|
||||
if (parsed.pathname !== '/' && parsed.pathname !== '') {
|
||||
return Promise.reject(new Error('Enter only scheme, host and port — no path.'));
|
||||
}
|
||||
const host = parsed.hostname.replace(/^\[|\]$/g, '').toLowerCase();
|
||||
if (host === 'localhost' || host === '127.0.0.1' || host === '::1' || host === '0.0.0.0') {
|
||||
return Promise.reject(new Error(
|
||||
'Loopback cannot work here: HAProxy resolves this address on the node, so it would mean the node itself. Use the management server\u2019s routable address.'
|
||||
));
|
||||
}
|
||||
return Promise.resolve();
|
||||
}
|
||||
}
|
||||
]}
|
||||
>
|
||||
<Input placeholder="http://10.90.1.4:80" />
|
||||
</Form.Item>
|
||||
|
||||
</Form>
|
||||
</Modal>
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import React, { useState, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Table, Button, Space, Modal, Form, Input, InputNumber, Select, Tag, message,
|
||||
Switch, Typography, Card, Alert, Tooltip, Spin
|
||||
Switch, Typography, Card, Alert, Tooltip, Spin, Checkbox
|
||||
} from 'antd';
|
||||
import {
|
||||
PlusOutlined, EditOutlined, DeleteOutlined, ReloadOutlined, WarningOutlined,
|
||||
@@ -46,6 +46,36 @@ const parseArr = (v) => {
|
||||
return [];
|
||||
};
|
||||
|
||||
// v1.10.4 — adoption blockers come back as prose from the parser. Two classes are resolvable by
|
||||
// the operator and the rest are not, so the modal has to tell them apart:
|
||||
// * `loss` — "our renderer cannot reproduce this, so adopting would delete it". A deliberate
|
||||
// choice, waivable with an explicit tick.
|
||||
// * `prefix` — the address has no explicit prefix length. Supplying it resolves the blocker;
|
||||
// we never guess a netmask for a live VIP.
|
||||
// * `hard` — an unknown VRID, a fractional advert_int, an unsupported auth_type. Not losses
|
||||
// but impossibilities; nothing in the UI may override them.
|
||||
const splitBlockers = (blockers) => {
|
||||
const list = blockers || [];
|
||||
return {
|
||||
loss: list.filter((b) => b.includes('would delete it')),
|
||||
prefix: list.filter((b) => b.includes('no explicit prefix length')),
|
||||
hard: list.filter((b) => !b.includes('would delete it') && !b.includes('no explicit prefix length')),
|
||||
};
|
||||
};
|
||||
|
||||
// v1.10.10 — every node of an instance reports the SAME problems about the SAME shared config,
|
||||
// so merging their blocker lists repeats each one per member. The line numbers differ between
|
||||
// the files, so exact-string dedup does not collapse them; key on the text WITHOUT the leading
|
||||
// "line N:" and keep the first occurrence. Four issues on a pair used to read as eight.
|
||||
const mergeBlockers = (lists) => {
|
||||
const seen = new Map();
|
||||
lists.flat().forEach((b) => {
|
||||
const key = String(b).replace(/^line \d+:\s*/, '');
|
||||
if (!seen.has(key)) seen.set(key, b);
|
||||
});
|
||||
return Array.from(seen.values());
|
||||
};
|
||||
|
||||
// This component uses raw fetch(), but extractApiError expects an axios-shaped error
|
||||
// (err.response.data). Read the fetch Response body and reuse the envelope-aware extractor
|
||||
// so backend messages — e.g. the 409 "node already in VIP X" — actually reach the user.
|
||||
@@ -55,7 +85,7 @@ const fetchApiError = async (res, fallback) => {
|
||||
};
|
||||
|
||||
const VIPManagement = () => {
|
||||
const { clusters } = useCluster();
|
||||
const { clusters, selectedCluster } = useCluster();
|
||||
const [vips, setVips] = useState([]);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [modalVisible, setModalVisible] = useState(false);
|
||||
@@ -67,6 +97,12 @@ const VIPManagement = () => {
|
||||
// Delete confirmation (with opt-in package uninstall) + diagnostics modal state.
|
||||
const [deleteTarget, setDeleteTarget] = useState(null);
|
||||
const [showL2Note, setShowL2Note] = useState(false);
|
||||
// v1.10.4 — VIP adoption from what the agents found on their nodes.
|
||||
const [discoveries, setDiscoveries] = useState([]);
|
||||
const [adoptTarget, setAdoptTarget] = useState(null); // { discovery, candidate }
|
||||
const [adoptAcceptLoss, setAdoptAcceptLoss] = useState(false);
|
||||
const [adopting, setAdopting] = useState(false);
|
||||
const [adoptForm] = Form.useForm();
|
||||
const [diagVip, setDiagVip] = useState(null);
|
||||
const [diagData, setDiagData] = useState(null);
|
||||
const [diagLoading, setDiagLoading] = useState(false);
|
||||
@@ -80,10 +116,15 @@ const VIPManagement = () => {
|
||||
return Array.from(seen, ([id, name]) => ({ id, name }));
|
||||
}, [clusters]);
|
||||
|
||||
// v1.10.6 — both lists follow the cluster picked in the header, like every other page. The
|
||||
// selector was always there but this page ignored it, so a fleet with several clusters saw
|
||||
// one undifferentiated list. Falls back to fleet-wide while the context is still resolving.
|
||||
const scopeQuery = selectedCluster?.id ? `?cluster_id=${selectedCluster.id}` : '';
|
||||
|
||||
const fetchVips = useCallback(async () => {
|
||||
setLoading(true);
|
||||
try {
|
||||
const res = await fetch('/api/vip', { headers: authHeaders() });
|
||||
const res = await fetch(`/api/vip${scopeQuery}`, { headers: authHeaders() });
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
setVips(data.vips || []);
|
||||
@@ -96,13 +137,161 @@ const VIPManagement = () => {
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
}, [scopeQuery]);
|
||||
|
||||
// v1.10.4 — keepalived configs the agents found on their nodes but do NOT manage. This is why
|
||||
// the page could be empty on a fleet that already runs keepalived: the flow was one-way, so
|
||||
// nothing ever read what was already there.
|
||||
const fetchDiscoveries = useCallback(async () => {
|
||||
try {
|
||||
const res = await fetch(`/api/vip/discoveries${scopeQuery}`, { headers: authHeaders() });
|
||||
if (!res.ok) { setDiscoveries([]); return; }
|
||||
const data = await res.json();
|
||||
// v1.10.8 — hide a node only while its adoption still STANDS. Filtering on adopted_vip_id
|
||||
// alone hid it forever after a reject: nothing clears that column, the VIP is only ever
|
||||
// soft-deleted, and the agent does not re-report an unchanged file.
|
||||
setDiscoveries((data.discoveries || [])
|
||||
.filter((d) => !d.is_managed && !d.adopted_vip_active));
|
||||
} catch (e) {
|
||||
console.error('fetchDiscoveries failed', e);
|
||||
}
|
||||
}, [scopeQuery]);
|
||||
|
||||
useEffect(() => {
|
||||
fetchVips();
|
||||
const t = setInterval(fetchVips, 30000); // live MASTER/BACKUP via existing detection pipeline
|
||||
fetchDiscoveries();
|
||||
const t = setInterval(() => { fetchVips(); fetchDiscoveries(); }, 30000); // live MASTER/BACKUP via existing detection pipeline
|
||||
return () => clearInterval(t);
|
||||
}, [fetchVips]);
|
||||
}, [fetchVips, fetchDiscoveries]);
|
||||
|
||||
// v1.10.8 — one row per VRRP INSTANCE, not per node. Adoption now takes the whole instance
|
||||
// (every node in the pool reporting the same VRID + address), so listing the nodes as separate
|
||||
// adoptable rows invited exactly the half-adoption the backend refuses: adopting the BACKUP
|
||||
// alone cannot be applied, and on a unicast pair adopting one side drops the peer list and
|
||||
// drops both nodes into a split brain. Identity is (VRID, address), same as keepalived's.
|
||||
const discoveryGroups = React.useMemo(() => {
|
||||
const groups = new Map();
|
||||
(discoveries || []).forEach((d) => {
|
||||
const cands = d.analysis?.candidates || [];
|
||||
if (cands.length === 0) {
|
||||
const key = `solo:${d.agent_id}`;
|
||||
groups.set(key, { key, instance_name: '—', vip: null, members: [{ discovery: d, candidate: null }] });
|
||||
return;
|
||||
}
|
||||
cands.forEach((c) => {
|
||||
const vrid = c.vip?.virtual_router_id;
|
||||
const addr = c.vip?.virtual_ip;
|
||||
const key = (vrid != null && addr) ? `${vrid}|${addr}` : `solo:${d.agent_id}:${c.instance_name}`;
|
||||
if (!groups.has(key)) {
|
||||
groups.set(key, { key, instance_name: c.instance_name, vip: c.vip, members: [] });
|
||||
}
|
||||
groups.get(key).members.push({ discovery: d, candidate: c });
|
||||
});
|
||||
});
|
||||
return Array.from(groups.values());
|
||||
}, [discoveries]);
|
||||
|
||||
// What stops a whole instance from being adopted. Mirrors the backend's checks so the button
|
||||
// state and the 422 it would return cannot drift apart.
|
||||
const groupState = (g) => {
|
||||
const parseFailed = g.members.filter((m) => m.discovery.parse_error);
|
||||
const noCandidate = g.members.filter((m) => !m.candidate);
|
||||
const blockers = mergeBlockers(g.members.map((m) => m.candidate?.blockers || []));
|
||||
const { hard, loss, prefix } = splitBlockers(blockers);
|
||||
const masters = g.members.filter((m) => m.candidate?.member?.role === 'MASTER').length;
|
||||
// Any reported config that mentions this address but is NOT one of this group's nodes would
|
||||
// be left behind when the others are taken over — unparseable, agent disabled, different
|
||||
// pool. The endpoint refuses on exactly that question, so ask it here too rather than letting
|
||||
// the operator click into a 422. This list is cluster-scoped, so a peer in another pool is
|
||||
// invisible from here; the endpoint still catches it.
|
||||
const inGroup = new Set(g.members.map((m) => m.discovery.agent_id));
|
||||
const strandedPeers = g.vip?.virtual_ip
|
||||
? discoveries.filter((d) => !inGroup.has(d.agent_id)
|
||||
&& (d.config_preview || '').includes(g.vip.virtual_ip))
|
||||
: [];
|
||||
// v1.10.11 — ONE ordered decision drives both the label and the button, because they used to
|
||||
// be computed separately and disagreed: a group blocked by an unreadable peer was tagged
|
||||
// "MASTER missing" (its peer's MASTER simply had not been counted) while the disabled button
|
||||
// gave the real reason in its own tooltip. Whatever stops adoption is what the label says.
|
||||
let reason = null;
|
||||
let label = null;
|
||||
let colour = null;
|
||||
if (parseFailed.length) {
|
||||
reason = `${parseFailed.map((m) => m.discovery.agent_name).join(', ')}: config could not be parsed`;
|
||||
label = 'unparseable'; colour = 'red';
|
||||
} else if (noCandidate.length) {
|
||||
reason = 'no vrrp_instance in the report';
|
||||
label = 'no vrrp_instance'; colour = undefined;
|
||||
} else if (strandedPeers.length) {
|
||||
const names = strandedPeers.map((d) => d.agent_name).join(', ');
|
||||
const why = strandedPeers.every((d) => d.parse_error)
|
||||
? 'their config could not be parsed'
|
||||
: 'they cannot be taken over with this instance';
|
||||
reason = `${names} also reference ${g.vip.virtual_ip} but ${why} — fix those nodes first, or `
|
||||
+ 'they would be left running an unmanaged config on the same address';
|
||||
label = 'blocked by peer'; colour = 'red';
|
||||
} else if (hard.length) {
|
||||
reason = hard.join(' · ');
|
||||
label = `${hard.length} blocker(s)`; colour = 'red';
|
||||
} else if (masters !== 1) {
|
||||
reason = masters === 0
|
||||
? 'no node in this instance declares state MASTER — enable the missing node\'s agent so it reports its config'
|
||||
: `${masters} nodes declare MASTER; exactly one must`;
|
||||
label = masters === 0 ? 'MASTER missing' : 'two MASTERs'; colour = 'red';
|
||||
} else if (blockers.length) {
|
||||
label = 'needs review'; colour = 'gold'; // resolvable in the adopt dialog
|
||||
} else {
|
||||
label = 'yes'; colour = 'green';
|
||||
}
|
||||
return { parseFailed, noCandidate, hard, loss, prefix, masters, blockers,
|
||||
strandedPeers, reason, label, colour };
|
||||
};
|
||||
|
||||
const openAdopt = (group) => {
|
||||
// Any member can carry the request: the backend resolves the whole instance from it. Prefer
|
||||
// the MASTER so the suggested name and the preview show the authoritative node.
|
||||
const primary = group.members.find((m) => m.candidate?.member?.role === 'MASTER') || group.members[0];
|
||||
setAdoptTarget({ group, primary, discovery: primary.discovery, candidate: primary.candidate });
|
||||
setAdoptAcceptLoss(false);
|
||||
adoptForm.setFieldsValue({
|
||||
name: `${group.vip?.virtual_ip || primary.discovery.agent_name}-vip`,
|
||||
prefix_length: group.vip?.prefix_length ?? undefined,
|
||||
});
|
||||
};
|
||||
|
||||
const submitAdopt = async () => {
|
||||
if (!adoptTarget) return;
|
||||
let values;
|
||||
try { values = await adoptForm.validateFields(); } catch { return; }
|
||||
setAdopting(true);
|
||||
try {
|
||||
const res = await fetch('/api/vip/adopt', {
|
||||
method: 'POST',
|
||||
headers: authHeaders(),
|
||||
body: JSON.stringify({
|
||||
agent_id: adoptTarget.discovery.agent_id,
|
||||
instance_name: adoptTarget.candidate.instance_name,
|
||||
name: values.name,
|
||||
description: values.description || undefined,
|
||||
prefix_length: values.prefix_length ?? undefined,
|
||||
accept_data_loss: adoptAcceptLoss || undefined,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
message.error(await fetchApiError(res, 'Adoption failed'), 8);
|
||||
return;
|
||||
}
|
||||
const data = await res.json();
|
||||
message.success(data.message || 'VIP adopted', 8);
|
||||
setAdoptTarget(null);
|
||||
fetchVips();
|
||||
fetchDiscoveries();
|
||||
} catch (e) {
|
||||
message.error('Adoption failed');
|
||||
} finally {
|
||||
setAdopting(false);
|
||||
}
|
||||
};
|
||||
|
||||
// Build the participating-nodes table from the pool's EXISTING agents (installed via the
|
||||
// standard Agent Management process). On edit, pre-select the VIP's current members.
|
||||
@@ -410,6 +599,200 @@ const VIPManagement = () => {
|
||||
<Table rowKey="id" columns={columns} dataSource={vips} loading={loading} pagination={{ pageSize: 10 }} />
|
||||
</Card>
|
||||
|
||||
{/* v1.10.4 — keepalived that already exists on a node. Shown separately from managed VIPs
|
||||
because OpenManager is NOT managing these: the agent found them, reported them, and
|
||||
deliberately left them untouched. */}
|
||||
{discoveries.length > 0 && (
|
||||
<Card style={{ marginTop: 16 }} title={
|
||||
<Space>
|
||||
<FileSearchOutlined />
|
||||
<span>Unmanaged keepalived detected on {discoveries.length} node(s)</span>
|
||||
</Space>
|
||||
}>
|
||||
<Alert
|
||||
type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message="These nodes already run keepalived, configured outside OpenManager"
|
||||
description={
|
||||
<span>
|
||||
The agent read each <Text code>keepalived.conf</Text> and left it untouched — nothing
|
||||
on these nodes has been changed. Adopting one creates a managed VIP from the values
|
||||
in that file, and the node's config is only handed over when you apply it from
|
||||
Apply Management. Adoption replaces the file with OpenManager's render, so anything
|
||||
it cannot reproduce is listed as a blocker rather than silently dropped.
|
||||
</span>
|
||||
}
|
||||
/>
|
||||
<Table
|
||||
rowKey={(r) => r.key}
|
||||
size="small"
|
||||
pagination={false}
|
||||
dataSource={discoveryGroups}
|
||||
columns={[
|
||||
{ title: 'Nodes', key: 'agents',
|
||||
render: (_v, r) => (
|
||||
<Space direction="vertical" size={0}>
|
||||
{r.members.map((m) => (
|
||||
<Text strong key={m.discovery.agent_id}>{m.discovery.agent_name}</Text>
|
||||
))}
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
{r.members[0]?.discovery.pool_name || 'no pool'}
|
||||
</Text>
|
||||
</Space>
|
||||
) },
|
||||
{ title: 'Instance', dataIndex: 'instance_name', key: 'instance' },
|
||||
{ title: 'Virtual IP', key: 'vip',
|
||||
render: (_v, r) => (r.vip?.virtual_ip
|
||||
? <Text code>{r.vip.virtual_ip}
|
||||
{r.vip.prefix_length != null ? `/${r.vip.prefix_length}` : ''}</Text>
|
||||
: <Text type="secondary">—</Text>) },
|
||||
{ title: 'VRID', key: 'vrid',
|
||||
render: (_v, r) => (r.vip?.virtual_router_id ?? <Text type="secondary">—</Text>) },
|
||||
{ title: 'Members', key: 'member',
|
||||
render: (_v, r) => (
|
||||
<Space direction="vertical" size={0}>
|
||||
{r.members.map((m) => (
|
||||
<Space size={4} key={m.discovery.agent_id}>
|
||||
{m.candidate ? (
|
||||
<>
|
||||
<Tag color={m.candidate.member.role === 'MASTER' ? 'green' : 'default'}>
|
||||
{m.candidate.member.role}
|
||||
</Tag>
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
prio {m.candidate.member.priority} · {m.candidate.member.network_interface}
|
||||
</Text>
|
||||
</>
|
||||
) : <Text type="secondary">—</Text>}
|
||||
</Space>
|
||||
))}
|
||||
</Space>
|
||||
) },
|
||||
{ title: 'Adoptable', key: 'adoptable',
|
||||
render: (_v, r) => {
|
||||
// Label, colour and tooltip all come from the single ordered decision in
|
||||
// groupState, so the tag can never name a different problem than the one that
|
||||
// actually disables the button.
|
||||
const st = groupState(r);
|
||||
const detail = st.parseFailed.length
|
||||
? st.parseFailed.map((m) => `${m.discovery.agent_name}: ${m.discovery.parse_error}`).join(' · ')
|
||||
: (st.reason || st.blockers.join(' · '));
|
||||
const tag = <Tag color={st.colour}>{st.label}</Tag>;
|
||||
return detail ? <Tooltip title={detail}>{tag}</Tooltip> : tag;
|
||||
} },
|
||||
{ title: 'Actions', key: 'actions',
|
||||
render: (_v, r) => {
|
||||
const st = groupState(r);
|
||||
const btn = (
|
||||
<Button size="small" type="primary" ghost
|
||||
disabled={!!st.reason} onClick={() => openAdopt(r)}>
|
||||
Adopt
|
||||
</Button>
|
||||
);
|
||||
// A disabled antd Button swallows mouse events, so the tooltip needs a live
|
||||
// wrapper or the operator never learns WHY adoption is unavailable.
|
||||
return st.reason
|
||||
? <Tooltip title={st.reason}><span style={{ display: 'inline-block' }}>{btn}</span></Tooltip>
|
||||
: btn;
|
||||
} },
|
||||
]}
|
||||
/>
|
||||
</Card>
|
||||
)}
|
||||
|
||||
{/* Adopt modal — shows what will be taken over, what was assumed, and what would be lost. */}
|
||||
<Modal
|
||||
title={adoptTarget
|
||||
? `Adopt ${adoptTarget.group.instance_name} — ${adoptTarget.group.members.length} node(s)`
|
||||
: 'Adopt VIP'}
|
||||
open={!!adoptTarget}
|
||||
onCancel={() => setAdoptTarget(null)}
|
||||
onOk={submitAdopt}
|
||||
confirmLoading={adopting}
|
||||
okText="Adopt as PENDING"
|
||||
width={720}
|
||||
okButtonProps={{
|
||||
disabled: !!adoptTarget && (() => {
|
||||
const { loss, hard } = splitBlockers(
|
||||
mergeBlockers(adoptTarget.group.members.map((m) => m.candidate?.blockers || [])));
|
||||
return hard.length > 0 || (loss.length > 0 && !adoptAcceptLoss);
|
||||
})(),
|
||||
}}
|
||||
>
|
||||
{adoptTarget && (() => {
|
||||
const cand = adoptTarget.candidate;
|
||||
// Blockers are aggregated across EVERY node of the instance, because adoption
|
||||
// overwrites every one of their files — the backend refuses on the same combined set.
|
||||
const { loss, prefix, hard } = splitBlockers(
|
||||
mergeBlockers(adoptTarget.group.members.map((m) => m.candidate?.blockers || [])));
|
||||
return (
|
||||
<>
|
||||
<Alert type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message={`These ${adoptTarget.group.members.length} node(s) will be taken over together`}
|
||||
description={
|
||||
<ul style={{ margin: 0, paddingLeft: 18 }}>
|
||||
{adoptTarget.group.members.map((m) => (
|
||||
<li key={m.discovery.agent_id}>
|
||||
<Text strong>{m.discovery.agent_name}</Text>
|
||||
{' — '}{m.candidate?.member?.role} · prio {m.candidate?.member?.priority}
|
||||
{' · '}{m.candidate?.member?.network_interface}
|
||||
{' · '}<Text code>{m.discovery.config_path}</Text>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
} />
|
||||
{hard.length > 0 && (
|
||||
<Alert type="error" showIcon style={{ marginBottom: 12 }}
|
||||
message="This config cannot be adopted"
|
||||
description={<ul style={{ margin: 0, paddingLeft: 18 }}>
|
||||
{hard.map((b, i) => <li key={i}>{b}</li>)}
|
||||
</ul>} />
|
||||
)}
|
||||
{loss.length > 0 && (
|
||||
<Alert type="warning" showIcon style={{ marginBottom: 12 }}
|
||||
message="Adopting would delete these directives from the node's config"
|
||||
description={
|
||||
<>
|
||||
<ul style={{ margin: '0 0 8px', paddingLeft: 18 }}>
|
||||
{loss.map((b, i) => <li key={i}>{b}</li>)}
|
||||
</ul>
|
||||
<Checkbox checked={adoptAcceptLoss} onChange={(e) => setAdoptAcceptLoss(e.target.checked)}>
|
||||
I understand these will be lost when the config is handed over
|
||||
</Checkbox>
|
||||
</>
|
||||
} />
|
||||
)}
|
||||
{(cand.defaulted || []).length > 0 && (
|
||||
<Alert type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message={`Assumed from keepalived's defaults (absent from the file): ${cand.defaulted.join(', ')}`} />
|
||||
)}
|
||||
<Form form={adoptForm} layout="vertical">
|
||||
<Form.Item name="name" label="VIP name"
|
||||
rules={[{ required: true, message: 'Give the managed VIP a name' }]}>
|
||||
<Input placeholder="e.g. dmz-web-vip" />
|
||||
</Form.Item>
|
||||
{prefix.length > 0 && (
|
||||
<Form.Item name="prefix_length" label="Prefix length"
|
||||
extra="The file has no explicit prefix, and guessing one would change this VIP's netmask on takeover. State it here."
|
||||
rules={[{ required: true, message: 'Required — the file does not state one' }]}>
|
||||
<InputNumber min={1} max={32} style={{ width: 160 }} />
|
||||
</Form.Item>
|
||||
)}
|
||||
<Form.Item name="description" label="Description (optional)">
|
||||
<Input placeholder={`Adopted from ${adoptTarget.discovery.agent_name}`} />
|
||||
</Form.Item>
|
||||
</Form>
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
Config found at <Text code>{adoptTarget.discovery.config_path}</Text> — the VRRP
|
||||
password is masked below and is carried over encrypted.
|
||||
</Text>
|
||||
<pre style={{ marginTop: 8, maxHeight: 220, overflow: 'auto', fontSize: 12,
|
||||
background: 'rgba(127,127,127,0.08)', padding: 8, borderRadius: 4 }}>
|
||||
{adoptTarget.discovery.config_preview || '(not available)'}
|
||||
</pre>
|
||||
</>
|
||||
);
|
||||
})()}
|
||||
</Modal>
|
||||
|
||||
<Modal
|
||||
title={editing ? `Edit VIP — ${editing.name}` : 'Create VIP'}
|
||||
open={modalVisible}
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* v1.10.3 regression tests: with more than one ACME account, the certificate wizard must honour the
|
||||
* account the operator picked — on the Review step AND in the request it submits.
|
||||
*
|
||||
* These drive the real component through all three wizard steps rather than testing a helper,
|
||||
* because the bug was invisible until the wizard ADVANCED PAST the step that owns the account
|
||||
* Select: Form.useWatch reports only fields that are currently rendered, so on Review the selection
|
||||
* read `undefined` and the wizard silently reverted to the default account. A unit test of any
|
||||
* single function would have passed the whole time.
|
||||
*/
|
||||
import React from 'react';
|
||||
import { render, screen, fireEvent, waitFor, act } from '@testing-library/react';
|
||||
import axios from 'axios';
|
||||
import ACMEAutomation from '../ACMEAutomation';
|
||||
|
||||
// These render the whole ACMEAutomation tree (antd Steps + Form + Select) three times per test and
|
||||
// take 4-7s each, so jest's default 5s per-test limit fails two of them. Raise it here rather than
|
||||
// relying on the runner being invoked with --testTimeout, so `npm test` passes as shipped.
|
||||
jest.setTimeout(30000);
|
||||
|
||||
jest.mock('axios');
|
||||
jest.mock('react-router-dom', () => ({ useNavigate: () => jest.fn() }));
|
||||
jest.mock('../../contexts/ClusterContext', () => ({
|
||||
useCluster: () => ({ clusters: [], selectCluster: jest.fn() }),
|
||||
}));
|
||||
|
||||
// The account list arrives ORDER BY id while the backend's default is ORDER BY created_at DESC, so
|
||||
// this fixture is deliberately the shape that made the two disagree: the DNS-01 account is BOTH the
|
||||
// lower id and the older account, the HTTP-01 account is the newest. Picking the first valid entry
|
||||
// (as the UI used to) yields the DNS-01 account; the backend would have used the HTTP-01 one.
|
||||
const DNS_ACCOUNT = {
|
||||
id: 1, email: 'dns@example.com', status: 'valid',
|
||||
challenge_type: 'dns-01', dns_provider: 'godaddy',
|
||||
created_at: '2026-05-06T00:00:00Z', directory_url: 'https://acme.zerossl.com/v2/DV90',
|
||||
};
|
||||
const HTTP_ACCOUNT = {
|
||||
id: 2, email: 'http@example.com', status: 'valid',
|
||||
challenge_type: 'http-01', dns_provider: null,
|
||||
created_at: '2026-08-08T00:00:00Z', directory_url: 'https://acme.zerossl.com/v2/DV90',
|
||||
};
|
||||
|
||||
const GET_ROUTES = {
|
||||
'/api/letsencrypt/orders': [],
|
||||
'/api/letsencrypt/accounts': [DNS_ACCOUNT, HTTP_ACCOUNT],
|
||||
'/api/letsencrypt/renewal-schedule': [],
|
||||
'/api/clusters': { clusters: [{ id: 10, name: 'cluster-a', acme_enabled: true, is_active: true }] },
|
||||
'/api/letsencrypt/prerequisites': { steps: [] },
|
||||
'/api/letsencrypt/dns-providers': {
|
||||
dns01_enabled: true,
|
||||
providers: [
|
||||
{ name: 'manual', label: 'Manual', automated: false, credential_fields: [] },
|
||||
{
|
||||
name: 'godaddy', label: 'GoDaddy', automated: true,
|
||||
credential_fields: [
|
||||
{ key: 'api_key', label: 'API Key', type: 'password', required: true, max_length: 200, help: 'Production key' },
|
||||
{ key: 'api_secret', label: 'API Secret', type: 'password', required: false, max_length: 200, help: 'Blank for a PAT' },
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
jest.clearAllMocks();
|
||||
axios.get.mockImplementation((url) =>
|
||||
Promise.resolve({ data: Object.prototype.hasOwnProperty.call(GET_ROUTES, url) ? GET_ROUTES[url] : {} })
|
||||
);
|
||||
axios.post.mockResolvedValue({ data: { message: 'ok', order_id: 99 } });
|
||||
});
|
||||
|
||||
const modal = () => document.querySelector('.ant-modal-content');
|
||||
|
||||
/** Open the wizard and wait for the Domains step. */
|
||||
async function openWizard() {
|
||||
render(<ACMEAutomation />);
|
||||
// Settle the initial fetch on a signal that does NOT depend on which account the component picks
|
||||
// as its default — that choice is one of the things under test, so waiting on an account address
|
||||
// here would make every test fail at the same early point instead of at its own assertion.
|
||||
await waitFor(() => expect(axios.get).toHaveBeenCalledWith('/api/letsencrypt/accounts'));
|
||||
await act(async () => {});
|
||||
fireEvent.click(screen.getByRole('button', { name: /Request Certificate/i }));
|
||||
await screen.findByText('Domain Names');
|
||||
}
|
||||
|
||||
/** antd wires the Form.Item name onto the inner input's id, which is the only stable handle. */
|
||||
function typeDomain(domain) {
|
||||
const input = document.getElementById('domains');
|
||||
fireEvent.change(input, { target: { value: domain } });
|
||||
fireEvent.keyDown(input, { key: 'Enter', keyCode: 13 });
|
||||
}
|
||||
|
||||
async function selectAccount(email) {
|
||||
await act(async () => {
|
||||
fireEvent.mouseDown(document.getElementById('account_id'));
|
||||
});
|
||||
const option = [...document.querySelectorAll('.ant-select-item-option')]
|
||||
.find((o) => o.textContent.includes(email));
|
||||
if (!option) throw new Error(`account option not found: ${email}`);
|
||||
await act(async () => {
|
||||
fireEvent.click(option);
|
||||
});
|
||||
}
|
||||
|
||||
const next = async () => {
|
||||
await act(async () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Next$/ }));
|
||||
});
|
||||
};
|
||||
const submitButton = () => screen.getByRole('button', { name: /Submit Request/i });
|
||||
|
||||
async function gotoReview({ domain = 'example.com', account } = {}) {
|
||||
await openWizard();
|
||||
typeDomain(domain);
|
||||
await next();
|
||||
// Wait for the Configuration step to actually MOUNT. The transition is async (validateFields
|
||||
// returns a promise) and the text "ACME Account" also appears on the dashboard card behind the
|
||||
// modal, so waiting on that text can resolve while the wizard is still on step 1.
|
||||
await waitFor(() => expect(document.getElementById('account_id')).toBeTruthy());
|
||||
if (account) await selectAccount(account);
|
||||
await next();
|
||||
await screen.findByText('Prerequisite Check');
|
||||
}
|
||||
|
||||
/** The Review step's rendered text. Compared as a whole string on purpose: the assertions must
|
||||
* describe BEHAVIOUR, not the markup this change happens to use, so that a failure means the
|
||||
* wizard resolved the wrong account rather than that a wrapper element moved. */
|
||||
const reviewText = () => modal().textContent;
|
||||
|
||||
async function submitAndGetBody() {
|
||||
fireEvent.click(submitButton());
|
||||
await waitFor(() => expect(axios.post).toHaveBeenCalled());
|
||||
const [url, body] = axios.post.mock.calls[0];
|
||||
expect(url).toBe('/api/letsencrypt/certificates');
|
||||
return body;
|
||||
}
|
||||
|
||||
describe('certificate wizard with multiple ACME accounts', () => {
|
||||
test('an explicitly picked HTTP-01 account survives the step change and is what gets submitted', async () => {
|
||||
await gotoReview({ account: HTTP_ACCOUNT.email });
|
||||
|
||||
// The payload is the real evidence. account_id used to come from the form store while
|
||||
// challenge_type came from an account object that had reverted to the default, so the API got
|
||||
// "HTTP-01 account + dns-01" and answered 422 "no DNS provider configured for DNS-01".
|
||||
const body = await submitAndGetBody();
|
||||
expect(body.account_id).toBe(HTTP_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('http-01');
|
||||
});
|
||||
|
||||
test('the Review step describes the picked HTTP-01 account, not the default', async () => {
|
||||
await gotoReview({ account: HTTP_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain(HTTP_ACCOUNT.email);
|
||||
expect(reviewText()).not.toContain(DNS_ACCOUNT.email);
|
||||
// "Challenge Method" and the provider name only render on the DNS-01 branch.
|
||||
expect(reviewText()).not.toContain('Challenge Method');
|
||||
expect(reviewText()).not.toContain('godaddy');
|
||||
});
|
||||
|
||||
test('an explicitly picked DNS-01 account is described and submitted as DNS-01', async () => {
|
||||
// Positive control: the DNS-01 path must keep working. This one passed before the fix too,
|
||||
// because the default the wizard fell back to happened to be the DNS-01 account.
|
||||
await gotoReview({ account: DNS_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain(DNS_ACCOUNT.email);
|
||||
expect(reviewText()).toContain('Challenge Method');
|
||||
expect(reviewText()).toContain('godaddy');
|
||||
|
||||
const body = await submitAndGetBody();
|
||||
expect(body.account_id).toBe(DNS_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('dns-01');
|
||||
});
|
||||
|
||||
test('with no explicit pick the wizard previews and sends the same default the backend would use', async () => {
|
||||
// The backend takes the NEWEST valid account; the UI used to preview the oldest entry of a
|
||||
// list ordered by id, so Review described an account the request never went to.
|
||||
await gotoReview();
|
||||
|
||||
expect(reviewText()).toContain(HTTP_ACCOUNT.email);
|
||||
expect(reviewText()).not.toContain(DNS_ACCOUNT.email);
|
||||
|
||||
const body = await submitAndGetBody();
|
||||
// Sent explicitly rather than left to the backend to guess a second time.
|
||||
expect(body.account_id).toBe(HTTP_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('http-01');
|
||||
});
|
||||
|
||||
test('the wildcard guard still applies on the Review step, where Submit lives', async () => {
|
||||
// Same root cause as the account bug: `domains` is entered on the first step, so a
|
||||
// non-preserving useWatch read undefined from Review onward and the guard evaporated at exactly
|
||||
// the point it had to hold.
|
||||
await gotoReview({ domain: '*.example.com', account: HTTP_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain('Wildcard requires a DNS-01 account');
|
||||
expect(submitButton()).toBeDisabled();
|
||||
fireEvent.click(submitButton());
|
||||
expect(axios.post).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,26 @@
|
||||
// Loaded automatically by react-scripts test (CRA convention).
|
||||
import '@testing-library/jest-dom';
|
||||
|
||||
// jsdom implements neither of these, and Ant Design 5 needs both: rc-select renders its dropdown
|
||||
// through rc-virtual-list (ResizeObserver) and the responsive Grid reads matchMedia. Without the
|
||||
// polyfills any test that opens a Select throws before it can assert anything.
|
||||
if (!global.ResizeObserver) {
|
||||
global.ResizeObserver = class {
|
||||
observe() {}
|
||||
unobserve() {}
|
||||
disconnect() {}
|
||||
};
|
||||
}
|
||||
|
||||
if (!window.matchMedia) {
|
||||
window.matchMedia = (query) => ({
|
||||
matches: false,
|
||||
media: query,
|
||||
onchange: null,
|
||||
addListener: () => {},
|
||||
removeListener: () => {},
|
||||
addEventListener: () => {},
|
||||
removeEventListener: () => {},
|
||||
dispatchEvent: () => false,
|
||||
});
|
||||
}
|
||||
Reference in New Issue
Block a user