mirror of
https://github.com/taylanbakircioglu/haproxy-openmanager.git
synced 2026-10-02 15:08:13 +00:00
Compare commits
45 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0ee227363e | |||
| 9d7a142cfa | |||
| 5f995d9d58 | |||
| 5f5c7f1c75 | |||
| fbe223250b | |||
| c5fbbd753f | |||
| bd4a50943f | |||
| bec0613ae5 | |||
| 4e2d936c27 | |||
| 82e6fe3f9c | |||
| 190d45fe09 | |||
| c50408026b | |||
| 4c84596215 | |||
| 1d4e4286af | |||
| 0eb587dfa8 | |||
| 9e64002f1c | |||
| 4f24d5bdd9 | |||
| a87994e06a | |||
| 7d95c737f0 | |||
| eda7f36c93 | |||
| 11e5bf57d9 | |||
| a4c74f2a52 | |||
| a36dd87a74 | |||
| 2f125da043 | |||
| 9e6e4dd03b | |||
| 47cc79dcf7 | |||
| bb774141d4 | |||
| acfd32dd63 | |||
| ef26860df9 | |||
| 78af849fdc | |||
| 709817fec3 | |||
| 822c441d34 | |||
| 1ca811e211 | |||
| 164841219a | |||
| d92a7e9660 | |||
| 7dfd31832a | |||
| 8ac567dfe0 | |||
| 02667bbda4 | |||
| c97df53da8 | |||
| be01ddd119 | |||
| bbd8359f50 | |||
| dd7b7822bf | |||
| c4139bb11a | |||
| dbb9189f16 | |||
| eee0a4716a |
+71
-3
@@ -17,11 +17,34 @@ REDIS_URL=redis://redis:6379
|
||||
# Change this to a strong random string in production
|
||||
SECRET_KEY=your-secret-key-change-this-in-production
|
||||
|
||||
# Optional: dedicated Fernet key for encrypting VRRP secrets of HA/VIP (Issue #27).
|
||||
# If unset, it is derived from SECRET_KEY (HKDF), exactly like MFA. Set an explicit
|
||||
# key (urlsafe-base64, 32 bytes) in production if you want independent key rotation.
|
||||
# ----------------------------------------------------------------------------
|
||||
# Optional per-purpose encryption keys.
|
||||
#
|
||||
# Every secret the application stores is encrypted at rest with Fernet. Each class
|
||||
# derives its own key, so rotating one never affects another. If a variable below is
|
||||
# unset, that class's key is derived from SECRET_KEY via HKDF — which works, but means
|
||||
# rotating SECRET_KEY makes the existing values of that class UNDECRYPTABLE. Set an
|
||||
# explicit key (urlsafe-base64, 32 bytes) in production if you want independent
|
||||
# rotation. Generate one with:
|
||||
# python -c "from cryptography.fernet import Fernet; print(Fernet.generate_key().decode())"
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
# VRRP secrets for HA/VIP (Issue #27).
|
||||
# VIP_ENCRYPTION_KEY=
|
||||
|
||||
# TOTP secrets for multi-factor authentication (Issue #18).
|
||||
# MFA_ENCRYPTION_KEY=
|
||||
|
||||
# Per-account DNS provider credentials for ACME DNS-01 (Issue #35).
|
||||
# Rotating this without re-entering credentials makes DNS-01 renewals fail until the
|
||||
# affected accounts' credentials are re-saved in ACME Automation.
|
||||
# DNS_PROVIDER_ENCRYPTION_KEY=
|
||||
|
||||
# Private keys of PENDING CSRs, held only until the signed certificate is imported
|
||||
# (Issue #53). Rotating this while CSRs are out for signature makes those CSRs
|
||||
# unusable — they must be deleted and re-created.
|
||||
# CSR_ENCRYPTION_KEY=
|
||||
|
||||
# ============================================================================
|
||||
# PUBLIC URL CONFIGURATION
|
||||
# ============================================================================
|
||||
@@ -36,6 +59,13 @@ PUBLIC_URL=http://localhost:8000
|
||||
|
||||
# Management base URL (defaults to PUBLIC_URL if not set)
|
||||
# Override this if your management interface is on a different URL
|
||||
#
|
||||
# This is also the last-resort fallback for the ACME HTTP-01 challenge backend,
|
||||
# i.e. the address written into haproxy.cfg as `server _acme_mgmt <host>:<port>`.
|
||||
# That address is resolved BY HAPROXY, ON THE HAPROXY NODE. If HAProxy runs
|
||||
# anywhere other than this machine, localhost points at the wrong box and HTTP-01
|
||||
# validation fails while DNS-01 keeps working. Set a routable address, with the
|
||||
# port, e.g. MANAGEMENT_BASE_URL=http://10.90.1.4:8080
|
||||
MANAGEMENT_BASE_URL=http://localhost:8000
|
||||
|
||||
# ============================================================================
|
||||
@@ -76,6 +106,44 @@ UVICORN_WORKERS=1
|
||||
# Example: http://haproxy-manager.example.com,http://localhost:8080
|
||||
CORS_ORIGINS=
|
||||
|
||||
# ============================================================================
|
||||
# REQUEST / RESPONSE LOG (v1.11.0)
|
||||
# ============================================================================
|
||||
# Records every inbound API call and every outbound HTTP call the backend makes
|
||||
# (ACME, DNS providers, agents) into the `request_logs` table, browsable under
|
||||
# "Request Log" in the UI.
|
||||
#
|
||||
# Only the four settings below are environment-level, because they decide
|
||||
# whether the middleware is registered at all and how much memory the writer
|
||||
# queue may hold. Everything an operator tunes day to day — retention windows,
|
||||
# body capture, sampling, excluded paths — lives in the database and is edited
|
||||
# in Settings -> Request Log.
|
||||
|
||||
# Hard kill-switch. When false the logging middleware is NEVER added to the ASGI
|
||||
# stack and neither the writer nor the retention task starts: zero overhead, not
|
||||
# even a settings lookup. Requires a restart to change.
|
||||
# (The `enabled` toggle in Settings is the no-restart equivalent.)
|
||||
REQUEST_LOG_ENABLED=true
|
||||
|
||||
# Per-worker in-process queue depth. When it fills, rows are DROPPED and counted
|
||||
# rather than blocking the request — the drop count is shown on the Request Log
|
||||
# page (per worker).
|
||||
REQUEST_LOG_QUEUE_MAX=2000
|
||||
|
||||
# Hard memory ceiling for that same queue, per worker. The row count above does
|
||||
# NOT bound memory on its own: `requestlog.max_body_bytes` is editable from
|
||||
# Settings up to 256 KB and a row can carry it twice, so at that ceiling a
|
||||
# 2000-row queue would hold ~1 GiB. Whichever limit is reached first stops the
|
||||
# queue. Raise this if you raise REQUEST_LOG_QUEUE_MAX.
|
||||
REQUEST_LOG_QUEUE_MAX_BYTES=67108864
|
||||
|
||||
# Rows per batched INSERT. One connection is taken from the pool per batch, not
|
||||
# per request.
|
||||
REQUEST_LOG_BATCH_SIZE=100
|
||||
|
||||
# Maximum wait before a partial batch is flushed, in milliseconds.
|
||||
REQUEST_LOG_FLUSH_MS=500
|
||||
|
||||
# ============================================================================
|
||||
# FRONTEND CONFIGURATION (React)
|
||||
# ============================================================================
|
||||
|
||||
@@ -131,6 +131,66 @@ if (window.location) {
|
||||
}
|
||||
```
|
||||
|
||||
### REQUEST_LOG_ENABLED (v1.11.0)
|
||||
|
||||
**Ne İşe Yarar**: Request/Response Log özelliğinin sert (hard) kill-switch'i. `false` yapıldığında
|
||||
loglama middleware'i ASGI zincirine **hiç eklenmez**, yazıcı ve retention görevleri başlatılmaz —
|
||||
yani sıfır ek yük, ayar okuması bile yapılmaz. Değişiklik için restart gerekir.
|
||||
|
||||
**Örnekler**:
|
||||
```bash
|
||||
# Varsayılan: açık
|
||||
REQUEST_LOG_ENABLED=true
|
||||
|
||||
# Tamamen kapat (ör. çok yüksek trafikli kurulum, veya regülasyon gereği)
|
||||
REQUEST_LOG_ENABLED=false
|
||||
```
|
||||
|
||||
**Nasıl Kullanılır**:
|
||||
1. Restart gerektirmeden kapatmak isterseniz bunun yerine **Settings → Request Log → Enable request
|
||||
log** anahtarını kullanın; o anında etkili olur.
|
||||
2. Retention süreleri, gövde (body) yakalama, örnekleme oranı ve hariç tutulan path'ler bu env
|
||||
değişkeniyle değil, veritabanındaki `requestlog.*` ayarlarıyla yönetilir — arayüzden düzenlenir.
|
||||
3. Disk büyümesi asıl operasyonel konudur: sırasıyla `sample_rate`'i düşürün, `capture_get`'i
|
||||
kapatın, `capture_bodies`'i kapatın, sonra `success_retention_days`'i kısaltın.
|
||||
|
||||
### REQUEST_LOG_QUEUE_MAX / REQUEST_LOG_QUEUE_MAX_BYTES / REQUEST_LOG_BATCH_SIZE / REQUEST_LOG_FLUSH_MS (v1.11.0)
|
||||
|
||||
**Ne İşe Yarar**: Log satırlarını yazan arka plan görevinin ayarları. Satırlar sınırlı bir kuyruğa
|
||||
konur ve toplu (batch) INSERT ile yazılır; böylece istek yolu asla veritabanını beklemez.
|
||||
|
||||
**Örnekler**:
|
||||
```bash
|
||||
# Worker başına kuyruk derinliği. Dolduğunda satırlar DÜŞÜRÜLÜR (sayılır ve
|
||||
# Request Log sayfasında gösterilir), istek bloklanmaz.
|
||||
REQUEST_LOG_QUEUE_MAX=2000
|
||||
|
||||
# Aynı kuyruğun BAYT tavanı (worker başına). Satır sayısı tek başına belleği
|
||||
# sınırlamaz: `max_body_bytes` Settings'ten 256 KB'a kadar ayarlanabilir ve bir
|
||||
# satır bunu iki kez taşıyabilir, o tavanda 2000 satırlık kuyruk ~1 GiB tutar.
|
||||
# Hangi sınır önce dolarsa kuyruk orada durur.
|
||||
REQUEST_LOG_QUEUE_MAX_BYTES=67108864
|
||||
|
||||
# Tek INSERT'te kaç satır yazılacağı (havuzdan istek başına değil, batch başına
|
||||
# bir bağlantı alınır)
|
||||
REQUEST_LOG_BATCH_SIZE=100
|
||||
|
||||
# Yarım dolu bir batch'in en fazla ne kadar bekletileceği (ms)
|
||||
REQUEST_LOG_FLUSH_MS=500
|
||||
```
|
||||
|
||||
**Nasıl Kullanılır**:
|
||||
1. Request Log sayfasında "rows dropped" uyarısı görüyorsanız önce `requestlog.max_body_bytes`
|
||||
veya `sample_rate`'i düşürün. `REQUEST_LOG_QUEUE_MAX`'ı artırmak bu worker'ın tutabileceği
|
||||
belleği de artırır; artıracaksanız `REQUEST_LOG_QUEUE_MAX_BYTES`'ı da birlikte artırın.
|
||||
2. Bu değerler worker başınadır — `UVICORN_WORKERS` arttıkça toplam bellek de o oranda artar.
|
||||
`GET /api/request-logs/stats` içindeki sayaçlar da worker başınadır ve yanıtta öyle
|
||||
etiketlenir; 4 worker'da gördüğünüz düşüş sayısı gerçeğin dörtte biridir.
|
||||
3. Büyük filolarda tek en etkili ayar `requestlog.capture_agent_success`'tir (varsayılan kapalı).
|
||||
Açık olsaydı 200 düğümlük bir filo günde ~2M satır yazar ve 500.000 satır tavanına 6 saatte
|
||||
ulaşırdı; yapılandırılmış 7 gün / 30 gün saklama o noktada birkaç saate iner. Başarısız ajan
|
||||
çağrıları bu ayardan bağımsız olarak her zaman loglanır.
|
||||
|
||||
## 🚀 Deployment Senaryoları
|
||||
|
||||
### Docker Compose
|
||||
|
||||
@@ -116,6 +116,7 @@ This architecture provides better security (no inbound connections to HAProxy se
|
||||
✅ **HA / VIP (Keepalived) Management** - Create virtual IPs from the UI; the agent installs & configures Keepalived (unicast VRRP) with a HAProxy health-check so the VIP fails over automatically; live MASTER/BACKUP detection per node
|
||||
✅ **Role-Based User Management** - Admin and user roles with granular permissions and access control
|
||||
✅ **User Activity Audit Logs** - Complete audit trail of all system events
|
||||
✅ **Request/Response Log** *(v1.11.0)* - Every inbound API call (GETs and errors included) and every outbound HTTP call the backend makes (ACME, Cloudflare, GoDaddy, agents) in one filterable timeline, with redacted, size-capped bodies and operator-configurable retention
|
||||
✅ **REST API** - Full programmatic access for automation and CI/CD integration
|
||||
|
||||
|
||||
@@ -2474,6 +2475,22 @@ Developed with ❤️ for the HAProxy community
|
||||
|
||||
## Release Notes
|
||||
|
||||
- **v1.11.1** (2026-08-15) — **A node can no longer be hidden from adoption for good, and Config Import reaches a freshly installed agent**: the agent posts an unmanaged `keepalived.conf` for adoption and caches the hash of what it sent, so the file (which carries the VRRP password) is re-posted only when it changes. Delivery was judged by `curl`'s exit code, which is **0 for 5xx as well**, so a report the server *rejected* was recorded as delivered — and because a hand-maintained config does not change on its own, that node dropped out of *Unmanaged keepalived detected* permanently, curable only by deleting a cache file on the node by hand. Now: the report is cached only on a 2xx; `GET /agents/{name}/keepalived-config` tells the agent whether the server actually holds a discovery for it, so nodes stuck from earlier releases recover by themselves on the next poll; a **400/413/422** records the refusal so identical bytes are not re-posted forever (4xx and 5xx are never sampled out of the request log, so an unattended loop would write a row carrying the whole config every cycle), while **401 and 404 keep retrying** because in this system they mean a token rotation or an agent row briefly absent, not a bad payload. The same exit-code mistake in the *clear* path is fixed too, where it left a managed node still being offered for adoption. Separately, **Config Import silently did nothing on any agent that had never self-upgraded**: `check_config_requests` was defined in the installer and in the self-upgrade daemon but not in the body a fresh install writes, and its call site is guarded by `type`, so the operator asked a node for its `haproxy.cfg` and nothing arrived, with no error anywhere. And the cluster whose `keepalived_config_path` is handed to an agent is now resolved deterministically — a pool may hold several clusters, and the unordered join could return a different one between polls, pointing the agent at a file that does not exist. Agent-script change: sync the script from Agent Management and let the agents upgrade. No schema change.
|
||||
- **v1.11.0** (2026-08-14) — **Unified request/response log with configurable retention**: until now the only record of what happened was `user_activity_logs`, which stores non-GET **2xx** operations with no bodies — so when something failed you could see *that* the count went up, never *what was sent or what came back*. This release adds one queryable timeline covering **both directions**: every inbound API call (**including GETs and including 4xx/5xx**) with the user, client IP, status, duration and — redacted and size-capped — the request and response bodies; and every **outbound** HTTP call the backend makes, tagged with who it went to (ACME/Let's Encrypt, Cloudflare, GoDaddy, HAProxy stats, agents, the ACME diagnostics probe). Outbound rows **inherit the inbound request's id**, so one operator action and the CA/DNS calls it triggered read as a single trace — opening a failed *Request Certificate* shows the exact `POST /acme/new-order` and the CA's `429` body underneath it. Capture is a **pure-ASGI middleware that tees** the request and response streams rather than draining them, so no downstream handler is affected (notably the raw-body agent heartbeat), and rows are written by a **batching background writer** with a bounded queue, so the request path never waits on the database and a saturated logger drops rows visibly instead of blocking. Secrets never land: headers are an allowlist (`Authorization`/`Cookie` reduced to a presence marker), body keys and value shapes are redacted (passwords, tokens, API keys, private-key PEMs, JWTs), the **ACME JWS request body is never stored** (a stored `protected`+`signature` pair is a replayable credential — a summary is logged instead), DNS-provider errors record only the exception **type**, and the ACME HTTP-01 challenge endpoint is excluded so `key_authorization` is never captured. **Retention is operator-configurable** in *Settings → Request Log*: separate day counts for successful and failed rows (defaults 7 and 30) plus a hard row cap (500 000), whichever is reached first, pruned in **batches** under a Postgres advisory lock so a multi-million-row table cannot time out the delete or have every replica scan it at once. New **Request Log** page (`requestlog.read`) and retention/purge permission (`requestlog.manage`); `super_admin` and `security_admin` get both, `operator` gets read, `viewer` gets neither. **Successful agent polls are not logged** (`capture_agent_success`, default off; failures always are), which is what keeps the table's size a function of operator activity rather than of node count: measured at 2 424 bytes/row, a 200-node fleet would otherwise write 2.0M rows/day and reach the row cap in six hours, silently reducing the configured 7-day/30-day retention to a few hours for everything in the table. Cost is measured, not estimated: 27.7 µs per request on the hot path, 18.8 µs per row on the writer task, **0.096 % of one core at 500 nodes**. Adds one new table (`request_logs`) and its settings seed — SCHEMA_VERSION 11 → 12 (not 11: that number was taken by v1.10.4 while this was in review, and the version gate would have skipped the migration entirely on every existing install), auto-migrated, no existing table altered, no agent or rendered-config change. Kill switches: `REQUEST_LOG_ENABLED=false` (environment — the middleware is then never registered and costs nothing) or the `enabled` toggle in Settings (no restart).
|
||||
- **v1.10.14** (2026-08-14) — **A converged node keeps acknowledging**: the deploy report is the server's only evidence that a member node applied its `keepalived.conf`, and it was sent on the write path alone. Once the rendered config was on disk the agent took the idempotency early return on every cycle and never reported again, so a **single lost report** — a backend restart, a 5xx, a network blip — left the VIP reading `SYNCING (0/n)` with an empty *Last ack* forever, while the node was demonstrably running the right config. Nothing would ever reconcile the two: the node was correct, the page was not, and the only way out was to change the rendered config so the agent wrote it again. The agent now re-asserts its state on the idempotent path too, which costs one request per node per ~2.5 minutes and touches nothing on the node — keepalived is not reloaded and the file is not rewritten. This is a long-standing gap from the original HA/VIP work, surfaced when acknowledgements were dropped for an unrelated reason in v1.10.12. Agent-script change: sync the script from Agent Management and let the agents upgrade. No schema or API change.
|
||||
- **v1.10.13** (2026-08-14) — **Agent deploy acknowledgements were silently dropped** (regression in v1.10.12, fix it before or with that release): the takeover-retirement clause added to `POST /agents/{name}/keepalived-status` in v1.10.12 reused one query placeholder for both the assignment `last_deploy_hash=$n` and the comparison inside its `CASE`. PostgreSQL deduces a type per **use**, so the same placeholder came out as `text` in one and `character varying` in the other, and asyncpg rejected the statement with `AmbiguousParameterError`. The failure was not partial: the whole UPDATE never ran, so **no member ever recorded an acknowledgement**. Every VIP sat at `SYNCING (0/n)` with an empty *Last ack*, even after the nodes had deployed the config successfully, and teardown acknowledgements were lost the same way. The hash is now bound to its own placeholder, which is only ever compared against the column and therefore unambiguous. Verified against a real PostgreSQL: both statements execute, a matching hash retires the takeover authorisation, a non-matching hash and a NULL `applied_config_hash` both leave it in place, and every case records the acknowledgement. A test now asserts every `$n` in these statements is bound exactly once and that the count matches the arguments passed. Backend only: no schema, agent or API-shape change.
|
||||
- **v1.10.12** (2026-08-14) — **A valid keepalived config is no longer rejected by its own warning**: before writing a rendered `keepalived.conf` the agent validates it with `keepalived -t` and, on failure, keeps the running config and does not restart keepalived. That fail-safe is right, but it treated **any** non-zero exit as invalid, and keepalived's config-test exit code does not separate fatal from benign. Measured on 2.2.8: a clean config exits 0, but `Truncating auth_pass to 8 characters` exits **5** and so does a missing `}` or an `Unknown keyword`. A VRRP password longer than eight characters was therefore enough to make every apply fail, including on nodes whose own running config produces the same warning and has been serving the VIP for weeks. The gate now judges the **output**: messages known to be benign are dropped and anything that remains still fails, so it fails **closed** and an unrecognised message is treated as fatal. Verified against real keepalived: a truncation warning passes while a missing brace, an unknown keyword and a `SECURITY VIOLATION` are all still refused. The agent also **reports what keepalived said** now, in the log and in the status the HA/VIP page shows; discarding it left a correct refusal with no way to act on it. Agent-script change: sync the script from Agent Management and let the agents upgrade for it to take effect. No schema or API change.
|
||||
- **v1.10.11** (2026-08-14) — **The *Adoptable* tag names the problem that actually blocks adoption**: the tag and the disabled *Adopt* button were computed separately and could disagree. A pair blocked because its peer's `keepalived.conf` could not be parsed was labelled **MASTER missing** — technically true, since the unreadable node's `state MASTER` had not been counted, but it pointed the operator at the wrong node while the real reason sat in the button's own tooltip. Both now come from one ordered decision, so the label, its colour and the tooltip always describe the condition that stops the adoption; a group held up by an unreadable or unreachable peer reads **blocked by peer**, and two MASTERs is now distinct from none. Display only: what the endpoint accepts or refuses is unchanged. On the public repo this is the first artifact carrying v1.10.4 through v1.10.10: none was released separately, because VIP adoption did not work end to end until these fixes landed.
|
||||
- **v1.10.10** (2026-08-14) — **Adoption blockers are listed once per instance**: with the instance-based panel a two-node pair reported the *same* problems about the *same* shared config twice, once per member, and the line numbers differ between the two files so plain de-duplication did not collapse them. Four issues on a pair read as eight, in both the *Adoptable* tooltip and the adopt dialog. They are now merged on the message text with the leading `line N:` ignored, so each distinct problem appears once. Display only: the endpoint already evaluated the combined set and its refusals are unchanged.
|
||||
- **v1.10.9** (2026-08-14) — **Adoption refuses to strand a node or silently normalise a peer's settings**: v1.10.8 adopted the whole VRRP instance, but it could only *match* a node it was able to read, that was enabled, and that sat in the same pool. Each of those was a door a real member of the group left through silently — the nodes that remained were rewritten while the one that left kept serving the same address from an unmanaged config. Seen on a live pool: one node of a pair had an unclosed `vrrp_instance` block, so it parsed to nothing while its partner parsed cleanly. Instead of guarding each door, adoption now asks the question directly — is there **any** reported `keepalived.conf` that mentions this virtual address and is not among the nodes being taken over — and refuses naming the node and the reason (unparseable, agent disabled, different pool). Separately, four VIP-level fields (`prefix_length`, unicast/multicast mode, HAProxy tracking and the VRRP password) are stored once and re-rendered onto **every** member, so taking them from whichever node was clicked imposed its settings on the others; the prefix length is the sharpest, because the design refuses to *guess* a netmask for a live VIP and copying one node's netmask onto another is that same change by another name. Adoption now requires the nodes to agree on all four, and compares the VRRP secret by decrypting each node's token (Fernet is non-deterministic, so the ciphertexts cannot be compared). Finally, the takeover authorisation is genuinely one-shot: `takeover_expected_hash` was written at adoption and never cleared, so it stayed valid for that file content indefinitely — it is now retired the moment a member acknowledges our rendered config, gated on the acked hash matching so a failed deploy never drops it. No schema change, no agent change.
|
||||
- **v1.10.8** (2026-08-13) — **VIP adoption takes the whole VRRP instance**: adoption used to take only the node whose row was clicked, which broke the exact case the feature exists for, a running HA pair. Adopting the **BACKUP** alone produced a VIP that could never be applied (*exactly one member must be MASTER*); adopting the **MASTER** alone left the peer unmanaged, and adopting it afterwards hit the VRID-collision guard with 409, so the pair could not be completed from the panel at all. Most serious, on a **unicast** instance the single-member render dropped the unicast block entirely — the renderer emits it only when it has peer addresses — so keepalived fell back to **multicast** on the adopted node while its peer stayed unicast: they stop seeing each other and **both** claim the VIP. Adoption now resolves the whole instance, keyed on `(virtual_router_id, virtual address)` exactly as keepalived groups nodes, and every participating node becomes a member with the role, priority and interface its own file declares and its own one-shot takeover hash. It refuses, with the reason, when the group does not have exactly one MASTER, when the nodes disagree on `advert_int`, when a declared unicast peer is not among the nodes being adopted, or when a node already belongs to a live VIP — a rule create/edit enforced and adoption did not. The panel now lists one row per **instance** instead of per node. Two further fixes: the Apply Management **View Change** diff did not recognise the `adopt` action, so it fell through to the generic HAProxy diff and rendered the cluster's entire `haproxy.cfg` as removed; and **rejecting** an adoption hid the node from the panel permanently, because `adopted_vip_id` is write-once, a VIP is only ever soft-deleted (so the column's `ON DELETE SET NULL` never fires) and the agent does not re-report an unchanged file — adoptability is now derived from whether the linked VIP is still active, which self-heals reject, undo-reject and approved teardown alike. No schema change, no agent change.
|
||||
- **v1.10.7** (2026-08-13) — **HA / VIP follows the selected cluster**: the page ignored the cluster picker in the header. On a multi-cluster install both the VIP table and the new *Unmanaged keepalived detected* panel listed every cluster's nodes at once and did not change when the selection did, so the panel appeared to be stuck on one cluster's keepalived. Both lists now send `cluster_id`, resolved to that cluster's pool exactly as the Apply Management view already did. The API parameter is **optional**: a caller that omits it still receives the whole fleet, so nothing outside the page changes. This is a deliberate behaviour change for the VIP table, which was fleet-wide before. Backend and frontend only: no schema, no agent change.
|
||||
- **v1.10.6** (2026-08-13) — **VIP adoption panel was unreachable**: v1.10.4's *Unmanaged keepalived detected* panel never appeared, even on a fleet where the agents had reported their configs correctly. `GET /discoveries` was declared **after** `GET /{vip_id}` in `routers/vip.py`, and FastAPI matches routes in declaration order, so every request for the discovery list was answered by the get-one-VIP handler, which takes `vip_id: int` and rejected `"discoveries"` with **422** before the real handler ran. Nothing surfaced the failure: the agents reported normally, the rows landed in `vip_discoveries`, and the HA/VIP page treats any non-OK response as "nothing to show" — so the whole feature was invisible with no error anywhere. The route is moved above the parameterised ones, and a static source scan now asserts that **no** literal path in **any** router is shadowed by an earlier parameterised one, so the class of bug cannot come back silently. Data reported under v1.10.4 is not lost: existing `vip_discoveries` rows appear as soon as the fixed backend is deployed, with no agent action needed. Backend-only fix. No schema, API-shape or agent change.
|
||||
- **v1.10.5** (2026-08-09) — **HTTP-01 challenge backend on split deployments**: on a deployment where the HAProxy nodes and the management stack are on different hosts, HTTP-01 issuance could fail silently for weeks while DNS-01 kept working — the rendered config pointed `server _acme_mgmt` at an address that resolves **on the HAProxy node**, defaulting to loopback, and every diagnostic still reported success. The per-cluster `acme_backend_url` now has a UI field, changing it actually mints a config version, and the value is validated where it is written. Three adjacent bugs are fixed with it: a config-generation failure was returned as `# Error ...` text and then stored as an APPLIED version and pushed to agents as the cluster's whole `haproxy.cfg` (both call sites now refuse with 422); a nullable `frontends.mode` was interpolated raw and emitted `mode None`, which HAProxy rejects and which takes down the entire cluster config; and cluster creation silently dropped the ACME fields. `docker-compose.yml` now interpolates `PUBLIC_URL` / `MANAGEMENT_BASE_URL` instead of hardcoding them, with the old literals as defaults. Diagnostics read the response body so an SPA answering 200 is no longer counted as healthy, and every new condition is a warning rather than a failure so no install is locked on upgrade. No schema, API-shape or agent change.
|
||||
- **v1.10.4** (2026-08-08) — **Adopt an existing keepalived VIP** (Issue #27 follow-up): on a fleet that already runs keepalived, the **HA / VIP** page came up empty, because the flow was one-way — VIPs were declared in OpenManager and pushed to the node, and nothing ever read what was already there. Agents now **report the `keepalived.conf` they find and do not own** (strictly read-only; the node is never touched), the page lists those nodes under *Unmanaged keepalived detected*, and **Adopt** turns one `vrrp_instance` into a managed VIP with the values from the file instead of retyping them. The heartbeat could not drive this: it carries the VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs **eleven** fields, and guessing them is not cosmetic — a wrong `virtual_router_id` puts the nodes in separate VRRP domains and a wrong `auth_pass` makes them reject each other, so both would claim the VIP. Because adoption **replaces** the operator's file with OpenManager's render, the parser reports every directive it cannot reproduce — a `notify_master` hook, an LVS `virtual_server` section, a `vrrp_sync_group`, a second address in one instance, a custom `track_script` — and **refuses** while any remain; the operator can waive that class explicitly, but a value that is simply *unknown* (an absent VRID or prefix length) can never be waived, only supplied. keepalived's own documented defaults (`state BACKUP`, `priority 100`, `advert_int 1`) are applied and shown as assumed. The agent's ownership guard is **not** weakened: adoption authorises exactly **one** takeover of exactly the file that was analysed, pinned to its hash, so a config edited between adoption and Apply is still refused. The adopted VIP is created **PENDING** like any other, so nothing reaches the node until it is applied from Apply Management. VRRP passwords are Fernet-encrypted at ingest and masked in the stored copy and the preview. Schema change: one new table `vip_discoveries` plus two additive columns (SCHEMA_VERSION 10 → 11, auto-migrated, no existing table altered) — **see the upgrade notes: this bump re-seeds the four built-in roles, and the Linux agent script must reach the nodes before discovery starts**.
|
||||
- **v1.10.3** (2026-08-08) — **Multi-account ACME: the certificate wizard honours the account you pick**: with more than one ACME account registered, picking an **HTTP-01** account in *Request ACME Certificate* still produced a **DNS-01** request. Three faults compounded. (1) `Form.useWatch` reports only fields that are currently **rendered**, and the account `Select` lives on the *Configuration* step — so as soon as the wizard advanced to *Review* the watch read `undefined` and the wizard silently reverted to the default account, even though the value was still in the form store; the watches now pass `preserve: true`. The same fault disabled the **wildcard guard** on *Review*, the one step where Submit lives. (2) The UI and the backend disagreed on which account is the *default*: the backend takes the **newest** valid account (`ORDER BY created_at DESC`), the UI took the **oldest** entry of a list ordered by id — the opposite account whenever the two differ. The wizard now resolves the same one, and sends `account_id` **explicitly** so there is no guess left to disagree about. (3) `account_id` was read from the form store while `challenge_type` came from the reverted account object, so the request asked for DNS-01 validation on an HTTP-01 account and the API answered `The selected ACME account has no DNS provider configured for DNS-01.` — both are now derived from one resolved account. The *Review* step also showed the default account's address instead of the chosen one, and Submit stayed enabled for a deactivated account; both fixed. Frontend only — no schema, API-shape, agent or rendered-config changes, and single-account installations behave exactly as before.
|
||||
- **v1.10.2** (2026-08-08) — **Dark mode fixes on Apply Management**: several panels on the Apply Management page were painted with light-mode colour literals, so in dark mode the **Pending Changes** box rendered as a cream panel with light text on it — measured contrast **1.03:1**, effectively unreadable, now **11.50:1**. The same bug affected the added/removed rows in the *View Change* diff (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME and pending-version panels, the VIP pending-delete row, and the agent-error recommendation box; all now derive from theme tokens. Separately, **static confirm dialogs came up white in dark mode**: in Ant Design 5 the static `Modal.confirm` / `message` / `notification` APIs render into their own detached root and never see the app's `ConfigProvider`, so they always used the light algorithm. Registering `ConfigProvider.config({ holderRender })` once at the app root fixes **every** static dialog in the application (12 components use them), not only this page. Light mode is byte-identical — each token resolves under the default algorithm to exactly the literal it replaced. Frontend only: no schema, API, environment or agent change.
|
||||
- **v1.10.1** (2026-08-08) — **CSR private key encrypted at rest** (Issue #53): the private key of a **pending** CSR is now Fernet-encrypted in the database instead of stored as PEM. It is the one key in the system worth protecting this way — it sits idle for the entire signing window (days to weeks), is never transmitted to an agent, and is destroyed the moment the signed certificate is imported; `ssl_certificates.private_key_content` and the ACME order keys are unchanged, because agents must receive those in plaintext on every poll. The token replaces the PEM in the **same column**, so there is **no schema change and no `SCHEMA_VERSION` bump** (and therefore no re-seed of the built-in roles). CSRs created before this release keep a raw PEM and are still read transparently, so anything already out for signature imports normally with no data migration. The key derives from `SECRET_KEY` via HKDF with its own info string, independent of the VIP/MFA/DNS keys, and an optional `CSR_ENCRYPTION_KEY` enables independent rotation — rotating `SECRET_KEY` without it makes pending CSR keys unrecoverable, which now fails with an explicit "delete and re-create this CSR" error rather than a misleading key-mismatch. `.env.template` now documents all four per-purpose encryption keys. No API, UI or agent change.
|
||||
- **v1.10.0** (2026-08-07) — **GoDaddy DNS provider for DNS-01** (Issue #35 follow-up): DNS-01 challenges can now be published and cleaned up automatically through **GoDaddy**, alongside the existing Manual and Cloudflare providers, so wildcard and internal-cluster certificates on GoDaddy-hosted zones **renew unattended**. Credentials are a **Production API Key + Secret** pair from `developer.godaddy.com/keys` (a **Personal Access Token** also works — paste it as the Key and leave the Secret blank, which is the forward path as GoDaddy retires `sso-key`); they are **verified against the GoDaddy API before being saved** and **encrypted at rest** (Fernet, the same path as Cloudflare), and are never returned by the API, logged, or written to an order event. GoDaddy's v1 API has **no per-value TXT write** — `PUT` replaces an entire RRset — so add/remove are read-modify-write with sibling values merged back, empty-`data` tombstones filtered out, and `DELETE` used for the last value (`PUT []` is rejected); this is what keeps the **apex + wildcard** case (two TXT values at one `_acme-challenge` name) working, and the record path is hard-gated so it can never collapse onto the zone-wide endpoint that would wipe SPF/DKIM/DMARC. Zone lookup probes the records API rather than the domain listing, so **delegated sub-zones** resolve and small accounts are not falsely rejected. Registry-only addition: one new provider module plus one registry line — no frontend change (the credential form is schema-driven). No schema, API-shape, agent, or rendered-config changes; Manual, Cloudflare and HTTP-01 are unaffected.
|
||||
- **v1.9.0** (2026-08-04) — **CSR creation** (in-app key + CSR generation and signed-certificate import): a new **CSR tab** on the SSL Certificates page generates a private key and Certificate Signing Request server-side (RSA 2048/4096 or ECDSA P-256/P-384; full subject — O/OU/L/ST/C/email — plus DNS SANs with wildcard support), for certificates signed by an **external or corporate CA**. The operator downloads/copies the CSR PEM, has it signed, then imports the signed certificate (+ optional chain): the backend verifies the certificate against the stored key (hard gate), rejects expired certs, warns on SAN drift, and creates a normal SSL certificate entry (source `CSR`) that flows through the standard **PENDING → Apply Management → agent pull** pipeline. The private key **never leaves the server** — no CSR endpoint returns it, and after import the CSR row's key copy is destroyed (the key then lives only on the certificate, like every other key). Additive schema change: one new table `ssl_csrs` (SCHEMA_VERSION 9 → 10, auto-migrated, no existing table altered); key generation runs off the event loop and is rate-limited per user; existing `ssl.*` permissions govern all new endpoints. No agent or rendered-config changes.
|
||||
- **v1.8.10** (2026-07-20) — **Security hardening** (GHSA-7rhv-c5pc-69r8, GHSA-3p5c-m5m4-mjpx, GHSA-3vh4): three advisory classes remediated, backend-only, no agent changes. (1) **RCE**: the agent script-template read/write endpoints now require the `agents.version` permission on top of authentication — a poisoned template is executed as root on every HAProxy node, so authentication alone was insufficient. (2) **Missing authentication**: operator/UI endpoints that were served without a JWT (dashboard stats, pool/cluster listings, agent inventory, WAF rules, config validate/optimize, SSL config-versions, health deep/agents/clusters) are now gated by a `require_authenticated_user` dependency, and agent data-plane endpoints that treated the `X-API-Key` header as *optional* (heartbeat, config, ssl-certificates, upgrade-status, pending-requests) now hard-reject a missing key. In every case the auth check was moved **ahead of** the handler's `try:` block so a 401 can no longer be rewritten into a 500 by the generic exception handler. (3) **SSRF**: a new `utils/ssrf_guard.py` (https-only, IPv4-pinned connector, all resolved addresses must be public, no redirects) protects the ACME directory fetch, the signed-request target and the ACME connection test, which accept operator- or DB-supplied URLs; the connection test also stopped reflecting arbitrary upstream JSON. Frontend dependency advisories patched in the same release. No schema, API-shape or rendered-config changes.
|
||||
|
||||
@@ -1,3 +1,523 @@
|
||||
# Upgrade Notes — v1.11.1 (adoption cannot hide a node; Config Import on fresh installs)
|
||||
|
||||
**Agent-script change, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. After deploying, sync the Linux agent script from **Agent Management** and let
|
||||
the agents upgrade, or none of this reaches the nodes.
|
||||
|
||||
- **A node that never appeared under *Unmanaged keepalived detected* now recovers by itself.**
|
||||
The agent caches the hash of its last discovery report and skips re-posting while it matches.
|
||||
Delivery was judged by `curl`'s exit code, which is 0 for 5xx too, so a rejected report was
|
||||
cached as delivered and a hand-maintained config — which never changes on its own — kept that
|
||||
node hidden. The report is now cached only on a 2xx, and the config endpoint reports whether
|
||||
the server actually holds a discovery for that agent, so the cache can only suppress while the
|
||||
server agrees. **No access to the nodes is needed**; affected nodes reappear within one poll
|
||||
cycle (~2.5 min) after the agents pick up the new script.
|
||||
- **Permanent refusals do not loop.** A 400, 413 or 422 means the payload itself is unacceptable,
|
||||
so the refusal is recorded and the same bytes are not re-posted; fixing the file releases the
|
||||
brake, because it is keyed to the content hash. **401 and 404 keep retrying** — in this system
|
||||
they mean a token rotation or an agent row briefly absent while it re-registers, and braking on
|
||||
them would have re-created the very failure above. This matters beyond noise: 4xx and 5xx agent
|
||||
calls are never sampled out of the request log, so a loop would write a row carrying the whole
|
||||
`keepalived.conf` every cycle on every affected node.
|
||||
- **The clear path had the same defect.** When a node becomes managed the agent tells the server
|
||||
to drop the discovery; that too was judged by the exit code, so a rejected clear left a stale
|
||||
row offering a **managed** node for adoption, with nothing to ever retry it.
|
||||
- **Config Import now works on a freshly installed agent.** `check_config_requests` uploads a
|
||||
node's live `haproxy.cfg` when you ask for it. It was defined in the installer and in the
|
||||
self-upgrade daemon but not in the body a fresh install writes, and its call site is guarded by
|
||||
`type`, so on such a node the feature was a silent no-op: the request was made and nothing ever
|
||||
arrived. Any agent that had self-upgraded at least once already had it, which is why it went
|
||||
unnoticed. A freshly installed agent now polls the pending-requests endpoint once per cycle,
|
||||
exactly as every upgraded agent already does — **no node running today changes behaviour**.
|
||||
- **The keepalived.conf path is resolved deterministically.** A pool may hold more than one
|
||||
cluster and the join was unordered, so the path handed to an agent could differ between polls
|
||||
whenever two clusters disagreed on it — the agent would inspect a file that is not there and the
|
||||
node would never appear, intermittently. A customised path now wins over the shipped default,
|
||||
then the lowest cluster id. With one cluster per pool, or when every cluster carries the
|
||||
default, the value is byte-identical to before.
|
||||
|
||||
**Rollback:** safe. No schema or data change; reverting restores the previous behaviour, in which
|
||||
a rejected discovery report is never retried and Config Import is absent on fresh installs.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.11.0 (Unified request/response log)
|
||||
|
||||
**Adds one new table and bumps `SCHEMA_VERSION` 11 → 12. The migration runs automatically on the
|
||||
first backend start.** No agent impact, no rendered-config change, no change to any existing API
|
||||
shape or response body.
|
||||
|
||||
> **12, not 11.** The feature was developed against v1.10.3, where `SCHEMA_VERSION` was still 10, and
|
||||
> proposed 11. v1.10.4 took 11 in the meantime. Shipping it as 11 would have been silently inert:
|
||||
> `run_all_migrations()` returns early on `applied_version >= SCHEMA_VERSION`, so every database
|
||||
> already at 11 would have skipped the entire sequence and received neither `request_logs` nor the
|
||||
> `requestlog.*` permissions, while a fresh install would have received both. If you are coming from
|
||||
> a pre-release build that recorded 11 for this feature, no action is needed: the bump to 12 re-runs
|
||||
> the (idempotent) sequence and creates whatever is missing.
|
||||
|
||||
- **New table `request_logs`** (BIGSERIAL primary key, 9 indexes). Created empty and starts filling
|
||||
immediately. The shipped defaults keep 7 days of successful requests, 30 days of failed ones, and
|
||||
at most 500 000 rows — whichever limit is reached first. Change any of it in
|
||||
*Settings → Request Log*.
|
||||
- **Successful agent polls are NOT logged** (`requestlog.capture_agent_success`, default off).
|
||||
Failed agent calls always are. This is what keeps the table's size a function of operator
|
||||
activity rather than of how many nodes you run. Measured at 2 424 bytes/row on PostgreSQL 15
|
||||
against the real schema and all nine indexes, with each agent issuing ~9 792 logged calls a day:
|
||||
|
||||
| fleet | if successful polls were logged | with the shipped default |
|
||||
|---|---|---|
|
||||
| 20 nodes | 196k rows/day, 453 MB/day, row cap in 61 h | ~21k rows/day, 48 MB/day, cap in 24 d |
|
||||
| 200 nodes | 2.0M rows/day, 4.5 GB/day, row cap in 6 h | ~30k rows/day, 69 MB/day, cap in 17 d |
|
||||
| 500 nodes | 4.9M rows/day, 11 GB/day, row cap in 2.5 h | ~44k rows/day, 103 MB/day, cap in 11 d |
|
||||
|
||||
The row cap always holds, but it holds by deleting — so without this default the configured
|
||||
"7 days of successes, 30 days of failures" silently becomes a few hours of both, and the forensic
|
||||
record the feature exists for is evicted by polling noise. Turn it on temporarily when debugging
|
||||
a specific node, then turn it back off.
|
||||
- **Runtime cost is measured, not estimated.** The middleware adds 27.7 µs (p50) per request and
|
||||
1.4 µs on an excluded path; redaction runs on the writer task, off the request path, at 18.8 µs
|
||||
per row. At 500 nodes that is **0.096 % of one core** in total.
|
||||
- **New permissions `requestlog.read` and `requestlog.manage`.** Both are granted to `super_admin`
|
||||
and `security_admin`; `operator` gets `requestlog.read` only; `viewer` gets neither, because
|
||||
captured request/response bodies are a broader disclosure surface than the read-only configuration
|
||||
views a viewer is meant to have. Custom roles can be granted either from **Users → Roles**.
|
||||
- **⚠️ Built-in roles are re-seeded to their defaults.** This is the pre-existing behaviour of every
|
||||
`SCHEMA_VERSION` bump, not something new in this release, but it bites here because this release
|
||||
bumps: the version gate re-runs the whole sequence and
|
||||
`update_system_roles_to_enterprise_rbac()` issues an unconditional
|
||||
`UPDATE roles SET … permissions = <defaults> WHERE name = …` for the four **built-in** roles
|
||||
(`super_admin`, `operator`, `security_admin`, `viewer`). **Any customisation you made to a
|
||||
built-in role is reverted.** Roles you created yourself are untouched (the update matches on
|
||||
name). To preserve customisation, export with `GET /api/roles` before upgrading and re-apply with
|
||||
`PUT /api/roles/{id}`, or move the customisation into a custom role.
|
||||
- **Bodies are captured, redacted and capped at 8 KB.** Passwords, tokens, API keys, private-key
|
||||
PEMs, JWT-shaped values, `Authorization` / `Cookie` headers, ACME JWS payloads and DNS-provider
|
||||
credentials are never stored. Headers use an allowlist — anything not on it is dropped rather than
|
||||
saved. Review *Settings → Request Log* before enabling body capture in a regulated environment;
|
||||
`capture_bodies` can be turned off while still recording who called what, with what result.
|
||||
- **Excluded by default:** health checks, the API docs, the ACME HTTP-01 challenge endpoint (it
|
||||
returns `key_authorization`), the agent heartbeat (the highest-volume POST in the system), static
|
||||
assets, and the log viewer's own endpoints. The list is editable, except the log viewer itself,
|
||||
which is a hard floor so the page cannot end up logging you reading it.
|
||||
- **Disk growth is the main operational consideration.** On a busy install, in order of bluntness:
|
||||
leave `capture_agent_success` off (the single biggest lever on a large fleet), lower `sample_rate`
|
||||
(errors are always kept at 100 %), turn off `capture_get`, turn off `capture_bodies`, or shorten
|
||||
`success_retention_days`.
|
||||
- **`operator` can see agent rows.** A caller holding only `requestlog.read` sees their own inbound
|
||||
requests plus the fleet's, and nothing belonging to another user. Agent rows are included because
|
||||
an apply fails on the *node* and the node reports it over its own API key — scoping to own-rows
|
||||
only would have hidden the diagnosis from exactly the role the grant exists for. Anonymous traffic
|
||||
(failed logins and the usernames they carry, unauthenticated probes) is **not** agent traffic and
|
||||
remains visible only to `requestlog.manage`.
|
||||
- **New environment variables**, all optional: `REQUEST_LOG_ENABLED` (default `true`),
|
||||
`REQUEST_LOG_QUEUE_MAX` (2000 rows), `REQUEST_LOG_QUEUE_MAX_BYTES` (64 MiB),
|
||||
`REQUEST_LOG_BATCH_SIZE` (100), `REQUEST_LOG_FLUSH_MS` (500). See `.env.template`.
|
||||
`REQUEST_LOG_QUEUE_MAX_BYTES` is a hard memory ceiling per worker: the row count alone does not
|
||||
bound memory, because `max_body_bytes` is operator-editable up to 256 KB and a row can carry it
|
||||
twice — at the ceiling the default 2 000-row queue would hold ~1 GiB, which is the whole pod
|
||||
limit. Whichever limit binds first stops the queue. If you raise `REQUEST_LOG_QUEUE_MAX`, raise
|
||||
this with it.
|
||||
- **`GET /api/request-logs/stats` sink counters are per worker**, labelled as such in the response.
|
||||
With `UVICORN_WORKERS > 1` each process keeps its own queue and its own drop counter.
|
||||
- **Clearing the exclude-path list does not mean "log everything."** An empty list falls back to the
|
||||
shipped defaults, so the log viewer and the raw-body agent heartbeat stay excluded; the UI now
|
||||
says so and shows what was actually applied.
|
||||
- **To disable entirely:** set `REQUEST_LOG_ENABLED=false` in the backend environment and restart —
|
||||
the middleware is then not registered at all and costs nothing, not even a settings lookup. The
|
||||
`enabled` toggle in Settings is the no-restart equivalent (it takes effect immediately).
|
||||
- **Default admin password is not reset** by this bump; user seeding is guarded by an existence
|
||||
check, not an upsert.
|
||||
- **Rollback:** downgrade the backend image freely. `request_logs` is purely additive and is simply
|
||||
ignored by v1.10.x. Drop the table manually if you want the space back:
|
||||
`DROP TABLE IF EXISTS request_logs;`
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.14 (a converged node keeps acknowledging)
|
||||
|
||||
**Agent-script change, no schema change.** No `SCHEMA_VERSION` bump. After deploying, sync the
|
||||
Linux agent script from **Agent Management** and let the agents upgrade, or the fix does not
|
||||
reach the nodes.
|
||||
|
||||
- **Symptom:** a VIP shows `SYNCING (0/n)` with an empty *Last ack* even though every member node
|
||||
has the rendered `keepalived.conf` on disk, keepalived is running and the VIP is held.
|
||||
- **Cause:** the deploy report was sent only when the agent actually wrote the config. Once the
|
||||
node matched, it took the idempotency early return every cycle and never reported again, so any
|
||||
report lost in transit was never retried and the server's view stayed stale permanently.
|
||||
- **Fix:** the agent re-asserts its state on the idempotent path as well. One request per node
|
||||
per poll cycle (~2.5 min); nothing is written and keepalived is not reloaded.
|
||||
- **Recovery is automatic.** A VIP stuck at SYNCING converges on the first poll after the agents
|
||||
pick up the new script. No action on the nodes, no re-apply, no edit to force a rewrite.
|
||||
- **This is not new in 1.10.12.** The gap dates from the original HA/VIP work; it only became
|
||||
visible when acknowledgements were dropped for an unrelated reason.
|
||||
|
||||
**Rollback:** safe. Reverting restores the previous behaviour, in which a lost acknowledgement is
|
||||
never recovered.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.13 (deploy acknowledgements were dropped)
|
||||
|
||||
**Backend only, no schema change.** No `SCHEMA_VERSION` bump, no agent change. If you deployed
|
||||
v1.10.12, deploy this one too.
|
||||
|
||||
- **Regression in v1.10.12.** The takeover-retirement clause added to the keepalived status
|
||||
endpoint reused a query placeholder for both an assignment and a comparison. PostgreSQL types a
|
||||
placeholder per use, so it was deduced as `text` in one place and `character varying` in the
|
||||
other, and asyncpg refused the statement outright.
|
||||
- **Symptom:** a VIP stayed at `SYNCING (0/n)` with an empty *Last ack* even though the agent log
|
||||
showed `applied config for VIP <id>` on every member. Teardown acknowledgements were lost the
|
||||
same way, so a deletion never showed as complete.
|
||||
- **Nothing was damaged.** The failure was on the write of the acknowledgement, not on the node.
|
||||
Configs were deployed correctly throughout; only the reporting was lost. Existing VIPs converge
|
||||
on the next poll once this is deployed, with no action on the nodes.
|
||||
- **Verified against a real PostgreSQL**, not by inspection: both statements execute, a matching
|
||||
hash retires the takeover authorisation, a non-matching hash and a NULL `applied_config_hash`
|
||||
leave it in place, and the acknowledgement is recorded in every case.
|
||||
|
||||
**Rollback:** do not roll back to v1.10.12; roll back to v1.10.11 instead, which predates the
|
||||
clause entirely.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.12 (valid config rejected by its own warning)
|
||||
|
||||
**Agent-script change, no schema change.** No `SCHEMA_VERSION` bump. After deploying, sync the
|
||||
Linux agent script from **Agent Management** and let the agents upgrade, or the fix does not
|
||||
reach the nodes.
|
||||
|
||||
- **Symptom:** applying a VIP left it stuck at `SYNCING`, the node kept its previous config and
|
||||
the agent logged only `config validation failed (keepalived -t)`.
|
||||
- **Cause:** the agent treated any non-zero exit from `keepalived -t` as invalid. keepalived's
|
||||
config-test exit code does not separate fatal from benign: on 2.2.8 a clean config exits 0,
|
||||
while `Truncating auth_pass to 8 characters` exits 5 and so do a missing `}` and an unknown
|
||||
keyword. A VRRP password longer than eight characters was enough to block every apply, even on
|
||||
a node whose own running config emits the same warning.
|
||||
- **Fix:** the gate judges the output instead. Known-benign messages are dropped and anything
|
||||
left still fails, so it fails closed. Verified against real keepalived: the truncation warning
|
||||
passes; a missing brace, an unknown keyword and a `SECURITY VIOLATION` are refused.
|
||||
- **Also:** the agent now reports what keepalived actually said, in its log and in the status the
|
||||
HA/VIP page shows. The refusal was correct but unactionable without reproducing it by hand.
|
||||
- **The fail-safe itself is unchanged:** a config that genuinely fails validation is never
|
||||
written and keepalived is never restarted.
|
||||
|
||||
**Rollback:** safe. Reverting restores the stricter gate, which rejects valid configs whose
|
||||
password exceeds eight characters.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.11 (Adoptable tag names the real blocker)
|
||||
|
||||
**Frontend only, no schema change.** No `SCHEMA_VERSION` bump, no API change, no agent change.
|
||||
|
||||
- The *Adoptable* tag and the disabled *Adopt* button were derived separately, so they could
|
||||
name different problems. A pair blocked by a peer whose config could not be parsed showed
|
||||
**MASTER missing**, because the unreadable node's `state MASTER` had not been counted — true,
|
||||
but it sent the operator to the wrong node. Both now come from one ordered decision.
|
||||
- New label **blocked by peer** for a group held up by a node that references the same address
|
||||
but cannot be taken over with it. Two MASTERs is now distinct from none.
|
||||
- Display only. The endpoint's checks and refusals are unchanged.
|
||||
|
||||
**Rollback:** safe; purely presentational.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.10 (Adoption blockers listed once per instance)
|
||||
|
||||
**Frontend only, no schema change.** No `SCHEMA_VERSION` bump, no API change, no agent change.
|
||||
|
||||
- The adoption panel merged every member's blocker list, so a two-node pair showed each shared
|
||||
problem twice. The two files report different line numbers for the same directive, so exact
|
||||
de-duplication did not collapse them. Blockers are now merged on the message with the leading
|
||||
`line N:` ignored.
|
||||
- Display only. The endpoint already evaluated the combined set across all nodes, and what it
|
||||
accepts or refuses is unchanged.
|
||||
|
||||
**Rollback:** safe; purely presentational.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.9 (Adoption refuses to strand a node)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent change.
|
||||
|
||||
- **Adoption will not leave a node behind.** v1.10.8 resolved the whole VRRP instance, but only
|
||||
from nodes it could parse, that were enabled and that were in the same pool. Anything else fell
|
||||
out of the set silently while its peers were rewritten. Adoption now refuses if any reported
|
||||
`keepalived.conf` mentions the virtual address and is not among the nodes being taken over, and
|
||||
says which node and why.
|
||||
- **The nodes must agree on the shared fields.** `prefix_length`, unicast/multicast mode, HAProxy
|
||||
tracking and the VRRP password live on the VIP and are re-rendered onto every member, so one
|
||||
node's value used to be imposed on the rest. A disagreement is now refused with both values
|
||||
shown.
|
||||
- **The takeover authorisation is now retired on acknowledgement.** It is the permission to
|
||||
overwrite a `keepalived.conf` that does not carry our marker. It was never cleared, so it stayed
|
||||
valid for that exact file content indefinitely; restoring the pre-adoption file would have been
|
||||
overwritten again without fresh approval. It is now dropped once the member acks our rendered
|
||||
config, gated on the acked hash matching `applied_config_hash` so a failed deploy cannot strand
|
||||
the VIP.
|
||||
- **Nothing to do on upgrade.** Existing adopted VIPs keep working; their authorisation is retired
|
||||
on the next successful acknowledgement.
|
||||
|
||||
**Rollback:** safe. No schema or data migration; reverting restores the previous (more permissive)
|
||||
adoption checks.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.8 (VIP adoption takes the whole VRRP instance)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent impact: nothing about what the agent reports or how it takes a
|
||||
config over changes.
|
||||
|
||||
- **Adoption is now per VRRP instance, not per node.** Every node in the pool reporting the same
|
||||
`virtual_router_id` and virtual address becomes a member of one VIP, each with the role,
|
||||
priority and interface its own `keepalived.conf` declares, and each with its own one-shot
|
||||
takeover hash. The panel lists one row per instance.
|
||||
- **Why this mattered:** single-node adoption could not produce a working pair. The BACKUP alone
|
||||
failed apply, the MASTER alone left the peer unmanaged and the peer could not then be adopted
|
||||
(VRID collision). On a **unicast** instance it was worse than inconvenient: the render drops
|
||||
the unicast block when there are no peers, so the adopted node fell back to multicast while its
|
||||
peer stayed unicast and both could hold the address.
|
||||
- **New refusals, each with the reason in the message:** the group does not have exactly one
|
||||
MASTER; the nodes disagree on `advert_int`; a declared unicast peer is not among the nodes being
|
||||
adopted; a node is already a member of a live VIP.
|
||||
- **Apply Management "View Change" now renders the adopt diff correctly.** It did not recognise
|
||||
the `adopt` action and fell through to the generic HAProxy diff, which compared the staged
|
||||
`keepalived.conf` against the cluster's previous `haproxy.cfg` and showed the whole HAProxy
|
||||
config as removed. Alarming, but display-only — nothing was ever applied from that view.
|
||||
- **Rejecting an adoption is recoverable again.** It used to hide the node from the panel
|
||||
permanently. Nothing clears `vip_discoveries.adopted_vip_id`, a VIP is only soft-deleted so the
|
||||
column's `ON DELETE SET NULL` never fires, and the agent does not re-report a file whose hash
|
||||
has not changed. Adoptability is now derived from whether the linked VIP is still active.
|
||||
- **If you adopted a VIP on 1.10.4-1.10.7**, check it before applying: it may have only one
|
||||
member. Add the peer from the VIP's edit form, or reject the pending adoption and adopt again —
|
||||
the node reappears in the panel under this release.
|
||||
|
||||
**Rollback:** safe. No schema or data change; reverting restores the previous single-node
|
||||
adoption behaviour.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.7 (HA / VIP follows the selected cluster)
|
||||
|
||||
**Backend + frontend, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No agent impact.
|
||||
|
||||
- **The HA / VIP page ignored the cluster picker.** Both the VIP table and the *Unmanaged
|
||||
keepalived detected* panel queried the whole fleet, so on an install with more than one
|
||||
cluster the lists never changed when the selection did. Both now pass `cluster_id`, mapped to
|
||||
the cluster's pool the same way `GET /api/vip?cluster_id=` already worked for Apply
|
||||
Management.
|
||||
- **Behaviour change worth knowing:** the VIP table is now scoped to the selected cluster. It
|
||||
used to show every VIP in the fleet. If you relied on the fleet-wide view, the API still
|
||||
supports it — `GET /api/vip` and `GET /api/vip/discoveries` without `cluster_id` return
|
||||
everything, unchanged.
|
||||
- **API compatibility:** `cluster_id` is optional on both endpoints. Existing integrations that
|
||||
do not send it behave exactly as before.
|
||||
|
||||
**Rollback:** safe. The change is a query parameter plus the page that sends it; reverting
|
||||
restores the fleet-wide lists and touches no data.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.6 (VIP adoption panel was unreachable)
|
||||
|
||||
**One backend fix, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are
|
||||
**not** re-seeded. No API-shape change, no frontend change and zero agent impact.
|
||||
|
||||
- **v1.10.4's adoption panel never appeared.** `GET /discoveries` was declared after
|
||||
`GET /{vip_id}` in `routers/vip.py`. FastAPI matches routes in declaration order, so the
|
||||
discovery list was routed into the get-one-VIP handler, which declares `vip_id: int` and
|
||||
answered **422** before the real handler ran. The HA/VIP page treats any non-OK response as
|
||||
"nothing to show", so the feature was invisible with no error in any log.
|
||||
- **Nothing was lost.** The agent side always worked: discoveries were reported and stored in
|
||||
`vip_discoveries`. Deploy this backend and the rows appear immediately — no agent upgrade, no
|
||||
re-sync of the agent script, no re-report needed.
|
||||
- **If you are upgrading straight from 1.10.3 or earlier**, follow the v1.10.4 notes below as
|
||||
well: that release does bump `SCHEMA_VERSION` (10 → 11), which re-seeds the four built-in
|
||||
roles, and its agent script has to reach the nodes before discovery starts.
|
||||
- **Regression guard.** A static source scan now fails the build if any literal API path in any
|
||||
router is declared after a parameterised route that would swallow it. The whole router tree is
|
||||
clean as of this release.
|
||||
|
||||
**Rollback:** safe and immediate. The change is a route declaration order plus a test; reverting
|
||||
to 1.10.5 restores the previous (broken-panel) behaviour and touches no data.
|
||||
|
||||
---
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.5 (HTTP-01 challenge backend on split deployments)
|
||||
|
||||
**Bug fixes, no schema change.** No `SCHEMA_VERSION` bump, so the built-in roles are **not**
|
||||
re-seeded. No API-shape change and zero agent impact.
|
||||
|
||||
- **HTTP-01 could fail silently when HAProxy runs on different hosts than the management stack.**
|
||||
The rendered config wrote `server _acme_mgmt <mgmt>:8080` from a value that defaults to
|
||||
loopback — and HAProxy resolves that address **on the HAProxy node**, so it pointed at the wrong
|
||||
box. Every check still reported success. The per-cluster `acme_backend_url` now has a UI field
|
||||
(Cluster Management), changing it actually mints a config version, and the value is validated at
|
||||
the write boundary.
|
||||
- **A config-generation failure could be pushed to agents as the cluster's whole `haproxy.cfg`.**
|
||||
The generator reported failure by *returning* `# Error ...` instead of raising, and the apply
|
||||
path hashed that comment and stored it as an APPLIED version. Both persisting call sites now
|
||||
refuse with 422 and leave the running config in force. **This is worth knowing even if you never
|
||||
touch ACME**, since any exception in the generator could trigger it.
|
||||
- **`frontends.mode` is nullable and was interpolated raw**, emitting a literal `mode None` that
|
||||
HAProxy rejects — which fails the whole cluster config, not just that frontend. Normalised now.
|
||||
- **Cluster creation ignored the ACME fields**: a cluster created with ACME switched on came back
|
||||
switched off, with no error.
|
||||
- **`docker-compose.yml` hardcoded `PUBLIC_URL` / `MANAGEMENT_BASE_URL`**, so a value in your
|
||||
`.env` or host environment was silently ignored. They are interpolated now, with the previous
|
||||
literals as defaults, so behaviour is unchanged unless you actually set them.
|
||||
- **Diagnostics stop over-reporting health.** The port-80 check now reads the body, so a reverse
|
||||
proxy answering 200 with a web page is no longer counted as a working challenge endpoint. Every
|
||||
new condition is a **warning, never a failure** — the Site Wizard blocks submit on a failing
|
||||
check, so a new failing condition would have locked installs on upgrade day.
|
||||
- **Rollback:** downgrade freely. No schema or data change.
|
||||
|
||||
---
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.4 (Adopt an existing keepalived VIP)
|
||||
|
||||
**Additive, but this release DOES bump the schema — read the role warning below.** Nothing on any
|
||||
node changes until you adopt a VIP and apply it.
|
||||
|
||||
- **Schema:** `SCHEMA_VERSION` bumps to `11`, so on first start the (idempotent) migration
|
||||
sequence re-runs once and adds **one new table** (`vip_discoveries`) plus two additive columns
|
||||
(`vip_instances.adopted_at`, `vip_members.takeover_expected_hash`). **No existing table is
|
||||
altered**, no existing row changes, and the admin password is not reset.
|
||||
- **⚠️ Built-in roles are re-seeded to their defaults** — the pre-existing behaviour of every
|
||||
`SCHEMA_VERSION` bump. If you customised `super_admin` / `operator` / `security_admin` /
|
||||
`viewer`, **re-apply those changes after upgrading**. (The three previous releases did not bump
|
||||
the version, so this is the first re-seed since v1.9.0.) No new permission strings are
|
||||
introduced: discovery and adoption are governed by the existing `vip.read` / `vip.create`.
|
||||
- **⚠️ The Linux agent script changed, and discovery does not start until nodes run it.** The
|
||||
fallback latest Linux agent version moves `2.0.0` → `2.1.0`, so nodes will pull the new script
|
||||
through the normal agent-upgrade path. The addition is **read-only**: the agent reads the
|
||||
`keepalived.conf` it does not own and reports it, rate-limited to once per content change. It
|
||||
writes nothing new to the node. Until a node has upgraded, it simply never appears under
|
||||
*Unmanaged keepalived detected*.
|
||||
- **Nothing is taken over implicitly.** The agent still refuses to overwrite a `keepalived.conf`
|
||||
that lacks OpenManager's ownership marker. Adoption authorises exactly **one** takeover of
|
||||
exactly the file that was analysed, pinned to its md5: if the file changes between adoption and
|
||||
Apply, the agent refuses again and reports `externally_managed` rather than clobbering your
|
||||
edit. Re-adopt to pick up the current file.
|
||||
- **Adoption can refuse, on purpose.** It replaces the file with OpenManager's render, so anything
|
||||
the renderer cannot reproduce would be destroyed. Those directives are listed as blockers —
|
||||
`notify_*` failover hooks, `vrrp_sync_group`, LVS `virtual_server` sections, a second address in
|
||||
one instance, a custom `track_script`, extra `global_defs`. You can accept that loss explicitly
|
||||
with a tick, but a value that is *unknown* rather than lost (an absent `virtual_router_id`, or
|
||||
an address with no prefix length) cannot be waived — the VRID is fatal to guess and the prefix
|
||||
has to be supplied, because picking a netmask for a live VIP would change its routing.
|
||||
- **Multi-node VIPs need every node.** Adoption covers the node that reported. Its unicast peers
|
||||
hold their own `keepalived.conf`, so adopt or add them as members before applying — otherwise
|
||||
the render has no peers. The UI says so after a successful adopt.
|
||||
- **Secrets:** the reported config may contain the VRRP `auth_pass`. It is split at ingest — the
|
||||
password is Fernet-encrypted into its own column (same key path as `vip_instances`,
|
||||
`VIP_ENCRYPTION_KEY` falling back to a key derived from `SECRET_KEY`) and the stored copy of the
|
||||
file has it masked, so nothing readable through the API, the UI preview or a DB dump carries it
|
||||
in cleartext.
|
||||
- **Rollback:** downgrading to 1.10.3 leaves `vip_discoveries` as an unused table and the two new
|
||||
columns unread; managed VIPs keep working. One caveat: a VIP adopted on 1.10.4 but **not yet
|
||||
applied** loses its takeover authorisation on downgrade, so the node's original config stays in
|
||||
place and the VIP sits PENDING — harmless, but re-adopt after upgrading again. Agents already on
|
||||
script 2.1.0 keep reporting discoveries to an endpoint that no longer exists; the report fails
|
||||
quietly and nothing on the node is affected.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.3 (Multi-account ACME wizard fix)
|
||||
|
||||
**Frontend only. Nothing to do on upgrade.** No schema, no `SCHEMA_VERSION` bump, no API change, no
|
||||
environment variable, zero agent impact. Installations with a single ACME account behave exactly as
|
||||
before.
|
||||
|
||||
- **What was broken:** with **more than one** ACME account registered, the *Request ACME
|
||||
Certificate* wizard did not honour the account you selected. Choosing an HTTP-01 account still
|
||||
submitted a DNS-01 request, which the API rejected with
|
||||
`The selected ACME account has no DNS provider configured for DNS-01.` The *Review* step also
|
||||
named the default account rather than the chosen one, so the mismatch was invisible before
|
||||
submitting.
|
||||
- **Default account:** the wizard previously previewed the **oldest** valid account while the
|
||||
backend uses the **newest** (`ORDER BY created_at DESC`). If you never picked an account
|
||||
explicitly and have several, requests were already going to the newest one — only the preview was
|
||||
wrong. The wizard now previews that same account, marks it `(default)`, and sends `account_id`
|
||||
explicitly so the two can no longer diverge.
|
||||
- **Wildcard guard:** the client-side "wildcard requires a DNS-01 account" block silently stopped
|
||||
applying on the *Review* step. Requests were still rejected by the backend, so nothing incorrect
|
||||
was ever issued — you now get the warning before submitting instead of an error after.
|
||||
- **No action needed on existing certificates or orders.** Nothing about issuance, renewal or the
|
||||
stored accounts changes; only how the wizard resolves which account a new request uses.
|
||||
- **Rollback:** downgrade freely. This release changes frontend behaviour only.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.2 (Dark mode fixes on Apply Management)
|
||||
|
||||
**Frontend only. Nothing to do on upgrade.** No schema, no `SCHEMA_VERSION` bump, no API change,
|
||||
no environment variable, zero agent impact. Light mode is byte-identical: every colour swapped in
|
||||
this release resolves, under the default algorithm, to exactly the literal it replaced
|
||||
(`colorWarningBg` → `#fffbe6`, `colorSuccessBg` → `#f6ffed`, `colorErrorBg` → `#fff2f0`, …), so
|
||||
only dark mode changes.
|
||||
|
||||
- **Apply Management panels** were painted with light-mode colour literals, so in dark mode the
|
||||
"Pending Changes" box rendered as a cream panel with light text on it. Measured contrast was
|
||||
**1.03:1** — effectively invisible. It is now **11.50:1**. The same class of bug affected the
|
||||
diff rows in *View Change* (2.21:1 and 2.99:1, now 5.49:1 and 4.01:1), the ACME/pending version
|
||||
panels, the VIP pending-delete row and the agent-error recommendation box.
|
||||
- **Static confirm dialogs came up white in dark mode.** In Ant Design 5 the static
|
||||
`Modal.confirm` / `message` / `notification` APIs render into their own detached root and never
|
||||
see the app's `ConfigProvider`, so they always used the light algorithm. This release registers
|
||||
`ConfigProvider.config({ holderRender })` once at the app root, which fixes **every** static
|
||||
dialog in the app (12 components use them), not just Apply Management.
|
||||
- **Rollback:** downgrade freely. This release changes rendering only.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.1 (CSR private key encrypted at rest)
|
||||
|
||||
**Backward compatible.** Nothing to do on upgrade, and nothing changes for existing clusters,
|
||||
agents or certificates:
|
||||
|
||||
- **Schema:** **no `SCHEMA_VERSION` bump and no migration.** The Fernet token replaces the PEM
|
||||
inside the *existing* `ssl_csrs.private_key_pem` TEXT column. As in v1.10.0, this means the
|
||||
four built-in roles are **not** re-seeded, so any customization of `super_admin` / `operator` /
|
||||
`security_admin` / `viewer` survives.
|
||||
- **Existing pending CSRs keep working.** Rows written before this release hold a raw PEM and are
|
||||
still read transparently, so a CSR that is already out for signature can be imported normally
|
||||
after the upgrade. There is no data migration and no downtime step. Those rows stay plaintext
|
||||
until they are imported (which NULLs the key) — if you want everything encrypted immediately,
|
||||
delete and re-create any long-pending CSRs.
|
||||
- **Scope:** this covers the PENDING CSR key only. It is the one key in the system that sits idle
|
||||
for the whole signing window and is never transmitted. `ssl_certificates.private_key_content`
|
||||
and the ACME order keys are unchanged, because agents must receive those in plaintext on every
|
||||
poll.
|
||||
- **Optional env:** `CSR_ENCRYPTION_KEY` (see `.env.template`). If unset, the key is derived from
|
||||
`SECRET_KEY` via HKDF with its own info string, so it is independent of the VIP, MFA and DNS
|
||||
provider keys.
|
||||
- **⚠️ Rotating `SECRET_KEY` while `CSR_ENCRYPTION_KEY` is unset makes pending CSR keys
|
||||
unrecoverable.** Import then fails with an explicit "delete this CSR and create a new one"
|
||||
error rather than a misleading key-mismatch. Set an explicit `CSR_ENCRYPTION_KEY` if you
|
||||
rotate `SECRET_KEY`. Certificates already imported are unaffected — their key lives on the
|
||||
certificate row.
|
||||
- **API / UI / agents:** unchanged. No CSR endpoint ever returned the private key before or now,
|
||||
and nothing about the CSR tab changes.
|
||||
- **Rollback:** the application downgrades cleanly — 1.10.0 starts normally against the same
|
||||
database and every other feature is unaffected. The one casualty is a CSR **created on 1.10.1
|
||||
and still pending**: 1.10.0 has no decrypt step, so it hands the Fernet token straight to the
|
||||
key-pairing check. Measured on a real downgrade, the import then fails with
|
||||
`HTTP 500 — Could not verify the certificate/key pair: key parse failed (encrypted?)`; it does
|
||||
**not** silently pair the wrong key, and it does not corrupt anything. Import or delete CSRs
|
||||
created on 1.10.1 before downgrading. Certificates already imported are unaffected, since their
|
||||
key lives on the certificate row, and CSRs created before 1.10.1 are plaintext and still work.
|
||||
|
||||
---
|
||||
|
||||
# Upgrade Notes — v1.10.0 (GoDaddy DNS-01 provider)
|
||||
|
||||
**Backward compatible & additive.** Nothing changes unless you select **GoDaddy** as an ACME
|
||||
|
||||
@@ -25,31 +25,42 @@ async def notify_agents_config_change(cluster_id: int, version_name: str) -> Lis
|
||||
for agent in agents:
|
||||
try:
|
||||
agent_url = f"http://{agent['ip_address']}:8081" # Agent default port
|
||||
|
||||
|
||||
async with aiohttp.ClientSession(timeout=aiohttp.ClientTimeout(total=10)) as session:
|
||||
payload = {
|
||||
"cluster_id": cluster_id,
|
||||
"version_name": version_name,
|
||||
"action": "config_update"
|
||||
}
|
||||
|
||||
async with session.post(f"{agent_url}/api/config/update", json=payload) as response:
|
||||
if response.status == 200:
|
||||
results.append({
|
||||
'node': agent['name'],
|
||||
'success': True,
|
||||
'message': f'Configuration updated successfully',
|
||||
'version': version_name
|
||||
})
|
||||
logger.info(f"✅ Agent {agent['name']} notified successfully")
|
||||
else:
|
||||
error_text = await response.text()
|
||||
results.append({
|
||||
'node': agent['name'],
|
||||
'success': False,
|
||||
'error': f'HTTP {response.status}: {error_text}'
|
||||
})
|
||||
logger.error(f"❌ Agent {agent['name']} notification failed: {response.status}")
|
||||
|
||||
# v1.11.0: instrumented so the code stays correct if the push
|
||||
# architecture is ever reverted. Unreachable today — see the
|
||||
# unconditional early return above.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_AGENT
|
||||
|
||||
push_url = f"{agent_url}/api/config/update"
|
||||
async with outbound_span(
|
||||
target=TARGET_AGENT, method="POST", url=push_url, request_body=payload
|
||||
) as span:
|
||||
async with session.post(push_url, json=payload) as response:
|
||||
if response.status == 200:
|
||||
span.set_response(response.status, getattr(response, "headers", None))
|
||||
results.append({
|
||||
'node': agent['name'],
|
||||
'success': True,
|
||||
'message': f'Configuration updated successfully',
|
||||
'version': version_name
|
||||
})
|
||||
logger.info(f"✅ Agent {agent['name']} notified successfully")
|
||||
else:
|
||||
error_text = await response.text()
|
||||
span.set_response(response.status, getattr(response, "headers", None), error_text)
|
||||
results.append({
|
||||
'node': agent['name'],
|
||||
'success': False,
|
||||
'error': f'HTTP {response.status}: {error_text}'
|
||||
})
|
||||
logger.error(f"❌ Agent {agent['name']} notification failed: {response.status}")
|
||||
|
||||
except asyncio.TimeoutError:
|
||||
results.append({
|
||||
|
||||
+62
-1
@@ -37,4 +37,65 @@ AGENT_CONFIG_SYNC_INTERVAL_SECONDS = 30
|
||||
|
||||
# Entity snapshot enabled by default (rollback functionality)
|
||||
# Set to "false" only if you need to disable snapshot temporarily
|
||||
ENTITY_SNAPSHOT_ENABLED = os.getenv("ENTITY_SNAPSHOT_ENABLED", "true").lower() == "true"
|
||||
ENTITY_SNAPSHOT_ENABLED = os.getenv("ENTITY_SNAPSHOT_ENABLED", "true").lower() == "true"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# v1.11.0 — unified request/response log
|
||||
# ---------------------------------------------------------------------------
|
||||
# These four are deliberately ENV-only (not database settings): they decide
|
||||
# whether the middleware is even registered and how much memory the writer
|
||||
# queue may hold, so they must be resolvable before the DB pool exists.
|
||||
# Everything the operator tunes at runtime (retention, body capture, sampling,
|
||||
# excluded paths) lives in `system_settings` under the `requestlog.` category
|
||||
# and is editable from Settings → Request Log.
|
||||
|
||||
def _bool_env(name: str, default: bool) -> bool:
|
||||
raw = os.getenv(name)
|
||||
if raw is None:
|
||||
return default
|
||||
return raw.strip().lower() not in ("0", "false", "no", "off", "")
|
||||
|
||||
|
||||
def _int_env(name: str, default: int, minimum: int, maximum: int) -> int:
|
||||
"""Read an int env var, clamped. A malformed value falls back to the
|
||||
default rather than crashing the process at import time."""
|
||||
raw = os.getenv(name)
|
||||
if raw is None or not raw.strip():
|
||||
return default
|
||||
try:
|
||||
value = int(raw.strip())
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
return max(minimum, min(maximum, value))
|
||||
|
||||
|
||||
# Hard kill-switch. When false the logging middleware is never added to the
|
||||
# ASGI stack and neither the writer nor the prune task is started — literally
|
||||
# zero overhead, not even a settings lookup.
|
||||
REQUEST_LOG_ENABLED = _bool_env("REQUEST_LOG_ENABLED", True)
|
||||
# Per-worker in-process queue depth. When full, rows are DROPPED (counted, and
|
||||
# reported through GET /api/request-logs/stats) — the request path never blocks
|
||||
# on the database.
|
||||
REQUEST_LOG_QUEUE_MAX = _int_env("REQUEST_LOG_QUEUE_MAX", 2000, 100, 100000)
|
||||
# HARD memory ceiling for the same queue, per worker. The row count above does
|
||||
# NOT bound memory on its own, because how much a row weighs is an operator
|
||||
# setting: `requestlog.max_body_bytes` is editable from Settings and its stated
|
||||
# ceiling is 256 KB, which a row can carry twice (request + response). Measured
|
||||
# on the real dataclass, the two limits multiply out to:
|
||||
#
|
||||
# defaults (2 000 rows x 8 KB) 33.9 MiB 3.3% of the 1 GiB pod limit
|
||||
# max_body_bytes at its 256 KB ceiling 1003 MiB at the pod limit
|
||||
# REQUEST_LOG_QUEUE_MAX at its ceiling 1695 MiB over the pod limit
|
||||
#
|
||||
# Both of those are reachable from documented, in-range values, and the drop
|
||||
# warning used to advise raising the queue - so following the tool's own advice
|
||||
# could OOM the worker. Whichever limit is hit FIRST now stops the queue, so
|
||||
# memory stays bounded no matter what the other is set to.
|
||||
REQUEST_LOG_QUEUE_MAX_BYTES = _int_env(
|
||||
"REQUEST_LOG_QUEUE_MAX_BYTES", 64 * 1024 * 1024, 1024 * 1024, 1024 * 1024 * 1024
|
||||
)
|
||||
# Rows per batched INSERT: one pool acquire per batch, not per request.
|
||||
REQUEST_LOG_BATCH_SIZE = _int_env("REQUEST_LOG_BATCH_SIZE", 100, 1, 1000)
|
||||
# Max wait before a partial batch is flushed (milliseconds).
|
||||
REQUEST_LOG_FLUSH_MS = _int_env("REQUEST_LOG_FLUSH_MS", 500, 50, 10000)
|
||||
|
||||
@@ -1365,6 +1365,10 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'roles.read', 'roles.create', 'roles.update', 'roles.delete', 'roles.permissions',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health', 'statistics.export',
|
||||
'activity.read', 'activity.all', 'activity.export',
|
||||
# v1.11.0 — request/response log. `read` browses the log,
|
||||
# `manage` edits retention/capture settings and triggers a
|
||||
# manual purge.
|
||||
'requestlog.read', 'requestlog.manage',
|
||||
'settings.read', 'settings.update', 'settings.system', 'settings.security',
|
||||
'system.restart', 'system.logs', 'system.database', 'system.services', 'system.emergency'
|
||||
]
|
||||
@@ -1384,7 +1388,11 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'vip.read', 'vip.create', 'vip.update', 'vip.delete', 'vip.apply',
|
||||
'config.read', 'config.update', 'config.download', 'config.history', 'config.bulk_import', 'config.view_request', 'config.download_request',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health',
|
||||
'activity.read'
|
||||
'activity.read',
|
||||
# v1.11.0 — operators debug failing applies and ACME orders,
|
||||
# so they get read access to the request log; retention and
|
||||
# purge stay with the admins.
|
||||
'requestlog.read'
|
||||
]
|
||||
},
|
||||
'security_admin': {
|
||||
@@ -1403,9 +1411,16 @@ async def update_system_roles_to_enterprise_rbac():
|
||||
'config.read', 'config.history', 'config.view_request', 'config.download_request',
|
||||
'statistics.read', 'statistics.performance', 'statistics.agents', 'statistics.health',
|
||||
'activity.read', 'activity.all', 'activity.export',
|
||||
# v1.11.0 — the request log is a security-forensics surface,
|
||||
# so the security admin gets both read and retention control.
|
||||
'requestlog.read', 'requestlog.manage',
|
||||
'settings.read', 'settings.security'
|
||||
]
|
||||
},
|
||||
# NOTE (v1.11.0): `viewer` deliberately gets NEITHER requestlog
|
||||
# permission. Even redacted, captured request/response bodies are a
|
||||
# far broader disclosure surface than the read-only configuration
|
||||
# views a viewer is meant to have.
|
||||
'viewer': {
|
||||
'display_name': 'Viewer',
|
||||
'description': 'Read-only access to view configurations, statistics, and monitor system status',
|
||||
@@ -1758,7 +1773,30 @@ async def ensure_agent_activity_logs_table():
|
||||
# until the operator imports the CA-signed certificate; the import creates a
|
||||
# normal ssl_certificates row and NULLs the key copy here. Additive + idempotent;
|
||||
# no existing table is altered, agents never read this table.
|
||||
SCHEMA_VERSION = 10
|
||||
# v1.10.4 (VIP adoption): bumped 10 -> 11 for the new `vip_discoveries` table plus two
|
||||
# additive columns (`vip_instances.adopted_at`, `vip_members.takeover_expected_hash`).
|
||||
# Holds the keepalived.conf an agent found already on a node so an existing VIP can be
|
||||
# adopted instead of retyped. Additive + idempotent; no existing table is altered and no
|
||||
# existing row changes. NOTE for the upgrade notes: a SCHEMA_VERSION bump re-seeds the four
|
||||
# built-in roles to their defaults, so role customizations are lost on this upgrade.
|
||||
# v1.11.0 (unified request/response log): bumped 11 -> 12 for the brand-new
|
||||
# `request_logs` table (ensure_request_logs_table), its retention-settings seed
|
||||
# (ensure_request_log_settings), and the new `requestlog.read` /
|
||||
# `requestlog.manage` permissions added to the built-in roles in
|
||||
# update_system_roles_to_enterprise_rbac().
|
||||
#
|
||||
# 12, NOT 11. The feature branch was cut when this constant was still 10 and
|
||||
# proposed 11, but 11 was taken in the meantime by v1.10.4 (vip_discoveries)
|
||||
# above. Landing it as 11 would be silently inert: run_all_migrations() returns
|
||||
# early on `applied_version >= SCHEMA_VERSION`, so every database already at 11
|
||||
# would skip the whole sequence and get neither the table nor the permissions,
|
||||
# while a fresh install would get both. Same class of bug the bump exists to
|
||||
# prevent, one number later.
|
||||
#
|
||||
# Additive + idempotent; no existing table is altered, agents never read this
|
||||
# table. Same caveat as v1.10.4: this bump re-seeds the four built-in roles to
|
||||
# their defaults, so export role customizations before upgrading.
|
||||
SCHEMA_VERSION = 12
|
||||
|
||||
|
||||
async def run_all_migrations():
|
||||
@@ -1902,6 +1940,12 @@ async def _run_all_migrations_inner():
|
||||
# ssl_certificates/users, both created above.
|
||||
await ensure_ssl_csrs_table()
|
||||
|
||||
# v1.11.0 — unified request/response log: brand-new request_logs table
|
||||
# (no FK targets) plus the seed for its operator-tunable retention
|
||||
# settings. Both are additive and idempotent.
|
||||
await ensure_request_logs_table()
|
||||
await ensure_request_log_settings()
|
||||
|
||||
logger.info("Database migrations completed successfully.")
|
||||
|
||||
|
||||
@@ -1973,6 +2017,162 @@ async def ensure_ssl_csrs_table():
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def ensure_request_logs_table():
|
||||
"""v1.11.0 — unified inbound/outbound request/response log.
|
||||
|
||||
Additive only: one brand-new table (request_logs) + indexes. No ALTER of
|
||||
any existing table; agents never read this table.
|
||||
|
||||
Deliberately has NO foreign key on user_id. This is the highest-volume
|
||||
table in the system — one row per API call — and per-insert FK validation
|
||||
is not worth it here; `username` is a denormalized snapshot so a row stays
|
||||
readable after the user who made the request is deleted. That is also the
|
||||
correct audit semantics: the record should outlive the account.
|
||||
|
||||
Fully idempotent (CREATE TABLE/INDEX IF NOT EXISTS). Uses only PostgreSQL
|
||||
9.5+ features (BIGSERIAL, JSONB, partial indexes, varchar_pattern_ops) so
|
||||
there is no server-version floor beyond what the rest of the schema needs.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS request_logs (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
request_id VARCHAR(64) NOT NULL,
|
||||
direction VARCHAR(8) NOT NULL,
|
||||
target VARCHAR(32),
|
||||
method VARCHAR(10) NOT NULL,
|
||||
url TEXT NOT NULL,
|
||||
path VARCHAR(512),
|
||||
query_params JSONB,
|
||||
status_code INTEGER,
|
||||
status_class SMALLINT NOT NULL DEFAULT 0,
|
||||
duration_ms INTEGER NOT NULL DEFAULT 0,
|
||||
user_id INTEGER,
|
||||
username VARCHAR(50),
|
||||
client_ip INET,
|
||||
user_agent TEXT,
|
||||
request_headers JSONB,
|
||||
request_body JSONB,
|
||||
request_body_bytes INTEGER NOT NULL DEFAULT 0,
|
||||
response_headers JSONB,
|
||||
response_body JSONB,
|
||||
response_body_bytes INTEGER NOT NULL DEFAULT 0,
|
||||
error TEXT,
|
||||
truncated BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
||||
CONSTRAINT request_logs_direction_check
|
||||
CHECK (direction IN ('inbound', 'outbound'))
|
||||
);
|
||||
""")
|
||||
|
||||
# Indexes run UNCONDITIONALLY on every startup, not only on first
|
||||
# creation (the R16-2 rule established for acme_order_events): an older
|
||||
# deploy that raced ahead of an index would otherwise be stuck doing
|
||||
# sequential scans forever. All are IF NOT EXISTS, so re-running is free.
|
||||
|
||||
# --- read paths: the filters the log viewer actually issues ---
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_created_at "
|
||||
"ON request_logs(created_at DESC);"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_dir_created "
|
||||
"ON request_logs(direction, created_at DESC);"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_status_created "
|
||||
"ON request_logs(status_class, created_at DESC);"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_user_created "
|
||||
"ON request_logs(user_id, created_at DESC) WHERE user_id IS NOT NULL;"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_target_created "
|
||||
"ON request_logs(target, created_at DESC) WHERE target IS NOT NULL;"
|
||||
)
|
||||
# Correlates one inbound row with the outbound calls it caused — this is
|
||||
# what makes "which request went where" readable as a single trace.
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_request_id "
|
||||
"ON request_logs(request_id);"
|
||||
)
|
||||
# Prefix search on path (LIKE 'x%') needs pattern_ops to be usable under
|
||||
# a non-C collation.
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_path_prefix "
|
||||
"ON request_logs(path varchar_pattern_ops);"
|
||||
)
|
||||
|
||||
# --- prune paths: the TTL delete is split by outcome, so a plain
|
||||
# (status_class, created_at) index would still range-scan the half it
|
||||
# is not interested in.
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_prune_ok "
|
||||
"ON request_logs(created_at) WHERE status_class BETWEEN 1 AND 3;"
|
||||
)
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_request_logs_prune_err "
|
||||
"ON request_logs(created_at) WHERE status_class = 0 OR status_class >= 4;"
|
||||
)
|
||||
|
||||
logger.info("request_logs table ensured (v1.11.0 request/response log)")
|
||||
except Exception as e:
|
||||
logger.error(f"Error ensuring request_logs table: {e}")
|
||||
# Re-raise (ssl_csrs precedent): this step is part of the
|
||||
# SCHEMA_VERSION=11 bump and the version marker is written only after
|
||||
# the inner sequence completes cleanly. Swallowing here would stamp
|
||||
# version 11 with no request_logs table, and the version gate would
|
||||
# then skip every future retry — permanently.
|
||||
raise
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def ensure_request_log_settings():
|
||||
"""v1.11.0 — seed the request/response-log retention defaults.
|
||||
|
||||
Runs UNCONDITIONALLY rather than inside an `if not table_exists:` branch,
|
||||
so an install that already has `system_settings` picks the rows up too.
|
||||
ON CONFLICT DO NOTHING means an operator's tuning is never overwritten by a
|
||||
later upgrade.
|
||||
|
||||
Defaults are mirrored in utils/request_log_settings.py; the pair is pinned
|
||||
by backend/tests/test_request_log_settings.py so they cannot drift apart.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
await conn.execute("""
|
||||
INSERT INTO system_settings (key, value, category, description) VALUES
|
||||
('requestlog.enabled', 'true', 'requestlog', 'Master switch for the request/response log'),
|
||||
('requestlog.capture_inbound', 'true', 'requestlog', 'Log inbound API calls'),
|
||||
('requestlog.capture_outbound', 'true', 'requestlog', 'Log outbound HTTP calls made by the backend'),
|
||||
('requestlog.capture_bodies', 'true', 'requestlog', 'Capture redacted, size-capped request/response bodies'),
|
||||
('requestlog.capture_get', 'true', 'requestlog', 'Log inbound GET requests'),
|
||||
('requestlog.capture_agent_success', 'false', 'requestlog', 'Log SUCCESSFUL agent polls too (failures are always logged); off by default because the row rate scales with fleet size, not operator activity'),
|
||||
('requestlog.max_body_bytes', '8192', 'requestlog', 'Per-body capture cap in bytes'),
|
||||
('requestlog.sample_rate', '1.0', 'requestlog', 'Sampling rate for successful inbound requests (errors always 1.0)'),
|
||||
('requestlog.exclude_paths', '["/api/request-logs","/api/health","/api/docs","/api/redoc","/api/openapi.json","/.well-known/acme-challenge","/api/agents/heartbeat","/static","/favicon.ico"]', 'requestlog', 'Path prefixes that are never logged'),
|
||||
('requestlog.success_retention_days', '7', 'requestlog', 'Retention for 1xx/2xx/3xx rows, in days'),
|
||||
('requestlog.error_retention_days', '30', 'requestlog', 'Retention for 4xx/5xx/transport-error rows, in days'),
|
||||
('requestlog.max_rows', '500000', 'requestlog', 'Hard row cap; oldest rows are pruned beyond this'),
|
||||
('requestlog.prune_interval_minutes', '60', 'requestlog', 'Minimum interval between retention prune passes')
|
||||
ON CONFLICT (key) DO NOTHING
|
||||
""")
|
||||
logger.info("request_log retention settings seeded (v1.11.0)")
|
||||
except Exception as e:
|
||||
logger.error(f"Error seeding request_log settings: {e}")
|
||||
raise
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
async def ensure_mfa_columns():
|
||||
"""Issue #18 — TOTP MFA (v1.6.0): additive columns on users + 3 new tables.
|
||||
|
||||
@@ -2161,6 +2361,45 @@ async def ensure_vip_tables():
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_members_agent ON vip_members(agent_id);"
|
||||
)
|
||||
|
||||
# ── v1.10.4 — VIP adoption: what the agent found already on the node ──────────
|
||||
# A node with a hand-maintained keepalived.conf reports it here so an existing VIP can
|
||||
# be adopted instead of retyped. One row per agent (the file is per-node); the agent
|
||||
# only reports a config it does NOT own, and only when the content changed.
|
||||
#
|
||||
# SECRETS: `raw_config` is stored MASKED (auth_pass replaced) because it is served to
|
||||
# the UI. The real VRRP password is Fernet-encrypted in auth_pass_encrypted, mirroring
|
||||
# vip_instances, so adoption can carry it into the managed VIP without it ever being
|
||||
# readable through the API or a DB dump. `analysis` is the parser output with auth_pass
|
||||
# stripped out.
|
||||
await conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS vip_discoveries (
|
||||
id SERIAL PRIMARY KEY,
|
||||
agent_id INTEGER NOT NULL REFERENCES agents(id) ON DELETE CASCADE,
|
||||
config_path VARCHAR(500) NOT NULL,
|
||||
config_hash VARCHAR(64) NOT NULL,
|
||||
is_managed BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
raw_config_masked TEXT,
|
||||
auth_pass_encrypted TEXT,
|
||||
analysis JSONB,
|
||||
parse_error TEXT,
|
||||
adopted_vip_id INTEGER REFERENCES vip_instances(id) ON DELETE SET NULL,
|
||||
reported_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
CONSTRAINT vip_discovery_agent_unique UNIQUE (agent_id)
|
||||
);
|
||||
""")
|
||||
await conn.execute(
|
||||
"CREATE INDEX IF NOT EXISTS idx_vip_discoveries_agent ON vip_discoveries(agent_id);"
|
||||
)
|
||||
# Adoption provenance + the one-shot takeover authorisation. The agent refuses to
|
||||
# overwrite a keepalived.conf that lacks our ownership marker, which is exactly the
|
||||
# guard adoption has to pass. Rather than weaken it, an adopted VIP carries the hash of
|
||||
# the file we analysed: the agent takes over ONLY if the file on disk still hashes to
|
||||
# that value, so a config that changed after adoption is never clobbered.
|
||||
await conn.execute(
|
||||
"ALTER TABLE vip_instances ADD COLUMN IF NOT EXISTS adopted_at TIMESTAMP;")
|
||||
await conn.execute(
|
||||
"ALTER TABLE vip_members ADD COLUMN IF NOT EXISTS takeover_expected_hash VARCHAR(64);")
|
||||
|
||||
logger.info("✅ VIP tables ensured (Issue #27 — HA/VIP Keepalived management)")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to ensure VIP tables: {e}")
|
||||
@@ -2303,19 +2542,19 @@ async def create_initial_system_data(conn):
|
||||
'name': 'super_admin',
|
||||
'display_name': 'Super Administrator',
|
||||
'description': 'Full system access with all permissions',
|
||||
'permissions': ["dashboard.read","dashboard.statistics","frontends.read","frontends.create","frontends.update","frontends.delete","backends.read","backends.create","backends.update","backends.delete","waf.read","waf.create","waf.update","waf.delete","ssl.read","ssl.create","ssl.update","ssl.delete","apply.read","apply.execute","agents.read","agents.create","agents.update","agents.delete","clusters.read","clusters.create","clusters.update","clusters.delete","config.read","config.update","config.bulk_import","config.view_request","config.download_request","users.read","users.create","users.update","users.delete","roles.read","roles.create","roles.update","roles.delete"]
|
||||
'permissions': ["dashboard.read","dashboard.statistics","frontends.read","frontends.create","frontends.update","frontends.delete","backends.read","backends.create","backends.update","backends.delete","waf.read","waf.create","waf.update","waf.delete","ssl.read","ssl.create","ssl.update","ssl.delete","apply.read","apply.execute","agents.read","agents.create","agents.update","agents.delete","clusters.read","clusters.create","clusters.update","clusters.delete","config.read","config.update","config.bulk_import","config.view_request","config.download_request","users.read","users.create","users.update","users.delete","roles.read","roles.create","roles.update","roles.delete","requestlog.read","requestlog.manage"]
|
||||
},
|
||||
{
|
||||
'name': 'operator',
|
||||
'display_name': 'Operator',
|
||||
'description': 'Daily operational access for managing HAProxy configurations',
|
||||
'permissions': ["dashboard.read","dashboard.statistics","frontends.read","frontends.create","frontends.update","backends.read","backends.create","backends.update","waf.read","waf.create","waf.update","ssl.read","ssl.create","ssl.update","apply.read","apply.execute","agents.read","clusters.read","config.read","config.update","config.bulk_import","config.view_request","config.download_request"]
|
||||
'permissions': ["dashboard.read","dashboard.statistics","frontends.read","frontends.create","frontends.update","backends.read","backends.create","backends.update","waf.read","waf.create","waf.update","ssl.read","ssl.create","ssl.update","apply.read","apply.execute","agents.read","clusters.read","config.read","config.update","config.bulk_import","config.view_request","config.download_request","requestlog.read"]
|
||||
},
|
||||
{
|
||||
'name': 'security_admin',
|
||||
'display_name': 'Security Administrator',
|
||||
'description': 'Security-focused access for WAF rules and SSL certificates',
|
||||
'permissions': ["dashboard.read","frontends.read","backends.read","waf.read","waf.create","waf.update","waf.delete","ssl.read","ssl.create","ssl.update","ssl.delete","apply.read","apply.execute","agents.read","clusters.read","config.read","config.view_request","config.download_request"]
|
||||
'permissions': ["dashboard.read","frontends.read","backends.read","waf.read","waf.create","waf.update","waf.delete","ssl.read","ssl.create","ssl.update","ssl.delete","apply.read","apply.execute","agents.read","clusters.read","config.read","config.view_request","config.download_request","requestlog.read","requestlog.manage"]
|
||||
},
|
||||
{
|
||||
'name': 'viewer',
|
||||
|
||||
@@ -61,14 +61,27 @@ class HAProxyClient:
|
||||
if self.stats_username and self.stats_password:
|
||||
auth = aiohttp.BasicAuth(self.stats_username, self.stats_password)
|
||||
|
||||
# v1.11.0: instrumented for completeness. NOTE the CSV body is
|
||||
# deliberately NOT handed to the span — a full stats dump is large,
|
||||
# changes every poll, and has no diagnostic value in an audit row;
|
||||
# status + duration is what matters. `auth` is likewise never logged:
|
||||
# aiohttp.BasicAuth is a NamedTuple whose repr contains the cleartext
|
||||
# password.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_HAPROXY_STATS
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(url, auth=auth, timeout=aiohttp.ClientTimeout(total=10)) as response:
|
||||
if response.status == 200:
|
||||
csv_data = await response.text()
|
||||
return self._parse_csv_stats(csv_data)
|
||||
else:
|
||||
logger.warning(f"HTTP stats request failed with status {response.status}")
|
||||
return self._get_fallback_stats()
|
||||
async with outbound_span(
|
||||
target=TARGET_HAPROXY_STATS, method="GET", url=url,
|
||||
capture_body=False, capture_response_body=False,
|
||||
) as span:
|
||||
async with session.get(url, auth=auth, timeout=aiohttp.ClientTimeout(total=10)) as response:
|
||||
span.set_response(response.status, getattr(response, "headers", None))
|
||||
if response.status == 200:
|
||||
csv_data = await response.text()
|
||||
return self._parse_csv_stats(csv_data)
|
||||
else:
|
||||
logger.warning(f"HTTP stats request failed with status {response.status}")
|
||||
return self._get_fallback_stats()
|
||||
except Exception as e:
|
||||
logger.error(f"HTTP stats request failed: {e}")
|
||||
return self._get_fallback_stats()
|
||||
|
||||
+104
-4
@@ -29,7 +29,7 @@ for _vpath in [os.path.join(os.path.dirname(__file__), "version.json"), "/app/ve
|
||||
|
||||
# Import configurations and database
|
||||
|
||||
from config import CORS_ORIGINS, REDIS_URL, LOG_LEVEL
|
||||
from config import CORS_ORIGINS, REDIS_URL, LOG_LEVEL, REQUEST_LOG_ENABLED
|
||||
from database.connection import redis_client, get_database_connection, close_database_connection, init_database_pool, close_database_pool
|
||||
from database.migrations import run_all_migrations
|
||||
|
||||
@@ -48,6 +48,7 @@ from routers.site_wizard import router as site_wizard_router
|
||||
from routers.mfa import router as mfa_router
|
||||
from routers.vip import router as vip_router # Issue #27 — HA/VIP (Keepalived) management
|
||||
from routers.csr import router as csr_router # v1.9.0 — CSR creation (in-app key+CSR generation, signed-cert import)
|
||||
from routers.request_logs import router as request_logs_router # v1.11.0 — unified request/response log
|
||||
|
||||
# Production logging configuration
|
||||
from utils.logging_config import setup_production_logging
|
||||
@@ -56,6 +57,10 @@ from middleware.error_handler import (
|
||||
GlobalExceptionHandler, get_error_statistics
|
||||
)
|
||||
from middleware.activity_logger import log_activity_middleware
|
||||
from middleware.request_logger import RequestResponseLogMiddleware # v1.11.0
|
||||
from utils.http_instrumentation import begin_background_trace # v1.11.0
|
||||
from utils.request_log_settings import refresh_config as refresh_request_log_config
|
||||
from utils.request_log_sink import request_log_sink
|
||||
|
||||
# Setup structured logging
|
||||
logger = setup_production_logging(LOG_LEVEL)
|
||||
@@ -232,6 +237,8 @@ Most operations are **cluster-scoped**:
|
||||
async def monitor_agent_status():
|
||||
"""Background task to monitor agent status and mark offline agents"""
|
||||
while True:
|
||||
# v1.11.0: see complete_pending_acme_orders — one id per tick.
|
||||
begin_background_trace("agent_status_monitor")
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
@@ -276,6 +283,10 @@ async def complete_pending_acme_orders():
|
||||
"""
|
||||
await asyncio.sleep(60)
|
||||
while True:
|
||||
# v1.11.0: one correlation id per TICK, so the outbound rows for this
|
||||
# pass group together and do not merge with every other pass this
|
||||
# process has ever run.
|
||||
begin_background_trace("acme_complete_orders")
|
||||
try:
|
||||
conn_check = await get_database_connection()
|
||||
try:
|
||||
@@ -645,6 +656,8 @@ async def check_letsencrypt_renewals():
|
||||
"""
|
||||
await asyncio.sleep(120)
|
||||
while True:
|
||||
# v1.11.0: see complete_pending_acme_orders — one id per tick.
|
||||
begin_background_trace("acme_renewals")
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
@@ -842,6 +855,41 @@ async def cleanup_stuck_agent_upgrades():
|
||||
# Wait 120 seconds (2 minutes) before next check
|
||||
await asyncio.sleep(120)
|
||||
|
||||
async def prune_request_logs_loop():
|
||||
"""v1.11.0 — retention prune for `request_logs`.
|
||||
|
||||
Kept independent of the ACME prune loop on purpose: that one is gated on
|
||||
the `letsencrypt_orders` table existing, which would silently disable this
|
||||
prune on an install that never uses ACME.
|
||||
|
||||
The 5-minute tick is only a heartbeat — the real gate is the DB watermark
|
||||
plus `requestlog.prune_interval_minutes`, so N replicas ticking every 5
|
||||
minutes still produce one pass per configured interval.
|
||||
"""
|
||||
# Stagger past startup so migrations and the first request burst are done.
|
||||
await asyncio.sleep(180)
|
||||
|
||||
while True:
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
table_exists = await conn.fetchval("""
|
||||
SELECT EXISTS (
|
||||
SELECT 1 FROM information_schema.tables
|
||||
WHERE table_name = 'request_logs'
|
||||
)
|
||||
""")
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
if table_exists:
|
||||
from utils.request_log_prune import prune_request_logs_if_due
|
||||
await prune_request_logs_if_due()
|
||||
except Exception as e:
|
||||
logger.error(f"Error in request_logs prune loop: {e}")
|
||||
|
||||
await asyncio.sleep(300)
|
||||
|
||||
# Production middleware stack (order matters!)
|
||||
app.add_middleware(PerformanceMonitoringMiddleware, slow_request_threshold_ms=1000)
|
||||
app.add_middleware(RequestLoggingMiddleware, exclude_paths=["/api/health/", "/docs", "/redoc"])
|
||||
@@ -849,15 +897,35 @@ app.add_middleware(RequestLoggingMiddleware, exclude_paths=["/api/health/", "/do
|
||||
# Activity logging middleware - must be before CORS
|
||||
app.middleware("http")(log_activity_middleware)
|
||||
|
||||
# CORS middleware
|
||||
# CORS middleware
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=CORS_ORIGINS,
|
||||
allow_credentials=True,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
# v1.11.0: without an explicit expose list, browser JS on a cross-origin
|
||||
# deployment cannot read ANY of these — so an operator could see the
|
||||
# X-Request-ID in devtools but the app could never quote it back. Same-origin
|
||||
# (nginx) deployments were already fine; this fixes the split-origin case.
|
||||
expose_headers=["X-Correlation-ID", "X-Response-Time", "X-Request-ID"],
|
||||
)
|
||||
|
||||
# v1.11.0 — unified request/response log.
|
||||
#
|
||||
# MUST be the LAST add_middleware call: Starlette inserts each new middleware at
|
||||
# index 0, so the last registration ends up OUTERMOST. Outermost is what we want:
|
||||
# (a) we see the exact status/headers/body the client receives, including the
|
||||
# JSONResponse that RequestLoggingMiddleware fabricates from an exception
|
||||
# it swallowed, and
|
||||
# (b) we seed correlation_id_context BEFORE RequestLoggingMiddleware calls
|
||||
# get_correlation_id(), so X-Correlation-ID matches request_logs.request_id.
|
||||
#
|
||||
# REQUEST_LOG_ENABLED=false keeps it out of the ASGI stack entirely — not a
|
||||
# runtime branch, genuinely zero overhead.
|
||||
if REQUEST_LOG_ENABLED:
|
||||
app.add_middleware(RequestResponseLogMiddleware)
|
||||
|
||||
# Global exception handlers
|
||||
from fastapi.exceptions import RequestValidationError
|
||||
from starlette.exceptions import HTTPException as StarletteHTTPException
|
||||
@@ -894,6 +962,7 @@ app.include_router(agent_router)
|
||||
app.include_router(waf_router)
|
||||
app.include_router(ssl_router)
|
||||
app.include_router(csr_router) # v1.9.0: CSR creation (in-app key+CSR generation, signed-cert import)
|
||||
app.include_router(request_logs_router) # v1.11.0: unified request/response log
|
||||
app.include_router(security_router)
|
||||
app.include_router(configuration_router)
|
||||
app.include_router(settings_router)
|
||||
@@ -1031,6 +1100,17 @@ async def startup_event():
|
||||
# Decoupled from auto_renew_enabled flag so user-initiated orders also complete.
|
||||
asyncio.create_task(complete_pending_acme_orders())
|
||||
logger.info("ACME order auto-completion task started (60s checks, replica-safe)")
|
||||
|
||||
# v1.11.0 — request/response log: load the operator's capture/retention
|
||||
# policy, then start the batching writer and the retention prune.
|
||||
# Guarded by the env kill-switch so a deployment that turned the log off
|
||||
# pays for neither task.
|
||||
if REQUEST_LOG_ENABLED:
|
||||
await refresh_request_log_config()
|
||||
asyncio.create_task(request_log_sink.run())
|
||||
logger.info("Request/response log sink started (batching writer)")
|
||||
asyncio.create_task(prune_request_logs_loop())
|
||||
logger.info("Request/response log retention prune task started")
|
||||
|
||||
# Create test activity log entry to verify system is working
|
||||
try:
|
||||
@@ -1059,6 +1139,16 @@ async def shutdown_event():
|
||||
"""Cleanup on shutdown"""
|
||||
logger.info("HAProxy OpenManager API shutting down...")
|
||||
|
||||
# v1.11.0: flush queued request-log rows FIRST. The sink's writer is a
|
||||
# `while True` loop, so it can never satisfy the asyncio.wait below — the
|
||||
# rows still sitting in its queue would be lost when the pool closes.
|
||||
try:
|
||||
flushed = await request_log_sink.flush(timeout=3.0)
|
||||
if flushed:
|
||||
logger.info(f"Flushed {flushed} queued request-log row(s)")
|
||||
except Exception as flush_err:
|
||||
logger.warning(f"request-log flush skipped: {flush_err}")
|
||||
|
||||
# R18c audit fix (round 3 #5): drain pending fire-and-forget
|
||||
# background tasks BEFORE closing the DB pool. The audit
|
||||
# logger middleware (`activity_logger.py`) and the wizard
|
||||
@@ -1111,9 +1201,19 @@ async def get_version():
|
||||
return _version_info
|
||||
|
||||
@app.get("/.well-known/acme-challenge/{token}")
|
||||
async def serve_acme_challenge(token: str):
|
||||
async def serve_acme_challenge(token: str, request: Request):
|
||||
"""Serve ACME HTTP-01 challenge token. Public endpoint, no auth required."""
|
||||
logger.info(f"ACME-CHALLENGE: Incoming request for token={token[:32]}...")
|
||||
# Log who reached us. When HTTP-01 fails, the first question is always "did the
|
||||
# request get here at all?" — and the answer separates a broken challenge-backend
|
||||
# address (nothing arrives) from a wrong response (arrives, wrong body). The peer
|
||||
# is normally the HAProxy node; X-Forwarded-For carries the CA when the frontend
|
||||
# sets `option forwardfor`.
|
||||
_peer = request.client.host if request.client else 'unknown'
|
||||
_xff = request.headers.get('x-forwarded-for') or '-'
|
||||
logger.info(
|
||||
f"ACME-CHALLENGE: Incoming request for token={token[:32]}... "
|
||||
f"peer={_peer} xff={_xff} host={request.headers.get('host') or '-'}"
|
||||
)
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
@@ -0,0 +1,321 @@
|
||||
"""v1.11.0 — inbound half of the unified request/response log.
|
||||
|
||||
Pure ASGI on purpose, NOT BaseHTTPMiddleware:
|
||||
|
||||
* `BaseHTTPMiddleware` hands the response back as a
|
||||
`starlette.middleware.base._StreamingResponse`, which has no `.body` to
|
||||
read, and
|
||||
* `await request.body()` inside a `dispatch()` DRAINS the receive channel.
|
||||
`POST /api/agents/heartbeat` (routers/agent.py) reads the raw stream
|
||||
itself, as does the validation-error body preview in
|
||||
middleware/error_handler.py — draining it here would break both.
|
||||
|
||||
So we never consume anything: we TEE. `receive` and `send` are wrapped, every
|
||||
message is forwarded verbatim, and a size-capped copy is kept for the log row.
|
||||
Cost per in-flight request is therefore bounded at ~2 × max_body_bytes (8 KB
|
||||
by default), not the size of the upload.
|
||||
|
||||
Registration: this MUST be the LAST `app.add_middleware(...)` call, because
|
||||
Starlette inserts at index 0 — the last registration is the OUTERMOST
|
||||
middleware. Outermost is what we want: we then see the exact status and body
|
||||
the client receives (including the JSONResponse that RequestLoggingMiddleware
|
||||
fabricates out of a swallowed exception), and we can seed
|
||||
`correlation_id_context` before anything downstream reads it.
|
||||
"""
|
||||
import logging
|
||||
import time
|
||||
import uuid
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from starlette.types import ASGIApp, Receive, Scope, Send
|
||||
|
||||
from utils.logging_config import correlation_id_context
|
||||
from utils.request_log_redaction import is_capturable_content_type, scrub_query_string
|
||||
from utils.request_log_settings import get_config
|
||||
from utils.request_log_sink import (
|
||||
TARGET_INBOUND_AGENT,
|
||||
RequestLogRow,
|
||||
request_id_context,
|
||||
request_log_sink,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
# Hard floor, NOT settable away through `requestlog.exclude_paths`. Without it
|
||||
# an operator who clears the exclude list turns the log viewer into a machine
|
||||
# that logs itself reading its own logs.
|
||||
_ALWAYS_EXCLUDED: Tuple[str, ...] = ("/api/request-logs",)
|
||||
|
||||
|
||||
def _header(scope: Scope, name: bytes) -> Optional[str]:
|
||||
for key, value in scope.get("headers") or ():
|
||||
if key == name:
|
||||
try:
|
||||
return value.decode("latin-1")
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _identify(scope: Scope) -> Tuple[Optional[int], Optional[str]]:
|
||||
"""Resolve the caller from the JWT locally — NO database round-trip.
|
||||
|
||||
`log_activity_middleware` already pays a `SELECT ... FROM users` per
|
||||
non-GET request; this middleware runs on every request including GETs, so a
|
||||
second lookup per call is not acceptable. The token issued at
|
||||
routers/auth.py carries both `user_id` and `username`, which is everything
|
||||
the log row needs.
|
||||
|
||||
A token that fails to decode simply yields (None, None): this is a logging
|
||||
path, not an authorization path — the real auth check still runs
|
||||
downstream.
|
||||
"""
|
||||
raw = _header(scope, b"authorization")
|
||||
if not raw:
|
||||
return None, None
|
||||
token = raw[7:].strip() if raw.lower().startswith("bearer ") else raw.strip()
|
||||
if not token or token in ("null", "undefined") or token.count(".") != 2:
|
||||
return None, None
|
||||
try:
|
||||
from jose import jwt
|
||||
from config import JWT_SECRET_KEY, JWT_ALGORITHM
|
||||
|
||||
payload = jwt.decode(token, JWT_SECRET_KEY, algorithms=[JWT_ALGORITHM])
|
||||
except Exception:
|
||||
return None, None
|
||||
|
||||
raw_uid = payload.get("user_id") or payload.get("sub")
|
||||
try:
|
||||
user_id = int(raw_uid) if raw_uid is not None else None
|
||||
except (TypeError, ValueError):
|
||||
user_id = None
|
||||
username = payload.get("username")
|
||||
return user_id, (str(username) if username else None)
|
||||
|
||||
|
||||
def _is_agent_call(scope: Scope) -> bool:
|
||||
"""True when the caller authenticated as an AGENT rather than as a user.
|
||||
|
||||
Every call the installed agent makes carries `X-API-Key` and never an
|
||||
`Authorization` header (linux_install.sh / macos_install.sh: heartbeat,
|
||||
config, pending-requests, upgrade-status, keepalived-*, config-response are
|
||||
all `-H "X-API-Key: $AGENT_TOKEN"`), while the UI carries a JWT and never an
|
||||
agent key. The one endpoint that accepts either -
|
||||
`POST /api/agents/generate-install-script`, used by agent self-upgrade -
|
||||
is correctly classified by the same rule: an operator generating a script
|
||||
sends Authorization, the self-upgrading agent sends only the key.
|
||||
|
||||
Header-only, so it costs two scope reads and no database round-trip.
|
||||
"""
|
||||
if _header(scope, b"authorization"):
|
||||
return False
|
||||
return bool(_header(scope, b"x-api-key"))
|
||||
|
||||
|
||||
def _client_ip(scope: Scope) -> Optional[str]:
|
||||
"""The peer address only.
|
||||
|
||||
`request_logs.client_ip` is an INET column, so a comma-joined
|
||||
X-Forwarded-For string would raise on INSERT (the same trap as
|
||||
`user_activity_logs.ip_address`). The XFF header is still captured — it is
|
||||
on the header allowlist — so the original client is not lost behind a proxy.
|
||||
"""
|
||||
client = scope.get("client")
|
||||
if not client:
|
||||
return None
|
||||
try:
|
||||
return str(client[0])
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
class RequestResponseLogMiddleware:
|
||||
def __init__(self, app: ASGIApp):
|
||||
self.app = app
|
||||
|
||||
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
|
||||
if scope.get("type") != "http":
|
||||
await self.app(scope, receive, send)
|
||||
return
|
||||
|
||||
cfg = get_config()
|
||||
path = scope.get("path", "") or ""
|
||||
method = scope.get("method", "") or ""
|
||||
|
||||
if (
|
||||
not cfg.enabled
|
||||
or not cfg.capture_inbound
|
||||
# OPTIONS never reaches a handler — CORSMiddleware short-circuits
|
||||
# it below us — and a preflight carries no information worth a row.
|
||||
or method == "OPTIONS"
|
||||
or (method == "GET" and not cfg.capture_get)
|
||||
or any(path.startswith(prefix) for prefix in _ALWAYS_EXCLUDED)
|
||||
or any(path.startswith(prefix) for prefix in cfg.exclude_paths)
|
||||
):
|
||||
await self.app(scope, receive, send)
|
||||
return
|
||||
|
||||
request_id = uuid.uuid4().hex
|
||||
# Seed the id BEFORE the downstream app runs so error_handler's
|
||||
# get_correlation_id() adopts ours instead of minting a second one; the
|
||||
# X-Correlation-ID header then matches request_logs.request_id.
|
||||
cid_token = correlation_id_context.set(request_id[:8])
|
||||
rid_token = request_id_context.set(request_id)
|
||||
|
||||
cap = cfg.max_body_bytes if cfg.capture_bodies else 0
|
||||
req_ctype = _header(scope, b"content-type")
|
||||
req_capturable = is_capturable_content_type(req_ctype)
|
||||
|
||||
req_buf = bytearray()
|
||||
res_buf = bytearray()
|
||||
state = {
|
||||
"req_bytes": 0,
|
||||
"res_bytes": 0,
|
||||
"status": None,
|
||||
"res_headers": {},
|
||||
"res_ctype": None,
|
||||
"res_capturable": True,
|
||||
}
|
||||
|
||||
async def tee_receive() -> Dict[str, Any]:
|
||||
message = await receive()
|
||||
try:
|
||||
if message.get("type") == "http.request":
|
||||
chunk = message.get("body", b"") or b""
|
||||
state["req_bytes"] += len(chunk)
|
||||
if cap and req_capturable and len(req_buf) < cap:
|
||||
req_buf.extend(chunk[: cap - len(req_buf)])
|
||||
except Exception:
|
||||
pass
|
||||
return message # forwarded verbatim, always
|
||||
|
||||
async def tee_send(message: Dict[str, Any]) -> None:
|
||||
try:
|
||||
mtype = message.get("type")
|
||||
if mtype == "http.response.start":
|
||||
state["status"] = message.get("status")
|
||||
raw_headers: List[Tuple[bytes, bytes]] = message.get("headers") or []
|
||||
headers = {}
|
||||
for key, value in raw_headers:
|
||||
try:
|
||||
headers[key.decode("latin-1").lower()] = value.decode("latin-1")
|
||||
except Exception:
|
||||
continue
|
||||
state["res_headers"] = headers
|
||||
state["res_ctype"] = headers.get("content-type")
|
||||
state["res_capturable"] = is_capturable_content_type(state["res_ctype"])
|
||||
# Hand the id to the client so a user reporting a problem can
|
||||
# quote it and an operator can find the exact row.
|
||||
if isinstance(raw_headers, list):
|
||||
raw_headers.append((b"x-request-id", request_id.encode("latin-1")))
|
||||
elif mtype == "http.response.body":
|
||||
chunk = message.get("body", b"") or b""
|
||||
state["res_bytes"] += len(chunk)
|
||||
if cap and state["res_capturable"] and len(res_buf) < cap:
|
||||
res_buf.extend(chunk[: cap - len(res_buf)])
|
||||
except Exception:
|
||||
pass
|
||||
await send(message) # forwarded verbatim, always
|
||||
|
||||
started = time.perf_counter()
|
||||
error_text: Optional[str] = None
|
||||
try:
|
||||
await self.app(scope, tee_receive, tee_send)
|
||||
except Exception as exc:
|
||||
# Almost never taken: RequestLoggingMiddleware sits below us and
|
||||
# converts exceptions into a JSONResponse first. It IS taken for
|
||||
# paths on that middleware's own exclude list, so the row still has
|
||||
# to be recorded before the exception continues upward.
|
||||
error_text = f"{type(exc).__name__}: {exc}"[:2000]
|
||||
raise
|
||||
finally:
|
||||
duration_ms = int((time.perf_counter() - started) * 1000)
|
||||
try:
|
||||
self._record(
|
||||
scope=scope,
|
||||
request_id=request_id,
|
||||
method=method,
|
||||
path=path,
|
||||
duration_ms=duration_ms,
|
||||
status=state["status"],
|
||||
req_buf=bytes(req_buf),
|
||||
req_bytes=state["req_bytes"],
|
||||
req_ctype=req_ctype,
|
||||
res_buf=bytes(res_buf),
|
||||
res_bytes=state["res_bytes"],
|
||||
res_ctype=state["res_ctype"],
|
||||
res_headers=state["res_headers"],
|
||||
error_text=error_text,
|
||||
)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"request_log: failed to record inbound row: {exc}")
|
||||
try:
|
||||
correlation_id_context.reset(cid_token)
|
||||
request_id_context.reset(rid_token)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@staticmethod
|
||||
def _record(
|
||||
*,
|
||||
scope: Scope,
|
||||
request_id: str,
|
||||
method: str,
|
||||
path: str,
|
||||
duration_ms: int,
|
||||
status: Optional[int],
|
||||
req_buf: bytes,
|
||||
req_bytes: int,
|
||||
req_ctype: Optional[str],
|
||||
res_buf: bytes,
|
||||
res_bytes: int,
|
||||
res_ctype: Optional[str],
|
||||
res_headers: Dict[str, str],
|
||||
error_text: Optional[str],
|
||||
) -> None:
|
||||
raw_query = scope.get("query_string") or b""
|
||||
try:
|
||||
query = raw_query.decode("latin-1")
|
||||
except Exception:
|
||||
query = ""
|
||||
scrubbed_query, query_params = scrub_query_string(query)
|
||||
|
||||
user_id, username = _identify(scope)
|
||||
|
||||
req_headers = {}
|
||||
for key, value in scope.get("headers") or ():
|
||||
try:
|
||||
req_headers[key.decode("latin-1").lower()] = value.decode("latin-1")
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
request_log_sink.offer(
|
||||
RequestLogRow(
|
||||
request_id=request_id,
|
||||
direction="inbound",
|
||||
# Who was on the other end. `offer()` uses this to drop
|
||||
# SUCCESSFUL agent polls, which are ~9 800 rows/day per node and
|
||||
# would otherwise make the table's size a function of fleet size.
|
||||
target=TARGET_INBOUND_AGENT if _is_agent_call(scope) else None,
|
||||
method=method,
|
||||
url=path + (("?" + scrubbed_query) if scrubbed_query else ""),
|
||||
path=path,
|
||||
query_string=scrubbed_query or None,
|
||||
query_params=query_params,
|
||||
status_code=status,
|
||||
duration_ms=duration_ms,
|
||||
user_id=user_id,
|
||||
username=username,
|
||||
client_ip=_client_ip(scope),
|
||||
user_agent=req_headers.get("user-agent"),
|
||||
request_headers=req_headers or None,
|
||||
response_headers=res_headers or None,
|
||||
request_body_raw=req_buf or None,
|
||||
request_body_bytes=req_bytes,
|
||||
request_content_type=req_ctype,
|
||||
response_body_raw=res_buf or None,
|
||||
response_body_bytes=res_bytes,
|
||||
response_content_type=res_ctype,
|
||||
error=error_text,
|
||||
)
|
||||
)
|
||||
@@ -1,6 +1,22 @@
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, field_validator
|
||||
from typing import Optional, List
|
||||
|
||||
from utils.acme_backend_url import AcmeBackendUrlError, validate_acme_backend_url
|
||||
|
||||
|
||||
def _validated_acme_backend_url(value: Optional[str]) -> Optional[str]:
|
||||
"""Shared field validator body for `acme_backend_url`.
|
||||
|
||||
Pydantic turns the raised ValueError into a 422 with this message attached, so
|
||||
the operator sees why the value was refused instead of discovering months later
|
||||
that HTTP-01 never worked. Returns the normalised value — callers persist THIS,
|
||||
not the raw input, so surrounding whitespace never reaches haproxy.cfg.
|
||||
"""
|
||||
try:
|
||||
return validate_acme_backend_url(value)
|
||||
except AcmeBackendUrlError as exc:
|
||||
raise ValueError(str(exc)) from None
|
||||
|
||||
class HAProxyClusterCreate(BaseModel):
|
||||
name: str
|
||||
description: Optional[str] = None
|
||||
@@ -10,6 +26,16 @@ class HAProxyClusterCreate(BaseModel):
|
||||
haproxy_bin_path: str = "/usr/sbin/haproxy" # HAProxy binary path
|
||||
keepalived_config_path: str = "/etc/keepalived/keepalived.conf" # HA/VIP: keepalived.conf path (Issue #27)
|
||||
pool_id: Optional[int] = None # Which pool this cluster belongs to
|
||||
# The create form submits both of these. Until they were declared here pydantic
|
||||
# dropped them and the INSERT never carried them, so a cluster created with ACME
|
||||
# switched on came back switched off with no error shown — the same silent-success
|
||||
# failure this work exists to remove.
|
||||
acme_enabled: Optional[bool] = None
|
||||
acme_backend_url: Optional[str] = None
|
||||
|
||||
_validate_acme_backend_url = field_validator("acme_backend_url")(
|
||||
_validated_acme_backend_url
|
||||
)
|
||||
|
||||
class HAProxyClusterUpdate(BaseModel):
|
||||
name: Optional[str] = None
|
||||
@@ -24,6 +50,10 @@ class HAProxyClusterUpdate(BaseModel):
|
||||
acme_enabled: Optional[bool] = None
|
||||
acme_backend_url: Optional[str] = None
|
||||
|
||||
_validate_acme_backend_url = field_validator("acme_backend_url")(
|
||||
_validated_acme_backend_url
|
||||
)
|
||||
|
||||
class HAProxyClusterResponse(BaseModel):
|
||||
id: int
|
||||
name: str
|
||||
|
||||
+169
-14
@@ -9,8 +9,15 @@ import os
|
||||
import json
|
||||
import ipaddress
|
||||
import hashlib
|
||||
import re
|
||||
# Pipeline trigger - force backend redeploy v2
|
||||
|
||||
# v1.10.4 — a discovered keepalived.conf is stored and served to the UI, so the VRRP password is
|
||||
# masked out of the stored copy (the real value lives Fernet-encrypted in its own column). Mask
|
||||
# the WHOLE remainder of the line, mirroring vip.py's version-diff masking, so a password
|
||||
# containing whitespace cannot partially leak.
|
||||
_AUTH_PASS_MASK_RE = re.compile(r"(auth_pass\s+).*")
|
||||
|
||||
from models import AgentCreate
|
||||
from models.agent import AgentToggle, AgentHeartbeat, AgentScriptRequest, AgentUpgradeRequest
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
@@ -26,7 +33,7 @@ logger = logging.getLogger(__name__)
|
||||
# Global version storage (acts as in-memory database)
|
||||
AGENT_VERSIONS = {
|
||||
"macos": "2.1.0", # Updated via endpoint
|
||||
"linux": "2.0.0"
|
||||
"linux": "2.1.0"
|
||||
}
|
||||
|
||||
|
||||
@@ -2318,21 +2325,46 @@ async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str]
|
||||
# to write/own-marker-check even on not_configured/teardown.
|
||||
agent = await conn.fetchrow("""
|
||||
SELECT a.id, a.name, COALESCE(a.enabled, TRUE) AS enabled,
|
||||
hc.keepalived_config_path
|
||||
hc.keepalived_config_path,
|
||||
-- v1.11.1: does the server already hold a discovery for this node? The agent
|
||||
-- caches the hash of its last discovery report next to the config and skips
|
||||
-- re-posting while it matches. That cache used to be written even when the
|
||||
-- POST was REJECTED, so a node could be hidden from the adoption panel for
|
||||
-- good: the file never changes, so the agent never speaks again. Telling it
|
||||
-- what we actually hold lets it recover on its own, with no extra request and
|
||||
-- no one having to touch the node.
|
||||
EXISTS (SELECT 1 FROM vip_discoveries vd WHERE vd.agent_id = a.id)
|
||||
AS discovery_known
|
||||
FROM agents a
|
||||
LEFT JOIN haproxy_clusters hc ON hc.pool_id = a.pool_id
|
||||
WHERE a.name = $1
|
||||
-- A pool may hold more than one cluster, and the join then multiplies this row. With
|
||||
-- no ordering the fetch took an arbitrary one, so the keepalived.conf PATH handed to
|
||||
-- the agent was non-deterministic whenever two clusters in a pool disagreed on it:
|
||||
-- the agent would look at the wrong file, find nothing there, and the node would
|
||||
-- never appear for adoption — intermittently, which is the worst way to fail.
|
||||
--
|
||||
-- A CUSTOMISED path wins over the shipped default, then the lowest cluster id. The
|
||||
-- column defaults to '/etc/keepalived/keepalived.conf' rather than NULL, so ordering
|
||||
-- by id alone could have picked a default-valued row over one the operator had
|
||||
-- deliberately set — turning "undefined" into "reliably wrong" for that install.
|
||||
-- When every cluster in the pool carries the default the string is identical, so the
|
||||
-- ordering cannot change what any working deployment already receives.
|
||||
ORDER BY (hc.keepalived_config_path IS NULL
|
||||
OR hc.keepalived_config_path = '/etc/keepalived/keepalived.conf'),
|
||||
hc.id
|
||||
LIMIT 1
|
||||
""", agent_name)
|
||||
if not agent:
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
config_path = agent['keepalived_config_path'] or '/etc/keepalived/keepalived.conf'
|
||||
if not agent['enabled']:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "discovery_known": bool(agent["discovery_known"]), "keepalived": None}
|
||||
|
||||
row = await conn.fetchrow("""
|
||||
SELECT v.id AS vip_id, v.name AS vip_name, v.is_active, v.track_haproxy,
|
||||
v.purge_on_teardown,
|
||||
m.applied_config_content, m.applied_config_hash
|
||||
m.applied_config_content, m.applied_config_hash, m.takeover_expected_hash
|
||||
FROM vip_members m JOIN vip_instances v ON v.id = m.vip_id
|
||||
WHERE m.agent_id = $1
|
||||
-- Active VIP first (an agent has at most one). With NO active VIP, pick the most
|
||||
@@ -2343,22 +2375,22 @@ async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str]
|
||||
""", agent['id'])
|
||||
|
||||
if not row:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "discovery_known": bool(agent["discovery_known"]), "keepalived": None}
|
||||
if not row['is_active']:
|
||||
# Soft-deleted VIP → teardown. purge carries the operator's opt-in package removal;
|
||||
# the agent still only purges on nodes where IT installed keepalived (install marker).
|
||||
return {"agent_name": agent_name, "status": "teardown", "vip_id": row['vip_id'],
|
||||
"config_path": config_path, "keepalived": None,
|
||||
"config_path": config_path, "discovery_known": bool(agent["discovery_known"]), "keepalived": None,
|
||||
"purge": bool(row['purge_on_teardown'])}
|
||||
if not row['applied_config_content']:
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "keepalived": None}
|
||||
return {"agent_name": agent_name, "status": "not_configured", "config_path": config_path, "discovery_known": bool(agent["discovery_known"]), "keepalived": None}
|
||||
|
||||
from services.keepalived_config import build_haproxy_check_script
|
||||
check_script = build_haproxy_check_script() if row['track_haproxy'] else ""
|
||||
return {
|
||||
"agent_name": agent_name,
|
||||
"status": "available",
|
||||
"config_path": config_path,
|
||||
"config_path": config_path, "discovery_known": bool(agent["discovery_known"]),
|
||||
"keepalived": {
|
||||
"desired_state": "enabled",
|
||||
"install_if_missing": True,
|
||||
@@ -2367,6 +2399,14 @@ async def get_agent_keepalived_config(agent_name: str, x_api_key: Optional[str]
|
||||
"config_content": row['applied_config_content'],
|
||||
"config_hash": row['applied_config_hash'],
|
||||
"check_script": check_script,
|
||||
# v1.10.4 adoption handoff. The agent refuses to overwrite a keepalived.conf
|
||||
# without our ownership marker — the guard that protects a hand-maintained
|
||||
# setup. Adoption does not weaken it: it authorises exactly ONE takeover, of
|
||||
# exactly the file we analysed, by pinning its hash. If the file changed since
|
||||
# adoption the hashes differ and the agent keeps refusing, so an edit made
|
||||
# between adoption and Apply can never be silently overwritten.
|
||||
"allow_takeover": bool(row['takeover_expected_hash']),
|
||||
"takeover_expected_hash": row['takeover_expected_hash'],
|
||||
},
|
||||
}
|
||||
except HTTPException:
|
||||
@@ -2408,19 +2448,40 @@ async def agent_keepalived_status(agent_name: str, status_data: dict, x_api_key:
|
||||
state = (status_data.get("state") or "").strip()[:24]
|
||||
config_hash = (status_data.get("config_hash") or "")[:64]
|
||||
message = status_data.get("message")
|
||||
# v1.10.9 — retire the adoption takeover authorisation once the node CONFIRMS it is
|
||||
# running our rendered config. `takeover_expected_hash` is the permission to overwrite a
|
||||
# keepalived.conf that lacks our ownership marker; it was written at adoption and never
|
||||
# cleared, so it stayed valid indefinitely and "one-shot" was only true in the sense of
|
||||
# "for exactly that file content". Clearing it the moment the member acks OUR hash makes
|
||||
# the claim real: if the file is replaced by hand afterwards the agent refuses and reports
|
||||
# "externally managed", which is the visible behaviour an operator should get.
|
||||
#
|
||||
# Gated on the acked hash MATCHING applied_config_hash, so a partial or failed deploy
|
||||
# never drops the authorisation and leaves the VIP unable to converge.
|
||||
# The hash is passed TWICE on purpose. Reusing one placeholder for both the assignment
|
||||
# (`last_deploy_hash=$n`, a VARCHAR column) and the comparison inside the CASE made
|
||||
# PostgreSQL deduce two different types for it and asyncpg refused the whole statement
|
||||
# with AmbiguousParameterError ("text versus character varying"). Because the failure is
|
||||
# in the UPDATE itself, not in one column, EVERY status ack was lost and every VIP sat at
|
||||
# SYNCING forever — including teardown acks. A separate placeholder is only ever compared
|
||||
# against the column, so its type is unambiguous.
|
||||
_retire_takeover = ("takeover_expected_hash = CASE WHEN applied_config_hash IS NOT NULL "
|
||||
"AND applied_config_hash = {p} THEN NULL ELSE takeover_expected_hash END")
|
||||
if vip_id is None:
|
||||
# No specific VIP (e.g. a teardown ack) — update all this agent's memberships.
|
||||
await conn.execute("""
|
||||
await conn.execute(f"""
|
||||
UPDATE vip_members SET last_deploy_state=$2, last_deploy_message=$3,
|
||||
last_deploy_hash=$4, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
last_deploy_hash=$4, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP,
|
||||
{_retire_takeover.format(p="$5")}
|
||||
WHERE agent_id=$1
|
||||
""", agent['id'], state, message, config_hash)
|
||||
""", agent['id'], state, message, config_hash, config_hash)
|
||||
else:
|
||||
await conn.execute("""
|
||||
await conn.execute(f"""
|
||||
UPDATE vip_members SET last_deploy_state=$3, last_deploy_message=$4,
|
||||
last_deploy_hash=$5, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP
|
||||
last_deploy_hash=$5, last_deploy_at=CURRENT_TIMESTAMP, updated_at=CURRENT_TIMESTAMP,
|
||||
{_retire_takeover.format(p="$6")}
|
||||
WHERE agent_id=$1 AND vip_id=$2
|
||||
""", agent['id'], int(vip_id), state, message, config_hash)
|
||||
""", agent['id'], int(vip_id), state, message, config_hash, config_hash)
|
||||
return {"status": "ok"}
|
||||
except HTTPException:
|
||||
raise
|
||||
@@ -2431,6 +2492,100 @@ async def agent_keepalived_status(agent_name: str, status_data: dict, x_api_key:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/{agent_name}/keepalived-discovery")
|
||||
async def agent_keepalived_discovery(agent_name: str, payload: dict, x_api_key: Optional[str] = Header(None)):
|
||||
"""v1.10.4 — the agent reports a keepalived.conf it found on the node but does NOT own.
|
||||
|
||||
This is what makes adopting a hand-maintained VIP possible: the heartbeat only carries the
|
||||
VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs eleven
|
||||
fields, so the file itself has to be read. Read-only on the agent side — reporting never
|
||||
changes anything on the node.
|
||||
|
||||
Auth mirrors /keepalived-status: a MISSING key is rejected outright, and because the token
|
||||
is a shared install token a name mismatch is an advisory audit log rather than a 403.
|
||||
|
||||
SECRETS: the reported content may contain the VRRP `auth_pass`. It is split immediately —
|
||||
the password is Fernet-encrypted into its own column and the stored copy of the file has it
|
||||
masked, so nothing readable through the API or a DB dump carries it in cleartext. The
|
||||
parse result is never logged.
|
||||
"""
|
||||
conn = None
|
||||
try:
|
||||
from auth_middleware import validate_agent_api_key
|
||||
agent_auth = await validate_agent_api_key(x_api_key)
|
||||
if not x_api_key or not agent_auth:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
if agent_auth['name'] != agent_name:
|
||||
logger.info(f"Agent '{agent_name}' reporting keepalived discovery using API key "
|
||||
f"from agent '{agent_auth['name']}'")
|
||||
|
||||
conn = await get_database_connection()
|
||||
agent = await conn.fetchrow("SELECT id FROM agents WHERE name = $1", agent_name)
|
||||
if not agent:
|
||||
raise HTTPException(status_code=404, detail=f"Agent '{agent_name}' not found")
|
||||
|
||||
config_path = (payload.get("config_path") or "/etc/keepalived/keepalived.conf")[:500]
|
||||
exists = bool(payload.get("exists"))
|
||||
if not exists:
|
||||
# The file is gone (keepalived removed, or we adopted and now own it) — drop the row
|
||||
# so the UI stops offering a stale candidate.
|
||||
await conn.execute("DELETE FROM vip_discoveries WHERE agent_id = $1", agent['id'])
|
||||
return {"status": "cleared"}
|
||||
|
||||
content = payload.get("config_content") or ""
|
||||
if len(content) > 256_000:
|
||||
raise HTTPException(status_code=413, detail="keepalived.conf too large to analyse")
|
||||
is_managed = bool(payload.get("is_managed"))
|
||||
config_hash = hashlib.md5(content.encode("utf-8", "replace")).hexdigest()
|
||||
|
||||
from services.keepalived_parser import analyse_keepalived_conf, KeepalivedParseError
|
||||
from services.keepalived_config import encrypt_vrrp_secret
|
||||
|
||||
parse_error = None
|
||||
analysis = None
|
||||
auth_enc = None
|
||||
try:
|
||||
analysis = analyse_keepalived_conf(content)
|
||||
# Split the secret out of everything we persist or serve.
|
||||
for cand in analysis.get("candidates", []):
|
||||
secret = (cand.get("vip") or {}).pop("auth_pass", None)
|
||||
cand["vip"]["has_auth_pass"] = bool(secret)
|
||||
if secret and auth_enc is None:
|
||||
auth_enc = encrypt_vrrp_secret(secret)
|
||||
except KeepalivedParseError as exc:
|
||||
parse_error = str(exc)[:500]
|
||||
except Exception as exc: # noqa: BLE001 — a malformed file must not 500 the agent loop
|
||||
parse_error = f"could not analyse the config ({type(exc).__name__})"
|
||||
|
||||
masked = _AUTH_PASS_MASK_RE.sub(r"\1********", content)
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_discoveries
|
||||
(agent_id, config_path, config_hash, is_managed, raw_config_masked,
|
||||
auth_pass_encrypted, analysis, parse_error, reported_at)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,CURRENT_TIMESTAMP)
|
||||
ON CONFLICT (agent_id) DO UPDATE SET
|
||||
config_path = EXCLUDED.config_path,
|
||||
config_hash = EXCLUDED.config_hash,
|
||||
is_managed = EXCLUDED.is_managed,
|
||||
raw_config_masked = EXCLUDED.raw_config_masked,
|
||||
auth_pass_encrypted = EXCLUDED.auth_pass_encrypted,
|
||||
analysis = EXCLUDED.analysis,
|
||||
parse_error = EXCLUDED.parse_error,
|
||||
reported_at = CURRENT_TIMESTAMP
|
||||
""", agent['id'], config_path, config_hash, is_managed, masked, auth_enc,
|
||||
json.dumps(analysis) if analysis is not None else None, parse_error)
|
||||
return {"status": "recorded", "config_hash": config_hash}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"keepalived-discovery failed for '{agent_name}': {e}")
|
||||
raise HTTPException(status_code=500, detail="keepalived-discovery failed")
|
||||
finally:
|
||||
if conn:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/script-version")
|
||||
async def get_latest_script_version(platform: str = "macos"):
|
||||
"""Get the latest available agent script version for specified platform"""
|
||||
|
||||
+134
-12
@@ -111,6 +111,14 @@ router = APIRouter(prefix="/api/clusters", tags=["clusters"])
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _AcmeNoConfigChange(Exception):
|
||||
"""Internal signal: an ACME edit renders the same config, so mint nothing.
|
||||
|
||||
Control flow, not an error — it unwinds out of the version-minting block without
|
||||
tripping the generic `except Exception` handler that would log it as a failure.
|
||||
"""
|
||||
|
||||
|
||||
class _ConcurrentlyDrained(Exception):
|
||||
"""Sentinel raised inside ``apply_pending_changes`` when the
|
||||
advisory-lock-protected re-fetch shows that another caller
|
||||
@@ -294,12 +302,14 @@ async def create_cluster(cluster: HAProxyClusterCreate, authorization: str = Hea
|
||||
cluster_id = await conn.fetchval("""
|
||||
INSERT INTO haproxy_clusters (name, description, connection_type, is_active,
|
||||
stats_socket_path, haproxy_config_path, haproxy_bin_path,
|
||||
keepalived_config_path, pool_id)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7, $8)
|
||||
keepalived_config_path, pool_id,
|
||||
acme_enabled, acme_backend_url)
|
||||
VALUES ($1, $2, $3, TRUE, $4, $5, $6, $7, $8, COALESCE($9, FALSE), $10)
|
||||
RETURNING id
|
||||
""", cluster.name, cluster.description, cluster.connection_type,
|
||||
cluster.stats_socket_path, cluster.haproxy_config_path, cluster.haproxy_bin_path,
|
||||
cluster.keepalived_config_path, cluster.pool_id)
|
||||
cluster.keepalived_config_path, cluster.pool_id,
|
||||
cluster.acme_enabled, cluster.acme_backend_url)
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -389,7 +399,8 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
# Check if cluster exists and get current values
|
||||
existing_cluster = await conn.fetchrow("""
|
||||
SELECT name, description, connection_type, is_active, stats_socket_path,
|
||||
haproxy_config_path, haproxy_bin_path, pool_id, acme_enabled
|
||||
haproxy_config_path, haproxy_bin_path, pool_id, acme_enabled,
|
||||
acme_backend_url
|
||||
FROM haproxy_clusters WHERE id = $1
|
||||
""", cluster_id)
|
||||
if not existing_cluster:
|
||||
@@ -453,7 +464,13 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
update_fields.append(f"acme_enabled = ${param_counter}")
|
||||
update_values.append(cluster.acme_enabled)
|
||||
param_counter += 1
|
||||
if cluster.acme_backend_url is not None:
|
||||
# Keyed on "was the field submitted?", not "is it non-None". With a plain
|
||||
# `is not None` test there is no way to CLEAR the value: the validator maps an
|
||||
# empty box to None, which is indistinguishable from "not supplied", so once an
|
||||
# operator set a per-cluster URL they could never revert to the global setting —
|
||||
# the field would accept the edit and silently keep the old value.
|
||||
_acme_url_submitted = 'acme_backend_url' in cluster.model_fields_set
|
||||
if _acme_url_submitted:
|
||||
update_fields.append(f"acme_backend_url = ${param_counter}")
|
||||
update_values.append(cluster.acme_backend_url)
|
||||
param_counter += 1
|
||||
@@ -463,14 +480,75 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
update_query = f"UPDATE haproxy_clusters SET {', '.join(update_fields)} WHERE id = $1"
|
||||
await conn.execute(update_query, *update_values)
|
||||
|
||||
# If acme_enabled actually changed, create a PENDING config version with entity snapshot
|
||||
if cluster.acme_enabled is not None and cluster.acme_enabled != existing_cluster.get('acme_enabled', False):
|
||||
# Create a PENDING config version when an ACME edit would change what the
|
||||
# HAProxy nodes actually run.
|
||||
#
|
||||
# This used to trigger only on `acme_enabled` flipping. `acme_backend_url` is
|
||||
# written to the DB a few lines above but minted nothing, so correcting a wrong
|
||||
# challenge backend from the panel was a silent no-op: the value changed, no
|
||||
# pending version existed, Apply answered "No pending changes to apply", and the
|
||||
# nodes kept the old address indefinitely. That made the one field an operator
|
||||
# needs to fix HTTP-01 impossible to actually apply.
|
||||
_acme_toggled = (
|
||||
cluster.acme_enabled is not None
|
||||
and cluster.acme_enabled != existing_cluster.get('acme_enabled', False)
|
||||
)
|
||||
if _acme_toggled or _acme_url_submitted:
|
||||
try:
|
||||
from services.haproxy_config import generate_haproxy_config_for_cluster
|
||||
from services.haproxy_config import (
|
||||
generate_haproxy_config_for_cluster,
|
||||
is_config_generation_error,
|
||||
extract_acme_backend_target,
|
||||
)
|
||||
config_content = await generate_haproxy_config_for_cluster(cluster_id)
|
||||
if is_config_generation_error(config_content):
|
||||
# Same sentinel-instead-of-exception contract as the apply path. A
|
||||
# PENDING version holding the sentinel is a landmine: the operator
|
||||
# sees a pending change and applies it, replacing the whole config.
|
||||
logger.error(
|
||||
f"ACME TOGGLE: config generation for cluster {cluster_id} returned an "
|
||||
f"error sentinel; no PENDING version created: {config_content!r}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=(
|
||||
"ACME setting was saved, but a configuration could not be generated "
|
||||
"for this cluster, so no pending change was created. "
|
||||
f"Generator reported: {config_content.strip()[:300]}"
|
||||
),
|
||||
)
|
||||
# A URL edit that renders the same `server _acme_mgmt` line changes
|
||||
# nothing on the nodes, so minting a version would put a no-op pending
|
||||
# change in front of the operator. Compare that one line rather than the
|
||||
# whole config: the generator also renders unrelated PENDING entities,
|
||||
# so a full-text diff reports a change on every edit anyone has queued.
|
||||
_active_config = await conn.fetchval("""
|
||||
SELECT config_content FROM config_versions
|
||||
WHERE cluster_id = $1 AND is_active = TRUE
|
||||
ORDER BY created_at DESC LIMIT 1
|
||||
""", cluster_id)
|
||||
_new_target = extract_acme_backend_target(config_content)
|
||||
_old_target = extract_acme_backend_target(_active_config)
|
||||
if not _acme_toggled and _new_target == _old_target:
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} URL updated but the rendered "
|
||||
f"challenge backend is unchanged ({_new_target!r}); no config "
|
||||
f"version created."
|
||||
)
|
||||
raise _AcmeNoConfigChange()
|
||||
|
||||
import time as _time
|
||||
import json as _json
|
||||
version_name = f"cluster-{cluster_id}-acme-{'enable' if cluster.acme_enabled else 'disable'}-{int(_time.time())}"
|
||||
if _acme_toggled:
|
||||
_acme_kind = 'enable' if cluster.acme_enabled else 'disable'
|
||||
else:
|
||||
_acme_kind = 'backend'
|
||||
version_name = f"cluster-{cluster_id}-acme-{_acme_kind}-{int(_time.time())}"
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} pending config version "
|
||||
f"'{version_name}' created — challenge backend {_old_target!r} -> "
|
||||
f"{_new_target!r}. Apply the cluster for the nodes to pick it up."
|
||||
)
|
||||
|
||||
from utils.entity_snapshot import save_entity_snapshot
|
||||
snapshot_metadata = await save_entity_snapshot(
|
||||
@@ -482,7 +560,14 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
"acme_backend_url": existing_cluster.get('acme_backend_url'),
|
||||
},
|
||||
new_values={
|
||||
"acme_enabled": cluster.acme_enabled,
|
||||
# `acme_enabled` is None when only the URL was submitted; the
|
||||
# snapshot must record the value that is actually in force, or a
|
||||
# rollback would write NULL over a working flag.
|
||||
"acme_enabled": (
|
||||
cluster.acme_enabled
|
||||
if cluster.acme_enabled is not None
|
||||
else existing_cluster.get('acme_enabled', False)
|
||||
),
|
||||
"acme_backend_url": getattr(cluster, 'acme_backend_url', None) or existing_cluster.get('acme_backend_url'),
|
||||
},
|
||||
operation="UPDATE"
|
||||
@@ -497,9 +582,20 @@ async def update_cluster(cluster_id: int, cluster: HAProxyClusterUpdate, authori
|
||||
""", cluster_id, version_name, config_content, current_user.get('id', 1), metadata_json)
|
||||
finally:
|
||||
await close_database_connection(conn2)
|
||||
except _AcmeNoConfigChange:
|
||||
# Not an error: the edit was accepted and simply renders the same
|
||||
# address, so there is nothing for the operator to apply.
|
||||
pass
|
||||
except HTTPException:
|
||||
# The sentinel guard above deliberately fails the request. Without this
|
||||
# clause the generic handler below would swallow it and report success
|
||||
# while no PENDING version exists — the exact silent-success failure
|
||||
# mode this change exists to remove.
|
||||
await close_database_connection(conn)
|
||||
raise
|
||||
except Exception as acme_err:
|
||||
logger.error(f"Failed to create ACME config version for cluster {cluster_id}: {acme_err}")
|
||||
|
||||
|
||||
await close_database_connection(conn)
|
||||
|
||||
# Log activity
|
||||
@@ -1914,6 +2010,28 @@ defaults
|
||||
logger.info(f"🧩 APPLY: Generating fresh configuration from database for cluster {cluster_id}")
|
||||
fresh_config_content = await generate_haproxy_config_for_cluster(cluster_id, conn)
|
||||
|
||||
# `generate_haproxy_config_for_cluster` reports failure by RETURNING a
|
||||
# one-line comment instead of raising (haproxy_config.py outer `except`).
|
||||
# Without this guard that sentinel is hashed, stored as an APPLIED version
|
||||
# and shipped to every agent — silently replacing the cluster's entire
|
||||
# configuration with a comment. Any generator exception (an out-of-range
|
||||
# port in acme_backend_url is enough) triggers it. Refuse the apply instead;
|
||||
# the previous APPLIED version stays in force.
|
||||
from services.haproxy_config import is_config_generation_error
|
||||
if is_config_generation_error(fresh_config_content):
|
||||
logger.error(
|
||||
f"APPLY ABORTED: config generation for cluster {cluster_id} returned an "
|
||||
f"error sentinel instead of a configuration: {fresh_config_content!r}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=(
|
||||
"Configuration could not be generated for this cluster, so nothing "
|
||||
"was applied and the running configuration is unchanged. "
|
||||
f"Generator reported: {fresh_config_content.strip()[:300]}"
|
||||
),
|
||||
)
|
||||
|
||||
# Create a new consolidated config version with fresh content
|
||||
import hashlib
|
||||
import time
|
||||
@@ -2586,7 +2704,11 @@ async def get_config_version_diff(cluster_id: int, version_id: int, authorizatio
|
||||
# HA/VIP (Issue #27): vip-{id}-{action} versions show the generated keepalived.conf
|
||||
# each member node will deploy as the change content (VRRP secret masked). Mirrors
|
||||
# the ssl-* special case above so VIP uses the STANDARD View Change diff modal.
|
||||
vip_match = re.search(r'vip-(\d+)-(create|update|delete)', current_version['version_name'])
|
||||
# `adopt` MUST stay in this alternation. A vip-* action missing here does not degrade
|
||||
# gracefully: the version falls through to the generic HAProxy diff, which compares this
|
||||
# row's keepalived.conf against the cluster's previous haproxy.cfg and shows the whole
|
||||
# HAProxy config as removed. v1.10.4 added `adopt` without it (fixed in v1.10.8).
|
||||
vip_match = re.search(r'vip-(\d+)-(create|update|delete|adopt)', current_version['version_name'])
|
||||
if vip_match:
|
||||
vip_id = int(vip_match.group(1))
|
||||
vip_action = vip_match.group(2)
|
||||
|
||||
@@ -926,12 +926,22 @@ async def import_le_ca_chain(authorization: str = Header(None)):
|
||||
("https://letsencrypt.org/certs/r11.pem", "R11 Intermediate"),
|
||||
]
|
||||
chain_parts = []
|
||||
# v1.11.0: each download is recorded as an outbound row. The bodies are
|
||||
# public CA certificates, not secrets, and the 8 KB body cap truncates them —
|
||||
# what matters here is which URL failed, with what status.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_LETSENCRYPT_CA
|
||||
|
||||
async with aiohttp.ClientSession() as session:
|
||||
for url, name in ca_urls:
|
||||
try:
|
||||
async with session.get(url, timeout=aiohttp.ClientTimeout(total=10)) as resp:
|
||||
if resp.status == 200:
|
||||
chain_parts.append(await resp.text())
|
||||
async with outbound_span(
|
||||
target=TARGET_LETSENCRYPT_CA, method="GET", url=url
|
||||
) as span:
|
||||
async with session.get(url, timeout=aiohttp.ClientTimeout(total=10)) as resp:
|
||||
text = await resp.text() if resp.status == 200 else None
|
||||
span.set_response(resp.status, getattr(resp, "headers", None), text)
|
||||
if resp.status == 200:
|
||||
chain_parts.append(text)
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to download {name}: {e}")
|
||||
|
||||
|
||||
@@ -0,0 +1,514 @@
|
||||
"""v1.11.0 — the read/administration API for the unified request/response log.
|
||||
|
||||
Endpoints (declaration order matters — see below):
|
||||
|
||||
GET /api/request-logs/settings requestlog.manage
|
||||
PUT /api/request-logs/settings requestlog.manage
|
||||
GET /api/request-logs/stats requestlog.read
|
||||
POST /api/request-logs/purge requestlog.manage
|
||||
GET /api/request-logs requestlog.read
|
||||
GET /api/request-logs/{log_id} requestlog.read
|
||||
|
||||
`/{log_id}` is a single-segment path, so FastAPI — which matches in declaration
|
||||
order — would shadow `/settings`, `/stats` and `/purge` if it came first. The
|
||||
literals are therefore declared before it. (This is the mirror image of the
|
||||
trap in routers/settings.py, where `GET /{category}` sits at the top of the
|
||||
file and swallows every literal route added after it.)
|
||||
|
||||
Settings are stored in `system_settings` under the `requestlog` category, so
|
||||
`GET /api/settings/requestlog` still reads them, but writes go through THIS
|
||||
router: the generic `PUT /api/settings/{category}` stringifies values with
|
||||
`str(value)`, which turns `True` into `'True'` — not valid JSON, and the
|
||||
`::jsonb` cast then fails.
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Header, HTTPException, Query
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
|
||||
from auth_middleware import check_user_permission, get_current_user_from_token
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.request_log_settings import (
|
||||
DEFAULT_CONFIG,
|
||||
DEFAULT_EXCLUDE_PATHS,
|
||||
MAX_EXCLUDE_PATHS,
|
||||
MAX_EXCLUDE_PATH_LENGTH,
|
||||
SETTINGS_CATEGORY,
|
||||
config_from_mapping,
|
||||
get_config,
|
||||
load_settings_rows,
|
||||
refresh_config,
|
||||
set_config,
|
||||
)
|
||||
from utils.request_log_sink import TARGET_INBOUND_AGENT, request_log_sink
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/api/request-logs", tags=["Request Logs"])
|
||||
|
||||
# Columns returned by the list endpoint. Bodies and headers are detail-only:
|
||||
# a 200-row page carrying two 8 KB JSONB blobs per row is a 3 MB response.
|
||||
_LIST_COLUMNS = """
|
||||
id, request_id, direction, target, method, url, path, status_code,
|
||||
status_class, duration_ms, user_id, username, host(client_ip) AS client_ip,
|
||||
error, request_body_bytes, response_body_bytes, truncated, created_at
|
||||
"""
|
||||
|
||||
_JSONB_COLUMNS = ("query_params", "request_headers", "request_body",
|
||||
"response_headers", "response_body")
|
||||
|
||||
|
||||
class RequestLogSettings(BaseModel):
|
||||
"""Operator-tunable capture + retention policy."""
|
||||
|
||||
enabled: bool = True
|
||||
capture_inbound: bool = True
|
||||
capture_outbound: bool = True
|
||||
capture_bodies: bool = True
|
||||
capture_get: bool = True
|
||||
capture_agent_success: bool = False
|
||||
max_body_bytes: int = Field(8192, ge=0, le=262144)
|
||||
sample_rate: float = Field(1.0, ge=0.0, le=1.0)
|
||||
exclude_paths: List[str] = Field(
|
||||
default_factory=lambda: list(DEFAULT_EXCLUDE_PATHS),
|
||||
max_length=MAX_EXCLUDE_PATHS,
|
||||
)
|
||||
success_retention_days: int = Field(7, ge=1, le=365)
|
||||
error_retention_days: int = Field(30, ge=1, le=365)
|
||||
max_rows: int = Field(500000, ge=1000, le=50_000_000)
|
||||
prune_interval_minutes: int = Field(60, ge=5, le=1440)
|
||||
|
||||
@field_validator("exclude_paths")
|
||||
@classmethod
|
||||
def _validate_paths(cls, value: List[str]) -> List[str]:
|
||||
for entry in value:
|
||||
if not entry.startswith("/"):
|
||||
raise ValueError("exclude_paths entries must start with '/'")
|
||||
if len(entry) > MAX_EXCLUDE_PATH_LENGTH:
|
||||
raise ValueError(
|
||||
f"exclude_paths entries must be <= {MAX_EXCLUDE_PATH_LENGTH} characters"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
async def _require(authorization: Optional[str], action: str) -> Dict[str, Any]:
|
||||
"""Authenticate, then enforce `requestlog.<action>`.
|
||||
|
||||
`current_user=` is passed through so the admin bypass in
|
||||
check_user_permission short-circuits without a second DB round-trip.
|
||||
"""
|
||||
current_user = await get_current_user_from_token(authorization)
|
||||
allowed = await check_user_permission(
|
||||
current_user["id"], "requestlog", action, current_user=current_user
|
||||
)
|
||||
if not allowed:
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail=f"Insufficient permissions: requestlog.{action} required",
|
||||
)
|
||||
return current_user
|
||||
|
||||
|
||||
async def _can_manage(current_user: Dict[str, Any]) -> bool:
|
||||
return await check_user_permission(
|
||||
current_user["id"], "requestlog", "manage", current_user=current_user
|
||||
)
|
||||
|
||||
|
||||
def _parse_jsonb(value: Any) -> Any:
|
||||
"""asyncpg has no JSONB codec on this pool, so JSONB comes back as raw
|
||||
text. This router is a new contract, so it parses server-side and returns
|
||||
real JSON rather than pushing a JSON.parse() into the UI."""
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
return json.loads(value)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return value
|
||||
return value
|
||||
|
||||
|
||||
def _row_to_dict(row) -> Dict[str, Any]:
|
||||
out = dict(row)
|
||||
for key in _JSONB_COLUMNS:
|
||||
if key in out:
|
||||
out[key] = _parse_jsonb(out[key])
|
||||
created = out.get("created_at")
|
||||
if isinstance(created, datetime):
|
||||
out["created_at"] = created.isoformat()
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Literal paths FIRST — see the module docstring.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@router.get("/settings")
|
||||
async def get_request_log_settings(authorization: Optional[str] = Header(None)):
|
||||
"""Current capture + retention policy, plus the shipped defaults so the UI
|
||||
can offer a 'reset' without hardcoding them."""
|
||||
await _require(authorization, "manage")
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
values = await load_settings_rows(conn)
|
||||
config = config_from_mapping(values) if values else DEFAULT_CONFIG
|
||||
return {
|
||||
"settings": config.as_dict(),
|
||||
"defaults": DEFAULT_CONFIG.as_dict(),
|
||||
"category": SETTINGS_CATEGORY,
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error fetching request log settings: {e}")
|
||||
raise HTTPException(status_code=500, detail="Failed to fetch request log settings")
|
||||
finally:
|
||||
if conn is not None:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.put("/settings")
|
||||
async def update_request_log_settings(
|
||||
body: RequestLogSettings,
|
||||
authorization: Optional[str] = Header(None),
|
||||
):
|
||||
"""Persist the policy and apply it immediately.
|
||||
|
||||
`refresh_config()` at the end is what makes an operator's change take
|
||||
effect on the very next request instead of up to 30 seconds later, when
|
||||
the writer loop would otherwise pick it up.
|
||||
"""
|
||||
current_user = await _require(authorization, "manage")
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
updated = []
|
||||
for suffix, value in body.model_dump().items():
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO system_settings (key, value, category, updated_at, updated_by)
|
||||
VALUES ($1, $2::jsonb, $3, $4, $5)
|
||||
ON CONFLICT (key) DO UPDATE SET
|
||||
value = EXCLUDED.value,
|
||||
updated_at = EXCLUDED.updated_at,
|
||||
updated_by = EXCLUDED.updated_by
|
||||
""",
|
||||
f"{SETTINGS_CATEGORY}.{suffix}",
|
||||
json.dumps(value),
|
||||
SETTINGS_CATEGORY,
|
||||
datetime.utcnow(),
|
||||
current_user.get("id"),
|
||||
)
|
||||
updated.append(suffix)
|
||||
|
||||
# Apply in-process right away, then re-read so this worker's snapshot
|
||||
# is exactly what is on disk.
|
||||
set_config(config_from_mapping(body.model_dump()))
|
||||
await refresh_config()
|
||||
|
||||
logger.info(
|
||||
f"Request log settings updated by {current_user.get('username')}: {len(updated)} keys"
|
||||
)
|
||||
return {"message": f"Updated {len(updated)} settings", "settings": get_config().as_dict()}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error updating request log settings: {e}")
|
||||
raise HTTPException(status_code=500, detail="Failed to update request log settings")
|
||||
finally:
|
||||
if conn is not None:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/stats")
|
||||
async def get_request_log_stats(
|
||||
authorization: Optional[str] = Header(None),
|
||||
hours: int = Query(24, ge=1, le=720),
|
||||
):
|
||||
"""Volume and error breakdown over a window, plus table-level totals and
|
||||
this worker's sink counters (so a saturated queue is visible)."""
|
||||
await _require(authorization, "read")
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
by_direction = await conn.fetch(
|
||||
"""
|
||||
SELECT direction,
|
||||
COUNT(*) AS total,
|
||||
COUNT(*) FILTER (WHERE status_class = 0 OR status_class >= 4) AS errors,
|
||||
COALESCE(ROUND(AVG(duration_ms))::int, 0) AS avg_duration_ms,
|
||||
COALESCE(MAX(duration_ms), 0) AS max_duration_ms
|
||||
FROM request_logs
|
||||
WHERE created_at > NOW() - ($1 || ' hours')::INTERVAL
|
||||
GROUP BY direction
|
||||
""",
|
||||
str(hours),
|
||||
)
|
||||
|
||||
by_status = await conn.fetch(
|
||||
"""
|
||||
SELECT status_class, COUNT(*) AS total
|
||||
FROM request_logs
|
||||
WHERE created_at > NOW() - ($1 || ' hours')::INTERVAL
|
||||
GROUP BY status_class
|
||||
ORDER BY status_class
|
||||
""",
|
||||
str(hours),
|
||||
)
|
||||
|
||||
by_target = await conn.fetch(
|
||||
"""
|
||||
SELECT target,
|
||||
COUNT(*) AS total,
|
||||
COUNT(*) FILTER (WHERE status_class = 0 OR status_class >= 4) AS errors
|
||||
FROM request_logs
|
||||
WHERE target IS NOT NULL
|
||||
AND created_at > NOW() - ($1 || ' hours')::INTERVAL
|
||||
GROUP BY target
|
||||
ORDER BY total DESC
|
||||
LIMIT 20
|
||||
""",
|
||||
str(hours),
|
||||
)
|
||||
|
||||
totals = await conn.fetchrow(
|
||||
"SELECT COUNT(*) AS total_rows, MIN(created_at) AS oldest_at, "
|
||||
"MAX(created_at) AS newest_at FROM request_logs"
|
||||
)
|
||||
|
||||
return {
|
||||
"window_hours": hours,
|
||||
"by_direction": [dict(r) for r in by_direction],
|
||||
"by_status_class": [dict(r) for r in by_status],
|
||||
"by_target": [dict(r) for r in by_target],
|
||||
"total_rows": (totals or {}).get("total_rows", 0),
|
||||
"oldest_at": totals["oldest_at"].isoformat() if totals and totals["oldest_at"] else None,
|
||||
"newest_at": totals["newest_at"].isoformat() if totals and totals["newest_at"] else None,
|
||||
# THIS WORKER only. The sink is a module global, so with
|
||||
# UVICORN_WORKERS > 1 each process keeps its own queue and its own
|
||||
# counters, and whichever worker happens to serve this request is
|
||||
# the one being reported. Labelled rather than aggregated: there is
|
||||
# no cross-process channel here, and a number that looks fleet-wide
|
||||
# but is not would understate drops by exactly the worker count.
|
||||
"sink": {**request_log_sink.stats, "scope": "this worker only"},
|
||||
"retention": {
|
||||
"success_retention_days": get_config().success_retention_days,
|
||||
"error_retention_days": get_config().error_retention_days,
|
||||
"max_rows": get_config().max_rows,
|
||||
},
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error fetching request log stats: {e}")
|
||||
raise HTTPException(status_code=500, detail="Failed to fetch request log stats")
|
||||
finally:
|
||||
if conn is not None:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.post("/purge")
|
||||
async def purge_request_logs(authorization: Optional[str] = Header(None)):
|
||||
"""Run a retention pass now, ignoring the watermark.
|
||||
|
||||
This applies the CONFIGURED retention — it is not a 'delete everything'
|
||||
button. It exists so an operator who has just lowered the retention does
|
||||
not have to wait for the next scheduled pass to reclaim the space.
|
||||
"""
|
||||
current_user = await _require(authorization, "manage")
|
||||
from utils.request_log_prune import prune_request_logs_if_due
|
||||
|
||||
counts = await prune_request_logs_if_due(force=True)
|
||||
logger.info(f"Manual request log purge by {current_user.get('username')}: {counts}")
|
||||
return {
|
||||
"message": "Retention pass completed",
|
||||
"removed": {
|
||||
"success": counts.get("success", 0),
|
||||
"error": counts.get("error", 0),
|
||||
"overflow": counts.get("overflow", 0),
|
||||
},
|
||||
"ran": bool(counts.get("ran")),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# List, then the catch-all detail route LAST.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@router.get("")
|
||||
async def list_request_logs(
|
||||
authorization: Optional[str] = Header(None),
|
||||
direction: Optional[str] = Query(None, pattern="^(inbound|outbound)$"),
|
||||
status_class: Optional[int] = Query(None, ge=0, le=5),
|
||||
method: Optional[str] = Query(None, max_length=10),
|
||||
target: Optional[str] = Query(None, max_length=32),
|
||||
user_id: Optional[int] = Query(None, ge=1),
|
||||
path_prefix: Optional[str] = Query(None, max_length=200),
|
||||
q: Optional[str] = Query(None, max_length=200),
|
||||
request_id: Optional[str] = Query(None, max_length=64),
|
||||
errors_only: bool = Query(False),
|
||||
since: Optional[datetime] = Query(None),
|
||||
until: Optional[datetime] = Query(None),
|
||||
min_duration_ms: Optional[int] = Query(None, ge=0),
|
||||
limit: int = Query(50, ge=1, le=500),
|
||||
offset: int = Query(0, ge=0),
|
||||
):
|
||||
"""Filtered, server-paginated list. Bodies are not included — use the
|
||||
detail endpoint for those."""
|
||||
current_user = await _require(authorization, "read")
|
||||
can_manage = await _can_manage(current_user)
|
||||
|
||||
where: List[str] = []
|
||||
params: List[Any] = []
|
||||
|
||||
def add(clause_template: str, value: Any) -> None:
|
||||
params.append(value)
|
||||
where.append(clause_template.format(n=len(params)))
|
||||
|
||||
if direction:
|
||||
add("direction = ${n}", direction)
|
||||
if status_class is not None:
|
||||
add("status_class = ${n}", status_class)
|
||||
if method:
|
||||
add("method = ${n}", method.upper())
|
||||
if target:
|
||||
add("target = ${n}", target)
|
||||
if user_id is not None:
|
||||
add("user_id = ${n}", user_id)
|
||||
if path_prefix:
|
||||
add("path LIKE ${n} || '%'", path_prefix)
|
||||
if q:
|
||||
# Substring search has no index to lean on; it is the deliberately slow
|
||||
# filter and should be combined with a time window.
|
||||
add("url ILIKE '%' || ${n} || '%'", q)
|
||||
if request_id:
|
||||
add("request_id = ${n}", request_id)
|
||||
if errors_only:
|
||||
where.append("(status_class = 0 OR status_class >= 4)")
|
||||
if since:
|
||||
add("created_at >= ${n}", since)
|
||||
if until:
|
||||
add("created_at <= ${n}", until)
|
||||
if min_duration_ms is not None:
|
||||
add("duration_ms >= ${n}", min_duration_ms)
|
||||
|
||||
# Self-scoping. Captured bodies are a broader disclosure surface than the
|
||||
# existing activity log, so a caller holding only `requestlog.read` sees
|
||||
# their OWN inbound requests, plus the fleet's. `requestlog.manage` (and the
|
||||
# is_admin bypass inside it) lifts the restriction.
|
||||
#
|
||||
# The agent clause is not a widening for its own sake, it is what makes the
|
||||
# `operator` grant do what the migration says it is for: "operators debug
|
||||
# failing applies and ACME orders, so they get read access to the request
|
||||
# log". An apply fails on the NODE, and the node reports that back over its
|
||||
# own API key - so the row carrying the diagnosis is an agent row with
|
||||
# `user_id IS NULL`, which own-rows-only scoping hid from exactly the role
|
||||
# the grant was written for. Scoped on `target`, not on `user_id IS NULL`:
|
||||
# anonymous inbound traffic (failed logins and their usernames, unauthorised
|
||||
# probes) is NOT agent traffic and stays admin-only.
|
||||
if not can_manage:
|
||||
params.append(current_user["id"])
|
||||
own = f"user_id = ${len(params)}"
|
||||
params.append(TARGET_INBOUND_AGENT)
|
||||
where.append(
|
||||
f"(direction = 'inbound' AND ({own} OR target = ${len(params)}))"
|
||||
)
|
||||
|
||||
where_sql = (" WHERE " + " AND ".join(where)) if where else ""
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
rows = await conn.fetch(
|
||||
f"SELECT {_LIST_COLUMNS} FROM request_logs{where_sql} "
|
||||
f"ORDER BY id DESC LIMIT ${len(params) + 1} OFFSET ${len(params) + 2}",
|
||||
*params, limit, offset,
|
||||
)
|
||||
|
||||
# Bounded count: an unfiltered COUNT(*) over a multi-million-row table
|
||||
# is a sequential scan on every page change. Cap it and tell the client
|
||||
# the number is a floor.
|
||||
count_cap = 10001
|
||||
counted = await conn.fetchval(
|
||||
f"SELECT COUNT(*) FROM (SELECT 1 FROM request_logs{where_sql} LIMIT {count_cap}) t",
|
||||
*params,
|
||||
)
|
||||
total = int(counted or 0)
|
||||
|
||||
return {
|
||||
"logs": [_row_to_dict(r) for r in rows],
|
||||
"total": total,
|
||||
"total_is_estimate": total >= count_cap,
|
||||
"limit": limit,
|
||||
"offset": offset,
|
||||
"scoped_to_self": not can_manage,
|
||||
}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error listing request logs: {e}")
|
||||
raise HTTPException(status_code=500, detail="Failed to list request logs")
|
||||
finally:
|
||||
if conn is not None:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{log_id}")
|
||||
async def get_request_log(log_id: int, authorization: Optional[str] = Header(None)):
|
||||
"""One exchange in full, plus every other row sharing its `request_id`.
|
||||
|
||||
That `related` list is the point of the feature: one inbound API call and
|
||||
the ACME / DNS / agent calls it triggered read as a single trace.
|
||||
"""
|
||||
current_user = await _require(authorization, "read")
|
||||
can_manage = await _can_manage(current_user)
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
row = await conn.fetchrow(
|
||||
"SELECT *, host(client_ip) AS client_ip_text FROM request_logs WHERE id = $1",
|
||||
log_id,
|
||||
)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404, detail="Request log entry not found")
|
||||
|
||||
record = _row_to_dict(row)
|
||||
record["client_ip"] = record.pop("client_ip_text", None)
|
||||
|
||||
if not can_manage and not (
|
||||
record.get("direction") == "inbound"
|
||||
and (
|
||||
record.get("user_id") == current_user["id"]
|
||||
or record.get("target") == TARGET_INBOUND_AGENT
|
||||
)
|
||||
):
|
||||
# Same self-scoping rule as the list endpoint. 404 rather than 403
|
||||
# so the endpoint does not confirm that a given id exists.
|
||||
raise HTTPException(status_code=404, detail="Request log entry not found")
|
||||
|
||||
related = await conn.fetch(
|
||||
f"SELECT {_LIST_COLUMNS} FROM request_logs "
|
||||
"WHERE request_id = $1 AND id <> $2 ORDER BY id ASC LIMIT 100",
|
||||
record["request_id"], log_id,
|
||||
)
|
||||
|
||||
return {"log": record, "related": [_row_to_dict(r) for r in related]}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error fetching request log {log_id}: {e}")
|
||||
raise HTTPException(status_code=500, detail="Failed to fetch request log entry")
|
||||
finally:
|
||||
if conn is not None:
|
||||
await close_database_connection(conn)
|
||||
+70
-19
@@ -1,10 +1,12 @@
|
||||
from fastapi import APIRouter, HTTPException, Header
|
||||
from pydantic import BaseModel
|
||||
from typing import Dict, Any
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.acme_backend_url import AcmeBackendUrlError, validate_acme_backend_url
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -50,6 +52,39 @@ async def get_settings_by_category(category: str, authorization: str = Header(No
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
def _validate_acme_challenge_backend_url(value):
|
||||
"""Validate `acme.challenge_backend_url` exactly as the config renderer reads it.
|
||||
|
||||
Settings values are stored as jsonb, so the renderer json.loads them before use
|
||||
(services/haproxy_config.py). Validating the raw column text instead of the
|
||||
decoded string would check the quoting rather than the URL.
|
||||
"""
|
||||
decoded = value
|
||||
if isinstance(decoded, str):
|
||||
try:
|
||||
decoded = json.loads(decoded)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if decoded is None:
|
||||
return
|
||||
try:
|
||||
validate_acme_backend_url(str(decoded))
|
||||
except AcmeBackendUrlError as exc:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail=f"acme.challenge_backend_url: {exc}",
|
||||
) from None
|
||||
|
||||
|
||||
# Per-key validators for settings that end up in generated configuration or in
|
||||
# outbound requests. Everything else is still written through unchecked; this is a
|
||||
# deliberate allow-list of the keys where a bad value causes silent breakage rather
|
||||
# than an obvious one.
|
||||
_SETTING_VALIDATORS = {
|
||||
"acme.challenge_backend_url": _validate_acme_challenge_backend_url,
|
||||
}
|
||||
|
||||
|
||||
@router.put("/{category}")
|
||||
async def update_settings_by_category(
|
||||
category: str,
|
||||
@@ -59,6 +94,13 @@ async def update_settings_by_category(
|
||||
current_user = await _get_admin_user(authorization)
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
# Validate the whole batch before writing any of it, so a rejected key cannot
|
||||
# leave the category half-applied.
|
||||
for key_suffix, value in body.settings.items():
|
||||
validator = _SETTING_VALIDATORS.get(f"{category}.{key_suffix}")
|
||||
if validator is not None:
|
||||
validator(value)
|
||||
|
||||
updated = []
|
||||
for key_suffix, value in body.settings.items():
|
||||
full_key = f"{category}.{key_suffix}"
|
||||
@@ -121,26 +163,35 @@ async def test_acme_connection(authorization: str = Header(None), directory_url:
|
||||
_KNOWN_ACME_FIELDS = ["newNonce", "newAccount", "newOrder", "newAuthz", "revokeCert", "keyChange"]
|
||||
try:
|
||||
import aiohttp
|
||||
# v1.11.0: this handler returns str(e) to the caller and logs nothing —
|
||||
# the span gives the failed probe a durable record.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_SETTINGS_PROBE
|
||||
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.get(
|
||||
directory_url,
|
||||
timeout=aiohttp.ClientTimeout(total=10),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if resp.status == 200:
|
||||
data = await resp.json(content_type=None)
|
||||
if not isinstance(data, dict):
|
||||
return {"success": False, "error": "Directory URL did not return a JSON object"}
|
||||
present = [k for k in _KNOWN_ACME_FIELDS if k in data]
|
||||
if not present:
|
||||
return {"success": False, "error": "Response is not a valid ACME directory"}
|
||||
return {
|
||||
"success": True,
|
||||
"directory": directory_url,
|
||||
"endpoints": present,
|
||||
}
|
||||
else:
|
||||
return {"success": False, "error": f"HTTP {resp.status} from directory URL"}
|
||||
async with outbound_span(
|
||||
target=TARGET_SETTINGS_PROBE, method="GET", url=directory_url
|
||||
) as span:
|
||||
async with session.get(
|
||||
directory_url,
|
||||
timeout=aiohttp.ClientTimeout(total=10),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if resp.status == 200:
|
||||
data = await resp.json(content_type=None)
|
||||
span.set_response(resp.status, getattr(resp, "headers", None), data)
|
||||
if not isinstance(data, dict):
|
||||
return {"success": False, "error": "Directory URL did not return a JSON object"}
|
||||
present = [k for k in _KNOWN_ACME_FIELDS if k in data]
|
||||
if not present:
|
||||
return {"success": False, "error": "Response is not a valid ACME directory"}
|
||||
return {
|
||||
"success": True,
|
||||
"directory": directory_url,
|
||||
"endpoints": present,
|
||||
}
|
||||
else:
|
||||
span.set_response(resp.status, getattr(resp, "headers", None))
|
||||
return {"success": False, "error": f"HTTP {resp.status} from directory URL"}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
|
||||
@@ -458,6 +458,46 @@ async def list_vips(cluster_id: Optional[int] = None, authorization: str = Heade
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
# ORDER MATTERS: every literal path under this router MUST be declared before the
|
||||
# `/{vip_id}` routes below. FastAPI matches in declaration order, so a literal placed after
|
||||
# `/{vip_id}` is swallowed by it and answered with 422 ("discoveries" is not an int) — the
|
||||
# handler never runs. That is not a visible failure either: the HA/VIP page treats any
|
||||
# non-OK response as "nothing to show", so the whole adoption feature silently disappears.
|
||||
# See the v1.10.4 adoption section further down for the endpoint's own documentation.
|
||||
@router.get("/discoveries")
|
||||
async def list_vip_discoveries(cluster_id: Optional[int] = None, authorization: str = Header(None)):
|
||||
"""Unmanaged keepalived configs the agents found on their nodes.
|
||||
|
||||
Read-only and safe to poll: this is what the HA/VIP page shows so an existing VIP is
|
||||
visible before anyone adopts it.
|
||||
|
||||
`cluster_id` scopes the result to the agents in that cluster's pool, resolved exactly like
|
||||
the VIP list above. Without it every discovery in the fleet is returned, which is what a
|
||||
caller that does not know about the parameter still gets.
|
||||
"""
|
||||
await _require(authorization, "read")
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
try:
|
||||
rows = await conn.fetch("""
|
||||
SELECT d.*, a.name AS agent_name, a.pool_id, p.name AS pool_name,
|
||||
av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN haproxy_cluster_pools p ON p.id = a.pool_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE $1::int IS NULL
|
||||
OR a.pool_id = (SELECT pool_id FROM haproxy_clusters WHERE id = $1::int)
|
||||
ORDER BY d.reported_at DESC, d.id DESC
|
||||
""", cluster_id)
|
||||
except Exception as exc: # noqa: BLE001 — a missing relation degrades to empty (B-7)
|
||||
logger.debug(f"vip_discoveries unavailable: {exc}")
|
||||
return {"discoveries": []}
|
||||
return {"discoveries": [_discovery_row_to_api(r) for r in rows]}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
|
||||
@router.get("/{vip_id}")
|
||||
async def get_vip(vip_id: int, authorization: str = Header(None)):
|
||||
await _require(authorization, "read")
|
||||
@@ -985,3 +1025,425 @@ async def vip_status(vip_id: int, authorization: str = Header(None)):
|
||||
# endpoint (cluster.py get_config_version_diff, vip-* branch) via render_vip_config_masked
|
||||
# above — there is no bespoke VIP preview endpoint, so VIP changes use the product's
|
||||
# standard "View Change" like every other entity (issue #27 follow-up).
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# v1.10.4 — Adoption of a keepalived setup that already exists on the nodes
|
||||
# ---------------------------------------------------------------------------
|
||||
# The HA/VIP page starts empty on a fleet that already runs keepalived, because the flow is
|
||||
# one-way: VIPs are declared here and pushed to the node, and nothing read what was already
|
||||
# there. The agent now reports the keepalived.conf it found (read-only) into vip_discoveries;
|
||||
# these two endpoints list those findings and turn one into a managed VIP.
|
||||
#
|
||||
# Adoption REPLACES the operator's file with our render, so it is gated hard: the parser
|
||||
# reports every directive we cannot reproduce and every value we cannot know, and adoption
|
||||
# refuses while any remain. See services/keepalived_parser.py for the reasoning.
|
||||
|
||||
|
||||
def _discovery_row_to_api(row) -> dict:
|
||||
"""Shape a vip_discoveries row for the UI. Never includes the VRRP password: the stored
|
||||
config copy is masked and the analysis has the secret replaced by a boolean."""
|
||||
analysis = row["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
try:
|
||||
analysis = json.loads(analysis)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
analysis = None
|
||||
return {
|
||||
"agent_id": row["agent_id"],
|
||||
"agent_name": row["agent_name"],
|
||||
"pool_id": row["pool_id"],
|
||||
"pool_name": row["pool_name"],
|
||||
"config_path": row["config_path"],
|
||||
"config_hash": row["config_hash"],
|
||||
"is_managed": row["is_managed"],
|
||||
"parse_error": row["parse_error"],
|
||||
"adopted_vip_id": row["adopted_vip_id"],
|
||||
# v1.10.8 — adoptability is derived from whether the linked VIP is STILL active, not
|
||||
# from the link existing. `adopted_vip_id` is write-once and nothing clears it, and a
|
||||
# VIP is only ever soft-deleted (is_active=FALSE), so the column's ON DELETE SET NULL
|
||||
# never fires. Keying the UI on the link alone made a rejected adoption hide the node
|
||||
# from the panel forever: the VIP was gone from the VIP list too, and the agent does not
|
||||
# re-report an unchanged file. Deriving it here self-heals reject, undo-reject, approved
|
||||
# teardown and anything added later, without a write on each path.
|
||||
"adopted_vip_active": bool(row["adopted_vip_active"]) if row["adopted_vip_id"] else False,
|
||||
"reported_at": row["reported_at"].isoformat() if row["reported_at"] else None,
|
||||
"config_preview": row["raw_config_masked"],
|
||||
"analysis": analysis,
|
||||
}
|
||||
|
||||
|
||||
# NOTE: list_vip_discoveries lives above the `/{vip_id}` routes — see the ordering comment
|
||||
# there. `/adopt` below is a POST and no `POST /{vip_id}` exists, so it is not shadowed.
|
||||
|
||||
|
||||
def _find_candidate(analysis: Optional[dict], instance_name: str) -> Optional[dict]:
|
||||
for cand in ((analysis or {}).get("candidates") or []):
|
||||
if cand.get("instance_name") == instance_name:
|
||||
return cand
|
||||
return None
|
||||
|
||||
|
||||
async def _collect_instance_participants(conn, *, pool_id: int, vrid: int, virtual_ip: str):
|
||||
"""Every discovered node in the pool that reports the SAME vrrp_instance.
|
||||
|
||||
v1.10.8. Adoption used to take only the node whose row was clicked, which broke the exact
|
||||
case the feature exists for — a running HA pair:
|
||||
|
||||
* adopting the BACKUP alone produced a VIP whose apply fails outright, because apply
|
||||
requires exactly one MASTER member;
|
||||
* adopting the MASTER alone left the peer unmanaged, and adopting it afterwards hit the
|
||||
VRID-collision guard with a 409, so the pair could never be completed from the panel;
|
||||
* worst, on a UNICAST pair the single-member render silently drops the unicast block —
|
||||
render_keepalived_conf only emits it when peer_ips is non-empty — so keepalived falls
|
||||
back to multicast on the adopted node while its peer stays unicast. They stop seeing
|
||||
each other and BOTH claim the VIP.
|
||||
|
||||
Identity is (virtual_router_id, virtual address), which is what keepalived itself uses to
|
||||
decide two nodes belong to one VRRP group, so it is the correct key. A node whose config
|
||||
failed to parse cannot be matched and is therefore skipped — the unicast peer check in the
|
||||
caller is what stops that turning into a silent half-adoption.
|
||||
"""
|
||||
rows = await conn.fetch("""
|
||||
SELECT d.agent_id, d.config_hash, d.analysis, d.parse_error, d.adopted_vip_id,
|
||||
d.auth_pass_encrypted,
|
||||
a.name AS agent_name, a.ip_address, av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE a.pool_id = $1 AND COALESCE(a.enabled, TRUE) = TRUE
|
||||
""", pool_id)
|
||||
|
||||
participants = []
|
||||
for r in rows:
|
||||
if r["parse_error"]:
|
||||
continue
|
||||
if r["adopted_vip_id"] and r["adopted_vip_active"]:
|
||||
continue # already under management by a VIP that still stands
|
||||
analysis = r["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
try:
|
||||
analysis = json.loads(analysis)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
for cand in ((analysis or {}).get("candidates") or []):
|
||||
v = cand.get("vip") or {}
|
||||
if v.get("virtual_router_id") == vrid and v.get("virtual_ip") == virtual_ip:
|
||||
participants.append({
|
||||
"agent_id": r["agent_id"], "agent_name": r["agent_name"],
|
||||
"ip_address": r["ip_address"], "config_hash": r["config_hash"],
|
||||
"auth_pass_encrypted": r["auth_pass_encrypted"],
|
||||
"candidate": cand,
|
||||
})
|
||||
break
|
||||
return participants
|
||||
|
||||
|
||||
@router.post("/adopt")
|
||||
async def adopt_vip(payload: dict, request: Request, authorization: str = Header(None)):
|
||||
"""Turn one discovered vrrp_instance into a managed VIP.
|
||||
|
||||
Body: agent_id, instance_name, name, [description], [prefix_length], [accept_data_loss].
|
||||
|
||||
Refuses while the parser reports blockers. Two of them are resolvable by the operator
|
||||
rather than fatal:
|
||||
* a missing prefix length can be supplied as `prefix_length` (we never guess a netmask
|
||||
for a live VIP);
|
||||
* "we would delete this directive" can be accepted with `accept_data_loss: true`, which
|
||||
is an explicit choice to lose e.g. a notify hook. Everything else — an unknown VRID, a
|
||||
fractional advert_int, an unsupported auth_type — is not a loss but an impossibility,
|
||||
and no flag overrides it.
|
||||
|
||||
The VIP is created PENDING like any other, so nothing reaches the node until the operator
|
||||
applies it from Apply Management.
|
||||
"""
|
||||
current_user = await _require(authorization, "create")
|
||||
agent_id = payload.get("agent_id")
|
||||
instance_name = (payload.get("instance_name") or "").strip()
|
||||
name = (payload.get("name") or "").strip()
|
||||
if not agent_id or not instance_name or not name:
|
||||
raise HTTPException(status_code=400, detail="agent_id, instance_name and name are required")
|
||||
|
||||
conn = await get_database_connection()
|
||||
try:
|
||||
disc = await conn.fetchrow("""
|
||||
SELECT d.*, a.name AS agent_name, a.pool_id, a.ip_address,
|
||||
av.is_active AS adopted_vip_active
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE d.agent_id = $1
|
||||
""", int(agent_id))
|
||||
if not disc:
|
||||
raise HTTPException(status_code=404, detail="No discovered keepalived config for that agent")
|
||||
# Only an adoption that is still STANDING blocks a new one. A rejected adoption leaves
|
||||
# adopted_vip_id pointing at a soft-deleted VIP, and nothing clears it, so keying on the
|
||||
# link alone made the node permanently unadoptable (v1.10.8).
|
||||
if disc["adopted_vip_id"] and disc["adopted_vip_active"]:
|
||||
raise HTTPException(status_code=409, detail="This discovery has already been adopted")
|
||||
if not disc["pool_id"]:
|
||||
raise HTTPException(status_code=400, detail="The agent is not in a pool; assign it first")
|
||||
if disc["parse_error"]:
|
||||
raise HTTPException(status_code=422,
|
||||
detail=f"Config could not be parsed: {disc['parse_error']}")
|
||||
|
||||
analysis = disc["analysis"]
|
||||
if isinstance(analysis, str):
|
||||
analysis = json.loads(analysis)
|
||||
cand = _find_candidate(analysis, instance_name)
|
||||
if not cand:
|
||||
raise HTTPException(status_code=404,
|
||||
detail=f"No vrrp_instance '{instance_name}' in the report")
|
||||
|
||||
vip_fields = dict(cand.get("vip") or {})
|
||||
member = dict(cand.get("member") or {})
|
||||
|
||||
# The operator may supply the one value we refuse to guess.
|
||||
supplied_prefix = payload.get("prefix_length")
|
||||
if vip_fields.get("prefix_length") is None and supplied_prefix is not None:
|
||||
try:
|
||||
vip_fields["prefix_length"] = int(supplied_prefix)
|
||||
except (TypeError, ValueError):
|
||||
raise HTTPException(status_code=400, detail="prefix_length must be an integer")
|
||||
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
vrid = vip_fields.get("virtual_router_id")
|
||||
virtual_ip = vip_fields.get("virtual_ip")
|
||||
if vrid is None or not virtual_ip or not member.get("network_interface"):
|
||||
raise HTTPException(status_code=422,
|
||||
detail="Incomplete candidate (vrid/address/interface)")
|
||||
|
||||
# v1.10.8 — adopt the whole VRRP instance. Every node in the pool reporting this same
|
||||
# VRID + address becomes a member, each with the role, priority and interface ITS OWN
|
||||
# file declares. See _collect_instance_participants for why single-node adoption was
|
||||
# unsafe on a unicast pair.
|
||||
participants = await _collect_instance_participants(
|
||||
conn, pool_id=disc["pool_id"], vrid=int(vrid), virtual_ip=virtual_ip)
|
||||
if not any(p["agent_id"] == int(agent_id) for p in participants):
|
||||
# The reporting node must be in its own instance; if it is not, something changed
|
||||
# underneath us (re-report, concurrent adoption) — refuse rather than guess.
|
||||
raise HTTPException(status_code=409,
|
||||
detail="The discovery changed while adopting; refresh and try again")
|
||||
|
||||
# One source of truth for which blockers an operator may resolve (see the docstring on
|
||||
# remaining_blockers): a supplied prefix, and an explicit acceptance of directives our
|
||||
# renderer would delete. Nothing else is waivable. Checked for EVERY node we are about
|
||||
# to overwrite, not only the one that was clicked.
|
||||
for p in participants:
|
||||
rem = remaining_blockers(
|
||||
list(p["candidate"].get("blockers") or []),
|
||||
prefix_supplied=vip_fields.get("prefix_length") is not None,
|
||||
accept_data_loss=bool(payload.get("accept_data_loss")),
|
||||
)
|
||||
if rem:
|
||||
raise HTTPException(status_code=422, detail={
|
||||
"message": f"This keepalived config cannot be adopted as-is ({p['agent_name']})",
|
||||
"node": p["agent_name"],
|
||||
"blockers": rem,
|
||||
})
|
||||
if not (p["candidate"].get("member") or {}).get("network_interface"):
|
||||
raise HTTPException(status_code=422,
|
||||
detail=f"{p['agent_name']} does not declare an interface for this instance")
|
||||
|
||||
# STRANDING GUARD. Participant resolution can only match a node it can READ, that is
|
||||
# ENABLED, and that is in THIS pool. Every one of those is a door through which a real
|
||||
# member of the VRRP group leaves the set silently — and a silent exit is the whole
|
||||
# failure mode this release exists to close, because the nodes we do adopt get rewritten
|
||||
# while the one that left keeps running an unmanaged config on the same address.
|
||||
#
|
||||
# So rather than guard each door, ask the question directly: is there any reported
|
||||
# keepalived.conf that mentions THIS virtual address and is not among the nodes we are
|
||||
# about to adopt? Scoping on the address keeps an unrelated file elsewhere in the fleet
|
||||
# from blocking every adoption.
|
||||
#
|
||||
# Two exclusions are legitimate rather than stranding:
|
||||
# * a node already held by a VIP that still STANDS is under management already — and it
|
||||
# cannot be this instance, because uq_vip_vrid_active forbids a second active VIP with
|
||||
# this VRID in the pool (the clash check above would have refused first);
|
||||
# * a node whose file carries our ownership marker (is_managed) is ours already.
|
||||
adopted_ids = [p["agent_id"] for p in participants]
|
||||
stranded = await conn.fetch("""
|
||||
SELECT a.name AS agent_name, a.pool_id, COALESCE(a.enabled, TRUE) AS enabled,
|
||||
d.parse_error
|
||||
FROM vip_discoveries d
|
||||
JOIN agents a ON a.id = d.agent_id
|
||||
LEFT JOIN vip_instances av ON av.id = d.adopted_vip_id
|
||||
WHERE d.raw_config_masked LIKE '%' || $1 || '%'
|
||||
AND NOT (a.id = ANY($2::int[]))
|
||||
AND COALESCE(av.is_active, FALSE) = FALSE
|
||||
AND d.is_managed = FALSE
|
||||
ORDER BY a.name
|
||||
""", virtual_ip, adopted_ids)
|
||||
if stranded:
|
||||
def _why(r):
|
||||
if r["parse_error"]:
|
||||
return f"its config could not be parsed: {r['parse_error']}"
|
||||
if not r["enabled"]:
|
||||
return "the agent is disabled, so it can never receive a config"
|
||||
if r["pool_id"] != disc["pool_id"]:
|
||||
return "it is in a different agent pool, and a VIP's members must share one pool"
|
||||
return "it does not report this vrrp_instance"
|
||||
named = "; ".join(f"{r['agent_name']} — {_why(r)}" for r in stranded)
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"These node(s) also reference {virtual_ip} but cannot be taken over with the rest "
|
||||
f"of the instance, so adopting now would rewrite the others and leave them running "
|
||||
f"an unmanaged config on the same address: {named}. Resolve that first, then adopt."))
|
||||
|
||||
# AGREEMENT ON THE SHARED VIP FIELDS. Everything below is stored once on the VIP row and
|
||||
# re-rendered onto EVERY member, so a value taken from the node that happened to be
|
||||
# clicked would be imposed on nodes whose own file said something else. prefix_length is
|
||||
# the sharpest: the design refuses to GUESS a netmask for a live VIP, and quietly copying
|
||||
# one node's netmask onto another is the same change by another name.
|
||||
def _vfield(p, key):
|
||||
return (p["candidate"].get("vip") or {}).get(key)
|
||||
|
||||
for key, label in (("prefix_length", "prefix length"),
|
||||
("use_unicast", "unicast/multicast mode"),
|
||||
("track_haproxy", "HAProxy tracking")):
|
||||
seen = {}
|
||||
for p in participants:
|
||||
seen.setdefault(_vfield(p, key), []).append(p["agent_name"])
|
||||
if len(seen) > 1:
|
||||
spread = "; ".join(f"{v!r}: {', '.join(names)}" for v, names in seen.items())
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes of this instance disagree on {label} ({spread}). One value is "
|
||||
f"stored on the VIP and re-rendered onto every member, so adopting would "
|
||||
f"impose one node's setting on the others. Align the files first."))
|
||||
|
||||
# The VRRP secret is stored per discovery as its own Fernet token, and Fernet is
|
||||
# non-deterministic, so the ciphertexts cannot be compared — decrypt and compare the
|
||||
# plaintexts. Nothing is logged. A node we cannot decrypt is treated as a mismatch rather
|
||||
# than assumed equal.
|
||||
secrets = {}
|
||||
for p in participants:
|
||||
enc = p.get("auth_pass_encrypted")
|
||||
plain = decrypt_vrrp_secret(enc) if enc else None
|
||||
if enc and not plain:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"The VRRP secret reported by {p['agent_name']} cannot be decrypted "
|
||||
f"(encryption key changed?). Re-report or fix that node before adopting."))
|
||||
secrets.setdefault(plain, []).append(p["agent_name"])
|
||||
if len(secrets) > 1:
|
||||
groups = "; ".join(
|
||||
("no password: " if k is None else "one password: ") + ", ".join(v)
|
||||
for k, v in secrets.items())
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes of this instance do not share one VRRP password ({groups}). "
|
||||
f"Adoption stores a single secret and renders it onto every member, so it would "
|
||||
f"silently change authentication on the others. Align the files first."))
|
||||
|
||||
# keepalived requires one MASTER and a matching advertisement interval across the group.
|
||||
# Both are checked here so the operator learns at adoption time instead of at apply.
|
||||
roles = [(p["candidate"].get("member") or {}).get("role") or "BACKUP" for p in participants]
|
||||
if roles.count("MASTER") != 1:
|
||||
listed = ", ".join(f"{p['agent_name']}={r}" for p, r in zip(participants, roles))
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"VRID {vrid} is reported by {len(participants)} node(s) with "
|
||||
f"{roles.count('MASTER')} MASTER ({listed}); exactly one must be MASTER. If a "
|
||||
f"member is missing, install or enable its agent so it reports its keepalived.conf, "
|
||||
f"then adopt again."))
|
||||
adverts = {int((p["candidate"].get("vip") or {}).get("advert_int") or 1) for p in participants}
|
||||
if len(adverts) > 1:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"The nodes disagree on advert_int ({sorted(adverts)}). keepalived needs the same "
|
||||
f"advertisement interval across a VRRP group, so align the files first."))
|
||||
|
||||
# UNICAST SAFETY. Our renderer emits the unicast block only when it has peer addresses,
|
||||
# so a peer that is not a member would be dropped and keepalived would fall back to
|
||||
# multicast on this node while the real peer stays unicast — both would then claim the
|
||||
# VIP. Refuse instead, naming the address that is unaccounted for.
|
||||
declared_peers = {str(x) for p in participants for x in (p["candidate"].get("peers") or [])}
|
||||
if declared_peers:
|
||||
missing_ip = [p["agent_name"] for p in participants if not p["ip_address"]]
|
||||
if missing_ip:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"These nodes have not reported an IP address yet: {', '.join(missing_ip)}. "
|
||||
f"The unicast peer list cannot be verified until they do."))
|
||||
member_ips = {str(p["ip_address"]) for p in participants}
|
||||
unknown = sorted(declared_peers - member_ips)
|
||||
if unknown:
|
||||
raise HTTPException(status_code=422, detail=(
|
||||
f"This instance is unicast and lists peer(s) {', '.join(unknown)} that are not "
|
||||
f"among the nodes being adopted. Adopting would drop them from the peer list and "
|
||||
f"keepalived would silently fall back to multicast, so both sides could end up "
|
||||
f"holding the VIP. Register those nodes as agents so they report their config, "
|
||||
f"then adopt again."))
|
||||
|
||||
# One active VIP per agent — the delivery endpoint serves a single keepalived.conf per
|
||||
# node, so a second active membership never converges. create/update enforce this via
|
||||
# _validate_members_against_pool; adoption did not call it at all.
|
||||
busy = await conn.fetchrow("""
|
||||
SELECT a.name AS agent_name, v.name AS vip_name
|
||||
FROM vip_members vm JOIN vip_instances v ON v.id = vm.vip_id
|
||||
JOIN agents a ON a.id = vm.agent_id
|
||||
WHERE vm.agent_id = ANY($1::int[]) AND v.is_active = TRUE LIMIT 1
|
||||
""", [p["agent_id"] for p in participants])
|
||||
if busy:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"node '{busy['agent_name']}' is already a member of VIP '{busy['vip_name']}'"))
|
||||
|
||||
# Adoption must keep the VRRP identity it found. A different VRID would create a second
|
||||
# VRRP domain on the wire, so a collision inside the pool is a hard conflict, never an
|
||||
# auto-reallocation the way create_vip does it.
|
||||
clash = await conn.fetchrow("""
|
||||
SELECT id, name FROM vip_instances
|
||||
WHERE pool_id = $1 AND is_active = TRUE AND virtual_router_id = $2
|
||||
""", disc["pool_id"], int(vrid))
|
||||
if clash:
|
||||
raise HTTPException(status_code=409, detail=(
|
||||
f"VRID {vrid} is already used by VIP '{clash['name']}' in this pool; the adopted "
|
||||
f"config must keep its VRID, so resolve the collision first"))
|
||||
|
||||
async with conn.transaction():
|
||||
vip_id = await conn.fetchval("""
|
||||
INSERT INTO vip_instances
|
||||
(name, description, pool_id, virtual_ip, prefix_length, virtual_router_id,
|
||||
advert_int, auth_pass_encrypted, use_unicast, track_haproxy,
|
||||
is_active, last_config_status, adopted_at, created_by)
|
||||
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,TRUE,'PENDING',CURRENT_TIMESTAMP,$11)
|
||||
RETURNING id
|
||||
""", name, payload.get("description") or f"Adopted from {disc['agent_name']}",
|
||||
disc["pool_id"], virtual_ip, int(vip_fields["prefix_length"]), int(vrid),
|
||||
int(vip_fields.get("advert_int") or 1),
|
||||
disc["auth_pass_encrypted"], # already Fernet-encrypted at ingest
|
||||
bool(vip_fields.get("use_unicast")), bool(vip_fields.get("track_haproxy")),
|
||||
current_user["id"])
|
||||
# Every node of the instance becomes a member with the role/priority/interface ITS
|
||||
# OWN file declares, and each carries its own one-shot takeover authorisation pinned
|
||||
# to the hash of the file we analysed on THAT node. The guard stays per-node: a file
|
||||
# edited on one member between adoption and Apply is still refused there.
|
||||
for p in participants:
|
||||
pm = p["candidate"].get("member") or {}
|
||||
await conn.execute("""
|
||||
INSERT INTO vip_members
|
||||
(vip_id, agent_id, network_interface, role, priority, takeover_expected_hash)
|
||||
VALUES ($1,$2,$3,$4,$5,$6)
|
||||
""", vip_id, p["agent_id"], pm["network_interface"],
|
||||
pm.get("role") or "BACKUP", int(pm.get("priority") or 100),
|
||||
p["config_hash"])
|
||||
await conn.execute(
|
||||
"UPDATE vip_discoveries SET adopted_vip_id = $2 WHERE agent_id = $1",
|
||||
p["agent_id"], vip_id)
|
||||
|
||||
await _stage_vip_version(conn, vip_id, "adopt", current_user["id"])
|
||||
await log_user_activity(
|
||||
user_id=current_user["id"], action="adopt", resource_type="vip",
|
||||
resource_id=str(vip_id),
|
||||
details={"name": name, "virtual_ip": virtual_ip, "vrid": vrid,
|
||||
"adopted_from_agent": disc["agent_name"],
|
||||
"members": [p["agent_name"] for p in participants],
|
||||
"accepted_data_loss": bool(payload.get("accept_data_loss"))},
|
||||
ip_address=_client_ip(request), user_agent=_user_agent(request))
|
||||
|
||||
node_names = [p["agent_name"] for p in participants]
|
||||
return {
|
||||
"id": vip_id,
|
||||
"members": node_names,
|
||||
"message": (
|
||||
f"VIP adopted with {len(node_names)} member node(s): {', '.join(node_names)} "
|
||||
f"(PENDING — review it in Apply Management, then apply to hand their "
|
||||
f"keepalived.conf over to OpenManager)"),
|
||||
}
|
||||
finally:
|
||||
await close_database_connection(conn)
|
||||
|
||||
@@ -8,9 +8,13 @@ suitable for an Antd Tabs/Steps display.
|
||||
Key constraints (Section 3.3 of the v1.5.0 plan):
|
||||
- DNS resolution uses stdlib socket.gethostbyname_ex via run_in_executor (we
|
||||
intentionally avoid pulling aiodns as a runtime dep for v1.5.0).
|
||||
- Port-80 probe is HEAD-only, target locked to the order's domains, success on
|
||||
HTTP 200 OR 404, warns on egress timeout (don't fail-hard — corp egress
|
||||
policies often blackhole outbound 80).
|
||||
- Port-80 probe is a GET (not HEAD) locked to the order's domains, because the
|
||||
status code alone cannot tell a working challenge endpoint from a web UI: a
|
||||
reverse proxy that has lost its /.well-known/acme-challenge/ location serves
|
||||
its SPA with HTTP 200. The body's shape decides. Warns rather than fails on
|
||||
egress timeout (corp egress policies often blackhole outbound 80) and on a
|
||||
wrong responder (this probe sees the PUBLIC domain, not the challenge backend,
|
||||
so it is evidence rather than a verdict).
|
||||
- All checks have hard wall-clock timeouts (asyncio.wait_for) to bound impact
|
||||
on the API event loop.
|
||||
- humanize_error_detail covers >= 11 RFC8555 problem types and is backwards
|
||||
@@ -330,10 +334,41 @@ async def check_dns(domains: List[str]) -> Dict[str, Any]:
|
||||
)
|
||||
|
||||
|
||||
# Enough to classify a response without turning the diagnostic into a way to pull
|
||||
# arbitrary amounts of a third party's content into our JSON.
|
||||
_PROBE_BODY_LIMIT = 65536
|
||||
|
||||
|
||||
def _classify_probe_body(body: bytes, content_type: str) -> str:
|
||||
"""Coarse shape of a probe response: html | json | text | empty | binary.
|
||||
|
||||
The body itself is deliberately NOT retained anywhere — the shape is all that is
|
||||
needed to tell "served me a web page" from "served me a token", and keeping the
|
||||
bytes would open a new read surface onto whatever is behind the address.
|
||||
"""
|
||||
if not body:
|
||||
return "empty"
|
||||
ct = (content_type or "").lower()
|
||||
head = body[:512].lstrip().lower()
|
||||
if ct.startswith("text/html") or head.startswith((b"<!doctype", b"<html")):
|
||||
return "html"
|
||||
if ct.startswith("application/json") or head[:1] in (b"{", b"["):
|
||||
return "json"
|
||||
try:
|
||||
body.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return "binary"
|
||||
return "text"
|
||||
|
||||
|
||||
async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict[str, Any]:
|
||||
"""Probe HTTP-01 readiness on port 80 with a HEAD request to a synthetic
|
||||
challenge URL. Success on 200 OR 404 (404 means the well-known path is
|
||||
served but no challenge yet — fine).
|
||||
"""Probe HTTP-01 readiness on port 80 with a GET to a synthetic challenge URL.
|
||||
|
||||
404 means the path is served but no challenge is outstanding, which is fine. A
|
||||
200 is only fine if the body is NOT a web page: a proxy that has lost its
|
||||
/.well-known/acme-challenge/ route falls through to its catch-all and answers
|
||||
200 with index.html, which a status-code-only check accepts as healthy while
|
||||
every real validation fails.
|
||||
|
||||
On egress timeout we WARN rather than FAIL because many corporate egress
|
||||
policies blackhole port 80 outbound; that does not impair LE's ingress
|
||||
@@ -398,12 +433,84 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
continue
|
||||
url = f"http://{d}/.well-known/acme-challenge/diagnostic-probe"
|
||||
try:
|
||||
async with session.head(url, allow_redirects=False) as resp:
|
||||
targets.append({
|
||||
"domain": d,
|
||||
"status": resp.status,
|
||||
"ok": resp.status in (200, 404),
|
||||
})
|
||||
# v1.11.0: recorded as an outbound row so a failing port-80 probe is
|
||||
# diagnosable after the fact, not only while the panel is open.
|
||||
#
|
||||
# MERGE NOTE: the feature branch instrumented a `session.head(...)`
|
||||
# probe, which is what this function did when that branch was cut.
|
||||
# It has since become a GET, because the status code alone cannot
|
||||
# tell a working challenge endpoint from a SPA catch-all serving
|
||||
# index.html with HTTP 200 (v1.10.x). Reverting to HEAD to gain the
|
||||
# log row would put that bug straight back, so the GET probe below
|
||||
# is the one that is wrapped.
|
||||
#
|
||||
# GET, not HEAD: the status code alone cannot tell a working challenge
|
||||
# endpoint from a SPA. A reverse proxy that has lost its
|
||||
# /.well-known/acme-challenge/ location falls through to its catch-all
|
||||
# and serves index.html with HTTP 200 — which the old
|
||||
# `status in (200, 404)` rule accepted as healthy while every real
|
||||
# validation failed. Only the body distinguishes them.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_ACME_DIAG
|
||||
|
||||
async with outbound_span(
|
||||
target=TARGET_ACME_DIAG, method="GET", url=url, capture_body=False
|
||||
) as span:
|
||||
async with session.get(url, allow_redirects=False) as resp:
|
||||
# The body is EVIDENCE, not a precondition. If it cannot be read —
|
||||
# connection reset mid-response, a server that hangs after headers —
|
||||
# fall back to the status-only semantics this check has always had
|
||||
# rather than turning a healthy 404 into a hard failure. The stricter
|
||||
# rule below applies only when there is something to judge.
|
||||
try:
|
||||
body = await resp.content.read(_PROBE_BODY_LIMIT)
|
||||
except Exception:
|
||||
body = None
|
||||
content_type = (resp.headers.get("content-type") or "").split(";")[0].strip()
|
||||
body_class = (
|
||||
_classify_probe_body(body, content_type) if body is not None else "unread"
|
||||
)
|
||||
target = {
|
||||
"domain": d,
|
||||
"status": resp.status,
|
||||
"content_type": content_type or None,
|
||||
"body_len": len(body) if body is not None else None,
|
||||
"body_class": body_class,
|
||||
}
|
||||
# The VERDICT goes in the log row, not the body. `body` here is
|
||||
# up to 64 KB of a third party's page (`_PROBE_BODY_LIMIT`);
|
||||
# storing it would put an arbitrary remote document into the
|
||||
# audit table for every probed domain. The classification is
|
||||
# what an operator reads back anyway.
|
||||
span.set_response(
|
||||
resp.status,
|
||||
getattr(resp, "headers", None),
|
||||
{
|
||||
"probe": "acme-http01",
|
||||
"content_type": content_type or None,
|
||||
"body_len": len(body) if body is not None else None,
|
||||
"body_class": body_class,
|
||||
},
|
||||
)
|
||||
if resp.status == 200 and body_class == "html":
|
||||
# Reachable, wrong responder. Reported as a warning rather than
|
||||
# a failure: this check probes the PUBLIC domain and cannot see
|
||||
# the challenge backend, so it is evidence, not a verdict — and
|
||||
# a new `fail` here would block the site wizard on upgrade day
|
||||
# for every install.
|
||||
target["warn"] = True
|
||||
target["diagnosis"] = (
|
||||
"responded 200 with an HTML page, not a challenge token — "
|
||||
"the request is reaching a web UI instead of the ACME endpoint"
|
||||
)
|
||||
elif resp.status in (301, 302, 303, 307, 308):
|
||||
target["warn"] = True
|
||||
target["redirect_location"] = resp.headers.get("location")
|
||||
target["diagnosis"] = (
|
||||
"redirected instead of serving the challenge path"
|
||||
)
|
||||
else:
|
||||
target["ok"] = resp.status in (200, 404)
|
||||
targets.append(target)
|
||||
except asyncio.TimeoutError:
|
||||
targets.append({"domain": d, "error": "egress timeout", "warn": True})
|
||||
skip_reason = "egress timeout"
|
||||
@@ -425,7 +532,11 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
details={"targets": targets},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
if warns and not [t for t in targets if t.get("ok")]:
|
||||
# Surface warnings even when OTHER domains answered correctly. The old condition
|
||||
# ("warn only if nothing succeeded") hid the single most diagnostic outcome there
|
||||
# is: a multi-domain certificate where one name reaches a web UI instead of the
|
||||
# challenge endpoint reported a clean pass.
|
||||
if warns:
|
||||
# R18b audit fix (round 7): branch the rollup message on the
|
||||
# actual cause. Pre-fix the message was always "Egress to
|
||||
# port 80 appears blocked" — even when every target was
|
||||
@@ -435,6 +546,36 @@ async def check_port80(domains: List[str], *, http_timeout: float = 5.0) -> Dict
|
||||
# corporate firewall logs while the real cause was an
|
||||
# internal-only DNS A record. Also harden against
|
||||
# `skip_reason=None` so the message never reads "(None)".
|
||||
# Wrong-responder warnings take priority over every other cause: they are the
|
||||
# only ones that mean "your server answered, and answered wrong", which is a
|
||||
# different problem from "we could not test".
|
||||
wrong_responder = [t for t in targets if t.get("diagnosis")]
|
||||
if wrong_responder:
|
||||
first = wrong_responder[0]
|
||||
if first.get("body_class") == "html":
|
||||
human = (
|
||||
f"{first['domain']} answered HTTP {first.get('status')} with an HTML "
|
||||
f"page ({first.get('content_type') or 'unknown type'}, "
|
||||
f"{first.get('body_len')} bytes) instead of a challenge token. The "
|
||||
"path is reaching a web interface, not the ACME endpoint — check "
|
||||
"that the reverse proxy in front of OpenManager routes "
|
||||
"/.well-known/acme-challenge/ to the API."
|
||||
)
|
||||
else:
|
||||
human = (
|
||||
f"{first['domain']} answered HTTP {first.get('status')} "
|
||||
f"({first.get('diagnosis')})"
|
||||
)
|
||||
return _check_result(
|
||||
"port80",
|
||||
"Port 80 reachability",
|
||||
"warn",
|
||||
human,
|
||||
severity="warn",
|
||||
details={"targets": targets},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
ssrf_skip = any(
|
||||
"non-public" in (t.get("skip") or "")
|
||||
or "SSRF" in (t.get("skip") or "")
|
||||
@@ -491,11 +632,20 @@ async def check_routing(conn, domains: List[str], cluster_ids: List[int]) -> Dic
|
||||
duration_ms=int((time.time() - started) * 1000),
|
||||
)
|
||||
|
||||
# The WHERE clause is deliberately identical to the pre-existing one, so `not rows`
|
||||
# still means exactly what it meant before and the `fail` branch below cannot fire
|
||||
# in any situation where it previously passed. Narrowing it here (e.g. by adding a
|
||||
# mode filter) would turn a tcp-only port-80 cluster from "ok" into "fail", and the
|
||||
# site wizard blocks submit on any failing check — locking those installs the day
|
||||
# this ships. Mode is examined afterwards, in Python, and only ever downgrades to
|
||||
# `warn`.
|
||||
rows = await conn.fetch(
|
||||
"""
|
||||
SELECT id, name, bind_address, bind_port, mode, default_backend
|
||||
FROM frontends
|
||||
WHERE cluster_id = ANY($1::int[]) AND is_active = TRUE AND bind_port = 80
|
||||
SELECT f.id, f.name, f.bind_address, f.bind_port, f.mode, f.default_backend,
|
||||
f.cluster_id, c.acme_enabled
|
||||
FROM frontends f
|
||||
JOIN haproxy_clusters c ON c.id = f.cluster_id
|
||||
WHERE f.cluster_id = ANY($1::int[]) AND f.is_active = TRUE AND f.bind_port = 80
|
||||
""",
|
||||
cluster_ids,
|
||||
)
|
||||
@@ -510,13 +660,144 @@ async def check_routing(conn, domains: List[str], cluster_ids: List[int]) -> Dic
|
||||
details={"cluster_ids": cluster_ids},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
# Character-for-character the renderer's normalisation (services/haproxy_config.py),
|
||||
# so this can never disagree with what actually gets emitted.
|
||||
http_rows = [r for r in rows if (r["mode"] or "http").strip().lower() == "http"]
|
||||
if not http_rows:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
"The only port-80 frontend(s) in the target cluster(s) are in tcp mode. "
|
||||
"A tcp-mode frontend cannot carry the /.well-known/acme-challenge/ ACL, "
|
||||
"so HTTP-01 cannot be served — use DNS-01, or add an http-mode frontend "
|
||||
"on port 80."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows]},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
rows = http_rows
|
||||
|
||||
# A frontend row proves only that the DATABASE describes port-80 routing. The
|
||||
# renderer gates the challenge ACL on `acme_enabled`, and the nodes run whatever
|
||||
# config was last APPLIED — so the row said "ok" during an incident where the
|
||||
# live config had no usable challenge route at all. Check the two things the row
|
||||
# cannot tell us. Both report `warn`, never `fail`: the site wizard blocks submit
|
||||
# on any `fail`, so a new failing condition would lock every install on the day
|
||||
# it ships.
|
||||
# `.get()` rather than `[]`: the column gates a WARNING, so a row shape without
|
||||
# it should not blow up the whole diagnostic. Absent means "assume enabled" —
|
||||
# the applied-config check below is the authoritative one either way.
|
||||
acme_off = sorted({r["cluster_id"] for r in rows if not r.get("acme_enabled", True)})
|
||||
if acme_off:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {acme_off} have ACME Challenge Routing disabled, so the "
|
||||
"generated config contains no /.well-known/acme-challenge/ route. "
|
||||
"Enable it in Cluster Management, then apply the cluster."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "acme_disabled_clusters": acme_off},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
missing_in_applied = []
|
||||
challenge_backends = {}
|
||||
for cluster_id in sorted({r["cluster_id"] for r in rows}):
|
||||
# Selector matched to the one the AGENT uses to fetch its config
|
||||
# (routers/agent.py: status='APPLIED' AND is_active=TRUE), because the question
|
||||
# here is "what are the nodes running right now?". Without `is_active` this can
|
||||
# read a superseded row and report on a config that was never delivered.
|
||||
# Extracting in SQL rather than pulling whole configs back per cluster: these
|
||||
# files run to hundreds of KB on real installs.
|
||||
applied = await conn.fetchrow(
|
||||
"""
|
||||
SELECT position('use_backend _acme_challenge_backend' in config_content) > 0
|
||||
AS has_route,
|
||||
substring(config_content from 'server _acme_mgmt [^\\n]*') AS server_line
|
||||
FROM config_versions
|
||||
WHERE cluster_id = $1 AND status = 'APPLIED' AND is_active = TRUE
|
||||
AND config_content IS NOT NULL
|
||||
ORDER BY created_at DESC LIMIT 1
|
||||
""",
|
||||
cluster_id,
|
||||
)
|
||||
if not applied or not applied["has_route"]:
|
||||
missing_in_applied.append(cluster_id)
|
||||
continue
|
||||
server_line = (applied["server_line"] or "").strip()
|
||||
challenge_backends[cluster_id] = (
|
||||
server_line[len("server _acme_mgmt "):].strip() if server_line else None
|
||||
)
|
||||
|
||||
if missing_in_applied:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {missing_in_applied} have no applied configuration carrying "
|
||||
"the challenge route. The change exists in the database but the nodes are "
|
||||
"still running an older config — apply the cluster."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "clusters_not_applied": missing_in_applied},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
# A backend section with no `server` line: the route exists, `haproxy -c` passes,
|
||||
# and every challenge request gets a 503 from an empty backend. Without this branch
|
||||
# the falsy target slips past the loopback filter below and the check reports "ok".
|
||||
serverless = sorted(cid for cid, target in challenge_backends.items() if not target)
|
||||
if serverless:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"Cluster(s) {serverless} route the challenge path to a backend that has "
|
||||
"no server line, so every request returns 503. The configured ACME "
|
||||
"Challenge Backend URL could not be resolved into an address — check it "
|
||||
"in Cluster Management, or Settings > ACME for the global value."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "clusters_without_server": serverless},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
loopback = {
|
||||
cid: target for cid, target in challenge_backends.items()
|
||||
if target and target.split(":")[0].strip("[]").lower()
|
||||
in ("localhost", "127.0.0.1", "::1", "0.0.0.0")
|
||||
}
|
||||
if loopback:
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"warn",
|
||||
(
|
||||
f"The applied config points the challenge backend at {sorted(loopback.values())}. "
|
||||
"HAProxy resolves that on the HAProxy node, so it means the node itself, not "
|
||||
"this management server. Set ACME Challenge Backend URL to a routable address."
|
||||
),
|
||||
severity="warn",
|
||||
details={"frontends": [dict(r) for r in rows], "challenge_backends": challenge_backends},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
return _check_result(
|
||||
"routing",
|
||||
"HAProxy routing",
|
||||
"ok",
|
||||
f"Found {len(rows)} HTTP frontend(s) on port 80",
|
||||
f"Found {len(rows)} HTTP frontend(s) on port 80; challenge route present in applied config",
|
||||
severity="info",
|
||||
details={"frontends": [dict(r) for r in rows]},
|
||||
details={"frontends": [dict(r) for r in rows], "challenge_backends": challenge_backends},
|
||||
duration_ms=duration_ms,
|
||||
)
|
||||
|
||||
|
||||
@@ -75,16 +75,23 @@ class ACMEService:
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
await assert_public_url(directory_url)
|
||||
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.get(directory_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
if resp.status != 200:
|
||||
raise Exception(f"Failed to fetch ACME directory: HTTP {resp.status}")
|
||||
data = await resp.json()
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
data['_fetched_at'] = time.time()
|
||||
self._directory_cache[directory_url] = data
|
||||
return data
|
||||
# v1.11.0: recorded in request_logs as an outbound call so an operator
|
||||
# can see exactly which CA was contacted and what it answered.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_ACME
|
||||
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url=directory_url) as span:
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.get(directory_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
if resp.status != 200:
|
||||
span.set_response(resp.status, dict(resp.headers))
|
||||
raise Exception(f"Failed to fetch ACME directory: HTTP {resp.status}")
|
||||
data = await resp.json()
|
||||
span.set_response(resp.status, dict(resp.headers), data)
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
data['_fetched_at'] = time.time()
|
||||
self._directory_cache[directory_url] = data
|
||||
return data
|
||||
|
||||
async def _get_nonce(self, directory_url: str) -> str:
|
||||
# Use a cached nonce for THIS CA only; otherwise fetch a fresh one from THIS CA's newNonce.
|
||||
@@ -104,9 +111,19 @@ class ACMEService:
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
nonce_url = directory['newNonce']
|
||||
await assert_public_url(nonce_url)
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.head(nonce_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
return resp.headers['Replay-Nonce']
|
||||
|
||||
# v1.11.0: a HEAD with no body and no status check — capture the status
|
||||
# and the allowlisted headers only. `Replay-Nonce` itself is redacted by
|
||||
# the header rules: it is a single-use credential.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_ACME
|
||||
|
||||
async with outbound_span(
|
||||
target=TARGET_ACME, method="HEAD", url=nonce_url, capture_body=False
|
||||
) as span:
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
async with session.head(nonce_url, timeout=aiohttp.ClientTimeout(total=15), allow_redirects=False) as resp:
|
||||
span.set_response(resp.status, dict(resp.headers))
|
||||
return resp.headers['Replay-Nonce']
|
||||
|
||||
def _generate_account_key(self) -> Tuple[str, dict]:
|
||||
private_key = rsa.generate_private_key(
|
||||
@@ -208,46 +225,75 @@ class ACMEService:
|
||||
from utils.ssrf_guard import assert_public_url, safe_connector
|
||||
await assert_public_url(url)
|
||||
|
||||
# v1.11.0: instrument each ATTEMPT separately (the span goes inside the
|
||||
# retry loop, the session stays outside it) so a badNonce retry shows up
|
||||
# as its own row instead of being folded into the successful one.
|
||||
#
|
||||
# capture_body=False is mandatory here. The JWS body is
|
||||
# {protected, payload, signature}: `protected` carries the nonce and the
|
||||
# account kid/jwk, and `signature` is made with the account private key.
|
||||
# The key itself never crosses the wire, but a stored (protected,
|
||||
# signature) pair is a REPLAYABLE ACME credential for the lifetime of the
|
||||
# nonce. We log a description of the request instead of the request.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_ACME
|
||||
|
||||
async with aiohttp.ClientSession(connector=safe_connector()) as session:
|
||||
for attempt in range(3):
|
||||
async with session.post(
|
||||
url,
|
||||
json=body,
|
||||
headers={"Content-Type": "application/jose+json"},
|
||||
timeout=aiohttp.ClientTimeout(total=30),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
jws_summary = {
|
||||
"jws": True,
|
||||
"acme_url": protected.get("url"),
|
||||
"kid_present": bool(protected.get("kid")),
|
||||
"jwk_present": bool(protected.get("jwk")),
|
||||
"payload_empty": payload == "",
|
||||
"attempt": attempt + 1,
|
||||
}
|
||||
async with outbound_span(
|
||||
target=TARGET_ACME,
|
||||
method="POST",
|
||||
url=url,
|
||||
request_body=jws_summary,
|
||||
capture_body=False,
|
||||
) as span:
|
||||
async with session.post(
|
||||
url,
|
||||
json=body,
|
||||
headers={"Content-Type": "application/jose+json"},
|
||||
timeout=aiohttp.ClientTimeout(total=30),
|
||||
allow_redirects=False,
|
||||
) as resp:
|
||||
if 'Replay-Nonce' in resp.headers:
|
||||
self._nonce_by_dir[directory_url] = resp.headers['Replay-Nonce']
|
||||
|
||||
if resp.status == 400 and attempt < 2:
|
||||
err = await resp.json()
|
||||
etype = (err.get('type') or '')
|
||||
edetail = (err.get('detail') or '').lower()
|
||||
# Retry on badNonce, and on any nonce-related malformed rejection (e.g.
|
||||
# "The Replay Nonce could not be base64url-decoded") — refetch a FRESH nonce
|
||||
# from the target CA and resign. With per-CA scoping the cross-CA cause is gone;
|
||||
# this is defense-in-depth so a stale/rejected nonce always self-heals.
|
||||
if etype.endswith('badNonce') or 'nonce' in edetail:
|
||||
nonce = resp.headers.get('Replay-Nonce') or await self._get_nonce(directory_url)
|
||||
protected['nonce'] = nonce
|
||||
body = self._sign_jws(private_key, protected, payload)
|
||||
continue
|
||||
if resp.status == 400 and attempt < 2:
|
||||
err = await resp.json()
|
||||
etype = (err.get('type') or '')
|
||||
edetail = (err.get('detail') or '').lower()
|
||||
# Retry on badNonce, and on any nonce-related malformed rejection (e.g.
|
||||
# "The Replay Nonce could not be base64url-decoded") — refetch a FRESH nonce
|
||||
# from the target CA and resign. With per-CA scoping the cross-CA cause is gone;
|
||||
# this is defense-in-depth so a stale/rejected nonce always self-heals.
|
||||
if etype.endswith('badNonce') or 'nonce' in edetail:
|
||||
span.set_response(resp.status, dict(resp.headers), err)
|
||||
nonce = resp.headers.get('Replay-Nonce') or await self._get_nonce(directory_url)
|
||||
protected['nonce'] = nonce
|
||||
body = self._sign_jws(private_key, protected, payload)
|
||||
continue
|
||||
|
||||
resp_data = {}
|
||||
content_type = resp.headers.get('Content-Type', '')
|
||||
if 'json' in content_type:
|
||||
resp_data = await resp.json()
|
||||
elif resp.status < 300:
|
||||
text = await resp.text()
|
||||
if text:
|
||||
try:
|
||||
resp_data = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
resp_data = {"raw": text}
|
||||
resp_data = {}
|
||||
content_type = resp.headers.get('Content-Type', '')
|
||||
if 'json' in content_type:
|
||||
resp_data = await resp.json()
|
||||
elif resp.status < 300:
|
||||
text = await resp.text()
|
||||
if text:
|
||||
try:
|
||||
resp_data = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
resp_data = {"raw": text}
|
||||
|
||||
headers = dict(resp.headers)
|
||||
return resp.status, resp_data, headers
|
||||
headers = dict(resp.headers)
|
||||
span.set_response(resp.status, headers, resp_data)
|
||||
return resp.status, resp_data, headers
|
||||
|
||||
raise Exception(f"ACME request to {url} failed after retries")
|
||||
|
||||
|
||||
@@ -19,8 +19,14 @@ subject instead of CN-only, ECDSA support, same PKCS8/NoEncryption key
|
||||
serialisation (the agent concatenates cert+key+chain into one PEM and HAProxy
|
||||
cannot read passphrase-protected keys).
|
||||
|
||||
Private keys are stored PLAINTEXT, consistent with every other key in the
|
||||
system (ssl_certificates.private_key_content, letsencrypt_orders.cert_private_key).
|
||||
Private keys are ENCRYPTED AT REST from v1.10.1 (Issue #53): the Fernet token
|
||||
replaces the PEM in the same `ssl_csrs.private_key_pem` column, so there is no
|
||||
schema change and no SCHEMA_VERSION bump. Rows written earlier hold a raw PEM
|
||||
and are still read transparently — see utils/csr_key_crypto.py for the format
|
||||
discriminator and the key-rotation caveat. The pending CSR key is the one key
|
||||
in the system worth encrypting: it sits idle for the whole signing window and
|
||||
is never transmitted, unlike ssl_certificates.private_key_content and the ACME
|
||||
order keys, which agents must receive in plaintext on every poll.
|
||||
The key is NEVER returned by any CSR API response — `csr_row_to_dict` strips
|
||||
it unconditionally.
|
||||
"""
|
||||
@@ -39,6 +45,7 @@ from cryptography.hazmat.primitives.asymmetric import ec, rsa
|
||||
from cryptography.x509.oid import NameOID
|
||||
|
||||
from services import ssl_service
|
||||
from utils.csr_key_crypto import decrypt_csr_private_key, encrypt_csr_private_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -186,8 +193,14 @@ async def assert_csr_name_available(conn, name: str) -> None:
|
||||
|
||||
|
||||
async def insert_csr_row(conn, payload: Any, bundle: Dict[str, Any], user_id: Optional[int]) -> int:
|
||||
"""Persist a freshly generated CSR bundle. Returns the new csr id."""
|
||||
"""Persist a freshly generated CSR bundle. Returns the new csr id.
|
||||
|
||||
Issue #53 (v1.10.1): the private key is Fernet-encrypted before it is stored. The token goes
|
||||
into the SAME private_key_pem TEXT column — no schema change — and is only ever decrypted
|
||||
in-process by import_signed_certificate. No CSR endpoint returns the column either way.
|
||||
"""
|
||||
await assert_csr_name_available(conn, payload.name)
|
||||
stored_key = encrypt_csr_private_key(bundle['private_key_pem'])
|
||||
try:
|
||||
csr_id = await conn.fetchval(
|
||||
"""
|
||||
@@ -203,7 +216,7 @@ async def insert_csr_row(conn, payload: Any, bundle: Dict[str, Any], user_id: Op
|
||||
json.dumps(bundle['sans']),
|
||||
payload.key_algorithm,
|
||||
bundle['csr_pem'],
|
||||
bundle['private_key_pem'],
|
||||
stored_key,
|
||||
user_id,
|
||||
)
|
||||
except asyncpg.exceptions.UniqueViolationError:
|
||||
@@ -243,8 +256,7 @@ async def import_signed_certificate(conn, csr_id: int, imp: Any, user_id: Option
|
||||
"Create a new CSR to reissue."
|
||||
),
|
||||
)
|
||||
stored_key = row['private_key_pem']
|
||||
if not stored_key:
|
||||
if not row['private_key_pem']:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
@@ -252,6 +264,23 @@ async def import_signed_certificate(conn, csr_id: int, imp: Any, user_id: Option
|
||||
"corrupt. Delete it and create a new CSR."
|
||||
),
|
||||
)
|
||||
# Issue #53: the column holds a Fernet token from v1.10.1 on, and a raw PEM for rows
|
||||
# written before it. decrypt_csr_private_key accepts both, so no data migration is
|
||||
# needed. A None here means the token cannot be decrypted — SECRET_KEY was rotated
|
||||
# without CSR_ENCRYPTION_KEY set. Fail loudly: the key is gone, so the CA's certificate
|
||||
# can never be paired with it, and silently falling through would surface as the far
|
||||
# more confusing "certificate does not match this CSR's private key".
|
||||
stored_key = decrypt_csr_private_key(row['private_key_pem'])
|
||||
if not stored_key:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=(
|
||||
f"The stored private key for CSR '{row['name']}' cannot be decrypted. This "
|
||||
"happens when SECRET_KEY was rotated while CSR_ENCRYPTION_KEY was not set. "
|
||||
"The key is unrecoverable, so this CSR can no longer be completed — delete "
|
||||
"it and create a new one (then have the new CSR signed)."
|
||||
),
|
||||
)
|
||||
|
||||
effective_name = getattr(imp, 'name', None) or row['name']
|
||||
|
||||
|
||||
@@ -73,26 +73,42 @@ class CloudflareDNSProvider(DnsProvider):
|
||||
"""One Cloudflare API call. Returns the parsed JSON body. Raises a SANITIZED
|
||||
DnsProviderError on transport/HTTP/API error (never echoes the token or raw headers)."""
|
||||
url = f"{CLOUDFLARE_API_BASE}{path}"
|
||||
# v1.11.0: single funnel for every Cloudflare call, so instrumenting here
|
||||
# covers all five logical endpoints. `safe_error_only=True` records only
|
||||
# the exception TYPE — the same stance the handlers below already take,
|
||||
# because a raw message can carry the request URL and through it the zone
|
||||
# identifier. The Authorization header is dropped to a presence marker by
|
||||
# the header allowlist.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_DNS_CLOUDFLARE
|
||||
|
||||
try:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
body = await resp.json()
|
||||
except Exception: # noqa: BLE001
|
||||
body = {}
|
||||
if resp.status in (401, 403):
|
||||
raise DnsProviderError("Cloudflare rejected the API token (check it has Zone:DNS:Edit + Zone:Read).")
|
||||
if resp.status >= 400 or not body.get("success", False):
|
||||
# Cloudflare returns {"errors":[{"code":..,"message":..}]} — surface only the
|
||||
# human message text, never the request (which carries the token header).
|
||||
msgs = "; ".join(
|
||||
str(e.get("message")) for e in (body.get("errors") or []) if e.get("message")
|
||||
)
|
||||
raise DnsProviderError(
|
||||
f"Cloudflare API error (HTTP {resp.status}){': ' + msgs if msgs else ''}"
|
||||
)
|
||||
return body
|
||||
async with outbound_span(
|
||||
target=TARGET_DNS_CLOUDFLARE,
|
||||
method=method,
|
||||
url=url,
|
||||
request_body=kwargs.get("json"),
|
||||
safe_error_only=True,
|
||||
) as span:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
body = await resp.json()
|
||||
except Exception: # noqa: BLE001
|
||||
body = {}
|
||||
span.set_response(resp.status, getattr(resp, "headers", None), body)
|
||||
if resp.status in (401, 403):
|
||||
raise DnsProviderError("Cloudflare rejected the API token (check it has Zone:DNS:Edit + Zone:Read).")
|
||||
if resp.status >= 400 or not body.get("success", False):
|
||||
# Cloudflare returns {"errors":[{"code":..,"message":..}]} — surface only the
|
||||
# human message text, never the request (which carries the token header).
|
||||
msgs = "; ".join(
|
||||
str(e.get("message")) for e in (body.get("errors") or []) if e.get("message")
|
||||
)
|
||||
raise DnsProviderError(
|
||||
f"Cloudflare API error (HTTP {resp.status}){': ' + msgs if msgs else ''}"
|
||||
)
|
||||
return body
|
||||
except DnsProviderError:
|
||||
raise
|
||||
except aiohttp.ClientError as exc:
|
||||
|
||||
@@ -343,30 +343,45 @@ class GoDaddyDNSProvider(DnsProvider):
|
||||
request, never a response body verbatim.
|
||||
"""
|
||||
url = f"{GODADDY_API_BASE}{path}"
|
||||
# v1.11.0: single funnel for every GoDaddy call. `safe_error_only=True`
|
||||
# keeps the recorded error to the exception TYPE, matching the stance the
|
||||
# handlers below already take — a raw message can carry the request URL.
|
||||
# The `Authorization: sso-key <key>:<secret>` header never reaches the log:
|
||||
# the header allowlist reduces it to a presence marker.
|
||||
from utils.http_instrumentation import outbound_span, TARGET_DNS_GODADDY
|
||||
|
||||
try:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
# content_type=None: every GoDaddy write answers 200/204 with an EMPTY body, and
|
||||
# aiohttp would otherwise raise on the missing/other content type before parsing.
|
||||
body = await resp.json(content_type=None)
|
||||
except ValueError:
|
||||
# ONLY a decode failure (JSONDecodeError subclasses ValueError) is swallowed —
|
||||
# an empty write body, or an HTML error page on a >=400. A transport failure
|
||||
# mid-read (ClientPayloadError, TimeoutError) must NOT land here: it would look
|
||||
# identical to "empty body", and a caller that reads an RRset would then see
|
||||
# None and could mistake it for an empty RRset. Those propagate to the handlers
|
||||
# below and become a real DnsProviderError.
|
||||
body = None
|
||||
# 2xx only. Redirects are deliberately not followed (aiohttp would forward the
|
||||
# Authorization header), so a 3xx is a failed call — treating `< 400` as success
|
||||
# would report a redirected write as a silent no-op.
|
||||
if 200 <= resp.status < 300:
|
||||
return body
|
||||
code, message = _error_fields(body)
|
||||
retry_after = _retry_after_seconds(resp.headers, body) if resp.status == 429 else None
|
||||
raise self._http_error(resp.status, code, message, retry_after)
|
||||
async with outbound_span(
|
||||
target=TARGET_DNS_GODADDY,
|
||||
method=method,
|
||||
url=url,
|
||||
request_body=kwargs.get("json"),
|
||||
safe_error_only=True,
|
||||
) as span:
|
||||
async with session.request(
|
||||
method, url, headers=self._headers(), allow_redirects=False, **kwargs
|
||||
) as resp:
|
||||
try:
|
||||
# content_type=None: every GoDaddy write answers 200/204 with an EMPTY body, and
|
||||
# aiohttp would otherwise raise on the missing/other content type before parsing.
|
||||
body = await resp.json(content_type=None)
|
||||
except ValueError:
|
||||
# ONLY a decode failure (JSONDecodeError subclasses ValueError) is swallowed —
|
||||
# an empty write body, or an HTML error page on a >=400. A transport failure
|
||||
# mid-read (ClientPayloadError, TimeoutError) must NOT land here: it would look
|
||||
# identical to "empty body", and a caller that reads an RRset would then see
|
||||
# None and could mistake it for an empty RRset. Those propagate to the handlers
|
||||
# below and become a real DnsProviderError.
|
||||
body = None
|
||||
span.set_response(resp.status, getattr(resp, "headers", None), body)
|
||||
# 2xx only. Redirects are deliberately not followed (aiohttp would forward the
|
||||
# Authorization header), so a 3xx is a failed call — treating `< 400` as success
|
||||
# would report a redirected write as a silent no-op.
|
||||
if 200 <= resp.status < 300:
|
||||
return body
|
||||
code, message = _error_fields(body)
|
||||
retry_after = _retry_after_seconds(resp.headers, body) if resp.status == 429 else None
|
||||
raise self._http_error(resp.status, code, message, retry_after)
|
||||
except DnsProviderError:
|
||||
raise
|
||||
except aiohttp.ClientError as exc:
|
||||
|
||||
@@ -6,9 +6,98 @@ import json
|
||||
import urllib.parse
|
||||
from typing import Optional, List, Dict, Any
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.acme_backend_url import resolve_acme_backend_target
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Sentinel prefixes returned by `generate_haproxy_config_for_cluster` INSTEAD of a
|
||||
# configuration when generation fails. They are plain strings (not exceptions) for
|
||||
# historical reasons: the function's outer `except` swallows everything and returns
|
||||
# a one-line comment as the "config".
|
||||
#
|
||||
# That is a data-loss primitive on its own: an exception anywhere in the generator —
|
||||
# e.g. `urllib.parse.urlparse('http://host:99999').port` raising ValueError for an
|
||||
# out-of-range port in `acme_backend_url` — collapses a whole cluster's haproxy.cfg
|
||||
# into a single comment line, which the apply path then stores as APPLIED and pushes
|
||||
# to every agent. Callers that PERSIST the returned text MUST reject it first; use
|
||||
# `is_config_generation_error()` rather than matching the string by hand.
|
||||
CONFIG_GENERATION_ERROR_PREFIXES = (
|
||||
"# Error generating configuration:",
|
||||
"# Error: Cluster not found",
|
||||
)
|
||||
|
||||
|
||||
def is_config_generation_error(config_content: Optional[str]) -> bool:
|
||||
"""True when `config_content` is a generator failure sentinel, not a configuration.
|
||||
|
||||
Persisting or shipping a sentinel silently destroys a cluster's configuration, so
|
||||
every call site that writes the generator's output to `config_versions` (or hands
|
||||
it to an agent) must guard with this.
|
||||
"""
|
||||
if not config_content:
|
||||
return True
|
||||
return config_content.lstrip().startswith(CONFIG_GENERATION_ERROR_PREFIXES)
|
||||
|
||||
|
||||
def select_acme_backend_source(candidates: List[tuple]):
|
||||
"""Pick the first candidate that resolves into a usable address.
|
||||
|
||||
`candidates` is an ordered list of ``(source_name, url)`` from most to least
|
||||
specific. Returns ``(source_name, url, target, skipped)`` where ``skipped`` lists
|
||||
the ``(source_name, url, target)`` of candidates that were rejected.
|
||||
|
||||
Falling through on UNUSABLE values, not just empty ones, is the point. Values
|
||||
predating validation are common — the settings field was free text — and a
|
||||
scheme-less ``10.90.1.4:8080`` cannot be resolved. Stopping at the first non-empty
|
||||
candidate would emit a backend section with no ``server`` line: ``haproxy -c``
|
||||
still passes because the section exists, Apply succeeds, and every challenge
|
||||
request then 503s from an empty backend with nothing to show for it.
|
||||
"""
|
||||
skipped = []
|
||||
for source, url in candidates:
|
||||
if not url:
|
||||
continue
|
||||
target = resolve_acme_backend_target(url)
|
||||
if target.error_code:
|
||||
skipped.append((source, url, target))
|
||||
continue
|
||||
return source, url, target, skipped
|
||||
|
||||
# Nothing usable. Report against the last non-empty candidate so the rendered
|
||||
# comment and the log name a concrete value rather than an empty one.
|
||||
if skipped:
|
||||
source, url, target = skipped[-1]
|
||||
return source, url, target, skipped[:-1]
|
||||
source, url = candidates[0] if candidates else ('none', '')
|
||||
return source, url, resolve_acme_backend_target(url), skipped
|
||||
|
||||
|
||||
def extract_acme_backend_target(config_content: Optional[str]) -> Optional[str]:
|
||||
"""Return the `server _acme_mgmt` argument string from a rendered config.
|
||||
|
||||
e.g. ``"10.90.1.4:80"`` or ``"mgmt.internal:443 ssl verify none"``; ``None`` when
|
||||
the cluster renders no ACME challenge backend at all.
|
||||
|
||||
Used to decide whether an ACME-related edit actually CHANGES the shipped
|
||||
configuration. Comparing whole config texts would report a difference on every
|
||||
unrelated pending edit; comparing this one line answers the only question that
|
||||
matters here — "would the HAProxy nodes start talking to a different address?"
|
||||
"""
|
||||
if not config_content:
|
||||
return None
|
||||
in_section = False
|
||||
for line in config_content.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("backend "):
|
||||
in_section = stripped == "backend _acme_challenge_backend"
|
||||
continue
|
||||
if stripped.startswith(("frontend ", "listen ", "defaults", "global")):
|
||||
in_section = False
|
||||
continue
|
||||
if in_section and stripped.startswith("server _acme_mgmt "):
|
||||
return stripped[len("server _acme_mgmt "):].strip()
|
||||
return None
|
||||
|
||||
|
||||
def _format_redirect_rule(rule: Any) -> Optional[str]:
|
||||
"""Render a single redirect rule into a HAProxy `redirect ...` line.
|
||||
@@ -878,7 +967,15 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
f" bind {frontend['bind_address']}:{frontend['bind_port']}"
|
||||
)
|
||||
|
||||
config_lines.append(f" mode {frontend['mode']}")
|
||||
# `frontends.mode` is `VARCHAR(10) DEFAULT 'http'` but NULLABLE, and rows
|
||||
# can arrive with it unset via agent sync or config import. Interpolating
|
||||
# the raw value then emits a literal `mode None`, which HAProxy rejects —
|
||||
# taking down the whole cluster config, not just this frontend. Normalise
|
||||
# once here and use the result everywhere below, so the ACME gate, the
|
||||
# backend-mode check and the rendered line can never disagree with each
|
||||
# other about what mode this frontend is in.
|
||||
frontend_mode = (frontend.get('mode') or 'http').strip().lower()
|
||||
config_lines.append(f" mode {frontend_mode}")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# R2.3 / R3.3 (PR-1 hotfix): emit ordering buckets.
|
||||
@@ -990,7 +1087,7 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
_fe_buckets[cat].append(line)
|
||||
|
||||
# ACME HTTP-01 Challenge routing (auto-managed)
|
||||
if frontend['mode'] == 'http' and cluster_info.get('acme_enabled', False):
|
||||
if frontend_mode == 'http' and cluster_info.get('acme_enabled', False):
|
||||
_emit_fe(" acl is_acme_challenge path_beg /.well-known/acme-challenge/")
|
||||
_emit_fe(" http-request allow if is_acme_challenge")
|
||||
_emit_fe(" use_backend _acme_challenge_backend if is_acme_challenge")
|
||||
@@ -1020,14 +1117,14 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
if default_backend_name and default_backend_name not in ('[]', '{}', 'null', 'None'):
|
||||
backend_mode = backend_modes.get(default_backend_name)
|
||||
|
||||
if backend_mode and backend_mode != frontend['mode']:
|
||||
logger.error(f"CONFIG ERROR: Frontend '{frontend['name']}' mode '{frontend['mode']}' does not match backend '{default_backend_name}' mode '{backend_mode}'")
|
||||
if backend_mode and backend_mode != frontend_mode:
|
||||
logger.error(f"CONFIG ERROR: Frontend '{frontend['name']}' mode '{frontend_mode}' does not match backend '{default_backend_name}' mode '{backend_mode}'")
|
||||
# FIX-10 marker: 'BACKEND-MODE-WARNING' keyword in
|
||||
# the comment body routes it to the 'default_be'
|
||||
# bucket via _categorize_haproxy_directive, so the
|
||||
# warning emits next to the actual default_backend
|
||||
# directive instead of at the top of the block.
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: Backend '{default_backend_name}' has mode '{backend_mode}' but frontend has mode '{frontend['mode']}'")
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: Backend '{default_backend_name}' has mode '{backend_mode}' but frontend has mode '{frontend_mode}'")
|
||||
_emit_fe(f" # BACKEND-MODE-WARNING: HAProxy will reject this configuration! Please fix the mode mismatch in UI.")
|
||||
|
||||
_emit_fe(f" default_backend {default_backend_name}")
|
||||
@@ -1589,36 +1686,82 @@ async def generate_haproxy_config_for_cluster(cluster_id: int, conn: Optional[An
|
||||
f"frontend found (would create orphan backend section)."
|
||||
)
|
||||
else:
|
||||
acme_url = cluster_info.get('acme_backend_url') or ''
|
||||
if not acme_url:
|
||||
try:
|
||||
acme_settings = await db_conn.fetchrow(
|
||||
"SELECT value FROM system_settings WHERE key = 'acme.challenge_backend_url'"
|
||||
)
|
||||
if acme_settings and acme_settings['value']:
|
||||
val = acme_settings['value']
|
||||
if isinstance(val, str):
|
||||
try:
|
||||
val = json.loads(val)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if val:
|
||||
acme_url = str(val)
|
||||
except Exception:
|
||||
pass
|
||||
if not acme_url:
|
||||
from config import MANAGEMENT_BASE_URL
|
||||
acme_url = MANAGEMENT_BASE_URL
|
||||
# Track WHERE the effective URL came from. Support has no way today to
|
||||
# tell an operator-set value from the shipped `localhost` default, and
|
||||
# this whole block emits no log line at all (contrast the skip branches
|
||||
# above), so a wrong challenge backend is invisible until Let's Encrypt
|
||||
# fails. `acme_source` is logged with the rendered host:port below.
|
||||
acme_candidates = [
|
||||
('cluster.acme_backend_url', cluster_info.get('acme_backend_url') or '')
|
||||
]
|
||||
_settings_url = ''
|
||||
try:
|
||||
acme_settings = await db_conn.fetchrow(
|
||||
"SELECT value FROM system_settings WHERE key = 'acme.challenge_backend_url'"
|
||||
)
|
||||
if acme_settings and acme_settings['value']:
|
||||
val = acme_settings['value']
|
||||
if isinstance(val, str):
|
||||
try:
|
||||
val = json.loads(val)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if val:
|
||||
_settings_url = str(val)
|
||||
except Exception:
|
||||
pass
|
||||
acme_candidates.append(
|
||||
('system_settings.acme.challenge_backend_url', _settings_url)
|
||||
)
|
||||
from config import MANAGEMENT_BASE_URL
|
||||
acme_candidates.append(('config.MANAGEMENT_BASE_URL', MANAGEMENT_BASE_URL))
|
||||
|
||||
parsed = urllib.parse.urlparse(acme_url)
|
||||
host = parsed.hostname or 'localhost'
|
||||
port = parsed.port or (443 if parsed.scheme == 'https' else 8080)
|
||||
ssl_flag = ' ssl verify none' if parsed.scheme == 'https' else ''
|
||||
acme_source, acme_url, target, _skipped = select_acme_backend_source(
|
||||
acme_candidates
|
||||
)
|
||||
for _s_source, _s_url, _s_target in _skipped:
|
||||
logger.warning(
|
||||
f"ACME-BACKEND: cluster {cluster_id} skipping unusable value from "
|
||||
f"{_s_source} (reason={_s_target.error_code}): "
|
||||
f"{_s_target.error_message} value={_s_url!r}"
|
||||
)
|
||||
|
||||
config_lines.append("# ACME Challenge Backend (auto-managed by HAProxy OpenManager)")
|
||||
config_lines.append("backend _acme_challenge_backend")
|
||||
config_lines.append(" mode http")
|
||||
config_lines.append(f" server _acme_mgmt {host}:{port}{ssl_flag}")
|
||||
if target.error_code:
|
||||
# The value cannot produce an address. Emit the section without a
|
||||
# `server` line rather than guessing: every HTTP frontend already
|
||||
# carries `use_backend _acme_challenge_backend`, and a use_backend
|
||||
# with no matching backend is fatal to `haproxy -c`. The operator's
|
||||
# raw value is deliberately NOT echoed into the file — a value
|
||||
# containing a newline would inject directives into a config pushed
|
||||
# to every node. It goes to the log instead.
|
||||
config_lines.append(
|
||||
f" # ACME challenge backend unavailable ({target.error_code}) — "
|
||||
f"see Cluster Management > ACME Challenge Backend URL"
|
||||
)
|
||||
logger.error(
|
||||
f"ACME-BACKEND: cluster {cluster_id} has an unusable challenge backend "
|
||||
f"URL (source={acme_source}, reason={target.error_code}): "
|
||||
f"{target.error_message} value={acme_url!r}"
|
||||
)
|
||||
else:
|
||||
config_lines.append(
|
||||
f" server _acme_mgmt {target.host}:{target.port}{target.ssl_flag}"
|
||||
)
|
||||
# One greppable line per render. `ACME-BACKEND` is the support
|
||||
# keyword: it answers "what address did we actually ship, and who
|
||||
# chose it?" without shell access to the node.
|
||||
logger.info(
|
||||
f"ACME-BACKEND: cluster {cluster_id} challenge backend rendered as "
|
||||
f"{target.host}:{target.port}{target.ssl_flag} "
|
||||
f"(source={acme_source}, url={acme_url!r})"
|
||||
)
|
||||
for _warning in target.warnings:
|
||||
logger.warning(
|
||||
f"ACME-BACKEND: cluster {cluster_id} (source={acme_source}): {_warning}"
|
||||
)
|
||||
config_lines.append("")
|
||||
|
||||
# Only close the connection if it was created within this function
|
||||
|
||||
@@ -0,0 +1,533 @@
|
||||
"""Issue #27 follow-up — parse an EXISTING keepalived.conf so a hand-maintained VIP can be
|
||||
adopted into OpenManager's model (v1.10.4).
|
||||
|
||||
Standalone and DB-free, like keepalived_config.py: the agent reports the file it found on a
|
||||
node, this module turns it into the fields `vip_instances` / `vip_members` need, and the
|
||||
adoption endpoint decides whether taking ownership is safe.
|
||||
|
||||
WHY A PARSER AND NOT THE HEARTBEAT. The heartbeat carries two keepalived facts —
|
||||
`keepalive_state` (MASTER/BACKUP, best-effort from logs) and `keepalive_ip` (the first
|
||||
address grepped out of `virtual_ipaddress`). Rendering a node's config needs eleven:
|
||||
virtual_router_id, auth_pass, interface, priority, prefix_length, advert_int, unicast
|
||||
peers, track_haproxy, role and the address itself. Guessing the missing ones is not a
|
||||
cosmetic risk — a wrong VRID puts the nodes in two separate VRRP domains and a wrong
|
||||
auth_pass makes them reject each other, and either way both nodes claim the VIP.
|
||||
|
||||
THE SAFETY CONTRACT. Adoption REPLACES the operator's file with our render, so anything in
|
||||
their file that `render_keepalived_conf` cannot reproduce would be silently destroyed on
|
||||
takeover — a `notify_master` failover hook, an LVS `virtual_server` section, a second
|
||||
address in one instance, a sync group. Extracting the fields is the easy half; the half
|
||||
that matters is `unsupported`, the list of directives we would drop. The caller must treat
|
||||
a non-empty `unsupported` as a refusal to adopt, not a warning to log.
|
||||
|
||||
Secrets: a parsed instance carries `auth_pass` in cleartext because that is the only way to
|
||||
re-render an identical config. NEVER log a parse result. Callers persist it through
|
||||
`encrypt_vrrp_secret` and mask it in anything UI-facing, exactly as the VIP router already
|
||||
does for `auth_pass` in version diffs.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
# Directives `render_keepalived_conf` emits, and therefore the only ones a takeover can
|
||||
# reproduce. Anything else found in a vrrp_instance is reported in `unsupported`.
|
||||
_SUPPORTED_INSTANCE_KEYS = {
|
||||
"state", "interface", "virtual_router_id", "priority", "advert_int",
|
||||
"authentication", "unicast_src_ip", "unicast_peer", "virtual_ipaddress", "track_script",
|
||||
}
|
||||
# Top-level blocks we can account for. `vrrp_script` is reproduced only when it is the
|
||||
# check script we generate ourselves (see _classify_script).
|
||||
_SUPPORTED_TOP_KEYS = {"global_defs", "vrrp_script", "vrrp_instance"}
|
||||
|
||||
# global_defs entries our render emits. An operator's file usually carries more (notification
|
||||
# email, router_id, ...) and losing those is a real change, so they are reported too.
|
||||
_SUPPORTED_GLOBAL_KEYS = {"enable_script_security", "script_user"}
|
||||
|
||||
_IDENT_RE = re.compile(r"^[A-Za-z0-9._:-]+$")
|
||||
|
||||
|
||||
class KeepalivedParseError(ValueError):
|
||||
"""The text is not a keepalived.conf we can reason about (unbalanced braces etc.)."""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tokenizer / block reader
|
||||
# ---------------------------------------------------------------------------
|
||||
def _strip_comment(line: str) -> str:
|
||||
"""Drop a trailing comment. keepalived treats BOTH `#` and `!` as comment starters, and
|
||||
neither is meaningful inside the quoted script paths we care about, so a quote-aware
|
||||
scan is enough (a `#` inside quotes stays)."""
|
||||
out: List[str] = []
|
||||
quote: Optional[str] = None
|
||||
for ch in line:
|
||||
if quote:
|
||||
out.append(ch)
|
||||
if ch == quote:
|
||||
quote = None
|
||||
continue
|
||||
if ch in ('"', "'"):
|
||||
quote = ch
|
||||
out.append(ch)
|
||||
continue
|
||||
if ch in ("#", "!"):
|
||||
break
|
||||
out.append(ch)
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def _split_tokens(line: str) -> List[str]:
|
||||
"""Whitespace split that keeps quoted strings whole and isolates braces, so
|
||||
`virtual_ipaddress { 10.0.0.1/24 dev eth0 }` tokenizes the same as its multi-line form."""
|
||||
tokens: List[str] = []
|
||||
buf: List[str] = []
|
||||
quote: Optional[str] = None
|
||||
|
||||
def flush() -> None:
|
||||
if buf:
|
||||
tokens.append("".join(buf))
|
||||
buf.clear()
|
||||
|
||||
for ch in line:
|
||||
if quote:
|
||||
if ch == quote:
|
||||
quote = None
|
||||
else:
|
||||
buf.append(ch)
|
||||
continue
|
||||
if ch in ('"', "'"):
|
||||
quote = ch
|
||||
continue
|
||||
if ch.isspace():
|
||||
flush()
|
||||
elif ch in ("{", "}"):
|
||||
flush()
|
||||
tokens.append(ch)
|
||||
else:
|
||||
buf.append(ch)
|
||||
flush()
|
||||
return tokens
|
||||
|
||||
|
||||
def _read_blocks(text: str) -> List[Dict[str, Any]]:
|
||||
"""Parse the file into nested entries.
|
||||
|
||||
Each entry is either
|
||||
{"kind": "block", "name": str, "args": [str], "body": [entries], "line": int}
|
||||
{"kind": "line", "tokens": [str], "line": int}
|
||||
|
||||
Line boundaries matter: inside `virtual_ipaddress` and `unicast_peer` each line is one
|
||||
bare value, so a flat token stream could not tell two addresses apart.
|
||||
"""
|
||||
root: List[Dict[str, Any]] = []
|
||||
stack: List[List[Dict[str, Any]]] = [root]
|
||||
# Blocks whose opening `{` we have seen, so a stray `}` can be reported with context.
|
||||
open_blocks: List[str] = []
|
||||
|
||||
for lineno, raw in enumerate(text.splitlines(), start=1):
|
||||
pending: List[str] = []
|
||||
for tok in _split_tokens(_strip_comment(raw)):
|
||||
if tok == "{":
|
||||
name = pending[0] if pending else ""
|
||||
args = pending[1:]
|
||||
block = {"kind": "block", "name": name, "args": args, "body": [], "line": lineno}
|
||||
stack[-1].append(block)
|
||||
stack.append(block["body"])
|
||||
open_blocks.append(name)
|
||||
pending = []
|
||||
elif tok == "}":
|
||||
if pending:
|
||||
stack[-1].append({"kind": "line", "tokens": pending, "line": lineno})
|
||||
pending = []
|
||||
if len(stack) == 1:
|
||||
raise KeepalivedParseError(f"unbalanced '}}' on line {lineno}")
|
||||
stack.pop()
|
||||
open_blocks.pop()
|
||||
else:
|
||||
pending.append(tok)
|
||||
if pending:
|
||||
stack[-1].append({"kind": "line", "tokens": pending, "line": lineno})
|
||||
|
||||
if len(stack) != 1:
|
||||
raise KeepalivedParseError(f"unclosed block '{open_blocks[-1] or '?'}' at end of file")
|
||||
return root
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Interpretation
|
||||
# ---------------------------------------------------------------------------
|
||||
def _as_int(tokens: List[str]) -> Optional[int]:
|
||||
if len(tokens) < 2:
|
||||
return None
|
||||
try:
|
||||
return int(tokens[1])
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _parse_vip_entry(tokens: List[str]) -> Optional[Dict[str, Any]]:
|
||||
"""One `virtual_ipaddress` line: `<addr>[/<prefix>] [dev <iface>] [label ...]`.
|
||||
|
||||
Returns None when the first token is not an address — a shape we do not understand must
|
||||
surface as unsupported rather than be silently dropped.
|
||||
"""
|
||||
spec = tokens[0]
|
||||
addr, _, prefix = spec.partition("/")
|
||||
try:
|
||||
ip = ipaddress.ip_address(addr)
|
||||
except ValueError:
|
||||
return None
|
||||
entry: Dict[str, Any] = {
|
||||
"address": str(ip),
|
||||
"prefix_length": None,
|
||||
"dev": None,
|
||||
"extra": [],
|
||||
}
|
||||
if prefix:
|
||||
try:
|
||||
entry["prefix_length"] = int(prefix)
|
||||
except ValueError:
|
||||
return None
|
||||
rest = tokens[1:]
|
||||
i = 0
|
||||
while i < len(rest):
|
||||
if rest[i] == "dev" and i + 1 < len(rest):
|
||||
entry["dev"] = rest[i + 1]
|
||||
i += 2
|
||||
continue
|
||||
# `label`, `scope`, `brd`, ... — all real directives we do not render.
|
||||
entry["extra"].append(rest[i])
|
||||
i += 1
|
||||
return entry
|
||||
|
||||
|
||||
def _classify_script(block: Dict[str, Any]) -> Tuple[str, Optional[str]]:
|
||||
"""Return (name, script_path) for a vrrp_script block."""
|
||||
name = block["args"][0] if block["args"] else (block["name"] or "")
|
||||
path = None
|
||||
for entry in block["body"]:
|
||||
if entry["kind"] == "line" and entry["tokens"] and entry["tokens"][0] == "script":
|
||||
path = " ".join(entry["tokens"][1:]) or None
|
||||
return name, path
|
||||
|
||||
|
||||
def _parse_instance(block: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Interpret one `vrrp_instance` block into VIP-model fields plus its own unsupported list."""
|
||||
inst: Dict[str, Any] = {
|
||||
"instance_name": block["args"][0] if block["args"] else "",
|
||||
"state": None,
|
||||
"interface": None,
|
||||
"virtual_router_id": None,
|
||||
"priority": None,
|
||||
"advert_int": None,
|
||||
"auth_type": None,
|
||||
"auth_pass": None,
|
||||
"unicast_src_ip": None,
|
||||
"unicast_peers": [],
|
||||
"virtual_ips": [],
|
||||
"track_scripts": [],
|
||||
"unsupported": [],
|
||||
"line": block["line"],
|
||||
}
|
||||
|
||||
def unsupported(what: str, lineno: int) -> None:
|
||||
inst["unsupported"].append({"directive": what, "line": lineno})
|
||||
|
||||
for entry in block["body"]:
|
||||
if entry["kind"] == "line":
|
||||
tokens = entry["tokens"]
|
||||
key = tokens[0]
|
||||
if key == "state":
|
||||
inst["state"] = (tokens[1].upper() if len(tokens) > 1 else None)
|
||||
elif key == "interface":
|
||||
inst["interface"] = tokens[1] if len(tokens) > 1 else None
|
||||
elif key == "virtual_router_id":
|
||||
inst["virtual_router_id"] = _as_int(tokens)
|
||||
elif key == "priority":
|
||||
inst["priority"] = _as_int(tokens)
|
||||
elif key == "advert_int":
|
||||
# keepalived accepts sub-second floats; our model column is an integer.
|
||||
raw = tokens[1] if len(tokens) > 1 else ""
|
||||
try:
|
||||
val = float(raw)
|
||||
except (TypeError, ValueError):
|
||||
val = None
|
||||
if val is None:
|
||||
unsupported(f"advert_int {raw}", entry["line"])
|
||||
elif val != int(val):
|
||||
# Rounding would change VRRP timing, so refuse rather than adopt-and-alter.
|
||||
unsupported(f"advert_int {raw} (fractional; model stores whole seconds)",
|
||||
entry["line"])
|
||||
else:
|
||||
inst["advert_int"] = int(val)
|
||||
elif key == "unicast_src_ip":
|
||||
inst["unicast_src_ip"] = tokens[1] if len(tokens) > 1 else None
|
||||
else:
|
||||
unsupported(" ".join(tokens), entry["line"])
|
||||
continue
|
||||
|
||||
name = entry["name"]
|
||||
if name == "authentication":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] != "line" or not sub["tokens"]:
|
||||
continue
|
||||
k = sub["tokens"][0]
|
||||
if k == "auth_type":
|
||||
inst["auth_type"] = (sub["tokens"][1].upper() if len(sub["tokens"]) > 1 else None)
|
||||
elif k == "auth_pass":
|
||||
# Everything after the keyword: a VRRP password may contain spaces.
|
||||
inst["auth_pass"] = " ".join(sub["tokens"][1:]) or None
|
||||
else:
|
||||
unsupported(f"authentication/{' '.join(sub['tokens'])}", sub["line"])
|
||||
elif name == "unicast_peer":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
inst["unicast_peers"].append(sub["tokens"][0])
|
||||
else:
|
||||
unsupported("unicast_peer/<block>", entry["line"])
|
||||
elif name == "virtual_ipaddress":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] != "line" or not sub["tokens"]:
|
||||
unsupported("virtual_ipaddress/<block>", entry["line"])
|
||||
continue
|
||||
parsed = _parse_vip_entry(sub["tokens"])
|
||||
if parsed is None:
|
||||
unsupported(f"virtual_ipaddress/{' '.join(sub['tokens'])}", sub["line"])
|
||||
else:
|
||||
if parsed["extra"]:
|
||||
unsupported(
|
||||
f"virtual_ipaddress/{parsed['address']} "
|
||||
f"({' '.join(parsed['extra'])})", sub["line"])
|
||||
inst["virtual_ips"].append(parsed)
|
||||
elif name == "track_script":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
inst["track_scripts"].append(sub["tokens"][0])
|
||||
else:
|
||||
unsupported(f"{name} {{...}}", entry["line"])
|
||||
|
||||
return inst
|
||||
|
||||
|
||||
def parse_keepalived_conf(text: str) -> Dict[str, Any]:
|
||||
"""Parse a keepalived.conf into VIP-model fields plus everything we could not model.
|
||||
|
||||
Raises KeepalivedParseError on structurally broken input. Never log the result: parsed
|
||||
instances carry `auth_pass` in cleartext.
|
||||
"""
|
||||
root = _read_blocks(text or "")
|
||||
result: Dict[str, Any] = {
|
||||
"instances": [],
|
||||
"scripts": {},
|
||||
"global_defs": {},
|
||||
"unsupported": [], # top-level directives our render would drop
|
||||
"sync_groups": [],
|
||||
}
|
||||
|
||||
for entry in root:
|
||||
if entry["kind"] == "line":
|
||||
# A bare top-level directive (e.g. `include /etc/keepalived/conf.d/*.conf`).
|
||||
result["unsupported"].append(
|
||||
{"directive": " ".join(entry["tokens"]), "line": entry["line"]})
|
||||
continue
|
||||
name = entry["name"]
|
||||
if name == "global_defs":
|
||||
for sub in entry["body"]:
|
||||
if sub["kind"] == "line" and sub["tokens"]:
|
||||
key = sub["tokens"][0]
|
||||
result["global_defs"][key] = " ".join(sub["tokens"][1:])
|
||||
if key not in _SUPPORTED_GLOBAL_KEYS:
|
||||
result["unsupported"].append(
|
||||
{"directive": f"global_defs/{' '.join(sub['tokens'])}",
|
||||
"line": sub["line"]})
|
||||
else:
|
||||
result["unsupported"].append(
|
||||
{"directive": f"global_defs/{sub.get('name', '?')} {{...}}",
|
||||
"line": sub["line"]})
|
||||
elif name == "vrrp_script":
|
||||
script_name, path = _classify_script(entry)
|
||||
result["scripts"][script_name] = {"script": path, "line": entry["line"]}
|
||||
elif name == "vrrp_instance":
|
||||
result["instances"].append(_parse_instance(entry))
|
||||
elif name == "vrrp_sync_group":
|
||||
# A sync group ties instances together so they fail over as a unit. Our render has
|
||||
# no equivalent, and dropping it changes failover semantics — never adopt silently.
|
||||
group = entry["args"][0] if entry["args"] else ""
|
||||
result["sync_groups"].append({"name": group, "line": entry["line"]})
|
||||
result["unsupported"].append(
|
||||
{"directive": f"vrrp_sync_group {group}", "line": entry["line"]})
|
||||
else:
|
||||
# virtual_server (LVS), static_routes, bfd_instance, ...
|
||||
args = " ".join(entry["args"])
|
||||
result["unsupported"].append(
|
||||
{"directive": f"{name} {args} {{...}}".replace(" ", " "), "line": entry["line"]})
|
||||
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mapping to the VIP model + the adoption gate
|
||||
# ---------------------------------------------------------------------------
|
||||
# keepalived defaults we are willing to apply when a directive is absent, because the value
|
||||
# is unambiguous and re-rendering it changes nothing on the wire.
|
||||
_DEFAULT_ADVERT_INT = 1
|
||||
_DEFAULT_PRIORITY = 100
|
||||
_DEFAULT_STATE = "BACKUP"
|
||||
|
||||
# The only track_script our renderer emits (keepalived_config.build_haproxy_check_script).
|
||||
OUR_CHECK_SCRIPT_NAME = "chk_haproxy"
|
||||
|
||||
|
||||
def build_adoption_candidate(parsed: Dict[str, Any], instance: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Map one parsed `vrrp_instance` onto vip_instances / vip_members fields.
|
||||
|
||||
Returns `adoptable` plus `blockers`. A blocker means taking ownership would change what
|
||||
is running — either because our render cannot reproduce something in the file, or because
|
||||
a value we must write is not knowable from the file. Adoption REPLACES the operator's
|
||||
config, so "we could not read it" and "we would change it" are the same hazard, and both
|
||||
have to stop the flow rather than be logged.
|
||||
|
||||
Never log the return value: `vip.auth_pass` is cleartext.
|
||||
"""
|
||||
blockers: List[str] = []
|
||||
|
||||
# Directives we would drop. Report the file's own line numbers so the operator can look.
|
||||
dropped = list(parsed.get("unsupported") or []) + list(instance.get("unsupported") or [])
|
||||
for d in dropped:
|
||||
blockers.append(
|
||||
f"line {d['line']}: `{d['directive']}` — OpenManager's renderer cannot reproduce "
|
||||
f"this, so adopting would delete it")
|
||||
|
||||
vips = instance.get("virtual_ips") or []
|
||||
if len(vips) == 0:
|
||||
blockers.append("the instance declares no virtual_ipaddress — nothing to adopt")
|
||||
elif len(vips) > 1:
|
||||
addrs = ", ".join(v["address"] for v in vips)
|
||||
blockers.append(
|
||||
f"the instance carries {len(vips)} addresses ({addrs}); a managed VIP holds exactly "
|
||||
f"one, so adopting would drop all but the first")
|
||||
|
||||
vip_entry = vips[0] if vips else None
|
||||
|
||||
if instance.get("virtual_router_id") is None:
|
||||
blockers.append("no virtual_router_id — it cannot be guessed: a wrong VRID puts the "
|
||||
"nodes in separate VRRP domains and both would claim the VIP")
|
||||
if not instance.get("interface"):
|
||||
blockers.append("no interface — required to render the instance and the address")
|
||||
|
||||
# An explicit prefix is required. Our renderer ALWAYS writes `<addr>/<prefix>`, the model
|
||||
# column defaults to 24, and keepalived's own default for a bare address is a host route.
|
||||
# Picking either one for the operator would silently change the VIP's netmask, so ask.
|
||||
if vip_entry is not None and vip_entry.get("prefix_length") is None:
|
||||
blockers.append(
|
||||
f"`{vip_entry['address']}` has no explicit prefix length; state it during adoption "
|
||||
f"so the netmask cannot change on takeover")
|
||||
|
||||
# The address must live on the instance's interface — that is the only `dev` we can render.
|
||||
if vip_entry is not None and vip_entry.get("dev") and instance.get("interface") \
|
||||
and vip_entry["dev"] != instance["interface"]:
|
||||
blockers.append(
|
||||
f"the address is bound to `dev {vip_entry['dev']}` but the instance uses "
|
||||
f"`interface {instance['interface']}`; the render always uses the instance interface")
|
||||
|
||||
auth_type = instance.get("auth_type")
|
||||
if auth_type not in (None, "PASS"):
|
||||
blockers.append(f"auth_type {auth_type} is not supported (only PASS is rendered)")
|
||||
|
||||
# A tracked script that is not ours would be replaced by our HAProxy check.
|
||||
tracked = [t for t in (instance.get("track_scripts") or [])]
|
||||
foreign = [t for t in tracked if t != OUR_CHECK_SCRIPT_NAME]
|
||||
if foreign:
|
||||
blockers.append(
|
||||
f"track_script {', '.join(foreign)} would be replaced by OpenManager's HAProxy "
|
||||
f"health check")
|
||||
|
||||
state = instance.get("state") or _DEFAULT_STATE
|
||||
if state not in ("MASTER", "BACKUP"):
|
||||
blockers.append(f"state {state} is not MASTER or BACKUP")
|
||||
|
||||
peers = list(instance.get("unicast_peers") or [])
|
||||
src = instance.get("unicast_src_ip")
|
||||
# Our renderer emits unicast_src_ip and unicast_peer together, or neither.
|
||||
if bool(src) != bool(peers):
|
||||
which = "unicast_src_ip without unicast_peer" if src else "unicast_peer without unicast_src_ip"
|
||||
blockers.append(f"{which} — the render emits both or neither")
|
||||
|
||||
candidate: Dict[str, Any] = {
|
||||
"instance_name": instance.get("instance_name") or "",
|
||||
"adoptable": not blockers,
|
||||
"blockers": blockers,
|
||||
"dropped_directives": dropped,
|
||||
"vip": {
|
||||
"virtual_ip": vip_entry["address"] if vip_entry else None,
|
||||
"prefix_length": vip_entry.get("prefix_length") if vip_entry else None,
|
||||
"virtual_router_id": instance.get("virtual_router_id"),
|
||||
"advert_int": instance.get("advert_int") if instance.get("advert_int") is not None
|
||||
else _DEFAULT_ADVERT_INT,
|
||||
"use_unicast": bool(peers),
|
||||
"track_haproxy": OUR_CHECK_SCRIPT_NAME in tracked,
|
||||
"auth_pass": instance.get("auth_pass"),
|
||||
},
|
||||
"member": {
|
||||
"network_interface": instance.get("interface"),
|
||||
"role": state,
|
||||
"priority": instance.get("priority") if instance.get("priority") is not None
|
||||
else _DEFAULT_PRIORITY,
|
||||
},
|
||||
"peers": peers,
|
||||
"unicast_src_ip": src,
|
||||
# Which values came from a keepalived default rather than the file, so the UI can say so.
|
||||
"defaulted": [
|
||||
k for k, present in (
|
||||
("advert_int", instance.get("advert_int") is not None),
|
||||
("priority", instance.get("priority") is not None),
|
||||
("state", instance.get("state") is not None),
|
||||
) if not present
|
||||
],
|
||||
}
|
||||
return candidate
|
||||
|
||||
|
||||
# Substrings that identify the two blocker classes an operator is allowed to resolve. They are
|
||||
# matched rather than typed because the blocker text is what the UI shows; keeping the marker in
|
||||
# the sentence means the message and the rule cannot drift apart.
|
||||
_LOSS_MARKER = "would delete it"
|
||||
_PREFIX_MARKER = "no explicit prefix length"
|
||||
|
||||
|
||||
def remaining_blockers(blockers: List[str], *, prefix_supplied: bool = False,
|
||||
accept_data_loss: bool = False) -> List[str]:
|
||||
"""Blockers that survive what the operator is permitted to resolve.
|
||||
|
||||
Exactly two classes are resolvable, and the distinction is the whole safety argument:
|
||||
|
||||
* a missing prefix length is *unknown*, and the operator can supply it — we refuse to pick
|
||||
a netmask for a live VIP ourselves;
|
||||
* "our renderer cannot reproduce this, so adopting would delete it" is a *loss*, and losing
|
||||
it can be an informed choice.
|
||||
|
||||
Everything else — an unknown virtual_router_id, a fractional advert_int, an unsupported
|
||||
auth_type, an address on a different interface — is neither unknown nor a loss but an
|
||||
impossibility, and no flag may wave it through. This is the single source of truth for that
|
||||
rule; the endpoint and the UI both derive from it.
|
||||
"""
|
||||
out: List[str] = []
|
||||
for b in blockers or []:
|
||||
if prefix_supplied and _PREFIX_MARKER in b:
|
||||
continue
|
||||
if accept_data_loss and _LOSS_MARKER in b:
|
||||
continue
|
||||
out.append(b)
|
||||
return out
|
||||
|
||||
|
||||
def analyse_keepalived_conf(text: str) -> Dict[str, Any]:
|
||||
"""Parse + map in one call: the shape the discovery endpoint stores and the UI renders."""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
return {
|
||||
"instance_count": len(parsed["instances"]),
|
||||
"sync_groups": parsed["sync_groups"],
|
||||
"global_defs": parsed["global_defs"],
|
||||
"candidates": [build_adoption_candidate(parsed, inst) for inst in parsed["instances"]],
|
||||
}
|
||||
@@ -0,0 +1,291 @@
|
||||
"""ACME challenge backend URL validation, resolution and change detection.
|
||||
|
||||
These pin the behaviour behind a real incident: a split deployment rendered
|
||||
`server _acme_mgmt <mgmt>:8080` against a port with no listener, HTTP-01 failed for
|
||||
weeks while DNS-01 kept working, and every existing check reported success. The
|
||||
three mechanisms below are what make that impossible to repeat silently.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from services.haproxy_config import (
|
||||
extract_acme_backend_target,
|
||||
is_config_generation_error,
|
||||
select_acme_backend_source,
|
||||
)
|
||||
from utils.acme_backend_url import (
|
||||
AcmeBackendUrlError,
|
||||
resolve_acme_backend_target,
|
||||
validate_acme_backend_url,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Boundary validation — what an operator may type.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"value,expected",
|
||||
[
|
||||
("http://10.90.1.4:80", "http://10.90.1.4:80"),
|
||||
("http://10.90.1.4", "http://10.90.1.4"),
|
||||
("https://mgmt.internal:8443", "https://mgmt.internal:8443"),
|
||||
# RFC1918 is the NORMAL answer here, unlike utils/ssrf_guard's policy: the
|
||||
# operator is naming their own management host, which on a split deployment
|
||||
# is private by definition.
|
||||
("http://192.168.1.5:8080", "http://192.168.1.5:8080"),
|
||||
# Empty means "inherit from the next level of the resolution chain".
|
||||
(None, None),
|
||||
("", None),
|
||||
(" ", None),
|
||||
# Surrounding whitespace is normalised, not rejected — and the NORMALISED
|
||||
# value is what callers persist, so it can never reach haproxy.cfg.
|
||||
(" http://10.0.0.5:80 ", "http://10.0.0.5:80"),
|
||||
],
|
||||
)
|
||||
def test_accepts_and_normalises_usable_values(value, expected):
|
||||
assert validate_acme_backend_url(value) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"value,code",
|
||||
[
|
||||
# Scheme-less values used to be accepted and then silently became `localhost`
|
||||
# in the renderer — the trap that makes a correct diagnosis un-actionable.
|
||||
("10.90.1.4:8080", "no_scheme"),
|
||||
("localhost:8080", "no_scheme"),
|
||||
("ftp://10.0.0.5", "bad_scheme"),
|
||||
# A newline would inject directives into a file pushed to every node.
|
||||
("http://10.90.1.4\nbind :9", "whitespace"),
|
||||
("http://10.90.1.4 x", "whitespace"),
|
||||
# Loopback by number AND by name: `localhost` is what both shipped defaults
|
||||
# contain, so catching only the numeric form would miss the common case.
|
||||
("http://localhost:8080", "loopback"),
|
||||
("http://LOCALHOST", "loopback"),
|
||||
("http://127.0.0.1", "loopback"),
|
||||
("http://[::1]:80", "loopback"),
|
||||
("http://0.0.0.0:80", "unspecified"),
|
||||
("http://169.254.169.254", "link_local"),
|
||||
("http://u:p@10.0.0.5", "userinfo"),
|
||||
("http://10.0.0.5/api", "has_path"),
|
||||
("http://10.0.0.5?x=1", "has_path"),
|
||||
# urlparse defers port parsing to attribute access; unguarded this raises
|
||||
# inside the config generator and destroys the cluster's whole config.
|
||||
("http://10.0.0.5:99999", "bad_port"),
|
||||
("http://10.0.0.5:abc", "bad_port"),
|
||||
("http://-bad-.com", "invalid_host"),
|
||||
],
|
||||
)
|
||||
def test_rejects_unusable_values_with_stable_codes(value, code):
|
||||
with pytest.raises(AcmeBackendUrlError) as exc:
|
||||
validate_acme_backend_url(value)
|
||||
assert exc.value.code == code
|
||||
assert str(exc.value), "every rejection must carry operator-facing prose"
|
||||
|
||||
|
||||
def test_rejects_values_longer_than_the_column():
|
||||
# VARCHAR(500); without this the write fails as an opaque asyncpg 22001 -> 500.
|
||||
with pytest.raises(AcmeBackendUrlError) as exc:
|
||||
validate_acme_backend_url("http://" + "a" * 600 + ".com")
|
||||
assert exc.value.code == "too_long"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Render-time resolution — never rejects, never raises.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"url,host,port,ssl_flag",
|
||||
[
|
||||
("http://10.90.1.4:80", "10.90.1.4", 80, ""),
|
||||
# Port-less http stays 8080, NOT the scheme default 80: the bundled compose
|
||||
# publishes nginx on 8080, so installs relying on this have a working path
|
||||
# today and changing it would break them silently in the renewal loop.
|
||||
("http://10.90.1.4", "10.90.1.4", 8080, ""),
|
||||
("https://m.io", "m.io", 443, " ssl verify none"),
|
||||
("http://localhost:8080", "localhost", 8080, ""),
|
||||
],
|
||||
)
|
||||
def test_resolution_preserves_existing_rendering(url, host, port, ssl_flag):
|
||||
target = resolve_acme_backend_target(url)
|
||||
assert (target.host, target.port, target.ssl_flag) == (host, port, ssl_flag)
|
||||
assert target.error_code is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"url",
|
||||
["", None, " ", "10.0.0.5:80", "http://h:99999", "http://10.0.0.5\nx", "http://ba d"],
|
||||
)
|
||||
def test_resolution_reports_instead_of_raising(url):
|
||||
target = resolve_acme_backend_target(url)
|
||||
assert target.error_code, "unusable values must be reported, not raised"
|
||||
assert target.error_message
|
||||
|
||||
|
||||
def test_resolution_warns_on_loopback_rather_than_refusing():
|
||||
# Refusing here would make every existing install unappliable: the shipped
|
||||
# defaults ARE loopback, and the failure would block changes unrelated to ACME.
|
||||
target = resolve_acme_backend_target("http://localhost:8080")
|
||||
assert target.error_code is None
|
||||
assert target.warnings and "Loopback" in target.warnings[0]
|
||||
|
||||
|
||||
def test_resolution_warns_when_the_port_is_omitted():
|
||||
target = resolve_acme_backend_target("http://10.0.0.5")
|
||||
assert target.port == 8080
|
||||
assert any("port" in w.lower() for w in target.warnings)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Change detection — what makes a panel edit actually reach the nodes.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
_CONFIG = """global
|
||||
daemon
|
||||
|
||||
frontend fe_http
|
||||
bind 10.90.1.100:80
|
||||
mode http
|
||||
acl is_acme_challenge path_beg /.well-known/acme-challenge/
|
||||
use_backend _acme_challenge_backend if is_acme_challenge
|
||||
default_backend app
|
||||
|
||||
backend app
|
||||
server s1 10.0.0.9:8080
|
||||
|
||||
# ACME Challenge Backend (auto-managed by HAProxy OpenManager)
|
||||
backend _acme_challenge_backend
|
||||
mode http
|
||||
server _acme_mgmt 10.90.1.4:80
|
||||
"""
|
||||
|
||||
|
||||
def test_extracts_the_challenge_backend_target():
|
||||
assert extract_acme_backend_target(_CONFIG) == "10.90.1.4:80"
|
||||
|
||||
|
||||
def test_extracts_target_with_ssl_flag():
|
||||
cfg = _CONFIG.replace("10.90.1.4:80", "m.io:443 ssl verify none")
|
||||
assert extract_acme_backend_target(cfg) == "m.io:443 ssl verify none"
|
||||
|
||||
|
||||
def test_ignores_server_lines_in_other_backends():
|
||||
# Comparing the whole config would flag every unrelated pending edit as a change;
|
||||
# this must key on the ACME section alone.
|
||||
cfg = _CONFIG.replace("backend _acme_challenge_backend", "backend something_else")
|
||||
assert extract_acme_backend_target(cfg) is None
|
||||
|
||||
|
||||
def test_returns_none_when_the_section_has_no_server_line():
|
||||
cfg = (
|
||||
"backend _acme_challenge_backend\n"
|
||||
" mode http\n"
|
||||
" # ACME challenge backend unavailable (loopback)\n"
|
||||
)
|
||||
assert extract_acme_backend_target(cfg) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", ["", None])
|
||||
def test_extraction_tolerates_empty_input(value):
|
||||
assert extract_acme_backend_target(value) is None
|
||||
|
||||
|
||||
def test_url_change_that_renders_the_same_target_is_not_a_change():
|
||||
# `http://10.90.1.4` and `http://10.90.1.4:8080` are different strings but the
|
||||
# same shipped address; minting a config version for that would put a no-op
|
||||
# pending change in front of the operator.
|
||||
a = resolve_acme_backend_target("http://10.90.1.4")
|
||||
b = resolve_acme_backend_target("http://10.90.1.4:8080")
|
||||
assert (a.host, a.port, a.ssl_flag) == (b.host, b.port, b.ssl_flag)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Source selection — an unusable value must not shadow a usable one.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_prefers_the_most_specific_usable_source():
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "http://10.0.0.1:80"),
|
||||
("settings", "http://10.0.0.2:80"),
|
||||
("env", "http://10.0.0.3:80"),
|
||||
])
|
||||
assert (source, url, target.host) == ("cluster", "http://10.0.0.1:80", "10.0.0.1")
|
||||
assert skipped == []
|
||||
|
||||
|
||||
def test_skips_an_unusable_value_and_uses_the_next_source():
|
||||
# The regression this guards: a scheme-less value left over from the era when the
|
||||
# settings field was free text resolves to nothing. Stopping there would emit a
|
||||
# backend section with no `server` line — `haproxy -c` passes, Apply succeeds, and
|
||||
# every challenge request 503s with no visible cause.
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "10.90.1.4:8080"),
|
||||
("settings", ""),
|
||||
("env", "http://10.90.1.4:8080"),
|
||||
])
|
||||
assert source == "env"
|
||||
assert target.error_code is None and target.host == "10.90.1.4"
|
||||
assert [s[0] for s in skipped] == ["cluster"]
|
||||
|
||||
|
||||
def test_empty_sources_are_skipped_without_being_reported():
|
||||
source, _url, target, skipped = select_acme_backend_source([
|
||||
("cluster", ""),
|
||||
("settings", None),
|
||||
("env", "http://10.0.0.9:80"),
|
||||
])
|
||||
assert source == "env" and target.error_code is None
|
||||
assert skipped == []
|
||||
|
||||
|
||||
def test_reports_the_last_attempted_value_when_nothing_resolves():
|
||||
source, url, target, skipped = select_acme_backend_source([
|
||||
("cluster", "10.0.0.1:80"),
|
||||
("env", "not a url"),
|
||||
])
|
||||
assert (source, url) == ("env", "not a url")
|
||||
assert target.error_code, "the caller needs an error to render and log"
|
||||
assert [s[0] for s in skipped] == ["cluster"]
|
||||
|
||||
|
||||
def test_all_sources_empty_yields_an_error_not_a_crash():
|
||||
source, _url, target, skipped = select_acme_backend_source([
|
||||
("cluster", ""), ("settings", ""), ("env", ""),
|
||||
])
|
||||
assert source == "cluster" and target.error_code == "empty" and skipped == []
|
||||
|
||||
|
||||
def test_loopback_is_usable_enough_to_render():
|
||||
# Warned about, never skipped: the shipped defaults are loopback, so treating it as
|
||||
# unusable would make the fallback chain fall off its own end on a stock install.
|
||||
source, _url, target, _skipped = select_acme_backend_source([
|
||||
("env", "http://localhost:8080"),
|
||||
])
|
||||
assert source == "env" and target.error_code is None
|
||||
assert target.warnings
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The generator's failure sentinel must never be mistaken for a config.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"content",
|
||||
[
|
||||
"# Error generating configuration: boom",
|
||||
"# Error: Cluster not found",
|
||||
"",
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_detects_generation_failure_sentinels(content):
|
||||
assert is_config_generation_error(content) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("content", [_CONFIG, "global\n daemon\n"])
|
||||
def test_real_configs_are_not_flagged(content):
|
||||
assert is_config_generation_error(content) is False
|
||||
@@ -104,13 +104,30 @@ async def test_check_dns_empty_ips_marks_failure(monkeypatch):
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Port-80 check (HEAD probe)
|
||||
# Port-80 check (GET probe)
|
||||
#
|
||||
# The probe is a GET, not a HEAD: a reverse proxy that has lost its
|
||||
# /.well-known/acme-challenge/ location falls through to its catch-all and serves
|
||||
# an SPA with HTTP 200, which a status-code-only check accepts as healthy while
|
||||
# every real validation fails. The fakes below therefore carry a body.
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
|
||||
class _FakeHEADResp:
|
||||
def __init__(self, status):
|
||||
class _FakeContent:
|
||||
def __init__(self, body):
|
||||
self._body = body
|
||||
|
||||
async def read(self, n=-1):
|
||||
if self._body is None:
|
||||
raise ConnectionResetError("reset mid-body")
|
||||
return self._body if n is None or n < 0 else self._body[:n]
|
||||
|
||||
|
||||
class _FakeGETResp:
|
||||
def __init__(self, status, body=b"", content_type="text/plain"):
|
||||
self.status = status
|
||||
self.headers = {"content-type": content_type}
|
||||
self.content = _FakeContent(body)
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
@@ -120,10 +137,13 @@ class _FakeHEADResp:
|
||||
|
||||
|
||||
class _FakeSession:
|
||||
def __init__(self, *, statuses=None, raise_timeout=False, raise_client_error=False):
|
||||
def __init__(self, *, statuses=None, raise_timeout=False, raise_client_error=False,
|
||||
bodies=None, content_types=None):
|
||||
self._statuses = list(statuses or [])
|
||||
self._raise_timeout = raise_timeout
|
||||
self._raise_client_error = raise_client_error
|
||||
self._bodies = list(bodies or [])
|
||||
self._content_types = list(content_types or [])
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
@@ -131,14 +151,16 @@ class _FakeSession:
|
||||
async def __aexit__(self, *args):
|
||||
return False
|
||||
|
||||
def head(self, url, allow_redirects=False):
|
||||
def get(self, url, allow_redirects=False):
|
||||
if self._raise_timeout:
|
||||
raise asyncio.TimeoutError()
|
||||
if self._raise_client_error:
|
||||
import aiohttp
|
||||
raise aiohttp.ClientError("connection refused")
|
||||
status = self._statuses.pop(0) if self._statuses else 200
|
||||
return _FakeHEADResp(status)
|
||||
body = self._bodies.pop(0) if self._bodies else b""
|
||||
ctype = self._content_types.pop(0) if self._content_types else "text/plain"
|
||||
return _FakeGETResp(status, body, ctype)
|
||||
|
||||
|
||||
def _mock_public_dns(monkeypatch, ip="93.184.216.34"):
|
||||
@@ -178,6 +200,65 @@ async def test_check_port80_ok_on_404(monkeypatch):
|
||||
assert out["status"] == "ok"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warns_when_200_carries_a_web_page(monkeypatch):
|
||||
"""The failure that motivated the GET probe.
|
||||
|
||||
A reverse proxy whose /.well-known/acme-challenge/ location has drifted away
|
||||
falls through to its catch-all and serves the SPA. The status is 200, so the old
|
||||
`status in (200, 404)` rule called the install healthy while every validation
|
||||
failed. Only the body distinguishes them.
|
||||
"""
|
||||
_mock_public_dns(monkeypatch)
|
||||
spa = b'<!doctype html><html><head><title>HAProxy OpenManager</title></head>'
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[200], bodies=[spa], content_types=["text/html"])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "warn"
|
||||
target = out["details"]["targets"][0]
|
||||
assert target["body_class"] == "html"
|
||||
assert not target.get("ok")
|
||||
assert "HTML" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_falls_back_to_status_when_the_body_cannot_be_read(monkeypatch):
|
||||
"""A body that cannot be read is missing evidence, not a verdict.
|
||||
|
||||
Turning a connection reset mid-response into a hard failure would make a healthy
|
||||
404 fail intermittently, so the check keeps its original status-only semantics
|
||||
whenever there is nothing to judge.
|
||||
"""
|
||||
_mock_public_dns(monkeypatch)
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[404], bodies=[None])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "ok"
|
||||
assert out["details"]["targets"][0]["body_class"] == "unread"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warns_on_a_redirect(monkeypatch):
|
||||
_mock_public_dns(monkeypatch)
|
||||
|
||||
def _ctor(*args, **kwargs):
|
||||
return _FakeSession(statuses=[301])
|
||||
|
||||
monkeypatch.setattr("aiohttp.ClientSession", _ctor)
|
||||
|
||||
out = await check_port80(["a.example.com"])
|
||||
assert out["status"] == "warn"
|
||||
assert out["details"]["targets"][0]["diagnosis"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_port80_warn_on_egress_timeout(monkeypatch):
|
||||
"""Corporate egress blocks port 80 outbound — warn, don't fail."""
|
||||
@@ -337,16 +418,92 @@ async def test_check_routing_fail_when_no_port80_frontend():
|
||||
assert "No HTTP frontend" in out["message"]
|
||||
|
||||
|
||||
def _routing_row(**over):
|
||||
row = {"id": 1, "name": "fe-http", "bind_address": "0.0.0.0", "bind_port": 80,
|
||||
"mode": "http", "default_backend": "be", "cluster_id": 1, "acme_enabled": True}
|
||||
row.update(over)
|
||||
return row
|
||||
|
||||
|
||||
def _applied(has_route=True, server_line="server _acme_mgmt 10.90.1.4:80"):
|
||||
return {"has_route": has_route, "server_line": server_line}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_ok_when_port80_frontend_present():
|
||||
async def test_check_routing_ok_when_challenge_route_is_in_the_applied_config():
|
||||
# A port-80 frontend row alone is NOT enough. It describes what the database
|
||||
# wants; the nodes run whatever was last applied. During the incident this
|
||||
# function reported "ok" from the row count while the live config had no usable
|
||||
# challenge route at all.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [
|
||||
{"id": 1, "name": "fe-http", "bind_address": "0.0.0.0", "bind_port": 80,
|
||||
"mode": "http", "default_backend": "be"},
|
||||
]
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied()
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "ok"
|
||||
assert len(out["details"]["frontends"]) == 1
|
||||
assert out["details"]["challenge_backends"] == {1: "10.90.1.4:80"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_acme_is_disabled_on_the_cluster():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(acme_enabled=False)]
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "disabled" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_route_is_not_in_the_applied_config():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(has_route=False)
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "apply the cluster" in out["message"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_challenge_backend_has_no_server_line():
|
||||
# `haproxy -c` passes because the section exists, so nothing else catches this;
|
||||
# every challenge request 503s from an empty backend.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(server_line=None)
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "503" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_when_the_challenge_backend_is_loopback():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row()]
|
||||
conn.fetchrow.return_value = _applied(server_line="server _acme_mgmt 127.0.0.1:8080")
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "HAProxy node" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_warns_rather_than_fails_on_a_tcp_only_port80_cluster():
|
||||
# The `fail` branch must stay reachable only when NO port-80 frontend exists at
|
||||
# all: SiteWizard blocks submit on any failing check, so turning this into a
|
||||
# failure would lock tcp-only installs the day it ships.
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(mode="tcp")]
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "warn"
|
||||
assert "tcp mode" in out["message"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_routing_treats_null_mode_as_http_like_the_renderer():
|
||||
conn = AsyncMock()
|
||||
conn.fetch.return_value = [_routing_row(mode=None)]
|
||||
conn.fetchrow.return_value = _applied()
|
||||
out = await check_routing(conn, ["a.example.com"], [1])
|
||||
assert out["status"] == "ok"
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Issue #53 (v1.10.1) — at-rest encryption for the pending CSR private key.
|
||||
|
||||
Pure logic: no DB, no network. Covers the round-trip, the backward-compatible read of rows
|
||||
written before this release, the unrecoverable-key path after a key rotation, and a static
|
||||
assertion that the write path can no longer store a raw PEM.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-csr-encryption-unit-tests")
|
||||
|
||||
from cryptography.fernet import Fernet
|
||||
|
||||
from utils.csr_key_crypto import (
|
||||
decrypt_csr_private_key,
|
||||
encrypt_csr_private_key,
|
||||
is_encrypted,
|
||||
reset_fernet_for_tests,
|
||||
)
|
||||
|
||||
_SAMPLE_PEM = (
|
||||
"-----BEGIN PRIVATE KEY-----\n"
|
||||
"MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj\n"
|
||||
"-----END PRIVATE KEY-----\n"
|
||||
)
|
||||
|
||||
|
||||
def test_roundtrip_and_ciphertext_does_not_contain_the_key():
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# The stored form must not be the PEM, and must not leak any recognisable fragment of it.
|
||||
assert token != _SAMPLE_PEM
|
||||
assert "-----BEGIN" not in token
|
||||
assert "MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC7VJTUt9Us8cKj" not in token
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_is_encrypted_discriminates_token_from_legacy_pem():
|
||||
reset_fernet_for_tests()
|
||||
assert is_encrypted(encrypt_csr_private_key(_SAMPLE_PEM)) is True
|
||||
assert is_encrypted(_SAMPLE_PEM) is False
|
||||
assert is_encrypted("") is False
|
||||
assert is_encrypted(None) is False
|
||||
|
||||
|
||||
def test_legacy_plaintext_row_is_read_unchanged():
|
||||
# Rows written before v1.10.1 hold a raw PEM. They must keep working with NO data migration,
|
||||
# otherwise upgrading would strand every CSR that is out for signature.
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(_SAMPLE_PEM) == _SAMPLE_PEM
|
||||
|
||||
|
||||
def test_empty_or_missing_value_returns_none():
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(None) is None
|
||||
assert decrypt_csr_private_key("") is None
|
||||
|
||||
|
||||
def test_key_rotation_makes_the_stored_key_unrecoverable_rather_than_wrong():
|
||||
"""After a rotation the caller must get None, never a silently wrong key."""
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
|
||||
# Rotate: an explicit, different CSR_ENCRYPTION_KEY takes precedence over the derived one.
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = Fernet.generate_key().decode()
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) is None
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_explicit_env_key_is_used_and_survives_reset():
|
||||
previous = os.environ.get("CSR_ENCRYPTION_KEY")
|
||||
key = Fernet.generate_key().decode()
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = key
|
||||
try:
|
||||
reset_fernet_for_tests()
|
||||
token = encrypt_csr_private_key(_SAMPLE_PEM)
|
||||
# Decryptable with the same explicit key from a fresh instance...
|
||||
reset_fernet_for_tests()
|
||||
assert decrypt_csr_private_key(token) == _SAMPLE_PEM
|
||||
# ...and independently verifiable with the raw Fernet key.
|
||||
assert Fernet(key.encode()).decrypt(token.encode()).decode() == _SAMPLE_PEM
|
||||
finally:
|
||||
if previous is None:
|
||||
os.environ.pop("CSR_ENCRYPTION_KEY", None)
|
||||
else:
|
||||
os.environ["CSR_ENCRYPTION_KEY"] = previous
|
||||
reset_fernet_for_tests()
|
||||
|
||||
|
||||
def test_derivation_uses_its_own_hkdf_info_string():
|
||||
"""Each secret class derives an independent key, so rotating one never affects another."""
|
||||
src = (Path(__file__).resolve().parent.parent / "utils" / "csr_key_crypto.py").read_text()
|
||||
assert b"csr-private-key-v1".decode() in src
|
||||
# Must NOT reuse another class's info string.
|
||||
for foreign in ("dns-provider-creds-v1", "vip-vrrp-secret-v1", "mfa-totp-secret-v1"):
|
||||
assert foreign not in src, f"CSR key derivation must not reuse the {foreign} info string"
|
||||
|
||||
|
||||
def test_write_path_stores_the_encrypted_form_not_the_pem():
|
||||
"""Static pin: insert_csr_row must encrypt before the INSERT.
|
||||
|
||||
A future refactor that passed bundle['private_key_pem'] straight through would silently
|
||||
reintroduce plaintext storage, and no unit test with a mocked connection would notice.
|
||||
"""
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
insert_fn = src[src.index("async def insert_csr_row("):]
|
||||
insert_fn = insert_fn[: insert_fn.index("\nasync def ")]
|
||||
assert "encrypt_csr_private_key(bundle['private_key_pem'])" in insert_fn
|
||||
# The raw PEM must not be a bind parameter of the INSERT itself.
|
||||
assert not re.search(r"^\s*bundle\['private_key_pem'\],\s*$", insert_fn, re.M)
|
||||
|
||||
|
||||
def test_import_path_decrypts_and_fails_closed_on_unrecoverable_key():
|
||||
src = (Path(__file__).resolve().parent.parent / "services" / "csr_service.py").read_text()
|
||||
fn = src[src.index("async def import_signed_certificate("):]
|
||||
assert "decrypt_csr_private_key(row['private_key_pem'])" in fn
|
||||
# A None decrypt must raise rather than fall through to the key-match comparison.
|
||||
assert "cannot be decrypted" in fn
|
||||
@@ -0,0 +1,341 @@
|
||||
"""Issue #27 follow-up (v1.10.4) — unit tests for parsing an EXISTING keepalived.conf so a
|
||||
hand-maintained VIP can be adopted.
|
||||
|
||||
Pure-function tests; no DB, no network. The parser exists because the heartbeat carries only
|
||||
the VIP address and a best-effort MASTER/BACKUP, while rendering a node's config needs eleven
|
||||
fields — and because adoption REPLACES the operator's file, so anything our renderer cannot
|
||||
reproduce has to be reported as a blocker rather than silently dropped.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
os.environ.setdefault("SECRET_KEY", "test-secret-key-for-keepalived-parser-tests")
|
||||
|
||||
from services import keepalived_config as kc # noqa: E402
|
||||
from services.keepalived_parser import ( # noqa: E402
|
||||
KeepalivedParseError, analyse_keepalived_conf, build_adoption_candidate,
|
||||
parse_keepalived_conf,
|
||||
)
|
||||
|
||||
|
||||
# A realistic hand-maintained config: two nodes, unicast VRRP, password auth, HAProxy check.
|
||||
HANDWRITTEN = """\
|
||||
! Configuration File for keepalived
|
||||
global_defs {
|
||||
enable_script_security
|
||||
script_user root
|
||||
}
|
||||
|
||||
vrrp_script chk_haproxy {
|
||||
script "/etc/keepalived/check_haproxy.sh"
|
||||
interval 2
|
||||
weight -21
|
||||
}
|
||||
|
||||
vrrp_instance VI_1 {
|
||||
state MASTER
|
||||
interface eth0 # public leg
|
||||
virtual_router_id 51
|
||||
priority 150
|
||||
advert_int 1
|
||||
authentication {
|
||||
auth_type PASS
|
||||
auth_pass s3cr3t
|
||||
}
|
||||
unicast_src_ip 10.0.0.11
|
||||
unicast_peer {
|
||||
10.0.0.12
|
||||
}
|
||||
virtual_ipaddress {
|
||||
10.0.0.100/24 dev eth0
|
||||
}
|
||||
track_script {
|
||||
chk_haproxy
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def _only_candidate(text):
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert len(parsed["instances"]) == 1
|
||||
return build_adoption_candidate(parsed, parsed["instances"][0])
|
||||
|
||||
|
||||
def test_parses_a_handwritten_config_into_model_fields():
|
||||
cand = _only_candidate(HANDWRITTEN)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["blockers"] == []
|
||||
assert cand["vip"] == {
|
||||
"virtual_ip": "10.0.0.100",
|
||||
"prefix_length": 24,
|
||||
"virtual_router_id": 51,
|
||||
"advert_int": 1,
|
||||
"use_unicast": True,
|
||||
"track_haproxy": True,
|
||||
"auth_pass": "s3cr3t",
|
||||
}
|
||||
assert cand["member"] == {"network_interface": "eth0", "role": "MASTER", "priority": 150}
|
||||
assert cand["peers"] == ["10.0.0.12"] and cand["unicast_src_ip"] == "10.0.0.11"
|
||||
assert cand["defaulted"] == [] # every value came from the file, nothing assumed
|
||||
|
||||
|
||||
def test_comment_and_layout_variants():
|
||||
# `!` and `#` both start comments; a block may open and close on one line; a quoted
|
||||
# script path keeps its spaces. None of this may change the parse.
|
||||
text = """\
|
||||
#!/not/a/shebang — this whole line is a comment
|
||||
vrrp_script chk { script "/opt/my scripts/chk.sh" }
|
||||
vrrp_instance VI_1 { state BACKUP
|
||||
interface eth1 ! trailing bang comment
|
||||
virtual_router_id 7
|
||||
priority 90
|
||||
virtual_ipaddress { 192.168.5.9/32 dev eth1 }
|
||||
}
|
||||
"""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert parsed["scripts"]["chk"]["script"] == "/opt/my scripts/chk.sh"
|
||||
inst = parsed["instances"][0]
|
||||
assert inst["state"] == "BACKUP" and inst["interface"] == "eth1"
|
||||
assert inst["virtual_router_id"] == 7 and inst["priority"] == 90
|
||||
assert inst["virtual_ips"] == [
|
||||
{"address": "192.168.5.9", "prefix_length": 32, "dev": "eth1", "extra": []}
|
||||
]
|
||||
|
||||
|
||||
def test_documented_defaults_are_applied_and_flagged():
|
||||
# keepalived's own defaults for absent directives. Applying them re-renders the same
|
||||
# behaviour, so they are allowed — but the UI must be able to say they were assumed.
|
||||
text = """\
|
||||
vrrp_instance VI_1 {
|
||||
interface eth0
|
||||
virtual_router_id 12
|
||||
virtual_ipaddress { 10.1.1.5/24 dev eth0 }
|
||||
}
|
||||
"""
|
||||
cand = _only_candidate(text)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["member"]["role"] == "BACKUP" and cand["member"]["priority"] == 100
|
||||
assert cand["vip"]["advert_int"] == 1
|
||||
assert sorted(cand["defaulted"]) == ["advert_int", "priority", "state"]
|
||||
# No authentication block and no track_script — both legal, both faithfully represented.
|
||||
assert cand["vip"]["auth_pass"] is None and cand["vip"]["track_haproxy"] is False
|
||||
assert cand["vip"]["use_unicast"] is False
|
||||
|
||||
|
||||
def _blockers_for(text):
|
||||
return " | ".join(_only_candidate(text)["blockers"])
|
||||
|
||||
|
||||
def test_directives_we_cannot_render_block_adoption():
|
||||
# THE central safety property: adoption overwrites the file, so a failover hook we do not
|
||||
# render would be destroyed. It must stop the flow, not warn.
|
||||
text = HANDWRITTEN.replace(
|
||||
" track_script {", ' notify_master "/usr/local/bin/promote.sh"\n track_script {')
|
||||
blockers = _blockers_for(text)
|
||||
assert "notify_master" in blockers and "would delete it" in blockers
|
||||
assert _only_candidate(text)["adoptable"] is False
|
||||
|
||||
|
||||
def test_multiple_addresses_in_one_instance_block_adoption():
|
||||
text = HANDWRITTEN.replace(" 10.0.0.100/24 dev eth0",
|
||||
" 10.0.0.100/24 dev eth0\n 10.0.0.101/24 dev eth0")
|
||||
blockers = _blockers_for(text)
|
||||
assert "2 addresses" in blockers and "10.0.0.101" in blockers
|
||||
|
||||
|
||||
def test_missing_vrid_blocks_adoption_with_the_split_brain_reason():
|
||||
text = HANDWRITTEN.replace(" virtual_router_id 51\n", "")
|
||||
blockers = _blockers_for(text)
|
||||
assert "no virtual_router_id" in blockers and "separate VRRP domains" in blockers
|
||||
|
||||
|
||||
def test_missing_prefix_blocks_adoption():
|
||||
# Our renderer always writes an explicit prefix; guessing one would change the netmask of a
|
||||
# live VIP, so the operator has to state it.
|
||||
text = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100 dev eth0")
|
||||
blockers = _blockers_for(text)
|
||||
assert "no explicit prefix length" in blockers
|
||||
|
||||
|
||||
def test_address_on_a_different_dev_blocks_adoption():
|
||||
text = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100/24 dev eth1")
|
||||
blockers = _blockers_for(text)
|
||||
assert "dev eth1" in blockers and "interface eth0" in blockers
|
||||
|
||||
|
||||
def test_foreign_track_script_blocks_adoption():
|
||||
text = HANDWRITTEN.replace(" chk_haproxy", " chk_custom")
|
||||
blockers = _blockers_for(text)
|
||||
assert "chk_custom" in blockers and "replaced by OpenManager" in blockers
|
||||
|
||||
|
||||
def test_unsupported_auth_type_blocks_adoption():
|
||||
text = HANDWRITTEN.replace("auth_type PASS", "auth_type AH")
|
||||
assert "auth_type AH" in _blockers_for(text)
|
||||
|
||||
|
||||
def test_fractional_advert_int_blocks_adoption():
|
||||
# Rounding 0.5s to 1s changes VRRP timing, so adopt-and-alter is not acceptable.
|
||||
text = HANDWRITTEN.replace("advert_int 1", "advert_int 0.5")
|
||||
blockers = _blockers_for(text)
|
||||
assert "advert_int 0.5" in blockers and "fractional" in blockers
|
||||
|
||||
|
||||
def test_half_configured_unicast_blocks_adoption():
|
||||
text = HANDWRITTEN.replace(" unicast_peer {\n 10.0.0.12\n }\n", "")
|
||||
assert "unicast_src_ip without unicast_peer" in _blockers_for(text)
|
||||
|
||||
|
||||
def test_sync_group_and_lvs_sections_block_adoption():
|
||||
text = HANDWRITTEN + """
|
||||
vrrp_sync_group VG1 {
|
||||
group {
|
||||
VI_1
|
||||
}
|
||||
}
|
||||
virtual_server 10.0.0.100 80 {
|
||||
lb_algo rr
|
||||
}
|
||||
"""
|
||||
parsed = parse_keepalived_conf(text)
|
||||
assert [g["name"] for g in parsed["sync_groups"]] == ["VG1"]
|
||||
directives = " ".join(d["directive"] for d in parsed["unsupported"])
|
||||
assert "vrrp_sync_group VG1" in directives and "virtual_server" in directives
|
||||
# Both are top-level, so EVERY candidate in the file is blocked — a sync group changes
|
||||
# failover semantics for the instances it groups.
|
||||
cand = build_adoption_candidate(parsed, parsed["instances"][0])
|
||||
assert cand["adoptable"] is False
|
||||
|
||||
|
||||
def test_extra_global_defs_are_reported_as_losses():
|
||||
text = HANDWRITTEN.replace(" script_user root",
|
||||
" script_user root\n router_id LVS_DEVEL")
|
||||
parsed = parse_keepalived_conf(text)
|
||||
directives = " ".join(d["directive"] for d in parsed["unsupported"])
|
||||
assert "global_defs/router_id LVS_DEVEL" in directives
|
||||
assert parsed["global_defs"]["router_id"] == "LVS_DEVEL"
|
||||
|
||||
|
||||
def test_multiple_instances_yield_one_candidate_each():
|
||||
text = HANDWRITTEN + """
|
||||
vrrp_instance VI_2 {
|
||||
state BACKUP
|
||||
interface eth0
|
||||
virtual_router_id 52
|
||||
priority 100
|
||||
advert_int 1
|
||||
virtual_ipaddress { 10.0.0.200/24 dev eth0 }
|
||||
}
|
||||
"""
|
||||
analysed = analyse_keepalived_conf(text)
|
||||
assert analysed["instance_count"] == 2
|
||||
names = [c["instance_name"] for c in analysed["candidates"]]
|
||||
assert names == ["VI_1", "VI_2"]
|
||||
assert [c["vip"]["virtual_ip"] for c in analysed["candidates"]] == ["10.0.0.100", "10.0.0.200"]
|
||||
assert all(c["adoptable"] for c in analysed["candidates"])
|
||||
|
||||
|
||||
def test_unbalanced_braces_raise():
|
||||
for bad in ("vrrp_instance VI_1 {\n state MASTER\n", "}\n"):
|
||||
raised = False
|
||||
try:
|
||||
parse_keepalived_conf(bad)
|
||||
except KeepalivedParseError:
|
||||
raised = True
|
||||
assert raised, f"should have raised for {bad!r}"
|
||||
|
||||
|
||||
def test_our_own_render_round_trips_with_zero_blockers():
|
||||
"""The invariant that keeps the parser honest: a config WE generated must parse back into
|
||||
the same model with nothing unsupported. If a future change to render_keepalived_conf emits
|
||||
a directive the parser does not know, this fails — instead of adoption silently reporting
|
||||
that OpenManager's own output is unadoptable."""
|
||||
vip = {"id": 3, "name": "web-vip", "virtual_ip": "10.0.0.100", "prefix_length": 24,
|
||||
"virtual_router_id": 51, "advert_int": 1, "use_unicast": True, "track_haproxy": True}
|
||||
members = [{"role": "MASTER", "priority": 150, "network_interface": "eth0",
|
||||
"agent_id": 1, "ip_address": "10.0.0.11"},
|
||||
{"role": "BACKUP", "priority": 100, "network_interface": "eth0",
|
||||
"agent_id": 2, "ip_address": "10.0.0.12"}]
|
||||
rendered = kc.render_keepalived_conf(
|
||||
vip=vip, members=members, this_agent=members[0],
|
||||
peer_ips=["10.0.0.12"], auth_pass_plain="s3cr3t")
|
||||
|
||||
cand = _only_candidate(rendered)
|
||||
assert cand["adoptable"] is True, cand["blockers"]
|
||||
assert cand["vip"]["virtual_ip"] == vip["virtual_ip"]
|
||||
assert cand["vip"]["prefix_length"] == vip["prefix_length"]
|
||||
assert cand["vip"]["virtual_router_id"] == vip["virtual_router_id"]
|
||||
assert cand["vip"]["track_haproxy"] is True and cand["vip"]["use_unicast"] is True
|
||||
assert cand["member"] == {"network_interface": "eth0", "role": "MASTER", "priority": 150}
|
||||
assert cand["vip"]["auth_pass"] == "s3cr3t"
|
||||
|
||||
# And the same for the no-auth / multicast / untracked shape, which renders fewer blocks.
|
||||
plain = kc.render_keepalived_conf(
|
||||
vip={**vip, "use_unicast": False, "track_haproxy": False},
|
||||
members=members, this_agent=members[1], peer_ips=[], auth_pass_plain=None)
|
||||
cand2 = _only_candidate(plain)
|
||||
assert cand2["adoptable"] is True, cand2["blockers"]
|
||||
assert cand2["vip"]["use_unicast"] is False and cand2["vip"]["track_haproxy"] is False
|
||||
assert cand2["vip"]["auth_pass"] is None
|
||||
|
||||
|
||||
# --- v1.10.4 adoption gate: which blockers an operator may resolve --------------------------
|
||||
|
||||
|
||||
def test_only_prefix_and_data_loss_are_waivable():
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
loss = "line 9: `notify_master \"/x.sh\"` — OpenManager's renderer cannot reproduce this, so adopting would delete it"
|
||||
prefix = "`10.0.0.5` has no explicit prefix length; state it during adoption so the netmask cannot change on takeover"
|
||||
hard_vrid = "no virtual_router_id — it cannot be guessed: a wrong VRID puts the nodes in separate VRRP domains"
|
||||
hard_auth = "auth_type AH is not supported (only PASS is rendered)"
|
||||
all_four = [loss, prefix, hard_vrid, hard_auth]
|
||||
|
||||
# Nothing waived: everything survives.
|
||||
assert remaining_blockers(all_four) == all_four
|
||||
# A supplied prefix resolves ONLY the prefix blocker.
|
||||
assert remaining_blockers(all_four, prefix_supplied=True) == [loss, hard_vrid, hard_auth]
|
||||
# Accepting data loss resolves ONLY the loss blocker.
|
||||
assert remaining_blockers(all_four, accept_data_loss=True) == [prefix, hard_vrid, hard_auth]
|
||||
# Both together still cannot wave through an impossibility — this is the property that stops
|
||||
# a UI flag from destroying a VIP whose VRID or auth_type we could not reproduce.
|
||||
assert remaining_blockers(all_four, prefix_supplied=True, accept_data_loss=True) == \
|
||||
[hard_vrid, hard_auth]
|
||||
# And an adoptable candidate stays adoptable.
|
||||
assert remaining_blockers([]) == []
|
||||
|
||||
|
||||
def test_waiver_markers_match_the_messages_the_parser_actually_emits():
|
||||
# The gate matches on substrings of the blocker prose, so a reworded message would silently
|
||||
# stop being waivable. Pin both directions against real parser output.
|
||||
from services.keepalived_parser import remaining_blockers
|
||||
|
||||
no_prefix = HANDWRITTEN.replace("10.0.0.100/24 dev eth0", "10.0.0.100 dev eth0")
|
||||
blockers = _only_candidate(no_prefix)["blockers"]
|
||||
assert blockers, "expected a prefix blocker"
|
||||
assert remaining_blockers(blockers, prefix_supplied=True) == []
|
||||
|
||||
with_hook = HANDWRITTEN.replace(
|
||||
" track_script {", ' notify_master "/usr/local/bin/promote.sh"\n track_script {')
|
||||
blockers = _only_candidate(with_hook)["blockers"]
|
||||
assert blockers, "expected a data-loss blocker"
|
||||
assert remaining_blockers(blockers, accept_data_loss=True) == []
|
||||
|
||||
|
||||
def test_auth_pass_masking_leaves_no_trace_of_the_secret():
|
||||
# The discovered config is stored and served to the UI, so the ingest endpoint masks the VRRP
|
||||
# password. Reuse the router's own regex so the test breaks if it is loosened.
|
||||
from routers.agent import _AUTH_PASS_MASK_RE
|
||||
|
||||
secret = "s3cr3t with spaces"
|
||||
text = HANDWRITTEN.replace("auth_pass s3cr3t", f"auth_pass {secret}")
|
||||
masked = _AUTH_PASS_MASK_RE.sub(r"\1********", text)
|
||||
assert secret not in masked and "s3cr3t" not in masked
|
||||
assert "auth_pass ********" in masked
|
||||
# Everything else survives, so the preview is still useful.
|
||||
assert "virtual_router_id 51" in masked and "10.0.0.100/24 dev eth0" in masked
|
||||
@@ -0,0 +1,327 @@
|
||||
"""v1.11.0: every outbound HTTP call is recorded, and instrumentation can never
|
||||
become the failure.
|
||||
|
||||
Two independent risks:
|
||||
|
||||
**Secrets.** The outbound calls carry the most sensitive material in the
|
||||
system: the ACME JWS (a replayable signed capability for the lifetime of its
|
||||
nonce) and the DNS provider API credentials. Those call sites must opt out of
|
||||
request-body capture and out of verbatim error text — the tests below assert
|
||||
that at the call site, not just in the helper.
|
||||
|
||||
**Availability.** Both DNS provider funnels end in
|
||||
`except Exception: raise DnsProviderError("Unexpected ... failure")`, and in
|
||||
GoDaddy's publish path that reverts `dns_record_published` and stalls the ACME
|
||||
order. So an exception escaping `outbound_span` would be reported to the
|
||||
operator as a provider outage. It must never raise — and it must never swallow.
|
||||
"""
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from dataclasses import replace # noqa: E402
|
||||
from unittest.mock import patch # noqa: E402
|
||||
|
||||
from utils import http_instrumentation # noqa: E402
|
||||
from utils import request_log_settings # noqa: E402
|
||||
from utils.http_instrumentation import ( # noqa: E402
|
||||
TARGET_ACME,
|
||||
TARGET_DNS_CLOUDFLARE,
|
||||
TARGET_DNS_GODADDY,
|
||||
outbound_span,
|
||||
)
|
||||
from utils.request_log_settings import DEFAULT_CONFIG # noqa: E402
|
||||
|
||||
_BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def _read(*parts):
|
||||
with open(os.path.join(_BACKEND, *parts), encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def _function_body(src, signature):
|
||||
start = src.index(signature)
|
||||
rest = src[start:]
|
||||
# Next def at the same or lower indentation ends the body.
|
||||
end = rest.find("\n async def ", 1)
|
||||
alt = rest.find("\n def ", 1)
|
||||
if alt != -1 and (end == -1 or alt < end):
|
||||
end = alt
|
||||
return rest if end == -1 else rest[:end]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def captured(monkeypatch):
|
||||
rows = []
|
||||
monkeypatch.setattr(http_instrumentation.request_log_sink, "offer", rows.append)
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE", DEFAULT_CONFIG)
|
||||
monkeypatch.setattr(http_instrumentation, "get_config", lambda: request_log_settings._CACHE)
|
||||
return rows
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# outbound_span behaviour
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_records_a_successful_call(captured):
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="POST",
|
||||
url="https://acme-v02.api.letsencrypt.org/acme/new-order") as span:
|
||||
span.set_response(201, {"content-type": "application/json"}, {"status": "pending"})
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
row = captured[0]
|
||||
assert row.direction == "outbound"
|
||||
assert row.target == TARGET_ACME
|
||||
assert row.method == "POST"
|
||||
assert row.status_code == 201
|
||||
assert row.status_class == 2
|
||||
assert row.response_body_value == {"status": "pending"}
|
||||
|
||||
|
||||
def test_exception_is_recorded_and_reraised_unchanged(captured):
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url="https://example.com/x"):
|
||||
raise ValueError("connection reset")
|
||||
|
||||
with pytest.raises(ValueError, match="connection reset"):
|
||||
asyncio.run(run())
|
||||
|
||||
row = captured[0]
|
||||
assert row.status_code is None
|
||||
assert row.status_class == 0, (
|
||||
"a call that never got a response must be status_class 0 — the sentinel the "
|
||||
"error-retention window keys off"
|
||||
)
|
||||
assert row.error.startswith("ValueError")
|
||||
|
||||
|
||||
def test_safe_error_only_records_the_type_not_the_message(captured):
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_DNS_GODADDY, method="PUT",
|
||||
url="https://api.godaddy.com/v1/domains/example.com/records/TXT/_acme-challenge",
|
||||
safe_error_only=True):
|
||||
raise RuntimeError("failed talking to https://api.godaddy.com/v1/domains/secret-zone")
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
asyncio.run(run())
|
||||
|
||||
assert captured[0].error == "RuntimeError"
|
||||
assert "secret-zone" not in (captured[0].error or "")
|
||||
|
||||
|
||||
def test_instrumentation_failure_never_becomes_a_provider_failure(captured, monkeypatch):
|
||||
"""A bug in row construction must not surface to the operator as
|
||||
'Unexpected GoDaddy API failure' and stall an ACME order."""
|
||||
def explode(row):
|
||||
raise RuntimeError("sink is broken")
|
||||
|
||||
monkeypatch.setattr(http_instrumentation.request_log_sink, "offer", explode)
|
||||
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_DNS_CLOUDFLARE, method="GET",
|
||||
url="https://api.cloudflare.com/client/v4/zones") as span:
|
||||
span.set_response(200, {}, {"success": True})
|
||||
return "provider-result"
|
||||
|
||||
assert asyncio.run(run()) == "provider-result", (
|
||||
"a broken sink propagated out of outbound_span; both DNS funnels would convert "
|
||||
"that into DnsProviderError('Unexpected ... failure'), and in GoDaddy's publish "
|
||||
"path that reverts dns_record_published and stalls the ACME order"
|
||||
)
|
||||
|
||||
|
||||
def test_block_exception_still_propagates_when_the_sink_is_broken(monkeypatch):
|
||||
monkeypatch.setattr(http_instrumentation.request_log_sink, "offer",
|
||||
lambda row: (_ for _ in ()).throw(RuntimeError("sink is broken")))
|
||||
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url="https://example.com"):
|
||||
raise KeyError("original")
|
||||
|
||||
with pytest.raises(KeyError, match="original"):
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_capture_body_false_stores_the_summary_not_the_payload(captured):
|
||||
async def run():
|
||||
async with outbound_span(
|
||||
target=TARGET_ACME, method="POST", url="https://acme/new-order",
|
||||
request_body={"jws": True, "kid_present": True, "payload_empty": False},
|
||||
capture_body=False,
|
||||
) as span:
|
||||
span.set_response(200, {}, {"status": "valid"})
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
row = captured[0]
|
||||
assert row.request_body_value == {"jws": True, "kid_present": True, "payload_empty": False}
|
||||
assert row.request_body_raw is None
|
||||
# The CA's RESPONSE is still captured — that is the half operators need.
|
||||
assert row.response_body_value == {"status": "valid"}
|
||||
|
||||
|
||||
def test_urls_are_scrubbed_before_storage(captured):
|
||||
async def run():
|
||||
async with outbound_span(
|
||||
target=TARGET_DNS_CLOUDFLARE, method="GET",
|
||||
url="https://user:hunter2@api.cloudflare.com/client/v4/zones?api_key=abc&page=1",
|
||||
) as span:
|
||||
span.set_response(200, {}, {})
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
url = captured[0].url
|
||||
assert "hunter2" not in url
|
||||
assert "abc" not in url
|
||||
assert "page=1" in url
|
||||
|
||||
|
||||
def test_outbound_rows_inherit_the_inbound_request_id(captured):
|
||||
from utils.request_log_sink import request_id_context
|
||||
|
||||
async def run():
|
||||
token = request_id_context.set("abc123def456")
|
||||
try:
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url="https://acme/dir") as span:
|
||||
span.set_response(200, {}, {})
|
||||
finally:
|
||||
request_id_context.reset(token)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
assert captured[0].request_id == "abc123def456", (
|
||||
"an outbound call must carry the inbound request's id, otherwise the detail view "
|
||||
"cannot show which API call triggered which CA/DNS call"
|
||||
)
|
||||
|
||||
|
||||
def test_background_calls_get_a_task_scoped_id(captured):
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url="https://acme/dir") as span:
|
||||
span.set_response(200, {}, {})
|
||||
|
||||
asyncio.run(run())
|
||||
assert captured[0].request_id.startswith("bg:")
|
||||
|
||||
|
||||
def test_disabled_outbound_capture_produces_no_row(captured, monkeypatch):
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE",
|
||||
replace(DEFAULT_CONFIG, capture_outbound=False))
|
||||
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="GET", url="https://acme/dir") as span:
|
||||
# The call site keeps working — set_response must still be callable.
|
||||
span.set_response(200, {}, {})
|
||||
|
||||
asyncio.run(run())
|
||||
assert captured == []
|
||||
|
||||
|
||||
def test_set_response_tolerates_a_response_without_headers(captured):
|
||||
"""Some call sites are driven in tests by minimal fakes exposing only
|
||||
`.status`."""
|
||||
async def run():
|
||||
async with outbound_span(target=TARGET_ACME, method="HEAD", url="https://acme/nonce") as span:
|
||||
span.set_response(200, None)
|
||||
|
||||
asyncio.run(run())
|
||||
assert captured[0].status_code == 200
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Call-site coverage
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("path,target", [
|
||||
(("services", "acme_service.py"), "TARGET_ACME"),
|
||||
(("services", "acme_diagnostics.py"), "TARGET_ACME_DIAG"),
|
||||
(("services", "dns_providers", "cloudflare.py"), "TARGET_DNS_CLOUDFLARE"),
|
||||
(("services", "dns_providers", "godaddy.py"), "TARGET_DNS_GODADDY"),
|
||||
(("routers", "letsencrypt.py"), "TARGET_LETSENCRYPT_CA"),
|
||||
(("routers", "settings.py"), "TARGET_SETTINGS_PROBE"),
|
||||
(("haproxy_client.py",), "TARGET_HAPROXY_STATS"),
|
||||
(("agent_notifications.py",), "TARGET_AGENT"),
|
||||
])
|
||||
def test_every_outbound_module_is_instrumented(path, target):
|
||||
src = _read(*path)
|
||||
assert "outbound_span(" in src, f"{'/'.join(path)} makes HTTP calls but records nothing"
|
||||
assert target in src, f"{'/'.join(path)} does not tag its rows with {target}"
|
||||
|
||||
|
||||
def test_acme_signed_request_never_captures_the_jws_body():
|
||||
"""The JWS body is {protected, payload, signature}: `protected` carries the
|
||||
nonce and account kid, `signature` is made with the account private key. A
|
||||
stored (protected, signature) pair is a replayable ACME credential."""
|
||||
src = _read("services", "acme_service.py")
|
||||
body = _function_body(src, " async def _signed_request(")
|
||||
|
||||
assert "capture_body=False" in body, (
|
||||
"the ACME JWS request body would be written to request_logs verbatim — that is a "
|
||||
"replayable signed credential sitting in an audit table"
|
||||
)
|
||||
assert '"jws": True' in body, "no synthetic summary replaces the suppressed JWS body"
|
||||
|
||||
|
||||
def test_acme_span_is_inside_the_retry_loop():
|
||||
"""The session is built outside `for attempt in range(3)`; the span must be
|
||||
inside it, so a badNonce retry is its own row rather than being folded into
|
||||
the successful attempt."""
|
||||
src = _read("services", "acme_service.py")
|
||||
body = _function_body(src, " async def _signed_request(")
|
||||
|
||||
loop_at = body.index("for attempt in range(3):")
|
||||
span_at = body.index("async with outbound_span(")
|
||||
assert loop_at < span_at, (
|
||||
"outbound_span wraps the retry loop instead of sitting inside it, so three "
|
||||
"attempts collapse into one log row and a nonce retry becomes invisible"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
("services", "dns_providers", "cloudflare.py"),
|
||||
("services", "dns_providers", "godaddy.py"),
|
||||
])
|
||||
def test_dns_providers_record_error_types_only(path):
|
||||
src = _read(*path)
|
||||
body = _function_body(src, " async def _request(")
|
||||
assert "safe_error_only=True" in body, (
|
||||
f"{'/'.join(path)} would record the full exception text, which can carry the "
|
||||
f"request URL and through it the tenant/zone identifier"
|
||||
)
|
||||
|
||||
|
||||
def test_godaddy_narrow_value_error_handling_is_preserved():
|
||||
"""R-round hardening: only a JSON decode failure may be swallowed. Widening
|
||||
it would make a mid-read transport failure look like an empty RRset, and the
|
||||
follow-up full-RRset PUT would then destroy coexisting TXT values."""
|
||||
src = _read("services", "dns_providers", "godaddy.py")
|
||||
body = _function_body(src, " async def _request(")
|
||||
assert "except ValueError:" in body
|
||||
assert "except Exception:\n body = None" not in body
|
||||
|
||||
|
||||
def test_acme_diagnostics_keeps_its_ipv4_pinned_connector():
|
||||
"""Duplicates an existing assertion on purpose: instrumenting this module
|
||||
must not have refactored the SSRF-guard connector away."""
|
||||
src = _read("services", "acme_diagnostics.py")
|
||||
assert "TCPConnector(family=socket.AF_INET" in src, (
|
||||
"the IPv4 pin was removed from the port-80 probe — that reopens the dual-stack "
|
||||
"AAAA bypass the SSRF guard closes"
|
||||
)
|
||||
|
||||
|
||||
def test_haproxy_stats_never_logs_basic_auth_or_the_csv():
|
||||
"""aiohttp.BasicAuth is a NamedTuple whose repr contains the cleartext
|
||||
password, and a full stats CSV has no audit value."""
|
||||
src = _read("haproxy_client.py")
|
||||
body = _function_body(src, " async def _get_stats_via_http(")
|
||||
assert "capture_body=False" in body
|
||||
assert "capture_response_body=False" in body
|
||||
assert "auth=auth" in body and "request_body=auth" not in body
|
||||
@@ -0,0 +1,327 @@
|
||||
"""v1.11.0: the log's cost must follow operator activity, not fleet size.
|
||||
|
||||
Every property here was a real defect measured on the feature branch, and each
|
||||
one only shows up at scale or at the edge of a setting's documented range, which
|
||||
is why none of them were caught by the rule-level tests.
|
||||
|
||||
* one row per API call becomes millions per day once the fleet is a few
|
||||
hundred nodes, and the row cap then evicts the forensic history the feature
|
||||
exists for;
|
||||
* the operator role could not see the rows its grant was written for;
|
||||
* every background call ever made shared one correlation id;
|
||||
* queue memory was a function of an operator-editable setting, not a limit.
|
||||
"""
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from dataclasses import replace # noqa: E402
|
||||
|
||||
from utils.http_instrumentation import _correlation_id, begin_background_trace # noqa: E402
|
||||
from utils.request_log_settings import ( # noqa: E402
|
||||
DEFAULT_CONFIG,
|
||||
get_config,
|
||||
set_config,
|
||||
)
|
||||
from utils.request_log_sink import ( # noqa: E402
|
||||
TARGET_INBOUND_AGENT,
|
||||
RequestLogRow,
|
||||
RequestLogSink,
|
||||
request_id_context,
|
||||
)
|
||||
|
||||
_BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
_ROUTER = os.path.join(_BACKEND, "routers", "request_logs.py")
|
||||
_MIDDLEWARE = os.path.join(_BACKEND, "middleware", "request_logger.py")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _restore_config():
|
||||
"""These tests mutate the module-global snapshot; put it back."""
|
||||
before = get_config()
|
||||
yield
|
||||
set_config(before)
|
||||
|
||||
|
||||
def _row(**kw):
|
||||
kw.setdefault("request_id", "a" * 32)
|
||||
kw.setdefault("direction", "inbound")
|
||||
kw.setdefault("method", "GET")
|
||||
kw.setdefault("url", "/api/agents/prod-lb-1/config")
|
||||
kw.setdefault("status_code", 200)
|
||||
return RequestLogRow(**kw)
|
||||
|
||||
|
||||
class _CountingSink(RequestLogSink):
|
||||
"""Counts what survives `offer()` without needing an event loop."""
|
||||
|
||||
def __init__(self, **kw):
|
||||
super().__init__(kw.pop("maxsize", 10000), 100, 500, **kw)
|
||||
self.accepted = []
|
||||
|
||||
def _ensure_queue(self):
|
||||
sink = self
|
||||
|
||||
class _Q:
|
||||
def put_nowait(self, row):
|
||||
sink.accepted.append(row)
|
||||
|
||||
def qsize(self):
|
||||
return len(sink.accepted)
|
||||
|
||||
return _Q()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Volume: successful agent polls are not rows
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_successful_agent_polls_are_dropped_by_default():
|
||||
"""~9 800 rows/day PER AGENT, all of them 200s meaning "nothing changed".
|
||||
|
||||
At 200 nodes that is ~2M rows/day and the 500 000 row cap is reached in
|
||||
about six hours, so the configured "7 days of successes, 30 days of
|
||||
failures" silently becomes about six hours of each — for everything in the
|
||||
table, not just for the agent rows.
|
||||
"""
|
||||
assert DEFAULT_CONFIG.capture_agent_success is False, (
|
||||
"the default must be off; on, the table's size is a function of node "
|
||||
"count rather than of anything anyone did"
|
||||
)
|
||||
sink = _CountingSink()
|
||||
for _ in range(100):
|
||||
sink.offer(_row(target=TARGET_INBOUND_AGENT, status_code=200))
|
||||
assert sink.accepted == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", [401, 422, 500, None])
|
||||
def test_failed_agent_calls_are_always_kept(status):
|
||||
"""The half an operator actually needs, and rare enough to be free.
|
||||
|
||||
`None` is a transport error with no HTTP response at all, which
|
||||
status_class reports as 0.
|
||||
"""
|
||||
sink = _CountingSink()
|
||||
sink.offer(_row(target=TARGET_INBOUND_AGENT, status_code=status))
|
||||
assert len(sink.accepted) == 1, f"a {status} agent call must be recorded"
|
||||
|
||||
|
||||
def test_operator_traffic_is_unaffected_by_the_agent_gate():
|
||||
sink = _CountingSink()
|
||||
sink.offer(_row(target=None, status_code=200, user_id=7))
|
||||
assert len(sink.accepted) == 1
|
||||
|
||||
|
||||
def test_the_gate_can_be_turned_on_for_debugging():
|
||||
set_config(replace(get_config(), capture_agent_success=True))
|
||||
sink = _CountingSink()
|
||||
sink.offer(_row(target=TARGET_INBOUND_AGENT, status_code=200))
|
||||
assert len(sink.accepted) == 1
|
||||
|
||||
|
||||
def test_agent_traffic_is_identified_by_headers_not_by_a_database_lookup():
|
||||
"""The hot path runs on every request; a lookup per call is not affordable.
|
||||
|
||||
The installed agent sends `X-API-Key` and never `Authorization`; the UI
|
||||
sends a JWT and never an agent key.
|
||||
"""
|
||||
from middleware.request_logger import _is_agent_call
|
||||
|
||||
def scope(headers):
|
||||
return {"type": "http", "headers": [(k.encode(), v.encode()) for k, v in headers.items()]}
|
||||
|
||||
assert _is_agent_call(scope({"x-api-key": "agt_x"})) is True
|
||||
assert _is_agent_call(scope({"authorization": "Bearer x.y.z"})) is False
|
||||
# generate-install-script accepts either; self-upgrade sends only the key.
|
||||
assert _is_agent_call(scope({"authorization": "Bearer x.y.z", "x-api-key": "agt_x"})) is False
|
||||
assert _is_agent_call(scope({})) is False
|
||||
|
||||
|
||||
def test_agent_gate_does_not_reach_for_a_connection():
|
||||
"""`offer()` is called from the request coroutine and must stay pure."""
|
||||
src = open(os.path.join(_BACKEND, "utils", "request_log_sink.py"), encoding="utf-8").read()
|
||||
body = src.split("def offer(", 1)[1].split("\n # -- consumer", 1)[0]
|
||||
for forbidden in ("await ", "get_database_connection", "fetch"):
|
||||
assert forbidden not in body, f"offer() must not {forbidden.strip()!r} — it runs on the hot path"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Visibility: the operator grant has to mean something
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_read_only_scoping_admits_agent_rows_but_not_other_users():
|
||||
"""`operator` holds requestlog.read to "debug failing applies" — but an
|
||||
apply fails on the NODE, and the node reports over its own API key, so that
|
||||
row has user_id NULL and own-rows-only scoping hid it.
|
||||
|
||||
Keyed on `target`, NOT on `user_id IS NULL`: anonymous traffic (failed
|
||||
logins and their usernames, unauthenticated probes) is not agent traffic
|
||||
and must stay admin-only.
|
||||
"""
|
||||
src = open(_ROUTER, encoding="utf-8").read()
|
||||
clause = re.search(r"if not can_manage:(.*?)where_sql =", src, re.S)
|
||||
assert clause, "the self-scoping block moved; re-check this test"
|
||||
# Code only: the comment above the clause explains what it deliberately
|
||||
# does NOT do, and would otherwise match the negative assertion below.
|
||||
body = "\n".join(
|
||||
line for line in clause.group(1).splitlines()
|
||||
if not line.lstrip().startswith("#")
|
||||
)
|
||||
assert "TARGET_INBOUND_AGENT" in body, "agent rows are still hidden from requestlog.read"
|
||||
assert "user_id IS NULL" not in body, (
|
||||
"scoping on NULL would also expose anonymous traffic, including failed "
|
||||
"logins and the usernames they carry"
|
||||
)
|
||||
|
||||
|
||||
def test_detail_endpoint_uses_the_same_scoping_rule_as_the_list():
|
||||
src = open(_ROUTER, encoding="utf-8").read()
|
||||
detail = src.split('@router.get("/{log_id}")', 1)[1]
|
||||
assert "TARGET_INBOUND_AGENT" in detail, (
|
||||
"the detail endpoint would 404 on the very rows the list now shows"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("decorator", [
|
||||
'@router.get("/settings")', '@router.put("/settings")',
|
||||
'@router.get("/stats")', '@router.post("/purge")',
|
||||
'@router.get("")', '@router.get("/{log_id}")',
|
||||
])
|
||||
def test_permission_is_enforced_before_the_try_block(decorator):
|
||||
"""The repo's GHSA-3p5c pattern: a permission check inside `try` gets
|
||||
swallowed by the handler's own `except Exception -> 500`, turning a 403
|
||||
into a server error and, worse, hiding that the check ran at all."""
|
||||
src = open(_ROUTER, encoding="utf-8").read()
|
||||
body = src.split(decorator, 1)[1]
|
||||
body = body.split("\n@router.")[0]
|
||||
require_at = body.find("_require(authorization")
|
||||
try_at = body.find("\n try:")
|
||||
assert require_at != -1, f"{decorator} does not call _require at all"
|
||||
assert try_at == -1 or require_at < try_at, (
|
||||
f"{decorator} checks permissions INSIDE its try block"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Correlation: a trace that groups the wrong rows is worse than no trace
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_each_background_pass_gets_its_own_correlation_id():
|
||||
"""Nothing in main.py names its tasks, so the old `bg:<task name>` fallback
|
||||
gave one long-lived loop a single id for its entire life — measured, 15
|
||||
ACME calls across 5 ticks came out as 1 id. `related` (LIMIT 100) then
|
||||
presents up to a hundred unrelated calls as this request's trace.
|
||||
"""
|
||||
async def loop():
|
||||
per_tick = []
|
||||
for _ in range(5):
|
||||
begin_background_trace("acme_renewals")
|
||||
per_tick.append([_correlation_id() for _ in range(3)])
|
||||
await asyncio.sleep(0)
|
||||
return per_tick
|
||||
|
||||
ticks = asyncio.run(loop())
|
||||
for tick in ticks:
|
||||
assert len(set(tick)) == 1, "calls within one pass must share an id"
|
||||
ids = [t[0] for t in ticks]
|
||||
assert len(set(ids)) == 5, f"passes must not share an id, got {ids}"
|
||||
|
||||
|
||||
def test_unwrapped_background_code_does_not_collapse_onto_one_id():
|
||||
"""Erring toward too little grouping: a row that stands alone is honest, a
|
||||
row falsely grouped with a hundred others is not."""
|
||||
async def unwrapped():
|
||||
request_id_context.set(None)
|
||||
return [_correlation_id() for _ in range(4)]
|
||||
|
||||
ids = asyncio.run(unwrapped())
|
||||
assert len(set(ids)) == 4
|
||||
|
||||
|
||||
def test_the_background_loops_that_make_outbound_calls_open_a_trace():
|
||||
src = open(os.path.join(_BACKEND, "main.py"), encoding="utf-8").read()
|
||||
for loop_name in ("complete_pending_acme_orders", "check_letsencrypt_renewals",
|
||||
"monitor_agent_status"):
|
||||
body = src.split(f"async def {loop_name}", 1)[1].split("\nasync def ")[0]
|
||||
assert "begin_background_trace(" in body, (
|
||||
f"{loop_name} makes outbound calls but never opens a per-pass trace"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Memory: a limit, not a setting
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_queue_memory_is_bounded_even_at_the_max_body_size_ceiling():
|
||||
"""`max_body_bytes` is editable from Settings and its documented ceiling is
|
||||
256 KB, which a row carries twice. Against the default 2 000-row queue that
|
||||
is ~1 GiB — the entire pod limit — reachable from in-range values.
|
||||
"""
|
||||
set_config(replace(get_config(), max_body_bytes=262144, capture_agent_success=True))
|
||||
budget = 8 * 1024 * 1024
|
||||
sink = _CountingSink(maxsize=2000, max_bytes=budget)
|
||||
blob = b"x" * 262144
|
||||
for _ in range(2000):
|
||||
sink.offer(_row(target=None, request_body_raw=blob, response_body_raw=blob))
|
||||
|
||||
held = sum(r.queue_weight() for r in sink.accepted)
|
||||
assert held <= budget, f"queue held {held} bytes against a {budget} byte budget"
|
||||
assert sink.stats["dropped"] > 0, "over-budget rows must be dropped, and counted"
|
||||
unbounded = 2000 * (2 * 262144 + 1400)
|
||||
assert held < unbounded / 10, (
|
||||
f"without the byte budget this queue would hold {unbounded // 1024 // 1024} MiB"
|
||||
)
|
||||
|
||||
|
||||
_DRAIN_ROWS = 50
|
||||
|
||||
|
||||
def test_the_byte_budget_is_released_as_rows_drain():
|
||||
"""A budget that only ever counts up is a slow leak, not a limit.
|
||||
|
||||
Deliberately time-INDEPENDENT. `_collect()` stops at whichever comes first,
|
||||
`batch_size` rows or the `flush_ms` deadline, so a batch size larger than
|
||||
the row count makes the result a function of how fast the runner happens to
|
||||
be. The first version of this test used batch=100/flush=10ms for 50 rows and
|
||||
passed on a native build while failing in CI, which builds
|
||||
linux/amd64 + linux/arm64 and therefore runs one of them under qemu
|
||||
emulation: 25 iterations of `asyncio.wait_for` were enough to exhaust 10 ms
|
||||
there, `_collect()` returned half a batch, and the assertion read
|
||||
`239800 == 0` - measuring the scheduler, not the sink.
|
||||
|
||||
So: batch size EQUAL to the row count, so the loop exits on the count and
|
||||
never consults the deadline; a generous flush window in case it somehow
|
||||
does; and a drain loop rather than a single call. Nothing here depends on
|
||||
wall-clock speed.
|
||||
"""
|
||||
async def drain():
|
||||
sink = RequestLogSink(2000, _DRAIN_ROWS, 5000, max_bytes=8 * 1024 * 1024)
|
||||
blob = b"x" * 4096
|
||||
set_config(replace(get_config(), capture_agent_success=True))
|
||||
for _ in range(_DRAIN_ROWS):
|
||||
sink.offer(_row(target=None, request_body_raw=blob, response_body_raw=blob))
|
||||
assert sink.stats["queued_bytes"] > 0, "nothing was queued, so nothing is being measured"
|
||||
assert sink._queue.qsize() == _DRAIN_ROWS, "the queue did not take every row"
|
||||
|
||||
drained = 0
|
||||
while not sink._queue.empty():
|
||||
drained += len(await sink._collect())
|
||||
assert drained == _DRAIN_ROWS, f"drained {drained} of {_DRAIN_ROWS} rows"
|
||||
return sink.stats["queued_bytes"]
|
||||
|
||||
assert asyncio.run(drain()) == 0
|
||||
|
||||
|
||||
def test_stats_say_the_sink_counters_are_per_worker():
|
||||
"""The sink is a module global; with UVICORN_WORKERS > 1 each process keeps
|
||||
its own. A number that looks fleet-wide but is not understates drops by
|
||||
exactly the worker count."""
|
||||
src = open(_ROUTER, encoding="utf-8").read()
|
||||
assert '"scope"' in src.split('"sink"', 1)[1][:400], (
|
||||
"the stats response must label the sink counters as this-worker-only"
|
||||
)
|
||||
@@ -0,0 +1,432 @@
|
||||
"""v1.11.0: the request/response logger must be invisible to everything below it.
|
||||
|
||||
This is the highest-risk piece of the feature. A logging middleware that reads
|
||||
the request body the naive way DRAINS the ASGI receive channel, and the handler
|
||||
underneath then sees an empty body — `POST /api/agents/heartbeat` reads the raw
|
||||
stream itself, so every agent in the fleet would start failing its heartbeat
|
||||
because someone wanted nicer logs.
|
||||
|
||||
The implementation therefore TEES rather than consumes. These tests drive the
|
||||
middleware over a stub ASGI app and assert that property directly: the
|
||||
downstream app sees the full body, the client sees the full response, and only
|
||||
a capped copy is kept.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from dataclasses import replace # noqa: E402
|
||||
|
||||
from middleware.request_logger import RequestResponseLogMiddleware # noqa: E402
|
||||
from utils.logging_config import correlation_id_context # noqa: E402
|
||||
from utils import request_log_settings # noqa: E402
|
||||
from utils.request_log_settings import DEFAULT_CONFIG # noqa: E402
|
||||
from utils import request_log_sink as sink_module # noqa: E402
|
||||
|
||||
|
||||
_MAIN = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "main.py")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def captured(monkeypatch):
|
||||
"""Collect the rows the middleware hands to the sink, instead of writing them."""
|
||||
rows = []
|
||||
monkeypatch.setattr(sink_module.request_log_sink, "offer", rows.append)
|
||||
# The middleware imports `request_log_sink` by value, so patch there too.
|
||||
import middleware.request_logger as rl
|
||||
monkeypatch.setattr(rl.request_log_sink, "offer", rows.append)
|
||||
return rows
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def default_config(monkeypatch):
|
||||
"""Every test starts from the shipped defaults, with a small body cap so the
|
||||
truncation paths are exercised without megabyte fixtures."""
|
||||
cfg = replace(DEFAULT_CONFIG, max_body_bytes=1024)
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE", cfg)
|
||||
import middleware.request_logger as rl
|
||||
monkeypatch.setattr(rl, "get_config", lambda: request_log_settings._CACHE)
|
||||
return cfg
|
||||
|
||||
|
||||
def set_config(monkeypatch, **overrides):
|
||||
cfg = replace(request_log_settings._CACHE, **overrides)
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE", cfg)
|
||||
return cfg
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# A minimal ASGI harness — no TestClient, no HTTP stack, just the protocol.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
async def drive(app, *, method="POST", path="/api/backends", body=b"", query=b"",
|
||||
headers=None, content_type="application/json"):
|
||||
"""Run one request through `app` and return (status, headers, body)."""
|
||||
raw_headers = [(b"host", b"testserver")]
|
||||
if content_type:
|
||||
raw_headers.append((b"content-type", content_type.encode()))
|
||||
for k, v in (headers or {}).items():
|
||||
raw_headers.append((k.encode().lower(), v.encode()))
|
||||
|
||||
scope = {
|
||||
"type": "http",
|
||||
"asgi": {"version": "3.0"},
|
||||
"http_version": "1.1",
|
||||
"method": method,
|
||||
"scheme": "http",
|
||||
"path": path,
|
||||
"raw_path": path.encode(),
|
||||
"query_string": query,
|
||||
"root_path": "",
|
||||
"headers": raw_headers,
|
||||
"client": ("10.1.2.3", 51234),
|
||||
"server": ("testserver", 80),
|
||||
}
|
||||
|
||||
# Deliver the body in three chunks so the tee is exercised across messages.
|
||||
chunks = [body[i:i + max(1, len(body) // 3 or 1)] for i in range(0, len(body), max(1, len(body) // 3 or 1))] or [b""]
|
||||
pending = list(chunks)
|
||||
|
||||
async def receive():
|
||||
if pending:
|
||||
chunk = pending.pop(0)
|
||||
return {"type": "http.request", "body": chunk, "more_body": bool(pending)}
|
||||
return {"type": "http.request", "body": b"", "more_body": False}
|
||||
|
||||
sent = {"status": None, "headers": [], "body": b""}
|
||||
|
||||
async def send(message):
|
||||
if message["type"] == "http.response.start":
|
||||
sent["status"] = message["status"]
|
||||
sent["headers"] = message.get("headers", [])
|
||||
elif message["type"] == "http.response.body":
|
||||
sent["body"] += message.get("body", b"") or b""
|
||||
|
||||
await app(scope, receive, send)
|
||||
return sent
|
||||
|
||||
|
||||
def echo_length_app(status=200, content_type=b"application/json"):
|
||||
"""Stub app that CONSUMES the whole request body and reports its length.
|
||||
|
||||
This is the regression shape: if the middleware drained the stream, the app
|
||||
below it would see 0 bytes.
|
||||
"""
|
||||
async def app(scope, receive, send):
|
||||
total = 0
|
||||
while True:
|
||||
message = await receive()
|
||||
total += len(message.get("body", b"") or b"")
|
||||
if not message.get("more_body"):
|
||||
break
|
||||
payload = json.dumps({"received_bytes": total}).encode()
|
||||
await send({"type": "http.response.start", "status": status,
|
||||
"headers": [(b"content-type", content_type)]})
|
||||
await send({"type": "http.response.body", "body": payload})
|
||||
return app
|
||||
|
||||
|
||||
def chunked_app(chunks, content_type=b"application/json"):
|
||||
async def app(scope, receive, send):
|
||||
await send({"type": "http.response.start", "status": 200,
|
||||
"headers": [(b"content-type", content_type)]})
|
||||
for i, chunk in enumerate(chunks):
|
||||
await send({"type": "http.response.body", "body": chunk,
|
||||
"more_body": i < len(chunks) - 1})
|
||||
return app
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The transparency guarantees
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_request_body_reaches_downstream_intact(captured):
|
||||
"""THE regression guard: draining the receive channel would break the raw-body
|
||||
agent heartbeat handler."""
|
||||
body = b"x" * 100_000
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
sent = asyncio.run(drive(app, body=body))
|
||||
|
||||
assert json.loads(sent["body"])["received_bytes"] == 100_000, (
|
||||
"the handler below the logger saw a different body length than the client sent — "
|
||||
"the middleware consumed the receive channel instead of teeing it"
|
||||
)
|
||||
|
||||
|
||||
def test_response_body_reaches_client_intact(captured):
|
||||
chunks = [b'{"part":', b'"one",', b'"n":2}']
|
||||
app = RequestResponseLogMiddleware(chunked_app(chunks))
|
||||
|
||||
sent = asyncio.run(drive(app, method="GET", body=b""))
|
||||
|
||||
assert sent["body"] == b"".join(chunks), "a response chunk was swallowed by the logger"
|
||||
assert sent["status"] == 200
|
||||
|
||||
|
||||
def test_only_the_capped_prefix_is_captured(captured):
|
||||
body = b"y" * 100_000
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
asyncio.run(drive(app, body=body))
|
||||
|
||||
row = captured[0]
|
||||
assert row.request_body_bytes == 100_000, "the on-the-wire size must be recorded in full"
|
||||
assert len(row.request_body_raw) <= 1024, (
|
||||
"the middleware buffered more than max_body_bytes — memory is unbounded per request"
|
||||
)
|
||||
|
||||
|
||||
def test_non_capturable_content_type_is_counted_but_not_buffered(captured):
|
||||
app = RequestResponseLogMiddleware(chunked_app([b"\x00\x01\x02" * 500],
|
||||
content_type=b"application/octet-stream"))
|
||||
|
||||
asyncio.run(drive(app, method="GET", content_type=None))
|
||||
|
||||
row = captured[0]
|
||||
assert row.response_body_bytes == 1500
|
||||
assert row.response_body_raw is None, (
|
||||
"a binary response body was buffered — this is what keeps streaming/file "
|
||||
"responses safe"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# What gets logged, and what does not
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_basic_row_fields(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
asyncio.run(drive(app, method="POST", path="/api/backends",
|
||||
body=b'{"name":"web"}', query=b"cluster_id=2&token=secret"))
|
||||
|
||||
row = captured[0]
|
||||
assert row.direction == "inbound"
|
||||
assert row.method == "POST"
|
||||
assert row.path == "/api/backends"
|
||||
assert row.status_code == 200
|
||||
assert row.status_class == 2
|
||||
assert row.client_ip == "10.1.2.3"
|
||||
assert row.duration_ms >= 0
|
||||
# The query string is scrubbed before it is stored, in the URL and the dict.
|
||||
assert "secret" not in row.url
|
||||
assert row.query_params["token"] == "***REDACTED***"
|
||||
assert row.query_params["cluster_id"] == "2"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/health",
|
||||
"/api/health/deep",
|
||||
"/api/docs",
|
||||
"/api/openapi.json",
|
||||
"/.well-known/acme-challenge/abc123",
|
||||
"/api/agents/heartbeat",
|
||||
"/favicon.ico",
|
||||
])
|
||||
def test_excluded_paths_produce_no_row(captured, path):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, method="GET", path=path))
|
||||
assert captured == [], f"{path} must not be logged by default"
|
||||
|
||||
|
||||
def test_log_viewer_path_cannot_be_un_excluded(captured, monkeypatch):
|
||||
"""`exclude_paths` is operator-editable, so the viewer's own endpoints have a
|
||||
hard floor — otherwise reading the log generates log entries about reading
|
||||
the log."""
|
||||
set_config(monkeypatch, exclude_paths=())
|
||||
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, method="GET", path="/api/request-logs?limit=50"))
|
||||
|
||||
assert captured == [], (
|
||||
"clearing exclude_paths re-enabled logging of the log viewer itself"
|
||||
)
|
||||
|
||||
|
||||
def test_options_preflight_is_skipped(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, method="OPTIONS", path="/api/backends"))
|
||||
assert captured == []
|
||||
|
||||
|
||||
def test_get_can_be_turned_off(captured, monkeypatch):
|
||||
set_config(monkeypatch, capture_get=False)
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
asyncio.run(drive(app, method="GET", path="/api/backends"))
|
||||
assert captured == []
|
||||
|
||||
asyncio.run(drive(app, method="POST", path="/api/backends", body=b"{}"))
|
||||
assert len(captured) == 1, "turning GETs off must not silence writes"
|
||||
|
||||
|
||||
def test_disabled_config_short_circuits_but_still_serves(captured, monkeypatch):
|
||||
set_config(monkeypatch, enabled=False)
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
sent = asyncio.run(drive(app, body=b"hello"))
|
||||
|
||||
assert captured == []
|
||||
assert sent["status"] == 200, "the kill-switch must not break request serving"
|
||||
|
||||
|
||||
def test_capture_bodies_off_keeps_sizes_but_drops_content(captured, monkeypatch):
|
||||
set_config(monkeypatch, capture_bodies=False)
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
|
||||
asyncio.run(drive(app, body=b'{"secret":"x"}'))
|
||||
|
||||
row = captured[0]
|
||||
assert row.request_body_raw is None
|
||||
assert row.request_body_bytes == 14, "size accounting must survive with bodies off"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Errors and correlation
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_exception_is_recorded_as_status_class_zero_and_reraised(captured):
|
||||
async def boom(scope, receive, send):
|
||||
raise RuntimeError("handler exploded")
|
||||
|
||||
app = RequestResponseLogMiddleware(boom)
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
asyncio.run(drive(app, method="GET"))
|
||||
|
||||
row = captured[0]
|
||||
assert row.status_code is None
|
||||
assert row.status_class == 0, (
|
||||
"a request that produced no HTTP response must be status_class 0 — that is the "
|
||||
"sentinel the error-retention prune keys off"
|
||||
)
|
||||
assert row.error.startswith("RuntimeError")
|
||||
|
||||
|
||||
def test_correlation_id_is_seeded_before_downstream_and_reset_after(captured):
|
||||
seen = {}
|
||||
|
||||
async def app(scope, receive, send):
|
||||
seen["cid"] = correlation_id_context.get()
|
||||
await send({"type": "http.response.start", "status": 204, "headers": []})
|
||||
await send({"type": "http.response.body", "body": b""})
|
||||
|
||||
wrapped = RequestResponseLogMiddleware(app)
|
||||
asyncio.run(drive(wrapped, method="GET"))
|
||||
|
||||
row = captured[0]
|
||||
assert seen["cid"] == row.request_id[:8], (
|
||||
"the downstream error handler would mint its own id, so X-Correlation-ID would "
|
||||
"not match request_logs.request_id"
|
||||
)
|
||||
assert correlation_id_context.get() is None, (
|
||||
"the ContextVar token was not reset — the next request on this task would inherit "
|
||||
"a stale correlation id"
|
||||
)
|
||||
|
||||
|
||||
def test_x_request_id_header_is_returned_to_the_client(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
sent = asyncio.run(drive(app, method="GET"))
|
||||
|
||||
names = {k.decode().lower() for k, _ in sent["headers"]}
|
||||
assert "x-request-id" in names, (
|
||||
"without this header a user reporting a problem has no id to quote"
|
||||
)
|
||||
|
||||
|
||||
def test_error_responses_are_logged_with_their_status(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app(status=422))
|
||||
asyncio.run(drive(app, body=b'{"bad":true}'))
|
||||
|
||||
row = captured[0]
|
||||
assert row.status_code == 422
|
||||
assert row.status_class == 4, "4xx must be classed as an error for retention purposes"
|
||||
|
||||
|
||||
def test_jwt_identity_is_resolved_without_a_database(captured):
|
||||
"""The middleware runs on every request; a DB lookup per call is not
|
||||
acceptable, so the user is read straight out of the token claims."""
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
from jose import jwt
|
||||
from config import JWT_ALGORITHM, JWT_SECRET_KEY
|
||||
|
||||
token = jwt.encode(
|
||||
{"user_id": 42, "username": "ops", "exp": datetime.utcnow() + timedelta(minutes=10)},
|
||||
JWT_SECRET_KEY, algorithm=JWT_ALGORITHM,
|
||||
)
|
||||
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, body=b"{}", headers={"authorization": f"Bearer {token}"}))
|
||||
|
||||
row = captured[0]
|
||||
assert row.user_id == 42
|
||||
assert row.username == "ops"
|
||||
|
||||
|
||||
def test_malformed_token_yields_an_anonymous_row(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, body=b"{}", headers={"authorization": "Bearer not.a.jwt"}))
|
||||
|
||||
row = captured[0]
|
||||
assert row.user_id is None
|
||||
assert row.username is None
|
||||
# Logging is not an auth path — a bad token must not turn into an exception.
|
||||
|
||||
|
||||
def test_authorization_header_is_never_stored_verbatim(captured):
|
||||
app = RequestResponseLogMiddleware(echo_length_app())
|
||||
asyncio.run(drive(app, body=b"{}", headers={"authorization": "Bearer super-secret"}))
|
||||
|
||||
params = captured[0].to_params()
|
||||
assert "super-secret" not in json.dumps(params, default=str)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Registration order in main.py
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_middleware_is_registered_last_so_it_is_outermost():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
log_at = src.index("app.add_middleware(RequestResponseLogMiddleware)")
|
||||
cors_at = src.index(" CORSMiddleware,")
|
||||
|
||||
assert log_at > cors_at, (
|
||||
"Starlette's add_middleware inserts at index 0, so the LAST registration is the "
|
||||
"OUTERMOST middleware. Registering the request logger before CORS would put it "
|
||||
"inside the stack, where it can no longer see the final client-visible response "
|
||||
"and can no longer seed the correlation id before the error handler reads it."
|
||||
)
|
||||
|
||||
|
||||
def test_env_kill_switch_guards_the_registration():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
assert re.search(
|
||||
r"if REQUEST_LOG_ENABLED:\s*\n\s*app\.add_middleware\(RequestResponseLogMiddleware\)",
|
||||
src,
|
||||
), (
|
||||
"REQUEST_LOG_ENABLED must gate the add_middleware call itself, not a branch inside "
|
||||
"the middleware — the whole point is that a disabled log costs nothing"
|
||||
)
|
||||
|
||||
|
||||
def test_cors_exposes_the_request_id_header():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
assert "expose_headers=" in src and "X-Request-ID" in src, (
|
||||
"without expose_headers the browser cannot read X-Request-ID on a cross-origin "
|
||||
"deployment, so the id is unusable from the app"
|
||||
)
|
||||
@@ -0,0 +1,251 @@
|
||||
"""v1.11.0: the request_logs migration actually runs on existing installs.
|
||||
|
||||
Source-scan tests (the sanctioned pattern here — there is no database in this
|
||||
suite). The failure mode being pinned is specific and silent: migrations are
|
||||
gated on `applied_version >= SCHEMA_VERSION`, so forgetting the bump means the
|
||||
whole sequence is skipped on every already-deployed database and neither the
|
||||
table nor the new permissions ever appear — while a fresh install works fine,
|
||||
so it looks correct in development.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
_BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
_MIGRATIONS = os.path.join(_BACKEND, "database", "migrations.py")
|
||||
_MAIN = os.path.join(_BACKEND, "main.py")
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def src():
|
||||
with open(_MIGRATIONS, encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def runner_body(src):
|
||||
"""The body of _run_all_migrations_inner, where steps are registered."""
|
||||
assert "async def _run_all_migrations_inner" in src
|
||||
return src.split("async def _run_all_migrations_inner", 1)[1].split("\nasync def ", 1)[0]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def rbac_body(src):
|
||||
return src.split("async def update_system_roles_to_enterprise_rbac", 1)[1].split("\nasync def ", 1)[0]
|
||||
|
||||
|
||||
def _role_block(rbac_body, role):
|
||||
"""Slice one role's permission list out of the enterprise_roles literal."""
|
||||
start = rbac_body.index(f"'{role}': {{")
|
||||
end = rbac_body.index("]", rbac_body.index("'permissions': [", start))
|
||||
return rbac_body[start:end]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The version gate
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_schema_version_bumped_to_at_least_11(src):
|
||||
match = re.search(r"^SCHEMA_VERSION\s*=\s*(\d+)", src, re.MULTILINE)
|
||||
assert match, "SCHEMA_VERSION assignment not found in migrations.py"
|
||||
assert int(match.group(1)) >= 11, (
|
||||
"SCHEMA_VERSION was not bumped for the request_logs table. run_all_migrations() "
|
||||
"returns early when the recorded version is already >= SCHEMA_VERSION, so every "
|
||||
"existing deployment would skip the whole run: no request_logs table, no "
|
||||
"requestlog.* permissions, and the feature would silently do nothing in production "
|
||||
"while working perfectly on a fresh database."
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Registration
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_both_migration_steps_are_registered(runner_body):
|
||||
assert "await ensure_request_logs_table()" in runner_body, (
|
||||
"ensure_request_logs_table is defined but never called from the migration runner"
|
||||
)
|
||||
assert "await ensure_request_log_settings()" in runner_body, (
|
||||
"the retention defaults are never seeded, so an upgraded install has no "
|
||||
"requestlog.* rows and Settings shows blanks"
|
||||
)
|
||||
|
||||
|
||||
def test_table_is_created_before_its_settings_are_seeded(runner_body):
|
||||
table_at = runner_body.index("await ensure_request_logs_table()")
|
||||
seed_at = runner_body.index("await ensure_request_log_settings()")
|
||||
assert table_at < seed_at, (
|
||||
"the settings seed runs before the table step; if the table step then raises, the "
|
||||
"run aborts with settings but no table"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The DDL itself
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def ddl_body(src):
|
||||
return src.split("async def ensure_request_logs_table", 1)[1].split("\nasync def ", 1)[0]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("fragment", [
|
||||
"CREATE TABLE IF NOT EXISTS request_logs",
|
||||
"id BIGSERIAL PRIMARY KEY",
|
||||
"request_id VARCHAR(64) NOT NULL",
|
||||
"direction VARCHAR(8) NOT NULL",
|
||||
"status_class SMALLINT",
|
||||
"created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()",
|
||||
"request_logs_direction_check",
|
||||
"client_ip INET",
|
||||
])
|
||||
def test_ddl_essentials(ddl_body, fragment):
|
||||
assert fragment in ddl_body, f"request_logs DDL is missing {fragment!r}"
|
||||
|
||||
|
||||
def test_ddl_is_idempotent(ddl_body):
|
||||
assert "CREATE TABLE IF NOT EXISTS" in ddl_body
|
||||
creates = re.findall(r"CREATE INDEX(?: IF NOT EXISTS)?", ddl_body)
|
||||
assert creates, "no indexes are created for request_logs"
|
||||
assert all(c == "CREATE INDEX IF NOT EXISTS" for c in creates), (
|
||||
"an index is created without IF NOT EXISTS — the second startup would raise and "
|
||||
"abort the whole migration run"
|
||||
)
|
||||
|
||||
|
||||
def test_prune_partial_indexes_are_present(ddl_body):
|
||||
"""The retention delete is split by outcome, so a plain
|
||||
(status_class, created_at) index would still range-scan the half it does not
|
||||
want."""
|
||||
assert "idx_request_logs_prune_ok" in ddl_body
|
||||
assert "idx_request_logs_prune_err" in ddl_body
|
||||
assert "WHERE status_class BETWEEN 1 AND 3" in ddl_body
|
||||
assert "WHERE status_class = 0 OR status_class >= 4" in ddl_body
|
||||
|
||||
|
||||
def test_request_id_index_exists_for_the_trace_view(ddl_body):
|
||||
assert "idx_request_logs_request_id" in ddl_body, (
|
||||
"without this index, opening one request to see the outbound calls it triggered "
|
||||
"is a sequential scan"
|
||||
)
|
||||
|
||||
|
||||
def test_no_foreign_key_on_user_id(ddl_body):
|
||||
"""Deliberate deviation from the house style — see the docstring in
|
||||
migrations.py. Pinned so it is not 'fixed' back into an FK later."""
|
||||
user_id_line = [line for line in ddl_body.splitlines() if "user_id " in line and "INTEGER" in line]
|
||||
assert user_id_line, "user_id column not found"
|
||||
assert "REFERENCES" not in user_id_line[0], (
|
||||
"an FK was added to request_logs.user_id — per-insert FK validation on the "
|
||||
"highest-volume table in the system, and audit rows must outlive the account"
|
||||
)
|
||||
|
||||
|
||||
def test_migration_step_reraises_on_failure(ddl_body):
|
||||
"""The version marker is written only after the inner sequence completes, so
|
||||
swallowing here would stamp version 11 with no table and the gate would then
|
||||
skip every retry, permanently."""
|
||||
assert re.search(r"\n\s+raise\n", ddl_body), (
|
||||
"ensure_request_logs_table swallows its exception instead of re-raising"
|
||||
)
|
||||
|
||||
|
||||
def test_settings_seed_does_not_overwrite_operator_tuning(src):
|
||||
seed_body = src.split("async def ensure_request_log_settings", 1)[1].split("\nasync def ", 1)[0]
|
||||
assert "ON CONFLICT (key) DO NOTHING" in seed_body, (
|
||||
"the seed uses DO UPDATE, so every upgrade would reset the operator's retention "
|
||||
"settings back to the defaults"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Permission seeding
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_super_admin_gets_both_permissions(rbac_body):
|
||||
block = _role_block(rbac_body, "super_admin")
|
||||
assert "'requestlog.read'" in block
|
||||
assert "'requestlog.manage'" in block
|
||||
|
||||
|
||||
def test_security_admin_gets_both_permissions(rbac_body):
|
||||
block = _role_block(rbac_body, "security_admin")
|
||||
assert "'requestlog.read'" in block
|
||||
assert "'requestlog.manage'" in block
|
||||
|
||||
|
||||
def test_operator_gets_read_only(rbac_body):
|
||||
block = _role_block(rbac_body, "operator")
|
||||
assert "'requestlog.read'" in block
|
||||
assert "'requestlog.manage'" not in block, (
|
||||
"operators should be able to read the log to debug an apply or an ACME order, but "
|
||||
"retention policy and purge belong to the admins"
|
||||
)
|
||||
|
||||
|
||||
def test_viewer_gets_neither(rbac_body):
|
||||
block = _role_block(rbac_body, "viewer")
|
||||
assert "requestlog" not in block, (
|
||||
"viewer was granted a requestlog permission. Even redacted, captured request and "
|
||||
"response bodies are a far broader disclosure surface than the read-only config "
|
||||
"views a viewer is meant to have."
|
||||
)
|
||||
|
||||
|
||||
def test_permission_strings_have_exactly_one_dot(rbac_body):
|
||||
"""get_user_permissions splits on the FIRST dot and silently drops any
|
||||
string without one."""
|
||||
for perm in re.findall(r"'(requestlog[^']*)'", rbac_body):
|
||||
assert perm.count(".") == 1, f"{perm!r} is not a <resource>.<action> pair"
|
||||
|
||||
|
||||
def test_initial_seed_lists_stay_in_sync(src):
|
||||
"""create_initial_system_data() is overwritten by the enterprise seeder on
|
||||
every run, but that seeder swallows all exceptions — keeping the two in sync
|
||||
is the safety net."""
|
||||
initial = src.split("system_roles = [", 1)[1].split("\n ]", 1)[0]
|
||||
assert '"requestlog.read"' in initial
|
||||
assert '"requestlog.manage"' in initial
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Runtime wiring
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_prune_loop_is_started_and_independent_of_the_acme_loop():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
main_src = f.read()
|
||||
|
||||
assert "async def prune_request_logs_loop" in main_src
|
||||
assert "asyncio.create_task(prune_request_logs_loop())" in main_src, (
|
||||
"the retention prune task is defined but never started, so request_logs grows "
|
||||
"without bound"
|
||||
)
|
||||
loop_body = main_src.split("async def prune_request_logs_loop", 1)[1].split("\n# Production middleware", 1)[0]
|
||||
assert "table_name = 'request_logs'" in loop_body, (
|
||||
"the prune loop does not check for its own table, so it would log an error every "
|
||||
"tick on a database where the migration has not run yet"
|
||||
)
|
||||
assert "table_name = 'letsencrypt_orders'" not in loop_body, (
|
||||
"the prune loop was gated on the ACME table, which would disable retention "
|
||||
"entirely on an install that never uses ACME"
|
||||
)
|
||||
|
||||
|
||||
def test_sink_is_flushed_before_the_pool_closes():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
main_src = f.read()
|
||||
|
||||
body = main_src.split("async def shutdown_event", 1)[1]
|
||||
flush_at = body.find("request_log_sink.flush")
|
||||
close_at = body.find("close_database_pool()")
|
||||
assert flush_at != -1, "queued request-log rows are never flushed on shutdown"
|
||||
assert flush_at < close_at, (
|
||||
"the sink is flushed after the pool is closed, so the queued rows are lost. The "
|
||||
"sink's writer is a `while True` loop and can never satisfy the generic "
|
||||
"asyncio.wait drain, so it needs its own explicit flush first."
|
||||
)
|
||||
@@ -0,0 +1,261 @@
|
||||
"""v1.11.0: retention actually reclaims space, and cannot be turned into an
|
||||
injection point or a 60-second lock.
|
||||
|
||||
`request_logs` is the highest-volume table in the system, so the prune has
|
||||
three properties that are easy to get wrong and expensive to get wrong:
|
||||
|
||||
* the operator-supplied retention day counts are BIND PARAMETERS, never
|
||||
string-interpolated into the SQL;
|
||||
* deletes are BATCHED, because the pool's command_timeout is 60s and an
|
||||
unbounded DELETE over millions of rows raises and then nothing is ever
|
||||
pruned;
|
||||
* the watermark is stamped only after a COMPLETE pass, so a pass that dies
|
||||
half-way is retried instead of being recorded as done.
|
||||
|
||||
The fake connection dispatches on the SQL text rather than on call order — an
|
||||
ordered side_effect list silently passes tests for the wrong reason as soon as
|
||||
the number of statements changes.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timedelta
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from dataclasses import replace # noqa: E402
|
||||
|
||||
from utils import request_log_prune # noqa: E402
|
||||
from utils.request_log_prune import ( # noqa: E402
|
||||
BATCH_SIZE,
|
||||
MAX_BATCHES,
|
||||
PRUNE_LOCK_KEY,
|
||||
prune_request_logs_if_due,
|
||||
)
|
||||
from utils.request_log_settings import DEFAULT_CONFIG # noqa: E402
|
||||
|
||||
_SUCCESS_MARKER = "status_class BETWEEN 1 AND 3"
|
||||
_ERROR_MARKER = "status_class = 0 OR status_class >= 4"
|
||||
_CAP_MARKER = "id <= $1"
|
||||
|
||||
|
||||
def _conn(*, lock=True, watermark_age_minutes=None, cutoff_id=None,
|
||||
success_batches=None, error_batches=None, cap_batches=None,
|
||||
fail_on=None):
|
||||
"""A fake asyncpg connection that answers by SQL shape."""
|
||||
conn = AsyncMock()
|
||||
|
||||
def fetchval(sql, *args):
|
||||
text = str(sql)
|
||||
if "pg_try_advisory_lock" in text:
|
||||
return lock
|
||||
if "ORDER BY id DESC OFFSET" in text:
|
||||
return cutoff_id
|
||||
return None
|
||||
|
||||
conn.fetchval = AsyncMock(side_effect=fetchval)
|
||||
|
||||
if watermark_age_minutes is None:
|
||||
conn.fetchrow = AsyncMock(return_value=None)
|
||||
else:
|
||||
stamp = (datetime.utcnow() - timedelta(minutes=watermark_age_minutes)).isoformat() + "Z"
|
||||
conn.fetchrow = AsyncMock(return_value={"value": json.dumps(stamp)})
|
||||
|
||||
queues = {
|
||||
_SUCCESS_MARKER: list(success_batches or ["DELETE 0"]),
|
||||
_ERROR_MARKER: list(error_batches or ["DELETE 0"]),
|
||||
_CAP_MARKER: list(cap_batches or ["DELETE 0"]),
|
||||
}
|
||||
|
||||
def execute(sql, *args):
|
||||
text = str(sql)
|
||||
if fail_on and fail_on in text:
|
||||
raise RuntimeError("statement timeout")
|
||||
for marker, queue in queues.items():
|
||||
if marker in text:
|
||||
return queue.pop(0) if queue else "DELETE 0"
|
||||
return ""
|
||||
|
||||
conn.execute = AsyncMock(side_effect=execute)
|
||||
return conn
|
||||
|
||||
|
||||
def _run(conn, *, force=False, **cfg_overrides):
|
||||
cfg = replace(DEFAULT_CONFIG, **cfg_overrides)
|
||||
with patch.object(request_log_prune, "get_config", lambda: cfg), \
|
||||
patch.object(request_log_prune, "get_database_connection", AsyncMock(return_value=conn)), \
|
||||
patch.object(request_log_prune, "close_database_connection", AsyncMock()):
|
||||
return asyncio.run(prune_request_logs_if_due(force=force))
|
||||
|
||||
|
||||
def _delete_sql(conn):
|
||||
return [str(c.args[0]) for c in conn.execute.call_args_list
|
||||
if "DELETE FROM request_logs" in str(c.args[0])]
|
||||
|
||||
|
||||
def test_skips_entirely_when_another_replica_holds_the_lock():
|
||||
conn = _conn(lock=False)
|
||||
counts = _run(conn)
|
||||
|
||||
assert counts == {"success": 0, "error": 0, "overflow": 0, "ran": 0}
|
||||
assert _delete_sql(conn) == [], (
|
||||
"a second replica ran the prune concurrently — pg_try_advisory_lock is what keeps "
|
||||
"N pods from all scanning the same table at once"
|
||||
)
|
||||
|
||||
|
||||
def test_skips_when_the_watermark_is_still_fresh():
|
||||
conn = _conn(watermark_age_minutes=10)
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
assert counts["ran"] == 0
|
||||
assert _delete_sql(conn) == []
|
||||
|
||||
|
||||
def test_runs_all_three_limits_when_due():
|
||||
conn = _conn(
|
||||
watermark_age_minutes=120, cutoff_id=999,
|
||||
success_batches=["DELETE 3"], error_batches=["DELETE 4"], cap_batches=["DELETE 5"],
|
||||
)
|
||||
counts = _run(conn, prune_interval_minutes=60, success_retention_days=7,
|
||||
error_retention_days=30, max_rows=500000)
|
||||
|
||||
sqls = _delete_sql(conn)
|
||||
assert len(sqls) == 3, f"expected success TTL + error TTL + row cap, got {len(sqls)}"
|
||||
assert _SUCCESS_MARKER in sqls[0]
|
||||
assert _ERROR_MARKER in sqls[1]
|
||||
assert _CAP_MARKER in sqls[2]
|
||||
|
||||
assert counts["success"] == 3
|
||||
assert counts["error"] == 4
|
||||
assert counts["overflow"] == 5
|
||||
assert counts["ran"] == 1
|
||||
|
||||
|
||||
def test_retention_days_travel_as_bind_parameters():
|
||||
"""Injection guard: the day counts come straight from an operator-editable
|
||||
setting, so they must never be formatted into the SQL text."""
|
||||
conn = _conn(watermark_age_minutes=120)
|
||||
_run(conn, prune_interval_minutes=60, success_retention_days=7, error_retention_days=30)
|
||||
|
||||
ttl_calls = [c for c in conn.execute.call_args_list
|
||||
if "created_at < NOW()" in str(c.args[0])]
|
||||
assert len(ttl_calls) == 2
|
||||
|
||||
for call in ttl_calls:
|
||||
assert "($1 || ' days')::INTERVAL" in str(call.args[0]), (
|
||||
"the retention window is interpolated into the SQL string instead of bound — "
|
||||
"an operator-supplied value reaching the parser is an injection point"
|
||||
)
|
||||
|
||||
assert ttl_calls[0].args[1] == "7"
|
||||
assert ttl_calls[1].args[1] == "30"
|
||||
assert ttl_calls[0].args[2] == BATCH_SIZE
|
||||
|
||||
|
||||
def test_deletes_are_batched_until_a_short_batch():
|
||||
conn = _conn(
|
||||
watermark_age_minutes=120,
|
||||
success_batches=[f"DELETE {BATCH_SIZE}", f"DELETE {BATCH_SIZE}", "DELETE 12"],
|
||||
)
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
assert counts["success"] == BATCH_SIZE * 2 + 12, (
|
||||
"the batch loop stopped early or double-counted"
|
||||
)
|
||||
success_calls = [s for s in _delete_sql(conn) if _SUCCESS_MARKER in s]
|
||||
assert len(success_calls) == 3, "the loop must stop on the first short batch"
|
||||
|
||||
|
||||
def test_batch_loop_respects_the_ceiling():
|
||||
"""A table so far behind that every batch comes back full must still hand the
|
||||
connection back rather than looping forever."""
|
||||
conn = _conn(
|
||||
watermark_age_minutes=120,
|
||||
success_batches=[f"DELETE {BATCH_SIZE}"] * (MAX_BATCHES * 3),
|
||||
)
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
assert counts["success"] == BATCH_SIZE * MAX_BATCHES
|
||||
success_calls = [s for s in _delete_sql(conn) if _SUCCESS_MARKER in s]
|
||||
assert len(success_calls) == MAX_BATCHES
|
||||
|
||||
|
||||
def test_watermark_is_not_stamped_when_a_step_fails():
|
||||
conn = _conn(watermark_age_minutes=120, cutoff_id=42, fail_on=_CAP_MARKER)
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
stamps = [c for c in conn.execute.call_args_list
|
||||
if "INSERT INTO system_settings" in str(c.args[0])]
|
||||
assert stamps == [], (
|
||||
"a partially-completed pass stamped the watermark, so the remainder would not be "
|
||||
"retried until the next interval"
|
||||
)
|
||||
assert counts["ran"] == 0
|
||||
|
||||
|
||||
def test_watermark_is_stamped_after_a_complete_pass():
|
||||
conn = _conn(watermark_age_minutes=120, cutoff_id=None)
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
stamps = [c for c in conn.execute.call_args_list
|
||||
if "INSERT INTO system_settings" in str(c.args[0])]
|
||||
assert len(stamps) == 1
|
||||
# args = (sql, key, json_value)
|
||||
assert stamps[0].args[1] == "requestlog.last_pruned_at"
|
||||
assert stamps[0].args[2].startswith('"'), (
|
||||
"the watermark must be stored as a JSON string — the ::jsonb cast rejects a bare "
|
||||
"timestamp, and the reader json.loads() it back"
|
||||
)
|
||||
assert counts["ran"] == 1
|
||||
|
||||
|
||||
def test_advisory_lock_is_released_even_on_failure():
|
||||
conn = _conn(watermark_age_minutes=120, fail_on=_SUCCESS_MARKER)
|
||||
_run(conn, prune_interval_minutes=60)
|
||||
|
||||
unlocks = [c for c in conn.execute.call_args_list if "pg_advisory_unlock" in str(c.args[0])]
|
||||
assert unlocks, "the advisory lock was leaked — every later pass on any replica would skip"
|
||||
assert unlocks[0].args[1] == PRUNE_LOCK_KEY
|
||||
|
||||
|
||||
def test_never_raises_when_the_pool_is_exhausted():
|
||||
with patch.object(request_log_prune, "get_database_connection",
|
||||
AsyncMock(side_effect=RuntimeError("pool exhausted"))), \
|
||||
patch.object(request_log_prune, "close_database_connection", AsyncMock()):
|
||||
counts = asyncio.run(prune_request_logs_if_due())
|
||||
|
||||
assert counts == {"success": 0, "error": 0, "overflow": 0, "ran": 0}
|
||||
|
||||
|
||||
def test_row_cap_is_a_noop_when_the_table_is_smaller_than_the_cap():
|
||||
conn = _conn(watermark_age_minutes=120, cutoff_id=None, cap_batches=["DELETE 77"])
|
||||
counts = _run(conn, prune_interval_minutes=60)
|
||||
|
||||
assert counts["overflow"] == 0, (
|
||||
"the cap deleted rows even though OFFSET max_rows found no cutoff — that would "
|
||||
"truncate a table that is under the limit"
|
||||
)
|
||||
assert not any(_CAP_MARKER in s for s in _delete_sql(conn))
|
||||
|
||||
|
||||
def test_force_bypasses_the_watermark():
|
||||
"""The manual purge button must not be a no-op just because the scheduled
|
||||
pass ran a minute ago."""
|
||||
conn = _conn(watermark_age_minutes=1, cutoff_id=None,
|
||||
success_batches=["DELETE 1"], error_batches=["DELETE 2"])
|
||||
counts = _run(conn, force=True, prune_interval_minutes=1440)
|
||||
|
||||
assert counts["ran"] == 1
|
||||
assert counts["success"] == 1
|
||||
assert counts["error"] == 2
|
||||
|
||||
|
||||
def test_lock_key_does_not_collide_with_the_existing_ones():
|
||||
# 18181818 drafts cap, 18181819 wizard create, 18181820 apply,
|
||||
# 0x41434D45 per-ACME-order, 1836016242 migration.
|
||||
assert PRUNE_LOCK_KEY not in (18181818, 18181819, 18181820, 0x41434D45, 1836016242)
|
||||
@@ -0,0 +1,289 @@
|
||||
"""v1.11.0: redaction pinned against THIS codebase's real payloads.
|
||||
|
||||
test_request_log_redaction.py pins the RULES — which key names match, which
|
||||
value shapes fire. It passed 72/72 while six real endpoints of this application
|
||||
still wrote secrets to `request_logs`, because a rule test proves the rule, not
|
||||
the coverage. Every case here is built from an actual handler's request or
|
||||
response shape, with the field names taken from the source and named in the
|
||||
docstring, so a future change to redaction is measured against what this system
|
||||
actually sends rather than against what someone remembered to imagine.
|
||||
|
||||
Method note: the payloads go through `decode_body()`, the same entry point the
|
||||
writer task uses, rather than calling `redact()` directly. That is deliberate —
|
||||
two of the findings below only appear on the way in (an oversized body never
|
||||
reaches `redact()` as a dict at all, it arrives as one `_raw` string), so a test
|
||||
that starts from a dict would report a pass on a payload that leaks in
|
||||
production.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from utils.request_log_redaction import ( # noqa: E402
|
||||
decode_body,
|
||||
is_secret_key,
|
||||
redact_headers,
|
||||
)
|
||||
|
||||
VRRP_SECRET = "S3cr3tVrrpPass!"
|
||||
TOTP_SECRET = "JBSWY3DPEHPK3PXP"
|
||||
STATS_PASSWORD = "StatsPa55word"
|
||||
USERLIST_HASH = "$6$rounds=5000$abcdefgh$XyZ"
|
||||
JWT = (
|
||||
"eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9"
|
||||
".eyJzdWIiOiIxIiwidXNlcm5hbWUiOiJhZG1pbiJ9"
|
||||
".dQw4w9WgXcQdQw4w9WgXcQdQw4w9WgXcQ"
|
||||
)
|
||||
PEM_KEY = (
|
||||
"-----BEGIN RSA PRIVATE KEY-----\n"
|
||||
+ "MIIEowIBAAKCAQEA" + "A" * 200 + "\n"
|
||||
+ "-----END RSA PRIVATE KEY-----\n"
|
||||
)
|
||||
|
||||
# The rendered file `GET /api/agents/{n}/keepalived-config` hands to an agent,
|
||||
# and the one `POST /api/agents/{n}/keepalived-discovery` sends back.
|
||||
KEEPALIVED_CONF = f"""! Managed by HAProxy OpenManager
|
||||
vrrp_instance VI_1 {{
|
||||
state MASTER
|
||||
interface eth0
|
||||
virtual_router_id 51
|
||||
priority 200
|
||||
advert_int 1
|
||||
authentication {{
|
||||
auth_type PASS
|
||||
auth_pass {VRRP_SECRET}
|
||||
}}
|
||||
virtual_ipaddress {{
|
||||
10.20.30.40/24
|
||||
}}
|
||||
}}
|
||||
"""
|
||||
|
||||
# A production haproxy.cfg as the agent uploads it verbatim from the node
|
||||
# (`config_content=$(cat "$config_path")` -> POST .../config-response).
|
||||
HAPROXY_CFG = f"""global
|
||||
log stdout local0
|
||||
stats socket /var/run/haproxy.sock mode 660
|
||||
|
||||
userlist admins
|
||||
user ops password {USERLIST_HASH}
|
||||
user dev insecure-password Hunter2Plain
|
||||
|
||||
listen stats
|
||||
bind *:8404
|
||||
stats enable
|
||||
stats auth admin:{STATS_PASSWORD}
|
||||
stats uri /stats
|
||||
|
||||
backend web
|
||||
server web1 10.0.0.1:80 check
|
||||
"""
|
||||
|
||||
|
||||
def _capture(payload, *, cap=8192, content_type="application/json"):
|
||||
"""Run a payload through the capture path exactly as the writer does.
|
||||
|
||||
`cap` is `requestlog.max_body_bytes`. Bodies larger than it arrive
|
||||
truncated, do not parse as JSON, and land in the `{"_raw": ...}` fallback —
|
||||
which is the common case for config uploads and the case a dict-based test
|
||||
never exercises.
|
||||
"""
|
||||
body = json.dumps(payload).encode()
|
||||
value, truncated = decode_body(body[:cap], content_type, len(body))
|
||||
return json.dumps(value, default=str), truncated
|
||||
|
||||
|
||||
def _assert_absent(rendered, *secrets):
|
||||
for secret in secrets:
|
||||
assert secret not in rendered, (
|
||||
f"{secret!r} reached request_logs. Rendered row: {rendered[:400]}"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The VRRP password. routers/vip.py: "the secret never leaves the server in
|
||||
# cleartext ... only the at-rest Fernet token and the agent-delivery endpoint
|
||||
# ever see the real value."
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_vip_create_body_does_not_store_auth_pass():
|
||||
"""POST/PUT /api/vip — `payload.auth_pass`, routers/vip.py:585,679."""
|
||||
rendered, _ = _capture({
|
||||
"name": "vip-prod", "virtual_ip": "10.20.30.40", "interface": "eth0",
|
||||
"virtual_router_id": 51, "auth_pass": VRRP_SECRET,
|
||||
})
|
||||
_assert_absent(rendered, VRRP_SECRET)
|
||||
|
||||
|
||||
def test_keepalived_config_delivery_does_not_store_the_rendered_secret():
|
||||
"""GET /api/agents/{n}/keepalived-config — `keepalived.config_content`.
|
||||
|
||||
Polled on the SSL cadence, so an unmasked capture rewrites the secret to the
|
||||
audit table roughly 576 times a day per member node.
|
||||
"""
|
||||
rendered, _ = _capture({
|
||||
"agent_name": "prod-lb-01", "status": "available",
|
||||
"config_path": "/etc/keepalived/keepalived.conf",
|
||||
"keepalived": {
|
||||
"vip_id": 3, "vip_name": "vip-prod",
|
||||
"config_content": KEEPALIVED_CONF, "config_hash": "abc123",
|
||||
},
|
||||
})
|
||||
_assert_absent(rendered, VRRP_SECRET)
|
||||
assert "auth_pass" in rendered, "the directive should stay visible, only its value masked"
|
||||
assert "vrrp_instance VI_1" in rendered, "masking must not destroy the rest of the config"
|
||||
|
||||
|
||||
def test_keepalived_discovery_body_does_not_store_the_found_secret():
|
||||
"""POST /api/agents/{n}/keepalived-discovery — `config_content`.
|
||||
|
||||
routers/agent.py already pops auth_pass out of the parsed analysis,
|
||||
Fernet-encrypts it into its own column and stores only
|
||||
`vip_discoveries.raw_config_masked`. Capturing the request that produced all
|
||||
that, unmasked, would put the plaintext straight back next to it.
|
||||
"""
|
||||
rendered, _ = _capture({
|
||||
"agent_name": "prod-lb-01", "exists": True, "is_managed": False,
|
||||
"config_path": "/etc/keepalived/keepalived.conf",
|
||||
"config_content": KEEPALIVED_CONF,
|
||||
})
|
||||
_assert_absent(rendered, VRRP_SECRET)
|
||||
|
||||
|
||||
def test_auth_pass_is_masked_when_the_body_is_too_large_to_parse():
|
||||
"""The truncated `_raw` path, where line breaks are the escape `\\n`.
|
||||
|
||||
A value pattern that stops only at a REAL newline runs to the end of the
|
||||
string here: no leak, but the whole remainder of the config is masked and
|
||||
the row is useless. Both properties are asserted.
|
||||
"""
|
||||
payload = {"config_content": KEEPALIVED_CONF + "backend b\n server s1 10.0.0.1:80 check\n" * 400}
|
||||
rendered, truncated = _capture(payload)
|
||||
assert truncated, "this fixture must exercise the truncated path"
|
||||
_assert_absent(rendered, VRRP_SECRET)
|
||||
assert "server s1 10.0.0.1:80" in rendered, (
|
||||
"masking ran past the end of the auth_pass line and ate the rest of the config"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# TOTP. routers/mfa.py logs `{"secret_len": ...}` with the comment
|
||||
# "NEVER log the secret itself".
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_mfa_enroll_response_does_not_store_the_totp_secret_in_either_field():
|
||||
"""POST /api/mfa/enroll — returns `secret` AND `otpauth_uri`.
|
||||
|
||||
Redacting one while the same value sits in the other is not redaction.
|
||||
"""
|
||||
rendered, _ = _capture({
|
||||
"secret": TOTP_SECRET,
|
||||
"otpauth_uri": f"otpauth://totp/OpenManager:admin?secret={TOTP_SECRET}&issuer=OpenManager",
|
||||
"qr_size": 256,
|
||||
})
|
||||
_assert_absent(rendered, TOTP_SECRET)
|
||||
|
||||
|
||||
def test_userinfo_credentials_in_a_url_are_dropped():
|
||||
rendered, _ = _capture({"webhook": "https://svc:Sup3rSecret@hooks.example.com/notify?api_key=abc123"})
|
||||
_assert_absent(rendered, "Sup3rSecret", "abc123")
|
||||
assert "hooks.example.com" in rendered, "the host is the diagnostic value; keep it"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# HAProxy config. We never RENDER credentials into one, but the agent uploads
|
||||
# the node's real file and the operator can paste one.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("cap,label", [(8192, "truncated _raw path"), (10 ** 6, "parsed path")])
|
||||
def test_uploaded_haproxy_config_masks_credentials_on_both_paths(cap, label):
|
||||
"""POST /api/configuration/agents/{n}/config-response, and
|
||||
POST /api/config/validate."""
|
||||
rendered, _ = _capture({"config_content": HAPROXY_CFG, "config_path": "/etc/haproxy/haproxy.cfg"}, cap=cap)
|
||||
_assert_absent(rendered, STATS_PASSWORD, USERLIST_HASH, "Hunter2Plain")
|
||||
assert "stats auth admin:" in rendered, f"[{label}] the account name is diagnostic; keep it"
|
||||
assert "server web1 10.0.0.1:80" in rendered, f"[{label}] the rest of the config must survive"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("prose", [
|
||||
"invalid password format",
|
||||
"the password must be at least 8 characters",
|
||||
"authentication failed for user admin",
|
||||
])
|
||||
def test_ordinary_prose_is_not_mangled(prose):
|
||||
"""Over-matching would blank the messages the log exists to show."""
|
||||
rendered, _ = _capture({"detail": prose})
|
||||
assert prose in rendered, f"redaction damaged an ordinary message: {rendered}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Regressions guarding what already worked, so a later rule change cannot
|
||||
# quietly trade one of these away for one of the above.
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_login_exchange_stores_neither_the_password_nor_the_token():
|
||||
req, _ = _capture({"username": "admin", "password": "hunter2hunter2"})
|
||||
_assert_absent(req, "hunter2hunter2")
|
||||
res, _ = _capture({"access_token": JWT, "token_type": "bearer", "user": {"id": 1}})
|
||||
_assert_absent(res, JWT)
|
||||
|
||||
|
||||
def test_private_key_is_redacted_even_under_an_innocent_key_name():
|
||||
"""The value-shape guard is the net under the key-name rules."""
|
||||
rendered, _ = _capture({"blob": PEM_KEY, "note": "backup"})
|
||||
_assert_absent(rendered, "MIIEowIBAAKCAQEA")
|
||||
|
||||
|
||||
def test_dns_provider_credentials_are_redacted():
|
||||
cf, _ = _capture({"provider": "cloudflare", "api_token": "cf_live_abcdefghijklmnop", "zone_id": "z1"})
|
||||
_assert_absent(cf, "cf_live_abcdefghijklmnop")
|
||||
gd, _ = _capture({"provider": "godaddy", "api_key": "gd_key_1234567890", "api_secret": "gd_secret_098"})
|
||||
_assert_absent(gd, "gd_key_1234567890", "gd_secret_098")
|
||||
|
||||
|
||||
def test_innocent_urls_survive_untouched():
|
||||
"""Scrubbing must not rewrite the ACME URLs an operator reads back."""
|
||||
for url in (
|
||||
"https://acme-v02.api.letsencrypt.org/directory",
|
||||
"https://acme-v02.api.letsencrypt.org/acme/acct/12345",
|
||||
):
|
||||
rendered, _ = _capture({"directory_url": url})
|
||||
assert url in rendered, f"an innocent URL was rewritten: {rendered}"
|
||||
|
||||
|
||||
def test_credential_headers_are_presence_only_and_the_rest_are_dropped():
|
||||
out = redact_headers({
|
||||
"authorization": f"Bearer {JWT}",
|
||||
"x-api-key": "agt_" + "a" * 32,
|
||||
"cookie": "session=abc123",
|
||||
"user-agent": "curl/8.4.0",
|
||||
"x-forwarded-for": "10.20.30.5",
|
||||
"x-internal-secret": "not-on-the-allowlist",
|
||||
})
|
||||
rendered = json.dumps(out)
|
||||
_assert_absent(rendered, JWT, "agt_" + "a" * 32, "abc123", "not-on-the-allowlist")
|
||||
assert out["user-agent"] == "curl/8.4.0"
|
||||
assert out["x-forwarded-for"] == "10.20.30.5"
|
||||
assert "x-internal-secret" not in out, "an unlisted header must be dropped, not kept"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", ["auth_pass", "authPass", "auth-pass", "AUTH_PASS"])
|
||||
def test_auth_pass_key_matches_in_every_spelling(key):
|
||||
assert is_secret_key(key), (
|
||||
f"{key!r} normalizes to something no rule matches. 'password' is not a "
|
||||
f"substring of 'authpass' and the bare 'auth' entry is an exact match."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", [
|
||||
"monkey", "key_suffix", "payload_size", "nonce_count", "keyboard_layout",
|
||||
"config_path", "authenticated", "author",
|
||||
])
|
||||
def test_innocent_field_names_are_still_kept(key):
|
||||
"""The other half of the trade: over-redaction blanks the fields the
|
||||
feature exists to show."""
|
||||
assert not is_secret_key(key)
|
||||
@@ -0,0 +1,262 @@
|
||||
"""v1.11.0: nothing secret reaches request_logs.
|
||||
|
||||
The request/response log stores bodies and headers, so redaction is the single
|
||||
control standing between "operators can debug a failing ACME order" and "the
|
||||
audit table is a credential store". These tests pin both halves of that: the
|
||||
things that MUST be redacted, and the innocent field names that must NOT be
|
||||
(over-matching would silently blank out the fields the feature exists to show).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from utils.request_log_redaction import ( # noqa: E402
|
||||
REDACTED,
|
||||
decode_body,
|
||||
is_capturable_content_type,
|
||||
is_secret_key,
|
||||
redact,
|
||||
redact_headers,
|
||||
safe_error_text,
|
||||
scrub_query_string,
|
||||
scrub_url,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Key matching
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("key", [
|
||||
"password", "PASSWORD", "Pass_Word", "passwd", "pwd",
|
||||
# api_token is the literal field name of the Cloudflare provider credential
|
||||
# (services/dns_providers/cloudflare.py) — it must never survive a round trip.
|
||||
"token", "access_token", "refreshToken", "MFA_TOKEN",
|
||||
"api_token", "agent_token", "csrf_token", "session_token",
|
||||
"api_key", "API-KEY", "apiKey", "x-api-key",
|
||||
"secret", "client_secret", "eab_hmac_key",
|
||||
"private_key", "cert_private_key", "csr_private_key", "jwk_private_key",
|
||||
"authorization", "cookie", "set-cookie",
|
||||
"signature", "protected", "payload", "nonce", "replay-nonce",
|
||||
"key_authorization", "backup_codes", "totp_secret",
|
||||
"stats_password", "credentials", "encryption_key",
|
||||
])
|
||||
def test_secret_keys_are_detected(key):
|
||||
assert is_secret_key(key), f"{key!r} must be treated as a secret field name"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", [
|
||||
# Every one of these has a secret-looking substring but is innocent. If any
|
||||
# starts redacting, the log stops being useful for the exact debugging it
|
||||
# was built for.
|
||||
"key_suffix", "monkey", "keyboard", "turkey",
|
||||
"payload_size", "nonce_count",
|
||||
"public_key_id", "keys_total",
|
||||
"name", "status_code", "duration_ms", "domain", "directory_url",
|
||||
])
|
||||
def test_innocent_keys_are_not_redacted(key):
|
||||
assert not is_secret_key(key), (
|
||||
f"{key!r} was redacted by over-matching — the log would blank out a field "
|
||||
f"operators need"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Recursive body redaction
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_nested_dicts_and_lists_are_redacted_recursively():
|
||||
body = {
|
||||
"user": {"username": "admin", "password": "hunter2"},
|
||||
"accounts": [
|
||||
{"email": "a@example.com", "eab_hmac_key": "s3cr3t"},
|
||||
{"email": "b@example.com", "api_token": "cf-token"},
|
||||
],
|
||||
"cluster_id": 3,
|
||||
}
|
||||
out = redact(body)
|
||||
|
||||
assert out["user"]["username"] == "admin"
|
||||
assert out["user"]["password"] == REDACTED
|
||||
assert out["accounts"][0]["email"] == "a@example.com"
|
||||
assert out["accounts"][0]["eab_hmac_key"] == REDACTED
|
||||
assert out["accounts"][1]["api_token"] == REDACTED
|
||||
assert out["cluster_id"] == 3
|
||||
|
||||
|
||||
def test_depth_limit_stops_runaway_nesting():
|
||||
deep = current = {}
|
||||
for _ in range(20):
|
||||
current["child"] = {}
|
||||
current = current["child"]
|
||||
current["password"] = "leak"
|
||||
|
||||
out = redact(deep)
|
||||
flattened = json.dumps(out)
|
||||
assert "***DEPTH_LIMIT***" in flattened
|
||||
assert "leak" not in flattened
|
||||
|
||||
|
||||
def test_node_budget_bounds_a_very_wide_body():
|
||||
wide = {f"field_{i}": i for i in range(5000)}
|
||||
out = redact(wide)
|
||||
assert out.get("_node_limit") is True
|
||||
assert len(out) < 5000, "node budget did not bound a pathologically wide body"
|
||||
|
||||
|
||||
def test_pem_private_key_is_redacted_by_value_shape():
|
||||
body = {"blob": "-----BEGIN RSA PRIVATE KEY-----\n" + "A" * 200 + "\n-----END RSA PRIVATE KEY-----"}
|
||||
out = redact(body)
|
||||
assert out["blob"] == REDACTED, (
|
||||
"a PEM private key under an innocent key name was stored verbatim"
|
||||
)
|
||||
|
||||
|
||||
def test_jwt_shaped_string_is_redacted_by_value_shape():
|
||||
jwt_like = "eyJhbGciOiJIUzI1NiJ9." + "a" * 40 + "." + "b" * 40
|
||||
out = redact({"data": jwt_like})
|
||||
assert out["data"] == REDACTED
|
||||
|
||||
|
||||
def test_long_strings_are_truncated_with_a_marker():
|
||||
out = redact({"note": "x" * 9000})
|
||||
assert out["note"].endswith("chars]")
|
||||
assert len(out["note"]) < 9000
|
||||
|
||||
|
||||
def test_redact_never_raises_on_odd_input():
|
||||
class Weird:
|
||||
def __repr__(self):
|
||||
raise RuntimeError("boom")
|
||||
|
||||
# Non-serializable leaf values must pass straight through, not explode.
|
||||
assert redact({"x": Weird()}) is not None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Headers (allowlist)
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_headers_use_an_allowlist_with_presence_markers():
|
||||
out = redact_headers({
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": "curl/8.0",
|
||||
"Authorization": "Bearer super-secret-token",
|
||||
"Cookie": "session=abc",
|
||||
"X-Custom-Internal": "some value",
|
||||
})
|
||||
|
||||
assert out["content-type"] == "application/json"
|
||||
assert out["user-agent"] == "curl/8.0"
|
||||
# Presence is useful when debugging a 401; the value is not.
|
||||
assert out["authorization"] == REDACTED
|
||||
assert out["cookie"] == REDACTED
|
||||
# Not on the allowlist and not a known credential header -> dropped entirely.
|
||||
assert "x-custom-internal" not in out
|
||||
|
||||
|
||||
def test_redact_headers_handles_none():
|
||||
assert redact_headers(None) is None
|
||||
assert redact_headers({}) is None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# URLs and query strings
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_query_string_secrets_are_scrubbed():
|
||||
scrubbed, as_dict = scrub_query_string("token=abc123&page=2&api_key=xyz")
|
||||
assert "abc123" not in scrubbed
|
||||
assert "xyz" not in scrubbed
|
||||
assert "page=2" in scrubbed
|
||||
assert as_dict["token"] == REDACTED
|
||||
assert as_dict["page"] == "2"
|
||||
|
||||
|
||||
def test_scrub_url_strips_userinfo_and_query_secrets():
|
||||
out = scrub_url("https://user:hunter2@api.example.com:8443/v1/zones?api_key=abc&page=1")
|
||||
assert "hunter2" not in out
|
||||
assert "user" not in out.split("/v1")[0].replace("api.example.com", "")
|
||||
assert "abc" not in out
|
||||
assert "api.example.com:8443" in out
|
||||
assert "page=1" in out
|
||||
|
||||
|
||||
def test_scrub_url_drops_the_fragment():
|
||||
# Fragments never reach a server, and they are a classic token carrier.
|
||||
assert "#" not in scrub_url("https://example.com/x?a=1#access_token=leak")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Body decoding, capping, truncation marker
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_decode_body_parses_and_redacts_json():
|
||||
raw = json.dumps({"username": "admin", "password": "hunter2"}).encode()
|
||||
value, truncated = decode_body(raw, "application/json", len(raw))
|
||||
assert value["username"] == "admin"
|
||||
assert value["password"] == REDACTED
|
||||
assert truncated is False
|
||||
|
||||
|
||||
def test_decode_body_marks_truncation_with_the_original_size():
|
||||
full = b"x" * 20000
|
||||
captured = full[:1024]
|
||||
value, truncated = decode_body(captured, "text/plain", len(full))
|
||||
assert truncated is True
|
||||
assert value["_truncated"] is True
|
||||
assert value["_original_bytes"] == 20000
|
||||
|
||||
|
||||
def test_decode_body_wraps_non_json_as_raw_object():
|
||||
value, _ = decode_body(b"plain text response", "text/plain", 19)
|
||||
assert value == {"_raw": "plain text response"}
|
||||
|
||||
|
||||
def test_decode_body_survives_truncated_json():
|
||||
# A JSON body cut off at the cap will not parse — keep the prefix rather
|
||||
# than losing the field entirely.
|
||||
value, truncated = decode_body(b'{"a": "bb', "application/json", 500)
|
||||
assert truncated is True
|
||||
assert "_raw" in value
|
||||
|
||||
|
||||
def test_decode_body_parses_form_encoded():
|
||||
value, _ = decode_body(b"username=admin&password=hunter2",
|
||||
"application/x-www-form-urlencoded", 30)
|
||||
assert value["username"] == "admin"
|
||||
assert value["password"] == REDACTED
|
||||
|
||||
|
||||
def test_decode_body_returns_none_for_empty():
|
||||
assert decode_body(b"", "application/json", 0) == (None, False)
|
||||
assert decode_body(None, "application/json", 0) == (None, False)
|
||||
|
||||
|
||||
def test_binary_content_types_are_not_capturable():
|
||||
assert is_capturable_content_type("application/json")
|
||||
assert is_capturable_content_type("application/json; charset=utf-8")
|
||||
assert is_capturable_content_type("text/plain")
|
||||
assert not is_capturable_content_type("application/octet-stream")
|
||||
assert not is_capturable_content_type("image/png")
|
||||
assert not is_capturable_content_type("text/event-stream")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Error rendering
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_safe_error_text_type_only_hides_the_message():
|
||||
exc = ValueError("https://api.godaddy.com/v1/domains/secret-zone/records failed")
|
||||
assert safe_error_text(exc, type_only=True) == "ValueError"
|
||||
assert "godaddy" not in safe_error_text(exc, type_only=True)
|
||||
|
||||
|
||||
def test_safe_error_text_includes_the_message_when_allowed():
|
||||
text = safe_error_text(RuntimeError("connection refused"))
|
||||
assert text.startswith("RuntimeError")
|
||||
assert "connection refused" in text
|
||||
@@ -0,0 +1,205 @@
|
||||
"""v1.11.0: the request log API is gated, and its routes resolve.
|
||||
|
||||
Two distinct failure modes are pinned here.
|
||||
|
||||
**Auth.** The table holds redacted-but-real request and response bodies for
|
||||
every user, so an unauthenticated or under-privileged caller must never get a
|
||||
row. There is no database in this suite, so the behavioural checks assert only
|
||||
that an anonymous call is rejected before any DB work — which is exactly the
|
||||
property that matters — and a source scan covers the per-endpoint permission.
|
||||
|
||||
**Route order.** `/{log_id}` is a single-segment path and FastAPI matches in
|
||||
declaration order, so declaring it before `/settings`, `/stats` or `/purge`
|
||||
makes those three unreachable (they parse as a log id and 422). This is the
|
||||
mirror image of the shadowing trap already present in routers/settings.py.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
_BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
_ROUTER = os.path.join(_BACKEND, "routers", "request_logs.py")
|
||||
_MAIN = os.path.join(_BACKEND, "main.py")
|
||||
|
||||
REJECT = (401, 403, 422)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def src():
|
||||
with open(_ROUTER, encoding="utf-8") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Behavioural: nothing is readable without credentials
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("method,path", [
|
||||
("get", "/api/request-logs"),
|
||||
("get", "/api/request-logs/1"),
|
||||
("get", "/api/request-logs/stats"),
|
||||
("get", "/api/request-logs/settings"),
|
||||
("put", "/api/request-logs/settings"),
|
||||
("post", "/api/request-logs/purge"),
|
||||
])
|
||||
def test_anonymous_access_is_rejected(client, method, path):
|
||||
res = getattr(client, method)(path) if method != "put" else client.put(path, json={})
|
||||
assert res.status_code in REJECT, (
|
||||
f"{method.upper()} {path} returned {res.status_code} without an Authorization "
|
||||
f"header — the request log contains captured bodies for every user"
|
||||
)
|
||||
|
||||
|
||||
def test_a_garbage_token_is_rejected(client):
|
||||
res = client.get("/api/request-logs", headers={"authorization": "Bearer not-a-token"})
|
||||
assert res.status_code in REJECT
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Source scan: per-endpoint permission
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def _handler_body(src, decorator):
|
||||
start = src.index(decorator)
|
||||
rest = src[start + len(decorator):]
|
||||
end = rest.find("\n@router.")
|
||||
return rest if end == -1 else rest[:end]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("decorator,action", [
|
||||
('@router.get("/settings")', "manage"),
|
||||
('@router.put("/settings")', "manage"),
|
||||
('@router.get("/stats")', "read"),
|
||||
('@router.post("/purge")', "manage"),
|
||||
('@router.get("")', "read"),
|
||||
('@router.get("/{log_id}")', "read"),
|
||||
])
|
||||
def test_every_endpoint_enforces_its_permission(src, decorator, action):
|
||||
body = _handler_body(src, decorator)
|
||||
assert f'_require(authorization, "{action}")' in body, (
|
||||
f"{decorator} does not enforce requestlog.{action}"
|
||||
)
|
||||
|
||||
|
||||
def test_require_helper_raises_403_not_a_silent_pass(src):
|
||||
helper = src.split("async def _require", 1)[1].split("\nasync def ", 1)[0]
|
||||
assert "check_user_permission" in helper
|
||||
assert "status_code=403" in helper
|
||||
assert "current_user=current_user" in helper, (
|
||||
"the admin bypass is skipped, so every call pays an extra SELECT on users"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Route declaration order
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("literal", ['@router.get("/settings")', '@router.put("/settings")',
|
||||
'@router.get("/stats")', '@router.post("/purge")'])
|
||||
def test_literal_routes_are_declared_before_the_catch_all(src, literal):
|
||||
catch_all = src.index('@router.get("/{log_id}")')
|
||||
assert src.index(literal) < catch_all, (
|
||||
f"{literal} is declared after GET /{{log_id}}. FastAPI matches in declaration "
|
||||
f"order and /{{log_id}} is a single-segment path, so it would swallow this route "
|
||||
f"and the request would fail parsing 'settings' as an int."
|
||||
)
|
||||
|
||||
|
||||
def test_list_route_is_declared_before_the_catch_all(src):
|
||||
assert src.index('@router.get("")') < src.index('@router.get("/{log_id}")')
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Query construction
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_filters_are_bound_never_interpolated(src):
|
||||
"""User-supplied filters reach the WHERE clause; they must arrive as $n
|
||||
parameters."""
|
||||
body = _handler_body(src, '@router.get("")')
|
||||
# The only f-string interpolation allowed into SQL is the placeholder index
|
||||
# and the assembled clause list, never a raw value.
|
||||
for match in re.findall(r'add\("([^"]+)"', body):
|
||||
assert "{n}" in match, f"filter clause {match!r} does not use a bound placeholder"
|
||||
|
||||
|
||||
def test_list_endpoint_scopes_non_privileged_callers(src):
|
||||
"""A caller with only `requestlog.read` sees their own rows plus the fleet's.
|
||||
|
||||
Widened from own-rows-only during review, deliberately. The `operator` role
|
||||
is granted requestlog.read to "debug failing applies and ACME orders", but
|
||||
an apply fails on the NODE and the node reports it over its own API key, so
|
||||
the row carrying the diagnosis has `user_id IS NULL` — own-rows-only hid it
|
||||
from exactly the role the grant was written for.
|
||||
|
||||
What must NOT widen is the part this test was written to protect: another
|
||||
USER's captured bodies. Both halves are asserted below.
|
||||
"""
|
||||
# Comments explain what the clause deliberately does NOT do, so match on
|
||||
# code only — otherwise the prose describing the rule fails the test for it.
|
||||
body = "\n".join(
|
||||
line for line in _handler_body(src, '@router.get("")').splitlines()
|
||||
if not line.lstrip().startswith("#")
|
||||
)
|
||||
assert "if not can_manage:" in body
|
||||
assert "direction = 'inbound'" in body, (
|
||||
"outbound rows are not scoped at all, so a caller with only "
|
||||
"requestlog.read would see every CA and DNS call the backend ever made"
|
||||
)
|
||||
assert "user_id = $" in body, (
|
||||
"a caller with only requestlog.read can see every other user's captured "
|
||||
"request bodies"
|
||||
)
|
||||
assert "TARGET_INBOUND_AGENT" in body, (
|
||||
"agent rows are hidden from requestlog.read, which is the one thing the "
|
||||
"operator grant exists for"
|
||||
)
|
||||
assert "user_id IS NULL" not in body, (
|
||||
"scoping on NULL rather than on target would also expose anonymous "
|
||||
"traffic — failed logins and the usernames they carry, unauthenticated "
|
||||
"probes — to any requestlog.read holder"
|
||||
)
|
||||
|
||||
|
||||
def test_detail_endpoint_applies_the_same_scoping(src):
|
||||
body = _handler_body(src, '@router.get("/{log_id}")')
|
||||
assert "can_manage" in body
|
||||
assert "404" in body, (
|
||||
"the detail endpoint should 404 rather than 403 for a row the caller may not see, "
|
||||
"so it does not confirm which ids exist"
|
||||
)
|
||||
|
||||
|
||||
def test_list_response_omits_bodies(src):
|
||||
"""A 200-row page carrying two 8 KB JSONB blobs per row is a multi-megabyte
|
||||
response; bodies belong to the detail endpoint."""
|
||||
columns = src.split("_LIST_COLUMNS = ", 1)[1].split('"""', 2)[1]
|
||||
assert "request_body," not in columns
|
||||
assert "response_body," not in columns
|
||||
assert "request_body_bytes" in columns, "the size is still useful in the list"
|
||||
|
||||
|
||||
def test_count_is_bounded(src):
|
||||
body = _handler_body(src, '@router.get("")')
|
||||
assert "LIMIT {count_cap}" in body or "count_cap" in body, (
|
||||
"an unbounded COUNT(*) over request_logs is a sequential scan on every page change"
|
||||
)
|
||||
assert "total_is_estimate" in body
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Registration
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_router_is_registered_in_main():
|
||||
with open(_MAIN, encoding="utf-8") as f:
|
||||
main_src = f.read()
|
||||
|
||||
assert "from routers.request_logs import router as request_logs_router" in main_src
|
||||
assert "app.include_router(request_logs_router)" in main_src, (
|
||||
"the router is imported but never mounted, so every endpoint 404s"
|
||||
)
|
||||
@@ -0,0 +1,258 @@
|
||||
"""v1.11.0: the retention policy the operator sees is the policy that runs.
|
||||
|
||||
Two things drift silently and are caught here:
|
||||
|
||||
1. The defaults live in TWO places — the seed SQL in migrations.py and the
|
||||
dataclass in utils/request_log_settings.py. If they disagree, a fresh
|
||||
install and an upgraded install behave differently, which is the worst
|
||||
kind of bug to chase.
|
||||
2. Values in `system_settings` are operator-editable and arrive from asyncpg
|
||||
as raw JSON *strings*. Anything out of range, mistyped or hand-edited must
|
||||
be clamped rather than crash the writer loop.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from utils import request_log_settings # noqa: E402
|
||||
from utils.request_log_settings import ( # noqa: E402
|
||||
DEFAULT_CONFIG,
|
||||
DEFAULT_EXCLUDE_PATHS,
|
||||
RequestLogConfig,
|
||||
config_from_mapping,
|
||||
get_config,
|
||||
normalize_exclude_paths,
|
||||
refresh_config,
|
||||
set_config,
|
||||
)
|
||||
|
||||
_MIGRATIONS = os.path.join(
|
||||
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "database", "migrations.py"
|
||||
)
|
||||
|
||||
|
||||
def _seeded_defaults():
|
||||
"""Parse the ('requestlog.x', 'value', ...) tuples out of the seed SQL."""
|
||||
with open(_MIGRATIONS, encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
body = src.split("async def ensure_request_log_settings", 1)[1].split("\nasync def ", 1)[0]
|
||||
out = {}
|
||||
for key, raw in re.findall(r"\('requestlog\.(\w+)', '(.*?)', 'requestlog'", body):
|
||||
try:
|
||||
out[key] = json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
out[key] = raw
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Defaults must not drift between the seed and the code
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_seed_and_dataclass_defaults_agree():
|
||||
seeded = _seeded_defaults()
|
||||
assert seeded, "could not parse the requestlog seed rows out of migrations.py"
|
||||
|
||||
code = DEFAULT_CONFIG.as_dict()
|
||||
for key, seed_value in seeded.items():
|
||||
assert key in code, f"migrations seeds requestlog.{key} but RequestLogConfig has no such field"
|
||||
assert code[key] == seed_value, (
|
||||
f"requestlog.{key} default drifted: migrations.py seeds {seed_value!r} but "
|
||||
f"RequestLogConfig has {code[key]!r}. A fresh install and an upgraded install "
|
||||
f"would then behave differently."
|
||||
)
|
||||
|
||||
for key in code:
|
||||
assert key in seeded, (
|
||||
f"RequestLogConfig has {key!r} but migrations.py does not seed requestlog.{key} — "
|
||||
f"existing installs would silently fall back to the in-code default"
|
||||
)
|
||||
|
||||
|
||||
def test_log_viewer_is_excluded_by_default():
|
||||
assert "/api/request-logs" in DEFAULT_EXCLUDE_PATHS
|
||||
assert "/api/health" in DEFAULT_EXCLUDE_PATHS
|
||||
assert "/.well-known/acme-challenge" in DEFAULT_EXCLUDE_PATHS, (
|
||||
"the ACME challenge endpoint returns key_authorization — logging it would store "
|
||||
"the challenge secret"
|
||||
)
|
||||
assert "/api/agents/heartbeat" in DEFAULT_EXCLUDE_PATHS, (
|
||||
"the agent heartbeat is the highest-volume POST in the system; logging it by "
|
||||
"default would dominate the table"
|
||||
)
|
||||
|
||||
|
||||
def test_error_retention_defaults_longer_than_success_retention():
|
||||
assert DEFAULT_CONFIG.error_retention_days > DEFAULT_CONFIG.success_retention_days, (
|
||||
"the whole point of splitting the two is to keep failures around after the "
|
||||
"ordinary traffic has aged out"
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Coercion and clamping
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_raw_json_strings_from_asyncpg_are_parsed():
|
||||
cfg = config_from_mapping({
|
||||
"enabled": True,
|
||||
"max_body_bytes": 4096,
|
||||
"sample_rate": 0.25,
|
||||
"success_retention_days": 3,
|
||||
"exclude_paths": ["/api/health", "/metrics"],
|
||||
})
|
||||
assert cfg.enabled is True
|
||||
assert cfg.max_body_bytes == 4096
|
||||
assert cfg.sample_rate == 0.25
|
||||
assert cfg.success_retention_days == 3
|
||||
assert cfg.exclude_paths == ("/api/health", "/metrics")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("raw,expected", [
|
||||
("true", True), ("false", False), ("1", True), ("0", False),
|
||||
("on", True), ("off", False), (1, True), (0, False), (True, True),
|
||||
])
|
||||
def test_boolean_coercion_accepts_hand_written_values(raw, expected):
|
||||
cfg = config_from_mapping({"enabled": raw})
|
||||
assert cfg.enabled is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field,value,expected", [
|
||||
("max_body_bytes", 10_000_000, 262144),
|
||||
("max_body_bytes", -5, 0),
|
||||
("success_retention_days", 0, 1),
|
||||
("success_retention_days", 9999, 365),
|
||||
("error_retention_days", 0, 1),
|
||||
("max_rows", 10, 1000),
|
||||
("prune_interval_minutes", 1, 5),
|
||||
("prune_interval_minutes", 99999, 1440),
|
||||
])
|
||||
def test_out_of_range_values_are_clamped_not_rejected(field, value, expected):
|
||||
"""A bad value in the table must not disable logging or crash the writer —
|
||||
it is clamped to the nearest sane bound."""
|
||||
cfg = config_from_mapping({field: value})
|
||||
assert getattr(cfg, field) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value,expected", [(1.5, 1.0), (-0.2, 0.0), ("0.4", 0.4)])
|
||||
def test_sample_rate_is_clamped(value, expected):
|
||||
assert config_from_mapping({"sample_rate": value}).sample_rate == expected
|
||||
|
||||
|
||||
def test_garbage_values_fall_back_to_the_default():
|
||||
cfg = config_from_mapping({"max_body_bytes": "not-a-number", "sample_rate": "abc"})
|
||||
assert cfg.max_body_bytes == DEFAULT_CONFIG.max_body_bytes
|
||||
assert cfg.sample_rate == DEFAULT_CONFIG.sample_rate
|
||||
|
||||
|
||||
def test_exclude_paths_shape_is_enforced():
|
||||
out = normalize_exclude_paths(
|
||||
["/good", "no-leading-slash", "/" + "x" * 500, 42, "/also-good"],
|
||||
DEFAULT_EXCLUDE_PATHS,
|
||||
)
|
||||
assert out == ("/good", "/also-good")
|
||||
|
||||
|
||||
def test_exclude_paths_count_is_bounded():
|
||||
out = normalize_exclude_paths([f"/p{i}" for i in range(500)], DEFAULT_EXCLUDE_PATHS)
|
||||
assert len(out) <= 64
|
||||
|
||||
|
||||
def test_empty_exclude_paths_falls_back_rather_than_logging_everything():
|
||||
"""An empty list would re-enable logging of health checks and the docs, and
|
||||
flood the table — treat it as 'not configured'."""
|
||||
assert normalize_exclude_paths([], DEFAULT_EXCLUDE_PATHS) == DEFAULT_EXCLUDE_PATHS
|
||||
assert normalize_exclude_paths(None, DEFAULT_EXCLUDE_PATHS) == DEFAULT_EXCLUDE_PATHS
|
||||
|
||||
|
||||
def test_partial_mapping_keeps_the_other_defaults():
|
||||
cfg = config_from_mapping({"sample_rate": 0.5})
|
||||
assert cfg.sample_rate == 0.5
|
||||
assert cfg.success_retention_days == DEFAULT_CONFIG.success_retention_days
|
||||
assert cfg.enabled is DEFAULT_CONFIG.enabled
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# refresh_config
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_refresh_config_parses_the_raw_jsonb_strings_asyncpg_returns():
|
||||
conn = AsyncMock()
|
||||
conn.fetch = AsyncMock(return_value=[
|
||||
{"key": "requestlog.enabled", "value": "false"},
|
||||
{"key": "requestlog.max_body_bytes", "value": "4096"},
|
||||
{"key": "requestlog.sample_rate", "value": "0.5"},
|
||||
{"key": "requestlog.exclude_paths", "value": '["/api/health","/metrics"]'},
|
||||
])
|
||||
|
||||
with patch.object(request_log_settings, "get_database_connection", AsyncMock(return_value=conn)), \
|
||||
patch.object(request_log_settings, "close_database_connection", AsyncMock()):
|
||||
cfg = asyncio.run(refresh_config())
|
||||
|
||||
assert cfg.enabled is False
|
||||
assert cfg.max_body_bytes == 4096
|
||||
assert cfg.sample_rate == 0.5
|
||||
assert cfg.exclude_paths == ("/api/health", "/metrics")
|
||||
|
||||
set_config(DEFAULT_CONFIG)
|
||||
|
||||
|
||||
def test_refresh_config_keeps_the_previous_snapshot_on_db_failure():
|
||||
"""A transient pool error must not silently flip logging on or off."""
|
||||
known = RequestLogConfig(enabled=False, sample_rate=0.1)
|
||||
set_config(known)
|
||||
|
||||
with patch.object(request_log_settings, "get_database_connection",
|
||||
AsyncMock(side_effect=RuntimeError("pool exhausted"))), \
|
||||
patch.object(request_log_settings, "close_database_connection", AsyncMock()):
|
||||
cfg = asyncio.run(refresh_config())
|
||||
|
||||
assert cfg.enabled is False
|
||||
assert cfg.sample_rate == 0.1
|
||||
set_config(DEFAULT_CONFIG)
|
||||
|
||||
|
||||
def test_get_config_is_synchronous_and_needs_no_database():
|
||||
"""The middleware calls this on every request; it must never await."""
|
||||
assert not asyncio.iscoroutinefunction(get_config)
|
||||
assert isinstance(get_config(), RequestLogConfig)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The Pydantic model the API exposes
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_api_model_defaults_match_the_dataclass():
|
||||
from routers.request_logs import RequestLogSettings
|
||||
|
||||
model = RequestLogSettings().model_dump()
|
||||
code = DEFAULT_CONFIG.as_dict()
|
||||
for key, value in code.items():
|
||||
assert model[key] == value, f"API model default for {key} disagrees with RequestLogConfig"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("payload", [
|
||||
{"max_body_bytes": 999999},
|
||||
{"success_retention_days": 0},
|
||||
{"error_retention_days": 400},
|
||||
{"sample_rate": 1.5},
|
||||
{"max_rows": 10},
|
||||
{"prune_interval_minutes": 1},
|
||||
{"exclude_paths": ["no-slash"]},
|
||||
{"exclude_paths": ["/" + "x" * 300]},
|
||||
])
|
||||
def test_api_model_rejects_out_of_range_input(payload):
|
||||
from pydantic import ValidationError
|
||||
|
||||
from routers.request_logs import RequestLogSettings
|
||||
|
||||
with pytest.raises(ValidationError):
|
||||
RequestLogSettings(**payload)
|
||||
@@ -0,0 +1,276 @@
|
||||
"""v1.11.0: the batching writer must never slow down or break a request.
|
||||
|
||||
One row per API call is the highest write volume in the system and the asyncpg
|
||||
pool (min=10/max=50) is shared with every handler and four background loops. So
|
||||
the hot path enqueues and returns; a single writer task batches and inserts.
|
||||
The properties pinned here:
|
||||
|
||||
* `offer()` never blocks and never raises — a full queue drops and counts;
|
||||
* the parameter list stays aligned with the INSERT placeholders (a column
|
||||
added to one and not the other would fail every write at runtime, in
|
||||
production, with the migration already applied);
|
||||
* a failed batch is dropped with a warning rather than killing the loop.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from dataclasses import replace # noqa: E402
|
||||
|
||||
from utils import request_log_settings # noqa: E402
|
||||
from utils import request_log_sink as sink_module # noqa: E402
|
||||
from utils.request_log_sink import ( # noqa: E402
|
||||
RequestLogRow,
|
||||
RequestLogSink,
|
||||
_INSERT_SQL,
|
||||
)
|
||||
from utils.request_log_settings import DEFAULT_CONFIG # noqa: E402
|
||||
|
||||
|
||||
def _row(**overrides):
|
||||
base = dict(
|
||||
request_id="abc123",
|
||||
direction="inbound",
|
||||
method="POST",
|
||||
url="/api/backends",
|
||||
path="/api/backends",
|
||||
status_code=200,
|
||||
duration_ms=12,
|
||||
created_at=datetime(2026, 8, 11, 9, 0, tzinfo=timezone.utc),
|
||||
)
|
||||
base.update(overrides)
|
||||
return RequestLogRow(**base)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def defaults(monkeypatch):
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE", DEFAULT_CONFIG)
|
||||
monkeypatch.setattr(sink_module, "get_config", lambda: request_log_settings._CACHE)
|
||||
|
||||
|
||||
def _set(monkeypatch, **overrides):
|
||||
monkeypatch.setattr(request_log_settings, "_CACHE", replace(DEFAULT_CONFIG, **overrides))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# SQL / parameter alignment
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_insert_placeholders_match_the_column_list():
|
||||
columns = _INSERT_SQL.split("(", 1)[1].split(")", 1)[0]
|
||||
n_columns = len([c for c in columns.split(",") if c.strip()])
|
||||
n_placeholders = len(set(re.findall(r"\$(\d+)", _INSERT_SQL)))
|
||||
|
||||
assert n_columns == n_placeholders, (
|
||||
f"the INSERT names {n_columns} columns but binds {n_placeholders} placeholders — "
|
||||
f"every write would fail at runtime, on a database where the migration has "
|
||||
f"already succeeded"
|
||||
)
|
||||
|
||||
|
||||
def test_row_produces_exactly_as_many_params_as_the_insert_binds():
|
||||
n_placeholders = len(set(re.findall(r"\$(\d+)", _INSERT_SQL)))
|
||||
assert len(_row().to_params()) == n_placeholders, (
|
||||
"RequestLogRow.to_params() drifted from _INSERT_SQL"
|
||||
)
|
||||
|
||||
|
||||
def test_jsonb_params_are_serialized_strings_not_dicts():
|
||||
"""No JSONB codec is registered on this pool, so JSONB values travel as text
|
||||
and are cast in SQL — handing asyncpg a dict raises."""
|
||||
row = _row(
|
||||
query_params={"page": "2"},
|
||||
request_headers={"content-type": "application/json"},
|
||||
request_body_value={"name": "web"},
|
||||
)
|
||||
params = row.to_params()
|
||||
|
||||
for value in params:
|
||||
assert not isinstance(value, (dict, list)), (
|
||||
f"{value!r} was passed as a Python container; asyncpg cannot bind it to a "
|
||||
f"jsonb parameter"
|
||||
)
|
||||
|
||||
assert json.loads(params[6]) == {"page": "2"}
|
||||
|
||||
|
||||
def test_client_ip_is_never_a_placeholder_string():
|
||||
"""client_ip is an INET column: 'unknown' or a comma-joined X-Forwarded-For
|
||||
raises on INSERT."""
|
||||
params = _row(client_ip=None).to_params()
|
||||
assert params[12] is None
|
||||
|
||||
|
||||
def test_status_class_is_zero_when_there_was_no_response():
|
||||
assert _row(status_code=None).status_class == 0
|
||||
assert _row(status_code=204).status_class == 2
|
||||
assert _row(status_code=503).status_class == 5
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# offer(): the hot path
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_offer_drops_and_counts_when_the_queue_is_full():
|
||||
sink = RequestLogSink(maxsize=3, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
for _ in range(10):
|
||||
sink.offer(_row())
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
assert sink.stats["queued"] == 3
|
||||
assert sink.stats["dropped"] == 7, (
|
||||
"a full queue must drop and count, never block the request or raise"
|
||||
)
|
||||
|
||||
|
||||
def test_offer_never_raises_on_a_broken_row():
|
||||
sink = RequestLogSink(maxsize=10, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
sink.offer(None) # not a RequestLogRow at all
|
||||
|
||||
asyncio.run(run()) # must not raise
|
||||
|
||||
|
||||
def test_offer_respects_the_kill_switch(monkeypatch):
|
||||
_set(monkeypatch, enabled=False)
|
||||
sink = RequestLogSink(maxsize=10, batch_size=10, flush_ms=10)
|
||||
|
||||
asyncio.run(_offer(sink, _row()))
|
||||
assert sink.stats["queued"] == 0
|
||||
|
||||
|
||||
def test_offer_respects_the_per_direction_switches(monkeypatch):
|
||||
_set(monkeypatch, capture_outbound=False)
|
||||
sink = RequestLogSink(maxsize=10, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
sink.offer(_row(direction="outbound", target="acme"))
|
||||
sink.offer(_row(direction="inbound"))
|
||||
|
||||
asyncio.run(run())
|
||||
assert sink.stats["queued"] == 1
|
||||
|
||||
|
||||
def test_sampling_never_drops_errors(monkeypatch):
|
||||
"""A sample rate of zero must still capture every failure — that is the whole
|
||||
point of sampling successes only."""
|
||||
_set(monkeypatch, sample_rate=0.0)
|
||||
sink = RequestLogSink(maxsize=100, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
for _ in range(20):
|
||||
sink.offer(_row(status_code=200))
|
||||
for _ in range(5):
|
||||
sink.offer(_row(status_code=500))
|
||||
for _ in range(5):
|
||||
sink.offer(_row(status_code=None))
|
||||
|
||||
asyncio.run(run())
|
||||
assert sink.stats["queued"] == 10, (
|
||||
"sampling removed error rows; only 1xx/2xx/3xx inbound traffic may be sampled out"
|
||||
)
|
||||
|
||||
|
||||
def test_sampling_does_not_touch_outbound_rows(monkeypatch):
|
||||
_set(monkeypatch, sample_rate=0.0)
|
||||
sink = RequestLogSink(maxsize=100, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
for _ in range(5):
|
||||
sink.offer(_row(direction="outbound", target="acme", status_code=200))
|
||||
|
||||
asyncio.run(run())
|
||||
assert sink.stats["queued"] == 5, (
|
||||
"outbound calls are low-volume and high-value; sampling them away hides which CA "
|
||||
"or DNS call was made"
|
||||
)
|
||||
|
||||
|
||||
def test_capture_bodies_off_strips_the_payload_before_queueing(monkeypatch):
|
||||
_set(monkeypatch, capture_bodies=False)
|
||||
sink = RequestLogSink(maxsize=10, batch_size=10, flush_ms=10)
|
||||
row = _row(request_body_raw=b'{"a":1}', request_body_bytes=7)
|
||||
|
||||
asyncio.run(_offer(sink, row))
|
||||
|
||||
assert row.request_body_raw is None
|
||||
assert row.request_body_bytes == 7, "the size must survive so growth is still measurable"
|
||||
|
||||
|
||||
async def _offer(sink, row):
|
||||
sink.offer(row)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# The writer
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
def test_a_batch_is_written_with_one_executemany():
|
||||
conn = AsyncMock()
|
||||
sink = RequestLogSink(maxsize=100, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
for _ in range(5):
|
||||
sink.offer(_row())
|
||||
with patch.object(sink_module, "get_database_connection", AsyncMock(return_value=conn)), \
|
||||
patch.object(sink_module, "close_database_connection", AsyncMock()):
|
||||
return await sink.flush(timeout=1.0)
|
||||
|
||||
written = asyncio.run(run())
|
||||
|
||||
assert written == 5
|
||||
assert conn.executemany.await_count == 1, (
|
||||
"rows were inserted one at a time; that is one pool acquire per API call and the "
|
||||
"pool has 50 connections"
|
||||
)
|
||||
sql, params = conn.executemany.await_args.args
|
||||
assert "INSERT INTO request_logs" in sql
|
||||
assert len(params) == 5
|
||||
|
||||
|
||||
def test_a_failed_batch_does_not_kill_the_writer():
|
||||
conn = AsyncMock()
|
||||
conn.executemany = AsyncMock(side_effect=RuntimeError("relation does not exist"))
|
||||
sink = RequestLogSink(maxsize=100, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
sink.offer(_row())
|
||||
with patch.object(sink_module, "get_database_connection", AsyncMock(return_value=conn)), \
|
||||
patch.object(sink_module, "close_database_connection", AsyncMock()):
|
||||
await sink.flush(timeout=1.0)
|
||||
|
||||
asyncio.run(run()) # must not raise
|
||||
assert sink.stats["failed_batches"] == 1
|
||||
|
||||
|
||||
def test_the_connection_is_released_even_when_the_write_fails():
|
||||
conn = AsyncMock()
|
||||
conn.executemany = AsyncMock(side_effect=RuntimeError("boom"))
|
||||
release = AsyncMock()
|
||||
sink = RequestLogSink(maxsize=100, batch_size=10, flush_ms=10)
|
||||
|
||||
async def run():
|
||||
sink.offer(_row())
|
||||
with patch.object(sink_module, "get_database_connection", AsyncMock(return_value=conn)), \
|
||||
patch.object(sink_module, "close_database_connection", release):
|
||||
await sink.flush(timeout=1.0)
|
||||
|
||||
asyncio.run(run())
|
||||
assert release.await_count == 1, "a failed batch leaked a pooled connection"
|
||||
|
||||
|
||||
def test_flush_on_an_empty_queue_is_a_noop():
|
||||
sink = RequestLogSink(maxsize=10, batch_size=10, flush_ms=10)
|
||||
assert asyncio.run(sink.flush(timeout=0.1)) == 0
|
||||
@@ -0,0 +1,141 @@
|
||||
"""
|
||||
v1.10.6 — literal API paths must never be declared after a parameterised one.
|
||||
|
||||
Found in production on the v1.10.4 VIP-adoption feature: `@router.get("/discoveries")`
|
||||
sat at the bottom of routers/vip.py, below `@router.get("/{vip_id}")`. FastAPI matches
|
||||
routes in DECLARATION order, so every `GET /api/vip/discoveries` was answered by the
|
||||
`/{vip_id}` handler, which declares `vip_id: int` and therefore rejected the request with
|
||||
422 before `list_vip_discoveries` ever ran.
|
||||
|
||||
Nothing about that failure was visible. The agents reported their discoveries correctly,
|
||||
the rows landed in `vip_discoveries`, and the HA/VIP page treats any non-OK response as
|
||||
"nothing to show" — so the adoption feature simply did not exist as far as the UI was
|
||||
concerned, with no error anywhere.
|
||||
|
||||
These tests are a STATIC source scan on purpose: no imports, no app construction, no DB.
|
||||
They therefore also cover routers that cannot be imported in a bare test environment, and
|
||||
they keep covering routes added in the future.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
ROUTERS_DIR = pathlib.Path(__file__).resolve().parents[1] / "routers"
|
||||
|
||||
# `@router.get("/x")`, `@some_router.post("/x", ...)` — the path is the first string arg.
|
||||
_DECORATOR = re.compile(r'^@(?:\w+)\.(get|post|put|delete|patch)\(\s*[\'"]([^\'"]*)[\'"]')
|
||||
|
||||
|
||||
def _routes(source: str):
|
||||
"""[(line_no, verb, path)] in declaration order."""
|
||||
out = []
|
||||
for line_no, line in enumerate(source.splitlines(), 1):
|
||||
match = _DECORATOR.match(line)
|
||||
if match:
|
||||
out.append((line_no, match.group(1), match.group(2)))
|
||||
return out
|
||||
|
||||
|
||||
def _shadows(earlier: str, later: str) -> bool:
|
||||
"""True if `earlier` (declared first) swallows the literal path `later`.
|
||||
|
||||
Only literal paths can be silently swallowed, and only by a path that has the same
|
||||
number of segments where every non-placeholder segment matches. The collection route
|
||||
("" or "/") is its own path and never collides.
|
||||
"""
|
||||
if not later.strip("/") or not earlier.strip("/"):
|
||||
return False
|
||||
if "{" in later:
|
||||
return False
|
||||
if "{" not in earlier:
|
||||
return False
|
||||
earlier_segments = earlier.strip("/").split("/")
|
||||
later_segments = later.strip("/").split("/")
|
||||
if len(earlier_segments) != len(later_segments):
|
||||
return False
|
||||
return all(
|
||||
e.startswith("{") or e == l
|
||||
for e, l in zip(earlier_segments, later_segments)
|
||||
)
|
||||
|
||||
|
||||
def _shadowed_routes(path: pathlib.Path):
|
||||
routes = _routes(path.read_text())
|
||||
found = []
|
||||
for index, (line_no, verb, route_path) in enumerate(routes):
|
||||
for prior_line, prior_verb, prior_path in routes[:index]:
|
||||
if prior_verb == verb and _shadows(prior_path, route_path):
|
||||
found.append(
|
||||
f"{path.name}:{line_no} {verb.upper()} {route_path} is swallowed by "
|
||||
f"{prior_path} declared at line {prior_line}"
|
||||
)
|
||||
return found
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 1. The specific regression: /api/vip/discoveries must outrank /{vip_id}
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_vip_discoveries_declared_before_vip_id():
|
||||
routes = _routes((ROUTERS_DIR / "vip.py").read_text())
|
||||
get_paths = [path for _line, verb, path in routes if verb == "get"]
|
||||
|
||||
assert "/discoveries" in get_paths, "the discoveries endpoint disappeared"
|
||||
assert "/{vip_id}" in get_paths, "the get-one endpoint disappeared"
|
||||
assert get_paths.index("/discoveries") < get_paths.index("/{vip_id}"), (
|
||||
"GET /discoveries is declared after GET /{vip_id}; FastAPI will route "
|
||||
"/api/vip/discoveries into get_vip and answer 422, silently emptying the "
|
||||
"adoption panel"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 2. The general guard: no literal path anywhere is shadowed
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_no_literal_route_is_shadowed_in_any_router():
|
||||
problems = []
|
||||
for router_file in sorted(ROUTERS_DIR.glob("*.py")):
|
||||
problems.extend(_shadowed_routes(router_file))
|
||||
|
||||
assert not problems, (
|
||||
"literal route(s) declared after a parameterised route that swallows them:\n "
|
||||
+ "\n ".join(problems)
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 3. The detector itself must actually detect (guards against a vacuous pass)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"earlier,later,expected",
|
||||
[
|
||||
("/{vip_id}", "/discoveries", True), # the v1.10.4 bug
|
||||
("/{vip_id}", "/{other}", False), # two placeholders never shadow
|
||||
("/{vip_id}", "", False), # collection route is its own path
|
||||
("/{vip_id}/apply", "/adopt", False), # different segment counts
|
||||
("/{vip_id}/apply", "/adopt/now", False), # literal mismatch in segment 2
|
||||
("/{vip_id}/{action}", "/adopt/now", True), # both segments placeheld
|
||||
("/vips", "/discoveries", False), # literal never shadows a literal
|
||||
],
|
||||
)
|
||||
def test_shadow_detector_semantics(earlier, later, expected):
|
||||
assert _shadows(earlier, later) is expected
|
||||
|
||||
|
||||
def test_detector_flags_the_original_declaration_order():
|
||||
"""A synthetic file in the pre-fix order must be reported, so a future refactor that
|
||||
breaks the detector cannot make the guard above pass vacuously."""
|
||||
source = (
|
||||
'@router.get("")\n'
|
||||
"async def list_vips(): ...\n"
|
||||
'@router.get("/{vip_id}")\n'
|
||||
"async def get_vip(vip_id: int): ...\n"
|
||||
'@router.get("/discoveries")\n'
|
||||
"async def list_vip_discoveries(): ...\n"
|
||||
)
|
||||
routes = _routes(source)
|
||||
assert [verb for _l, verb, _p in routes] == ["get", "get", "get"]
|
||||
assert _shadows(routes[1][2], routes[2][2]) is True
|
||||
@@ -0,0 +1,641 @@
|
||||
"""
|
||||
v1.10.8 — the four adoption defects found while testing v1.10.4 on a live HA pair.
|
||||
|
||||
B1 The Apply Management "View Change" diff did not recognise the `adopt` action, so the
|
||||
version fell through to the generic HAProxy diff and rendered the cluster's whole
|
||||
haproxy.cfg as removed.
|
||||
|
||||
B2 `vip_discoveries.adopted_vip_id` is write-once and nothing clears it, while a VIP is only
|
||||
ever SOFT-deleted — so `ON DELETE SET NULL` never fires. Rejecting an adoption therefore
|
||||
hid the node from the panel permanently: the VIP was gone from the VIP list too, and the
|
||||
agent does not re-report a file whose hash has not changed. Adoptability is now derived
|
||||
from whether the linked VIP is still active.
|
||||
|
||||
B3 Adoption took only the node that was clicked. On a two-node pair that meant: adopting the
|
||||
BACKUP produced a VIP whose apply fails ("exactly one member must be MASTER"), adopting the
|
||||
MASTER left the peer unmanaged, and adopting the peer afterwards hit the VRID-collision
|
||||
guard with 409. The pair could never be completed from the panel.
|
||||
|
||||
B4 Worst of the four. `render_keepalived_conf` emits the unicast block only when it has peer
|
||||
addresses, so a single-member adoption of a UNICAST instance silently dropped it and
|
||||
keepalived fell back to multicast on that node while its peer stayed unicast — they stop
|
||||
seeing each other and BOTH claim the VIP.
|
||||
|
||||
B3 and B4 share one root and one fix: adoption now resolves the whole VRRP instance, keyed on
|
||||
(virtual_router_id, virtual address) exactly as keepalived groups nodes.
|
||||
|
||||
These are source-level and unit tests: the adoption endpoint needs a live database, so the
|
||||
behaviour that can be exercised without one is pinned here, and the SQL/flow invariants are
|
||||
pinned by reading the module.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
BACKEND = pathlib.Path(__file__).resolve().parents[1]
|
||||
VIP_ROUTER = (BACKEND / "routers" / "vip.py").read_text()
|
||||
CLUSTER_ROUTER = (BACKEND / "routers" / "cluster.py").read_text()
|
||||
RENDERER = (BACKEND / "services" / "keepalived_config.py").read_text()
|
||||
|
||||
|
||||
def _adopt_body() -> str:
|
||||
start = VIP_ROUTER.index("async def adopt_vip")
|
||||
return VIP_ROUTER[start:]
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B1 — the diff must recognise `adopt`
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_view_change_diff_recognises_the_adopt_action():
|
||||
match = re.search(r"vip_match = re\.search\(r'vip-\(\\d\+\)-\(([^)]+)\)'", CLUSTER_ROUTER)
|
||||
assert match, "the vip version regex moved; re-point this test"
|
||||
actions = set(match.group(1).split("|"))
|
||||
assert actions == {"create", "update", "delete", "adopt"}, (
|
||||
f"the View Change diff recognises {sorted(actions)}. An action missing here does not "
|
||||
f"degrade gracefully: the version falls through to the generic HAProxy diff and shows "
|
||||
f"the cluster's whole haproxy.cfg as removed."
|
||||
)
|
||||
|
||||
|
||||
def test_every_staged_vip_action_is_covered_by_the_diff_regex():
|
||||
"""Whatever _stage_vip_version can be called with must be in that alternation."""
|
||||
staged = set(re.findall(r'_stage_vip_version\(conn, vip_id, "(\w+)"', VIP_ROUTER))
|
||||
match = re.search(r"vip_match = re\.search\(r'vip-\(\\d\+\)-\(([^)]+)\)'", CLUSTER_ROUTER)
|
||||
recognised = set(match.group(1).split("|"))
|
||||
assert staged, "no _stage_vip_version call sites found; re-point this test"
|
||||
assert staged <= recognised, (
|
||||
f"staged action(s) {sorted(staged - recognised)} are not recognised by the View Change "
|
||||
f"diff regex {sorted(recognised)}"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B2 — adoptability follows the VIP's liveness, not the bare link
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_discovery_api_exposes_whether_the_adopted_vip_still_stands():
|
||||
assert "adopted_vip_active" in VIP_ROUTER, (
|
||||
"the discovery payload no longer reports whether the adopted VIP is still active; the "
|
||||
"UI would go back to hiding a rejected adoption forever"
|
||||
)
|
||||
assert re.search(r"LEFT JOIN vip_instances av ON av\.id = d\.adopted_vip_id", VIP_ROUTER), (
|
||||
"the discovery query must join the linked VIP to report its is_active"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_only_while_the_previous_adoption_still_stands():
|
||||
body = _adopt_body()
|
||||
assert re.search(r'if disc\["adopted_vip_id"\] and disc\["adopted_vip_active"\]', body), (
|
||||
"adopt must refuse only when the linked VIP is still ACTIVE — refusing on the bare link "
|
||||
"makes a rejected adoption impossible to retry, because nothing ever clears the column"
|
||||
)
|
||||
|
||||
|
||||
def test_nothing_clears_adopted_vip_id_so_the_derivation_is_load_bearing():
|
||||
"""If a future change starts clearing the column, this test should be revisited rather than
|
||||
silently left in place — the derived flag is what makes reject recoverable today."""
|
||||
writes = re.findall(r"UPDATE vip_discoveries SET adopted_vip_id = (\S+)", VIP_ROUTER)
|
||||
assert writes, "no adopted_vip_id write found; re-point this test"
|
||||
assert all(w != "NULL" for w in writes), (
|
||||
"adopted_vip_id is now cleared somewhere — re-check that the adopted_vip_active "
|
||||
"derivation and this test still describe reality"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B3 — the whole VRRP instance is adopted, not one node
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_participants_are_resolved_by_vrid_and_address():
|
||||
assert "async def _collect_instance_participants" in VIP_ROUTER
|
||||
start = VIP_ROUTER.index("async def _collect_instance_participants")
|
||||
end = VIP_ROUTER.index("@router.post(\"/adopt\")", start)
|
||||
body = VIP_ROUTER[start:end]
|
||||
assert 'v.get("virtual_router_id") == vrid' in body and 'v.get("virtual_ip") == virtual_ip' in body, (
|
||||
"instance identity must be (VRID, address) — the same key keepalived uses to decide two "
|
||||
"nodes are one VRRP group"
|
||||
)
|
||||
assert 'if r["parse_error"]' in body, "a node whose config failed to parse must not become a member"
|
||||
assert 'r["adopted_vip_id"] and r["adopted_vip_active"]' in body, (
|
||||
"a node already held by a STANDING VIP must not be pulled into a second one"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_inserts_one_member_per_participant():
|
||||
body = _adopt_body()
|
||||
insert = body.index("INSERT INTO vip_members")
|
||||
preceding = body[:insert]
|
||||
assert "for p in participants:" in preceding, (
|
||||
"members must be inserted in a loop over the resolved participants; a single insert is "
|
||||
"the half-adoption bug"
|
||||
)
|
||||
assert 'p["config_hash"]' in body[insert:insert + 800], (
|
||||
"each member must carry ITS OWN takeover hash — the one-shot takeover guard is per node"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_exactly_one_master_across_the_instance():
|
||||
body = _adopt_body()
|
||||
assert 'roles.count("MASTER") != 1' in body, (
|
||||
"adoption must reject an instance that does not have exactly one MASTER, instead of "
|
||||
"letting apply fail later with 'exactly one member must be MASTER'"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_enforces_one_active_vip_per_agent():
|
||||
body = _adopt_body()
|
||||
assert "already a member of VIP" in body, (
|
||||
"adoption must enforce the one-active-VIP-per-agent rule that create/update enforce via "
|
||||
"_validate_members_against_pool; a second membership never converges"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# B4 — a unicast instance can never be half-adopted
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_renderer_only_emits_unicast_when_it_has_peers():
|
||||
"""The property that makes B4 dangerous. Pinned so the guard below keeps its reason."""
|
||||
assert re.search(r"if use_unicast and peer_ips:", RENDERER), (
|
||||
"the renderer no longer gates the unicast block on having peers; re-derive whether the "
|
||||
"adoption guard is still needed"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_a_unicast_peer_that_is_not_being_adopted():
|
||||
body = _adopt_body()
|
||||
assert "declared_peers" in body and "member_ips" in body, (
|
||||
"adoption must verify every declared unicast peer is among the nodes being adopted"
|
||||
)
|
||||
assert "fall back to multicast" in body, (
|
||||
"the refusal must explain the consequence — silently dropping a peer puts both nodes in "
|
||||
"MASTER state on the same address"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_refuses_to_strand_any_node_that_references_the_address():
|
||||
"""The half-adoption hole that instance resolution alone does not close.
|
||||
|
||||
Participant resolution can only match a node it can READ, that is ENABLED and that is in the
|
||||
SAME pool. Each of those is a door a real member leaves through silently, and the nodes that
|
||||
remain get rewritten while it keeps serving the same address unmanaged. Seen for real: one
|
||||
node of a pair had a missing closing brace, so it parsed to nothing while its partner parsed
|
||||
cleanly. Rather than guard each door, the endpoint asks whether ANY reported config mentions
|
||||
this address and is not among the nodes being adopted.
|
||||
"""
|
||||
body = _adopt_body()
|
||||
assert "raw_config_masked LIKE" in body, (
|
||||
"the check must be scoped to configs that reference THIS virtual address, so an unrelated "
|
||||
"file elsewhere in the fleet does not block every adoption"
|
||||
)
|
||||
assert "NOT (a.id = ANY($2::int[]))" in body, (
|
||||
"the guard must catch every node that is NOT a participant, not just the unparseable "
|
||||
"ones — a disabled agent and a peer in another pool are stranded exactly the same way"
|
||||
)
|
||||
assert "COALESCE(av.is_active, FALSE) = FALSE" in body and "d.is_managed = FALSE" in body, (
|
||||
"a node already under management is not stranded and must not block adoption"
|
||||
)
|
||||
for reason in ("could not be parsed", "agent is disabled", "different agent pool"):
|
||||
assert reason in body, f"the refusal must be able to explain '{reason}'"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field,label", [
|
||||
("prefix_length", "prefix length"),
|
||||
("use_unicast", "unicast/multicast mode"),
|
||||
("track_haproxy", "HAProxy tracking"),
|
||||
])
|
||||
def test_adopt_requires_agreement_on_shared_vip_fields(field, label):
|
||||
"""These live on the VIP row and are re-rendered onto EVERY member, so taking them from the
|
||||
node that happened to be clicked imposes its settings on the others. prefix_length is the
|
||||
sharpest: the design refuses to guess a netmask for a live VIP, and copying one node's
|
||||
netmask onto another is that same change by another name."""
|
||||
body = _adopt_body()
|
||||
assert f'("{field}", "{label}")' in body, (
|
||||
f"{field} is written to the VIP row from one node's report; adoption must refuse when the "
|
||||
f"nodes disagree about it"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_one_shared_vrrp_password():
|
||||
body = _adopt_body()
|
||||
assert "decrypt_vrrp_secret(enc)" in body, (
|
||||
"Fernet is non-deterministic, so the per-node tokens cannot be compared as ciphertext — "
|
||||
"they must be decrypted and compared as plaintext"
|
||||
)
|
||||
assert "do not share one VRRP password" in body, (
|
||||
"adoption stores ONE secret and renders it onto every member, so a mismatch must be "
|
||||
"refused rather than silently normalised"
|
||||
)
|
||||
assert "cannot be decrypted" in body, (
|
||||
"a token we cannot decrypt must be an error, not silently treated as equal to another"
|
||||
)
|
||||
|
||||
|
||||
def test_adopt_requires_reported_ips_before_trusting_the_peer_check():
|
||||
body = _adopt_body()
|
||||
assert "have not reported an IP address yet" in body, (
|
||||
"the unicast peer check compares against member IPs, so a member without a reported IP "
|
||||
"must block the check rather than silently pass it"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# The takeover authorisation is genuinely one-shot
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_takeover_authorisation_is_retired_once_the_node_acks_our_config():
|
||||
"""`takeover_expected_hash` is the permission to overwrite a keepalived.conf that does NOT
|
||||
carry our ownership marker. It was written at adoption and never cleared, so the release
|
||||
notes' "authorises exactly ONE takeover" was only true as "for exactly that file content,
|
||||
indefinitely" — restoring the pre-adoption file would have been silently overwritten again
|
||||
with no fresh human approval."""
|
||||
agent_router = (BACKEND / "routers" / "agent.py").read_text()
|
||||
assert "takeover_expected_hash = CASE WHEN applied_config_hash IS NOT NULL" in agent_router, (
|
||||
"the status ack must retire the takeover authorisation once the node confirms our config"
|
||||
)
|
||||
assert "ELSE takeover_expected_hash END" in agent_router, (
|
||||
"a non-matching ack must LEAVE the authorisation in place — dropping it on a partial or "
|
||||
"failed deploy would leave the VIP unable to converge"
|
||||
)
|
||||
|
||||
|
||||
def test_validation_judges_the_output_not_the_exit_code():
|
||||
"""keepalived's config-test exit code cannot separate fatal from benign.
|
||||
|
||||
Measured on keepalived 2.2.8 rather than assumed:
|
||||
|
||||
clean config ..................... 0
|
||||
auth_pass longer than 8 chars .... 5 "Truncating auth_pass to 8 characters"
|
||||
missing '}' ...................... 5 "There are 1 missing '}'s"
|
||||
unknown keyword .................. 5 "Unknown keyword '...'"
|
||||
script without script_security ... 6 "SECURITY VIOLATION ..."
|
||||
|
||||
So 5 covers both a harmless truncation and a broken file. Treating any non-zero exit as
|
||||
invalid rejected VALID configs: a VRRP password over 8 characters is enough, and keepalived
|
||||
truncates it to 8 anyway, exactly as it does for the file the operator already runs. Found
|
||||
on the first live adoption, where the node's OWN running config also exited non-zero.
|
||||
|
||||
The gate must therefore judge the output, and it must fail CLOSED: anything not on the
|
||||
benign list still counts as fatal.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count("grep -v 'Truncating auth_pass to 8 characters'") == 2, (
|
||||
"both daemon copies must drop the known-benign truncation warning before judging"
|
||||
)
|
||||
assert script.count('if [[ -n "$kp_fatal" ]]; then') == 2, (
|
||||
"the decision must be made on what REMAINS after the benign lines are dropped"
|
||||
)
|
||||
# Fail-closed: the benign list is an allowlist, never a denylist of fatal messages. The
|
||||
# measured table above is quoted in a comment inside the script, so assert on what is used
|
||||
# as a grep PATTERN rather than on the text appearing anywhere.
|
||||
patterns = re.findall(r"grep -v '([^']*)'", script)
|
||||
assert set(patterns) == {"Truncating auth_pass to 8 characters", "^[[:space:]]*$"}, (
|
||||
f"the validation filter greps for {sorted(set(patterns))}. It must drop only messages "
|
||||
f"known to be benign; matching on fatal messages instead would let an unrecognised "
|
||||
f"error through."
|
||||
)
|
||||
|
||||
|
||||
def test_validation_fails_closed_when_keepalived_says_nothing():
|
||||
"""A non-zero exit with no output must stay fatal.
|
||||
|
||||
Filtering the output introduces a way to reach the accept path with an EMPTY filter result,
|
||||
and some keepalived builds log to syslog rather than stderr — on such a host every config
|
||||
would then be accepted regardless of what is wrong with it. Verified against a stub that
|
||||
exits non-zero silently, and against one that emits only whitespace.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count('if [[ -z "${kp_out//[[:space:]]/}" ]]; then') == 2, (
|
||||
"both daemon copies must treat a non-zero exit with no readable output as fatal"
|
||||
)
|
||||
assert script.count('kp_fatal="keepalived -t exited non-zero without output"') == 2
|
||||
|
||||
|
||||
def test_validation_failure_reports_what_keepalived_said():
|
||||
"""The fail-safe protected the node correctly on a live adoption but logged only
|
||||
"config validation failed", with keepalived's own output sent to /dev/null. The operator
|
||||
had no way to act on it without reproducing the check by hand on the node."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert 'keepalived -t -f "$tmp_conf" >/dev/null 2>&1' not in script, (
|
||||
"keepalived's output must not be discarded — a fail-safe that cannot say why it fired "
|
||||
"is only half a safety feature"
|
||||
)
|
||||
assert script.count('kp_out=$(keepalived -t -f "$tmp_conf" 2>&1)') == 2, (
|
||||
"both daemon copies must capture the validation output"
|
||||
)
|
||||
# The text is interpolated into JSON by both `log` and _kp_report, so it must be sanitised.
|
||||
assert script.count(r"""tr -d '"\\'""") == 2, (
|
||||
"captured output must have quotes and backslashes stripped before it reaches the JSON "
|
||||
"log line and the status report"
|
||||
)
|
||||
assert script.count('keepalived -t failed: ${kp_err}') == 2, (
|
||||
"the reason must also travel to the server so the UI can show it"
|
||||
)
|
||||
|
||||
|
||||
def test_converged_node_keeps_acknowledging():
|
||||
"""The idempotent path must still report, or a lost ack is never recovered.
|
||||
|
||||
The status report is the server's ONLY evidence that a member converged, and it used to be
|
||||
sent solely on the write path. Once the rendered config was on disk the agent took the
|
||||
idempotency early return every cycle and never spoke again, so a single lost report - a
|
||||
backend restart, a 5xx, a network blip - left the VIP reading SYNCING forever with an empty
|
||||
"Last ack" while the node was demonstrably running the right config. Seen in the field after
|
||||
acks were dropped for an unrelated reason: the node was correct, the page was not, and
|
||||
nothing would ever reconcile them.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert script.count('_kp_report "enabled" "$vip_id" "$new_hash" "already converged"') == 2, (
|
||||
"both daemon copies must re-assert the deploy state on the idempotent path; without it "
|
||||
"the server can never recover a lost acknowledgement"
|
||||
)
|
||||
# The report has to come BEFORE the early return in both copies.
|
||||
for m in re.finditer(r'if \[\[ -n "\$cur_hash" && "\$cur_hash" == "\$would_hash" \]\]; then(.*?)fi',
|
||||
script, re.S):
|
||||
body = m.group(1)
|
||||
assert body.index("_kp_report") < body.index("return 0"), (
|
||||
"the acknowledgement must be sent before returning, or the early return skips it"
|
||||
)
|
||||
|
||||
|
||||
def test_status_ack_statements_bind_each_placeholder_once():
|
||||
"""Every `$n` in the keepalived-status UPDATEs must be used exactly once, and the count must
|
||||
match the arguments passed.
|
||||
|
||||
Reusing one placeholder for both the assignment (`last_deploy_hash=$n`, a VARCHAR column)
|
||||
and the comparison inside the takeover-retirement CASE made PostgreSQL deduce two types for
|
||||
it, and asyncpg rejected the whole statement with AmbiguousParameterError. The failure was
|
||||
not partial: no ack was written at all, so every VIP sat at SYNCING forever and teardown acks
|
||||
were lost too. Shipped in v1.10.12 and caught in the field.
|
||||
|
||||
The suite has no database, so this pins the shape that made it possible rather than the SQL
|
||||
behaviour: one placeholder, one binding site.
|
||||
"""
|
||||
src = (BACKEND / "routers" / "agent.py").read_text()
|
||||
start = src.index("async def agent_keepalived_status")
|
||||
seg = src[start:src.index('return {"status": "ok"}', start)]
|
||||
|
||||
retire = "".join(re.findall(r'"([^"]*)"',
|
||||
re.search(r"_retire_takeover = \((.*?)\)\n", seg, re.S).group(1)))
|
||||
calls = re.findall(r'await conn\.execute\(f"""(.*?)""",\s*(.*?)\)\n', seg, re.S)
|
||||
assert len(calls) == 2, f"expected the two ack UPDATEs, found {len(calls)}"
|
||||
|
||||
for sql, args in calls:
|
||||
placeholder = re.search(r'_retire_takeover\.format\(p="(\$\d+)"\)', sql).group(1)
|
||||
rendered = re.sub(r"\{_retire_takeover\.format\(p=\"\$\d+\"\)\}",
|
||||
retire.replace("{p}", placeholder), sql)
|
||||
used = re.findall(r"\$(\d+)", rendered)
|
||||
dupes = {n for n in used if used.count(n) > 1}
|
||||
assert not dupes, (
|
||||
f"placeholder(s) {sorted('$'+d for d in dupes)} are bound more than once. PostgreSQL "
|
||||
f"deduces a type per USE, so a placeholder that is both assigned to a column and "
|
||||
f"compared against one is ambiguous and the whole UPDATE is rejected."
|
||||
)
|
||||
n_args = len([a for a in args.split(",") if a.strip()])
|
||||
assert max(int(n) for n in used) == n_args, (
|
||||
f"the statement uses ${max(int(n) for n in used)} but {n_args} arguments are passed"
|
||||
)
|
||||
|
||||
|
||||
def test_takeover_still_requires_the_pinned_hash_to_match_on_disk():
|
||||
"""The guard that stops an edit between adoption and Apply from being overwritten."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
guard = '[[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]'
|
||||
assert script.count(guard) == 2, (
|
||||
f"the takeover guard must be present in BOTH daemon copies (found {script.count(guard)}); "
|
||||
f"a self-upgraded agent runs the in-script copy, a freshly installed one the heredoc"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# Backward compatibility
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("guard", [
|
||||
"version_name NOT LIKE 'vip-%'", # bulk apply/reject still skip VIP versions
|
||||
])
|
||||
def test_vip_versions_stay_excluded_from_the_haproxy_apply_flow(guard):
|
||||
assert guard in CLUSTER_ROUTER, (
|
||||
"vip-* versions must stay out of the HAProxy apply/reject sweep; they are owned by the "
|
||||
"VIP endpoints and are never served as haproxy.cfg"
|
||||
)
|
||||
|
||||
|
||||
def test_vip_version_transition_matches_any_action():
|
||||
"""_transition_vip_versions must key on the VIP id alone, or a new action's PENDING row
|
||||
would be stranded in Apply Management after apply/reject."""
|
||||
assert 'f"vip-{vip_id}-%"' in VIP_ROUTER, (
|
||||
"the PENDING -> APPLIED/REJECTED transition must match every action for the VIP"
|
||||
)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# v1.11.1 — a discovery report that was never accepted must be retried
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_discovery_is_cached_only_when_the_server_accepted_it():
|
||||
"""`curl` without -f exits 0 on 500/403/404, so the previous `if curl ...` recorded a
|
||||
REJECTED discovery as delivered. The cache then suppressed every later attempt, and since
|
||||
the file never changes on its own the node stayed out of the adoption panel permanently:
|
||||
the only cure was deleting the cache on the node by hand."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert "if curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -X POST" not in script, (
|
||||
"the discovery POST must not be judged by curl's exit code; it is 0 for 5xx as well"
|
||||
)
|
||||
assert script.count('if [[ "$disc_code" =~ ^2[0-9][0-9]$ ]]; then') == 2, (
|
||||
"both daemon copies must cache only on a 2xx"
|
||||
)
|
||||
|
||||
|
||||
def test_discovery_cache_defers_to_the_server():
|
||||
"""Recovery without touching the node. The server reports whether it actually holds a
|
||||
discovery for this agent; only an explicit `false` overrides the cache, so a backend older
|
||||
than v1.11.1 (which omits the field) keeps the previous behaviour instead of being flooded
|
||||
with re-posts."""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
agent_router = (BACKEND / "routers" / "agent.py").read_text()
|
||||
|
||||
assert agent_router.count('AS discovery_known') == 1, (
|
||||
"the keepalived-config query must report whether a discovery row exists"
|
||||
)
|
||||
assert agent_router.count('"discovery_known": bool(agent["discovery_known"])') == 5, (
|
||||
"every response path that knows the agent must carry the flag — the adoptable nodes are "
|
||||
"precisely the not_configured ones"
|
||||
)
|
||||
assert script.count('jq -r \'if has("discovery_known")') == 2
|
||||
assert script.count('[[ "$disc_known" != "false" ]] && return 0') == 2, (
|
||||
"only an explicit false may override the cache, or an older backend — which omits the "
|
||||
"field — would cause a re-post on every cycle"
|
||||
)
|
||||
assert script.count('conf chk disc_known=""') == 2, (
|
||||
"disc_known must be function-local; in the in-script daemon the enclosing scope is the "
|
||||
"poll loop, so a stale value would outlive the response it came from"
|
||||
)
|
||||
|
||||
|
||||
def test_keepalived_config_path_is_resolved_deterministically():
|
||||
"""A pool may hold more than one cluster, and the join multiplies the agent row. Without an
|
||||
ordering the fetch took an arbitrary cluster, so the keepalived.conf PATH handed to the agent
|
||||
was non-deterministic whenever two clusters in a pool disagreed on it: the agent would look at
|
||||
the wrong file, find nothing, and the node would never appear for adoption."""
|
||||
src = (BACKEND / "routers" / "agent.py").read_text()
|
||||
start = src.index("async def get_agent_keepalived_config")
|
||||
q_start = src.index('agent = await conn.fetchrow("""', start)
|
||||
query = src[q_start:src.index('""", agent_name)', q_start)]
|
||||
assert "LEFT JOIN haproxy_clusters" in query, "re-point this test; the join moved"
|
||||
assert "ORDER BY" in query and "LIMIT 1" in query, (
|
||||
"the cluster row must be picked deterministically, or the config path the agent is told "
|
||||
"to inspect can change between polls"
|
||||
)
|
||||
assert "hc.keepalived_config_path = '/etc/keepalived/keepalived.conf'" in query, (
|
||||
"the ordering must prefer a CUSTOMISED path over the shipped default. The column defaults "
|
||||
"to that path rather than NULL, so ordering by id alone could pick a default-valued row "
|
||||
"over one the operator deliberately set, turning 'undefined' into 'reliably wrong'."
|
||||
)
|
||||
|
||||
|
||||
def test_daemon_copies_agree_on_the_whole_keepalived_path():
|
||||
"""Everything the adoption flow depends on must behave identically on BOTH install routes:
|
||||
a freshly installed agent runs the heredoc body, a self-upgraded one runs the in-script
|
||||
daemon. Comments may differ; logic may not."""
|
||||
import difflib
|
||||
lines = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text().splitlines()
|
||||
term = next(i for i, l in enumerate(lines) if l.strip() == "AGENT_SCRIPT")
|
||||
|
||||
def strip_comment(s):
|
||||
out, q, esc = [], None, False
|
||||
for ch in s:
|
||||
if esc:
|
||||
out.append(ch); esc = False; continue
|
||||
if ch == "\\":
|
||||
out.append(ch); esc = True; continue
|
||||
if q:
|
||||
out.append(ch)
|
||||
if ch == q:
|
||||
q = None
|
||||
continue
|
||||
if ch in ('"', "'"):
|
||||
q = ch; out.append(ch); continue
|
||||
if ch == "#":
|
||||
break
|
||||
out.append(ch)
|
||||
return "".join(out).rstrip()
|
||||
|
||||
def funcs(block):
|
||||
found = {}
|
||||
for idx, l in enumerate(block):
|
||||
m = re.match(r"^(\s*)([a-zA-Z_][a-zA-Z0-9_]*)\(\)\s*\{\s*(#.*)?$", l)
|
||||
if not m:
|
||||
continue
|
||||
close = m.group(1) + "}"
|
||||
end = next((j for j in range(idx + 1, len(block)) if block[j].rstrip() == close), None)
|
||||
if end is None:
|
||||
continue
|
||||
found[m.group(2)] = [re.sub(r"\s+", " ", strip_comment(x).strip())
|
||||
for x in block[idx + 1:end] if strip_comment(x).strip()]
|
||||
return found
|
||||
|
||||
here, insc = funcs(lines[923:term]), funcs(lines[term + 1:])
|
||||
for name in ("_kp_discover", "_kp_report", "_kp_teardown",
|
||||
"fetch_and_deploy_keepalived_config", "get_keepalive_state"):
|
||||
assert name in here and name in insc, f"{name} is missing from one daemon copy"
|
||||
if here[name] != insc[name]:
|
||||
d = "\n".join(x for x in difflib.unified_diff(here[name], insc[name], lineterm="")
|
||||
if x[:1] in "+-" and x[:3] not in ("+++", "---"))
|
||||
raise AssertionError(
|
||||
f"{name}() differs between the daemon copies, so a self-upgraded agent would "
|
||||
f"behave differently from a freshly installed one:\n{d}"
|
||||
)
|
||||
|
||||
|
||||
def test_discovery_flag_distinguishes_false_from_absent():
|
||||
"""jq's `//` returns the alternative for **false** as well as null.
|
||||
|
||||
`.discovery_known // empty` therefore yields an empty string both when the backend omits the
|
||||
field (older release) and when it explicitly says `false` (no discovery on record) — the one
|
||||
case the recovery exists for. Written that way the fix is inert: the cache is never overridden
|
||||
and a stuck node stays hidden. Caught in review, before it shipped, by parsing a real response
|
||||
rather than passing the value in by hand.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
assert "'.discovery_known // empty'" not in script, (
|
||||
"jq's // treats false like null, so this cannot tell 'no record' from 'old backend'"
|
||||
)
|
||||
expected = ('disc_known=$(echo "$resp" | jq -r \'if has("discovery_known") '
|
||||
'then (.discovery_known|tostring) else "" end\' 2>/dev/null)')
|
||||
assert script.count(expected) == 2, (
|
||||
"both daemon copies must distinguish an explicit false from an absent field"
|
||||
)
|
||||
|
||||
|
||||
def test_discovery_backs_off_on_a_permanent_rejection():
|
||||
"""A 4xx means the payload itself is unacceptable, so re-posting the same bytes cannot help.
|
||||
|
||||
Retrying forever is not free here: 4xx and 5xx agent calls are never sampled out of the
|
||||
request log (see request_log_sink), so an unattended loop writes a row carrying the whole
|
||||
keepalived.conf every poll cycle, on every affected node — the exact "polling noise evicts
|
||||
the forensic record" failure the log's own defaults exist to prevent. The cache therefore
|
||||
records the rejection and stays quiet until the file changes; a 5xx or a transport failure
|
||||
is still retried, which is what the recovery depends on.
|
||||
"""
|
||||
script = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text()
|
||||
brake = ('elif [[ "$disc_code" == "400" || "$disc_code" == "413" '
|
||||
'|| "$disc_code" == "422" ]]; then')
|
||||
assert script.count(brake) == 2, (
|
||||
"the brake must be limited to the codes that mean 'these bytes are unacceptable'"
|
||||
)
|
||||
# 401 and 404 are 4xx but TRANSIENT here: a token rotation, or an agent row briefly absent
|
||||
# while it re-registers. Braking on them would silence discovery for every affected node
|
||||
# until its keepalived.conf changed, which for a hand-maintained file may be never — the
|
||||
# exact failure this release removes.
|
||||
for transient in ('"401"', '"404"'):
|
||||
assert transient not in brake, (
|
||||
f"{transient} must stay in the retry class; it does not mean the payload is bad"
|
||||
)
|
||||
assert script.count("""printf '%s rejected' "$cur_hash" > "$cache" 2>/dev/null""") == 2
|
||||
assert script.count('[[ "$cached_state" == "rejected" ]] && return 0') == 2, (
|
||||
"a recorded rejection must suppress the post even when the server reports no record, "
|
||||
"or the flag override turns into an unbounded retry loop"
|
||||
)
|
||||
# The rejection must be keyed to the CONTENT, so a fixed config is retried.
|
||||
assert script.count('cached_hash="${cached_line%% *}"') == 2, (
|
||||
"the rejection is stored against the hash; changing the file must clear the brake"
|
||||
)
|
||||
|
||||
|
||||
def test_config_import_reaches_a_freshly_installed_agent():
|
||||
"""`check_config_requests` uploads the node's live haproxy.cfg when the operator asks for it.
|
||||
|
||||
It was defined in the installer body and in the in-script daemon, but NOT in the heredoc a
|
||||
fresh install writes to /usr/local/bin/haproxy-agent. Its call site is guarded by
|
||||
`type check_config_requests`, so on a freshly installed agent the whole feature was a silent
|
||||
no-op: the operator requested a config from the node and nothing ever arrived, with no error.
|
||||
Agents that had self-upgraded at least once did have it, which is why it went unnoticed.
|
||||
"""
|
||||
lines = (BACKEND / "utils" / "agent_scripts" / "linux_install.sh").read_text().splitlines()
|
||||
term = next(i for i, l in enumerate(lines) if l.strip() == "AGENT_SCRIPT")
|
||||
pattern = re.compile(r"^\s*check_config_requests\(\)\s*\{")
|
||||
|
||||
in_heredoc = sum(1 for l in lines[923:term] if pattern.match(l))
|
||||
in_daemon = sum(1 for l in lines[term + 1:] if pattern.match(l))
|
||||
assert in_heredoc == 1, (
|
||||
"the heredoc a fresh install writes must define check_config_requests, or Config Import "
|
||||
"silently does nothing on any node that has never self-upgraded"
|
||||
)
|
||||
assert in_daemon == 1, "the self-upgrade daemon must keep its definition"
|
||||
|
||||
# And the two must be the same function, not two drifting implementations.
|
||||
def body(block):
|
||||
idx = next(i for i, l in enumerate(block) if pattern.match(l))
|
||||
out = []
|
||||
for l in block[idx:]:
|
||||
out.append(l.strip())
|
||||
if l.strip() == "}" and len(out) > 5:
|
||||
break
|
||||
return out
|
||||
|
||||
assert body(lines[923:term]) == body(lines[term + 1:]), (
|
||||
"the two copies of check_config_requests have drifted"
|
||||
)
|
||||
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
v1.10.6 — GET /api/vip/discoveries resolves to its own handler and honours cluster_id.
|
||||
|
||||
Two defects, one endpoint, found in that order on a live fleet:
|
||||
|
||||
1. The route was declared after `GET /{vip_id}`, so FastAPI matched it there and answered
|
||||
422 ("discoveries" is not an int) before the handler ran. The static declaration-order
|
||||
guard lives in test_router_path_shadowing.py; this file pins the observable behaviour,
|
||||
because a 422 is what the browser actually saw.
|
||||
|
||||
2. Once reachable, it returned every discovery in the fleet regardless of the cluster
|
||||
selected in the header, so a multi-cluster install saw one undifferentiated list. The
|
||||
endpoint now takes the same optional `cluster_id` the VIP list takes.
|
||||
|
||||
The auth tests here deliberately assert `!= 422`: the repo's generic endpoint-auth tests
|
||||
accept 401/403/422 together, which is precisely why defect 1 slipped through them.
|
||||
"""
|
||||
import pathlib
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
VIP_ROUTER = pathlib.Path(__file__).resolve().parents[1] / "routers" / "vip.py"
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 1. The route reaches its own handler (defect 1)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("path", [
|
||||
"/api/vip/discoveries",
|
||||
"/api/vip/discoveries?cluster_id=7",
|
||||
])
|
||||
def test_discoveries_route_is_not_captured_by_the_vip_id_route(client, path):
|
||||
res = client.get(path)
|
||||
assert res.status_code != 422, (
|
||||
f"GET {path} returned 422 — the request was routed into the get-one-VIP handler, "
|
||||
f"which parses the path segment as an int. Declaration order regressed. "
|
||||
f"Body: {res.text[:200]}"
|
||||
)
|
||||
assert res.status_code in (401, 403), (
|
||||
f"GET {path} without a token should be refused by the vip.read gate, got "
|
||||
f"{res.status_code}. Body: {res.text[:200]}"
|
||||
)
|
||||
|
||||
|
||||
def test_get_one_vip_still_parses_a_numeric_id(client):
|
||||
"""Moving /discoveries above /{vip_id} must not shadow the numeric route itself."""
|
||||
res = client.get("/api/vip/12")
|
||||
assert res.status_code in (401, 403), res.text[:200]
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------
|
||||
# 2. cluster_id is accepted and actually scopes the query (defect 2)
|
||||
# ----------------------------------------------------------------------------
|
||||
|
||||
def test_handler_accepts_cluster_id():
|
||||
from routers.vip import list_vip_discoveries
|
||||
import inspect
|
||||
|
||||
params = inspect.signature(list_vip_discoveries).parameters
|
||||
assert "cluster_id" in params, (
|
||||
"list_vip_discoveries no longer takes cluster_id; the HA/VIP page would show every "
|
||||
"cluster's nodes at once again"
|
||||
)
|
||||
assert params["cluster_id"].default is None, (
|
||||
"cluster_id must stay optional — omitting it returns the whole fleet, which is what "
|
||||
"a caller that predates the parameter expects"
|
||||
)
|
||||
|
||||
|
||||
def test_discovery_query_scopes_by_the_cluster_pool():
|
||||
"""The filter must resolve cluster -> pool the same way the VIP list does, and must be a
|
||||
no-op when the parameter is absent."""
|
||||
source = VIP_ROUTER.read_text()
|
||||
start = source.index("async def list_vip_discoveries")
|
||||
end = source.index("def _find_candidate", start)
|
||||
body = source[start:end]
|
||||
|
||||
assert "FROM vip_discoveries" in body, "the discovery query moved; re-point this test"
|
||||
assert re.search(r"a\.pool_id\s*=\s*\(\s*SELECT\s+pool_id\s+FROM\s+haproxy_clusters", body), (
|
||||
"the cluster filter must map cluster -> pool via haproxy_clusters, matching list_vips"
|
||||
)
|
||||
assert "IS NULL" in body, (
|
||||
"the filter must short-circuit when cluster_id is absent, so an unscoped call still "
|
||||
"returns the whole fleet"
|
||||
)
|
||||
@@ -0,0 +1,266 @@
|
||||
"""Validation and resolution for the ACME HTTP-01 challenge backend URL.
|
||||
|
||||
This URL tells HAProxy where to proxy ``/.well-known/acme-challenge/*``. It is
|
||||
rendered into ``backend _acme_challenge_backend`` as ``server _acme_mgmt host:port``
|
||||
and — this is the part that makes it unlike every other URL in the product —
|
||||
**resolved on the HAProxy node, not on the management host**. A value that works
|
||||
when pasted into the management server's own browser can be completely dead from
|
||||
the data plane.
|
||||
|
||||
Two entry points, deliberately asymmetric:
|
||||
|
||||
``validate_acme_backend_url``
|
||||
Called at the WRITE BOUNDARY (cluster PUT, settings PUT). Rejects values that
|
||||
cannot express a reachable target. Strict here is safe: it only ever affects a
|
||||
value an operator is typing right now, and the error text can teach.
|
||||
|
||||
``resolve_acme_backend_target``
|
||||
Called at RENDER TIME. Never raises, never rejects. Strictness here would be a
|
||||
catastrophe: the shipped defaults (``config.py`` ``http://localhost:8000``,
|
||||
``docker-compose.yml`` ``http://localhost:8080``) mean essentially every
|
||||
existing install resolves to loopback today, and refusing to render would make
|
||||
every ``acme_enabled`` cluster unappliable — including for urgent changes that
|
||||
have nothing to do with ACME. It reports problems instead of enforcing them.
|
||||
|
||||
Two conscious departures from ``utils/ssrf_guard.py``, whose policy is the exact
|
||||
opposite of what is needed here:
|
||||
|
||||
* **RFC1918 is allowed, and is usually the correct answer.** The guard exists to
|
||||
stop the server being tricked into dialling internal space. Here the operator is
|
||||
deliberately naming their own management host, which on a split deployment is
|
||||
private by definition.
|
||||
* **No DNS resolution.** Resolving from the management host would re-introduce the
|
||||
very wrong-vantage-point mistake this work exists to remove: what this box can
|
||||
resolve says nothing about what the HAProxy node can reach.
|
||||
"""
|
||||
import ipaddress
|
||||
import re
|
||||
from typing import List, NamedTuple, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
# `haproxy_clusters.acme_backend_url` / `system_settings.value` are VARCHAR(500).
|
||||
# Without this check asyncpg raises 22001 and the operator gets an opaque 500.
|
||||
MAX_URL_LENGTH = 500
|
||||
|
||||
ALLOWED_SCHEMES = ("http", "https")
|
||||
|
||||
# Port assumed when the URL omits one. NOT the scheme's default: the bundled
|
||||
# docker-compose publishes nginx on 8080 (`nginx/nginx.conf` listens 8080,
|
||||
# `docker-compose.yml` maps 8080:8080), so an operator who wrote a bare
|
||||
# `http://10.0.0.5` has a WORKING path today that resolves to :8080. Changing this
|
||||
# to 80 would break those installs silently — the first symptom would be the
|
||||
# unattended renewal loop failing months later. The value is kept and the omission
|
||||
# is surfaced as a warning instead.
|
||||
DEFAULT_HTTP_PORT = 8080
|
||||
DEFAULT_HTTPS_PORT = 443
|
||||
|
||||
# RFC 1123 host label set. Deliberately not a full IDN implementation: an operator
|
||||
# naming their management host in a config file pushed to HAProxy nodes should use
|
||||
# ASCII, and HAProxy itself would not accept anything else on a `server` line.
|
||||
_HOSTNAME_RE = re.compile(
|
||||
r"^(?=.{1,253}$)[A-Za-z0-9]([A-Za-z0-9-]{0,61}[A-Za-z0-9])?"
|
||||
r"(\.[A-Za-z0-9]([A-Za-z0-9-]{0,61}[A-Za-z0-9])?)*\.?$"
|
||||
)
|
||||
|
||||
_CONTROL_CHARS = frozenset("\t\n\r\v\f\x00")
|
||||
|
||||
|
||||
class AcmeBackendUrlError(ValueError):
|
||||
"""A value that cannot express a usable challenge backend target.
|
||||
|
||||
``code`` is stable and machine-readable so the UI can map it to help text;
|
||||
``args[0]`` is operator-facing prose.
|
||||
"""
|
||||
|
||||
def __init__(self, code: str, message: str):
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
|
||||
|
||||
class AcmeBackendTarget(NamedTuple):
|
||||
"""What the renderer should emit, plus everything worth telling the operator."""
|
||||
|
||||
host: str
|
||||
port: int
|
||||
ssl_flag: str
|
||||
#: Non-fatal observations. Rendered anyway; surfaced in logs and the panel.
|
||||
warnings: List[str]
|
||||
#: Set when the value could not be parsed at all and the caller must not emit
|
||||
#: a `server` line. None on success.
|
||||
error_code: Optional[str]
|
||||
error_message: Optional[str]
|
||||
|
||||
|
||||
def _classify_host(host: str) -> Optional[str]:
|
||||
"""Return a rejection code for hosts that cannot be a management address."""
|
||||
try:
|
||||
ip = ipaddress.ip_address(host)
|
||||
except ValueError:
|
||||
lowered = host.rstrip(".").lower()
|
||||
# `localhost` is loopback by name and is the single most likely wrong value
|
||||
# here — it is what both shipped defaults contain. Catching only the numeric
|
||||
# form would let the exact failure this module exists to prevent straight
|
||||
# through. RFC 6761 also reserves the whole `.localhost` tree.
|
||||
if lowered == "localhost" or lowered.endswith(".localhost"):
|
||||
return "loopback"
|
||||
return None if _HOSTNAME_RE.match(host) else "invalid_host"
|
||||
|
||||
if isinstance(ip, ipaddress.IPv6Address) and ip.ipv4_mapped is not None:
|
||||
ip = ip.ipv4_mapped
|
||||
if ip.is_loopback:
|
||||
return "loopback"
|
||||
if ip.is_unspecified:
|
||||
return "unspecified"
|
||||
# Includes 169.254.169.254, the cloud metadata endpoint.
|
||||
if ip.is_link_local:
|
||||
return "link_local"
|
||||
if ip.is_multicast:
|
||||
return "multicast"
|
||||
# NOTE: private (RFC1918) addresses fall through on purpose — see module docstring.
|
||||
return None
|
||||
|
||||
|
||||
_REJECTION_PROSE = {
|
||||
"too_long": f"URL must be at most {MAX_URL_LENGTH} characters.",
|
||||
"whitespace": (
|
||||
"URL must not contain spaces or line breaks. A trailing space survives parsing "
|
||||
"and would be written into haproxy.cfg as part of the address."
|
||||
),
|
||||
"no_scheme": (
|
||||
"URL must start with http:// or https://. Without a scheme the value cannot be "
|
||||
"parsed as an address and silently falls back to localhost, which on a HAProxy "
|
||||
"node means the node itself."
|
||||
),
|
||||
"bad_scheme": "URL scheme must be http or https.",
|
||||
"userinfo": "URL must not contain credentials.",
|
||||
"has_path": (
|
||||
"Enter only the scheme, host and port — no path, query or fragment. The "
|
||||
"challenge path is appended by HAProxy."
|
||||
),
|
||||
"bad_port": "Port must be a number between 1 and 65535.",
|
||||
"no_host": "URL must contain a host.",
|
||||
"invalid_host": "Host is not a valid IP address or hostname.",
|
||||
"loopback": (
|
||||
"Loopback addresses cannot work here. HAProxy resolves this address on the "
|
||||
"HAProxy node, so 127.0.0.1 means the node itself, not the management server. "
|
||||
"Use the management server's routable address."
|
||||
),
|
||||
"unspecified": (
|
||||
"0.0.0.0 is a listen address, not a destination. Use the management server's "
|
||||
"routable address."
|
||||
),
|
||||
"link_local": "Link-local addresses cannot be used as a management address.",
|
||||
"multicast": "Multicast addresses cannot be used as a management address.",
|
||||
}
|
||||
|
||||
|
||||
def validate_acme_backend_url(value: Optional[str]) -> Optional[str]:
|
||||
"""Validate an operator-supplied URL at the write boundary.
|
||||
|
||||
Returns the normalised value (stripped), or ``None`` for empty input, which
|
||||
legitimately means "inherit from the next level of the resolution chain".
|
||||
Raises :class:`AcmeBackendUrlError` otherwise.
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
if not isinstance(value, str):
|
||||
raise AcmeBackendUrlError("invalid_host", _REJECTION_PROSE["invalid_host"])
|
||||
|
||||
stripped = value.strip()
|
||||
if not stripped:
|
||||
return None
|
||||
|
||||
if len(stripped) > MAX_URL_LENGTH:
|
||||
raise AcmeBackendUrlError("too_long", _REJECTION_PROSE["too_long"])
|
||||
if any(c in _CONTROL_CHARS for c in stripped) or " " in stripped:
|
||||
raise AcmeBackendUrlError("whitespace", _REJECTION_PROSE["whitespace"])
|
||||
|
||||
parsed = urlparse(stripped)
|
||||
|
||||
if not parsed.scheme:
|
||||
raise AcmeBackendUrlError("no_scheme", _REJECTION_PROSE["no_scheme"])
|
||||
if parsed.scheme.lower() not in ALLOWED_SCHEMES:
|
||||
# `10.0.0.5:8080` parses as scheme='10.0.0.5' with no netloc, and
|
||||
# `localhost:8080` as scheme='localhost'. Both are the same operator mistake,
|
||||
# so point at the missing scheme rather than the nonsense one.
|
||||
if not parsed.netloc:
|
||||
raise AcmeBackendUrlError("no_scheme", _REJECTION_PROSE["no_scheme"])
|
||||
raise AcmeBackendUrlError("bad_scheme", _REJECTION_PROSE["bad_scheme"])
|
||||
|
||||
if parsed.username is not None or parsed.password is not None:
|
||||
raise AcmeBackendUrlError("userinfo", _REJECTION_PROSE["userinfo"])
|
||||
if parsed.path not in ("", "/") or parsed.query or parsed.fragment:
|
||||
raise AcmeBackendUrlError("has_path", _REJECTION_PROSE["has_path"])
|
||||
|
||||
try:
|
||||
port = parsed.port
|
||||
except ValueError:
|
||||
# urlparse defers port parsing to attribute access; an out-of-range or
|
||||
# non-numeric port raises here. Unguarded, this exception reaches the config
|
||||
# generator's blanket `except` and collapses the cluster's whole config.
|
||||
raise AcmeBackendUrlError("bad_port", _REJECTION_PROSE["bad_port"]) from None
|
||||
if port is not None and not (1 <= port <= 65535):
|
||||
raise AcmeBackendUrlError("bad_port", _REJECTION_PROSE["bad_port"])
|
||||
|
||||
host = parsed.hostname
|
||||
if not host:
|
||||
raise AcmeBackendUrlError("no_host", _REJECTION_PROSE["no_host"])
|
||||
|
||||
code = _classify_host(host)
|
||||
if code is not None:
|
||||
raise AcmeBackendUrlError(code, _REJECTION_PROSE[code])
|
||||
|
||||
return stripped
|
||||
|
||||
|
||||
def resolve_acme_backend_target(url: Optional[str]) -> AcmeBackendTarget:
|
||||
"""Resolve a stored URL into what the renderer emits. Never raises.
|
||||
|
||||
Values already in the database predate validation (and the shipped defaults are
|
||||
themselves loopback), so anything unparseable or discouraged is reported through
|
||||
``warnings`` / ``error_code`` rather than refused.
|
||||
"""
|
||||
warnings: List[str] = []
|
||||
raw = (url or "").strip()
|
||||
|
||||
if not raw:
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "empty", "No challenge backend URL configured."
|
||||
)
|
||||
|
||||
if any(c in _CONTROL_CHARS for c in raw) or " " in raw:
|
||||
# Must never reach haproxy.cfg: a newline here writes attacker- or
|
||||
# accident-chosen directives into a file pushed to every node.
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "whitespace", _REJECTION_PROSE["whitespace"]
|
||||
)
|
||||
|
||||
parsed = urlparse(raw)
|
||||
scheme = (parsed.scheme or "").lower()
|
||||
|
||||
try:
|
||||
port = parsed.port
|
||||
except ValueError:
|
||||
return AcmeBackendTarget("", 0, "", warnings, "bad_port", _REJECTION_PROSE["bad_port"])
|
||||
|
||||
host = parsed.hostname
|
||||
if not host or scheme not in ALLOWED_SCHEMES:
|
||||
return AcmeBackendTarget(
|
||||
"", 0, "", warnings, "no_scheme", _REJECTION_PROSE["no_scheme"]
|
||||
)
|
||||
|
||||
if port is None:
|
||||
port = DEFAULT_HTTPS_PORT if scheme == "https" else DEFAULT_HTTP_PORT
|
||||
warnings.append(
|
||||
f"No port given, assuming {port}. State the port explicitly — the assumed "
|
||||
f"value is the bundled reverse proxy's port, not the scheme's default."
|
||||
)
|
||||
|
||||
code = _classify_host(host)
|
||||
if code == "invalid_host":
|
||||
return AcmeBackendTarget("", 0, "", warnings, code, _REJECTION_PROSE[code])
|
||||
if code is not None:
|
||||
warnings.append(_REJECTION_PROSE[code])
|
||||
|
||||
ssl_flag = " ssl verify none" if scheme == "https" else ""
|
||||
return AcmeBackendTarget(host, port, ssl_flag, warnings, None, None)
|
||||
@@ -1691,7 +1691,7 @@ check_ssl_updates() {
|
||||
# (kept in sync with the live token in both daemon loops).
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local marker="# Managed by HAProxy OpenManager"
|
||||
local resp status http_code we_own="false" conf chk
|
||||
local resp status http_code we_own="false" conf chk disc_known=""
|
||||
|
||||
# Timeouts so a hung management server can never stall the daemon loop.
|
||||
resp=$(curl -k -s --connect-timeout 10 --max-time 30 -w '\n%{http_code}' -X GET \
|
||||
@@ -1710,6 +1710,95 @@ fetch_and_deploy_keepalived_config() {
|
||||
[[ -z "$conf" || "$conf" == "null" ]] && conf="/etc/keepalived/keepalived.conf"
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
# Whether the SERVER already holds a discovery for this node. Absent on backends older than
|
||||
# v1.11.1, in which case the discovery cache keeps its previous meaning.
|
||||
disc_known=$(echo "$resp" | jq -r 'if has("discovery_known") then (.discovery_known|tostring) else "" end' 2>/dev/null)
|
||||
|
||||
# v1.10.4 — VIP adoption discovery. Report a keepalived.conf we do NOT own so an existing
|
||||
# VIP can be adopted from the UI instead of retyped. STRICTLY READ-ONLY: this never writes
|
||||
# to the node. The heartbeat cannot carry this — it has the VIP address and a best-effort
|
||||
# MASTER/BACKUP, while rendering a node's config needs eleven fields.
|
||||
#
|
||||
# Rate limited by content: the hash of the last report is cached next to the config, so the
|
||||
# file (which may contain the VRRP password) is posted only when it actually changes, not on
|
||||
# every cycle. Once we own the file there is nothing to adopt, so the record is cleared once.
|
||||
_kp_discover() {
|
||||
local cache="$(dirname "$conf")/.hom_discovery_hash" cur_hash="" body content_json
|
||||
if [[ "$we_own" == "true" || ! -f "$conf" ]]; then
|
||||
# Nothing adoptable here. Clear a previous report exactly once.
|
||||
[[ -f "$cache" ]] || return 0
|
||||
# Clear ONLY when the server accepted it. curl's exit code is 0 for 5xx too, so
|
||||
# dropping the cache on a rejected clear lost the fact that this node is ours: the
|
||||
# stale discovery row would keep offering a MANAGED node for adoption, and with the
|
||||
# cache gone nothing would ever send the clear again.
|
||||
local clr_code
|
||||
clr_code=$(curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -w '%{http_code}' \
|
||||
-X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"config_path\":\"$conf\",\"exists\":false}" 2>/dev/null)
|
||||
[[ "$clr_code" =~ ^2[0-9][0-9]$ ]] || return 0
|
||||
rm -f "$cache"
|
||||
return 0
|
||||
fi
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
[[ -z "$cur_hash" ]] && return 0
|
||||
# The cache may only suppress a report while the SERVER agrees it already holds one.
|
||||
# v1.11.1: it suppressed unconditionally, so a report the server REJECTED was recorded
|
||||
# as delivered and the node stayed out of the adoption panel for good — the file never
|
||||
# changes, so nothing ever triggered another attempt and the only cure was deleting this
|
||||
# file on the node by hand. `discovery_known` is absent on older backends, and only an
|
||||
# explicit "false" overrides the cache, so an old server keeps the previous behaviour
|
||||
# rather than being flooded with re-posts.
|
||||
# Cache line is "<hash>" after a success, or "<hash> rejected" after a 4xx.
|
||||
local cached_line cached_hash cached_state=""
|
||||
cached_line=$(cat "$cache" 2>/dev/null)
|
||||
cached_hash="${cached_line%% *}"
|
||||
[[ "$cached_line" == *" "* ]] && cached_state="${cached_line#* }"
|
||||
if [[ -n "$cached_hash" && "$cached_hash" == "$cur_hash" ]]; then
|
||||
# A 4xx means the server refuses THESE BYTES. Re-posting them cannot succeed, and
|
||||
# every attempt is stored as a failed agent call (4xx/5xx are never sampled out), so
|
||||
# an unattended loop would write a row carrying the whole config every cycle. Stay
|
||||
# quiet until the file changes — the reason was logged when it was rejected.
|
||||
[[ "$cached_state" == "rejected" ]] && return 0
|
||||
# Otherwise the cache only holds while the SERVER agrees it has the report.
|
||||
[[ "$disc_known" != "false" ]] && return 0
|
||||
fi
|
||||
# jq -Rs makes the file a single JSON string with its newlines intact, so the content the
|
||||
# server hashes is byte-identical to what is on disk — the takeover authorisation is
|
||||
# pinned to that hash.
|
||||
content_json=$(jq -Rs . < "$conf" 2>/dev/null) || return 0
|
||||
body=$(jq -n --arg p "$conf" --argjson c "$content_json" \
|
||||
'{config_path:$p, exists:true, is_managed:false, config_content:$c}' 2>/dev/null) || return 0
|
||||
# Cache ONLY on a 2xx. curl without -f exits 0 on 500/403/404 too, so the previous
|
||||
# `if curl ...` recorded a REJECTED report as delivered — and since the cache suppresses
|
||||
# every later attempt until the file itself changes, one server-side error hid the node
|
||||
# from the adoption panel permanently. Same failure shape as the deploy acknowledgement
|
||||
# fixed in v1.10.14, on the discovery path.
|
||||
local disc_code
|
||||
disc_code=$(curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -w '%{http_code}' \
|
||||
-X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
--data-binary "$body" 2>/dev/null)
|
||||
if [[ "$disc_code" =~ ^2[0-9][0-9]$ ]]; then
|
||||
printf '%s' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "INFO" "KEEPALIVED: reported an unmanaged keepalived.conf for adoption"
|
||||
elif [[ "$disc_code" == "400" || "$disc_code" == "413" || "$disc_code" == "422" ]]; then
|
||||
# ONLY the codes that mean "these bytes are unacceptable" stop the retry: too large
|
||||
# to analyse, malformed, rejected by validation. Re-posting identical content cannot
|
||||
# change any of those answers.
|
||||
#
|
||||
# 401 and 404 are deliberately NOT here even though they are 4xx. Both are transient
|
||||
# in this system — a token rotation, or an agent row briefly absent while it
|
||||
# re-registers — and treating them as permanent would silence discovery for every
|
||||
# affected node until its keepalived.conf changed, which for a hand-maintained file
|
||||
# may be never. That is the exact failure this release set out to remove.
|
||||
printf '%s rejected' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "WARN" "KEEPALIVED: discovery report refused (HTTP $disc_code); not retrying until the config changes"
|
||||
else
|
||||
log "WARN" "KEEPALIVED: discovery report failed (HTTP ${disc_code:-none}); will retry next cycle"
|
||||
fi
|
||||
}
|
||||
_kp_discover
|
||||
|
||||
_kp_report() { # $1=state $2=vip_id(or empty) $3=hash $4=message
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
@@ -1774,10 +1863,29 @@ fetch_and_deploy_keepalived_config() {
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
# Ownership guard: never overwrite a keepalived.conf we don't own.
|
||||
#
|
||||
# v1.10.4 adoption is the ONE exception, and it does not weaken the guard: the server
|
||||
# authorises a single takeover of a specific file by pinning the md5 the operator adopted
|
||||
# from. We overwrite only when that hash still matches what is on disk, so a config edited
|
||||
# between adoption and Apply is still refused — the operator's later edit wins over a stale
|
||||
# adoption rather than being silently destroyed.
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
local allow_takeover expected_hash disk_hash
|
||||
allow_takeover=$(echo "$resp" | jq -r '.keepalived.allow_takeover // false' 2>/dev/null)
|
||||
expected_hash=$(echo "$resp" | jq -r '.keepalived.takeover_expected_hash // empty' 2>/dev/null)
|
||||
disk_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
if [[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]; then
|
||||
log "INFO" "KEEPALIVED: adopting $conf (one-shot takeover authorised; on-disk hash matches)"
|
||||
elif [[ "$allow_takeover" == "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: adoption authorised but $conf changed since it was adopted — refusing"
|
||||
_kp_report "externally_managed" "$vip_id" "$disk_hash" \
|
||||
"config changed after adoption; re-adopt to pick up the current file"
|
||||
return 0
|
||||
else
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# Hybrid install: install keepalived only if missing.
|
||||
@@ -1807,6 +1915,13 @@ fetch_and_deploy_keepalived_config() {
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
# STILL ACK. This report is the server's only evidence that the node converged, and
|
||||
# it used to be sent on the write path alone — so a single lost ack (a backend
|
||||
# restart, a 5xx, a network blip) left the VIP reading SYNCING forever: the node was
|
||||
# already correct on disk, took this early return every cycle, and never spoke again.
|
||||
# Re-asserting the state makes the loop self-healing, costs one request per ~2.5
|
||||
# minutes, and touches nothing on the node.
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "already converged"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
@@ -1822,11 +1937,43 @@ fetch_and_deploy_keepalived_config() {
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
# Capture what keepalived actually said. Discarding it made the fail-safe useless in
|
||||
# practice: the node was correctly protected, but neither the log nor the UI could say WHY,
|
||||
# so the only way forward was to reproduce the check by hand on the node. Sanitised hard
|
||||
# (quotes, backslashes and newlines removed, tail kept) because both `log` and _kp_report
|
||||
# embed the text in JSON built by string interpolation.
|
||||
local kp_out kp_err kp_fatal
|
||||
if ! kp_out=$(keepalived -t -f "$tmp_conf" 2>&1); then
|
||||
# keepalived's config-test EXIT CODE does not separate a fatal config error from a
|
||||
# harmless warning. Measured on 2.2.8, not assumed:
|
||||
# clean config .................. 0
|
||||
# auth_pass longer than 8 chars . 5 "Truncating auth_pass to 8 characters"
|
||||
# missing '}' ................... 5 "There are 1 missing '}'s"
|
||||
# unknown keyword ............... 5 "Unknown keyword '...'"
|
||||
# script without script_security 6 "SECURITY VIOLATION ..."
|
||||
# So 5 covers both a benign truncation and a broken file, and treating any non-zero
|
||||
# exit as invalid rejected VALID configs: a VRRP password over 8 characters is enough,
|
||||
# and keepalived truncates it to 8 regardless, exactly as it does for the file the
|
||||
# operator already runs. Judge on the OUTPUT instead, dropping only messages known to
|
||||
# be benign; anything unrecognised is still fatal, so this fails CLOSED.
|
||||
if [[ -z "${kp_out//[[:space:]]/}" ]]; then
|
||||
# Non-zero with NOTHING to read. We cannot confirm the reason is benign, and some
|
||||
# builds log to syslog rather than stderr, so proceeding here would silently accept
|
||||
# every config on such a host. Treat as fatal — the gate must fail closed.
|
||||
kp_fatal="keepalived -t exited non-zero without output"
|
||||
else
|
||||
kp_fatal=$(printf '%s\n' "$kp_out" \
|
||||
| grep -v 'Truncating auth_pass to 8 characters' \
|
||||
| grep -v '^[[:space:]]*$' || true)
|
||||
fi
|
||||
if [[ -n "$kp_fatal" ]]; then
|
||||
rm -f "$tmp_conf"
|
||||
kp_err=$(printf '%s' "$kp_fatal" | tr '\n\r\t' ' ' | tr -d '"\\' | tail -c 300)
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t): ${kp_err} — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed: ${kp_err}"
|
||||
return 0
|
||||
fi
|
||||
log "WARN" "KEEPALIVED: keepalived -t exited non-zero with only known-benign warnings; proceeding"
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
@@ -2096,6 +2243,95 @@ UPGRADE_EOF
|
||||
}
|
||||
|
||||
# Check and apply configuration updates (Linux)
|
||||
# Config Import: upload this node's live haproxy.cfg when the operator asks for it.
|
||||
#
|
||||
# This function existed ONLY in the installer body and in the in-script daemon, never in the
|
||||
# heredoc that a FRESH install writes to /usr/local/bin/haproxy-agent. The call site below is
|
||||
# guarded by `type check_config_requests`, so on a freshly installed agent the whole feature
|
||||
# was a silent no-op: the operator requested a config from the node and nothing ever arrived,
|
||||
# with no error anywhere. Agents that had self-upgraded at least once did have it, which is why
|
||||
# it went unnoticed. Copied verbatim from the in-script daemon so both install routes behave
|
||||
# identically (v1.11.1).
|
||||
check_config_requests() {
|
||||
local curl_bin="${CURL_BIN:-$(find_binary curl)}"
|
||||
|
||||
# Get pending config requests from backend
|
||||
log "DEBUG" "CONFIG: Checking for pending config requests from: ${MANAGEMENT_URL}/api/configuration/agents/${AGENT_NAME}/pending-requests"
|
||||
|
||||
local response=$("$curl_bin" -k -s -X GET "${MANAGEMENT_URL}/api/configuration/agents/${AGENT_NAME}/pending-requests" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "X-API-Key: ${AGENT_TOKEN}")
|
||||
|
||||
log "DEBUG" "CONFIG: Response: ${response:0:200}..."
|
||||
|
||||
# Check if there are pending requests
|
||||
local pending_count=$(echo "$response" | jq -r '.pending_requests | length' 2>/dev/null)
|
||||
|
||||
log "DEBUG" "CONFIG: Pending count: $pending_count"
|
||||
|
||||
if [[ "$pending_count" -gt 0 ]]; then
|
||||
log "INFO" "CONFIG: Found $pending_count pending config request(s)"
|
||||
|
||||
# Process each request
|
||||
echo "$response" | jq -c '.pending_requests[]' 2>/dev/null | while read -r request; do
|
||||
local request_id=$(echo "$request" | jq -r '.request_id')
|
||||
local request_type=$(echo "$request" | jq -r '.request_type')
|
||||
|
||||
log "INFO" "CONFIG: Processing config request #$request_id (type: $request_type)"
|
||||
|
||||
# Read haproxy.cfg content
|
||||
local config_path="${HAPROXY_CONFIG_PATH}"
|
||||
if [[ -f "$config_path" ]]; then
|
||||
local file_size=$(wc -c < "$config_path")
|
||||
log "INFO" "CONFIG: Reading config from: $config_path (size: $file_size bytes)"
|
||||
|
||||
local config_content=$(cat "$config_path")
|
||||
|
||||
# Escape config content for JSON
|
||||
log "DEBUG" "CONFIG: Escaping config content for JSON..."
|
||||
local escaped_content=$(echo "$config_content" | jq -Rs .)
|
||||
|
||||
if [[ -z "$escaped_content" ]]; then
|
||||
log "ERROR" "CONFIG: JSON escaping failed for config content!"
|
||||
continue
|
||||
fi
|
||||
|
||||
local escaped_size=${#escaped_content}
|
||||
log "DEBUG" "CONFIG: JSON escaped content size: $escaped_size bytes"
|
||||
|
||||
# Submit config response
|
||||
local response_payload=$(cat <<CONFIG_RESPONSE_EOF
|
||||
{
|
||||
"request_id": $request_id,
|
||||
"config_content": $escaped_content,
|
||||
"config_path": "$config_path"
|
||||
}
|
||||
CONFIG_RESPONSE_EOF
|
||||
)
|
||||
|
||||
local payload_size=${#response_payload}
|
||||
log "INFO" "CONFIG: Sending config response (payload size: $payload_size bytes)..."
|
||||
|
||||
local http_response=$("$curl_bin" -k -s -w "\n%{http_code}" -X POST "${MANAGEMENT_URL}/api/configuration/agents/${AGENT_NAME}/config-response" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "X-API-Key: ${AGENT_TOKEN}" \
|
||||
-d "$response_payload")
|
||||
|
||||
local http_code=$(echo "$http_response" | tail -n1)
|
||||
local response_body=$(echo "$http_response" | head -n-1)
|
||||
|
||||
if [[ "$http_code" == "200" || "$http_code" == "201" ]]; then
|
||||
log "INFO" "CONFIG: Config response sent for request #$request_id (size: $file_size bytes, HTTP $http_code)"
|
||||
else
|
||||
log "ERROR" "CONFIG: Config response failed for request #$request_id (HTTP $http_code): $response_body"
|
||||
fi
|
||||
else
|
||||
log "ERROR" "CONFIG: Config file not found: $config_path"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
}
|
||||
|
||||
check_config_updates() {
|
||||
# Only log config check in debug mode to reduce log spam
|
||||
[[ "${DEBUG_MODE:-0}" == "1" ]] && log "INFO" "Checking for configuration updates..."
|
||||
@@ -3242,7 +3478,7 @@ CONFIG_RESPONSE_EOF
|
||||
# the live token at the top of each loop iteration below.
|
||||
fetch_and_deploy_keepalived_config() {
|
||||
local marker="# Managed by HAProxy OpenManager"
|
||||
local resp status http_code we_own="false" conf chk
|
||||
local resp status http_code we_own="false" conf chk disc_known=""
|
||||
|
||||
# Timeouts so a hung management server can never stall the daemon loop.
|
||||
resp=$(curl -k -s --connect-timeout 10 --max-time 30 -w '\n%{http_code}' -X GET \
|
||||
@@ -3261,6 +3497,84 @@ CONFIG_RESPONSE_EOF
|
||||
[[ -z "$conf" || "$conf" == "null" ]] && conf="/etc/keepalived/keepalived.conf"
|
||||
chk="$(dirname "$conf")/check_haproxy.sh"
|
||||
if [[ -f "$conf" ]] && grep -q "$marker" "$conf" 2>/dev/null; then we_own="true"; fi
|
||||
# See the heredoc copy.
|
||||
disc_known=$(echo "$resp" | jq -r 'if has("discovery_known") then (.discovery_known|tostring) else "" end' 2>/dev/null)
|
||||
|
||||
# v1.10.4 — VIP adoption discovery. Report a keepalived.conf we do NOT own so an existing
|
||||
# VIP can be adopted from the UI instead of retyped. STRICTLY READ-ONLY: this never writes
|
||||
# to the node. The heartbeat cannot carry this — it has the VIP address and a best-effort
|
||||
# MASTER/BACKUP, while rendering a node's config needs eleven fields.
|
||||
#
|
||||
# Rate limited by content: the hash of the last report is cached next to the config, so the
|
||||
# file (which may contain the VRRP password) is posted only when it actually changes, not on
|
||||
# every cycle. Once we own the file there is nothing to adopt, so the record is cleared once.
|
||||
_kp_discover() {
|
||||
local cache="$(dirname "$conf")/.hom_discovery_hash" cur_hash="" body content_json
|
||||
if [[ "$we_own" == "true" || ! -f "$conf" ]]; then
|
||||
# Nothing adoptable here. Clear a previous report exactly once.
|
||||
[[ -f "$cache" ]] || return 0
|
||||
# See the heredoc copy: clear only on a 2xx, or a managed node keeps being
|
||||
# offered for adoption and nothing ever retries the clear.
|
||||
local clr_code
|
||||
clr_code=$(curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -w '%{http_code}' \
|
||||
-X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
-d "{\"config_path\":\"$conf\",\"exists\":false}" 2>/dev/null)
|
||||
[[ "$clr_code" =~ ^2[0-9][0-9]$ ]] || return 0
|
||||
rm -f "$cache"
|
||||
return 0
|
||||
fi
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
[[ -z "$cur_hash" ]] && return 0
|
||||
# See the heredoc copy: the cache may only suppress while the server agrees it holds
|
||||
# a discovery for this node, or a rejected report hides the node permanently.
|
||||
# Cache line is "<hash>" after a success, or "<hash> rejected" after a 4xx.
|
||||
local cached_line cached_hash cached_state=""
|
||||
cached_line=$(cat "$cache" 2>/dev/null)
|
||||
cached_hash="${cached_line%% *}"
|
||||
[[ "$cached_line" == *" "* ]] && cached_state="${cached_line#* }"
|
||||
if [[ -n "$cached_hash" && "$cached_hash" == "$cur_hash" ]]; then
|
||||
# A 4xx means the server refuses THESE BYTES. Re-posting them cannot succeed, and
|
||||
# every attempt is stored as a failed agent call (4xx/5xx are never sampled out), so
|
||||
# an unattended loop would write a row carrying the whole config every cycle. Stay
|
||||
# quiet until the file changes — the reason was logged when it was rejected.
|
||||
[[ "$cached_state" == "rejected" ]] && return 0
|
||||
# Otherwise the cache only holds while the SERVER agrees it has the report.
|
||||
[[ "$disc_known" != "false" ]] && return 0
|
||||
fi
|
||||
# jq -Rs makes the file a single JSON string with its newlines intact, so the content the
|
||||
# server hashes is byte-identical to what is on disk — the takeover authorisation is
|
||||
# pinned to that hash.
|
||||
content_json=$(jq -Rs . < "$conf" 2>/dev/null) || return 0
|
||||
body=$(jq -n --arg p "$conf" --argjson c "$content_json" \
|
||||
'{config_path:$p, exists:true, is_managed:false, config_content:$c}' 2>/dev/null) || return 0
|
||||
# See the heredoc copy: cache ONLY on a 2xx, or a rejected report is recorded as
|
||||
# delivered and the node never reappears in the adoption panel.
|
||||
local disc_code
|
||||
disc_code=$(curl -k -s --connect-timeout 10 --max-time 30 -o /dev/null -w '%{http_code}' \
|
||||
-X POST "$MANAGEMENT_URL/api/agents/$AGENT_NAME/keepalived-discovery" \
|
||||
-H "X-API-Key: $AGENT_TOKEN" -H "Content-Type: application/json" \
|
||||
--data-binary "$body" 2>/dev/null)
|
||||
if [[ "$disc_code" =~ ^2[0-9][0-9]$ ]]; then
|
||||
printf '%s' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "INFO" "KEEPALIVED: reported an unmanaged keepalived.conf for adoption"
|
||||
elif [[ "$disc_code" == "400" || "$disc_code" == "413" || "$disc_code" == "422" ]]; then
|
||||
# ONLY the codes that mean "these bytes are unacceptable" stop the retry: too large
|
||||
# to analyse, malformed, rejected by validation. Re-posting identical content cannot
|
||||
# change any of those answers.
|
||||
#
|
||||
# 401 and 404 are deliberately NOT here even though they are 4xx. Both are transient
|
||||
# in this system — a token rotation, or an agent row briefly absent while it
|
||||
# re-registers — and treating them as permanent would silence discovery for every
|
||||
# affected node until its keepalived.conf changed, which for a hand-maintained file
|
||||
# may be never. That is the exact failure this release set out to remove.
|
||||
printf '%s rejected' "$cur_hash" > "$cache" 2>/dev/null
|
||||
log "WARN" "KEEPALIVED: discovery report refused (HTTP $disc_code); not retrying until the config changes"
|
||||
else
|
||||
log "WARN" "KEEPALIVED: discovery report failed (HTTP ${disc_code:-none}); will retry next cycle"
|
||||
fi
|
||||
}
|
||||
_kp_discover
|
||||
|
||||
_kp_report() {
|
||||
local vid="${2:-null}"; [[ -z "$2" ]] && vid="null"
|
||||
@@ -3321,11 +3635,26 @@ CONFIG_RESPONSE_EOF
|
||||
[[ -z "$new_conf" ]] && return 0
|
||||
|
||||
if [[ -f "$conf" && "$we_own" != "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
local allow_takeover expected_hash disk_hash
|
||||
allow_takeover=$(echo "$resp" | jq -r '.keepalived.allow_takeover // false' 2>/dev/null)
|
||||
expected_hash=$(echo "$resp" | jq -r '.keepalived.takeover_expected_hash // empty' 2>/dev/null)
|
||||
disk_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
if [[ "$allow_takeover" == "true" && -n "$expected_hash" && "$disk_hash" == "$expected_hash" ]]; then
|
||||
log "INFO" "KEEPALIVED: adopting $conf (one-shot takeover authorised; on-disk hash matches)"
|
||||
elif [[ "$allow_takeover" == "true" ]]; then
|
||||
log "WARN" "KEEPALIVED: adoption authorised but $conf changed since it was adopted — refusing"
|
||||
_kp_report "externally_managed" "$vip_id" "$disk_hash" \
|
||||
"config changed after adoption; re-adopt to pick up the current file"
|
||||
return 0
|
||||
else
|
||||
log "WARN" "KEEPALIVED: $conf is externally managed — refusing to overwrite"
|
||||
_kp_report "externally_managed" "$vip_id" "" "pre-existing unmanaged keepalived.conf"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
# Hybrid install: install keepalived only if missing.
|
||||
|
||||
if ! command -v keepalived >/dev/null 2>&1; then
|
||||
if [[ "$install_if" == "true" ]]; then
|
||||
log "INFO" "KEEPALIVED: installing package..."
|
||||
@@ -3351,6 +3680,9 @@ CONFIG_RESPONSE_EOF
|
||||
cur_hash=$(md5sum "$conf" 2>/dev/null | awk '{print $1}')
|
||||
would_hash=$(printf '%s' "$new_conf" | md5sum 2>/dev/null | awk '{print $1}')
|
||||
if [[ -n "$cur_hash" && "$cur_hash" == "$would_hash" ]]; then
|
||||
# See the heredoc copy: the ack must be re-asserted here or a single lost report
|
||||
# leaves the VIP reading SYNCING forever.
|
||||
_kp_report "enabled" "$vip_id" "$new_hash" "already converged"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
@@ -3366,11 +3698,26 @@ CONFIG_RESPONSE_EOF
|
||||
local tmp_conf="${conf}.hom.tmp"
|
||||
printf '%s' "$new_conf" > "$tmp_conf"
|
||||
chmod 0644 "$tmp_conf"
|
||||
if ! keepalived -t -f "$tmp_conf" >/dev/null 2>&1; then
|
||||
rm -f "$tmp_conf"
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t) — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed"
|
||||
return 0
|
||||
# See the heredoc copy for why the output is captured rather than discarded.
|
||||
# See the heredoc copy for the measured exit-code table and why the OUTPUT, not the
|
||||
# exit code, decides. Fails closed on anything not known to be benign.
|
||||
local kp_out kp_err kp_fatal
|
||||
if ! kp_out=$(keepalived -t -f "$tmp_conf" 2>&1); then
|
||||
if [[ -z "${kp_out//[[:space:]]/}" ]]; then
|
||||
kp_fatal="keepalived -t exited non-zero without output"
|
||||
else
|
||||
kp_fatal=$(printf '%s\n' "$kp_out" \
|
||||
| grep -v 'Truncating auth_pass to 8 characters' \
|
||||
| grep -v '^[[:space:]]*$' || true)
|
||||
fi
|
||||
if [[ -n "$kp_fatal" ]]; then
|
||||
rm -f "$tmp_conf"
|
||||
kp_err=$(printf '%s' "$kp_fatal" | tr '\n\r\t' ' ' | tr -d '"\\' | tail -c 300)
|
||||
log "ERROR" "KEEPALIVED: config validation failed (keepalived -t): ${kp_err} — keeping current config, not (re)starting"
|
||||
_kp_report "error" "$vip_id" "$new_hash" "keepalived -t failed: ${kp_err}"
|
||||
return 0
|
||||
fi
|
||||
log "WARN" "KEEPALIVED: keepalived -t exited non-zero with only known-benign warnings; proceeding"
|
||||
fi
|
||||
mv -f "$tmp_conf" "$conf"
|
||||
chown root:root "$conf" 2>/dev/null
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Issue #53 — at-rest encryption for the pending CSR private key (v1.10.1).
|
||||
|
||||
Mirrors the established Fernet + HKDF(SECRET_KEY) pattern already used for the VRRP secret
|
||||
(services/keepalived_config.py), TOTP secrets (services/mfa_service.py) and DNS provider
|
||||
credentials (utils/dns_credentials.py): prefer an explicit CSR_ENCRYPTION_KEY env var (enables
|
||||
key rotation), else derive a stable key from SECRET_KEY via HKDF with its own versioned info
|
||||
string, so a rotation of one secret class never affects another.
|
||||
|
||||
WHY this key and not every key in the system: the CSR private key is the one key that sits IDLE.
|
||||
It is generated at CSR creation, waits for an external CA to sign the request (days to weeks),
|
||||
and is destroyed the moment the signed certificate is imported — it is never transmitted to an
|
||||
agent and never leaves the server. `ssl_certificates.private_key_content` and the ACME order keys
|
||||
are different: agents must receive them in plaintext on every poll, so encrypting them at rest
|
||||
buys nothing without an end-to-end redesign.
|
||||
|
||||
STORAGE: the Fernet token replaces the PEM in the SAME `ssl_csrs.private_key_pem` TEXT column.
|
||||
No new column, no new table, and deliberately NO `SCHEMA_VERSION` bump — a bump would re-run the
|
||||
migration sequence and re-seed the four built-in roles to their defaults (see UPGRADE_GUIDE.md),
|
||||
which is a needless side effect for a storage-format change.
|
||||
|
||||
BACKWARD COMPATIBILITY: rows written before this release hold a raw PEM. `decrypt_csr_private_key`
|
||||
detects those by their `-----BEGIN` header and returns them unchanged. The discriminator is exact,
|
||||
not a heuristic: a Fernet token is base64url text and can never contain "-----". Legacy rows drain
|
||||
naturally, since a CSR's key copy is NULLed on import.
|
||||
|
||||
KEY ROTATION: if SECRET_KEY rotates while CSR_ENCRYPTION_KEY is unset, previously stored keys
|
||||
become undecryptable and `decrypt_csr_private_key` returns None. Callers MUST surface a clear
|
||||
"delete this CSR and create a new one" error — the CSR is unusable at that point, because the
|
||||
signed certificate can no longer be paired with its key.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from cryptography.fernet import Fernet, InvalidToken
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
from config import SECRET_KEY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A PEM private key always carries this header; a Fernet token is base64url and never can.
|
||||
_PEM_MARKER = "-----BEGIN"
|
||||
|
||||
_fernet_instance: Optional[Fernet] = None
|
||||
|
||||
|
||||
def _resolve_fernet_key() -> bytes:
|
||||
"""Prefer an explicit CSR_ENCRYPTION_KEY; else derive from SECRET_KEY via HKDF with a
|
||||
versioned info string (so stored keys survive restarts)."""
|
||||
explicit = os.getenv("CSR_ENCRYPTION_KEY", "").strip()
|
||||
if explicit:
|
||||
try:
|
||||
Fernet(explicit.encode())
|
||||
return explicit.encode()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("CSR_ENCRYPTION_KEY env var present but invalid: %s", exc)
|
||||
logger.warning(
|
||||
"CSR_ENCRYPTION_KEY not set; deriving the CSR private-key encryption key from SECRET_KEY. "
|
||||
"Set CSR_ENCRYPTION_KEY to a Fernet key to enable key rotation."
|
||||
)
|
||||
hkdf = HKDF(algorithm=hashes.SHA256(), length=32, salt=None, info=b"csr-private-key-v1")
|
||||
derived = hkdf.derive(SECRET_KEY.encode("utf-8"))
|
||||
return base64.urlsafe_b64encode(derived)
|
||||
|
||||
|
||||
def _get_fernet() -> Fernet:
|
||||
global _fernet_instance
|
||||
if _fernet_instance is None:
|
||||
_fernet_instance = Fernet(_resolve_fernet_key())
|
||||
return _fernet_instance
|
||||
|
||||
|
||||
def reset_fernet_for_tests() -> None:
|
||||
"""Test-only hook to force re-resolution after env mutation."""
|
||||
global _fernet_instance
|
||||
_fernet_instance = None
|
||||
|
||||
|
||||
def is_encrypted(stored: Optional[str]) -> bool:
|
||||
"""True when the stored value is a Fernet token rather than a legacy raw PEM.
|
||||
|
||||
Single source of the format discriminator: `decrypt_csr_private_key` branches on this, so
|
||||
the "what does a stored value look like" rule is stated exactly once.
|
||||
"""
|
||||
return bool(stored) and _PEM_MARKER not in stored
|
||||
|
||||
|
||||
def encrypt_csr_private_key(pem: str) -> str:
|
||||
"""Fernet-encrypt a PEM private key to a storable token string."""
|
||||
return _get_fernet().encrypt(pem.encode("utf-8")).decode("utf-8")
|
||||
|
||||
|
||||
def decrypt_csr_private_key(stored: Optional[str]) -> Optional[str]:
|
||||
"""Return the PEM private key for a stored value.
|
||||
|
||||
Accepts BOTH shapes so an upgrade needs no data migration:
|
||||
- a raw PEM written before v1.10.1 -> returned unchanged
|
||||
- a Fernet token -> decrypted
|
||||
|
||||
Returns None when the value is empty or cannot be decrypted (e.g. SECRET_KEY rotated without
|
||||
CSR_ENCRYPTION_KEY). Callers MUST treat None as "this CSR's key is unrecoverable" and tell the
|
||||
operator to delete it and create a new one; never fall through to a pairing attempt.
|
||||
"""
|
||||
if not stored:
|
||||
return None
|
||||
if not is_encrypted(stored):
|
||||
return stored # legacy plaintext row, pre-v1.10.1
|
||||
try:
|
||||
return _get_fernet().decrypt(stored.encode("utf-8")).decode("utf-8")
|
||||
except InvalidToken:
|
||||
logger.warning("Failed to decrypt a stored CSR private key (invalid Fernet token)")
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Unexpected error decrypting a stored CSR private key: %s", exc)
|
||||
return None
|
||||
@@ -0,0 +1,328 @@
|
||||
"""v1.11.0 — outbound half of the unified request/response log.
|
||||
|
||||
This is deliberately NOT a session or connector factory. Three incompatible
|
||||
connector policies coexist in this codebase:
|
||||
|
||||
* `utils.ssrf_guard.safe_connector()` — IPv4-pinned, TLS verification on;
|
||||
returns a NEW connector per call because `ClientSession` closes the one it
|
||||
owns, so a shared long-lived connector would raise "Connector is closed".
|
||||
* `services/acme_diagnostics.py` — IPv4-pinned with `ssl=False` for the
|
||||
plain-HTTP port-80 probe.
|
||||
* the DNS providers and the CA-chain import — the default dual-stack
|
||||
connector.
|
||||
|
||||
On top of that, `backend/tests/test_acme_diagnostics.py` monkeypatches
|
||||
`aiohttp.ClientSession` globally with fakes that implement only
|
||||
`__aenter__/__aexit__/head(...)`. Centralising session construction would break
|
||||
all of it. So this module wraps the CALL, never the session.
|
||||
|
||||
Two hard rules, both load-bearing:
|
||||
|
||||
1. `outbound_span` NEVER raises. Both DNS provider funnels end in
|
||||
`except Exception: raise DnsProviderError("Unexpected ... failure")`, and in
|
||||
GoDaddy's publish path that reverts `dns_record_published` and stalls the
|
||||
ACME order — an instrumentation bug must not masquerade as a provider
|
||||
outage.
|
||||
2. `outbound_span` NEVER swallows. An exception raised inside the block is
|
||||
recorded (status_class 0) and re-raised unchanged.
|
||||
"""
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
import uuid
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from utils.request_log_redaction import safe_error_text, scrub_query_string, scrub_url
|
||||
from utils.request_log_settings import get_config
|
||||
from utils.request_log_sink import RequestLogRow, request_id_context, request_log_sink
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
# Stable identifiers for the `request_logs.target` column — this is the
|
||||
# "kime gitti" (who did we call) axis of the log.
|
||||
TARGET_ACME = "acme"
|
||||
TARGET_ACME_DIAG = "acme_diag"
|
||||
TARGET_LETSENCRYPT_CA = "letsencrypt_ca"
|
||||
TARGET_DNS_CLOUDFLARE = "dns_cloudflare"
|
||||
TARGET_DNS_GODADDY = "dns_godaddy"
|
||||
TARGET_AGENT = "agent"
|
||||
TARGET_HAPROXY_STATS = "haproxy_stats"
|
||||
TARGET_SETTINGS_PROBE = "settings_probe"
|
||||
|
||||
|
||||
def begin_background_trace(label: str) -> str:
|
||||
"""Open a fresh correlation id for ONE iteration of a background loop.
|
||||
|
||||
Without this, background outbound rows fell back to `bg:<asyncio task
|
||||
name>`. Nothing in main.py passes `name=` to `create_task`, so a loop is
|
||||
`Task-5` for its entire life and EVERY call it ever makes carries the same
|
||||
`request_id` — measured: fifteen ACME calls across five renewal ticks came
|
||||
out as one id. `GET /api/request-logs/{id}` then answers with up to 100 rows
|
||||
under `related`, presented as "the calls this request made", which in a
|
||||
forensics tool is worse than having no trace: an operator reading a failed
|
||||
renewal is shown a hundred unrelated calls spanning days. Task numbers are
|
||||
also reused across restarts, so `bg:Task-5` can mean a different loop after
|
||||
a redeploy.
|
||||
|
||||
Called at the top of each iteration; the loop task is dedicated, so the next
|
||||
iteration simply overwrites it and there is nothing to reset.
|
||||
"""
|
||||
trace_id = f"bg:{label}:{uuid.uuid4().hex[:12]}"[:64]
|
||||
request_id_context.set(trace_id)
|
||||
return trace_id
|
||||
|
||||
|
||||
def _correlation_id() -> str:
|
||||
"""Inherit the inbound request's id when there is one, so an API call and
|
||||
the CA/DNS calls it triggered share a trace. Background work gets the id
|
||||
opened by begin_background_trace() for the current iteration."""
|
||||
existing = request_id_context.get()
|
||||
if existing:
|
||||
return existing
|
||||
# No inbound request and no iteration trace: background code that has not
|
||||
# been wrapped. Mint a unique id rather than falling back to the task name,
|
||||
# which would silently re-collapse every such call into one row group.
|
||||
try:
|
||||
task = asyncio.current_task()
|
||||
name = task.get_name() if task else "unknown"
|
||||
except Exception:
|
||||
name = "unknown"
|
||||
return f"bg:{name}:{uuid.uuid4().hex[:12]}"[:64]
|
||||
|
||||
|
||||
class OutboundSpan:
|
||||
"""Handle passed to the `async with` body so the call site can attach the
|
||||
response it just read."""
|
||||
|
||||
__slots__ = (
|
||||
"target", "method", "url", "capture_request_body", "capture_response_body",
|
||||
"safe_error_only",
|
||||
"_status", "_response_headers", "_response_body", "_response_bytes",
|
||||
"_response_content_type", "_request_body", "_request_headers", "_error",
|
||||
)
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
target: str,
|
||||
method: str,
|
||||
url: str,
|
||||
capture_request_body: bool,
|
||||
capture_response_body: bool,
|
||||
safe_error_only: bool,
|
||||
request_body: Any = None,
|
||||
request_headers: Optional[Dict[str, str]] = None,
|
||||
):
|
||||
self.target = target
|
||||
self.method = (method or "GET").upper()
|
||||
self.url = url
|
||||
# Two independent switches on purpose: the ACME JWS request body is a
|
||||
# replayable credential and must never be stored, but the CA's RESPONSE
|
||||
# (problem JSON, order state) is exactly what an operator needs to see.
|
||||
self.capture_request_body = capture_request_body
|
||||
self.capture_response_body = capture_response_body
|
||||
self.safe_error_only = safe_error_only
|
||||
self._request_body = request_body
|
||||
self._request_headers = request_headers
|
||||
self._status: Optional[int] = None
|
||||
self._response_headers: Optional[Dict[str, str]] = None
|
||||
self._response_body: Any = None
|
||||
self._response_bytes: int = 0
|
||||
self._response_content_type: Optional[str] = None
|
||||
self._error: Optional[str] = None
|
||||
|
||||
def set_response(
|
||||
self,
|
||||
status: Optional[int],
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
body: Any = None,
|
||||
) -> None:
|
||||
"""Record what came back. Safe to call with a partially-read response;
|
||||
never raises, so a call site can hand us whatever it happens to have."""
|
||||
try:
|
||||
self._status = int(status) if status is not None else None
|
||||
except (TypeError, ValueError):
|
||||
self._status = None
|
||||
try:
|
||||
if headers:
|
||||
self._response_headers = {str(k).lower(): str(v) for k, v in dict(headers).items()}
|
||||
self._response_content_type = self._response_headers.get("content-type")
|
||||
except Exception:
|
||||
self._response_headers = None
|
||||
|
||||
if body is None or not self.capture_response_body:
|
||||
return
|
||||
try:
|
||||
if isinstance(body, (bytes, bytearray)):
|
||||
self._response_bytes = len(body)
|
||||
cap = get_config().max_body_bytes
|
||||
self._response_body = bytes(body[:cap]) if cap else None
|
||||
elif isinstance(body, str):
|
||||
encoded = body.encode("utf-8", "replace")
|
||||
self._response_bytes = len(encoded)
|
||||
cap = get_config().max_body_bytes
|
||||
self._response_body = encoded[:cap] if cap else None
|
||||
else:
|
||||
# Already-decoded JSON (the common case: `await resp.json()`).
|
||||
self._response_body = body
|
||||
except Exception:
|
||||
self._response_body = None
|
||||
|
||||
def set_error(self, exc: BaseException, *, type_only: Optional[bool] = None) -> None:
|
||||
try:
|
||||
only = self.safe_error_only if type_only is None else type_only
|
||||
self._error = safe_error_text(exc, type_only=only)
|
||||
except Exception:
|
||||
self._error = "UnknownError"
|
||||
|
||||
def to_row(self, duration_ms: int) -> RequestLogRow:
|
||||
scrubbed = scrub_url(self.url)
|
||||
path = None
|
||||
query_params = None
|
||||
try:
|
||||
import urllib.parse
|
||||
|
||||
parts = urllib.parse.urlsplit(self.url)
|
||||
path = parts.path or "/"
|
||||
_, query_params = scrub_query_string(parts.query)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
row = RequestLogRow(
|
||||
request_id=_correlation_id(),
|
||||
direction="outbound",
|
||||
target=self.target,
|
||||
method=self.method,
|
||||
url=scrubbed,
|
||||
path=path,
|
||||
query_params=query_params,
|
||||
status_code=self._status,
|
||||
duration_ms=duration_ms,
|
||||
request_headers=self._request_headers,
|
||||
response_headers=self._response_headers,
|
||||
error=self._error,
|
||||
)
|
||||
|
||||
if self._request_body is not None:
|
||||
if not self.capture_request_body:
|
||||
# The call site handed us a synthetic SUMMARY instead of the real
|
||||
# payload (the ACME JWS case) — store the summary as-is.
|
||||
row.request_body_value = _redacted_value(self._request_body)
|
||||
elif isinstance(self._request_body, (bytes, bytearray)):
|
||||
row.request_body_bytes = len(self._request_body)
|
||||
cap = get_config().max_body_bytes
|
||||
row.request_body_raw = bytes(self._request_body[:cap]) if cap else None
|
||||
else:
|
||||
row.request_body_value = _redacted_value(self._request_body)
|
||||
|
||||
if isinstance(self._response_body, (bytes, bytearray)):
|
||||
row.response_body_raw = bytes(self._response_body)
|
||||
row.response_body_bytes = self._response_bytes or len(self._response_body)
|
||||
row.response_content_type = self._response_content_type
|
||||
elif self._response_body is not None:
|
||||
row.response_body_value = _redacted_value(self._response_body)
|
||||
|
||||
return row
|
||||
|
||||
|
||||
def _redacted_value(value: Any) -> Any:
|
||||
from utils.request_log_redaction import redact
|
||||
|
||||
return redact(value)
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def outbound_span(
|
||||
*,
|
||||
target: str,
|
||||
method: str,
|
||||
url: str,
|
||||
request_body: Any = None,
|
||||
request_headers: Optional[Dict[str, str]] = None,
|
||||
capture_body: bool = True,
|
||||
capture_response_body: bool = True,
|
||||
safe_error_only: bool = False,
|
||||
):
|
||||
"""Time an outbound HTTP call and record one `direction='outbound'` row.
|
||||
|
||||
`capture_body=False` applies to the REQUEST body only, for payloads that
|
||||
are themselves credentials — the ACME JWS body is a replayable, signed
|
||||
capability for the lifetime of its nonce, so the call site passes a
|
||||
description of it instead. The CA's response is still captured, because
|
||||
that is the half an operator actually needs when an order fails.
|
||||
|
||||
`safe_error_only=True` reduces a recorded exception to its type name, for
|
||||
the DNS providers whose own error handling already refuses to surface
|
||||
`str(exc)` (it can carry the request URL and, through it, zone identifiers).
|
||||
"""
|
||||
span: Optional[OutboundSpan] = None
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
cfg = get_config()
|
||||
if cfg.enabled and cfg.capture_outbound:
|
||||
span = OutboundSpan(
|
||||
target=target,
|
||||
method=method,
|
||||
url=url,
|
||||
capture_request_body=capture_body and cfg.capture_bodies,
|
||||
capture_response_body=capture_response_body and cfg.capture_bodies,
|
||||
safe_error_only=safe_error_only,
|
||||
request_body=request_body,
|
||||
request_headers=request_headers,
|
||||
)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"outbound_span: could not start span for {target}: {exc}")
|
||||
span = None
|
||||
|
||||
if span is None:
|
||||
# Logging is off (or failed to initialise) — yield a throwaway span so
|
||||
# the call site's `span.set_response(...)` still works.
|
||||
span = OutboundSpan(
|
||||
target=target, method=method, url=url,
|
||||
capture_request_body=False, capture_response_body=False,
|
||||
safe_error_only=safe_error_only,
|
||||
)
|
||||
try:
|
||||
yield span
|
||||
finally:
|
||||
pass
|
||||
return
|
||||
|
||||
try:
|
||||
yield span
|
||||
except BaseException as exc:
|
||||
try:
|
||||
span.set_error(exc)
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
finally:
|
||||
try:
|
||||
duration_ms = int((time.perf_counter() - started) * 1000)
|
||||
request_log_sink.offer(span.to_row(duration_ms))
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"outbound_span: failed to record row for {target}: {exc}")
|
||||
|
||||
|
||||
async def instrumented_request(session, method: str, url: str, *, target: str,
|
||||
safe_error_only: bool = True, capture_body: bool = True,
|
||||
**kwargs):
|
||||
"""Convenience wrapper for the call sites that already funnel through
|
||||
`session.request(...)` (the two DNS providers).
|
||||
|
||||
Returns `(status, headers, text)` and leaves error handling entirely to the
|
||||
caller — this helper only adds the log row.
|
||||
"""
|
||||
async with outbound_span(
|
||||
target=target,
|
||||
method=method,
|
||||
url=url,
|
||||
request_body=kwargs.get("json"),
|
||||
capture_body=capture_body,
|
||||
safe_error_only=safe_error_only,
|
||||
) as span:
|
||||
async with session.request(method, url, **kwargs) as resp:
|
||||
text = await resp.text()
|
||||
span.set_response(resp.status, dict(resp.headers), text)
|
||||
return resp.status, dict(resp.headers), text
|
||||
@@ -0,0 +1,201 @@
|
||||
"""v1.11.0 — retention prune for `request_logs`.
|
||||
|
||||
Three independent limits, applied in order:
|
||||
|
||||
1. successful rows (`status_class` 1..3) older than `success_retention_days`
|
||||
2. errored rows (`status_class` 0, 4, 5 — 0 meaning "no HTTP response at
|
||||
all") older than `error_retention_days`
|
||||
3. a hard row cap: anything below the `max_rows`-th newest id
|
||||
|
||||
Splitting success from error is the point of the design: a busy install can
|
||||
keep a week of ordinary traffic while still holding three months of failures
|
||||
for forensics, without paying for both.
|
||||
|
||||
Deliberately NOT folded into `utils/activity_log.prune_acme_events_and_drafts_if_due`:
|
||||
that function is driven by tests with fixed `execute.side_effect` lists and an
|
||||
exact return dict, and it is gated behind a `letsencrypt_orders`-exists check
|
||||
that would silently disable this prune on an ACME-free install.
|
||||
|
||||
Three safety properties, all of which matter at scale:
|
||||
|
||||
* **Batched deletes.** The pool sets `command_timeout=60`; an unbounded
|
||||
DELETE over a multi-million-row table raises `asyncpg.TimeoutError` and
|
||||
then nothing is ever pruned.
|
||||
* **Advisory lock.** `pg_try_advisory_lock` (try, never block) so N replicas
|
||||
× M uvicorn workers do not all scan at once.
|
||||
* **Watermark stamped only after a complete pass.** A pass that times out
|
||||
mid-way is retried at the next tick instead of being recorded as done.
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Dict, Optional
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.request_log_settings import get_config
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
# Fresh namespace. Already taken in this codebase: 18181818 (draft cap),
|
||||
# 18181819 (wizard create), 18181820 (apply), 0x41434D45 (per-ACME-order),
|
||||
# 1836016242 (migration lock).
|
||||
PRUNE_LOCK_KEY = 18181821
|
||||
|
||||
WATERMARK_KEY = "requestlog.last_pruned_at"
|
||||
|
||||
BATCH_SIZE = 5000
|
||||
MAX_BATCHES = 40 # ceiling of 200k rows removed per pass
|
||||
|
||||
# Retention days ALWAYS travel as a bind parameter. They are operator-supplied,
|
||||
# so interpolating them into the SQL string would be an injection point.
|
||||
_SQL_TTL_SUCCESS = """
|
||||
DELETE FROM request_logs
|
||||
WHERE ctid IN (
|
||||
SELECT ctid FROM request_logs
|
||||
WHERE status_class BETWEEN 1 AND 3
|
||||
AND created_at < NOW() - ($1 || ' days')::INTERVAL
|
||||
LIMIT $2
|
||||
)
|
||||
"""
|
||||
|
||||
_SQL_TTL_ERROR = """
|
||||
DELETE FROM request_logs
|
||||
WHERE ctid IN (
|
||||
SELECT ctid FROM request_logs
|
||||
WHERE (status_class = 0 OR status_class >= 4)
|
||||
AND created_at < NOW() - ($1 || ' days')::INTERVAL
|
||||
LIMIT $2
|
||||
)
|
||||
"""
|
||||
|
||||
_SQL_CAP_CUTOFF = "SELECT id FROM request_logs ORDER BY id DESC OFFSET $1 LIMIT 1"
|
||||
|
||||
_SQL_CAP_DELETE = """
|
||||
DELETE FROM request_logs
|
||||
WHERE ctid IN (
|
||||
SELECT ctid FROM request_logs WHERE id <= $1 LIMIT $2
|
||||
)
|
||||
"""
|
||||
|
||||
|
||||
def _deleted_count(result) -> int:
|
||||
"""asyncpg returns the command tag ('DELETE 42') from execute()."""
|
||||
if isinstance(result, str) and result.startswith("DELETE "):
|
||||
try:
|
||||
return int(result.split()[-1])
|
||||
except (ValueError, IndexError):
|
||||
return 0
|
||||
return 0
|
||||
|
||||
|
||||
async def _batched_delete(conn, sql: str, first_param) -> int:
|
||||
"""Run `sql` repeatedly until a short batch comes back or the ceiling hits."""
|
||||
total = 0
|
||||
for _ in range(MAX_BATCHES):
|
||||
result = await conn.execute(sql, first_param, BATCH_SIZE)
|
||||
count = _deleted_count(result)
|
||||
total += count
|
||||
if count < BATCH_SIZE:
|
||||
break
|
||||
else:
|
||||
logger.info(
|
||||
f"request_logs prune hit the {MAX_BATCHES}-batch ceiling "
|
||||
f"({total} rows this pass); the remainder is removed on the next run"
|
||||
)
|
||||
return total
|
||||
|
||||
|
||||
async def _is_due(conn, key: str, min_interval_seconds: int) -> bool:
|
||||
"""Watermark gate. Unlike the hardcoded 24h in utils/activity_log.py the
|
||||
interval here is operator-configurable."""
|
||||
row = await conn.fetchrow("SELECT value FROM system_settings WHERE key = $1", key)
|
||||
if not row or row["value"] is None:
|
||||
return True
|
||||
raw = row["value"]
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
raw = json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return True
|
||||
if not isinstance(raw, str):
|
||||
return True
|
||||
try:
|
||||
last = datetime.fromisoformat(raw.replace("Z", "+00:00"))
|
||||
except ValueError:
|
||||
return True
|
||||
age = (datetime.utcnow() - last.replace(tzinfo=None)).total_seconds()
|
||||
return age >= min_interval_seconds
|
||||
|
||||
|
||||
async def _stamp(conn, key: str) -> None:
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO system_settings (key, value, category, description)
|
||||
VALUES ($1, $2::jsonb, 'requestlog', 'Internal: last request_logs prune timestamp')
|
||||
ON CONFLICT (key) DO UPDATE
|
||||
SET value = EXCLUDED.value, updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
key,
|
||||
json.dumps(datetime.utcnow().isoformat() + "Z"),
|
||||
)
|
||||
|
||||
|
||||
async def _prune_row_cap(conn, max_rows: int) -> int:
|
||||
"""Delete everything below the `max_rows`-th newest id."""
|
||||
cutoff: Optional[int] = await conn.fetchval(_SQL_CAP_CUTOFF, max_rows)
|
||||
if cutoff is None:
|
||||
return 0 # fewer rows than the cap — nothing to do
|
||||
return await _batched_delete(conn, _SQL_CAP_DELETE, cutoff)
|
||||
|
||||
|
||||
async def prune_request_logs_if_due(force: bool = False) -> Dict[str, int]:
|
||||
"""Run one retention pass if the watermark says it is due.
|
||||
|
||||
Never raises: a prune failure must not take down the loop that calls it.
|
||||
`force=True` skips the watermark gate (used by the manual purge endpoint).
|
||||
"""
|
||||
counts = {"success": 0, "error": 0, "overflow": 0, "ran": 0}
|
||||
cfg = get_config()
|
||||
|
||||
conn = None
|
||||
locked = False
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
|
||||
# One replica only. try-lock: never block a pod waiting on another's pass.
|
||||
locked = await conn.fetchval("SELECT pg_try_advisory_lock($1)", PRUNE_LOCK_KEY)
|
||||
if not locked:
|
||||
return counts
|
||||
|
||||
if not force and not await _is_due(conn, WATERMARK_KEY, cfg.prune_interval_minutes * 60):
|
||||
return counts
|
||||
|
||||
counts["success"] = await _batched_delete(conn, _SQL_TTL_SUCCESS, str(cfg.success_retention_days))
|
||||
counts["error"] = await _batched_delete(conn, _SQL_TTL_ERROR, str(cfg.error_retention_days))
|
||||
counts["overflow"] = await _prune_row_cap(conn, cfg.max_rows)
|
||||
counts["ran"] = 1
|
||||
|
||||
# Only after all three steps completed — a partial pass must be retried,
|
||||
# not recorded as done.
|
||||
await _stamp(conn, WATERMARK_KEY)
|
||||
|
||||
if counts["success"] or counts["error"] or counts["overflow"]:
|
||||
logger.info(
|
||||
f"request_logs prune: {counts['success']} successful, {counts['error']} errored, "
|
||||
f"{counts['overflow']} over-cap row(s) removed"
|
||||
)
|
||||
return counts
|
||||
except Exception as exc:
|
||||
logger.warning(f"prune_request_logs_if_due: {exc}")
|
||||
return counts
|
||||
finally:
|
||||
if conn is not None:
|
||||
if locked:
|
||||
try:
|
||||
await conn.execute("SELECT pg_advisory_unlock($1)", PRUNE_LOCK_KEY)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
await close_database_connection(conn)
|
||||
except Exception:
|
||||
pass
|
||||
@@ -0,0 +1,517 @@
|
||||
"""v1.11.0 — redaction for the unified request/response log.
|
||||
|
||||
Everything that lands in `request_logs.request_body` / `response_body` /
|
||||
`request_headers` / `response_headers` / `query_params` passes through here
|
||||
first. The rules, in order of how much they are trusted:
|
||||
|
||||
1. **Headers are an ALLOWLIST.** Anything not explicitly listed is dropped.
|
||||
A small set of high-signal headers (`Authorization`, `Cookie`, …) is kept
|
||||
as a presence marker with the value replaced, so an operator debugging a
|
||||
401 can still see *that* a credential was sent.
|
||||
2. **Body keys are matched by a normalized name** (lowercased, punctuation
|
||||
stripped), against an exact set for short generic names that would
|
||||
over-match as substrings (`key`, `payload`) and a contains set for the
|
||||
compound ones (`cert_private_key`, `eab_hmac_key`, …).
|
||||
3. **Values are shape-checked too.** A PEM private key or a JWT-shaped string
|
||||
is redacted no matter what key it arrived under — this is the net that
|
||||
catches a route echoing a secret under a renamed field.
|
||||
|
||||
None of these functions raise: a redaction failure must never turn into a
|
||||
failed request or a failed provider call, so callers get a safe placeholder
|
||||
instead of an exception.
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import urllib.parse
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
REDACTED = "***REDACTED***"
|
||||
|
||||
# Short, generic names. Matched EXACTLY after normalization, because as
|
||||
# substrings they would swallow innocent fields (`key_suffix`, `monkey`,
|
||||
# `payload_size`, `keyboard`, `nonce_count`).
|
||||
REDACT_EXACT = {
|
||||
"password", "passwd", "pwd", "secret", "token", "key", "auth",
|
||||
"authorization", "cookie", "signature", "protected", "payload",
|
||||
"nonce", "credentials", "credential", "otp", "pin", "jwk", "csr",
|
||||
# `auth_pass` is keepalived's VRRP password and it is the plaintext field
|
||||
# name on `POST/PUT /api/vip` (routers/vip.py binds `payload.auth_pass`
|
||||
# straight into encrypt_vrrp_secret). It normalizes to "authpass", which
|
||||
# matches NOTHING above: "password" is not a substring of "authpass", and
|
||||
# the bare "auth" entry is an EXACT match, not a prefix. Without this line
|
||||
# the VRRP secret is written to request_logs in cleartext on every VIP
|
||||
# create and edit.
|
||||
"authpass",
|
||||
}
|
||||
|
||||
# Compound names. Matched as SUBSTRINGS of the normalized key.
|
||||
#
|
||||
# `token` is in here on purpose, not just its compounds. In this domain EVERY
|
||||
# field whose name contains "token" is a credential — api_token (the Cloudflare
|
||||
# provider credential), agent_token, access_token, session_token — and the cost
|
||||
# of over-redacting a hypothetical innocent one is a blanked field, while the
|
||||
# cost of under-redacting is a live credential sitting in an audit table.
|
||||
REDACT_CONTAINS = {
|
||||
"password", "passwordhash", "secret", "apisecret", "clientsecret",
|
||||
"token", "accesstoken", "refreshtoken", "mfatoken", "resettoken",
|
||||
"sessiontoken", "apitoken", "agenttoken", "csrftoken",
|
||||
"apikey", "xapikey", "privatekey", "publicprivate", "jwkprivatekey",
|
||||
"certprivatekey", "csrprivatekey", "keypem", "privkey",
|
||||
"hmac", "eabhmackey", "eabkid",
|
||||
"credentialsencrypted", "encryptedcredentials", "dnscredentials",
|
||||
"authorization", "cookie", "setcookie", "keyauthorization",
|
||||
"backupcode", "backupcodes", "totp", "totpcode", "totpsecret",
|
||||
"replaynonce", "sessionid", "statspassword", "encryptionkey",
|
||||
"bearer", "signature",
|
||||
}
|
||||
|
||||
_NORMALIZE_RE = re.compile(r"[^a-z0-9]")
|
||||
|
||||
# Value-shaped guards — these fire regardless of the key name.
|
||||
_PEM_RE = re.compile(r"-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----")
|
||||
_JWT_RE = re.compile(r"^[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}$")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Secrets embedded INSIDE a value, not carried as their own field.
|
||||
# ---------------------------------------------------------------------------
|
||||
# Key-name and whole-value matching both miss the biggest source of secrets in
|
||||
# this system: a rendered config file handed around as one long string under an
|
||||
# innocent key.
|
||||
#
|
||||
# GET /api/agents/{n}/keepalived-config -> keepalived.config_content
|
||||
# POST /api/agents/{n}/keepalived-discovery -> config_content
|
||||
#
|
||||
# Both carry a full keepalived.conf whose `auth_pass <secret>` line is the VRRP
|
||||
# password in cleartext, and the delivery endpoint is polled every ~2.5 minutes
|
||||
# per member node. routers/vip.py states the rule this restores: "the secret
|
||||
# never leaves the server in cleartext ... only the at-rest Fernet token and the
|
||||
# agent-delivery endpoint ever see the real value" — and routers/agent.py's
|
||||
# discovery handler already masks the very same text with the very same pattern
|
||||
# before storing it in `vip_discoveries.raw_config_masked`. Capturing the
|
||||
# unmasked original into request_logs would sit that plaintext right next to the
|
||||
# masked copy the codebase went to three separate lengths to produce.
|
||||
#
|
||||
# ONE compiled alternation, applied in a single pass over each string value:
|
||||
# `re.sub` with no match costs one scan, whereas a pre-filter plus N patterns
|
||||
# costs a scan each. Mask the WHOLE remainder of the line (vip.py's reasoning),
|
||||
# so a secret containing whitespace cannot partially leak.
|
||||
#
|
||||
# The `uri` branch is the same class of miss in a different shape. Redacting a
|
||||
# key called `secret` does nothing when the SAME secret is also handed back
|
||||
# inside a URI under a key called `otpauth_uri`:
|
||||
#
|
||||
# POST /api/mfa/enroll -> {"secret": "***REDACTED***",
|
||||
# "otpauth_uri": "otpauth://totp/X?secret=JBSWY3DP..."}
|
||||
#
|
||||
# routers/mfa.py's own activity log says "NEVER log the secret itself" and
|
||||
# records only `secret_len`. scrub_query_string already knows `secret` is a
|
||||
# credential; it was simply never pointed at query strings that arrive inside a
|
||||
# body value rather than on the request line. Any URI in any captured string is
|
||||
# now run through scrub_url, which also strips userinfo (`https://u:p@host`)
|
||||
# and drops the fragment.
|
||||
#
|
||||
# The `statsauth` / `userlist` branches cover the same shape for HAProxy. This
|
||||
# application never RENDERS a credential into a haproxy.cfg, so nothing we
|
||||
# generate is at risk - but the agent uploads the node's REAL on-disk file:
|
||||
#
|
||||
# linux_install.sh: config_content=$(cat "$config_path")
|
||||
# -> POST /api/configuration/agents/{n}/config-response
|
||||
#
|
||||
# plus `POST /api/config/validate` and the bulk import, which take whatever the
|
||||
# operator pastes. A production haproxy.cfg routinely carries `stats auth
|
||||
# admin:<password>` and a `userlist` block, so the content is credential-bearing
|
||||
# even though our generator's output is not. Patterns are anchored to HAProxy
|
||||
# keyword syntax rather than the bare word "password", so ordinary prose in an
|
||||
# error message ("invalid password format") is left alone.
|
||||
#
|
||||
# `_EOL` instead of a plain `[^\r\n]` / `\S`: a captured body is only decoded
|
||||
# from JSON when it PARSES, and a config upload is routinely larger than the
|
||||
# capture cap, so the common case for exactly these payloads is the truncated
|
||||
# `{"_raw": ...}` fallback - where the line breaks are still the two-character
|
||||
# escape `\n`, not real newlines. Matching to "end of line" without knowing that
|
||||
# makes `auth_pass ...` swallow the entire rest of the string: no leak, but the
|
||||
# whole remainder of the config is masked and the row becomes useless. So every
|
||||
# value here stops at a real newline OR at a literal backslash-n.
|
||||
#
|
||||
# PERFORMANCE. These patterns run on the writer task, on every string value of
|
||||
# every captured body, so their cost is paid per row forever. Measured on an
|
||||
# 8 KB config body, 300 iterations:
|
||||
#
|
||||
# one combined alternation, IGNORECASE, \b-anchored 308.2 us
|
||||
# the same four patterns run separately (sum) 219.7 us
|
||||
# text.lower() once + four substring pre-checks 5.3 us
|
||||
#
|
||||
# Two things make the naive version expensive, and neither is the matching:
|
||||
# `\b` and IGNORECASE both defeat the regex engine's literal-prefix scan, so
|
||||
# every alphanumeric position in 8 KB becomes a candidate start. A body with
|
||||
# none of these keywords - which is almost every body - paid the full 308 us to
|
||||
# find nothing.
|
||||
#
|
||||
# So: split the alternation, and gate each pattern behind a substring test on
|
||||
# one lowercased copy. A `str.lower()` and an `in` are C-level scans; the regex
|
||||
# only runs when its keyword is actually present, and then it runs on text that
|
||||
# genuinely contains it. Cost becomes O(total string bytes) rather than
|
||||
# O(bytes x patterns), and the common case is ~5 us instead of ~308 us.
|
||||
#
|
||||
# The pre-checks are on the LOWERCASED copy, so the patterns must stay
|
||||
# IGNORECASE: the marker may well have been `AUTH_PASS` in the original.
|
||||
_EOL = r"(?:(?!\\n)[^\r\n])"
|
||||
_NON_SPACE = r"(?:(?!\\n)\S)"
|
||||
|
||||
_RE_AUTH_PASS = re.compile(r"(auth_pass[ \t]+)\S" + _EOL + r"*", re.IGNORECASE)
|
||||
# `stats auth <user>:<passwd>` - keep the user, mask the password half, so an
|
||||
# operator can still tell WHICH account a 401 was about.
|
||||
_RE_STATS_AUTH = re.compile(
|
||||
r"(stats[ \t]+auth[ \t]+[^\s:]*:)" + _NON_SPACE + r"+", re.IGNORECASE
|
||||
)
|
||||
# `user <name> password <hash>` / `user <name> insecure-password <plain>`
|
||||
_RE_USERLIST = re.compile(
|
||||
r"(user[ \t]+" + _NON_SPACE + r"+[ \t]+(?:insecure-)?password[ \t]+)" + _NON_SPACE + r"+",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# No IGNORECASE and no `\b`: the character class already covers both cases, and
|
||||
# `://` gives the engine a literal to scan for.
|
||||
_RE_URI = re.compile(r"[a-zA-Z][a-zA-Z0-9+.\-]*://[^\s\"'<>\\]+")
|
||||
|
||||
_EMBEDDED_MASK = "********"
|
||||
|
||||
# Only strings long enough to hold `auth_pass ` plus a value are worth scanning.
|
||||
_EMBEDDED_MIN_LENGTH = 11
|
||||
|
||||
def _keep_prefix(match: "re.Match") -> str:
|
||||
"""Keep group 1 (the directive, and for `stats auth` the account name too),
|
||||
replace the value that follows it."""
|
||||
return f"{match.group(1)}{_EMBEDDED_MASK}"
|
||||
|
||||
|
||||
def _mask_uri(match: "re.Match") -> str:
|
||||
uri = match.group(0)
|
||||
# A URI with neither a query nor userinfo cannot carry a credential in the
|
||||
# place we scrub, and rebuilding it would only risk changing a value for no
|
||||
# benefit.
|
||||
if "?" not in uri and "@" not in uri:
|
||||
return uri
|
||||
try:
|
||||
scrubbed = scrub_url(uri)
|
||||
except Exception: # pragma: no cover - scrub_url already swallows
|
||||
return uri
|
||||
# scrub_url reports its own failure as a placeholder string. Inside a larger
|
||||
# text value that would corrupt the surrounding sentence, so keep the
|
||||
# original: it parsed badly enough that urlsplit found no query to scrub.
|
||||
if not scrubbed or scrubbed == "***URL_PARSE_ERROR***":
|
||||
return uri
|
||||
return scrubbed
|
||||
|
||||
|
||||
def scrub_embedded_secrets(text: str) -> str:
|
||||
"""Mask credentials that live inside a larger text value (config blobs).
|
||||
|
||||
Each pattern is gated behind a substring test on one lowercased copy: see
|
||||
the block comment above for the measurements. Never raises - a scrub
|
||||
failure must not turn into a failed log write, and returning the input
|
||||
unchanged would be the wrong failure direction, so the whole value is
|
||||
replaced instead.
|
||||
"""
|
||||
if not text or len(text) < _EMBEDDED_MIN_LENGTH:
|
||||
return text
|
||||
try:
|
||||
lowered = text.lower()
|
||||
if "auth_pass" in lowered:
|
||||
text = _RE_AUTH_PASS.sub(_keep_prefix, text)
|
||||
# Not "stats auth": the directive may be separated by tabs or by more
|
||||
# than one space, which the pattern allows and a substring test does not.
|
||||
if "stats" in lowered and "auth" in lowered:
|
||||
text = _RE_STATS_AUTH.sub(_keep_prefix, text)
|
||||
if "password" in lowered:
|
||||
text = _RE_USERLIST.sub(_keep_prefix, text)
|
||||
if "://" in text:
|
||||
text = _RE_URI.sub(_mask_uri, text)
|
||||
return text
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"scrub_embedded_secrets() failed, blanking value: {exc}")
|
||||
return REDACTED
|
||||
|
||||
|
||||
_MAX_DEPTH = 6
|
||||
_MAX_NODES = 2000
|
||||
_MAX_STRING = 4096
|
||||
_MAX_LIST_ITEMS = 200
|
||||
|
||||
# Header handling. Allowlist wins; presence-only names are emitted with the
|
||||
# value replaced so the operator knows the header was there.
|
||||
HEADER_ALLOWLIST = {
|
||||
"content-type", "content-length", "content-encoding", "accept",
|
||||
"accept-encoding", "accept-language", "user-agent", "referer", "origin",
|
||||
"host", "connection", "cache-control", "pragma", "date", "server",
|
||||
"x-correlation-id", "x-request-id", "x-response-time",
|
||||
"x-forwarded-for", "x-forwarded-proto", "x-forwarded-host", "x-real-ip",
|
||||
"location", "retry-after", "ratelimit-reset", "ratelimit-remaining",
|
||||
"link", "etag", "vary",
|
||||
}
|
||||
|
||||
HEADER_PRESENCE_ONLY = {
|
||||
"authorization", "cookie", "set-cookie", "x-api-key", "api-key",
|
||||
"proxy-authorization", "replay-nonce", "www-authenticate",
|
||||
"x-auth-token", "x-agent-token", "x-agent-api-key",
|
||||
}
|
||||
|
||||
_MAX_HEADERS = 40
|
||||
|
||||
|
||||
class _NodeBudget:
|
||||
"""Shared mutable counter so a single body can't blow the CPU budget by
|
||||
being wide as well as deep."""
|
||||
|
||||
__slots__ = ("remaining",)
|
||||
|
||||
def __init__(self, remaining: int = _MAX_NODES):
|
||||
self.remaining = remaining
|
||||
|
||||
def spend(self) -> int:
|
||||
self.remaining -= 1
|
||||
return self.remaining
|
||||
|
||||
|
||||
def _normalize_key(key: Any) -> str:
|
||||
try:
|
||||
return _NORMALIZE_RE.sub("", str(key).lower())
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def is_secret_key(key: Any) -> bool:
|
||||
"""True when a dict key / query param name names a secret."""
|
||||
norm = _normalize_key(key)
|
||||
if not norm:
|
||||
return False
|
||||
if norm in REDACT_EXACT:
|
||||
return True
|
||||
return any(needle in norm for needle in REDACT_CONTAINS)
|
||||
|
||||
|
||||
def _is_secret_value(value: str) -> bool:
|
||||
"""Shape-based guard for secrets that arrive under an innocent key."""
|
||||
if len(value) < 32:
|
||||
# Neither a PEM block nor a JWT fits in less than this; skip the
|
||||
# regex work on the overwhelmingly common short-string case.
|
||||
return False
|
||||
if _PEM_RE.search(value):
|
||||
return True
|
||||
return bool(_JWT_RE.match(value.strip()))
|
||||
|
||||
|
||||
def redact(value: Any, *, depth: int = 0, budget: Optional[_NodeBudget] = None) -> Any:
|
||||
"""Recursively redact a decoded body.
|
||||
|
||||
Depth- and node-capped so a hostile or merely pathological payload cannot
|
||||
burn CPU on the writer task. Never raises.
|
||||
"""
|
||||
if budget is None:
|
||||
budget = _NodeBudget()
|
||||
|
||||
try:
|
||||
if depth > _MAX_DEPTH:
|
||||
return "***DEPTH_LIMIT***"
|
||||
|
||||
if isinstance(value, dict):
|
||||
out: Dict[str, Any] = {}
|
||||
for k, v in value.items():
|
||||
if budget.spend() <= 0:
|
||||
out["_node_limit"] = True
|
||||
break
|
||||
if is_secret_key(k):
|
||||
out[str(k)] = REDACTED
|
||||
else:
|
||||
out[str(k)] = redact(v, depth=depth + 1, budget=budget)
|
||||
return out
|
||||
|
||||
if isinstance(value, (list, tuple)):
|
||||
out_list = []
|
||||
for item in list(value)[:_MAX_LIST_ITEMS]:
|
||||
if budget.spend() <= 0:
|
||||
out_list.append("_node_limit")
|
||||
break
|
||||
out_list.append(redact(item, depth=depth + 1, budget=budget))
|
||||
if len(value) > _MAX_LIST_ITEMS:
|
||||
out_list.append(f"…[{len(value) - _MAX_LIST_ITEMS} more items]")
|
||||
return out_list
|
||||
|
||||
if isinstance(value, str):
|
||||
if _is_secret_value(value):
|
||||
return REDACTED
|
||||
# Scrub BEFORE truncating. Truncation is not a security control:
|
||||
# `auth_pass` sits inside the first 400 bytes of a rendered
|
||||
# keepalived.conf, well under _MAX_STRING, so relying on the cut to
|
||||
# drop it would be relying on luck about where the secret happens
|
||||
# to fall in the file.
|
||||
value = scrub_embedded_secrets(value)
|
||||
if len(value) > _MAX_STRING:
|
||||
return value[:_MAX_STRING] + f"…[truncated {len(value) - _MAX_STRING} chars]"
|
||||
return value
|
||||
|
||||
return value
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"redact() failed, substituting placeholder: {exc}")
|
||||
return "***REDACTION_ERROR***"
|
||||
|
||||
|
||||
def redact_headers(headers: Optional[Dict[str, str]]) -> Optional[Dict[str, str]]:
|
||||
"""Allowlist-filter a header mapping.
|
||||
|
||||
Allowlisted headers keep their value, `HEADER_PRESENCE_ONLY` headers keep
|
||||
only the fact they were present, everything else is dropped silently.
|
||||
"""
|
||||
if not headers:
|
||||
return None
|
||||
try:
|
||||
out: Dict[str, str] = {}
|
||||
for raw_name, raw_value in headers.items():
|
||||
name = str(raw_name).lower()
|
||||
if name in HEADER_PRESENCE_ONLY:
|
||||
out[name] = REDACTED
|
||||
elif name in HEADER_ALLOWLIST:
|
||||
value = str(raw_value)
|
||||
out[name] = value[:1024]
|
||||
if len(out) >= _MAX_HEADERS:
|
||||
break
|
||||
return out or None
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"redact_headers() failed: {exc}")
|
||||
return None
|
||||
|
||||
|
||||
def scrub_query_string(query: Optional[str]) -> Tuple[str, Optional[Dict[str, str]]]:
|
||||
"""Return (scrubbed_query_string, scrubbed_dict) for a raw query string."""
|
||||
if not query:
|
||||
return "", None
|
||||
try:
|
||||
pairs = urllib.parse.parse_qsl(query, keep_blank_values=True)
|
||||
scrubbed = [(k, REDACTED if is_secret_key(k) else v) for k, v in pairs]
|
||||
return urllib.parse.urlencode(scrubbed), dict(scrubbed)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"scrub_query_string() failed: {exc}")
|
||||
return "", None
|
||||
|
||||
|
||||
def scrub_url(url: str) -> str:
|
||||
"""Strip userinfo and scrub the query string of an absolute URL.
|
||||
|
||||
`https://user:pass@api.example.com/v1?api_key=x`
|
||||
→ `https://api.example.com/v1?api_key=***REDACTED***`
|
||||
"""
|
||||
if not url:
|
||||
return ""
|
||||
try:
|
||||
parts = urllib.parse.urlsplit(url)
|
||||
netloc = parts.hostname or ""
|
||||
if parts.port:
|
||||
netloc = f"{netloc}:{parts.port}"
|
||||
query, _ = scrub_query_string(parts.query)
|
||||
# Fragments are dropped: they never reach a server and can carry tokens.
|
||||
return urllib.parse.urlunsplit((parts.scheme, netloc, parts.path, query, ""))
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug(f"scrub_url() failed: {exc}")
|
||||
return "***URL_PARSE_ERROR***"
|
||||
|
||||
|
||||
# Content types whose bodies are worth buffering. Anything else (octet-stream,
|
||||
# images, text/event-stream) is size-counted but never copied, which is what
|
||||
# keeps streaming and file responses safe.
|
||||
CAPTURABLE_CONTENT_TYPES = (
|
||||
"application/json",
|
||||
"application/problem+json",
|
||||
"application/jose+json",
|
||||
"application/x-www-form-urlencoded",
|
||||
"text/plain",
|
||||
"text/html",
|
||||
"text/xml",
|
||||
"application/xml",
|
||||
)
|
||||
|
||||
|
||||
def is_capturable_content_type(content_type: Optional[str]) -> bool:
|
||||
if not content_type:
|
||||
# No Content-Type on a body-bearing message is rare; assume JSON-ish
|
||||
# rather than dropping the one field the operator wanted to see.
|
||||
return True
|
||||
ct = content_type.split(";")[0].strip().lower()
|
||||
return any(ct.startswith(prefix) for prefix in CAPTURABLE_CONTENT_TYPES)
|
||||
|
||||
|
||||
def decode_body(
|
||||
raw: Optional[bytes],
|
||||
content_type: Optional[str],
|
||||
total_bytes: int = 0,
|
||||
) -> Tuple[Optional[Any], bool]:
|
||||
"""Decode + redact a captured body fragment.
|
||||
|
||||
`raw` is what the middleware managed to buffer (already capped);
|
||||
`total_bytes` is how large the body actually was on the wire. Returns
|
||||
`(jsonb_value, truncated)`. Non-JSON payloads are wrapped as
|
||||
`{"_raw": "..."}` so the column stays a uniform JSONB object that the
|
||||
detail view and any future `->>` query can rely on.
|
||||
"""
|
||||
if not raw:
|
||||
return None, False
|
||||
|
||||
truncated = total_bytes > len(raw)
|
||||
ct = (content_type or "").split(";")[0].strip().lower()
|
||||
|
||||
try:
|
||||
text = raw.decode("utf-8", "replace")
|
||||
except Exception: # pragma: no cover - decode with 'replace' can't raise
|
||||
return {"_raw": "***DECODE_ERROR***"}, truncated
|
||||
|
||||
value: Any
|
||||
if ct in ("application/json", "application/problem+json", "application/jose+json") or (
|
||||
not ct and text[:1] in ("{", "[")
|
||||
):
|
||||
try:
|
||||
value = redact(json.loads(text))
|
||||
except Exception:
|
||||
# A truncated JSON body will not parse — keep the raw prefix so the
|
||||
# operator still sees what was sent.
|
||||
value = {"_raw": redact(text)}
|
||||
elif ct == "application/x-www-form-urlencoded":
|
||||
try:
|
||||
value = redact(dict(urllib.parse.parse_qsl(text, keep_blank_values=True)))
|
||||
except Exception:
|
||||
value = {"_raw": redact(text)}
|
||||
else:
|
||||
value = {"_raw": redact(text)}
|
||||
|
||||
if truncated:
|
||||
if isinstance(value, dict):
|
||||
value["_truncated"] = True
|
||||
value["_original_bytes"] = total_bytes
|
||||
else:
|
||||
value = {
|
||||
"_value": value,
|
||||
"_truncated": True,
|
||||
"_original_bytes": total_bytes,
|
||||
}
|
||||
|
||||
return value, truncated
|
||||
|
||||
|
||||
def safe_error_text(exc: BaseException, *, type_only: bool = False, limit: int = 2000) -> str:
|
||||
"""Render an exception for the `error` column.
|
||||
|
||||
`type_only=True` is used for the DNS providers, whose own error paths
|
||||
deliberately never surface `str(exc)` — it can carry the request URL and,
|
||||
through it, tenant/zone identifiers (see services/dns_providers/*.py).
|
||||
"""
|
||||
try:
|
||||
name = type(exc).__name__
|
||||
if type_only:
|
||||
return name
|
||||
text = f"{name}: {exc}"
|
||||
redacted = redact(text)
|
||||
if not isinstance(redacted, str):
|
||||
return name
|
||||
return redacted[:limit]
|
||||
except Exception: # pragma: no cover - defensive
|
||||
return "UnknownError"
|
||||
@@ -0,0 +1,272 @@
|
||||
"""v1.11.0 — operator-tunable settings for the request/response log.
|
||||
|
||||
The middleware runs on EVERY request, so the hot path must not touch the
|
||||
database. `get_config()` returns a module-global immutable snapshot with no
|
||||
`await`; `refresh_config()` reloads it from `system_settings` and is called
|
||||
|
||||
* once at startup, right after migrations,
|
||||
* every `_TTL_SECONDS` from the sink's writer loop (off the request path),
|
||||
* synchronously at the end of `PUT /api/request-logs/settings`, so an
|
||||
operator's change takes effect immediately instead of up to 30s later.
|
||||
|
||||
asyncpg has no JSONB codec registered on this pool (see
|
||||
database/connection.py), so every value comes back as a raw JSON *string* and
|
||||
needs the `isinstance(v, str)` + `json.loads` guard used elsewhere in this
|
||||
codebase (services/acme_service.py, utils/activity_log.py).
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
SETTINGS_CATEGORY = "requestlog"
|
||||
|
||||
# Kept in sync with the seed in database/migrations.ensure_request_log_settings().
|
||||
# backend/tests/test_request_log_settings.py asserts the two agree, so a change
|
||||
# here without a change there fails the suite rather than drifting silently.
|
||||
DEFAULT_EXCLUDE_PATHS = (
|
||||
"/api/request-logs",
|
||||
"/api/health",
|
||||
"/api/docs",
|
||||
"/api/redoc",
|
||||
"/api/openapi.json",
|
||||
"/.well-known/acme-challenge",
|
||||
"/api/agents/heartbeat",
|
||||
"/static",
|
||||
"/favicon.ico",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RequestLogConfig:
|
||||
enabled: bool = True
|
||||
capture_inbound: bool = True
|
||||
capture_outbound: bool = True
|
||||
capture_bodies: bool = True
|
||||
capture_get: bool = True
|
||||
# SUCCESSFUL agent polls only. Off by default because the row rate of this
|
||||
# table is otherwise a linear function of fleet size, not of operator
|
||||
# activity: each agent runs a 30s cycle that issues three logged calls
|
||||
# (config, pending-requests, upgrade-status; the heartbeat is already
|
||||
# excluded) plus two more every fifth cycle. Measured, that is ~9 800 rows
|
||||
# per day PER AGENT, so a 200-node fleet writes ~2M rows/day and reaches the
|
||||
# 500 000 max_rows cap in about six hours - at which point the shipped
|
||||
# "7 days of successes, 30 days of failures" is not 7 and 30, it is 0.25.
|
||||
# FAILED agent calls are always kept regardless of this flag: they are the
|
||||
# half an operator actually needs, and they are rare.
|
||||
capture_agent_success: bool = False
|
||||
max_body_bytes: int = 8192
|
||||
sample_rate: float = 1.0
|
||||
exclude_paths: Tuple[str, ...] = DEFAULT_EXCLUDE_PATHS
|
||||
success_retention_days: int = 7
|
||||
error_retention_days: int = 30
|
||||
max_rows: int = 500000
|
||||
prune_interval_minutes: int = 60
|
||||
|
||||
def as_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"enabled": self.enabled,
|
||||
"capture_inbound": self.capture_inbound,
|
||||
"capture_outbound": self.capture_outbound,
|
||||
"capture_bodies": self.capture_bodies,
|
||||
"capture_get": self.capture_get,
|
||||
"capture_agent_success": self.capture_agent_success,
|
||||
"max_body_bytes": self.max_body_bytes,
|
||||
"sample_rate": self.sample_rate,
|
||||
"exclude_paths": list(self.exclude_paths),
|
||||
"success_retention_days": self.success_retention_days,
|
||||
"error_retention_days": self.error_retention_days,
|
||||
"max_rows": self.max_rows,
|
||||
"prune_interval_minutes": self.prune_interval_minutes,
|
||||
}
|
||||
|
||||
|
||||
DEFAULT_CONFIG = RequestLogConfig()
|
||||
|
||||
_CACHE: RequestLogConfig = DEFAULT_CONFIG
|
||||
_CACHE_AT: float = 0.0
|
||||
_TTL_SECONDS: float = 30.0
|
||||
|
||||
# Bounds, mirrored by the Pydantic model in routers/request_logs.py. Kept here
|
||||
# too because refresh_config() reads whatever is in the table, which may have
|
||||
# been written by an older build or by hand.
|
||||
_BOUNDS = {
|
||||
"max_body_bytes": (0, 262144),
|
||||
"success_retention_days": (1, 365),
|
||||
"error_retention_days": (1, 365),
|
||||
"max_rows": (1000, 50_000_000),
|
||||
"prune_interval_minutes": (5, 1440),
|
||||
}
|
||||
|
||||
MAX_EXCLUDE_PATHS = 64
|
||||
MAX_EXCLUDE_PATH_LENGTH = 200
|
||||
|
||||
|
||||
def get_config() -> RequestLogConfig:
|
||||
"""Hot-path read: no await, no DB, no lock. Returns the last snapshot."""
|
||||
return _CACHE
|
||||
|
||||
|
||||
def set_config(config: RequestLogConfig) -> None:
|
||||
"""Replace the snapshot directly. Used by the settings PUT handler (which
|
||||
already has the validated values) and by tests."""
|
||||
global _CACHE, _CACHE_AT
|
||||
_CACHE = config
|
||||
_CACHE_AT = time.monotonic()
|
||||
|
||||
|
||||
def _clamp_int(raw: Any, field: str, fallback: int) -> int:
|
||||
try:
|
||||
value = int(raw)
|
||||
except (TypeError, ValueError):
|
||||
return fallback
|
||||
low, high = _BOUNDS[field]
|
||||
return max(low, min(high, value))
|
||||
|
||||
|
||||
def _clamp_float(raw: Any, fallback: float, low: float, high: float) -> float:
|
||||
try:
|
||||
value = float(raw)
|
||||
except (TypeError, ValueError):
|
||||
return fallback
|
||||
return max(low, min(high, value))
|
||||
|
||||
|
||||
def _as_bool(raw: Any, fallback: bool) -> bool:
|
||||
if isinstance(raw, bool):
|
||||
return raw
|
||||
if isinstance(raw, (int, float)):
|
||||
return bool(raw)
|
||||
if isinstance(raw, str):
|
||||
lowered = raw.strip().lower()
|
||||
if lowered in ("true", "1", "yes", "on"):
|
||||
return True
|
||||
if lowered in ("false", "0", "no", "off"):
|
||||
return False
|
||||
return fallback
|
||||
|
||||
|
||||
def normalize_exclude_paths(raw: Any, fallback: Tuple[str, ...]) -> Tuple[str, ...]:
|
||||
"""Coerce whatever is stored into a bounded tuple of path prefixes."""
|
||||
if not isinstance(raw, (list, tuple)):
|
||||
return fallback
|
||||
out = []
|
||||
for item in raw:
|
||||
if not isinstance(item, str):
|
||||
continue
|
||||
candidate = item.strip()
|
||||
if not candidate.startswith("/") or len(candidate) > MAX_EXCLUDE_PATH_LENGTH:
|
||||
continue
|
||||
out.append(candidate)
|
||||
if len(out) >= MAX_EXCLUDE_PATHS:
|
||||
break
|
||||
return tuple(out) if out else fallback
|
||||
|
||||
|
||||
def config_from_mapping(values: Dict[str, Any], base: Optional[RequestLogConfig] = None) -> RequestLogConfig:
|
||||
"""Build a config from a plain suffix→value mapping, clamping every field.
|
||||
|
||||
Unknown keys are ignored and missing keys keep the value from `base`
|
||||
(default: the shipped defaults), so a partially-seeded table still yields a
|
||||
complete, usable config.
|
||||
"""
|
||||
base = base or DEFAULT_CONFIG
|
||||
return replace(
|
||||
base,
|
||||
enabled=_as_bool(values.get("enabled", base.enabled), base.enabled),
|
||||
capture_inbound=_as_bool(values.get("capture_inbound", base.capture_inbound), base.capture_inbound),
|
||||
capture_outbound=_as_bool(values.get("capture_outbound", base.capture_outbound), base.capture_outbound),
|
||||
capture_bodies=_as_bool(values.get("capture_bodies", base.capture_bodies), base.capture_bodies),
|
||||
capture_get=_as_bool(values.get("capture_get", base.capture_get), base.capture_get),
|
||||
capture_agent_success=_as_bool(
|
||||
values.get("capture_agent_success", base.capture_agent_success),
|
||||
base.capture_agent_success,
|
||||
),
|
||||
max_body_bytes=_clamp_int(values.get("max_body_bytes", base.max_body_bytes), "max_body_bytes", base.max_body_bytes),
|
||||
sample_rate=_clamp_float(values.get("sample_rate", base.sample_rate), base.sample_rate, 0.0, 1.0),
|
||||
exclude_paths=normalize_exclude_paths(values.get("exclude_paths"), base.exclude_paths),
|
||||
success_retention_days=_clamp_int(
|
||||
values.get("success_retention_days", base.success_retention_days),
|
||||
"success_retention_days", base.success_retention_days,
|
||||
),
|
||||
error_retention_days=_clamp_int(
|
||||
values.get("error_retention_days", base.error_retention_days),
|
||||
"error_retention_days", base.error_retention_days,
|
||||
),
|
||||
max_rows=_clamp_int(values.get("max_rows", base.max_rows), "max_rows", base.max_rows),
|
||||
prune_interval_minutes=_clamp_int(
|
||||
values.get("prune_interval_minutes", base.prune_interval_minutes),
|
||||
"prune_interval_minutes", base.prune_interval_minutes,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _decode_setting_value(raw: Any) -> Any:
|
||||
"""JSONB comes back as a raw string on this pool — parse it, but keep the
|
||||
original text if it is not valid JSON (an operator may have hand-written
|
||||
`7` or `seven`)."""
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return raw
|
||||
return raw
|
||||
|
||||
|
||||
async def load_settings_rows(conn) -> Dict[str, Any]:
|
||||
"""Read the `requestlog.*` rows into a suffix→value mapping."""
|
||||
rows = await conn.fetch(
|
||||
"SELECT key, value FROM system_settings WHERE category = $1",
|
||||
SETTINGS_CATEGORY,
|
||||
)
|
||||
values: Dict[str, Any] = {}
|
||||
for row in rows:
|
||||
key = row["key"]
|
||||
suffix = key.split(".", 1)[1] if "." in key else key
|
||||
values[suffix] = _decode_setting_value(row["value"])
|
||||
return values
|
||||
|
||||
|
||||
async def refresh_config(force: bool = True) -> RequestLogConfig:
|
||||
"""Reload the snapshot from the database.
|
||||
|
||||
Never raises and never leaves a half-built config behind: on any failure
|
||||
the previous snapshot is kept, so a transient DB blip cannot silently turn
|
||||
logging off (or on).
|
||||
"""
|
||||
global _CACHE_AT
|
||||
if not force and (time.monotonic() - _CACHE_AT) < _TTL_SECONDS:
|
||||
return _CACHE
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
values = await load_settings_rows(conn)
|
||||
if values:
|
||||
set_config(config_from_mapping(values))
|
||||
else:
|
||||
# Table not seeded yet (fresh install mid-migration) — keep the
|
||||
# in-code defaults but stamp the timestamp so we don't re-query
|
||||
# every tick.
|
||||
_CACHE_AT = time.monotonic()
|
||||
return _CACHE
|
||||
except Exception as exc:
|
||||
logger.debug(f"refresh_config: keeping previous snapshot ({exc})")
|
||||
_CACHE_AT = time.monotonic()
|
||||
return _CACHE
|
||||
finally:
|
||||
if conn is not None:
|
||||
try:
|
||||
await close_database_connection(conn)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
async def maybe_refresh_config() -> RequestLogConfig:
|
||||
"""TTL-gated refresh, called from the sink's writer loop."""
|
||||
return await refresh_config(force=False)
|
||||
@@ -0,0 +1,388 @@
|
||||
"""v1.11.0 — batching writer for the unified request/response log.
|
||||
|
||||
One row per API call is the highest write volume in this system, and the
|
||||
asyncpg pool (min=10/max=50, see database/connection.py) is shared with every
|
||||
request handler and four background loops. Acquiring a connection per logged
|
||||
request would exhaust it under any real load, so instead:
|
||||
|
||||
hot path ──offer(row)──▶ bounded asyncio.Queue ──▶ single writer task
|
||||
(drops when full) (executemany batches)
|
||||
|
||||
The hot path never awaits I/O and never raises. When the queue is full rows are
|
||||
counted as dropped and reported through `GET /api/request-logs/stats`, so a
|
||||
saturated logger is visible rather than silent.
|
||||
|
||||
Redaction deliberately happens HERE, on the writer task, not in the middleware:
|
||||
the recursive walk is the most expensive part of building a row and it has no
|
||||
business running inside the request coroutine.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import random
|
||||
from contextvars import ContextVar
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from config import (
|
||||
REQUEST_LOG_BATCH_SIZE,
|
||||
REQUEST_LOG_FLUSH_MS,
|
||||
REQUEST_LOG_QUEUE_MAX,
|
||||
REQUEST_LOG_QUEUE_MAX_BYTES,
|
||||
)
|
||||
from database.connection import get_database_connection, close_database_connection
|
||||
from utils.request_log_redaction import decode_body, redact_headers
|
||||
from utils.request_log_settings import get_config, maybe_refresh_config
|
||||
|
||||
logger = logging.getLogger("haproxy_openmanager.request_log")
|
||||
|
||||
# Set by the inbound middleware; read by outbound_span so an outbound call
|
||||
# inherits the id of the inbound request that caused it. That is what turns
|
||||
# "operator clicked Issue Certificate" and "we POSTed to Let's Encrypt" into
|
||||
# one readable trace.
|
||||
request_id_context: ContextVar[Optional[str]] = ContextVar(
|
||||
"request_log_request_id", default=None
|
||||
)
|
||||
|
||||
# `target` on an INBOUND row means the same thing it means on an outbound one:
|
||||
# who was on the other end. Set by the middleware when the caller authenticated
|
||||
# with an agent API key rather than a user JWT, which is how agent traffic is
|
||||
# told apart from operator traffic without a database lookup on the hot path.
|
||||
# Deliberately the same literal as http_instrumentation.TARGET_AGENT, so
|
||||
# `target = 'agent'` selects the whole conversation with the fleet in both
|
||||
# directions; `direction` separates them when that matters.
|
||||
TARGET_INBOUND_AGENT = "agent"
|
||||
|
||||
_INSERT_SQL = """
|
||||
INSERT INTO request_logs (
|
||||
request_id, direction, target, method, url, path, query_params,
|
||||
status_code, status_class, duration_ms, user_id, username, client_ip,
|
||||
user_agent, request_headers, request_body, request_body_bytes,
|
||||
response_headers, response_body, response_body_bytes, error, truncated,
|
||||
created_at
|
||||
) VALUES (
|
||||
$1, $2, $3, $4, $5, $6, $7::jsonb,
|
||||
$8, $9, $10, $11, $12, $13::inet,
|
||||
$14, $15::jsonb, $16::jsonb, $17,
|
||||
$18::jsonb, $19::jsonb, $20, $21, $22,
|
||||
$23
|
||||
)
|
||||
"""
|
||||
|
||||
|
||||
def _jsonb(value: Any) -> Optional[str]:
|
||||
"""asyncpg has no JSONB codec on this pool, so JSONB params travel as text
|
||||
and are cast in SQL — the house idiom (utils/activity_log.py)."""
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return json.dumps(value, default=str)
|
||||
except Exception:
|
||||
return json.dumps({"_serialize_error": True})
|
||||
|
||||
|
||||
@dataclass
|
||||
class RequestLogRow:
|
||||
"""One captured exchange, still holding RAW body bytes.
|
||||
|
||||
Decoding and redaction run in `to_params()` on the writer task.
|
||||
"""
|
||||
|
||||
request_id: str
|
||||
direction: str
|
||||
method: str
|
||||
url: str
|
||||
target: Optional[str] = None
|
||||
path: Optional[str] = None
|
||||
query_string: Optional[str] = None
|
||||
query_params: Optional[Dict[str, Any]] = None
|
||||
status_code: Optional[int] = None
|
||||
duration_ms: int = 0
|
||||
user_id: Optional[int] = None
|
||||
username: Optional[str] = None
|
||||
client_ip: Optional[str] = None
|
||||
user_agent: Optional[str] = None
|
||||
request_headers: Optional[Dict[str, str]] = None
|
||||
response_headers: Optional[Dict[str, str]] = None
|
||||
request_body_raw: Optional[bytes] = None
|
||||
request_body_bytes: int = 0
|
||||
request_content_type: Optional[str] = None
|
||||
response_body_raw: Optional[bytes] = None
|
||||
response_body_bytes: int = 0
|
||||
response_content_type: Optional[str] = None
|
||||
# Pre-decoded body override, used by outbound spans that hold a dict/str
|
||||
# rather than wire bytes (e.g. the synthetic JWS summary).
|
||||
request_body_value: Optional[Any] = None
|
||||
response_body_value: Optional[Any] = None
|
||||
error: Optional[str] = None
|
||||
created_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
||||
|
||||
def queue_weight(self) -> int:
|
||||
"""Approximate bytes this row holds while it waits in the queue.
|
||||
|
||||
The buffered bodies are the only part that varies by orders of
|
||||
magnitude (0 to 2 x max_body_bytes); everything else is a handful of
|
||||
short strings and two small header dicts, measured at ~1.4 KB per row.
|
||||
Used to enforce a byte budget alongside the row count, so memory does
|
||||
not become a function of an operator-editable setting.
|
||||
"""
|
||||
weight = 1400
|
||||
if self.request_body_raw:
|
||||
weight += len(self.request_body_raw)
|
||||
if self.response_body_raw:
|
||||
weight += len(self.response_body_raw)
|
||||
return weight
|
||||
|
||||
@property
|
||||
def status_class(self) -> int:
|
||||
"""`status_code // 100`, or 0 when there was no HTTP response at all
|
||||
(transport error / unhandled exception). 0 is what the error-retention
|
||||
prune treats as an error alongside >= 4."""
|
||||
if not self.status_code:
|
||||
return 0
|
||||
return int(self.status_code) // 100
|
||||
|
||||
def to_params(self) -> List[Any]:
|
||||
req_body, req_truncated = (self.request_body_value, False)
|
||||
if req_body is None:
|
||||
req_body, req_truncated = decode_body(
|
||||
self.request_body_raw, self.request_content_type, self.request_body_bytes
|
||||
)
|
||||
|
||||
res_body, res_truncated = (self.response_body_value, False)
|
||||
if res_body is None:
|
||||
res_body, res_truncated = decode_body(
|
||||
self.response_body_raw, self.response_content_type, self.response_body_bytes
|
||||
)
|
||||
|
||||
return [
|
||||
self.request_id[:64],
|
||||
self.direction,
|
||||
self.target[:32] if self.target else None,
|
||||
(self.method or "")[:10],
|
||||
self.url or "",
|
||||
self.path[:512] if self.path else None,
|
||||
_jsonb(self.query_params),
|
||||
self.status_code,
|
||||
self.status_class,
|
||||
max(0, int(self.duration_ms)),
|
||||
self.user_id,
|
||||
self.username[:50] if self.username else None,
|
||||
self.client_ip,
|
||||
self.user_agent[:1024] if self.user_agent else None,
|
||||
_jsonb(redact_headers(self.request_headers)),
|
||||
_jsonb(req_body),
|
||||
max(0, int(self.request_body_bytes)),
|
||||
_jsonb(redact_headers(self.response_headers)),
|
||||
_jsonb(res_body),
|
||||
max(0, int(self.response_body_bytes)),
|
||||
self.error[:4000] if self.error else None,
|
||||
bool(req_truncated or res_truncated),
|
||||
self.created_at,
|
||||
]
|
||||
|
||||
|
||||
class RequestLogSink:
|
||||
"""Bounded queue + single batching writer task (one per uvicorn worker)."""
|
||||
|
||||
def __init__(self, maxsize: int, batch_size: int, flush_ms: int,
|
||||
max_bytes: int = REQUEST_LOG_QUEUE_MAX_BYTES):
|
||||
self._maxsize = maxsize
|
||||
self._max_bytes = max_bytes
|
||||
self._queued_bytes = 0
|
||||
self._batch_size = batch_size
|
||||
self._flush_seconds = flush_ms / 1000.0
|
||||
self._queue: Optional[asyncio.Queue] = None
|
||||
self._dropped = 0
|
||||
self._written = 0
|
||||
self._failed = 0
|
||||
self._running = False
|
||||
|
||||
# -- lifecycle ---------------------------------------------------------
|
||||
|
||||
def _ensure_queue(self) -> asyncio.Queue:
|
||||
# Created lazily so importing this module never needs a running loop
|
||||
# (matters for the test suite, which imports main.py without one).
|
||||
if self._queue is None:
|
||||
self._queue = asyncio.Queue(maxsize=self._maxsize)
|
||||
return self._queue
|
||||
|
||||
@property
|
||||
def stats(self) -> Dict[str, int]:
|
||||
return {
|
||||
"queued": self._queue.qsize() if self._queue is not None else 0,
|
||||
"queue_capacity": self._maxsize,
|
||||
"queued_bytes": self._queued_bytes,
|
||||
"queue_capacity_bytes": self._max_bytes,
|
||||
"written": self._written,
|
||||
"dropped": self._dropped,
|
||||
"failed_batches": self._failed,
|
||||
"running": 1 if self._running else 0,
|
||||
}
|
||||
|
||||
# -- producer side (hot path) -----------------------------------------
|
||||
|
||||
def offer(self, row: RequestLogRow) -> None:
|
||||
"""Enqueue a row. NEVER blocks, NEVER raises.
|
||||
|
||||
Sampling is applied here rather than in the middleware so both
|
||||
directions go through one policy: successful *inbound* traffic can be
|
||||
sampled down, errors never are.
|
||||
"""
|
||||
try:
|
||||
cfg = get_config()
|
||||
if not cfg.enabled:
|
||||
return
|
||||
if row.direction == "inbound" and not cfg.capture_inbound:
|
||||
return
|
||||
if row.direction == "outbound" and not cfg.capture_outbound:
|
||||
return
|
||||
# Fleet-scale gate, and the reason this table's size is a function of
|
||||
# operator activity rather than of node count. An agent's 30s cycle
|
||||
# issues three logged calls, plus two more every fifth cycle: ~9 800
|
||||
# rows/day PER AGENT, all of them 200s saying "nothing changed". At a
|
||||
# few hundred nodes that is millions of rows a day, and the row cap is
|
||||
# then reached in hours, which silently shortens the configured
|
||||
# retention for EVERYTHING else in the table - including the failures
|
||||
# the log exists for. Successes are dropped, failures never are.
|
||||
if (
|
||||
row.direction == "inbound"
|
||||
and row.target == TARGET_INBOUND_AGENT
|
||||
and not cfg.capture_agent_success
|
||||
and row.status_class in (1, 2, 3)
|
||||
):
|
||||
return
|
||||
if (
|
||||
row.direction == "inbound"
|
||||
and cfg.sample_rate < 1.0
|
||||
and row.status_class in (1, 2, 3)
|
||||
and random.random() > cfg.sample_rate
|
||||
):
|
||||
return
|
||||
if not cfg.capture_bodies:
|
||||
row.request_body_raw = None
|
||||
row.response_body_raw = None
|
||||
row.request_body_value = None
|
||||
row.response_body_value = None
|
||||
|
||||
# Byte budget, checked BEFORE the row count. The count alone does
|
||||
# not bound memory: how much a row weighs is an operator setting,
|
||||
# and `max_body_bytes` at its documented 256 KB ceiling puts the
|
||||
# default 2 000-row queue at ~1 GiB, which is the whole pod limit.
|
||||
# Dropping here is the same visible, counted drop as a full queue.
|
||||
weight = row.queue_weight()
|
||||
if self._queued_bytes + weight > self._max_bytes:
|
||||
raise asyncio.QueueFull
|
||||
|
||||
self._ensure_queue().put_nowait(row)
|
||||
self._queued_bytes += weight
|
||||
except asyncio.QueueFull:
|
||||
self._dropped += 1
|
||||
if self._dropped % 500 == 1:
|
||||
logger.warning(
|
||||
f"request_log: queue full, {self._dropped} row(s) dropped so far "
|
||||
f"({self._queue.qsize() if self._queue is not None else 0}/{self._maxsize} rows, "
|
||||
f"{self._queued_bytes // 1024} KiB/{self._max_bytes // 1024} KiB). "
|
||||
f"Lower requestlog.max_body_bytes or requestlog.sample_rate. "
|
||||
f"Raising REQUEST_LOG_QUEUE_MAX also raises the memory this "
|
||||
f"worker can hold, so raise REQUEST_LOG_QUEUE_MAX_BYTES with it."
|
||||
)
|
||||
except Exception as exc:
|
||||
# Instrumentation must never break the thing it instruments.
|
||||
logger.debug(f"request_log: offer() failed: {exc}")
|
||||
|
||||
# -- consumer side (writer task) --------------------------------------
|
||||
|
||||
async def _collect(self) -> List[RequestLogRow]:
|
||||
"""Wait for at least one row, then drain up to batch_size or flush_ms."""
|
||||
queue = self._ensure_queue()
|
||||
first = await queue.get()
|
||||
self._queued_bytes = max(0, self._queued_bytes - first.queue_weight())
|
||||
batch = [first]
|
||||
loop = asyncio.get_running_loop()
|
||||
deadline = loop.time() + self._flush_seconds
|
||||
while len(batch) < self._batch_size:
|
||||
remaining = deadline - loop.time()
|
||||
if remaining <= 0:
|
||||
break
|
||||
try:
|
||||
nxt = await asyncio.wait_for(queue.get(), timeout=remaining)
|
||||
self._queued_bytes = max(0, self._queued_bytes - nxt.queue_weight())
|
||||
batch.append(nxt)
|
||||
except asyncio.TimeoutError:
|
||||
break
|
||||
return batch
|
||||
|
||||
async def _write(self, batch: List[RequestLogRow]) -> None:
|
||||
if not batch:
|
||||
return
|
||||
params = []
|
||||
for row in batch:
|
||||
try:
|
||||
params.append(row.to_params())
|
||||
except Exception as exc:
|
||||
logger.debug(f"request_log: row serialization failed, skipped: {exc}")
|
||||
if not params:
|
||||
return
|
||||
|
||||
conn = None
|
||||
try:
|
||||
conn = await get_database_connection()
|
||||
await conn.executemany(_INSERT_SQL, params)
|
||||
self._written += len(params)
|
||||
except Exception as exc:
|
||||
self._failed += 1
|
||||
# A missing table (pre-migration) or a transient pool error must not
|
||||
# take the writer loop down — drop the batch and carry on.
|
||||
logger.warning(f"request_log: batch write failed ({len(params)} rows): {exc}")
|
||||
finally:
|
||||
if conn is not None:
|
||||
try:
|
||||
await close_database_connection(conn)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
async def run(self) -> None:
|
||||
"""Writer loop. Started once per worker from startup_event()."""
|
||||
self._running = True
|
||||
logger.info(
|
||||
f"request_log sink started (queue={self._maxsize}, batch={self._batch_size}, "
|
||||
f"flush={int(self._flush_seconds * 1000)}ms)"
|
||||
)
|
||||
try:
|
||||
while True:
|
||||
try:
|
||||
batch = await self._collect()
|
||||
await self._write(batch)
|
||||
await maybe_refresh_config()
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.error(f"request_log sink loop error: {exc}")
|
||||
await asyncio.sleep(1)
|
||||
finally:
|
||||
self._running = False
|
||||
|
||||
async def flush(self, timeout: float = 3.0) -> int:
|
||||
"""Drain and persist whatever is queued. Called on shutdown."""
|
||||
queue = self._queue
|
||||
if queue is None or queue.empty():
|
||||
return 0
|
||||
written = 0
|
||||
loop = asyncio.get_running_loop()
|
||||
deadline = loop.time() + timeout
|
||||
while not queue.empty() and loop.time() < deadline:
|
||||
batch: List[RequestLogRow] = []
|
||||
while not queue.empty() and len(batch) < self._batch_size:
|
||||
row = queue.get_nowait()
|
||||
self._queued_bytes = max(0, self._queued_bytes - row.queue_weight())
|
||||
batch.append(row)
|
||||
await self._write(batch)
|
||||
written += len(batch)
|
||||
return written
|
||||
|
||||
|
||||
request_log_sink = RequestLogSink(
|
||||
REQUEST_LOG_QUEUE_MAX, REQUEST_LOG_BATCH_SIZE, REQUEST_LOG_FLUSH_MS
|
||||
)
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": "1.10.0",
|
||||
"releaseName": "GoDaddy DNS provider for ACME DNS-01",
|
||||
"releaseDate": "2026-08-07"
|
||||
"version": "1.11.1",
|
||||
"releaseName": "Adoption cannot hide a node; Config Import on fresh installs",
|
||||
"releaseDate": "2026-08-15"
|
||||
}
|
||||
|
||||
+11
-2
@@ -49,13 +49,22 @@ services:
|
||||
- SECRET_KEY=your-secret-key-change-this-in-production
|
||||
- DEBUG=False
|
||||
- LOG_LEVEL=INFO
|
||||
- PUBLIC_URL=http://localhost:8080
|
||||
- MANAGEMENT_BASE_URL=http://localhost:8080
|
||||
# Interpolated, not hardcoded: these were literals, so a value set in the
|
||||
# host environment or .env was silently ignored and every install kept the
|
||||
# localhost default. That default is also what the ACME challenge backend
|
||||
# falls back to, and HAProxy resolves it ON THE HAPROXY NODE — so on any
|
||||
# deployment where HAProxy is not this machine, it points at the wrong box.
|
||||
- PUBLIC_URL=${PUBLIC_URL:-http://localhost:8080}
|
||||
- MANAGEMENT_BASE_URL=${MANAGEMENT_BASE_URL:-http://localhost:8080}
|
||||
# Empty when unset on the host: the image CMD then falls back to
|
||||
# WEB_CONCURRENCY (uvicorn's native env) and finally to 1.
|
||||
- UVICORN_WORKERS=${UVICORN_WORKERS:-}
|
||||
volumes:
|
||||
- haproxy_configs:/etc/haproxy
|
||||
# NOT published to the host: the API is reachable only through the nginx
|
||||
# service (host :8080). Host port 8000 is therefore NOT this API — on a box
|
||||
# running Portainer it is Portainer's edge tunnel, which answers 404 and looks
|
||||
# deceptively like a working challenge endpoint.
|
||||
expose:
|
||||
- "8000"
|
||||
depends_on:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "haproxy-openmanager-frontend",
|
||||
"version": "1.10.0",
|
||||
"version": "1.11.1",
|
||||
"description": "HAProxy Load Balancer Management UI",
|
||||
"license": "AGPL-3.0-or-later",
|
||||
"dependencies": {
|
||||
|
||||
@@ -209,4 +209,24 @@
|
||||
|
||||
[data-theme='dark'] .agent-offline:hover > td {
|
||||
background-color: #321518 !important;
|
||||
}
|
||||
|
||||
/* v1.11.0 — Request Log: failed exchanges (4xx/5xx and transport errors) are
|
||||
tinted so a page of traffic reads at a glance. Defined for BOTH themes here
|
||||
rather than as an inline style, so the dark variant is not forgotten (the
|
||||
v1.10.2 regression class). */
|
||||
.request-log-error-row > td {
|
||||
background-color: #fff2f0;
|
||||
}
|
||||
|
||||
.request-log-error-row:hover > td {
|
||||
background-color: #ffe7e5 !important;
|
||||
}
|
||||
|
||||
[data-theme='dark'] .request-log-error-row > td {
|
||||
background-color: #2a1215 !important;
|
||||
}
|
||||
|
||||
[data-theme='dark'] .request-log-error-row:hover > td {
|
||||
background-color: #321518 !important;
|
||||
}
|
||||
+41
-6
@@ -23,7 +23,8 @@ import {
|
||||
BulbOutlined,
|
||||
BulbFilled,
|
||||
PlusOutlined,
|
||||
ThunderboltOutlined
|
||||
ThunderboltOutlined,
|
||||
FileSearchOutlined
|
||||
} from '@ant-design/icons';
|
||||
|
||||
import Dashboard from './components/DashboardV2';
|
||||
@@ -49,6 +50,7 @@ import APIDocumentation from './components/APIDocumentation';
|
||||
import IPInventory from './components/IPInventory';
|
||||
import SiteWizard from './components/SiteWizard';
|
||||
import SiteDrafts from './components/SiteDrafts';
|
||||
import RequestLog from './components/RequestLog'; // v1.11.0 — request/response log
|
||||
import { AuthProvider, useAuth } from './contexts/AuthContext';
|
||||
import { ClusterProvider } from './contexts/ClusterContext';
|
||||
import { ThemeProvider, useTheme } from './contexts/ThemeContext';
|
||||
@@ -183,6 +185,18 @@ const { Text } = Typography;
|
||||
icon: <SecurityScanOutlined />,
|
||||
label: <Link to="/security">Security</Link>,
|
||||
},
|
||||
// v1.11.0 — sits directly above Settings on purpose: the Request Log page
|
||||
// and its retention/capture policy (Settings -> Request Log) are the two
|
||||
// halves of one feature, so they are adjacent in the sidebar.
|
||||
{
|
||||
key: '/request-log',
|
||||
icon: <FileSearchOutlined />,
|
||||
label: (
|
||||
<Tooltip placement="right" title="Request Log — every API call in, and every HTTP call this backend made out (ACME, DNS, agents)">
|
||||
<Link to="/request-log">Request Log</Link>
|
||||
</Tooltip>
|
||||
),
|
||||
},
|
||||
{
|
||||
key: '/settings',
|
||||
icon: <SettingOutlined />,
|
||||
@@ -469,6 +483,9 @@ function AppContent() {
|
||||
<Route path="/security" element={<Security />} />
|
||||
<Route path="/agents" element={<AgentManagement />} />
|
||||
<Route path="/configuration" element={<Configuration />} />
|
||||
{/* v1.11.0 — the page gates itself on requestlog.read; the route
|
||||
is unconditional like every other route in this app. */}
|
||||
<Route path="/request-log" element={<RequestLog />} />
|
||||
<Route path="/api-docs" element={<APIDocumentation />} />
|
||||
<Route path="/pools" element={<PoolManagement />} />
|
||||
<Route path="/settings" element={<Settings />} />
|
||||
@@ -484,12 +501,30 @@ function AppContent() {
|
||||
|
||||
function ThemedApp() {
|
||||
const { isDarkMode } = useTheme();
|
||||
const antdTheme = {
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
};
|
||||
|
||||
// Ant Design 5: the STATIC message/notification/Modal.confirm APIs render into their own
|
||||
// detached root, so they do not see this ConfigProvider and always fall back to the light
|
||||
// algorithm — a confirm dialog came up white while the app was in dark mode. `holderRender`
|
||||
// wraps that detached root in the same ConfigProvider, which fixes every static call in the
|
||||
// app at once (12 components use Modal.confirm) instead of migrating each one to
|
||||
// App.useApp(). In an effect rather than during render: ConfigProvider.config() mutates
|
||||
// antd module state, and effects still run long before a user can click anything that opens
|
||||
// a static modal. Re-registered on theme change so the toggle takes effect immediately.
|
||||
React.useEffect(() => {
|
||||
ConfigProvider.config({
|
||||
holderRender: (children) => (
|
||||
<ConfigProvider theme={{ algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm }}>
|
||||
{children}
|
||||
</ConfigProvider>
|
||||
),
|
||||
});
|
||||
}, [isDarkMode]);
|
||||
|
||||
return (
|
||||
<ConfigProvider
|
||||
theme={{
|
||||
algorithm: isDarkMode ? theme.darkAlgorithm : theme.defaultAlgorithm,
|
||||
}}
|
||||
>
|
||||
<ConfigProvider theme={antdTheme}>
|
||||
<AuthProvider>
|
||||
<ClusterProvider>
|
||||
<ProgressProvider>
|
||||
|
||||
@@ -107,8 +107,14 @@ const ACMEAutomation = () => {
|
||||
const [confirming, setConfirming] = useState(false);
|
||||
const regChallengeType = Form.useWatch('challenge_type', registerForm);
|
||||
const regDnsProvider = Form.useWatch('dns_provider', registerForm);
|
||||
const wizardAccountId = Form.useWatch('account_id', wizardForm);
|
||||
const wizardDomains = Form.useWatch('domains', wizardForm);
|
||||
// `preserve: true` is load-bearing, not a nicety. Each wizard step renders only its own fields —
|
||||
// the domain list on Domains, the account Select on Configuration — and a plain useWatch reports
|
||||
// only fields that are currently REGISTERED, so every one of these read `undefined` from the
|
||||
// Review step onward even though the values were still in the form store. That is what made
|
||||
// Review (and the request it submits) silently fall back to the default ACME account, and it
|
||||
// disabled the wildcard guard at exactly the step where Submit lives.
|
||||
const wizardAccountId = Form.useWatch('account_id', { form: wizardForm, preserve: true });
|
||||
const wizardDomains = Form.useWatch('domains', { form: wizardForm, preserve: true });
|
||||
const selectedDnsProvider = dnsProviders.find(p => p.name === regDnsProvider) || null;
|
||||
|
||||
// Issue #35: per-account DNS credential management (view/replace/clear after creation).
|
||||
@@ -217,7 +223,16 @@ const ACMEAutomation = () => {
|
||||
const pendingOrders = orders.filter(o =>
|
||||
o.status === 'pending' || o.status === 'processing' || o.status === 'ready' || isOrderStuck(o)
|
||||
);
|
||||
const activeAccount = accounts.find(a => a.status === 'valid') || null;
|
||||
// The account a request lands on when it carries no explicit account_id. This MUST match the
|
||||
// backend, which takes `ORDER BY created_at DESC LIMIT 1` (routers/letsencrypt.py). The list
|
||||
// arrives ORDER BY id, so picking the first valid entry would preview the OLDEST account — the
|
||||
// opposite one. With two accounts of different challenge methods that made the wizard describe
|
||||
// DNS-01 while the request would actually have gone to an HTTP-01 account.
|
||||
const activeAccount = accounts.filter(a => a.status === 'valid').reduce((best, a) => {
|
||||
if (!best) return a;
|
||||
const delta = new Date(a.created_at) - new Date(best.created_at);
|
||||
return delta > 0 || (delta === 0 && a.id > best.id) ? a : best;
|
||||
}, null);
|
||||
const acmeAccount = activeAccount || (accounts.length > 0 ? accounts[accounts.length - 1] : null);
|
||||
const acmeEnabledClusters = clusters.filter(c => c.acme_enabled && c.is_active);
|
||||
// Issue #35: the cert wizard adapts to the selected account's challenge method.
|
||||
@@ -260,18 +275,26 @@ const ACMEAutomation = () => {
|
||||
return;
|
||||
}
|
||||
setSubmitting(true);
|
||||
// Resolve the chosen account's challenge method so DNS-01/wildcard requests are explicit.
|
||||
// Use the same resolution as the wizard description (wizardAccount) so what the user reviewed
|
||||
// matches what is sent.
|
||||
const challengeType = wizardAccount?.challenge_type; // 'http-01' | 'dns-01' | undefined
|
||||
// Resolve the account ONCE and derive everything else from that single object. The id and the
|
||||
// challenge method used to come from different places — account_id from the form store,
|
||||
// challenge_type from wizardAccount — so whenever those two disagreed the request asked for
|
||||
// DNS-01 validation on an HTTP-01 account and the backend answered "The selected ACME account
|
||||
// has no DNS provider configured for DNS-01."
|
||||
const selectedAccountId = values.account_id ?? wizardAccount?.id ?? null;
|
||||
const account = accounts.find(a => a.id === selectedAccountId) || wizardAccount || null;
|
||||
const challengeType = account?.challenge_type; // 'http-01' | 'dns-01' | undefined
|
||||
// Manual DNS-01 can't auto-renew (the wizard shows the switch off+disabled). Send false to
|
||||
// match the displayed state rather than relying only on the backend to override it.
|
||||
const autoRenew = wizardDnsManual ? false : (values.auto_renew !== false);
|
||||
const isManualDns01 = challengeType === 'dns-01' && (account?.dns_provider || 'manual') === 'manual';
|
||||
const autoRenew = isManualDns01 ? false : (values.auto_renew !== false);
|
||||
const res = await axios.post('/api/letsencrypt/certificates', {
|
||||
domains: values.domains,
|
||||
cluster_ids: values.cluster_ids || [],
|
||||
auto_renew: autoRenew,
|
||||
account_id: values.account_id || null,
|
||||
// Always explicit: sending the resolved id removes the frontend/backend "default account"
|
||||
// guess, which disagreed (the UI previewed the oldest valid account, the backend used the
|
||||
// newest) and made the Review step describe an account the request never went to.
|
||||
account_id: account?.id ?? null,
|
||||
challenge_type: challengeType || undefined,
|
||||
});
|
||||
message.success(res.data?.message || 'Certificate request submitted');
|
||||
@@ -1092,7 +1115,17 @@ const ACMEAutomation = () => {
|
||||
message="Prerequisite Check"
|
||||
description={
|
||||
<ul style={{ margin: 0, paddingLeft: 20 }}>
|
||||
<li>ACME Account: {activeAccount ? <Tag color="success">Active ({activeAccount.email})</Tag> : <Tag color="error">No active account</Tag>}</li>
|
||||
{/* The account the request will actually use — NOT `activeAccount`, which is only
|
||||
the default and would name a different account whenever the user picked one. */}
|
||||
<li>ACME Account: {wizardAccount
|
||||
? <Tag color={wizardAccount.status === 'valid' ? 'success' : 'error'}>
|
||||
{wizardAccount.email}{wizardAccount.status !== 'valid' ? ` (${wizardAccount.status})` : ''}
|
||||
</Tag>
|
||||
: <Tag color="error">No active account</Tag>}
|
||||
{wizardAccount && accounts.length > 1 && !wizardAccountId && (
|
||||
<Typography.Text type="secondary" style={{ fontSize: 12 }}> (default)</Typography.Text>
|
||||
)}
|
||||
</li>
|
||||
{wizardIsDns01 ? (
|
||||
<>
|
||||
<li>Challenge Method: <Tag>DNS-01</Tag> (TXT record; no port 80 / ACME routing needed)</li>
|
||||
@@ -1324,8 +1357,11 @@ const ACMEAutomation = () => {
|
||||
Next
|
||||
</Button>
|
||||
)}
|
||||
{/* Gate on the account the request will actually use, and on its status: with no valid
|
||||
account wizardAccount falls back to the newest (deactivated) one, and the Select lists
|
||||
deactivated accounts too, so a bare null-check would leave Submit enabled. */}
|
||||
{wizardStep === wizardSteps.length - 1 && (
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={!activeAccount || wizardWildcardBlocked || wizardDns01Disabled || (!wizardIsDns01 && acmeEnabledClusters.length === 0)}>
|
||||
<Button type="primary" onClick={handleRequestCert} loading={submitting} disabled={wizardAccount?.status !== 'valid' || wizardWildcardBlocked || wizardDns01Disabled || (!wizardIsDns01 && acmeEnabledClusters.length === 0)}>
|
||||
Submit Request
|
||||
</Button>
|
||||
)}
|
||||
|
||||
@@ -1092,7 +1092,7 @@ const ApplyManagement = () => {
|
||||
style={{
|
||||
marginBottom: 24,
|
||||
borderRadius: 8,
|
||||
border: '1px solid #ffccc7',
|
||||
border: `1px solid ${token.colorErrorBorder}`,
|
||||
boxShadow: '0 2px 8px rgba(255, 77, 79, 0.15)'
|
||||
}}
|
||||
message={
|
||||
@@ -1348,7 +1348,7 @@ const ApplyManagement = () => {
|
||||
HA / VIP Changes ({pendingChanges.vips.length})
|
||||
</Title>
|
||||
{pendingChanges.vips.map(item => (
|
||||
<div key={`vip-${item.id}`} style={{ display: 'flex', alignItems: 'center', gap: 8, padding: '8px 12px', marginBottom: 6, border: item.pending_delete ? '1px solid #ffccc7' : '1px solid #f0f0f0', borderRadius: 6, background: item.pending_delete ? '#fff1f0' : undefined }}>
|
||||
<div key={`vip-${item.id}`} style={{ display: 'flex', alignItems: 'center', gap: 8, padding: '8px 12px', marginBottom: 6, border: `1px solid ${item.pending_delete ? token.colorErrorBorder : token.colorBorderSecondary}`, borderRadius: 6, background: item.pending_delete ? token.colorErrorBg : undefined }}>
|
||||
<CloudServerOutlined style={{ color: item.pending_delete ? '#cf1322' : '#13c2c2' }} />
|
||||
<span style={{ fontWeight: 500 }}>{item.name}</span>
|
||||
<Tag>{item.virtual_ip}/{item.prefix_length}</Tag>
|
||||
@@ -1374,8 +1374,8 @@ const ApplyManagement = () => {
|
||||
const isEnable = /^cluster-\d+-acme-enable-/.test(v.version_name);
|
||||
return (
|
||||
<div key={v.id} style={{
|
||||
padding: 10, border: '1px dashed #1890ff', borderRadius: 6, marginBottom: 8,
|
||||
display: 'flex', alignItems: 'center', justifyContent: 'space-between', backgroundColor: '#f0f8ff'
|
||||
padding: 10, border: `1px dashed ${token.colorPrimary}`, borderRadius: 6, marginBottom: 8,
|
||||
display: 'flex', alignItems: 'center', justifyContent: 'space-between', backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<span>
|
||||
@@ -1419,7 +1419,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f0f8ff'
|
||||
backgroundColor: token.colorInfoBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1443,7 +1443,7 @@ const ApplyManagement = () => {
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
backgroundColor: '#f6ffed'
|
||||
backgroundColor: token.colorSuccessBg
|
||||
}}>
|
||||
<span style={{ fontFamily: 'monospace' }}>{v.version_name}</span>
|
||||
<Tag color="orange">PENDING</Tag>
|
||||
@@ -1518,9 +1518,13 @@ const ApplyManagement = () => {
|
||||
</Descriptions>
|
||||
</div>
|
||||
|
||||
{/* Pending Versions Section */}
|
||||
{/* Pending Versions Section.
|
||||
Theme tokens, not the light-mode literals #fffbe6/#ffe58f: in dark mode those
|
||||
produced a cream panel with light text on it, so the version name, timestamp
|
||||
and "View Change" link were unreadable. colorWarningBg/Border track the
|
||||
algorithm, so the "pending" tint survives in both themes. */}
|
||||
{pendingVersions.length > 0 && (
|
||||
<div style={{ marginBottom: 24, background: '#fffbe6', borderRadius: 8, border: '1px solid #ffe58f', padding: '16px 16px 8px' }}>
|
||||
<div style={{ marginBottom: 24, background: token.colorWarningBg, borderRadius: 8, border: `1px solid ${token.colorWarningBorder}`, padding: '16px 16px 8px' }}>
|
||||
<Title level={5} style={{ marginTop: 0 }}>
|
||||
<ClockCircleOutlined style={{ marginRight: 8, color: '#faad14' }} />
|
||||
Pending Changes ({pendingVersions.length})
|
||||
@@ -1765,8 +1769,8 @@ const ApplyManagement = () => {
|
||||
<div>
|
||||
{/* Parsed suggestion */}
|
||||
{agentSync.parsed_error?.suggestion && (
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: '#fff7e6', borderRadius: 4, border: '1px solid #ffd591' }}>
|
||||
<Text strong style={{ color: '#ad4e00' }}>Recommendation: </Text>
|
||||
<div style={{ marginBottom: 12, padding: '8px 12px', background: token.colorWarningBg, borderRadius: 4, border: `1px solid ${token.colorWarningBorder}` }}>
|
||||
<Text strong style={{ color: token.colorWarningText }}>Recommendation: </Text>
|
||||
<Text>{agentSync.parsed_error.suggestion}</Text>
|
||||
</div>
|
||||
)}
|
||||
@@ -1951,13 +1955,17 @@ const ApplyManagement = () => {
|
||||
{diffData.changes && diffData.changes.length > 0 ? (
|
||||
diffData.changes.map((change, index) => (
|
||||
<div key={index} style={{ marginBottom: '4px' }}>
|
||||
{/* Token-based, not the light literals #f6ffed/#fff2f0: those stayed
|
||||
near-white in dark mode, so the added/removed rows glared against
|
||||
the dark diff panel around them. colorSuccessBg/colorErrorBg darken
|
||||
with the algorithm while keeping the green/red semantics. */}
|
||||
{change.type === 'added' && (
|
||||
<div style={{ backgroundColor: '#f6ffed', color: '#52c41a', padding: '2px 8px', borderLeft: '3px solid #52c41a' }}>
|
||||
<div style={{ backgroundColor: token.colorSuccessBg, color: token.colorSuccessText, padding: '2px 8px', borderLeft: `3px solid ${token.colorSuccess}` }}>
|
||||
+ {change.line}
|
||||
</div>
|
||||
)}
|
||||
{change.type === 'removed' && (
|
||||
<div style={{ backgroundColor: '#fff2f0', color: '#ff4d4f', padding: '2px 8px', borderLeft: '3px solid #ff4d4f' }}>
|
||||
<div style={{ backgroundColor: token.colorErrorBg, color: token.colorErrorText, padding: '2px 8px', borderLeft: `3px solid ${token.colorError}` }}>
|
||||
- {change.line}
|
||||
</div>
|
||||
)}
|
||||
|
||||
@@ -160,7 +160,8 @@ const ClusterManagement = () => {
|
||||
agent_pool_id: cluster.pool_id || undefined,
|
||||
haproxy_user: cluster.haproxy_user || '',
|
||||
haproxy_group: cluster.haproxy_group || '',
|
||||
acme_enabled: cluster.acme_enabled || false
|
||||
acme_enabled: cluster.acme_enabled || false,
|
||||
acme_backend_url: cluster.acme_backend_url || ''
|
||||
};
|
||||
|
||||
console.log('🔍 CLUSTER EDIT DEBUG - Form values being set:', formValues);
|
||||
@@ -202,7 +203,11 @@ const ClusterManagement = () => {
|
||||
pool_id: values.agent_pool_id || null,
|
||||
haproxy_user: values.haproxy_user || null,
|
||||
haproxy_group: values.haproxy_group || null,
|
||||
acme_enabled: values.acme_enabled || false
|
||||
acme_enabled: values.acme_enabled || false,
|
||||
// Send null, not '', when cleared: null means "inherit the global setting",
|
||||
// and the backend validator treats empty as inherit too. Omitting the key
|
||||
// entirely would make the field look saved while silently discarding it.
|
||||
acme_backend_url: (values.acme_backend_url || '').trim() || null
|
||||
};
|
||||
|
||||
// For new clusters, explicitly set is_active to true
|
||||
@@ -767,6 +772,47 @@ const ClusterManagement = () => {
|
||||
<Switch checkedChildren="Enabled" unCheckedChildren="Disabled" />
|
||||
</Form.Item>
|
||||
|
||||
<Form.Item
|
||||
label="ACME Challenge Backend URL"
|
||||
name="acme_backend_url"
|
||||
tooltip="Where this cluster's HAProxy nodes reach OpenManager to fetch HTTP-01 challenge tokens. Leave empty to use the global setting under Settings > ACME."
|
||||
extra="This address is resolved on the HAProxy node, not here — localhost would mean the HAProxy box itself. Include the port: without one, 8080 is assumed. Example: http://10.90.1.4:80"
|
||||
rules={[
|
||||
{
|
||||
validator: (_, value) => {
|
||||
const v = (value || '').trim();
|
||||
if (!v) return Promise.resolve();
|
||||
if (!/^https?:\/\//i.test(v)) {
|
||||
return Promise.reject(new Error(
|
||||
'Start with http:// or https:// — without a scheme the value cannot be parsed as an address.'
|
||||
));
|
||||
}
|
||||
if (/\s/.test(v)) {
|
||||
return Promise.reject(new Error('Must not contain spaces or line breaks.'));
|
||||
}
|
||||
let parsed;
|
||||
try {
|
||||
parsed = new URL(v);
|
||||
} catch (e) {
|
||||
return Promise.reject(new Error('Not a valid URL.'));
|
||||
}
|
||||
if (parsed.pathname !== '/' && parsed.pathname !== '') {
|
||||
return Promise.reject(new Error('Enter only scheme, host and port — no path.'));
|
||||
}
|
||||
const host = parsed.hostname.replace(/^\[|\]$/g, '').toLowerCase();
|
||||
if (host === 'localhost' || host === '127.0.0.1' || host === '::1' || host === '0.0.0.0') {
|
||||
return Promise.reject(new Error(
|
||||
'Loopback cannot work here: HAProxy resolves this address on the node, so it would mean the node itself. Use the management server\u2019s routable address.'
|
||||
));
|
||||
}
|
||||
return Promise.resolve();
|
||||
}
|
||||
}
|
||||
]}
|
||||
>
|
||||
<Input placeholder="http://10.90.1.4:80" />
|
||||
</Form.Item>
|
||||
|
||||
</Form>
|
||||
</Modal>
|
||||
|
||||
|
||||
@@ -0,0 +1,606 @@
|
||||
/**
|
||||
* v1.11.0 — Request Log.
|
||||
*
|
||||
* One timeline for both directions of HTTP traffic:
|
||||
* - inbound: which user called which API endpoint, with what result
|
||||
* - outbound: which CA / DNS provider / agent this backend called, and what
|
||||
* came back
|
||||
*
|
||||
* Both live in the same table, so opening one inbound request shows the
|
||||
* outbound calls it triggered (they share a request_id) — that "related" list
|
||||
* is the whole point of the page. Bodies are captured redacted and size-capped
|
||||
* by the backend; nothing is unredacted here.
|
||||
*
|
||||
* Pagination is SERVER-side (a first for this frontend — every other table
|
||||
* filters an already-fetched array). The table can hold millions of rows, so
|
||||
* fetching them to slice client-side is not an option.
|
||||
*/
|
||||
import React, { useCallback, useEffect, useMemo, useState } from 'react';
|
||||
import {
|
||||
Alert, Button, Card, DatePicker, Descriptions, Empty, Input, Modal, Select,
|
||||
Space, Spin, Statistic, Switch, Table, Tag, Tooltip, Typography, message, theme
|
||||
} from 'antd';
|
||||
import {
|
||||
ApiOutlined, ClockCircleOutlined, CloudDownloadOutlined, DeleteOutlined,
|
||||
EyeOutlined, ReloadOutlined, UserOutlined, WarningOutlined
|
||||
} from '@ant-design/icons';
|
||||
import axios from 'axios';
|
||||
|
||||
import { extractApiError } from '../utils/apiError';
|
||||
import { useAuth } from '../contexts/AuthContext';
|
||||
|
||||
const { Text, Paragraph } = Typography;
|
||||
|
||||
const DIRECTION_OPTIONS = [
|
||||
{ label: 'Inbound (API calls to us)', value: 'inbound' },
|
||||
{ label: 'Outbound (calls we made)', value: 'outbound' },
|
||||
];
|
||||
|
||||
const STATUS_OPTIONS = [
|
||||
{ label: '2xx Success', value: 2 },
|
||||
{ label: '3xx Redirect', value: 3 },
|
||||
{ label: '4xx Client error', value: 4 },
|
||||
{ label: '5xx Server error', value: 5 },
|
||||
{ label: 'No response (transport error)', value: 0 },
|
||||
];
|
||||
|
||||
const METHOD_OPTIONS = ['GET', 'POST', 'PUT', 'PATCH', 'DELETE', 'HEAD'].map((m) => ({
|
||||
label: m, value: m,
|
||||
}));
|
||||
|
||||
// Mirrors utils/http_instrumentation.py's TARGET_* constants.
|
||||
const TARGET_LABELS = {
|
||||
acme: "ACME / Let's Encrypt",
|
||||
acme_diag: 'ACME diagnostics probe',
|
||||
letsencrypt_ca: "Let's Encrypt CA chain",
|
||||
dns_cloudflare: 'Cloudflare DNS',
|
||||
dns_godaddy: 'GoDaddy DNS',
|
||||
agent: 'HAProxy agent',
|
||||
haproxy_stats: 'HAProxy stats',
|
||||
settings_probe: 'ACME directory probe',
|
||||
};
|
||||
|
||||
const statusColor = (statusClass) => {
|
||||
if (statusClass === 2) return 'green';
|
||||
if (statusClass === 3) return 'blue';
|
||||
if (statusClass === 4) return 'orange';
|
||||
if (statusClass >= 5) return 'red';
|
||||
return 'red';
|
||||
};
|
||||
|
||||
const isFailure = (row) => row?.status_class === 0 || row?.status_class >= 4;
|
||||
|
||||
const formatTime = (value) => {
|
||||
if (!value) return '—';
|
||||
// created_at is TIMESTAMPTZ, so the ISO string already carries an offset —
|
||||
// no manual 'Z' suffix needed here (unlike the naive-TIMESTAMP columns
|
||||
// elsewhere in this app).
|
||||
const d = new Date(value);
|
||||
return Number.isNaN(d.getTime()) ? String(value) : d.toLocaleString();
|
||||
};
|
||||
|
||||
const JsonBlock = ({ value, token }) => {
|
||||
if (value === null || value === undefined) {
|
||||
return <Text type="secondary">Not captured</Text>;
|
||||
}
|
||||
return (
|
||||
<pre
|
||||
style={{
|
||||
fontSize: 11,
|
||||
margin: 0,
|
||||
maxHeight: 260,
|
||||
overflow: 'auto',
|
||||
whiteSpace: 'pre-wrap',
|
||||
wordBreak: 'break-word',
|
||||
background: token.colorFillQuaternary,
|
||||
border: `1px solid ${token.colorBorderSecondary}`,
|
||||
borderRadius: token.borderRadius,
|
||||
padding: 8,
|
||||
}}
|
||||
>
|
||||
{typeof value === 'string' ? value : JSON.stringify(value, null, 2)}
|
||||
</pre>
|
||||
);
|
||||
};
|
||||
|
||||
const RequestLog = () => {
|
||||
const { token } = theme.useToken();
|
||||
const { hasPermission, isAdmin } = useAuth();
|
||||
|
||||
const canRead = hasPermission('requestlog', 'read') || isAdmin();
|
||||
const canManage = hasPermission('requestlog', 'manage') || isAdmin();
|
||||
|
||||
const [rows, setRows] = useState([]);
|
||||
const [total, setTotal] = useState(0);
|
||||
const [totalIsEstimate, setTotalIsEstimate] = useState(false);
|
||||
const [scopedToSelf, setScopedToSelf] = useState(false);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [loadError, setLoadError] = useState(null);
|
||||
|
||||
const [page, setPage] = useState(1);
|
||||
const [pageSize, setPageSize] = useState(50);
|
||||
|
||||
const [direction, setDirection] = useState(undefined);
|
||||
const [statusClass, setStatusClass] = useState(undefined);
|
||||
const [methods, setMethods] = useState([]);
|
||||
const [target, setTarget] = useState(undefined);
|
||||
const [errorsOnly, setErrorsOnly] = useState(false);
|
||||
const [range, setRange] = useState(null);
|
||||
const [searchTyped, setSearchTyped] = useState('');
|
||||
const [search, setSearch] = useState('');
|
||||
|
||||
const [stats, setStats] = useState(null);
|
||||
const [purging, setPurging] = useState(false);
|
||||
|
||||
const [detailOpen, setDetailOpen] = useState(false);
|
||||
const [detailLoading, setDetailLoading] = useState(false);
|
||||
const [detailError, setDetailError] = useState(null);
|
||||
const [detail, setDetail] = useState(null);
|
||||
|
||||
const params = useMemo(() => {
|
||||
const p = { limit: pageSize, offset: (page - 1) * pageSize };
|
||||
if (direction) p.direction = direction;
|
||||
if (statusClass !== undefined && statusClass !== null) p.status_class = statusClass;
|
||||
// The API takes one method; a single selection is the common case and
|
||||
// keeps the query index-friendly.
|
||||
if (methods.length === 1) p.method = methods[0];
|
||||
if (target) p.target = target;
|
||||
if (errorsOnly) p.errors_only = true;
|
||||
if (search) p.q = search;
|
||||
if (range && range[0]) p.since = range[0].toISOString();
|
||||
if (range && range[1]) p.until = range[1].toISOString();
|
||||
return p;
|
||||
}, [page, pageSize, direction, statusClass, methods, target, errorsOnly, search, range]);
|
||||
|
||||
const fetchLogs = useCallback(async () => {
|
||||
if (!canRead) return;
|
||||
setLoading(true);
|
||||
setLoadError(null);
|
||||
try {
|
||||
const res = await axios.get('/api/request-logs', { params });
|
||||
setRows(res.data?.logs || []);
|
||||
setTotal(res.data?.total || 0);
|
||||
setTotalIsEstimate(Boolean(res.data?.total_is_estimate));
|
||||
setScopedToSelf(Boolean(res.data?.scoped_to_self));
|
||||
} catch (err) {
|
||||
const msg = extractApiError(err, 'Failed to load request logs');
|
||||
setLoadError(msg);
|
||||
setRows([]);
|
||||
setTotal(0);
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, [canRead, params]);
|
||||
|
||||
const fetchStats = useCallback(async () => {
|
||||
if (!canRead) return;
|
||||
try {
|
||||
const res = await axios.get('/api/request-logs/stats', { params: { hours: 24 } });
|
||||
setStats(res.data || null);
|
||||
} catch (err) {
|
||||
// Stats are a nice-to-have header; a failure here must not hide the table.
|
||||
setStats(null);
|
||||
}
|
||||
}, [canRead]);
|
||||
|
||||
useEffect(() => { fetchLogs(); }, [fetchLogs]);
|
||||
useEffect(() => { fetchStats(); }, [fetchStats]);
|
||||
|
||||
const openDetail = useCallback(async (id) => {
|
||||
setDetailOpen(true);
|
||||
setDetailLoading(true);
|
||||
setDetailError(null);
|
||||
setDetail(null);
|
||||
try {
|
||||
const res = await axios.get(`/api/request-logs/${id}`);
|
||||
setDetail(res.data || null);
|
||||
} catch (err) {
|
||||
setDetailError(extractApiError(err, 'Failed to load this request'));
|
||||
} finally {
|
||||
setDetailLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
const runPurge = useCallback(() => {
|
||||
Modal.confirm({
|
||||
title: 'Apply retention now?',
|
||||
icon: <DeleteOutlined />,
|
||||
content:
|
||||
'This runs the configured retention immediately instead of waiting for the next ' +
|
||||
'scheduled pass. It removes rows that are already past their retention window or ' +
|
||||
'beyond the row cap — it does not delete everything.',
|
||||
okText: 'Run retention pass',
|
||||
onOk: async () => {
|
||||
setPurging(true);
|
||||
try {
|
||||
const res = await axios.post('/api/request-logs/purge');
|
||||
const removed = res.data?.removed || {};
|
||||
const count = (removed.success || 0) + (removed.error || 0) + (removed.overflow || 0);
|
||||
message.success(`Retention pass completed — ${count} row(s) removed`);
|
||||
fetchLogs();
|
||||
fetchStats();
|
||||
} catch (err) {
|
||||
message.error(extractApiError(err, 'Retention pass failed'));
|
||||
} finally {
|
||||
setPurging(false);
|
||||
}
|
||||
},
|
||||
});
|
||||
}, [fetchLogs, fetchStats]);
|
||||
|
||||
const columns = useMemo(() => ([
|
||||
{
|
||||
title: 'Time',
|
||||
dataIndex: 'created_at',
|
||||
width: 180,
|
||||
render: (v) => <Text style={{ fontSize: 12 }}>{formatTime(v)}</Text>,
|
||||
},
|
||||
{
|
||||
title: 'Direction',
|
||||
dataIndex: 'direction',
|
||||
width: 110,
|
||||
render: (v) => (
|
||||
<Tag color={v === 'inbound' ? 'blue' : 'purple'}>{v === 'inbound' ? 'IN' : 'OUT'}</Tag>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Method',
|
||||
dataIndex: 'method',
|
||||
width: 90,
|
||||
render: (v) => <Tag>{v}</Tag>,
|
||||
},
|
||||
{
|
||||
title: 'URL',
|
||||
dataIndex: 'url',
|
||||
ellipsis: true,
|
||||
render: (v) => (
|
||||
<Tooltip title={v} placement="topLeft">
|
||||
<Text style={{ fontSize: 12 }} ellipsis>{v}</Text>
|
||||
</Tooltip>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Status',
|
||||
dataIndex: 'status_code',
|
||||
width: 100,
|
||||
render: (v, row) => (
|
||||
<Tag color={statusColor(row.status_class)}>{v ?? 'ERR'}</Tag>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Duration',
|
||||
dataIndex: 'duration_ms',
|
||||
width: 110,
|
||||
render: (v) => (
|
||||
// 1000ms matches the backend's slow-request threshold, so "red here"
|
||||
// means "logged as slow there".
|
||||
<Text type={v > 1000 ? 'danger' : undefined} style={{ fontSize: 12 }}>{v} ms</Text>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: 'Who / Where',
|
||||
key: 'who',
|
||||
width: 200,
|
||||
render: (_, row) => {
|
||||
if (row.direction === 'inbound') {
|
||||
return (
|
||||
<Space size={4}>
|
||||
<UserOutlined />
|
||||
<Text style={{ fontSize: 12 }}>{row.username || (row.user_id ? `#${row.user_id}` : 'anonymous')}</Text>
|
||||
</Space>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<Tooltip title={TARGET_LABELS[row.target] || row.target}>
|
||||
<Tag icon={<ApiOutlined />}>{row.target || '—'}</Tag>
|
||||
</Tooltip>
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
title: 'Error',
|
||||
dataIndex: 'error',
|
||||
width: 200,
|
||||
ellipsis: true,
|
||||
render: (v) => (v ? (
|
||||
<Tooltip title={v}><Text type="danger" style={{ fontSize: 12 }}>{v}</Text></Tooltip>
|
||||
) : <Text type="secondary">—</Text>),
|
||||
},
|
||||
{
|
||||
title: '',
|
||||
key: 'actions',
|
||||
width: 90,
|
||||
fixed: 'right',
|
||||
render: (_, row) => (
|
||||
<Button size="small" icon={<EyeOutlined />} onClick={() => openDetail(row.id)}>
|
||||
Detail
|
||||
</Button>
|
||||
),
|
||||
},
|
||||
]), [openDetail]);
|
||||
|
||||
if (!canRead) {
|
||||
return (
|
||||
<Card>
|
||||
<Alert
|
||||
type="error"
|
||||
showIcon
|
||||
message="Access denied"
|
||||
description="You need the requestlog.read permission to view the request log."
|
||||
/>
|
||||
</Card>
|
||||
);
|
||||
}
|
||||
|
||||
const inboundStats = stats?.by_direction?.find((d) => d.direction === 'inbound');
|
||||
const outboundStats = stats?.by_direction?.find((d) => d.direction === 'outbound');
|
||||
const dropped = stats?.sink?.dropped || 0;
|
||||
|
||||
const detailRow = detail?.log;
|
||||
|
||||
return (
|
||||
<div>
|
||||
<Card
|
||||
title={<Space><ClockCircleOutlined />Request Log</Space>}
|
||||
extra={
|
||||
<Space>
|
||||
<Button icon={<ReloadOutlined />} onClick={() => { fetchLogs(); fetchStats(); }} loading={loading}>
|
||||
Refresh
|
||||
</Button>
|
||||
{canManage && (
|
||||
<Button icon={<DeleteOutlined />} onClick={runPurge} loading={purging}>
|
||||
Apply retention now
|
||||
</Button>
|
||||
)}
|
||||
</Space>
|
||||
}
|
||||
>
|
||||
<Alert
|
||||
type="info"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Every API call in, and every HTTP call out"
|
||||
description={
|
||||
<>
|
||||
Inbound rows show which user called which endpoint and what came back.
|
||||
Outbound rows show which CA, DNS provider or agent this backend contacted.
|
||||
Bodies are captured <strong>redacted and size-capped</strong> — credentials,
|
||||
tokens, private keys and ACME signatures are never stored. Retention is
|
||||
configured in <Text code>Settings → Request Log</Text>.
|
||||
</>
|
||||
}
|
||||
/>
|
||||
|
||||
{scopedToSelf && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Showing your own requests only"
|
||||
description="Seeing every user's traffic, and all outbound calls, requires the requestlog.manage permission."
|
||||
/>
|
||||
)}
|
||||
|
||||
{dropped > 0 && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
icon={<WarningOutlined />}
|
||||
style={{ marginBottom: 16 }}
|
||||
message={`${dropped} row(s) dropped by this worker`}
|
||||
description="The writer queue filled up. Lower the sampling rate, turn off body capture, or raise REQUEST_LOG_QUEUE_MAX."
|
||||
/>
|
||||
)}
|
||||
|
||||
{stats && (
|
||||
<Space size="large" wrap style={{ marginBottom: 16 }}>
|
||||
<Statistic
|
||||
title="Inbound (24h)"
|
||||
value={inboundStats?.total || 0}
|
||||
suffix={inboundStats?.errors ? <Text type="danger" style={{ fontSize: 14 }}>{`/ ${inboundStats.errors} failed`}</Text> : null}
|
||||
/>
|
||||
<Statistic
|
||||
title="Outbound (24h)"
|
||||
value={outboundStats?.total || 0}
|
||||
suffix={outboundStats?.errors ? <Text type="danger" style={{ fontSize: 14 }}>{`/ ${outboundStats.errors} failed`}</Text> : null}
|
||||
/>
|
||||
<Statistic title="Rows stored" value={stats.total_rows || 0} />
|
||||
<Statistic title="Oldest entry" valueRender={() => <span style={{ fontSize: 16 }}>{formatTime(stats.oldest_at)}</span>} />
|
||||
</Space>
|
||||
)}
|
||||
|
||||
<Space style={{ marginBottom: 16 }} wrap>
|
||||
<Select
|
||||
placeholder="All directions"
|
||||
allowClear
|
||||
style={{ width: 220 }}
|
||||
value={direction}
|
||||
onChange={(v) => { setDirection(v); setPage(1); }}
|
||||
options={DIRECTION_OPTIONS}
|
||||
/>
|
||||
<Select
|
||||
placeholder="All statuses"
|
||||
allowClear
|
||||
style={{ width: 210 }}
|
||||
value={statusClass}
|
||||
onChange={(v) => { setStatusClass(v); setPage(1); }}
|
||||
options={STATUS_OPTIONS}
|
||||
/>
|
||||
<Select
|
||||
placeholder="All methods"
|
||||
allowClear
|
||||
mode="multiple"
|
||||
maxTagCount={1}
|
||||
style={{ width: 180 }}
|
||||
value={methods}
|
||||
onChange={(v) => { setMethods(v); setPage(1); }}
|
||||
options={METHOD_OPTIONS}
|
||||
/>
|
||||
<Select
|
||||
placeholder="All targets"
|
||||
allowClear
|
||||
style={{ width: 220 }}
|
||||
value={target}
|
||||
disabled={direction === 'inbound'}
|
||||
onChange={(v) => { setTarget(v); setPage(1); }}
|
||||
options={Object.entries(TARGET_LABELS).map(([value, label]) => ({ value, label }))}
|
||||
/>
|
||||
<DatePicker.RangePicker
|
||||
showTime
|
||||
value={range}
|
||||
onChange={(v) => { setRange(v); setPage(1); }}
|
||||
/>
|
||||
<Input.Search
|
||||
placeholder="Search URL…"
|
||||
allowClear
|
||||
enterButton
|
||||
style={{ width: 300 }}
|
||||
value={searchTyped}
|
||||
onChange={(e) => setSearchTyped(e.target.value)}
|
||||
onSearch={(v) => { setSearch(v); setPage(1); }}
|
||||
/>
|
||||
<Space size={4}>
|
||||
<Switch
|
||||
checked={errorsOnly}
|
||||
onChange={(v) => { setErrorsOnly(v); setPage(1); }}
|
||||
checkedChildren="Errors"
|
||||
unCheckedChildren="All"
|
||||
/>
|
||||
</Space>
|
||||
</Space>
|
||||
|
||||
{loadError && (
|
||||
<Alert type="error" showIcon style={{ marginBottom: 16 }} message={loadError} />
|
||||
)}
|
||||
|
||||
<Table
|
||||
rowKey="id"
|
||||
size="small"
|
||||
loading={loading}
|
||||
dataSource={rows}
|
||||
columns={columns}
|
||||
scroll={{ x: 1400 }}
|
||||
rowClassName={(row) => (isFailure(row) ? 'request-log-error-row' : '')}
|
||||
locale={{
|
||||
emptyText: <Empty description="No requests match these filters" />,
|
||||
}}
|
||||
pagination={{
|
||||
current: page,
|
||||
pageSize,
|
||||
total,
|
||||
showSizeChanger: true,
|
||||
pageSizeOptions: ['25', '50', '100', '200'],
|
||||
showTotal: (t, r) => `${r[0]}-${r[1]} of ${totalIsEstimate ? `${t}+` : t} requests`,
|
||||
onChange: (p, ps) => { setPage(p); setPageSize(ps); },
|
||||
}}
|
||||
/>
|
||||
</Card>
|
||||
|
||||
<Modal
|
||||
open={detailOpen}
|
||||
onCancel={() => setDetailOpen(false)}
|
||||
width={960}
|
||||
destroyOnClose
|
||||
title="Request detail"
|
||||
footer={<Button onClick={() => setDetailOpen(false)}>Close</Button>}
|
||||
>
|
||||
{detailLoading ? (
|
||||
<div style={{ textAlign: 'center', padding: 48 }}><Spin /></div>
|
||||
) : detailError ? (
|
||||
<Alert type="error" showIcon message={detailError} />
|
||||
) : !detailRow ? (
|
||||
<Empty description="Nothing to show" />
|
||||
) : (
|
||||
<Space direction="vertical" size="middle" style={{ width: '100%' }}>
|
||||
<Descriptions size="small" column={2} bordered>
|
||||
<Descriptions.Item label="Time">{formatTime(detailRow.created_at)}</Descriptions.Item>
|
||||
<Descriptions.Item label="Direction">
|
||||
<Tag color={detailRow.direction === 'inbound' ? 'blue' : 'purple'}>{detailRow.direction}</Tag>
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="Method"><Tag>{detailRow.method}</Tag></Descriptions.Item>
|
||||
<Descriptions.Item label="Status">
|
||||
<Tag color={statusColor(detailRow.status_class)}>{detailRow.status_code ?? 'no response'}</Tag>
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="URL" span={2}>
|
||||
<Text copyable style={{ fontSize: 12 }}>{detailRow.url}</Text>
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="Duration">{detailRow.duration_ms} ms</Descriptions.Item>
|
||||
<Descriptions.Item label="Target">
|
||||
{detailRow.target ? (TARGET_LABELS[detailRow.target] || detailRow.target) : '—'}
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="User">
|
||||
{detailRow.username || (detailRow.user_id ? `#${detailRow.user_id}` : 'anonymous')}
|
||||
</Descriptions.Item>
|
||||
<Descriptions.Item label="Client IP">{detailRow.client_ip || '—'}</Descriptions.Item>
|
||||
<Descriptions.Item label="Request id" span={2}>
|
||||
<Text code copyable style={{ fontSize: 11 }}>{detailRow.request_id}</Text>
|
||||
</Descriptions.Item>
|
||||
{detailRow.error && (
|
||||
<Descriptions.Item label="Error" span={2}>
|
||||
<Text type="danger">{detailRow.error}</Text>
|
||||
</Descriptions.Item>
|
||||
)}
|
||||
</Descriptions>
|
||||
|
||||
{detailRow.truncated && (
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
message="Body truncated"
|
||||
description={
|
||||
`Only the first part of the body was captured (request ${detailRow.request_body_bytes} bytes, ` +
|
||||
`response ${detailRow.response_body_bytes} bytes on the wire). Raise the body cap in ` +
|
||||
`Settings → Request Log if you need more.`
|
||||
}
|
||||
/>
|
||||
)}
|
||||
|
||||
<Card size="small" title="Request headers">
|
||||
<JsonBlock value={detailRow.request_headers} token={token} />
|
||||
</Card>
|
||||
<Card size="small" title="Request body">
|
||||
<JsonBlock value={detailRow.request_body} token={token} />
|
||||
</Card>
|
||||
<Card size="small" title="Response headers">
|
||||
<JsonBlock value={detailRow.response_headers} token={token} />
|
||||
</Card>
|
||||
<Card size="small" title="Response body">
|
||||
<JsonBlock value={detailRow.response_body} token={token} />
|
||||
</Card>
|
||||
|
||||
<Card
|
||||
size="small"
|
||||
title={<Space><CloudDownloadOutlined />Calls triggered by this request</Space>}
|
||||
>
|
||||
{detail?.related?.length ? (
|
||||
<Table
|
||||
rowKey="id"
|
||||
size="small"
|
||||
pagination={false}
|
||||
dataSource={detail.related}
|
||||
columns={[
|
||||
{ title: 'Dir', dataIndex: 'direction', width: 70,
|
||||
render: (v) => <Tag color={v === 'inbound' ? 'blue' : 'purple'}>{v === 'inbound' ? 'IN' : 'OUT'}</Tag> },
|
||||
{ title: 'Target', dataIndex: 'target', width: 140, render: (v) => v || '—' },
|
||||
{ title: 'Method', dataIndex: 'method', width: 80 },
|
||||
{ title: 'URL', dataIndex: 'url', ellipsis: true },
|
||||
{ title: 'Status', dataIndex: 'status_code', width: 80,
|
||||
render: (v, r) => <Tag color={statusColor(r.status_class)}>{v ?? 'ERR'}</Tag> },
|
||||
{ title: '', key: 'go', width: 70,
|
||||
render: (_, r) => <Button size="small" type="link" onClick={() => openDetail(r.id)}>Open</Button> },
|
||||
]}
|
||||
/>
|
||||
) : (
|
||||
<Paragraph type="secondary" style={{ margin: 0 }}>
|
||||
No other calls share this request id.
|
||||
</Paragraph>
|
||||
)}
|
||||
</Card>
|
||||
</Space>
|
||||
)}
|
||||
</Modal>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
export default RequestLog;
|
||||
@@ -1,6 +1,6 @@
|
||||
import React, { useEffect, useState } from 'react';
|
||||
import { Card, Form, Switch, Button, InputNumber, message, Tabs, Input, Select, Collapse, Space, Alert, Tag, Spin, Tooltip } from 'antd';
|
||||
import { SafetyCertificateOutlined, ApiOutlined, CheckCircleOutlined, CloseCircleOutlined, InfoCircleOutlined } from '@ant-design/icons';
|
||||
import { SafetyCertificateOutlined, ApiOutlined, CheckCircleOutlined, CloseCircleOutlined, InfoCircleOutlined, FileSearchOutlined } from '@ant-design/icons';
|
||||
import { useSearchParams } from 'react-router-dom';
|
||||
import axios from 'axios';
|
||||
|
||||
@@ -43,6 +43,15 @@ const Settings = () => {
|
||||
const [testResult, setTestResult] = useState(null);
|
||||
const [testing, setTesting] = useState(false);
|
||||
|
||||
// v1.11.0 — request/response log retention. Read/written through
|
||||
// /api/request-logs/settings, NOT the generic /api/settings/{category}: that
|
||||
// endpoint stringifies values with str(), which turns True into 'True' and
|
||||
// fails the ::jsonb cast.
|
||||
const [rlForm] = Form.useForm();
|
||||
const [rlLoading, setRlLoading] = useState(false);
|
||||
const [rlSaving, setRlSaving] = useState(false);
|
||||
const [rlDenied, setRlDenied] = useState(false);
|
||||
|
||||
const onFinish = (values) => {
|
||||
try {
|
||||
localStorage.setItem('app_settings', JSON.stringify({
|
||||
@@ -70,8 +79,62 @@ const Settings = () => {
|
||||
|
||||
useEffect(() => {
|
||||
loadAcmeSettings();
|
||||
loadRequestLogSettings();
|
||||
}, []);
|
||||
|
||||
const loadRequestLogSettings = async () => {
|
||||
setRlLoading(true);
|
||||
try {
|
||||
const res = await axios.get('/api/request-logs/settings');
|
||||
rlForm.setFieldsValue(res.data?.settings || {});
|
||||
setRlDenied(false);
|
||||
} catch (err) {
|
||||
// A viewer can open Settings but has no requestlog.manage — show the tab
|
||||
// read-only-with-explanation rather than a scary console error.
|
||||
if (err?.response?.status === 403) {
|
||||
setRlDenied(true);
|
||||
} else {
|
||||
console.error('Error loading request log settings:', err);
|
||||
}
|
||||
} finally {
|
||||
setRlLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const onRequestLogSave = async (values) => {
|
||||
setRlSaving(true);
|
||||
try {
|
||||
const res = await axios.put('/api/request-logs/settings', {
|
||||
...values,
|
||||
// Values come back from the InputNumber controls as numbers already;
|
||||
// the endpoint is properly typed, so no per-value JSON.stringify here
|
||||
// (unlike the ACME form above, which talks to the legacy endpoint).
|
||||
exclude_paths: values.exclude_paths || [],
|
||||
});
|
||||
// Show what the server ACTUALLY applied, not what was typed. Values are
|
||||
// clamped server-side, and clearing the exclude list does not mean "log
|
||||
// everything": normalize_exclude_paths() falls back to the shipped
|
||||
// defaults so the log viewer and the raw-body heartbeat stay excluded.
|
||||
// Without this the form would keep displaying an empty list that is not
|
||||
// in effect.
|
||||
const applied = res?.data?.settings;
|
||||
if (applied) {
|
||||
rlForm.setFieldsValue(applied);
|
||||
const typed = values.exclude_paths || [];
|
||||
if (typed.length === 0 && (applied.exclude_paths || []).length > 0) {
|
||||
message.warning(
|
||||
'An empty exclude list is not applied as "log everything" — the shipped defaults were restored.'
|
||||
);
|
||||
}
|
||||
}
|
||||
message.success('Request log settings saved');
|
||||
} catch (err) {
|
||||
message.error(err?.response?.data?.detail || 'Failed to save request log settings');
|
||||
} finally {
|
||||
setRlSaving(false);
|
||||
}
|
||||
};
|
||||
|
||||
const loadAcmeSettings = async () => {
|
||||
setAcmeLoading(true);
|
||||
try {
|
||||
@@ -371,6 +434,158 @@ const Settings = () => {
|
||||
</Spin>
|
||||
),
|
||||
},
|
||||
{
|
||||
key: 'requestlog',
|
||||
label: (
|
||||
<span><FileSearchOutlined /> Request Log</span>
|
||||
),
|
||||
children: (
|
||||
<Spin spinning={rlLoading}>
|
||||
<Card>
|
||||
<Alert
|
||||
message="Request / Response Log"
|
||||
description="Records every inbound API call and every outbound HTTP call this backend makes (ACME, DNS providers, agents), with redacted and size-capped request and response bodies. Browse it under Request Log in the sidebar."
|
||||
type="info"
|
||||
showIcon
|
||||
icon={<FileSearchOutlined />}
|
||||
style={{ marginBottom: 24 }}
|
||||
/>
|
||||
|
||||
{rlDenied && (
|
||||
<Alert
|
||||
message="Read-only"
|
||||
description="Changing the request log policy requires the requestlog.manage permission."
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginBottom: 24 }}
|
||||
/>
|
||||
)}
|
||||
|
||||
<Form
|
||||
form={rlForm}
|
||||
layout="vertical"
|
||||
onFinish={onRequestLogSave}
|
||||
disabled={rlDenied}
|
||||
initialValues={{
|
||||
enabled: true,
|
||||
capture_inbound: true,
|
||||
capture_outbound: true,
|
||||
capture_get: true,
|
||||
capture_bodies: true,
|
||||
max_body_bytes: 8192,
|
||||
sample_rate: 1.0,
|
||||
success_retention_days: 7,
|
||||
error_retention_days: 30,
|
||||
max_rows: 500000,
|
||||
prune_interval_minutes: 60,
|
||||
exclude_paths: [],
|
||||
}}
|
||||
>
|
||||
<Card size="small" title="Capture" style={{ marginBottom: 24 }}>
|
||||
<Form.Item
|
||||
name="enabled"
|
||||
label="Enable request log"
|
||||
valuePropName="checked"
|
||||
tooltip="Turning this off stops all capture immediately, without a restart. To remove the middleware entirely set REQUEST_LOG_ENABLED=false in the backend environment."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item name="capture_inbound" label="Log inbound API calls" valuePropName="checked">
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="capture_outbound"
|
||||
label="Log outbound HTTP calls"
|
||||
valuePropName="checked"
|
||||
tooltip="Calls this backend makes to Let's Encrypt / ACME, Cloudflare, GoDaddy, agents and HAProxy stats."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="capture_get"
|
||||
label="Include GET requests"
|
||||
valuePropName="checked"
|
||||
tooltip="GETs are the bulk of the traffic. Turning this off keeps writes and errors only."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="capture_agent_success"
|
||||
label="Include successful agent polls"
|
||||
valuePropName="checked"
|
||||
tooltip="Each agent writes about 9,800 rows a day just saying nothing changed, so on a large fleet this fills the row cap in hours and shortens retention for everything else. FAILED agent calls are always logged regardless. Turn this on only while debugging a specific node, and turn it back off."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="capture_bodies"
|
||||
label="Capture bodies (redacted)"
|
||||
valuePropName="checked"
|
||||
tooltip="Passwords, tokens, API keys, private-key PEMs and ACME signatures are never stored, whatever this is set to."
|
||||
>
|
||||
<Switch />
|
||||
</Form.Item>
|
||||
<Form.Item name="max_body_bytes" label="Maximum body size captured (bytes)">
|
||||
<InputNumber min={0} max={262144} step={1024} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="sample_rate"
|
||||
label="Sampling rate for successful requests"
|
||||
tooltip="1.0 logs everything. Errors are always captured at 100%, whatever this is set to."
|
||||
>
|
||||
<InputNumber min={0} max={1} step={0.05} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
</Card>
|
||||
|
||||
<Card size="small" title="Retention" style={{ marginBottom: 24 }}>
|
||||
<Alert
|
||||
type="warning"
|
||||
showIcon
|
||||
style={{ marginBottom: 16 }}
|
||||
message="Whichever limit is reached first wins"
|
||||
description="Rows are removed when they pass their retention window OR when the table exceeds the row cap — the cap is the backstop for a sudden traffic spike."
|
||||
/>
|
||||
<Form.Item name="success_retention_days" label="Keep successful requests for (days)">
|
||||
<InputNumber min={1} max={365} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
<Form.Item
|
||||
name="error_retention_days"
|
||||
label="Keep failed requests for (days)"
|
||||
tooltip="4xx, 5xx and calls that got no response at all. Usually set longer than the success window."
|
||||
>
|
||||
<InputNumber min={1} max={365} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
<Form.Item name="max_rows" label="Maximum stored rows">
|
||||
<InputNumber min={1000} max={50000000} step={10000} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
<Form.Item name="prune_interval_minutes" label="Minimum interval between prune passes (minutes)">
|
||||
<InputNumber min={5} max={1440} style={{ width: 200 }} />
|
||||
</Form.Item>
|
||||
</Card>
|
||||
|
||||
<Card size="small" title="Excluded paths" style={{ marginBottom: 24 }}>
|
||||
<Form.Item
|
||||
name="exclude_paths"
|
||||
label="Never log these path prefixes"
|
||||
tooltip="Health checks, the API docs, the ACME challenge endpoint and the agent heartbeat are excluded by default. The log viewer's own endpoints are always excluded and cannot be re-enabled."
|
||||
>
|
||||
<Select
|
||||
mode="tags"
|
||||
tokenSeparators={[',', ' ']}
|
||||
placeholder="/api/health"
|
||||
style={{ width: '100%' }}
|
||||
/>
|
||||
</Form.Item>
|
||||
</Card>
|
||||
|
||||
<Button type="primary" htmlType="submit" loading={rlSaving} disabled={rlDenied}>
|
||||
Save Request Log Settings
|
||||
</Button>
|
||||
</Form>
|
||||
</Card>
|
||||
</Spin>
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
return (
|
||||
|
||||
@@ -216,6 +216,17 @@ const PERMISSION_TREE = [
|
||||
{ title: 'Export Activity Logs', key: 'activity.export' }
|
||||
]
|
||||
},
|
||||
// v1.11.0 — the API does no server-side whitelist of permission strings, so
|
||||
// this tree is the ONLY catalogue an admin can grant from. Without an entry
|
||||
// here the permission exists but is unreachable for custom roles.
|
||||
{
|
||||
title: '🧾 Request Log',
|
||||
key: 'requestlog',
|
||||
children: [
|
||||
{ title: 'View Request/Response Logs', key: 'requestlog.read' },
|
||||
{ title: 'Manage Retention & Purge', key: 'requestlog.manage' }
|
||||
]
|
||||
},
|
||||
{
|
||||
title: '⚙️ Settings',
|
||||
key: 'settings',
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import React, { useState, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Table, Button, Space, Modal, Form, Input, InputNumber, Select, Tag, message,
|
||||
Switch, Typography, Card, Alert, Tooltip, Spin
|
||||
Switch, Typography, Card, Alert, Tooltip, Spin, Checkbox
|
||||
} from 'antd';
|
||||
import {
|
||||
PlusOutlined, EditOutlined, DeleteOutlined, ReloadOutlined, WarningOutlined,
|
||||
@@ -46,6 +46,36 @@ const parseArr = (v) => {
|
||||
return [];
|
||||
};
|
||||
|
||||
// v1.10.4 — adoption blockers come back as prose from the parser. Two classes are resolvable by
|
||||
// the operator and the rest are not, so the modal has to tell them apart:
|
||||
// * `loss` — "our renderer cannot reproduce this, so adopting would delete it". A deliberate
|
||||
// choice, waivable with an explicit tick.
|
||||
// * `prefix` — the address has no explicit prefix length. Supplying it resolves the blocker;
|
||||
// we never guess a netmask for a live VIP.
|
||||
// * `hard` — an unknown VRID, a fractional advert_int, an unsupported auth_type. Not losses
|
||||
// but impossibilities; nothing in the UI may override them.
|
||||
const splitBlockers = (blockers) => {
|
||||
const list = blockers || [];
|
||||
return {
|
||||
loss: list.filter((b) => b.includes('would delete it')),
|
||||
prefix: list.filter((b) => b.includes('no explicit prefix length')),
|
||||
hard: list.filter((b) => !b.includes('would delete it') && !b.includes('no explicit prefix length')),
|
||||
};
|
||||
};
|
||||
|
||||
// v1.10.10 — every node of an instance reports the SAME problems about the SAME shared config,
|
||||
// so merging their blocker lists repeats each one per member. The line numbers differ between
|
||||
// the files, so exact-string dedup does not collapse them; key on the text WITHOUT the leading
|
||||
// "line N:" and keep the first occurrence. Four issues on a pair used to read as eight.
|
||||
const mergeBlockers = (lists) => {
|
||||
const seen = new Map();
|
||||
lists.flat().forEach((b) => {
|
||||
const key = String(b).replace(/^line \d+:\s*/, '');
|
||||
if (!seen.has(key)) seen.set(key, b);
|
||||
});
|
||||
return Array.from(seen.values());
|
||||
};
|
||||
|
||||
// This component uses raw fetch(), but extractApiError expects an axios-shaped error
|
||||
// (err.response.data). Read the fetch Response body and reuse the envelope-aware extractor
|
||||
// so backend messages — e.g. the 409 "node already in VIP X" — actually reach the user.
|
||||
@@ -55,7 +85,7 @@ const fetchApiError = async (res, fallback) => {
|
||||
};
|
||||
|
||||
const VIPManagement = () => {
|
||||
const { clusters } = useCluster();
|
||||
const { clusters, selectedCluster } = useCluster();
|
||||
const [vips, setVips] = useState([]);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [modalVisible, setModalVisible] = useState(false);
|
||||
@@ -67,6 +97,12 @@ const VIPManagement = () => {
|
||||
// Delete confirmation (with opt-in package uninstall) + diagnostics modal state.
|
||||
const [deleteTarget, setDeleteTarget] = useState(null);
|
||||
const [showL2Note, setShowL2Note] = useState(false);
|
||||
// v1.10.4 — VIP adoption from what the agents found on their nodes.
|
||||
const [discoveries, setDiscoveries] = useState([]);
|
||||
const [adoptTarget, setAdoptTarget] = useState(null); // { discovery, candidate }
|
||||
const [adoptAcceptLoss, setAdoptAcceptLoss] = useState(false);
|
||||
const [adopting, setAdopting] = useState(false);
|
||||
const [adoptForm] = Form.useForm();
|
||||
const [diagVip, setDiagVip] = useState(null);
|
||||
const [diagData, setDiagData] = useState(null);
|
||||
const [diagLoading, setDiagLoading] = useState(false);
|
||||
@@ -80,10 +116,15 @@ const VIPManagement = () => {
|
||||
return Array.from(seen, ([id, name]) => ({ id, name }));
|
||||
}, [clusters]);
|
||||
|
||||
// v1.10.6 — both lists follow the cluster picked in the header, like every other page. The
|
||||
// selector was always there but this page ignored it, so a fleet with several clusters saw
|
||||
// one undifferentiated list. Falls back to fleet-wide while the context is still resolving.
|
||||
const scopeQuery = selectedCluster?.id ? `?cluster_id=${selectedCluster.id}` : '';
|
||||
|
||||
const fetchVips = useCallback(async () => {
|
||||
setLoading(true);
|
||||
try {
|
||||
const res = await fetch('/api/vip', { headers: authHeaders() });
|
||||
const res = await fetch(`/api/vip${scopeQuery}`, { headers: authHeaders() });
|
||||
if (res.ok) {
|
||||
const data = await res.json();
|
||||
setVips(data.vips || []);
|
||||
@@ -96,13 +137,161 @@ const VIPManagement = () => {
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
}, [scopeQuery]);
|
||||
|
||||
// v1.10.4 — keepalived configs the agents found on their nodes but do NOT manage. This is why
|
||||
// the page could be empty on a fleet that already runs keepalived: the flow was one-way, so
|
||||
// nothing ever read what was already there.
|
||||
const fetchDiscoveries = useCallback(async () => {
|
||||
try {
|
||||
const res = await fetch(`/api/vip/discoveries${scopeQuery}`, { headers: authHeaders() });
|
||||
if (!res.ok) { setDiscoveries([]); return; }
|
||||
const data = await res.json();
|
||||
// v1.10.8 — hide a node only while its adoption still STANDS. Filtering on adopted_vip_id
|
||||
// alone hid it forever after a reject: nothing clears that column, the VIP is only ever
|
||||
// soft-deleted, and the agent does not re-report an unchanged file.
|
||||
setDiscoveries((data.discoveries || [])
|
||||
.filter((d) => !d.is_managed && !d.adopted_vip_active));
|
||||
} catch (e) {
|
||||
console.error('fetchDiscoveries failed', e);
|
||||
}
|
||||
}, [scopeQuery]);
|
||||
|
||||
useEffect(() => {
|
||||
fetchVips();
|
||||
const t = setInterval(fetchVips, 30000); // live MASTER/BACKUP via existing detection pipeline
|
||||
fetchDiscoveries();
|
||||
const t = setInterval(() => { fetchVips(); fetchDiscoveries(); }, 30000); // live MASTER/BACKUP via existing detection pipeline
|
||||
return () => clearInterval(t);
|
||||
}, [fetchVips]);
|
||||
}, [fetchVips, fetchDiscoveries]);
|
||||
|
||||
// v1.10.8 — one row per VRRP INSTANCE, not per node. Adoption now takes the whole instance
|
||||
// (every node in the pool reporting the same VRID + address), so listing the nodes as separate
|
||||
// adoptable rows invited exactly the half-adoption the backend refuses: adopting the BACKUP
|
||||
// alone cannot be applied, and on a unicast pair adopting one side drops the peer list and
|
||||
// drops both nodes into a split brain. Identity is (VRID, address), same as keepalived's.
|
||||
const discoveryGroups = React.useMemo(() => {
|
||||
const groups = new Map();
|
||||
(discoveries || []).forEach((d) => {
|
||||
const cands = d.analysis?.candidates || [];
|
||||
if (cands.length === 0) {
|
||||
const key = `solo:${d.agent_id}`;
|
||||
groups.set(key, { key, instance_name: '—', vip: null, members: [{ discovery: d, candidate: null }] });
|
||||
return;
|
||||
}
|
||||
cands.forEach((c) => {
|
||||
const vrid = c.vip?.virtual_router_id;
|
||||
const addr = c.vip?.virtual_ip;
|
||||
const key = (vrid != null && addr) ? `${vrid}|${addr}` : `solo:${d.agent_id}:${c.instance_name}`;
|
||||
if (!groups.has(key)) {
|
||||
groups.set(key, { key, instance_name: c.instance_name, vip: c.vip, members: [] });
|
||||
}
|
||||
groups.get(key).members.push({ discovery: d, candidate: c });
|
||||
});
|
||||
});
|
||||
return Array.from(groups.values());
|
||||
}, [discoveries]);
|
||||
|
||||
// What stops a whole instance from being adopted. Mirrors the backend's checks so the button
|
||||
// state and the 422 it would return cannot drift apart.
|
||||
const groupState = (g) => {
|
||||
const parseFailed = g.members.filter((m) => m.discovery.parse_error);
|
||||
const noCandidate = g.members.filter((m) => !m.candidate);
|
||||
const blockers = mergeBlockers(g.members.map((m) => m.candidate?.blockers || []));
|
||||
const { hard, loss, prefix } = splitBlockers(blockers);
|
||||
const masters = g.members.filter((m) => m.candidate?.member?.role === 'MASTER').length;
|
||||
// Any reported config that mentions this address but is NOT one of this group's nodes would
|
||||
// be left behind when the others are taken over — unparseable, agent disabled, different
|
||||
// pool. The endpoint refuses on exactly that question, so ask it here too rather than letting
|
||||
// the operator click into a 422. This list is cluster-scoped, so a peer in another pool is
|
||||
// invisible from here; the endpoint still catches it.
|
||||
const inGroup = new Set(g.members.map((m) => m.discovery.agent_id));
|
||||
const strandedPeers = g.vip?.virtual_ip
|
||||
? discoveries.filter((d) => !inGroup.has(d.agent_id)
|
||||
&& (d.config_preview || '').includes(g.vip.virtual_ip))
|
||||
: [];
|
||||
// v1.10.11 — ONE ordered decision drives both the label and the button, because they used to
|
||||
// be computed separately and disagreed: a group blocked by an unreadable peer was tagged
|
||||
// "MASTER missing" (its peer's MASTER simply had not been counted) while the disabled button
|
||||
// gave the real reason in its own tooltip. Whatever stops adoption is what the label says.
|
||||
let reason = null;
|
||||
let label = null;
|
||||
let colour = null;
|
||||
if (parseFailed.length) {
|
||||
reason = `${parseFailed.map((m) => m.discovery.agent_name).join(', ')}: config could not be parsed`;
|
||||
label = 'unparseable'; colour = 'red';
|
||||
} else if (noCandidate.length) {
|
||||
reason = 'no vrrp_instance in the report';
|
||||
label = 'no vrrp_instance'; colour = undefined;
|
||||
} else if (strandedPeers.length) {
|
||||
const names = strandedPeers.map((d) => d.agent_name).join(', ');
|
||||
const why = strandedPeers.every((d) => d.parse_error)
|
||||
? 'their config could not be parsed'
|
||||
: 'they cannot be taken over with this instance';
|
||||
reason = `${names} also reference ${g.vip.virtual_ip} but ${why} — fix those nodes first, or `
|
||||
+ 'they would be left running an unmanaged config on the same address';
|
||||
label = 'blocked by peer'; colour = 'red';
|
||||
} else if (hard.length) {
|
||||
reason = hard.join(' · ');
|
||||
label = `${hard.length} blocker(s)`; colour = 'red';
|
||||
} else if (masters !== 1) {
|
||||
reason = masters === 0
|
||||
? 'no node in this instance declares state MASTER — enable the missing node\'s agent so it reports its config'
|
||||
: `${masters} nodes declare MASTER; exactly one must`;
|
||||
label = masters === 0 ? 'MASTER missing' : 'two MASTERs'; colour = 'red';
|
||||
} else if (blockers.length) {
|
||||
label = 'needs review'; colour = 'gold'; // resolvable in the adopt dialog
|
||||
} else {
|
||||
label = 'yes'; colour = 'green';
|
||||
}
|
||||
return { parseFailed, noCandidate, hard, loss, prefix, masters, blockers,
|
||||
strandedPeers, reason, label, colour };
|
||||
};
|
||||
|
||||
const openAdopt = (group) => {
|
||||
// Any member can carry the request: the backend resolves the whole instance from it. Prefer
|
||||
// the MASTER so the suggested name and the preview show the authoritative node.
|
||||
const primary = group.members.find((m) => m.candidate?.member?.role === 'MASTER') || group.members[0];
|
||||
setAdoptTarget({ group, primary, discovery: primary.discovery, candidate: primary.candidate });
|
||||
setAdoptAcceptLoss(false);
|
||||
adoptForm.setFieldsValue({
|
||||
name: `${group.vip?.virtual_ip || primary.discovery.agent_name}-vip`,
|
||||
prefix_length: group.vip?.prefix_length ?? undefined,
|
||||
});
|
||||
};
|
||||
|
||||
const submitAdopt = async () => {
|
||||
if (!adoptTarget) return;
|
||||
let values;
|
||||
try { values = await adoptForm.validateFields(); } catch { return; }
|
||||
setAdopting(true);
|
||||
try {
|
||||
const res = await fetch('/api/vip/adopt', {
|
||||
method: 'POST',
|
||||
headers: authHeaders(),
|
||||
body: JSON.stringify({
|
||||
agent_id: adoptTarget.discovery.agent_id,
|
||||
instance_name: adoptTarget.candidate.instance_name,
|
||||
name: values.name,
|
||||
description: values.description || undefined,
|
||||
prefix_length: values.prefix_length ?? undefined,
|
||||
accept_data_loss: adoptAcceptLoss || undefined,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
message.error(await fetchApiError(res, 'Adoption failed'), 8);
|
||||
return;
|
||||
}
|
||||
const data = await res.json();
|
||||
message.success(data.message || 'VIP adopted', 8);
|
||||
setAdoptTarget(null);
|
||||
fetchVips();
|
||||
fetchDiscoveries();
|
||||
} catch (e) {
|
||||
message.error('Adoption failed');
|
||||
} finally {
|
||||
setAdopting(false);
|
||||
}
|
||||
};
|
||||
|
||||
// Build the participating-nodes table from the pool's EXISTING agents (installed via the
|
||||
// standard Agent Management process). On edit, pre-select the VIP's current members.
|
||||
@@ -410,6 +599,200 @@ const VIPManagement = () => {
|
||||
<Table rowKey="id" columns={columns} dataSource={vips} loading={loading} pagination={{ pageSize: 10 }} />
|
||||
</Card>
|
||||
|
||||
{/* v1.10.4 — keepalived that already exists on a node. Shown separately from managed VIPs
|
||||
because OpenManager is NOT managing these: the agent found them, reported them, and
|
||||
deliberately left them untouched. */}
|
||||
{discoveries.length > 0 && (
|
||||
<Card style={{ marginTop: 16 }} title={
|
||||
<Space>
|
||||
<FileSearchOutlined />
|
||||
<span>Unmanaged keepalived detected on {discoveries.length} node(s)</span>
|
||||
</Space>
|
||||
}>
|
||||
<Alert
|
||||
type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message="These nodes already run keepalived, configured outside OpenManager"
|
||||
description={
|
||||
<span>
|
||||
The agent read each <Text code>keepalived.conf</Text> and left it untouched — nothing
|
||||
on these nodes has been changed. Adopting one creates a managed VIP from the values
|
||||
in that file, and the node's config is only handed over when you apply it from
|
||||
Apply Management. Adoption replaces the file with OpenManager's render, so anything
|
||||
it cannot reproduce is listed as a blocker rather than silently dropped.
|
||||
</span>
|
||||
}
|
||||
/>
|
||||
<Table
|
||||
rowKey={(r) => r.key}
|
||||
size="small"
|
||||
pagination={false}
|
||||
dataSource={discoveryGroups}
|
||||
columns={[
|
||||
{ title: 'Nodes', key: 'agents',
|
||||
render: (_v, r) => (
|
||||
<Space direction="vertical" size={0}>
|
||||
{r.members.map((m) => (
|
||||
<Text strong key={m.discovery.agent_id}>{m.discovery.agent_name}</Text>
|
||||
))}
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
{r.members[0]?.discovery.pool_name || 'no pool'}
|
||||
</Text>
|
||||
</Space>
|
||||
) },
|
||||
{ title: 'Instance', dataIndex: 'instance_name', key: 'instance' },
|
||||
{ title: 'Virtual IP', key: 'vip',
|
||||
render: (_v, r) => (r.vip?.virtual_ip
|
||||
? <Text code>{r.vip.virtual_ip}
|
||||
{r.vip.prefix_length != null ? `/${r.vip.prefix_length}` : ''}</Text>
|
||||
: <Text type="secondary">—</Text>) },
|
||||
{ title: 'VRID', key: 'vrid',
|
||||
render: (_v, r) => (r.vip?.virtual_router_id ?? <Text type="secondary">—</Text>) },
|
||||
{ title: 'Members', key: 'member',
|
||||
render: (_v, r) => (
|
||||
<Space direction="vertical" size={0}>
|
||||
{r.members.map((m) => (
|
||||
<Space size={4} key={m.discovery.agent_id}>
|
||||
{m.candidate ? (
|
||||
<>
|
||||
<Tag color={m.candidate.member.role === 'MASTER' ? 'green' : 'default'}>
|
||||
{m.candidate.member.role}
|
||||
</Tag>
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
prio {m.candidate.member.priority} · {m.candidate.member.network_interface}
|
||||
</Text>
|
||||
</>
|
||||
) : <Text type="secondary">—</Text>}
|
||||
</Space>
|
||||
))}
|
||||
</Space>
|
||||
) },
|
||||
{ title: 'Adoptable', key: 'adoptable',
|
||||
render: (_v, r) => {
|
||||
// Label, colour and tooltip all come from the single ordered decision in
|
||||
// groupState, so the tag can never name a different problem than the one that
|
||||
// actually disables the button.
|
||||
const st = groupState(r);
|
||||
const detail = st.parseFailed.length
|
||||
? st.parseFailed.map((m) => `${m.discovery.agent_name}: ${m.discovery.parse_error}`).join(' · ')
|
||||
: (st.reason || st.blockers.join(' · '));
|
||||
const tag = <Tag color={st.colour}>{st.label}</Tag>;
|
||||
return detail ? <Tooltip title={detail}>{tag}</Tooltip> : tag;
|
||||
} },
|
||||
{ title: 'Actions', key: 'actions',
|
||||
render: (_v, r) => {
|
||||
const st = groupState(r);
|
||||
const btn = (
|
||||
<Button size="small" type="primary" ghost
|
||||
disabled={!!st.reason} onClick={() => openAdopt(r)}>
|
||||
Adopt
|
||||
</Button>
|
||||
);
|
||||
// A disabled antd Button swallows mouse events, so the tooltip needs a live
|
||||
// wrapper or the operator never learns WHY adoption is unavailable.
|
||||
return st.reason
|
||||
? <Tooltip title={st.reason}><span style={{ display: 'inline-block' }}>{btn}</span></Tooltip>
|
||||
: btn;
|
||||
} },
|
||||
]}
|
||||
/>
|
||||
</Card>
|
||||
)}
|
||||
|
||||
{/* Adopt modal — shows what will be taken over, what was assumed, and what would be lost. */}
|
||||
<Modal
|
||||
title={adoptTarget
|
||||
? `Adopt ${adoptTarget.group.instance_name} — ${adoptTarget.group.members.length} node(s)`
|
||||
: 'Adopt VIP'}
|
||||
open={!!adoptTarget}
|
||||
onCancel={() => setAdoptTarget(null)}
|
||||
onOk={submitAdopt}
|
||||
confirmLoading={adopting}
|
||||
okText="Adopt as PENDING"
|
||||
width={720}
|
||||
okButtonProps={{
|
||||
disabled: !!adoptTarget && (() => {
|
||||
const { loss, hard } = splitBlockers(
|
||||
mergeBlockers(adoptTarget.group.members.map((m) => m.candidate?.blockers || [])));
|
||||
return hard.length > 0 || (loss.length > 0 && !adoptAcceptLoss);
|
||||
})(),
|
||||
}}
|
||||
>
|
||||
{adoptTarget && (() => {
|
||||
const cand = adoptTarget.candidate;
|
||||
// Blockers are aggregated across EVERY node of the instance, because adoption
|
||||
// overwrites every one of their files — the backend refuses on the same combined set.
|
||||
const { loss, prefix, hard } = splitBlockers(
|
||||
mergeBlockers(adoptTarget.group.members.map((m) => m.candidate?.blockers || [])));
|
||||
return (
|
||||
<>
|
||||
<Alert type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message={`These ${adoptTarget.group.members.length} node(s) will be taken over together`}
|
||||
description={
|
||||
<ul style={{ margin: 0, paddingLeft: 18 }}>
|
||||
{adoptTarget.group.members.map((m) => (
|
||||
<li key={m.discovery.agent_id}>
|
||||
<Text strong>{m.discovery.agent_name}</Text>
|
||||
{' — '}{m.candidate?.member?.role} · prio {m.candidate?.member?.priority}
|
||||
{' · '}{m.candidate?.member?.network_interface}
|
||||
{' · '}<Text code>{m.discovery.config_path}</Text>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
} />
|
||||
{hard.length > 0 && (
|
||||
<Alert type="error" showIcon style={{ marginBottom: 12 }}
|
||||
message="This config cannot be adopted"
|
||||
description={<ul style={{ margin: 0, paddingLeft: 18 }}>
|
||||
{hard.map((b, i) => <li key={i}>{b}</li>)}
|
||||
</ul>} />
|
||||
)}
|
||||
{loss.length > 0 && (
|
||||
<Alert type="warning" showIcon style={{ marginBottom: 12 }}
|
||||
message="Adopting would delete these directives from the node's config"
|
||||
description={
|
||||
<>
|
||||
<ul style={{ margin: '0 0 8px', paddingLeft: 18 }}>
|
||||
{loss.map((b, i) => <li key={i}>{b}</li>)}
|
||||
</ul>
|
||||
<Checkbox checked={adoptAcceptLoss} onChange={(e) => setAdoptAcceptLoss(e.target.checked)}>
|
||||
I understand these will be lost when the config is handed over
|
||||
</Checkbox>
|
||||
</>
|
||||
} />
|
||||
)}
|
||||
{(cand.defaulted || []).length > 0 && (
|
||||
<Alert type="info" showIcon style={{ marginBottom: 12 }}
|
||||
message={`Assumed from keepalived's defaults (absent from the file): ${cand.defaulted.join(', ')}`} />
|
||||
)}
|
||||
<Form form={adoptForm} layout="vertical">
|
||||
<Form.Item name="name" label="VIP name"
|
||||
rules={[{ required: true, message: 'Give the managed VIP a name' }]}>
|
||||
<Input placeholder="e.g. dmz-web-vip" />
|
||||
</Form.Item>
|
||||
{prefix.length > 0 && (
|
||||
<Form.Item name="prefix_length" label="Prefix length"
|
||||
extra="The file has no explicit prefix, and guessing one would change this VIP's netmask on takeover. State it here."
|
||||
rules={[{ required: true, message: 'Required — the file does not state one' }]}>
|
||||
<InputNumber min={1} max={32} style={{ width: 160 }} />
|
||||
</Form.Item>
|
||||
)}
|
||||
<Form.Item name="description" label="Description (optional)">
|
||||
<Input placeholder={`Adopted from ${adoptTarget.discovery.agent_name}`} />
|
||||
</Form.Item>
|
||||
</Form>
|
||||
<Text type="secondary" style={{ fontSize: 12 }}>
|
||||
Config found at <Text code>{adoptTarget.discovery.config_path}</Text> — the VRRP
|
||||
password is masked below and is carried over encrypted.
|
||||
</Text>
|
||||
<pre style={{ marginTop: 8, maxHeight: 220, overflow: 'auto', fontSize: 12,
|
||||
background: 'rgba(127,127,127,0.08)', padding: 8, borderRadius: 4 }}>
|
||||
{adoptTarget.discovery.config_preview || '(not available)'}
|
||||
</pre>
|
||||
</>
|
||||
);
|
||||
})()}
|
||||
</Modal>
|
||||
|
||||
<Modal
|
||||
title={editing ? `Edit VIP — ${editing.name}` : 'Create VIP'}
|
||||
open={modalVisible}
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* v1.10.3 regression tests: with more than one ACME account, the certificate wizard must honour the
|
||||
* account the operator picked — on the Review step AND in the request it submits.
|
||||
*
|
||||
* These drive the real component through all three wizard steps rather than testing a helper,
|
||||
* because the bug was invisible until the wizard ADVANCED PAST the step that owns the account
|
||||
* Select: Form.useWatch reports only fields that are currently rendered, so on Review the selection
|
||||
* read `undefined` and the wizard silently reverted to the default account. A unit test of any
|
||||
* single function would have passed the whole time.
|
||||
*/
|
||||
import React from 'react';
|
||||
import { render, screen, fireEvent, waitFor, act } from '@testing-library/react';
|
||||
import axios from 'axios';
|
||||
import ACMEAutomation from '../ACMEAutomation';
|
||||
|
||||
// These render the whole ACMEAutomation tree (antd Steps + Form + Select) three times per test and
|
||||
// take 4-7s each, so jest's default 5s per-test limit fails two of them. Raise it here rather than
|
||||
// relying on the runner being invoked with --testTimeout, so `npm test` passes as shipped.
|
||||
jest.setTimeout(30000);
|
||||
|
||||
jest.mock('axios');
|
||||
jest.mock('react-router-dom', () => ({ useNavigate: () => jest.fn() }));
|
||||
jest.mock('../../contexts/ClusterContext', () => ({
|
||||
useCluster: () => ({ clusters: [], selectCluster: jest.fn() }),
|
||||
}));
|
||||
|
||||
// The account list arrives ORDER BY id while the backend's default is ORDER BY created_at DESC, so
|
||||
// this fixture is deliberately the shape that made the two disagree: the DNS-01 account is BOTH the
|
||||
// lower id and the older account, the HTTP-01 account is the newest. Picking the first valid entry
|
||||
// (as the UI used to) yields the DNS-01 account; the backend would have used the HTTP-01 one.
|
||||
const DNS_ACCOUNT = {
|
||||
id: 1, email: 'dns@example.com', status: 'valid',
|
||||
challenge_type: 'dns-01', dns_provider: 'godaddy',
|
||||
created_at: '2026-05-06T00:00:00Z', directory_url: 'https://acme.zerossl.com/v2/DV90',
|
||||
};
|
||||
const HTTP_ACCOUNT = {
|
||||
id: 2, email: 'http@example.com', status: 'valid',
|
||||
challenge_type: 'http-01', dns_provider: null,
|
||||
created_at: '2026-08-08T00:00:00Z', directory_url: 'https://acme.zerossl.com/v2/DV90',
|
||||
};
|
||||
|
||||
const GET_ROUTES = {
|
||||
'/api/letsencrypt/orders': [],
|
||||
'/api/letsencrypt/accounts': [DNS_ACCOUNT, HTTP_ACCOUNT],
|
||||
'/api/letsencrypt/renewal-schedule': [],
|
||||
'/api/clusters': { clusters: [{ id: 10, name: 'cluster-a', acme_enabled: true, is_active: true }] },
|
||||
'/api/letsencrypt/prerequisites': { steps: [] },
|
||||
'/api/letsencrypt/dns-providers': {
|
||||
dns01_enabled: true,
|
||||
providers: [
|
||||
{ name: 'manual', label: 'Manual', automated: false, credential_fields: [] },
|
||||
{
|
||||
name: 'godaddy', label: 'GoDaddy', automated: true,
|
||||
credential_fields: [
|
||||
{ key: 'api_key', label: 'API Key', type: 'password', required: true, max_length: 200, help: 'Production key' },
|
||||
{ key: 'api_secret', label: 'API Secret', type: 'password', required: false, max_length: 200, help: 'Blank for a PAT' },
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
jest.clearAllMocks();
|
||||
axios.get.mockImplementation((url) =>
|
||||
Promise.resolve({ data: Object.prototype.hasOwnProperty.call(GET_ROUTES, url) ? GET_ROUTES[url] : {} })
|
||||
);
|
||||
axios.post.mockResolvedValue({ data: { message: 'ok', order_id: 99 } });
|
||||
});
|
||||
|
||||
const modal = () => document.querySelector('.ant-modal-content');
|
||||
|
||||
/** Open the wizard and wait for the Domains step. */
|
||||
async function openWizard() {
|
||||
render(<ACMEAutomation />);
|
||||
// Settle the initial fetch on a signal that does NOT depend on which account the component picks
|
||||
// as its default — that choice is one of the things under test, so waiting on an account address
|
||||
// here would make every test fail at the same early point instead of at its own assertion.
|
||||
await waitFor(() => expect(axios.get).toHaveBeenCalledWith('/api/letsencrypt/accounts'));
|
||||
await act(async () => {});
|
||||
fireEvent.click(screen.getByRole('button', { name: /Request Certificate/i }));
|
||||
await screen.findByText('Domain Names');
|
||||
}
|
||||
|
||||
/** antd wires the Form.Item name onto the inner input's id, which is the only stable handle. */
|
||||
function typeDomain(domain) {
|
||||
const input = document.getElementById('domains');
|
||||
fireEvent.change(input, { target: { value: domain } });
|
||||
fireEvent.keyDown(input, { key: 'Enter', keyCode: 13 });
|
||||
}
|
||||
|
||||
async function selectAccount(email) {
|
||||
await act(async () => {
|
||||
fireEvent.mouseDown(document.getElementById('account_id'));
|
||||
});
|
||||
const option = [...document.querySelectorAll('.ant-select-item-option')]
|
||||
.find((o) => o.textContent.includes(email));
|
||||
if (!option) throw new Error(`account option not found: ${email}`);
|
||||
await act(async () => {
|
||||
fireEvent.click(option);
|
||||
});
|
||||
}
|
||||
|
||||
const next = async () => {
|
||||
await act(async () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Next$/ }));
|
||||
});
|
||||
};
|
||||
const submitButton = () => screen.getByRole('button', { name: /Submit Request/i });
|
||||
|
||||
async function gotoReview({ domain = 'example.com', account } = {}) {
|
||||
await openWizard();
|
||||
typeDomain(domain);
|
||||
await next();
|
||||
// Wait for the Configuration step to actually MOUNT. The transition is async (validateFields
|
||||
// returns a promise) and the text "ACME Account" also appears on the dashboard card behind the
|
||||
// modal, so waiting on that text can resolve while the wizard is still on step 1.
|
||||
await waitFor(() => expect(document.getElementById('account_id')).toBeTruthy());
|
||||
if (account) await selectAccount(account);
|
||||
await next();
|
||||
await screen.findByText('Prerequisite Check');
|
||||
}
|
||||
|
||||
/** The Review step's rendered text. Compared as a whole string on purpose: the assertions must
|
||||
* describe BEHAVIOUR, not the markup this change happens to use, so that a failure means the
|
||||
* wizard resolved the wrong account rather than that a wrapper element moved. */
|
||||
const reviewText = () => modal().textContent;
|
||||
|
||||
async function submitAndGetBody() {
|
||||
fireEvent.click(submitButton());
|
||||
await waitFor(() => expect(axios.post).toHaveBeenCalled());
|
||||
const [url, body] = axios.post.mock.calls[0];
|
||||
expect(url).toBe('/api/letsencrypt/certificates');
|
||||
return body;
|
||||
}
|
||||
|
||||
describe('certificate wizard with multiple ACME accounts', () => {
|
||||
test('an explicitly picked HTTP-01 account survives the step change and is what gets submitted', async () => {
|
||||
await gotoReview({ account: HTTP_ACCOUNT.email });
|
||||
|
||||
// The payload is the real evidence. account_id used to come from the form store while
|
||||
// challenge_type came from an account object that had reverted to the default, so the API got
|
||||
// "HTTP-01 account + dns-01" and answered 422 "no DNS provider configured for DNS-01".
|
||||
const body = await submitAndGetBody();
|
||||
expect(body.account_id).toBe(HTTP_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('http-01');
|
||||
});
|
||||
|
||||
test('the Review step describes the picked HTTP-01 account, not the default', async () => {
|
||||
await gotoReview({ account: HTTP_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain(HTTP_ACCOUNT.email);
|
||||
expect(reviewText()).not.toContain(DNS_ACCOUNT.email);
|
||||
// "Challenge Method" and the provider name only render on the DNS-01 branch.
|
||||
expect(reviewText()).not.toContain('Challenge Method');
|
||||
expect(reviewText()).not.toContain('godaddy');
|
||||
});
|
||||
|
||||
test('an explicitly picked DNS-01 account is described and submitted as DNS-01', async () => {
|
||||
// Positive control: the DNS-01 path must keep working. This one passed before the fix too,
|
||||
// because the default the wizard fell back to happened to be the DNS-01 account.
|
||||
await gotoReview({ account: DNS_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain(DNS_ACCOUNT.email);
|
||||
expect(reviewText()).toContain('Challenge Method');
|
||||
expect(reviewText()).toContain('godaddy');
|
||||
|
||||
const body = await submitAndGetBody();
|
||||
expect(body.account_id).toBe(DNS_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('dns-01');
|
||||
});
|
||||
|
||||
test('with no explicit pick the wizard previews and sends the same default the backend would use', async () => {
|
||||
// The backend takes the NEWEST valid account; the UI used to preview the oldest entry of a
|
||||
// list ordered by id, so Review described an account the request never went to.
|
||||
await gotoReview();
|
||||
|
||||
expect(reviewText()).toContain(HTTP_ACCOUNT.email);
|
||||
expect(reviewText()).not.toContain(DNS_ACCOUNT.email);
|
||||
|
||||
const body = await submitAndGetBody();
|
||||
// Sent explicitly rather than left to the backend to guess a second time.
|
||||
expect(body.account_id).toBe(HTTP_ACCOUNT.id);
|
||||
expect(body.challenge_type).toBe('http-01');
|
||||
});
|
||||
|
||||
test('the wildcard guard still applies on the Review step, where Submit lives', async () => {
|
||||
// Same root cause as the account bug: `domains` is entered on the first step, so a
|
||||
// non-preserving useWatch read undefined from Review onward and the guard evaporated at exactly
|
||||
// the point it had to hold.
|
||||
await gotoReview({ domain: '*.example.com', account: HTTP_ACCOUNT.email });
|
||||
|
||||
expect(reviewText()).toContain('Wildcard requires a DNS-01 account');
|
||||
expect(submitButton()).toBeDisabled();
|
||||
fireEvent.click(submitButton());
|
||||
expect(axios.post).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,274 @@
|
||||
/**
|
||||
* v1.11.0 regression tests for the Request Log page.
|
||||
*
|
||||
* Three properties matter here and none of them are visible from a snapshot:
|
||||
*
|
||||
* 1. Pagination is SERVER-side. Every other table in this app fetches once and
|
||||
* slices in the browser; this table can hold millions of rows, so changing
|
||||
* the page MUST issue a new request with a new offset. A client-side slice
|
||||
* would look identical on a two-row fixture and fall over in production.
|
||||
* 2. The page gates itself on requestlog.read. The sidebar and the router are
|
||||
* unconditional (App.js has no hook access at module level), so this
|
||||
* component is the only gate — and it must not even call the API when the
|
||||
* caller lacks the permission.
|
||||
* 3. Redacted values are what the modal shows. A test that only asserted the
|
||||
* modal opens would still pass if the UI un-redacted anything.
|
||||
*/
|
||||
import React from 'react';
|
||||
import { render, screen, fireEvent, waitFor, act } from '@testing-library/react';
|
||||
import axios from 'axios';
|
||||
import RequestLog from '../RequestLog';
|
||||
|
||||
// Rendering the antd Table + filter row + modal is slow; keep the limit here so
|
||||
// plain `npm test` passes as shipped rather than needing --testTimeout.
|
||||
jest.setTimeout(30000);
|
||||
|
||||
jest.mock('axios');
|
||||
|
||||
let mockPermissions = { read: true, manage: true, admin: false };
|
||||
jest.mock('../../contexts/AuthContext', () => ({
|
||||
useAuth: () => ({
|
||||
hasPermission: (resource, action) =>
|
||||
resource === 'requestlog' && Boolean(mockPermissions[action]),
|
||||
isAdmin: () => mockPermissions.admin,
|
||||
}),
|
||||
}));
|
||||
|
||||
const INBOUND_ROW = {
|
||||
id: 2,
|
||||
request_id: 'abc123',
|
||||
direction: 'inbound',
|
||||
target: null,
|
||||
method: 'POST',
|
||||
url: '/api/letsencrypt/certificates',
|
||||
path: '/api/letsencrypt/certificates',
|
||||
status_code: 500,
|
||||
status_class: 5,
|
||||
duration_ms: 2431,
|
||||
user_id: 1,
|
||||
username: 'admin',
|
||||
client_ip: '10.0.0.5',
|
||||
error: null,
|
||||
request_body_bytes: 120,
|
||||
response_body_bytes: 88,
|
||||
truncated: false,
|
||||
created_at: '2026-08-11T09:00:00+00:00',
|
||||
};
|
||||
|
||||
const OUTBOUND_ROW = {
|
||||
id: 1,
|
||||
request_id: 'abc123',
|
||||
direction: 'outbound',
|
||||
target: 'acme',
|
||||
method: 'POST',
|
||||
url: 'https://acme-v02.api.letsencrypt.org/acme/new-order',
|
||||
path: '/acme/new-order',
|
||||
status_code: 429,
|
||||
status_class: 4,
|
||||
duration_ms: 812,
|
||||
user_id: null,
|
||||
username: null,
|
||||
client_ip: null,
|
||||
error: null,
|
||||
request_body_bytes: 0,
|
||||
response_body_bytes: 210,
|
||||
truncated: false,
|
||||
created_at: '2026-08-11T09:00:01+00:00',
|
||||
};
|
||||
|
||||
const LIST_RESPONSE = {
|
||||
logs: [INBOUND_ROW, OUTBOUND_ROW],
|
||||
total: 2,
|
||||
total_is_estimate: false,
|
||||
limit: 50,
|
||||
offset: 0,
|
||||
scoped_to_self: false,
|
||||
};
|
||||
|
||||
const DETAIL_RESPONSE = {
|
||||
log: {
|
||||
...INBOUND_ROW,
|
||||
query_params: null,
|
||||
request_headers: { 'content-type': 'application/json', authorization: '***REDACTED***' },
|
||||
request_body: { domains: ['example.com'], eab_hmac_key: '***REDACTED***' },
|
||||
response_headers: { 'content-type': 'application/json' },
|
||||
response_body: { error: { message: 'ACME rate limited' } },
|
||||
},
|
||||
related: [OUTBOUND_ROW],
|
||||
};
|
||||
|
||||
const STATS_RESPONSE = {
|
||||
window_hours: 24,
|
||||
by_direction: [
|
||||
{ direction: 'inbound', total: 120, errors: 3, avg_duration_ms: 45, max_duration_ms: 2431 },
|
||||
{ direction: 'outbound', total: 18, errors: 1, avg_duration_ms: 300, max_duration_ms: 812 },
|
||||
],
|
||||
by_status_class: [],
|
||||
by_target: [],
|
||||
total_rows: 138,
|
||||
oldest_at: '2026-08-04T09:00:00+00:00',
|
||||
newest_at: '2026-08-11T09:00:01+00:00',
|
||||
sink: { queued: 0, queue_capacity: 2000, written: 138, dropped: 0, failed_batches: 0, running: 1 },
|
||||
retention: { success_retention_days: 7, error_retention_days: 30, max_rows: 500000 },
|
||||
};
|
||||
|
||||
const listCalls = () => axios.get.mock.calls.filter((c) => c[0] === '/api/request-logs');
|
||||
|
||||
beforeEach(() => {
|
||||
jest.clearAllMocks();
|
||||
mockPermissions = { read: true, manage: true, admin: false };
|
||||
axios.get.mockImplementation((url) => {
|
||||
if (url === '/api/request-logs') return Promise.resolve({ data: LIST_RESPONSE });
|
||||
if (url === '/api/request-logs/stats') return Promise.resolve({ data: STATS_RESPONSE });
|
||||
if (url.startsWith('/api/request-logs/')) return Promise.resolve({ data: DETAIL_RESPONSE });
|
||||
return Promise.resolve({ data: {} });
|
||||
});
|
||||
axios.post.mockResolvedValue({ data: { removed: { success: 1, error: 0, overflow: 0 } } });
|
||||
});
|
||||
|
||||
async function renderPage() {
|
||||
render(<RequestLog />);
|
||||
await waitFor(() => expect(listCalls().length).toBeGreaterThan(0));
|
||||
await act(async () => {});
|
||||
}
|
||||
|
||||
test('renders both directions of a request from the list endpoint', async () => {
|
||||
await renderPage();
|
||||
|
||||
expect(await screen.findByText('/api/letsencrypt/certificates')).toBeInTheDocument();
|
||||
expect(screen.getByText('https://acme-v02.api.letsencrypt.org/acme/new-order')).toBeInTheDocument();
|
||||
// Inbound shows the user; outbound shows the target it called.
|
||||
expect(screen.getByText('admin')).toBeInTheDocument();
|
||||
expect(screen.getByText('acme')).toBeInTheDocument();
|
||||
expect(screen.getByText('500')).toBeInTheDocument();
|
||||
expect(screen.getByText('429')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('a caller without requestlog.read sees a denial and the API is never called', async () => {
|
||||
mockPermissions = { read: false, manage: false, admin: false };
|
||||
|
||||
render(<RequestLog />);
|
||||
|
||||
expect(await screen.findByText('Access denied')).toBeInTheDocument();
|
||||
expect(axios.get).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test('an admin without an explicit grant still gets in', async () => {
|
||||
mockPermissions = { read: false, manage: false, admin: true };
|
||||
await renderPage();
|
||||
expect(await screen.findByText('/api/letsencrypt/certificates')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('the errors-only switch is sent to the server, not applied in the browser', async () => {
|
||||
await renderPage();
|
||||
const before = listCalls().length;
|
||||
|
||||
await act(async () => {
|
||||
fireEvent.click(document.querySelector('.ant-switch'));
|
||||
});
|
||||
|
||||
await waitFor(() => expect(listCalls().length).toBeGreaterThan(before));
|
||||
const params = listCalls()[listCalls().length - 1][1].params;
|
||||
expect(params.errors_only).toBe(true);
|
||||
});
|
||||
|
||||
test('the direction filter is sent as a query parameter', async () => {
|
||||
await renderPage();
|
||||
const before = listCalls().length;
|
||||
|
||||
await act(async () => {
|
||||
fireEvent.mouseDown(document.querySelectorAll('.ant-select-selector')[0]);
|
||||
});
|
||||
await act(async () => {
|
||||
const option = Array.from(document.querySelectorAll('.ant-select-item-option'))
|
||||
.find((el) => el.textContent.includes('Outbound'));
|
||||
fireEvent.click(option);
|
||||
});
|
||||
|
||||
await waitFor(() => expect(listCalls().length).toBeGreaterThan(before));
|
||||
const params = listCalls()[listCalls().length - 1][1].params;
|
||||
expect(params.direction).toBe('outbound');
|
||||
});
|
||||
|
||||
test('the first request asks for a bounded page, not the whole table', async () => {
|
||||
await renderPage();
|
||||
const params = listCalls()[0][1].params;
|
||||
expect(params.limit).toBe(50);
|
||||
expect(params.offset).toBe(0);
|
||||
});
|
||||
|
||||
test('opening a row fetches the detail and shows the REDACTED body', async () => {
|
||||
await renderPage();
|
||||
|
||||
await act(async () => {
|
||||
fireEvent.click(screen.getAllByText('Detail')[0]);
|
||||
});
|
||||
|
||||
await waitFor(() =>
|
||||
expect(axios.get).toHaveBeenCalledWith(expect.stringMatching(/\/api\/request-logs\/\d+$/))
|
||||
);
|
||||
|
||||
const modal = await waitFor(() => document.querySelector('.ant-modal-content'));
|
||||
expect(modal.textContent).toContain('***REDACTED***');
|
||||
// The redaction happens server-side; the UI must not attempt to show a raw value.
|
||||
expect(modal.textContent).not.toContain('eab_hmac_key":"');
|
||||
expect(modal.textContent).toContain('ACME rate limited');
|
||||
});
|
||||
|
||||
test('the detail modal lists the outbound calls triggered by the same request', async () => {
|
||||
await renderPage();
|
||||
|
||||
await act(async () => {
|
||||
fireEvent.click(screen.getAllByText('Detail')[0]);
|
||||
});
|
||||
|
||||
const modal = await waitFor(() => document.querySelector('.ant-modal-content'));
|
||||
await waitFor(() => expect(modal.textContent).toContain('Calls triggered by this request'));
|
||||
expect(modal.textContent).toContain('acme');
|
||||
});
|
||||
|
||||
test('a self-scoped response explains why the list is narrower', async () => {
|
||||
axios.get.mockImplementation((url) => {
|
||||
if (url === '/api/request-logs') {
|
||||
return Promise.resolve({ data: { ...LIST_RESPONSE, scoped_to_self: true } });
|
||||
}
|
||||
if (url === '/api/request-logs/stats') return Promise.resolve({ data: STATS_RESPONSE });
|
||||
return Promise.resolve({ data: {} });
|
||||
});
|
||||
|
||||
await renderPage();
|
||||
expect(await screen.findByText('Showing your own requests only')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('dropped rows are surfaced as a warning', async () => {
|
||||
axios.get.mockImplementation((url) => {
|
||||
if (url === '/api/request-logs') return Promise.resolve({ data: LIST_RESPONSE });
|
||||
if (url === '/api/request-logs/stats') {
|
||||
return Promise.resolve({ data: { ...STATS_RESPONSE, sink: { ...STATS_RESPONSE.sink, dropped: 42 } } });
|
||||
}
|
||||
return Promise.resolve({ data: {} });
|
||||
});
|
||||
|
||||
await renderPage();
|
||||
expect(await screen.findByText(/42 row\(s\) dropped/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('a failed load surfaces the server message instead of a blank table', async () => {
|
||||
axios.get.mockImplementation((url) => {
|
||||
if (url === '/api/request-logs') {
|
||||
return Promise.reject({
|
||||
response: { status: 500, data: { error: { message: 'Failed to list request logs' } } },
|
||||
});
|
||||
}
|
||||
return Promise.resolve({ data: STATS_RESPONSE });
|
||||
});
|
||||
|
||||
await renderPage();
|
||||
expect(await screen.findByText('Failed to list request logs')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
test('the retention button is hidden without requestlog.manage', async () => {
|
||||
mockPermissions = { read: true, manage: false, admin: false };
|
||||
await renderPage();
|
||||
expect(screen.queryByText('Apply retention now')).not.toBeInTheDocument();
|
||||
});
|
||||
@@ -0,0 +1,26 @@
|
||||
// Loaded automatically by react-scripts test (CRA convention).
|
||||
import '@testing-library/jest-dom';
|
||||
|
||||
// jsdom implements neither of these, and Ant Design 5 needs both: rc-select renders its dropdown
|
||||
// through rc-virtual-list (ResizeObserver) and the responsive Grid reads matchMedia. Without the
|
||||
// polyfills any test that opens a Select throws before it can assert anything.
|
||||
if (!global.ResizeObserver) {
|
||||
global.ResizeObserver = class {
|
||||
observe() {}
|
||||
unobserve() {}
|
||||
disconnect() {}
|
||||
};
|
||||
}
|
||||
|
||||
if (!window.matchMedia) {
|
||||
window.matchMedia = (query) => ({
|
||||
matches: false,
|
||||
media: query,
|
||||
onchange: null,
|
||||
addListener: () => {},
|
||||
removeListener: () => {},
|
||||
addEventListener: () => {},
|
||||
removeEventListener: () => {},
|
||||
dispatchEvent: () => false,
|
||||
});
|
||||
}
|
||||
Reference in New Issue
Block a user