Compare commits

..

2 Commits

Author SHA1 Message Date
lebaudantoine 7ea0cd4145 🔒️(summary) redact meeting content from Sentry events
Sentry attaches stack frame locals to its events. When storing a
transcript in S3 failed, the full transcript held in `data` and
`transcript` was sent to Sentry.

Keep locals for debugging, but scrub variables and nested dict keys
known to hold transcripts, summaries, LLM prompts, participants'
personal data or pre-signed URLs. Share the Sentry init between the
API and the Celery worker, and never send request bodies.
2026-10-01 19:49:13 +02:00
lebaudantoine 7b1593c3c0 🐛(summary) disable default S3 checksums for GCS-compatible storage
Since boto3/botocore 1.36, the S3 client computes CRC32 checksums on
uploads by default (request_checksum_calculation="when_supported").
PutObject requests are then sent with aws-chunked encoding, a trailing
x-amz-checksum-crc32 header and a STREAMING-UNSIGNED-PAYLOAD-TRAILER
content hash.

Our production storage (S3NS, storage.s3nsapis.fr) is built on Google
Cloud Storage and exposes it through GCS's S3-compatible XML API, which
does not support these flexible checksums. It rejects the request with
a 403 SignatureDoesNotMatch ("Invalid argument"), so storing transcripts
failed in the Celery worker. Garage, used locally, supports them, which
is why the issue only appeared in production after switching the
client to boto3.

Add aws_request_checksum_calculation and
aws_response_checksum_validation settings, defaulting to
"when_required", and pass them to the botocore Config of the summary S3
client and the backend S3 client. Checksums are then only sent for
operations that require them, restoring the pre-1.36 behavior.
2026-10-01 19:37:39 +02:00
11 changed files with 349 additions and 17 deletions
+2
View File
@@ -19,6 +19,8 @@ and this project adheres to
- 🔒️(agents) fix util-linux CVEs reported by Cyberwatch
- 🔒️(backend) fix HIGH CVEs in Django and urllib3
- 🔒️(agents) upgrade libpcre2-8-0 to fix CVE-2026-103111
- 🐛(summary) disable default S3 checksums for GCS-compatible storage
- 🔒️(summary) redact meeting content from Sentry events
## [1.33.0] - 2026-09-30
+3 -2
View File
@@ -37,13 +37,14 @@ RUN --mount=type=cache,target=/root/.cache/uv \
uv sync --locked --no-dev
# ---- mails ----
FROM node:22-alpine AS mail-builder
FROM node:22 AS mail-builder
COPY ./src/mail /mail/app
WORKDIR /mail/app
RUN npm ci --ignore-scripts && npm run build
RUN yarn install --frozen-lockfile && \
yarn build
# ---- static link collector ----
+2 -2
View File
@@ -172,7 +172,7 @@ services:
working_dir: /app
node:
image: node:22-alpine
image: node:22
user: "${DOCKER_USER:-1000}"
environment:
HOME: /tmp
@@ -271,7 +271,7 @@ services:
- /app/.venv
redis-summary:
image: redis:5
image: redis
ports:
- "6379:6379"
+1 -1
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env bash
set -e pipefail
set -eo pipefail
# Run html-to-text to convert all html files to text files
DIR_MAILS="../backend/core/templates/mail/"
+2 -2
View File
@@ -9,8 +9,8 @@
},
"private": true,
"scripts": {
"build-mjml-to-html": "sh ./bin/mjml-to-html",
"build-html-to-plain-text": "sh ./bin/html-to-plain-text",
"build-mjml-to-html": "bash ./bin/mjml-to-html",
"build-html-to-plain-text": "bash ./bin/html-to-plain-text",
"build": "npm run build-mjml-to-html && npm run build-html-to-plain-text"
},
"volta": {
+5 -6
View File
@@ -9,7 +9,6 @@ from typing import Any
from urllib.parse import urljoin
import requests
import sentry_sdk
from celery import Celery, signals
from celery.utils.log import get_task_logger
from openai.types.audio import Transcription
@@ -42,6 +41,7 @@ from summary.core.prompt import (
PROMPT_SYSTEM_TLDR,
PROMPT_USER_PART,
)
from summary.core.sentry import init_sentry
from summary.core.shared_models import (
SummarizeWebhookFailurePayload,
SummarizeWebhookSuccessPayload,
@@ -75,12 +75,11 @@ celery = Celery(
celery.config_from_object("summary.core.celery_config")
if settings.sentry_dsn and settings.sentry_is_enabled:
@signals.celeryd_init.connect
def init_sentry(**_kwargs):
"""Initialize sentry."""
sentry_sdk.init(dsn=settings.sentry_dsn, enable_tracing=True)
@signals.celeryd_init.connect
def init_celery_sentry(**_kwargs):
"""Initialize Sentry in the Celery worker."""
init_sentry()
file_service = FileService()
+2
View File
@@ -84,6 +84,8 @@ class Settings(BaseSettings):
aws_s3_secret_access_key: SecretStr
aws_s3_secure_access: bool = True
aws_s3_region_name: str | None = None
aws_s3_request_checksum_calculation: str | None = None
aws_s3_response_checksum_validation: str | None = None
aws_transcript_path: str = "transcripts"
aws_summary_path: str = "summaries"
+6 -1
View File
@@ -286,7 +286,12 @@ def _build_s3_client():
aws_access_key_id=settings.aws_s3_access_key_id,
aws_secret_access_key=settings.aws_s3_secret_access_key.get_secret_value(),
region_name=settings.aws_s3_region_name,
config=Config(signature_version="s3v4", s3={"addressing_style": "path"}),
config=Config(
signature_version="s3v4",
s3={"addressing_style": "path"},
request_checksum_calculation=settings.aws_s3_request_checksum_calculation,
response_checksum_validation=settings.aws_s3_response_checksum_validation,
),
)
+90
View File
@@ -0,0 +1,90 @@
"""Sentry configuration."""
import sentry_sdk
from sentry_sdk.scrubber import DEFAULT_DENYLIST, DEFAULT_PII_DENYLIST, EventScrubber
from summary.core.config import get_settings
# Exact names (case-insensitive) of variables and dict keys to redact.
SENSITIVE_DATA_DENYLIST = [
# Raw payloads and serialized bodies
"data",
"body",
"payload",
"args",
"kwargs",
"response",
"res",
# Transcripts
"transcript",
"transcription",
"transcription_json",
"transcription_res",
"new_transcription",
"segments",
"word_segments",
"words",
"text",
"content",
"formatted_output",
# Summaries and LLM exchanges
"summary",
"raw_summary",
"cleaned_summary",
"tldr",
"part",
"parts",
"parts_summarized",
"next_steps",
"title",
"titles",
"action",
"line",
"lines",
"user_prompt",
"prompt_user_part",
"messages",
"json_data", # OpenAI client internals
"opts",
"options",
"input_options",
# Participants' personal data
"email",
"user_email",
"assignees",
"participant_name",
"participant_names",
"participants_info",
"speaker_to_name",
# Signed / pre-authenticated URLs
"cloud_storage_url",
"transcription_data_url",
"summary_data_url",
]
def build_event_scrubber() -> EventScrubber:
"""Build the scrubber redacting meeting content and personal data."""
return EventScrubber(
denylist=DEFAULT_DENYLIST + SENSITIVE_DATA_DENYLIST,
pii_denylist=DEFAULT_PII_DENYLIST,
recursive=True,
)
def init_sentry() -> None:
"""Initialize Sentry if enabled in the settings."""
settings = get_settings()
if not (settings.sentry_dsn and settings.sentry_is_enabled):
return
sentry_sdk.init(
dsn=settings.sentry_dsn,
enable_tracing=True,
# Never attach request bodies, Celery task arguments or user data.
send_default_pii=False,
# Task creation requests carry the content to summarize.
max_request_body_size="never",
include_local_variables=True,
event_scrubber=build_event_scrubber(),
)
+2 -3
View File
@@ -1,18 +1,17 @@
"""Application."""
import sentry_sdk
from dockerflow.fastapi import router as dockerflow_router
from fastapi import FastAPI
from summary.api.main import api_router_v2
from summary.core import checks # noqa: F401 -- registers the Dockerflow checks
from summary.core.config import get_settings
from summary.core.sentry import init_sentry
settings = get_settings()
if settings.sentry_dsn and settings.sentry_is_enabled:
sentry_sdk.init(dsn=settings.sentry_dsn, enable_tracing=True)
init_sentry()
app = FastAPI(
title=settings.app_name,
+234
View File
@@ -0,0 +1,234 @@
"""Tests for the Sentry configuration.
Each test raises an error from code handling meeting content and inspects the
event Sentry would send: the content must be redacted, while harmless local
variables are kept for debugging.
"""
import json
from collections.abc import Callable, Iterator
from unittest.mock import Mock
import httpx
import openai
import pytest
import sentry_sdk
from botocore.exceptions import ClientError
from botocore.stub import Stubber
from sentry_sdk.transport import Transport
from summary.core import file_service
from summary.core import sentry as sentry_module
from summary.core.file_service import FileService
from summary.core.llm_service import LLMException, LLMService
from summary.core.shared_models import WhisperXResponse
CANARY = "CANARY-MEETING-CONTENT"
SentryEvents = Callable[[], list[str]]
class _CapturingTransport(Transport):
"""Keep serialized events in memory instead of sending them."""
def __init__(self):
super().__init__()
self.events: list[str] = []
def capture_envelope(self, envelope):
"""Store each serialized event of the envelope."""
for item in envelope.items:
if item.type == "event":
self.events.append(item.payload.get_bytes().decode())
@pytest.fixture
def sentry_events() -> Iterator[SentryEvents]:
"""Initialize Sentry as in production, with an in-memory transport."""
transport = _CapturingTransport()
sentry_sdk.init(
dsn="https://public@sentry.example.com/1",
transport=transport,
send_default_pii=False,
include_local_variables=True,
event_scrubber=sentry_module.build_event_scrubber(),
default_integrations=False,
)
def flush() -> list[str]:
sentry_sdk.flush()
return transport.events
yield flush
sentry_sdk.init() # Disable Sentry for the following tests
def _transcript() -> WhisperXResponse:
return WhisperXResponse.model_validate(
{
"segments": [
{
"start": 0.0,
"end": 1.0,
"text": f"I don't know {CANARY}",
"speaker": "SPEAKER_01",
"words": [
{
"word": CANARY,
"start": 0.0,
"end": 1.0,
"score": 0.9,
"speaker": "SPEAKER_01",
}
],
}
]
}
)
def _local_vars(event: str) -> list[dict]:
"""Return the local variables of every frame of the event."""
return [
frame.get("vars", {})
for exception in json.loads(event)["exception"]["values"]
for frame in exception["stacktrace"]["frames"]
]
@pytest.fixture
def s3_stubber(monkeypatch: pytest.MonkeyPatch) -> Iterator[Stubber]:
"""Stub the S3 client built by the file service."""
monkeypatch.setattr(
file_service,
"settings",
file_service.settings.model_copy(
update={
"aws_s3_endpoint_url": "garage:9000",
"aws_s3_secure_access": False,
"aws_s3_region_name": "fr-par",
"aws_storage_bucket_name": "meet-media-storage",
}
),
)
stubber = Stubber(file_service._build_s3_client())
stubber.activate()
monkeypatch.setattr(file_service, "_build_s3_client", lambda: stubber.client)
yield stubber
stubber.assert_no_pending_responses()
def test_sentry_store_transcript_failure_redacts_transcript(
sentry_events: SentryEvents, s3_stubber: Stubber
) -> None:
"""A failed S3 upload does not send the transcript, but keeps the job id."""
s3_stubber.add_client_error("put_object", service_error_code="InvalidDigest")
try:
FileService().store_transcript(transcript=_transcript(), job_id="job-1")
except ClientError:
sentry_sdk.capture_exception()
else:
pytest.fail("store_transcript should have failed")
[event] = sentry_events()
assert CANARY not in event
store_transcript_vars = next(
frame_vars
for frame_vars in _local_vars(event)
if "transcript_path" in frame_vars
)
assert store_transcript_vars["job_id"] == "'job-1'"
assert store_transcript_vars["transcript_path"] == "'transcripts/job-1.json'"
assert store_transcript_vars["data"] == "[Filtered]"
assert store_transcript_vars["transcript"] == "[Filtered]"
def test_sentry_redacts_nested_content_in_dicts(sentry_events: SentryEvents) -> None:
"""Content nested in dicts held by harmless names is redacted too."""
def process(job_id: str, task_payload: dict, dumped: dict) -> None:
raise RuntimeError("boom")
try:
process(
job_id="job-1",
task_payload={"tenant_id": "tenant-1", "content": CANARY},
dumped=_transcript().model_dump(),
)
except RuntimeError:
sentry_sdk.capture_exception()
[event] = sentry_events()
assert CANARY not in event
process_vars = next(
frame_vars for frame_vars in _local_vars(event) if "task_payload" in frame_vars
)
assert process_vars["job_id"] == "'job-1'"
assert process_vars["task_payload"]["tenant_id"] == "'tenant-1'"
assert process_vars["task_payload"]["content"] == "[Filtered]"
assert process_vars["dumped"]["segments"] == "[Filtered]"
def test_sentry_llm_failure_redacts_prompts(sentry_events: SentryEvents) -> None:
"""A failed LLM call does not send the prompts, even from OpenAI internals."""
client = openai.OpenAI(
api_key="test-key",
base_url="https://llm.example.com/v1",
max_retries=0,
http_client=httpx.Client(
transport=httpx.MockTransport(lambda request: httpx.Response(500))
),
)
observability = Mock(is_enabled=False)
observability.get_openai_client.return_value = client
try:
LLMService(observability).call(
system_prompt="Summarize this meeting.",
user_prompt=f"Transcript: {CANARY}",
name="tldr",
)
except LLMException:
sentry_sdk.capture_exception()
else:
pytest.fail("the LLM call should have failed")
[event] = sentry_events()
assert CANARY not in event
def test_init_sentry_uses_the_event_scrubber(monkeypatch: pytest.MonkeyPatch) -> None:
"""Sentry is initialized with local variables and the content scrubber."""
settings = sentry_module.get_settings().model_copy(
update={"sentry_is_enabled": True, "sentry_dsn": "https://k@example.com/1"}
)
monkeypatch.setattr(sentry_module, "get_settings", lambda: settings)
init = Mock()
monkeypatch.setattr(sentry_module.sentry_sdk, "init", init)
sentry_module.init_sentry()
init.assert_called_once()
kwargs = init.call_args.kwargs
assert kwargs["send_default_pii"] is False
assert kwargs["max_request_body_size"] == "never"
assert kwargs["include_local_variables"] is True
denylist = kwargs["event_scrubber"].denylist
assert {"data", "transcript", "content", "summary", "password"} <= set(denylist)
def test_init_sentry_disabled(monkeypatch: pytest.MonkeyPatch) -> None:
"""Sentry is not initialized when disabled."""
settings = sentry_module.get_settings().model_copy(
update={"sentry_is_enabled": False, "sentry_dsn": "https://k@example.com/1"}
)
monkeypatch.setattr(sentry_module, "get_settings", lambda: settings)
init = Mock()
monkeypatch.setattr(sentry_module.sentry_sdk, "init", init)
sentry_module.init_sentry()
init.assert_not_called()