diff --git a/.env.example b/.env.example index 8aba3b2..0dca713 100644 --- a/.env.example +++ b/.env.example @@ -216,3 +216,27 @@ KEYCLOAK_ADMIN_CLIENT_IDS=["opentaberna-admin-ui"] # How long signing keys are cached before being refetched. Keycloak rotates # keys, so this must expire rather than being fetched once at startup. KEYCLOAK_JWKS_CACHE_SECONDS=300 + +# ---------------------------------- +# Analytics / reporting +# ---------------------------------- +# IANA timezone the shop trades in. Analytics buckets days in this zone rather +# than UTC, so "today" matches the operator's day. +SHOP_TIMEZONE=Europe/Berlin + +# Accept anonymous shopper events from the storefront. Off by default: cloning +# OpenTaberna must not silently start collecting anything. While false the +# ingest endpoint returns 404. +STOREFRONT_ANALYTICS_ENABLED=false + +# ---------------------------------- +# OpenTelemetry +# ---------------------------------- +# Export traces and metrics over OTLP. Off by default, for the same reason. +OTEL_ENABLED=false + +# The seam: pointing this at a vendor's collector is the whole change needed to +# use one, because no application code imports a vendor SDK. +OTEL_EXPORTER_OTLP_ENDPOINT=http://opentaberna-otel-collector:4318 +OTEL_SERVICE_NAME=opentaberna-api +OTEL_METRIC_EXPORT_INTERVAL_SECONDS=30 diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index 6d24317..da2d22e 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -11,6 +11,8 @@ services: KEYCLOAK_URL: http://opentaberna-keycloak:8080 KEYCLOAK_PUBLIC_URL: http://localhost:8080 STORAGE_ENDPOINT_URL: http://opentaberna-minio:9000 + OTEL_EXPORTER_OTLP_ENDPOINT: http://opentaberna-otel-collector:4318 + OTEL_SERVICE_NAME: opentaberna-api volumes: - stripe_webhook_secret:/run/secrets:ro ports: @@ -178,6 +180,8 @@ services: environment: REDIS_URL: redis://opentaberna-redis:6379/0 STORAGE_ENDPOINT_URL: http://opentaberna-minio:9000 + OTEL_EXPORTER_OTLP_ENDPOINT: http://opentaberna-otel-collector:4318 + OTEL_SERVICE_NAME: opentaberna-worker restart: unless-stopped container_name: opentaberna-worker healthcheck: @@ -207,6 +211,61 @@ services: - "8081:8080" # GreenMail web UI restart: unless-stopped + # -------------------------------------------------------------------- + # Observability (S3). Self-hosted by default; the collector is the seam, + # so pointing OTEL_EXPORTER_OTLP_ENDPOINT at a vendor replaces all three + # without touching application code. + # -------------------------------------------------------------------- + + opentaberna-otel-collector: + image: otel/opentelemetry-collector-contrib:0.116.1 + command: ["--config=/etc/otel/config.yaml"] + volumes: + - ./src/docker/observability/otel-collector.yaml:/etc/otel/config.yaml:ro + ports: + - "4318:4318" # OTLP/HTTP + - "8889:8889" # Prometheus scrape target + restart: unless-stopped + container_name: opentaberna-otel-collector + + opentaberna-prometheus: + image: prom/prometheus:v3.1.0 + command: + - "--config.file=/etc/prometheus/prometheus.yml" + - "--storage.tsdb.retention.time=15d" + volumes: + - ./src/docker/observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro + - prometheus_data:/prometheus + ports: + - "9090:9090" + restart: unless-stopped + container_name: opentaberna-prometheus + depends_on: + - opentaberna-otel-collector + + opentaberna-grafana: + image: grafana/grafana:11.4.0 + environment: + # Development only. The dashboard is read-only and carries no secrets, + # and requiring a login to look at a latency graph on your own laptop + # helps nobody. + GF_AUTH_ANONYMOUS_ENABLED: "true" + GF_AUTH_ANONYMOUS_ORG_ROLE: Admin + GF_AUTH_DISABLE_LOGIN_FORM: "true" + GF_USERS_DEFAULT_THEME: light + volumes: + - ./src/docker/observability/grafana/datasources:/etc/grafana/provisioning/datasources:ro + - ./src/docker/observability/grafana/dashboards:/etc/grafana/provisioning/dashboards:ro + - grafana_data:/var/lib/grafana + ports: + - "3001:3000" + restart: unless-stopped + container_name: opentaberna-grafana + depends_on: + - opentaberna-prometheus + volumes: keycloak_data: stripe_webhook_secret: + prometheus_data: + grafana_data: diff --git a/docs/observability.md b/docs/observability.md new file mode 100644 index 0000000..8385428 --- /dev/null +++ b/docs/observability.md @@ -0,0 +1,133 @@ +# Observability + +OpenTelemetry traces and metrics from the API and the worker, exported over OTLP. + +Before this the API had structured logs and correlation IDs and nothing else. +"The shop feels slow" could not be answered with anything but a guess, and a +regression in one endpoint stayed invisible until somebody reported it. + +## OTLP is the seam + +No application module imports a vendor SDK. Everything speaks OTLP to a +collector, and the collector decides where telemetry goes. Using Datadog or +Grafana Cloud instead of the bundled stack is a change to +`OTEL_EXPORTER_OTLP_ENDPOINT` and the collector's config — not a change to any +service. + +``` +API ────┐ + ├──▶ OTel Collector ──▶ Prometheus ──▶ Grafana +Worker ─┘ (:4318) (:9090) (:3001) +``` + +## Off by default + +`OTEL_ENABLED` defaults to `false`, and while it is off `setup()` returns before +creating an exporter — so a deployment that has not opted in opens no +connection and sends nothing anywhere. + +## It must never take the application down + +Every step of the wiring is wrapped. A collector that is absent, unreachable or +misconfigured produces a warning and a running API, not a failed start. +Observability that can cause the outage it exists to diagnose is a bad trade, +and the tests pin this: an exporter that raises on construction still leaves +`setup()` returning `False` rather than propagating. + +## What is instrumented + +| Source | Gives you | +|---|---| +| FastAPI | Request rate, latency histogram, status codes, in-flight requests | +| SQLAlchemy | Query spans and connection pool usage | +| Redis | Command spans | +| httpx | Outbound calls — Stripe, DHL, Keycloak | + +Health endpoints are excluded. A liveness probe every few seconds would +otherwise dominate the trace volume and the request-rate metric, burying real +traffic under a heartbeat. + +## Business gauges + +`Deployment.md` names the queue states worth alerting on. They used to be +queries an operator had to remember to run, which means nobody ran them and the +first sign of a stalled pipeline was a customer asking where their parcel was. + +| Metric | Non-zero means | +|---|---| +| `opentaberna.outbox.pending` | Rising: the worker is not running | +| `opentaberna.outbox.failed` | Events never reached the queue — Redis or the poller | +| `opentaberna.outbox.dead` | Jobs ran and gave up — usually the carrier API | +| `opentaberna.webhooks.unprocessed` | Payments arriving, not handled. **Page on this.** | +| `opentaberna.orders.awaiting_shipment` | The work queue, not necessarily a fault | + +**These are collected by the worker, on a 30-second cron.** The first version +used observable gauges whose callbacks run on the metrics SDK's own thread — and +the only database engine here is asynchronous, so driving it from outside the +event loop failed. The gauges registered cleanly and then silently produced +nothing, which is the worst kind of monitoring bug. The worker already runs a +scheduler and holds an async session, and is the process that most needs to be +alive for these numbers to matter. + +## Traces and logs are joined + +Every log record carries `trace_id` alongside `correlation_id`. Without it the +two systems describe the same request and cannot be put side by side — you would +find a slow span in Grafana and have no way to reach the log lines explaining it. +The field is empty when tracing is off, so it is always present and a formatter +never raises. + +## Running it + +```bash +# in .env +OTEL_ENABLED=true + +docker compose -f docker-compose.dev.yml up -d +``` + +| Service | URL | +|---|---| +| Grafana | http://localhost:3001 — anonymous, dashboard provisioned | +| Prometheus | http://localhost:9090 | +| Collector metrics | http://localhost:8889/metrics | + +Grafana is anonymous **in the development compose file only**. The dashboard is +read-only and holds no secrets, and requiring a login to look at a latency graph +on your own laptop helps nobody. Do not copy that setting to production. + +## The dashboard + +`OpenTaberna — Health`, provisioned from +`src/docker/observability/grafana/dashboards/`. Three rows: the queue gauges +above, request rate/error rate/latency percentiles, and dependency health. + +Two details worth keeping if you edit it: + +**The error-rate panel ends in `or vector(0)`.** Without it a healthy shop shows +"No data", which is indistinguishable from a broken scrape — and is exactly the +wrong thing to be uncertain about during an incident. + +**Latency panels use p95 and p99, never an average.** An average hides the slow +tail that customers actually notice. + +The FastAPI instrumentation labels the path as `http_target`, not `http_route`. +Grouping by `http_route` silently collapses every route into one unnamed series, +which looks like a working panel. That mistake is already made and fixed here. + +## Production + +- Set `OTEL_EXPORTER_OTLP_ENDPOINT` to your collector. +- Set `OTEL_SERVICE_NAME` per process — the worker overrides it in compose. +- Do not expose Grafana anonymously. +- Alert on `webhooks_unprocessed` and `outbox_failed` first: both mean money has + moved and the system has not noticed. + +## Testing + +- `tests/test_telemetry_unit.py` — off unless opted in, idempotent setup, and + every failure mode degrading to "no telemetry" rather than "no API". +- `tests/test_telemetry_integration.py` — asserts against the collector's + output and Prometheus, not against the fact that setup was called. That + distinction caught the real bug: instrumenting the app before configuring + telemetry logged a clean start and produced no HTTP metrics at all. diff --git a/pyproject.toml b/pyproject.toml index b0accc7..8ce9d3f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -36,6 +36,12 @@ dependencies = [ "sqlalchemy[asyncio]>=2.0.49", "stripe>=15.0.1", "uvicorn>=0.43.0", + "opentelemetry-sdk>=1.44.0", + "opentelemetry-exporter-otlp-proto-http>=1.44.0", + "opentelemetry-instrumentation-fastapi>=0.65b0", + "opentelemetry-instrumentation-sqlalchemy>=0.65b0", + "opentelemetry-instrumentation-redis>=0.65b0", + "opentelemetry-instrumentation-httpx>=0.65b0", ] [project.optional-dependencies] diff --git a/src/app/chore/lifespan.py b/src/app/chore/lifespan.py index b0d38bc..7b05d6d 100644 --- a/src/app/chore/lifespan.py +++ b/src/app/chore/lifespan.py @@ -14,6 +14,8 @@ from app.shared.database.base import Base from app.shared.database.engine import close_database, get_engine, init_database from app.shared.logger import get_logger +from app.shared.observability import instrument_engine +from app.shared.observability import setup as setup_telemetry from app.shared.storage.minio_adapter import build_minio_adapter logger = get_logger(__name__) @@ -43,9 +45,16 @@ async def lifespan(app: FastAPI): # Startup: validate secrets before doing anything else _validate_critical_secrets() + # Telemetry before anything else, so startup itself is traced. A disabled + # or unreachable collector logs a warning and the API starts regardless — + # observability must not be able to cause the outage it exists to diagnose. + settings = get_settings() + setup_telemetry(settings) + # Startup: Initialize database and create tables await init_database() engine = get_engine() + instrument_engine(engine, settings) async with engine.begin() as conn: # This creates all tables from SQLAlchemy models that inherit from Base await conn.run_sync(Base.metadata.create_all) diff --git a/src/app/main.py b/src/app/main.py index 0eb5d07..ab0f014 100644 --- a/src/app/main.py +++ b/src/app/main.py @@ -21,6 +21,8 @@ from app.shared.exceptions import AppException, InternalError from app.shared.logger import get_logger from app.shared.middleware import CorrelationIDMiddleware +from app.shared.observability import instrument_app +from app.shared.observability import setup as setup_telemetry from app.shared.rate_limit import limiter from app.shared.responses import ErrorResponse, ValidationErrorResponse from app.shared.config import get_settings @@ -127,6 +129,16 @@ async def generic_exception_handler(request: Request, exc: Exception) -> JSONRes origins = ["*"] # Consider restricting this in a production environment +# Telemetry must be configured before instrumenting, and this module is +# imported long before lifespan startup runs — instrumenting first silently +# produced no HTTP metrics at all. setup() is idempotent, so lifespan calling +# it again is harmless. +setup_telemetry(_settings) + +# Traces HTTP requests. Health endpoints are excluded inside instrument_app: +# a liveness probe every few seconds would otherwise bury real traffic. +instrument_app(app, _settings) + app.add_middleware(CorrelationIDMiddleware) app.add_middleware(SlowAPIMiddleware) app.add_exception_handler(RateLimitExceeded, _rate_limit_exceeded_handler) diff --git a/src/app/shared/config/settings.py b/src/app/shared/config/settings.py index 80b589c..76f2202 100644 --- a/src/app/shared/config/settings.py +++ b/src/app/shared/config/settings.py @@ -288,6 +288,31 @@ class Settings(BaseSettings): description="Default label format requested from DHL: 'pdf' or 'zpl'", ) + # OpenTelemetry — see app/shared/observability + otel_enabled: bool = Field( + default=False, + description=( + "Export traces and metrics over OTLP. Off by default: a deployment " + "that has not opted in must send nothing anywhere." + ), + ) + otel_exporter_otlp_endpoint: str = Field( + default="http://opentaberna-otel-collector:4318", + description=( + "OTLP/HTTP endpoint. This is the seam — pointing it at a vendor's " + "collector is the whole change needed to use one, because no " + "application code imports a vendor SDK." + ), + ) + otel_service_name: str = Field( + default="opentaberna-api", + description="service.name on every span and metric", + ) + otel_metric_export_interval_seconds: int = Field( + default=30, + description="Seconds between metric exports", + ) + # Analytics / reporting storefront_analytics_enabled: bool = Field( default=False, diff --git a/src/app/shared/logger/filters.py b/src/app/shared/logger/filters.py index f53ceb7..9d5ffda 100644 --- a/src/app/shared/logger/filters.py +++ b/src/app/shared/logger/filters.py @@ -90,6 +90,7 @@ class CorrelationIdFilter(ILogFilter): """ ATTRIBUTE = "correlation_id" + TRACE_ATTRIBUTE = "trace_id" def filter(self, record: logging.LogRecord) -> bool: """Never blocks — only enriches the record.""" @@ -100,6 +101,15 @@ def filter(self, record: logging.LogRecord) -> bool: if not hasattr(record, self.ATTRIBUTE): setattr(record, self.ATTRIBUTE, get_correlation_id()) + + # The trace id is what joins a span found in Grafana to the log lines + # for that same request. Without it the two systems describe the same + # work and cannot be put side by side. Empty when tracing is off, so + # the field is always present and a formatter never KeyErrors. + if not hasattr(record, self.TRACE_ATTRIBUTE): + from app.shared.observability import current_trace_id + + setattr(record, self.TRACE_ATTRIBUTE, current_trace_id() or "") return True def sanitize(self, data: Dict[str, Any]) -> Dict[str, Any]: diff --git a/src/app/shared/logger/formatters.py b/src/app/shared/logger/formatters.py index 69f7e12..d30a9dd 100644 --- a/src/app/shared/logger/formatters.py +++ b/src/app/shared/logger/formatters.py @@ -39,6 +39,12 @@ def format(self, record: logging.LogRecord) -> str: if correlation_id: log_data["correlation_id"] = correlation_id + # Promoted to a top-level field so a log backend can index it and a + # trace found in Grafana leads straight to these lines. + trace_id = getattr(record, "trace_id", "") + if trace_id: + log_data["trace_id"] = trace_id + # Add context data context = get_log_context() if context: @@ -71,6 +77,7 @@ def format(self, record: logging.LogRecord) -> str: "stack_info", "taskName", "correlation_id", + "trace_id", } extra_fields = { diff --git a/src/app/shared/observability/__init__.py b/src/app/shared/observability/__init__.py new file mode 100644 index 0000000..79bfc25 --- /dev/null +++ b/src/app/shared/observability/__init__.py @@ -0,0 +1,25 @@ +""" +Observability + +OpenTelemetry traces and metrics, exported over OTLP. + +OTLP is the seam: no application module imports a vendor SDK, so using a +different backend is a change to OTEL_EXPORTER_OTLP_ENDPOINT. +""" + +from .business_metrics import QUEUE_GAUGES, collect as collect_business_metrics +from .telemetry import ( + current_trace_id, + instrument_app, + instrument_engine, + setup, +) + +__all__ = [ + "current_trace_id", + "instrument_app", + "instrument_engine", + "QUEUE_GAUGES", + "collect_business_metrics", + "setup", +] diff --git a/src/app/shared/observability/business_metrics.py b/src/app/shared/observability/business_metrics.py new file mode 100644 index 0000000..258d5bc --- /dev/null +++ b/src/app/shared/observability/business_metrics.py @@ -0,0 +1,112 @@ +""" +Business Metrics + +Gauges for the queue states that mean the shop has quietly stopped working. + +`Deployment.md` names the queries worth alerting on. Until now they were +something an operator had to remember to run by hand, which means nobody ran +them and the first sign of a stalled fulfillment pipeline was a customer asking +where their parcel was. These export the same figures on a schedule. + +**Collected by the worker, not by the metrics SDK.** An observable gauge's +callback runs on the SDK's own thread, and the only database engine this +application has is asynchronous — driving it from outside the event loop fails, +which is exactly what the first version of this module did. The worker already +runs a scheduler, already holds an async session, and is the process that most +needs to be alive for these numbers to matter. So it computes them on a cron and +sets synchronous gauges. +""" + +from __future__ import annotations + +from typing import Any + +from sqlalchemy import text +from sqlalchemy.ext.asyncio import AsyncSession + +from app.shared.logger import get_logger + +logger = get_logger(__name__) + +# name -> (SQL, description). Each of these being non-zero and staying non-zero +# means something is wrong that no HTTP status code will reveal. +QUEUE_GAUGES: dict[str, tuple[str, str]] = { + "opentaberna.outbox.pending": ( + "SELECT count(*) FROM outbox_events WHERE status = 'pending'", + "Outbox events awaiting enqueue. Rising means the worker is not running.", + ), + "opentaberna.outbox.failed": ( + "SELECT count(*) FROM outbox_events WHERE status = 'failed'", + "Events that never reached the queue. Points at Redis or the poller.", + ), + "opentaberna.outbox.dead": ( + "SELECT count(*) FROM outbox_events WHERE status = 'dead'", + "Jobs that ran and exhausted their retries. Usually the carrier API.", + ), + "opentaberna.webhooks.unprocessed": ( + "SELECT count(*) FROM webhook_events WHERE processed_at IS NULL", + "Payments arriving but not handled. Money is at stake here.", + ), + "opentaberna.orders.awaiting_shipment": ( + "SELECT count(*) FROM orders WHERE deleted_at IS NULL " + "AND status IN ('paid', 'ready_to_ship')", + "Paid orders not yet handed to a carrier. The work queue.", + ), +} + +_gauges: dict[str, Any] = {} + + +def _ensure_gauges() -> dict[str, Any]: + """Create the gauge instruments once, on first collection.""" + global _gauges + + if _gauges: + return _gauges + + from opentelemetry import metrics + + meter = metrics.get_meter("opentaberna.business") + _gauges = { + name: meter.create_gauge(name, description=description) + for name, (_, description) in QUEUE_GAUGES.items() + } + logger.info("Business gauges created", extra={"count": len(_gauges)}) + return _gauges + + +async def collect(session: AsyncSession, settings: Any) -> dict[str, int]: + """ + Read the queue counts and publish them as gauges. + + Returns the values, so the caller can log them and a test can assert on + them without reaching into the metrics SDK. + + A failure here is logged and swallowed: a metrics collection that raises + would take down the scheduler it runs on, turning an observability problem + into a fulfillment one. + """ + if not settings.otel_enabled: + return {} + + values: dict[str, int] = {} + + try: + gauges = _ensure_gauges() + except Exception as exc: # noqa: BLE001 + logger.warning("Business gauges unavailable", extra={"error": str(exc)}) + return {} + + for name, (sql, _) in QUEUE_GAUGES.items(): + try: + result = await session.execute(text(sql)) + value = int(result.scalar() or 0) + values[name] = value + gauges[name].set(value) + except Exception as exc: # noqa: BLE001 + logger.warning( + "Business metric could not be collected", + extra={"metric": name, "error": str(exc)}, + ) + + return values diff --git a/src/app/shared/observability/settings_fields.py b/src/app/shared/observability/settings_fields.py new file mode 100644 index 0000000..31d8232 --- /dev/null +++ b/src/app/shared/observability/settings_fields.py @@ -0,0 +1,34 @@ +""" +Observability configuration fields. + +Kept beside the telemetry code rather than inline in Settings so the whole +feature — its switches, its wiring and its meters — reads as one thing. +""" + +from pydantic import Field + +OTEL_FIELDS = { + "otel_enabled": Field( + default=False, + description=( + "Export traces and metrics over OTLP. Off by default: a deployment " + "that has not opted in must send nothing anywhere." + ), + ), + "otel_exporter_otlp_endpoint": Field( + default="http://opentaberna-otel-collector:4318", + description=( + "OTLP/HTTP endpoint. This is the seam: pointing it at a vendor's " + "collector is the whole change needed to use one, because no " + "application code imports a vendor SDK." + ), + ), + "otel_service_name": Field( + default="opentaberna-api", + description="service.name on every span and metric", + ), + "otel_metric_export_interval_seconds": Field( + default=30, + description="Seconds between metric exports", + ), +} diff --git a/src/app/shared/observability/telemetry.py b/src/app/shared/observability/telemetry.py new file mode 100644 index 0000000..a0d4808 --- /dev/null +++ b/src/app/shared/observability/telemetry.py @@ -0,0 +1,198 @@ +""" +OpenTelemetry Wiring + +Traces and metrics for the API and the worker, exported over OTLP. + +**OTLP is the seam.** Nothing in application code imports a vendor SDK, and no +module outside this package knows telemetry exists beyond calling `setup()`. +Sending data to a different backend is a change to +`OTEL_EXPORTER_OTLP_ENDPOINT`, not a change to any service. + +**Off by default.** `setup()` returns immediately unless `OTEL_ENABLED` is set, +so a deployment that has not opted in creates no exporter, opens no connection +and sends nothing. + +**It must never take the application down.** Every step is wrapped: a collector +that is absent, unreachable or misconfigured produces a warning and a running +API, not a failed start. Observability that can cause the outage it exists to +diagnose is a bad trade. +""" + +from __future__ import annotations + +from typing import Any + +from app.shared.logger import get_logger + +logger = get_logger(__name__) + +_configured = False + + +def _resource(settings: Any): + from opentelemetry.sdk.resources import Resource + + return Resource.create( + { + "service.name": settings.otel_service_name, + "service.version": settings.app_version, + "deployment.environment": str( + getattr(settings.environment, "value", settings.environment) + ), + } + ) + + +def _setup_tracing(settings: Any, resource) -> None: + from opentelemetry import trace + from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter + from opentelemetry.sdk.trace import TracerProvider + from opentelemetry.sdk.trace.export import BatchSpanProcessor + + provider = TracerProvider(resource=resource) + provider.add_span_processor( + BatchSpanProcessor( + OTLPSpanExporter( + endpoint=f"{settings.otel_exporter_otlp_endpoint}/v1/traces" + ) + ) + ) + trace.set_tracer_provider(provider) + + +def _setup_metrics(settings: Any, resource) -> None: + from opentelemetry import metrics + from opentelemetry.exporter.otlp.proto.http.metric_exporter import ( + OTLPMetricExporter, + ) + from opentelemetry.sdk.metrics import MeterProvider + from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader + + reader = PeriodicExportingMetricReader( + OTLPMetricExporter( + endpoint=f"{settings.otel_exporter_otlp_endpoint}/v1/metrics" + ), + export_interval_millis=settings.otel_metric_export_interval_seconds * 1000, + ) + metrics.set_meter_provider( + MeterProvider(resource=resource, metric_readers=[reader]) + ) + + +def setup(settings: Any) -> bool: + """ + Configure tracing and metrics. + + Safe to call more than once; only the first call does anything, because the + API and the worker share this module and a second provider would silently + replace the first. + + Returns: + True when telemetry is now running, False when disabled or unavailable. + """ + global _configured + + if not settings.otel_enabled: + logger.debug("OpenTelemetry disabled (OTEL_ENABLED is false)") + return False + + if _configured: + return True + + try: + resource = _resource(settings) + _setup_tracing(settings, resource) + _setup_metrics(settings, resource) + except Exception as exc: # noqa: BLE001 — see the module docstring + logger.warning( + "OpenTelemetry could not be configured; continuing without it", + extra={"error": str(exc), "error_type": type(exc).__name__}, + ) + return False + + _configured = True + logger.info( + "OpenTelemetry configured", + extra={ + "endpoint": settings.otel_exporter_otlp_endpoint, + "service": settings.otel_service_name, + }, + ) + return True + + +def instrument_app(app: Any, settings: Any) -> None: + """ + Instrument the FastAPI application and its clients. + + Health endpoints are excluded. A liveness probe every few seconds would + otherwise dominate the trace volume and the request-rate metric, burying + real traffic under a heartbeat. + """ + if not settings.otel_enabled or not _configured: + return + + try: + from opentelemetry.instrumentation.fastapi import FastAPIInstrumentor + + FastAPIInstrumentor.instrument_app(app, excluded_urls="health,health/ready") + except Exception as exc: # noqa: BLE001 + logger.warning("Could not instrument FastAPI", extra={"error": str(exc)}) + + _instrument_clients(settings) + + +def _instrument_clients(settings: Any) -> None: + """Instrument Redis and outbound HTTP. Each failure is isolated.""" + from opentelemetry.instrumentation.httpx import HTTPXClientInstrumentor + from opentelemetry.instrumentation.redis import RedisInstrumentor + + for name, instrumentor in ( + ("redis", RedisInstrumentor), + ("httpx", HTTPXClientInstrumentor), + ): + try: + instrumentor().instrument() + except Exception as exc: # noqa: BLE001 + logger.warning(f"Could not instrument {name}", extra={"error": str(exc)}) + + +def instrument_engine(engine: Any, settings: Any) -> None: + """ + Instrument SQLAlchemy. + + Takes the sync engine behind an async one: the instrumentation hooks + SQLAlchemy's own events, which live on the underlying engine rather than on + the async facade. + """ + if not settings.otel_enabled or not _configured: + return + + try: + from opentelemetry.instrumentation.sqlalchemy import SQLAlchemyInstrumentor + + SQLAlchemyInstrumentor().instrument( + engine=getattr(engine, "sync_engine", engine) + ) + except Exception as exc: # noqa: BLE001 + logger.warning("Could not instrument SQLAlchemy", extra={"error": str(exc)}) + + +def current_trace_id() -> str | None: + """ + The active trace id as hex, or None. + + Logged alongside the correlation ID so a trace found in Grafana leads to the + log lines for that request. Without it the two systems describe the same + request and cannot be joined. + """ + try: + from opentelemetry import trace + + span = trace.get_current_span() + context = span.get_span_context() + if not context.is_valid: + return None + return format(context.trace_id, "032x") + except Exception: # noqa: BLE001 + return None diff --git a/src/app/worker.py b/src/app/worker.py index d6ac291..8adffac 100644 --- a/src/app/worker.py +++ b/src/app/worker.py @@ -53,6 +53,8 @@ ) from app.shared.config import get_settings from app.shared.logger import get_logger +from app.shared.observability import collect_business_metrics, instrument_engine +from app.shared.observability import setup as setup_telemetry from app.shared.storage.minio_adapter import build_minio_adapter logger = get_logger(__name__) @@ -79,12 +81,18 @@ async def startup(ctx: dict) -> None: settings = get_settings() ctx["settings"] = settings + # The worker is where fulfillment actually happens, so it needs tracing at + # least as much as the API does — a label job that hangs against a carrier + # is invisible from the HTTP side entirely. + setup_telemetry(settings) + engine = create_async_engine( settings.database_url, pool_size=settings.database_pool_size, max_overflow=settings.database_max_overflow, pool_pre_ping=settings.database_pool_pre_ping, ) + instrument_engine(engine, settings) ctx["session_factory"] = async_sessionmaker(engine, expire_on_commit=False) ctx["carrier_adapters"] = { @@ -320,6 +328,32 @@ async def on_job_abort(ctx: dict, job_id: str, function: str, args, kwargs) -> N # --------------------------------------------------------------------------- +async def export_business_metrics(ctx: dict) -> None: + """ + Publish the queue gauges an operator needs to alert on. + + Runs here rather than in the metrics SDK's own collection callback: that + callback executes on the SDK's thread, and the only database engine this + application has is asynchronous. Driving it from outside the event loop + fails, which is what the first version of this did — the gauges registered + cleanly and then silently produced nothing. + + The worker is the right home anyway. It already runs a scheduler, already + holds an async session, and is the process that most needs to be alive for + these numbers to mean anything. + """ + settings = ctx["settings"] + if not settings.otel_enabled: + return + + session_factory = ctx["session_factory"] + async with session_factory() as session: + values = await collect_business_metrics(session, settings) + + if values: + logger.debug("Business metrics exported", extra=values) + + def _build_outbox_cron(coroutine) -> CronJob: """ Build the outbox poller CronJob honouring settings.outbox_poll_interval. @@ -410,6 +444,10 @@ class WorkerSettings: _build_outbox_cron(poll_outbox), # Phase 4.2 — release expired stock reservations every 5 minutes cron(expire_reservations_sweep, minute={*range(0, 60, 5)}, second=0), + # S3 — publish queue gauges every 30 seconds. Cheap: five counts over + # indexed status columns, and it does nothing at all when OTEL_ENABLED + # is false. + cron(export_business_metrics, second={0, 30}), ] on_startup = startup diff --git a/src/docker/observability/grafana/dashboards/opentaberna-health.json b/src/docker/observability/grafana/dashboards/opentaberna-health.json new file mode 100644 index 0000000..32c6ef8 --- /dev/null +++ b/src/docker/observability/grafana/dashboards/opentaberna-health.json @@ -0,0 +1,527 @@ +{ + "uid": "opentaberna-health", + "title": "OpenTaberna \u2014 Health", + "tags": [ + "opentaberna" + ], + "timezone": "browser", + "schemaVersion": 39, + "version": 1, + "refresh": "30s", + "time": { + "from": "now-6h", + "to": "now" + }, + "panels": [ + { + "id": 100, + "type": "row", + "title": "Is the shop quietly broken?", + "gridPos": { + "x": 0, + "y": 0, + "w": 24, + "h": 1 + }, + "collapsed": false + }, + { + "id": 1, + "title": "Outbox pending", + "type": "stat", + "gridPos": { + "x": 0, + "y": 1, + "w": 4, + "h": 4 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "opentaberna_outbox_pending", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 20 + } + ] + } + }, + "overrides": [] + }, + "options": { + "colorMode": "background", + "graphMode": "area", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ] + } + }, + "description": "Events waiting to be enqueued. Rising and staying up means the worker is not running \u2014 orders reach paid and stop." + }, + { + "id": 2, + "title": "Outbox failed", + "type": "stat", + "gridPos": { + "x": 4, + "y": 1, + "w": 4, + "h": 4 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "opentaberna_outbox_failed", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + }, + "overrides": [] + }, + "options": { + "colorMode": "background", + "graphMode": "area", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ] + } + }, + "description": "Never reached the queue. Points at Redis or the poller, not at the job." + }, + { + "id": 3, + "title": "Outbox dead", + "type": "stat", + "gridPos": { + "x": 8, + "y": 1, + "w": 4, + "h": 4 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "opentaberna_outbox_dead", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + }, + "overrides": [] + }, + "options": { + "colorMode": "background", + "graphMode": "area", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ] + } + }, + "description": "The job ran and exhausted its retries. Usually the carrier API." + }, + { + "id": 4, + "title": "Webhooks unprocessed", + "type": "stat", + "gridPos": { + "x": 12, + "y": 1, + "w": 4, + "h": 4 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "opentaberna_webhooks_unprocessed", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + }, + "overrides": [] + }, + "options": { + "colorMode": "background", + "graphMode": "area", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ] + } + }, + "description": "Payments arriving but not handled. Money is at stake here \u2014 this is the one to page on." + }, + { + "id": 5, + "title": "Awaiting shipment", + "type": "stat", + "gridPos": { + "x": 16, + "y": 1, + "w": 4, + "h": 4 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "opentaberna_orders_awaiting_shipment", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + } + }, + "overrides": [] + }, + "options": { + "colorMode": "background", + "graphMode": "area", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ] + } + }, + "description": "Paid orders not yet handed to a carrier. The work queue, not necessarily a fault." + }, + { + "id": 101, + "type": "row", + "title": "Requests", + "gridPos": { + "x": 0, + "y": 5, + "w": 24, + "h": 1 + }, + "collapsed": false + }, + { + "id": 10, + "title": "Request rate by route", + "type": "timeseries", + "gridPos": { + "x": 0, + "y": 6, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (http_target) (rate(http_server_duration_milliseconds_count[5m]))", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + }, + "unit": "reqps" + }, + "overrides": [] + }, + "description": "Health endpoints are excluded at the source, so a liveness probe does not bury real traffic." + }, + { + "id": 11, + "title": "Error rate (5xx)", + "type": "timeseries", + "gridPos": { + "x": 12, + "y": 6, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "(sum(rate(http_server_duration_milliseconds_count{http_status_code=~\"5..\"}[5m])) or vector(0)) / clamp_min(sum(rate(http_server_duration_milliseconds_count[5m])), 0.001)", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + }, + "unit": "percentunit" + }, + "overrides": [] + }, + "description": "Share of requests failing. `or vector(0)` so a healthy shop reads as zero rather than 'No data' \u2014 which is indistinguishable from the scrape being broken, and is the wrong thing to see during an incident." + }, + { + "id": 12, + "title": "Latency p95 by route", + "type": "timeseries", + "gridPos": { + "x": 0, + "y": 14, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "histogram_quantile(0.95, sum by (le, http_target) (rate(http_server_duration_milliseconds_bucket[5m])))", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + }, + "unit": "ms" + }, + "overrides": [] + }, + "description": "p95 rather than an average: an average hides the slow tail that customers actually notice." + }, + { + "id": 13, + "title": "Latency p99", + "type": "timeseries", + "gridPos": { + "x": 12, + "y": 14, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "histogram_quantile(0.99, sum by (le) (rate(http_server_duration_milliseconds_bucket[5m])))", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + }, + "unit": "ms" + }, + "overrides": [] + } + }, + { + "id": 102, + "type": "row", + "title": "Dependencies", + "gridPos": { + "x": 0, + "y": 22, + "w": 24, + "h": 1 + }, + "collapsed": false + }, + { + "id": 20, + "title": "Database connection pool", + "type": "timeseries", + "gridPos": { + "x": 0, + "y": 23, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (state) (db_client_connections_usage)", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + } + }, + "overrides": [] + }, + "description": "Connections in use against idle. Saturation here is what a slow endpoint looks like from below, before it shows up as latency." + }, + { + "id": 21, + "title": "Requests in flight", + "type": "timeseries", + "gridPos": { + "x": 12, + "y": 23, + "w": 12, + "h": 8 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum(http_server_active_requests)", + "refId": "A", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + } + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "lineWidth": 2, + "fillOpacity": 8 + } + }, + "overrides": [] + }, + "description": "Concurrent requests. Climbing while the request rate is flat means requests are taking longer, not that there are more of them." + } + ] +} diff --git a/src/docker/observability/grafana/dashboards/provider.yml b/src/docker/observability/grafana/dashboards/provider.yml new file mode 100644 index 0000000..4b706e0 --- /dev/null +++ b/src/docker/observability/grafana/dashboards/provider.yml @@ -0,0 +1,10 @@ +apiVersion: 1 + +providers: + - name: OpenTaberna + folder: '' + type: file + disableDeletion: false + updateIntervalSeconds: 30 + options: + path: /etc/grafana/provisioning/dashboards diff --git a/src/docker/observability/grafana/datasources/prometheus.yml b/src/docker/observability/grafana/datasources/prometheus.yml new file mode 100644 index 0000000..8688c47 --- /dev/null +++ b/src/docker/observability/grafana/datasources/prometheus.yml @@ -0,0 +1,9 @@ +apiVersion: 1 + +datasources: + - name: Prometheus + type: prometheus + access: proxy + url: http://opentaberna-prometheus:9090 + isDefault: true + editable: false diff --git a/src/docker/observability/otel-collector.yaml b/src/docker/observability/otel-collector.yaml new file mode 100644 index 0000000..3abb08e --- /dev/null +++ b/src/docker/observability/otel-collector.yaml @@ -0,0 +1,56 @@ +# OpenTelemetry Collector — the seam between the application and wherever +# telemetry ends up. +# +# The API speaks OTLP and knows nothing else. Swapping Prometheus and Grafana +# for a vendor is a change to this file, not to any application code. + +receivers: + otlp: + protocols: + http: + endpoint: 0.0.0.0:4318 + grpc: + endpoint: 0.0.0.0:4317 + +processors: + # Bounds memory so a burst of spans cannot push the collector into the OOM + # killer and take the metrics pipeline down with it. + memory_limiter: + check_interval: 1s + limit_mib: 256 + spike_limit_mib: 64 + batch: + timeout: 5s + send_batch_size: 512 + +exporters: + # Prometheus scrapes this endpoint rather than the API directly, so the API + # never has to expose a metrics port of its own. + prometheus: + endpoint: 0.0.0.0:8889 + debug: + verbosity: basic + +service: + pipelines: + traces: + receivers: [otlp] + processors: [memory_limiter, batch] + exporters: [debug] + metrics: + receivers: [otlp] + processors: [memory_limiter, batch] + exporters: [prometheus] + telemetry: + logs: + level: warn + metrics: + # The collector's own health. Without this the scrape target Prometheus + # is configured for refuses connections, and "is the collector coping?" + # has no answer. + readers: + - pull: + exporter: + prometheus: + host: 0.0.0.0 + port: 8888 diff --git a/src/docker/observability/prometheus.yml b/src/docker/observability/prometheus.yml new file mode 100644 index 0000000..10b66e3 --- /dev/null +++ b/src/docker/observability/prometheus.yml @@ -0,0 +1,13 @@ +global: + scrape_interval: 15s + evaluation_interval: 15s + +scrape_configs: + # The collector re-exposes everything the API and worker sent over OTLP. + - job_name: opentaberna + static_configs: + - targets: ['opentaberna-otel-collector:8889'] + + - job_name: collector-self + static_configs: + - targets: ['opentaberna-otel-collector:8888'] diff --git a/tests/test_telemetry_integration.py b/tests/test_telemetry_integration.py new file mode 100644 index 0000000..c545124 --- /dev/null +++ b/tests/test_telemetry_integration.py @@ -0,0 +1,159 @@ +""" +Integration tests for observability (S3). + +Runs against the live stack with OTEL_ENABLED=true: + + docker compose -f docker-compose.dev.yml up -d + +Asserts against the collector's Prometheus endpoint and Prometheus itself, +because "the SDK was configured" is not the same claim as "a number an operator +can alert on actually arrived". +""" + +import os + +import pytest +import requests + +COLLECTOR = os.getenv("TEST_COLLECTOR_URL", "http://localhost:8889") +PROMETHEUS = os.getenv("TEST_PROMETHEUS_URL", "http://localhost:9090") +API = os.getenv("TEST_API_URL", "http://localhost:8000") + + +def _collector_metrics() -> str: + response = requests.get(f"{COLLECTOR}/metrics", timeout=10) + response.raise_for_status() + return response.text + + +def _promql(query: str) -> list: + response = requests.get( + f"{PROMETHEUS}/api/v1/query", params={"query": query}, timeout=10 + ) + response.raise_for_status() + return response.json()["data"]["result"] + + +@pytest.fixture(scope="module", autouse=True) +def stack_or_skip(): + """Nothing to assert on a deployment that has not opted into telemetry.""" + try: + requests.get(f"{COLLECTOR}/metrics", timeout=5).raise_for_status() + requests.get(f"{PROMETHEUS}/-/healthy", timeout=5).raise_for_status() + except Exception: + pytest.skip("collector or Prometheus not running") + + # Give the pipeline something to measure. + for _ in range(5): + requests.get(f"{API}/v1/items/", timeout=10) + yield + + +# --------------------------------------------------------------------------- +# The pipeline actually carries data +# --------------------------------------------------------------------------- + + +def test_http_request_metrics_reach_the_collector(): + """ + This is the one that caught a real bug: instrumenting the app before + configuring telemetry produced a clean startup log and no HTTP metrics at + all. Asserting on the pipeline's output catches that; asserting on the + setup call would not. + """ + assert "http_server_duration_milliseconds" in _collector_metrics() + + +def test_both_services_report(): + """The worker matters as much as the API — fulfillment happens there.""" + metrics = _collector_metrics() + assert 'job="opentaberna-api"' in metrics + assert 'job="opentaberna-worker"' in metrics + + +def test_prometheus_is_scraping_the_collector(): + targets = requests.get(f"{PROMETHEUS}/api/v1/targets", timeout=10).json() + jobs = { + t["labels"].get("job"): t["health"] for t in targets["data"]["activeTargets"] + } + assert jobs.get("opentaberna") == "up" + + +# --------------------------------------------------------------------------- +# The gauges an operator alerts on +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "metric", + [ + "opentaberna_outbox_pending", + "opentaberna_outbox_failed", + "opentaberna_outbox_dead", + "opentaberna_webhooks_unprocessed", + "opentaberna_orders_awaiting_shipment", + ], +) +def test_each_documented_alert_query_has_a_live_metric(metric): + """ + Deployment.md tells operators to watch these. Each must be queryable, or + the documented alert cannot be built. + """ + assert _promql(metric), f"{metric} is not in Prometheus" + + +def test_awaiting_shipment_matches_the_database(): + """ + A gauge that is exported but wrong is worse than one that is missing. This + checks the number against the query it claims to represent. + """ + import subprocess + + result = subprocess.run( + [ + "docker", + "exec", + "opentaberna-db", + "psql", + "-U", + "opentaberna", + "-d", + "opentaberna", + "-t", + "-A", + "-c", + "SELECT count(*) FROM orders WHERE deleted_at IS NULL " + "AND status IN ('paid', 'ready_to_ship');", + ], + check=True, + capture_output=True, + text=True, + ) + expected = int(result.stdout.strip()) + + series = _promql("opentaberna_orders_awaiting_shipment") + assert series + assert int(float(series[0]["value"][1])) == expected + + +# --------------------------------------------------------------------------- +# Noise control +# --------------------------------------------------------------------------- + + +def test_health_probes_are_not_traced_as_traffic(): + """ + A liveness probe every few seconds would dominate the request-rate metric + and bury real traffic under a heartbeat. + """ + for _ in range(5): + requests.get(f"{API}/health", timeout=10) + + metrics = _collector_metrics() + health_lines = [ + line + for line in metrics.splitlines() + if line.startswith("http_server_duration_milliseconds_count") + and 'http_target="/health"' in line + ] + assert not health_lines, "health checks are being counted as traffic" diff --git a/tests/test_telemetry_unit.py b/tests/test_telemetry_unit.py new file mode 100644 index 0000000..1eeec5d --- /dev/null +++ b/tests/test_telemetry_unit.py @@ -0,0 +1,151 @@ +""" +Unit tests for the observability wiring — no collector, no network. + +Two properties matter more than the plumbing and are pinned here: + + - telemetry stays off unless the operator opts in + - telemetry can never take the application down +""" + +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from app.shared.observability import telemetry +from app.shared.observability.business_metrics import QUEUE_GAUGES + + +def _settings(**overrides) -> SimpleNamespace: + base = { + "otel_enabled": True, + "otel_exporter_otlp_endpoint": "http://collector:4318", + "otel_service_name": "test-service", + "otel_metric_export_interval_seconds": 30, + "app_version": "0.0.0", + "environment": "testing", + } + base.update(overrides) + return SimpleNamespace(**base) + + +@pytest.fixture(autouse=True) +def _reset_module_state(): + """setup() is idempotent by design, so tests must reset the latch.""" + telemetry._configured = False + yield + telemetry._configured = False + + +# --------------------------------------------------------------------------- +# Off unless opted in +# --------------------------------------------------------------------------- + + +def test_setup_does_nothing_when_disabled(): + """ + A deployment that has not opted in must create no exporter and open no + connection — not merely send nothing useful. + """ + with ( + patch.object(telemetry, "_setup_tracing") as tracing, + patch.object(telemetry, "_setup_metrics") as metrics, + ): + assert telemetry.setup(_settings(otel_enabled=False)) is False + + tracing.assert_not_called() + metrics.assert_not_called() + + +def test_instrument_app_does_nothing_when_disabled(): + app = object() + with patch.object(telemetry, "_instrument_clients") as clients: + telemetry.instrument_app(app, _settings(otel_enabled=False)) + clients.assert_not_called() + + +def test_instrument_engine_does_nothing_before_setup(): + """ + Instrumenting before setup silently produced no metrics at all once. The + guard makes that a no-op rather than a half-configured pipeline. + """ + engine = object() + telemetry._configured = False + # Must not raise, and must not reach the instrumentation library. + telemetry.instrument_engine(engine, _settings()) + + +# --------------------------------------------------------------------------- +# Never take the application down +# --------------------------------------------------------------------------- + + +def test_an_unreachable_collector_does_not_stop_startup(): + """ + Observability that can cause the outage it exists to diagnose is a bad + trade. A broken exporter must degrade to "no telemetry", not "no API". + """ + with patch.object( + telemetry, "_setup_tracing", side_effect=OSError("connection refused") + ): + assert telemetry.setup(_settings()) is False + + +def test_a_broken_metrics_pipeline_does_not_stop_startup(): + with ( + patch.object(telemetry, "_setup_tracing"), + patch.object( + telemetry, "_setup_metrics", side_effect=RuntimeError("bad endpoint") + ), + ): + assert telemetry.setup(_settings()) is False + + +def test_setup_is_idempotent(): + """ + The API and the worker share this module. A second provider would silently + replace the first, so the second call must be a no-op. + """ + with ( + patch.object(telemetry, "_setup_tracing") as tracing, + patch.object(telemetry, "_setup_metrics"), + ): + assert telemetry.setup(_settings()) is True + assert telemetry.setup(_settings()) is True + + assert tracing.call_count == 1 + + +def test_current_trace_id_returns_none_outside_a_span(): + """Callers log this unconditionally, so it must never raise.""" + assert telemetry.current_trace_id() is None + + +# --------------------------------------------------------------------------- +# The gauges an operator alerts on +# --------------------------------------------------------------------------- + + +def test_every_queue_state_deployment_docs_name_has_a_gauge(): + """ + Deployment.md tells operators to watch these four. If a gauge is dropped, + the documented alert quietly stops being possible. + """ + assert "opentaberna.outbox.pending" in QUEUE_GAUGES + assert "opentaberna.outbox.failed" in QUEUE_GAUGES + assert "opentaberna.outbox.dead" in QUEUE_GAUGES + assert "opentaberna.webhooks.unprocessed" in QUEUE_GAUGES + + +def test_each_gauge_carries_sql_and_a_description(): + for name, (sql, description) in QUEUE_GAUGES.items(): + assert sql.lower().startswith("select count(*)"), name + # The description is what an operator reads at 3am; an empty one makes + # the panel a number with no meaning. + assert len(description) > 20, name + + +def test_gauge_queries_exclude_soft_deleted_orders(): + """A cancelled order must not be reported as work waiting to be done.""" + sql, _ = QUEUE_GAUGES["opentaberna.orders.awaiting_shipment"] + assert "deleted_at IS NULL" in sql diff --git a/uv.lock b/uv.lock index 400d701..bd84372 100644 --- a/uv.lock +++ b/uv.lock @@ -167,6 +167,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4d/89/e28a3a82da9b3d5ccacf143013b8911501c3593cf976b14b34711766fd4c/arq-0.27.0-py3-none-any.whl", hash = "sha256:4ca085671520472e45f09f5c37a1495097e5c4a79faaf18efcb9a759f8917f87", size = 26026, upload-time = "2026-02-02T14:38:19.375Z" }, ] +[[package]] +name = "asgiref" +version = "3.12.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e6/26/3b59f2bdae5f640389becb1f673cded775287f5fc4f816309d9ca9a3f93d/asgiref-3.12.1.tar.gz", hash = "sha256:59dcb51c272ad209d59bed5708a64a333083e86017d7fcdd67498eeab7784340", size = 42378, upload-time = "2026-07-14T09:56:18.087Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c0/1b/54f4ad77cd8a584fa70746c47df988e002cf1ee1eba43364d46f87803647/asgiref-3.12.1-py3-none-any.whl", hash = "sha256:fe386d1c2bff7259ea95929266d12a8cf9a8b5a1c2598402967d8792e7a7c094", size = 25478, upload-time = "2026-07-14T09:56:16.926Z" }, +] + [[package]] name = "asyncpg" version = "0.31.0" @@ -498,6 +507,12 @@ dependencies = [ { name = "email-validator" }, { name = "fastapi" }, { name = "httpx" }, + { name = "opentelemetry-exporter-otlp-proto-http" }, + { name = "opentelemetry-instrumentation-fastapi" }, + { name = "opentelemetry-instrumentation-httpx" }, + { name = "opentelemetry-instrumentation-redis" }, + { name = "opentelemetry-instrumentation-sqlalchemy" }, + { name = "opentelemetry-sdk" }, { name = "pydantic" }, { name = "pydantic-settings" }, { name = "pyjwt", extra = ["crypto"] }, @@ -532,6 +547,12 @@ requires-dist = [ { name = "email-validator", specifier = ">=2.3.0" }, { name = "fastapi", specifier = ">=0.135.3" }, { name = "httpx", specifier = ">=0.28.1" }, + { name = "opentelemetry-exporter-otlp-proto-http", specifier = ">=1.44.0" }, + { name = "opentelemetry-instrumentation-fastapi", specifier = ">=0.65b0" }, + { name = "opentelemetry-instrumentation-httpx", specifier = ">=0.65b0" }, + { name = "opentelemetry-instrumentation-redis", specifier = ">=0.65b0" }, + { name = "opentelemetry-instrumentation-sqlalchemy", specifier = ">=0.65b0" }, + { name = "opentelemetry-sdk", specifier = ">=1.44.0" }, { name = "pydantic", specifier = ">=2.12.5" }, { name = "pydantic-settings", specifier = ">=2.13.1" }, { name = "pyjwt", extras = ["crypto"], specifier = ">=2.12.1" }, @@ -591,6 +612,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9a/9a/e35b4a917281c0b8419d4207f4334c8e8c5dbf4f3f5f9ada73958d937dcc/frozenlist-1.8.0-py3-none-any.whl", hash = "sha256:0c18a16eab41e82c295618a77502e17b195883241c563b00f0aa5106fc4eaa0d", size = 13409, upload-time = "2025-10-06T05:38:16.721Z" }, ] +[[package]] +name = "googleapis-common-protos" +version = "1.75.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c0/90/fb8f1c84537fbf210c1f53a53ae473a805f6599c5a40b93c1bbadd211f7a/googleapis_common_protos-1.75.2.tar.gz", hash = "sha256:8829a3d1e4508c5b7b9a6b9525f7fccff611f8531644579a76466c29295d4bb2", size = 154083, upload-time = "2026-08-25T19:19:13.028Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/47/5b/1c9e55363c3b1890a98cae813de5b4ea327845756cd8fb7ee690140c7eac/googleapis_common_protos-1.75.2-py3-none-any.whl", hash = "sha256:6b83302f554ea93a0f48409c7fc2050f954bcbcddb7e3a9c76d4a823cb22920e", size = 307002, upload-time = "2026-08-25T19:18:08.927Z" }, +] + [[package]] name = "greenlet" version = "3.3.2" @@ -828,6 +861,190 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/81/08/7036c080d7117f28a4af526d794aab6a84463126db031b007717c1a6676e/multidict-6.7.1-py3-none-any.whl", hash = "sha256:55d97cc6dae627efa6a6e548885712d4864b81110ac76fa4e534c03819fa4a56", size = 12319, upload-time = "2026-01-26T02:46:44.004Z" }, ] +[[package]] +name = "opentelemetry-api" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ee/8b/aa9e2d8b8dfa7c946f7dec5d1f8f6ba8eca062f43509a06bdb5ce93d26c0/opentelemetry_api-1.44.0.tar.gz", hash = "sha256:67647e5e9566edcf421166fdf022b3537f818635daa852b289e34604dc6fb33a", size = 72406, upload-time = "2026-07-16T15:25:32.678Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ca/6f/a04e900f465ff3221ccc395522503e2d10e79fa21f2723c8e177aae1e0d1/opentelemetry_api-1.44.0-py3-none-any.whl", hash = "sha256:94b98c893a91b88657eaac1e3ba89618cdb85be6918196705354f34728b2cdef", size = 60018, upload-time = "2026-07-16T15:25:11.657Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-common" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-proto" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/09/4d717852c1cf3f854b76c7110a5d00883bc3c99288b9b0dbcbeb9e306eb6/opentelemetry_exporter_otlp_proto_common-1.44.0.tar.gz", hash = "sha256:dc87a5a5bc58f149a56d1547e4691588fa12994cdc3bc039a694ccb3375862ac", size = 20202, upload-time = "2026-07-16T15:25:37.658Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5e/71/65fd9d54c10b860f87c045ccee1264cab7011268895d3528818a29c1172a/opentelemetry_exporter_otlp_proto_common-1.44.0-py3-none-any.whl", hash = "sha256:9a9fe61bba73d802904bc989f1d6b4a7b1ee40f06c40e98d6f85af65aaebb694", size = 17045, upload-time = "2026-07-16T15:25:18.201Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-http" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "googleapis-common-protos" }, + { name = "opentelemetry-api" }, + { name = "opentelemetry-exporter-otlp-proto-common" }, + { name = "opentelemetry-proto" }, + { name = "opentelemetry-sdk" }, + { name = "requests" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1a/87/95e2a5aaa795b4e2260d74e16df2d5541deb2ea9de010bcd615f4dee2654/opentelemetry_exporter_otlp_proto_http-1.44.0.tar.gz", hash = "sha256:c633d7270ad6b57cd4cfbe8b0007a9e2e7c0cb50bd6c50fe2a7b245f721a09d8", size = 25806, upload-time = "2026-07-16T15:25:39.162Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cd/d0/fdeb1a98d8d3a6205f5f297c51b4a9bfe65126ab60339669bbe3dd54c2e2/opentelemetry_exporter_otlp_proto_http-1.44.0-py3-none-any.whl", hash = "sha256:838592fce774c1c8bb7b9a0a7facbfa82e17be5a8a4e94cef10cb84ae026bae3", size = 21850, upload-time = "2026-07-16T15:25:20.006Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "packaging" }, + { name = "wrapt" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/13/91/3c58961cb0360cd60509064734f0be4275383c8681d73c580a40ca83ddce/opentelemetry_instrumentation-0.65b0.tar.gz", hash = "sha256:071d9d9eced9bd6460444ec3b0c77229870ed05a881c22c84fdede58e4eed09b", size = 42689, upload-time = "2026-07-16T15:25:50.275Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/40/7b/85eab1215f72adf0e68d3dc4a679b9bff993fa679ff34cd8dd378e2659fd/opentelemetry_instrumentation-0.65b0-py3-none-any.whl", hash = "sha256:ea967a72b9939b5fcfdad572753b4306c59dcb99e3f382d95dae04286805e137", size = 36717, upload-time = "2026-07-16T15:24:51.424Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation-asgi" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "asgiref" }, + { name = "opentelemetry-api" }, + { name = "opentelemetry-instrumentation" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "opentelemetry-util-http" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/17/83/8e8e83b7ac285281687c7be2fd305213ccccbb8c0a2dd4fb45a8ccaf12c7/opentelemetry_instrumentation_asgi-0.65b0.tar.gz", hash = "sha256:892bca67c56522ffa85a8a83cf934d7b50b3be2132e45cbee705825f0a5ba426", size = 26140, upload-time = "2026-07-16T15:25:54.544Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0b/9c/376962840b619d2d55fe8ee2285f8c70971c090e5fff614516fc654a6f3a/opentelemetry_instrumentation_asgi-0.65b0-py3-none-any.whl", hash = "sha256:3a845a8ebd1c4ef0d8263401e6545f5b219b2feee612090d50f578a87e71fd65", size = 15903, upload-time = "2026-07-16T15:24:57.198Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation-fastapi" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-instrumentation" }, + { name = "opentelemetry-instrumentation-asgi" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "opentelemetry-util-http" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/30/23/b057f8196d06efdc1b50e3ff11fbc499a7d96b35c87f217eb7885542f4ea/opentelemetry_instrumentation_fastapi-0.65b0.tar.gz", hash = "sha256:10a3a95486036230413a58fe4fdf4a83fa6bba46918407e527476994bd92bd97", size = 26236, upload-time = "2026-07-16T15:26:05.954Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fa/b0/c9b0300d33349ecc3dfd2362516eaffc44877e90970e6a52178ff953fec3/opentelemetry_instrumentation_fastapi-0.65b0-py3-none-any.whl", hash = "sha256:cda2610a0ec1b22d19886f33e4d861e9f5dbb886aeaa3a1263b47aff82c36943", size = 13261, upload-time = "2026-07-16T15:25:12.429Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation-httpx" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-instrumentation" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "opentelemetry-util-http" }, + { name = "wrapt" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/03/a529140241addd4d0acc73bafbd6f74691651b92fc0ae9b4513cf80f07fa/opentelemetry_instrumentation_httpx-0.65b0.tar.gz", hash = "sha256:4627aa9c6bb99bf4462c8b565b0ef6aeb9ffad95c6c92868be1ef7895de112ee", size = 26309, upload-time = "2026-07-16T15:26:07.973Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9d/0f/c6144096b4914bbf44b43ba21c962e8f333ff045770b50a3e79ed8bd455f/opentelemetry_instrumentation_httpx-0.65b0-py3-none-any.whl", hash = "sha256:400f1b78afa4ee2332b5debe58e1ed1b317913d58812c952576be76660aeadb1", size = 17436, upload-time = "2026-07-16T15:25:15.772Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation-redis" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-instrumentation" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "wrapt" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/21/e5/b369897a9ff902eb028f1f999c9f9a4602f0b30fe1e97298d2c15fe5b8b0/opentelemetry_instrumentation_redis-0.65b0.tar.gz", hash = "sha256:f16409f189092984ff922f26939e7be79509365f5c9d202308c13fddf1368147", size = 17218, upload-time = "2026-07-16T15:26:17.243Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9d/2a/5b874bf8824e7cb995b22e1e61e1b5bc42810023d81612c44edbfd2c48a3/opentelemetry_instrumentation_redis-0.65b0-py3-none-any.whl", hash = "sha256:7d135f61db9d72416e1f382012fbc13de6f5e726491bebd0344a9c42a60e6769", size = 14658, upload-time = "2026-07-16T15:25:29.146Z" }, +] + +[[package]] +name = "opentelemetry-instrumentation-sqlalchemy" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-instrumentation" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "packaging" }, + { name = "wrapt" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ae/9d/72af21527ce6f286ac78dc2c75ce44b76cc45343a28f0bcbb13f213421fc/opentelemetry_instrumentation_sqlalchemy-0.65b0.tar.gz", hash = "sha256:8ec2e79f1e00808c5dc639ab5b2cfcdf9dcdef55efd6442c6019707fe2894028", size = 18007, upload-time = "2026-07-16T15:26:19.197Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/94/3eb3195a60b894cc1852a5657a2b64764a998085c50d7fb3452148af8f18/opentelemetry_instrumentation_sqlalchemy-0.65b0-py3-none-any.whl", hash = "sha256:4d8a2e5afc7b505a48d05cb1fb6db5f8b31681814c1d7767bd6d4b5bdd3a3047", size = 14410, upload-time = "2026-07-16T15:25:32.04Z" }, +] + +[[package]] +name = "opentelemetry-proto" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/64/01/40ac4ae9a149263cc52c2cee200ddd80cb6d8db1a4610abf8eabce0fe771/opentelemetry_proto-1.44.0.tar.gz", hash = "sha256:c547a79c2f8c0c515d31509154682e5921c7cfd5ca67b70e1f9266e2c3e103f3", size = 46488, upload-time = "2026-07-16T15:25:45.34Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d1/7c/8be563d68e93bbefa5c8affb82ddcff91b3ad858ce49957ba7b16fd3e0ab/opentelemetry_proto-1.44.0-py3-none-any.whl", hash = "sha256:898b155a0e1557afd867478fb6158e8122a46329ca0bb8dc53cc55e98f017f56", size = 72483, upload-time = "2026-07-16T15:25:28.429Z" }, +] + +[[package]] +name = "opentelemetry-sdk" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5d/77/a6592cbc7c8d9bcc9d6757a9df45e04a7c585e3e6e7a13456da522b21109/opentelemetry_sdk-1.44.0.tar.gz", hash = "sha256:cebe7f65dc12f26ead75c6064de12fd2a9052e5060c0272d402cfa203aae123b", size = 208624, upload-time = "2026-07-16T15:25:46.078Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e7/23/ff077e61886ee020a17ce9c8b6fa11c601c8d8345b09ea24f605445df62a/opentelemetry_sdk-1.44.0-py3-none-any.whl", hash = "sha256:df081c4c6bcfdb1211e3e86140376792643128a25f8d72d1d27675936e7e96ad", size = 137221, upload-time = "2026-07-16T15:25:29.534Z" }, +] + +[[package]] +name = "opentelemetry-semantic-conventions" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/8f/73/0cbdebcb4cf545fdd328da14f5137e37d0770c3f26185e478b0d15d94f50/opentelemetry_semantic_conventions-0.65b0.tar.gz", hash = "sha256:f9b2b81e9d5b64f11bc952075e7e9c7fb0aab075c7fd1c46d597f1b919852d60", size = 148774, upload-time = "2026-07-16T15:25:46.902Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a6/0e/49df70d9b81fb5cbae4bbf2a49d865b09bcbcbc4eb53f5851b1027738d78/opentelemetry_semantic_conventions-0.65b0-py3-none-any.whl", hash = "sha256:1cacde7b0ad306f84c5ef08c3dbe1bbaf20165bba6f8bff43b670e555a086bcb", size = 204645, upload-time = "2026-07-16T15:25:30.688Z" }, +] + +[[package]] +name = "opentelemetry-util-http" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/32/a9/d7525a59fdd240e69b5af4a6338e78fafa1b4203394122cbd6701fb5f84a/opentelemetry_util_http-0.65b0.tar.gz", hash = "sha256:84f82d826978bba416ab453460ff6a7391cdc3534c93a786595e4068680016b7", size = 11243, upload-time = "2026-07-16T15:26:27.898Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/23/3f/ab8d29df207ce5f470a07fa96ebb48af4e95b7fab7e7635311b9a32f2fab/opentelemetry_util_http-0.65b0-py3-none-any.whl", hash = "sha256:7553b606f963097cb190536dc30556cce85090692e471a422fff30ca29b04348", size = 8245, upload-time = "2026-07-16T15:25:46.482Z" }, +] + [[package]] name = "packaging" version = "26.0" @@ -885,6 +1102,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/5b/5a/bc7b4a4ef808fa59a816c17b20c4bef6884daebbdf627ff2a161da67da19/propcache-0.4.1-py3-none-any.whl", hash = "sha256:af2a6052aeb6cf17d3e46ee169099044fd8224cbaf75c76a2ef596e8163e2237", size = 13305, upload-time = "2025-10-08T19:49:00.792Z" }, ] +[[package]] +name = "protobuf" +version = "7.36.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a7/e7/0553e21d25ca4d9f573135775348a372c3ec34a93a71d5f297c3bac38341/protobuf-7.36.0.tar.gz", hash = "sha256:e8e09cb0d794c6687926fa558a8a6e72aa10edb997d5ca61da0765f12a3e00ea", size = 510034, upload-time = "2026-08-20T16:34:01.071Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/8f/ae/58e3ca96cb2e118cc546b677359b3c6659f79a140935c08dec94c7998585/protobuf-7.36.0-cp310-abi3-macosx_10_9_universal2.whl", hash = "sha256:9103532dffd80c6fab7e50c65a31007680a06eb57537d437bb1b35812c138a37", size = 453256, upload-time = "2026-08-20T16:33:53.945Z" }, + { url = "https://files.pythonhosted.org/packages/f0/15/5162230af4912697f0fe406f6800f80760945babcff0e2c2fe6c84ef2d5d/protobuf-7.36.0-cp310-abi3-manylinux2014_aarch64.whl", hash = "sha256:bf94a5917c71058262de683669bc0a797a7669d3de71f0b36d058e3194f47b44", size = 341436, upload-time = "2026-08-20T16:33:55.134Z" }, + { url = "https://files.pythonhosted.org/packages/d7/09/1670b2bfc9a45e807e520c3e9be36524db9ccc7dc05ea17af7681cabdc61/protobuf-7.36.0-cp310-abi3-manylinux2014_s390x.whl", hash = "sha256:3297e60abdff301e5f74393d87f6cc59dacab5f024a89548a6e8de1d26576b16", size = 354440, upload-time = "2026-08-20T16:33:56.077Z" }, + { url = "https://files.pythonhosted.org/packages/c7/f8/bd5804695ba400e423c33fd4d9f58c28d86633d5ba1945c36ff3967d98cb/protobuf-7.36.0-cp310-abi3-manylinux2014_x86_64.whl", hash = "sha256:70f5ec8eb0da81a44360c0dc0beac99a0d78071d21956a7076bae8bd2051841b", size = 340439, upload-time = "2026-08-20T16:33:56.992Z" }, + { url = "https://files.pythonhosted.org/packages/ef/9f/acd02338235a3e7d03168c4303478347b7624fc8189ff4e7f0d2654bbe86/protobuf-7.36.0-cp310-abi3-win32.whl", hash = "sha256:7326fd717bdc419162a735938d89d4032332bcc3408804012b24ff3a37086071", size = 440216, upload-time = "2026-08-20T16:33:57.99Z" }, + { url = "https://files.pythonhosted.org/packages/0e/4e/12cb93270967a2affff5b3f720694700d4d87712a67afd05c8cb3f6fa52c/protobuf-7.36.0-cp310-abi3-win_amd64.whl", hash = "sha256:1781cc1de61249b750848029bca452c0a8b7e990080316b9bbc2518b2117b488", size = 453731, upload-time = "2026-08-20T16:33:58.951Z" }, + { url = "https://files.pythonhosted.org/packages/01/c3/629999e78d46c1115c11886d51c6bd68c17ce4a944f1ea3e153a91316a33/protobuf-7.36.0-py3-none-any.whl", hash = "sha256:53374d53fc29a67f7dbbf0ade47d7526a0f0137bf0f9c90e48d8a60790ef748c", size = 177024, upload-time = "2026-08-20T16:34:00.053Z" }, +] + [[package]] name = "pycparser" version = "3.0"