OTEL collector service was configured.
This commit is contained in:
parent
18916fd017
commit
1dee1ef0c4
6 changed files with 169 additions and 33 deletions
17
.env.example
17
.env.example
|
|
@ -49,12 +49,16 @@ REFUND_FULL_USAGE_THRESHOLD=0.20 # ≤20% consumed + within window ⇒ full ref
|
|||
# --- Observability (leave empty to disable) ---
|
||||
SENTRY_DSN=
|
||||
# OTLP endpoint for traces/metrics/logs.
|
||||
# profile: observer → http://openobserve:5080/api/default (OTLP/HTTP)
|
||||
# profile: observer → http://otel-collector:4318 (OTLP/HTTP via collector)
|
||||
# profile: obs → configure your own otel-collector/tempo, or leave empty
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT=
|
||||
# Required when sending to OpenObserve. Generate with: echo -n 'user:pass' | base64
|
||||
# Default value is for root@example.com:Complexpass#123 (percent-encoded space).
|
||||
OTEL_EXPORTER_OTLP_HEADERS=Authorization=Basic%20cm9vdEBleGFtcGxlLmNvbTpDb21wbGV4cGFzcyMxMjM=
|
||||
# Auth header for OpenObserve. Used by the otel-collector to forward telemetry.
|
||||
# Generate with: echo -n 'user:pass' | base64
|
||||
# Default value is for root@example.com:Complexpass#123.
|
||||
OPENOBSERVE_AUTH_TOKEN=cm9vdEBleGFtcGxlLmNvbTpDb21wbGV4cGFzcyMxMjM=
|
||||
# Only needed if sending OTLP directly to OpenObserve (without collector).
|
||||
# The collector uses OPENOBSERVE_AUTH_TOKEN instead.
|
||||
OTEL_EXPORTER_OTLP_HEADERS=
|
||||
OTEL_SERVICE_NAME=contract-check
|
||||
|
||||
# --- OpenObserve (profile: observer) ---
|
||||
|
|
@ -73,6 +77,11 @@ OPENOBSERVE_MEMORY_LIMIT=2G
|
|||
OPENOBSERVE_CPU_RESERVATION=0.5
|
||||
OPENOBSERVE_MEMORY_RESERVATION=512M
|
||||
|
||||
# --- OTel Collector (profile: observer) ---
|
||||
OTEL_COLLECTOR_OTLP_GRPC_PORT=4317
|
||||
OTEL_COLLECTOR_OTLP_HTTP_PORT=4318
|
||||
OTEL_COLLECTOR_METRICS_PORT=8889
|
||||
|
||||
# --- Grafana / Loki (profile: obs) ---
|
||||
GRAFANA_PORT=3000
|
||||
GRAFANA_ADMIN_USER=admin
|
||||
|
|
|
|||
23
Makefile
23
Makefile
|
|
@ -78,22 +78,25 @@ services-logs: ## Tail all service logs
|
|||
services-ps: ## Show running containers
|
||||
docker compose --profile services ps
|
||||
|
||||
services-obs: ## Start services + observability (logs)
|
||||
services-obs: ## Start services + Grafana/Loki/Prometheus observability
|
||||
docker compose --profile services --profile obs up -d --build --remove-orphans
|
||||
|
||||
services-observer: ## Start services + OpenObserve collector stack
|
||||
docker compose --profile services --profile observer up -d --build --remove-orphans
|
||||
|
||||
services-nginx: ## Start with nginx reverse proxy
|
||||
docker compose --profile services --profile edge up -d --build --remove-orphans
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Docker: observability (Grafana + Loki logs)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
obs-up: ## Start observability containers (needs infra or services running)
|
||||
obs-up: ## Start Grafana/Loki/Prometheus containers
|
||||
docker compose --profile obs up -d --build --remove-orphans
|
||||
|
||||
obs-down: ## Stop observability containers
|
||||
obs-down: ## Stop Grafana/Loki/Prometheus containers
|
||||
docker compose --profile obs down
|
||||
|
||||
obs-logs: ## Tail observability containers logs
|
||||
obs-logs: ## Tail Grafana/Loki/Prometheus containers logs
|
||||
docker compose --profile obs logs -f
|
||||
|
||||
obs-url: ## Print Grafana URL and default credentials
|
||||
|
|
@ -103,6 +106,18 @@ obs-url: ## Print Grafana URL and default credentials
|
|||
@echo "Password: $${GRAFANA_ADMIN_PASSWORD:-admin}"
|
||||
@echo "Dashboard: $${GRAFANA_ROOT_URL:-http://localhost}/grafana/d/contract-check-logs"
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Docker: OpenObserve collector stack
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
observer-up: ## Start OpenObserve + otel-collector containers
|
||||
docker compose --profile observer up -d --build --remove-orphans
|
||||
|
||||
observer-down: ## Stop OpenObserve + otel-collector containers
|
||||
docker compose --profile observer down
|
||||
|
||||
observer-logs: ## Tail OpenObserve + otel-collector containers logs
|
||||
docker compose --profile observer logs -f
|
||||
|
||||
obs-reset: ## Reset Grafana and Loki volumes (wipes dashboards/logs data)
|
||||
docker compose --profile obs down -v
|
||||
rm -rf deploy/observability/grafana/dashboards/*.json.tmp
|
||||
|
|
|
|||
61
deploy/observability/otel-collector-config.yaml
Normal file
61
deploy/observability/otel-collector-config.yaml
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
receivers:
|
||||
otlp:
|
||||
protocols:
|
||||
grpc:
|
||||
endpoint: 0.0.0.0:4317
|
||||
http:
|
||||
endpoint: 0.0.0.0:4318
|
||||
|
||||
# Scrape the Prometheus /metrics endpoints exposed by api and workers.
|
||||
prometheus:
|
||||
config:
|
||||
scrape_configs:
|
||||
- job_name: contract-check
|
||||
static_configs:
|
||||
- targets:
|
||||
- api:9100
|
||||
- worker-extract:9101
|
||||
- worker-analyze:9102
|
||||
- worker-prescreen:9104
|
||||
- worker-billing:9105
|
||||
- worker-notify:9103
|
||||
|
||||
processors:
|
||||
batch:
|
||||
timeout: 1s
|
||||
send_batch_size: 1024
|
||||
|
||||
exporters:
|
||||
# Forward everything to the OpenObserve OTLP/HTTP endpoint.
|
||||
otlphttp/openobserve:
|
||||
endpoint: http://openobserve:5080/api/default
|
||||
headers:
|
||||
Authorization: Basic ${env:OPENOBSERVE_AUTH_TOKEN}
|
||||
tls:
|
||||
insecure: true
|
||||
|
||||
# Echo pipeline data to the collector's own logs (useful for debugging).
|
||||
debug:
|
||||
verbosity: detailed
|
||||
|
||||
service:
|
||||
pipelines:
|
||||
traces:
|
||||
receivers: [otlp]
|
||||
processors: [batch]
|
||||
exporters: [otlphttp/openobserve]
|
||||
metrics:
|
||||
receivers: [otlp, prometheus]
|
||||
processors: [batch]
|
||||
exporters: [otlphttp/openobserve]
|
||||
logs:
|
||||
receivers: [otlp]
|
||||
processors: [batch]
|
||||
exporters: [otlphttp/openobserve]
|
||||
|
||||
# Collector's own telemetry (optional, useful for debugging).
|
||||
telemetry:
|
||||
logs:
|
||||
level: info
|
||||
metrics:
|
||||
level: detailed
|
||||
|
|
@ -392,9 +392,10 @@ services:
|
|||
# Use for production-grade visibility.
|
||||
#
|
||||
# App services send OTLP traces/metrics/logs to the endpoint configured in
|
||||
# OTEL_EXPORTER_OTLP_ENDPOINT (.env). Point it at openobserve:5080/api/default
|
||||
# for the lightweight profile (the app uses OTLP/HTTP exporters), or at your
|
||||
# own otel-collector/tempo for the obs profile.
|
||||
# OTEL_EXPORTER_OTLP_ENDPOINT (.env). Point it at otel-collector:4318 for the
|
||||
# lightweight profile (the app uses OTLP/HTTP exporters). The collector then
|
||||
# forwards logs/metrics/traces to OpenObserve. For the obs profile you can
|
||||
# point this at your own otel-collector/tempo or leave it empty.
|
||||
|
||||
# ── LIGHTWEIGHT OBSERVABILITY (profile: observer) ───────────────────────────
|
||||
openobserve:
|
||||
|
|
@ -426,6 +427,24 @@ services:
|
|||
cpus: "${OPENOBSERVE_CPU_RESERVATION:-0.5}"
|
||||
memory: "${OPENOBSERVE_MEMORY_RESERVATION:-512M}"
|
||||
|
||||
otel-collector:
|
||||
profiles: ["observer"]
|
||||
image: otel/opentelemetry-collector-contrib:latest
|
||||
container_name: contract_check-otel-collector
|
||||
restart: unless-stopped
|
||||
command: ["--config", "/etc/otelcol/config.yaml"]
|
||||
environment:
|
||||
OPENOBSERVE_AUTH_TOKEN: ${OPENOBSERVE_AUTH_TOKEN}
|
||||
volumes:
|
||||
- ./deploy/observability/otel-collector-config.yaml:/etc/otelcol/config.yaml:ro
|
||||
ports:
|
||||
- "${OTEL_COLLECTOR_OTLP_GRPC_PORT:-4317}:4317" # OTLP gRPC receiver
|
||||
- "${OTEL_COLLECTOR_OTLP_HTTP_PORT:-4318}:4318" # OTLP HTTP receiver
|
||||
- "${OTEL_COLLECTOR_METRICS_PORT:-8889}:8889" # Collector self-metrics
|
||||
depends_on:
|
||||
openobserve:
|
||||
condition: service_started
|
||||
|
||||
# ── MATURE OBSERVABILITY (profile: obs) ─────────────────────────────────────
|
||||
# Grafana + Loki logs + Prometheus metrics. Promtail scrapes all compose
|
||||
# container logs via the local Docker socket. See deploy/observability/.
|
||||
|
|
|
|||
|
|
@ -167,9 +167,10 @@ DealDocumentScreening/
|
|||
│ │ ├── templates/contract-check.conf.template (reverse proxy + TLS)
|
||||
│ │ └── certbot-init.sh (initial cert + nginx reload)
|
||||
│ └── observability/
|
||||
│ ├── prometheus.yml (planned — scrape api + workers :9100/:9101/:9102)
|
||||
│ ├── otel-collector-config.yaml (planned)
|
||||
│ ├── tempo.yaml (planned — trace storage)
|
||||
│ ├── prometheus/prometheus.yml (scrape api + workers :9100..:9105)
|
||||
│ ├── otel-collector-config.yaml (OpenObserve profile: receiver + forwarder)
|
||||
│ ├── loki-config.yaml (Grafana/Loki profile)
|
||||
│ ├── tempo.yaml (planned — trace storage for Grafana stack)
|
||||
│ └── grafana/provisioning/
|
||||
│ ├── datasources/ (prometheus + tempo)
|
||||
│ └── dashboards/ (starter: queue depth, job latency, LLM tokens)
|
||||
|
|
@ -1088,8 +1089,12 @@ None of 1–5 requires touching `core/` application code — only compose/infra.
|
|||
| `REFUND_WINDOW_DAYS` | `14` | full-refund window (§8a) |
|
||||
| `REFUND_FULL_USAGE_THRESHOLD` | `0.20` | usage ratio for full refund |
|
||||
| `SENTRY_DSN` | (empty) | if set, init sentry |
|
||||
| `OTEL_EXPORTER_OTLP_ENDPOINT` | (empty) | OTel collector |
|
||||
| `OTEL_EXPORTER_OTLP_ENDPOINT` | (empty) | OTLP endpoint; with profile `observer` use `http://otel-collector:4318` |
|
||||
| `OTEL_SERVICE_NAME` | per-service | overridden in each service settings |
|
||||
| `OPENOBSERVE_AUTH_TOKEN` | default creds base64 | Basic auth token the collector uses to forward to OpenObserve |
|
||||
| `OTEL_COLLECTOR_OTLP_GRPC_PORT` | `4317` | host port for collector OTLP/gRPC receiver |
|
||||
| `OTEL_COLLECTOR_OTLP_HTTP_PORT` | `4318` | host port for collector OTLP/HTTP receiver |
|
||||
| `OTEL_COLLECTOR_METRICS_PORT` | `8889` | host port for collector self-metrics |
|
||||
| `GRAFANA_PORT` | `3000` | host port for the Grafana UI (profile `obs`) |
|
||||
| `GRAFANA_ADMIN_USER` | `admin` | initial Grafana admin user |
|
||||
| `GRAFANA_ADMIN_PASSWORD` | `admin` | initial Grafana admin password |
|
||||
|
|
@ -1273,8 +1278,9 @@ dev = [{ include-group = "api" }, { include-group = "extract" },
|
|||
### docker-compose.yml structure
|
||||
|
||||
One file, profiles. Default (`docker compose up`) = infra only. Services
|
||||
behind `--profile services`. Observability behind `--profile obs`. Edge
|
||||
behind `--profile edge`.
|
||||
behind `--profile services`. Two observability stacks are mutually exclusive:
|
||||
`--profile obs` (Grafana + Loki + Prometheus) or `--profile observer`
|
||||
(OpenObserve + OTel Collector). Edge behind `--profile edge`.
|
||||
|
||||
```
|
||||
services:
|
||||
|
|
@ -1284,9 +1290,6 @@ services:
|
|||
rabbitmq: rabbitmq:4-management-alpine, healthcheck, volume; AMQP 5672, mgmt UI 15672
|
||||
minio: minio/minio, healthcheck, volume, console on :9001; S3 API :9000
|
||||
minio-init: one-shot: mc alias + mb + ilm rule; depends_on minio healthy
|
||||
loki: grafana/loki; filesystem-backed single-node log store
|
||||
promtail: grafana/promtail; ships Docker container logs into Loki
|
||||
grafana: grafana/grafana; provisioned Loki datasource + logs dashboard
|
||||
|
||||
# ── SERVICES (profile: services) ──
|
||||
api: build srv/api/Dockerfile; depends_on pg/rabbit/minio-init healthy;
|
||||
|
|
@ -1294,19 +1297,26 @@ services:
|
|||
worker-extract: build srv/worker-extract/Dockerfile; depends_on pg/rabbit/minio-init healthy; 9101
|
||||
worker-prescreen: build srv/worker-prescreen/Dockerfile; depends_on pg/rabbit/minio-init healthy; 9104
|
||||
worker-analyze: build srv/worker-analyze/Dockerfile; depends_on pg/rabbit/minio-init healthy; 9102
|
||||
worker-billing: build srv/worker-billing/Dockerfile; depends_on pg/minio-init healthy; 9105
|
||||
worker-notify: build srv/worker-notify/Dockerfile; depends_on rabbit healthy; 9103
|
||||
bot: build srv/bot/Dockerfile; depends_on api healthy (NOT pg/rabbit)
|
||||
|
||||
# ── OBSERVABILITY (profile: obs) ──
|
||||
# Logs are live (Grafana + Loki + Promtail). Metrics/traces still planned.
|
||||
# ── OBSERVABILITY — GRAFANA STACK (profile: obs) ──
|
||||
loki: grafana/loki; filesystem-backed single-node log store
|
||||
promtail: grafana/promtail; ships Docker container logs into Loki
|
||||
grafana: grafana/grafana; provisioned Loki datasource + logs dashboard
|
||||
prometheus: prom/prometheus; scrapes service /metrics endpoints
|
||||
|
||||
# ── EDGE (profile: edge) ── PLANNED (T-E1-009)
|
||||
nginx: nginx:alpine; reverse proxy → api; TLS via certbot
|
||||
# ── OBSERVABILITY — OPENOBSERVE STACK (profile: observer) ──
|
||||
openobserve: public.ecr.aws/zinclabs/openobserve:latest; UI + OTLP backend on :5080, gRPC on :5081
|
||||
otel-collector: otel/opentelemetry-collector-contrib:latest; OTLP receiver 4317/4318,
|
||||
scrapes /metrics from services, forwards logs/metrics/traces to OpenObserve
|
||||
|
||||
# ── EDGE (profile: edge) ──
|
||||
nginx: nginx:alpine; reverse proxy → api/openobserve/grafana; TLS via certbot
|
||||
certbot: certbot/certbot; renew cron sidecar
|
||||
|
||||
volumes: { pgdata, pgwal, redisdata, rabbitmq, minio, loki-data, grafana-data }
|
||||
volumes: { pgdata, pgwal, redisdata, rabbitmq, minio, loki-data, grafana-data, prometheus-data, openobserve-data }
|
||||
```
|
||||
|
||||
`bot` depends on `api` healthy (not on infra) — it speaks HTTP to the api,
|
||||
|
|
@ -1369,17 +1379,33 @@ enforcing the adapter boundary even in dependency ordering.
|
|||
highlighted as annotations. Use correlation_id values to trace one upload
|
||||
across `api` → `worker-extract` → `worker-prescreen` → `worker-analyze`.
|
||||
|
||||
### Prometheus + Tempo (planned)
|
||||
### Prometheus + Tempo (Grafana stack, planned enhancements)
|
||||
|
||||
- Prometheus will scrape `:9100`/`:9101`/`:9102` metrics exposed by the services.
|
||||
- Tempo will receive OTLP traces via the OpenTelemetry collector.
|
||||
- `prometheus` in the `obs` profile scrapes `:9100`/`:9101`/`:9102`/`:9104`/`:9105`/`:9103`
|
||||
metrics exposed by the services.
|
||||
- Tempo will receive OTLP traces via the OpenTelemetry collector when added.
|
||||
- Starter dashboards for queue depth, job latency, LLM tokens/min, refund rate,
|
||||
and 429/fallback rate will ship with the metrics/traces datasources.
|
||||
|
||||
### OpenObserve + OTel Collector (`observer` profile)
|
||||
|
||||
- `otel-collector` runs in the `observer` profile and listens for OTLP on
|
||||
`4317` (gRPC) and `4318` (HTTP).
|
||||
- Services export OTLP traces, metrics, and logs to `http://otel-collector:4318`
|
||||
(`OTEL_EXPORTER_OTLP_ENDPOINT`).
|
||||
- The collector also scrapes Prometheus `/metrics` from `api:9100` and each
|
||||
worker (`:9101`/`:9102`/`:9104`/`:9105`/`:9103`).
|
||||
- A dedicated `otlphttp/openobserve` exporter forwards all signals to
|
||||
`http://openobserve:5080/api/default` using `OPENOBSERVE_AUTH_TOKEN`.
|
||||
- OpenObserve UI is exposed under `/openobserve/` via nginx.
|
||||
|
||||
### OpenTelemetry (`core/telemetry.py`)
|
||||
|
||||
- OTLP exporter to `otel-collector` → Tempo.
|
||||
- OTLP exporter sends traces, metrics, and logs to the endpoint configured in
|
||||
`OTEL_EXPORTER_OTLP_ENDPOINT` (collector in `observer` profile, Tempo in `obs`).
|
||||
- Auto-instrument FastAPI (api), httpx (all outbound, incl. Ollama calls).
|
||||
- `configure_logging` routes structlog through stdlib logging so the OTEL log
|
||||
handler captures application logs with their structured fields.
|
||||
- Spans carry the same `correlation_id` as logs. Trace from HTTP request →
|
||||
RabbitMQ publish → consume → LLM call is one trace tree.
|
||||
|
||||
|
|
|
|||
|
|
@ -134,7 +134,9 @@ PRICE_PER_DOC_KOPECKS=19900 # 199 ₽ за документ без подпис
|
|||
|
||||
# --- Observability (опционально) ---
|
||||
SENTRY_DSN=https://...@sentry.io/...
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT=http://otel-collector:4317
|
||||
# OTLP/HTTP endpoint. With profile `observer` use the local collector.
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT=http://otel-collector:4318
|
||||
OPENOBSERVE_AUTH_TOKEN=cm9vdEBleGFtcGxlLmNvbTpDb21wbGV4cGFzcyMxMjM=
|
||||
```
|
||||
|
||||
> Полный список: см. `.env.example` и `ARCHITECTURE.md §11`. Включение
|
||||
|
|
@ -187,7 +189,8 @@ make shell-db # UPDATE plans SET price_kopecks = 59000 WHERE code = 'lite';
|
|||
|---|---|
|
||||
| *(default)* | `postgres`, `redis`, `rabbitmq`, `minio`, `minio-init` |
|
||||
| `services` | `api`, `worker-extract`, `worker-prescreen`, `worker-analyze`, `worker-billing`, `worker-notify`, `bot` |
|
||||
| `obs` | `loki`, `promtail`, `grafana` (logs only; metrics/traces — позже) |
|
||||
| `obs` | `loki`, `promtail`, `grafana`, `prometheus` (logs + metrics; traces — позже) |
|
||||
| `observer` | `openobserve`, `otel-collector` (logs + metrics + traces через collector) |
|
||||
| `edge` | `nginx`, `certbot` |
|
||||
|
||||
```bash
|
||||
|
|
@ -197,11 +200,14 @@ docker compose up -d
|
|||
# + сервисы (сборка + запуск):
|
||||
docker compose --profile services up -d --build
|
||||
|
||||
# + observability (логи):
|
||||
# + observability Grafana/Loki/Prometheus (логи + метрики):
|
||||
docker compose --profile services --profile obs up -d --build
|
||||
|
||||
# + observability OpenObserve (логи + метрики + трейсы через otel-collector):
|
||||
docker compose --profile services --profile observer up -d --build
|
||||
|
||||
# + edge (nginx reverse proxy + TLS; deploy/nginx готов):
|
||||
docker compose --profile services --profile obs --profile edge up -d --build
|
||||
docker compose --profile services --profile observer --profile edge up -d --build
|
||||
```
|
||||
|
||||
### 4.2 depends_on и healthchecks
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue