Ingest tenant-awareness was named "undesigned, not just unbuilt" across CLAUDE.md/threat-model.md/the runbook since early Phase 4 -- the last major standing gap. Scoping was agreed via AskUserQuestion: a config-supplied tenant_id + shared-secret token ingest validates (smaller real implementation, no new PKI), over per-tenant mTLS certs. This change builds that identity mechanism end to end and attaches it to every record at the point it enters the system; it deliberately does NOT build per-tenant write-routing for ClickHouse or Tantivy -- that's real, separately-scoped follow-up work, disclosed explicitly everywhere this was previously called undesigned, not silently left half-done. New pieces: - metadata/migrations/0034 + enterprise/internal/rbacstore/ ingest_credentials.go: a per-tenant bearer credential, only its SHA-256 hash ever persisted (same reasoning a password gets hashed, not stored raw) -- CreateIngestCredential returns the plaintext exactly once, ValidateIngestCredential/RevokeIngestCredential/ ListIngestCredentialsForTenant round it out. - enterprise-auth gains -create-ingest-credential-tenant/ -list-ingest-credentials-tenant/-revoke-ingest-credential (same offline-operator-flag shape as every other credential-minting flag in this binary) and a new POST /internal/authorize-ingest endpoint (internal/authhandler) validating a presented token and resolving its tenant -- a genuinely different credential type from session-backed /internal/authorize, so it doesn't touch session.Manager at all. - ingest (AGPL core) gains an optional TenantResolver (internal/grpcserver, nil by default) and its HTTP client implementation (internal/tenantresolver.HTTPResolver) -- a plain HTTP call to enterprise-auth's new endpoint, never an enterprise/ import, same "network boundary, not import boundary" shape api/authz.HTTPAuthorizer already uses for the query path. PushBatch now requires an `authorization: Bearer <token>` gRPC metadata entry once a resolver is configured, fails the whole batch closed on a missing/invalid credential (never falls back to "no tenant"), and attaches the resolved tenant ID to every record as a `tenant_id` Kafka message header before producing it. Verified with real round trips at every layer, no Docker needed: rbacstore's credential CRUD (skip-gated on live Postgres, same as every other rbacstore integration test this phase), authhandler's new endpoint (real HTTP via httptest, including the regression test that a session token must not validate as an ingest credential), tenantresolver (real HTTP client against httptest, same pattern as authz.HTTPAuthorizer's own tests), and grpcserver's PushBatch (fake resolver/producer -- no resolver leaves messages unchanged, a configured resolver attaches the right header or fails closed on a bad/missing token). Helm: ingest.requireTenantCredential (default false) is a deliberate, separate opt-in from enterprise.enabled -- turning ENTERPRISE_AUTH_URL on for ingest requires every agent to already hold a credential or be refused outright, so it must not default on just because enterprise.enabled does (same reasoning api.yaml's ENTERPRISE_AUTH_URL isn't tied to enterprise.enabled directly either). docker-compose.yml leaves it unset, same as ever. Docs updated everywhere this was called "undesigned": CLAUDE.md, docs/architecture.md, docs/security/threat-model.md (including its summary table, now split into "identity: built" vs "write-routing: not yet"), docs/phase-4-runbook.md (new §13), enterprise/README.md.
404 lines
16 KiB
YAML
404 lines
16 KiB
YAML
# Phase 0+1 stack: Redpanda -> ingest -> ClickHouse -> api -> web, plus
|
|
# search (Tantivy full-text indexing, reads the same Redpanda topic
|
|
# ingest's consumer does).
|
|
#
|
|
# Does NOT include the Rust agent — see /agent/README.md: journald
|
|
# sourcing needs the host's journal, which isn't something a container
|
|
# gets for free. Run the agent natively on the host per
|
|
# /docs/phase-0-runbook.md, pointed at ingest's mapped port (localhost:4317).
|
|
# Windows Event Log/ETW sourcing needs a real Windows host regardless —
|
|
# see /docs/phase-1-runbook.md.
|
|
#
|
|
# Before first run: generate dev mTLS certs (hack/dev-certs/generate.sh).
|
|
# See /docs/phase-0-runbook.md (Linux pipeline) and
|
|
# /docs/phase-1-runbook.md (Windows + full-text search) for the full
|
|
# sequences.
|
|
services:
|
|
redpanda:
|
|
image: docker.redpanda.com/redpandadata/redpanda:v24.2.7
|
|
container_name: sentry-redpanda
|
|
command:
|
|
- redpanda
|
|
- start
|
|
- --smp=1
|
|
- --memory=1G
|
|
- --reserve-memory=0M
|
|
- --overprovisioned
|
|
- --node-id=0
|
|
- --check=false
|
|
- --kafka-addr=PLAINTEXT://0.0.0.0:9092
|
|
- --advertise-kafka-addr=PLAINTEXT://redpanda:9092
|
|
ports:
|
|
- "9092:9092"
|
|
volumes:
|
|
- redpanda-data:/var/lib/redpanda/data
|
|
healthcheck:
|
|
test: ["CMD", "rpk", "cluster", "health", "--exit-when-healthy"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
# One-shot: creates the sentry.logs.raw topic, then exits 0. ingest
|
|
# waits on this completing successfully before it starts.
|
|
redpanda-provision:
|
|
build:
|
|
context: ./transport
|
|
container_name: sentry-redpanda-provision
|
|
depends_on:
|
|
redpanda:
|
|
condition: service_healthy
|
|
environment:
|
|
REDPANDA_BROKERS: "redpanda:9092"
|
|
REDPANDA_ADMIN_HOSTS: "redpanda:9644"
|
|
# Explicit rather than relying on both this script's and /search's
|
|
# defaults happening to agree — search consumes this same topic and
|
|
# needs to know the partition count up front (see /search/README.md).
|
|
REDPANDA_TOPIC_PARTITIONS: "6"
|
|
|
|
clickhouse:
|
|
image: clickhouse/clickhouse-server:24.8
|
|
container_name: sentry-clickhouse
|
|
ports:
|
|
- "8123:8123" # HTTP interface, used by the migrate step
|
|
- "9000:9000" # native protocol, used by ingest and api
|
|
environment:
|
|
# The official image disables *network* access entirely for the
|
|
# default user (even from sibling containers) unless
|
|
# CLICKHOUSE_USER or CLICKHOUSE_PASSWORD is set to a genuinely
|
|
# non-empty value — confirmed by testing, not just reading docs: an
|
|
# explicitly-empty CLICKHOUSE_PASSWORD="" still triggers the
|
|
# lockdown, silently returning 403 to every other container. This
|
|
# password isn't a real secret (mTLS between agent and ingest is
|
|
# the actual security boundary here) — it exists purely to satisfy
|
|
# this image's login gate for local/homelab use.
|
|
CLICKHOUSE_PASSWORD: "sentry-dev-only"
|
|
volumes:
|
|
- clickhouse-data:/var/lib/clickhouse
|
|
ulimits:
|
|
nofile:
|
|
soft: 262144
|
|
hard: 262144
|
|
healthcheck:
|
|
test: ["CMD", "wget", "--no-verbose", "--tries=1", "--spider", "http://localhost:8123/ping"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
# One-shot: applies /storage/migrations/*.sql, then exits 0. ingest and
|
|
# api both wait on this completing successfully.
|
|
clickhouse-migrate:
|
|
build:
|
|
context: ./storage
|
|
container_name: sentry-clickhouse-migrate
|
|
depends_on:
|
|
clickhouse:
|
|
condition: service_healthy
|
|
environment:
|
|
CLICKHOUSE_HTTP: "http://clickhouse:8123"
|
|
CLICKHOUSE_PASSWORD: "sentry-dev-only"
|
|
|
|
# Control-plane metadata store (dashboards, alert rules -- see
|
|
# /docs/phase-3-dashboard-design.md for why this is Postgres rather
|
|
# than new ClickHouse tables). Log data stays on ClickHouse/Tantivy
|
|
# only, unaffected.
|
|
metadata-postgres:
|
|
image: postgres:16-alpine
|
|
container_name: sentry-metadata-postgres
|
|
environment:
|
|
POSTGRES_DB: sentry_metadata
|
|
POSTGRES_USER: sentry
|
|
POSTGRES_PASSWORD: "sentry-dev-only" # not a real secret, same framing as CLICKHOUSE_PASSWORD above
|
|
volumes:
|
|
- metadata-postgres-data:/var/lib/postgresql/data
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U sentry -d sentry_metadata"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
# One-shot: applies /metadata/migrations/*.sql, then exits 0. api waits
|
|
# on this completing successfully, same shape as clickhouse-migrate.
|
|
metadata-migrate:
|
|
build:
|
|
context: ./metadata
|
|
container_name: sentry-metadata-migrate
|
|
depends_on:
|
|
metadata-postgres:
|
|
condition: service_healthy
|
|
environment:
|
|
POSTGRES_HOST: "metadata-postgres"
|
|
POSTGRES_PORT: "5432"
|
|
POSTGRES_USER: "sentry"
|
|
POSTGRES_PASSWORD: "sentry-dev-only"
|
|
POSTGRES_DATABASE: "sentry_metadata"
|
|
# Password for the restricted audit_writer Postgres role (Phase 4
|
|
# task 4) -- INSERT+SELECT only on audit_log, never UPDATE/DELETE,
|
|
# via its own connection pool distinct from the shared "sentry"
|
|
# role every other store uses. See /docs/phase-4-isolation-design.md.
|
|
AUDIT_WRITER_PASSWORD: "audit-writer-dev-only"
|
|
|
|
ingest:
|
|
build:
|
|
context: . # needs both ingest/ and proto/
|
|
dockerfile: ingest/Dockerfile
|
|
container_name: sentry-ingest
|
|
depends_on:
|
|
redpanda-provision:
|
|
condition: service_completed_successfully
|
|
clickhouse-migrate:
|
|
condition: service_completed_successfully
|
|
ports:
|
|
- "4317:4317" # gRPC, mTLS — this is what the host-run agent connects to
|
|
environment:
|
|
REDPANDA_BROKERS: "redpanda:9092"
|
|
CLICKHOUSE_ADDR: "clickhouse:9000"
|
|
CLICKHOUSE_PASSWORD: "sentry-dev-only"
|
|
# TLS_*_FILE env vars are left at their defaults
|
|
# (/etc/sentry-ingest/{server,server-key,ca}.pem) — matches where
|
|
# the volume below mounts the generated dev certs.
|
|
#
|
|
# ENTERPRISE_AUTH_URL is deliberately NOT set here (see
|
|
# ingest/internal/grpcserver's TenantResolver): with it unset,
|
|
# PushBatch attaches no tenant_id header to any record, matching
|
|
# every Phase 0-3 deployment's behavior. Setting it to
|
|
# "http://enterprise-auth:8082" would require every agent to
|
|
# present a valid `Authorization: Bearer <ingest token>` (minted
|
|
# via `enterprise-auth -create-ingest-credential-tenant=<id>`) or
|
|
# be refused outright -- not turned on here since nothing in this
|
|
# compose file provisions one.
|
|
volumes:
|
|
- ./hack/dev-certs/out:/etc/sentry-ingest:ro
|
|
|
|
# Reads the same sentry.logs.raw topic ingest's consumer does (own
|
|
# offset tracking, own failure domain — see /search/README.md) and
|
|
# builds a Tantivy full-text index over the message field.
|
|
search:
|
|
build:
|
|
context: . # needs both search/ and proto/
|
|
dockerfile: search/Dockerfile
|
|
container_name: sentry-search
|
|
depends_on:
|
|
redpanda-provision:
|
|
condition: service_completed_successfully
|
|
environment:
|
|
REDPANDA_BROKERS: "redpanda:9092"
|
|
REDPANDA_TOPIC_PARTITIONS: "6" # must match redpanda-provision's above
|
|
# tracing-subscriber's default filter suppresses INFO without this
|
|
# -- found by actually checking `docker compose logs search` and
|
|
# seeing nothing, same silent-logging gap the agent had in Phase 0.
|
|
RUST_LOG: "info"
|
|
volumes:
|
|
- search-index-data:/var/lib/sentry-search
|
|
|
|
# Mutually exclusive with enterprise-api below, same choice Helm makes
|
|
# via enterprise.enabled (deploy/helm/sentry/templates/api.yaml vs
|
|
# enterprise-api.yaml) -- selected by the COMPOSE_PROFILES value in
|
|
# .env (checked in as "single-tenant", the zero-config default) or an
|
|
# override on the command line, e.g. `COMPOSE_PROFILES=enterprise
|
|
# docker compose up`. `docker compose run api ...` (as the manual RBAC
|
|
# testing steps in enterprise/README.md/phase-4-runbook.md §4 use)
|
|
# still works regardless of the active profile -- an explicit service
|
|
# reference on the command line bypasses profile filtering.
|
|
api:
|
|
profiles: ["single-tenant"]
|
|
build:
|
|
context: . # needs both api/ and proto/ (gRPC client to search)
|
|
dockerfile: api/Dockerfile
|
|
container_name: sentry-api
|
|
depends_on:
|
|
clickhouse-migrate:
|
|
condition: service_completed_successfully
|
|
metadata-migrate:
|
|
condition: service_completed_successfully
|
|
ports:
|
|
- "8080:8080"
|
|
environment:
|
|
CLICKHOUSE_ADDR: "clickhouse:9000"
|
|
CLICKHOUSE_PASSWORD: "sentry-dev-only"
|
|
SEARCH_GRPC_ADDR: "search:50052"
|
|
POSTGRES_ADDR: "metadata-postgres:5432"
|
|
POSTGRES_DATABASE: "sentry_metadata"
|
|
POSTGRES_USERNAME: "sentry"
|
|
POSTGRES_PASSWORD: "sentry-dev-only"
|
|
healthcheck:
|
|
# alerting (Phase 3 task 5) depends_on api -- without this, that
|
|
# dependency can only mean "container started," not "actually
|
|
# listening," and would hammer a not-yet-ready api with errors on
|
|
# every evaluator tick during stack startup. api's image is
|
|
# distroless (no shell, no wget) so this execs the api binary's own
|
|
# -healthcheck self-check mode instead of an external tool.
|
|
test: ["CMD", "/api", "-healthcheck"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
alerting:
|
|
build:
|
|
context: alerting # self-contained, no /proto needed -- see alerting/Dockerfile
|
|
dockerfile: Dockerfile
|
|
container_name: sentry-alerting
|
|
depends_on:
|
|
metadata-migrate:
|
|
condition: service_completed_successfully
|
|
# Both optional (required: false): whichever of api/enterprise-api
|
|
# is actually in the active profile set is the one this waits on
|
|
# -- the other isn't defined for this run at all, and without
|
|
# `required: false` compose would error on the inactive one rather
|
|
# than just skipping it. See api's doc comment above.
|
|
api:
|
|
condition: service_healthy
|
|
required: false
|
|
enterprise-api:
|
|
condition: service_healthy
|
|
required: false
|
|
ports:
|
|
- "8081:8081"
|
|
environment:
|
|
POSTGRES_ADDR: "metadata-postgres:5432"
|
|
POSTGRES_DATABASE: "sentry_metadata"
|
|
POSTGRES_USERNAME: "sentry"
|
|
POSTGRES_PASSWORD: "sentry-dev-only"
|
|
# Resolves to whichever of api/enterprise-api is actually active --
|
|
# enterprise-api declares a `default.aliases: [api]` network alias
|
|
# below specifically so this never needs to change based on which
|
|
# profile is selected.
|
|
API_QUERY_URL: "http://api:8080"
|
|
healthcheck:
|
|
test: ["CMD", "/alerting", "-healthcheck"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
# Commercial-license SSO/RBAC service (Phase 4) -- see
|
|
# /docs/phase-4-isolation-design.md and enterprise/README.md. Included
|
|
# here so it can be built/run/curled like every other service, but
|
|
# deliberately NOT wired into api's ENTERPRISE_AUTH_URL or alerting's
|
|
# API_SERVICE_TOKEN below: turning that on makes every /query and
|
|
# /dashboards request require a valid session/service token. Both
|
|
# OIDC and SAML login flows now exist (enterprise/internal/loginhandler),
|
|
# but this compose file sets neither OIDC_ISSUER_URL nor
|
|
# SAML_IDP_METADATA_URL, so both stay disabled here, and there's still
|
|
# no admin UI to create the first tenant_memberships row -- see
|
|
# /docs/phase-4-runbook.md sections 3a/3b for wiring a real IdP and
|
|
# bootstrapping that row by hand. Flipping enforcement on by default
|
|
# without that would break the web UI and sentryctl with no way to log
|
|
# in. See enterprise/README.md for how to turn enforcement on for
|
|
# manual testing (mint a service token, set the two env vars, restart).
|
|
enterprise-auth:
|
|
build:
|
|
context: enterprise
|
|
dockerfile: Dockerfile
|
|
container_name: sentry-enterprise-auth
|
|
depends_on:
|
|
metadata-migrate:
|
|
condition: service_completed_successfully
|
|
ports:
|
|
- "8082:8082"
|
|
environment:
|
|
# Dev-only, same framing as CLICKHOUSE_PASSWORD above -- not a real
|
|
# secret. Must be at least 32 bytes (see internal/config.Load).
|
|
ENTERPRISE_SESSION_SIGNING_KEY: "sentry-dev-only-session-signing-key-32bytes+"
|
|
POSTGRES_ADDR: "metadata-postgres:5432"
|
|
POSTGRES_DATABASE: "sentry_metadata"
|
|
POSTGRES_USERNAME: "sentry"
|
|
POSTGRES_PASSWORD: "sentry-dev-only"
|
|
# Where the browser lands after internal/loginhandler sets a
|
|
# session cookie -- web's mapped host port (see web's build args
|
|
# for why this is localhost:3000, not the compose network's
|
|
# service DNS name: the browser resolves this, not a sibling
|
|
# container).
|
|
POST_LOGIN_REDIRECT_URL: "http://localhost:3000"
|
|
healthcheck:
|
|
test: ["CMD", "/enterprise-auth", "-healthcheck"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
# Multi-tenant-aware alternative to `api` (Phase 4) -- see
|
|
# enterprise/cmd/enterprise-api/main.go's doc comment for why this is
|
|
# a second binary rather than a flag on `api`. Mutually exclusive with
|
|
# `api` above via COMPOSE_PROFILES (see that service's doc comment);
|
|
# when the "enterprise" profile is active this replaces `api` in the
|
|
# traffic path transparently, same as Helm: HTTP_LISTEN_ADDR is
|
|
# overridden to :8080 (this binary's own default is :8083) and the
|
|
# `default.aliases` entry below makes this reachable at the hostname
|
|
# `api` too, so alerting's API_QUERY_URL and web's VITE_API_BASE_URL
|
|
# need zero conditional logic -- whichever binary is actually running
|
|
# transparently answers on the same name/port either way. Nothing here
|
|
# provisions any tenants on its own (see -provision-tenant).
|
|
# CLICKHOUSE_ADMIN_USERNAME/PASSWORD reuse the same admin credential
|
|
# `clickhouse-migrate` uses, since tenantprovision needs
|
|
# access_management, not a tenant-scoped grant.
|
|
enterprise-api:
|
|
profiles: ["enterprise"]
|
|
build:
|
|
context: .
|
|
dockerfile: enterprise/cmd/enterprise-api/Dockerfile
|
|
container_name: sentry-enterprise-api
|
|
depends_on:
|
|
clickhouse-migrate:
|
|
condition: service_completed_successfully
|
|
metadata-migrate:
|
|
condition: service_completed_successfully
|
|
networks:
|
|
default:
|
|
aliases:
|
|
- api
|
|
ports:
|
|
- "8080:8080"
|
|
environment:
|
|
HTTP_LISTEN_ADDR: ":8080"
|
|
CLICKHOUSE_ADDR: "clickhouse:9000"
|
|
CLICKHOUSE_ADMIN_USERNAME: "default"
|
|
CLICKHOUSE_ADMIN_PASSWORD: "sentry-dev-only"
|
|
SEARCH_GRPC_ADDR: "search:50052"
|
|
POSTGRES_ADDR: "metadata-postgres:5432"
|
|
POSTGRES_DATABASE: "sentry_metadata"
|
|
POSTGRES_USERNAME: "sentry"
|
|
POSTGRES_PASSWORD: "sentry-dev-only"
|
|
AUDIT_WRITER_USERNAME: "audit_writer"
|
|
AUDIT_WRITER_PASSWORD: "audit-writer-dev-only"
|
|
ENTERPRISE_AUTH_URL: "http://enterprise-auth:8082"
|
|
healthcheck:
|
|
test: ["CMD", "/enterprise-api", "-healthcheck"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 30
|
|
|
|
web:
|
|
build:
|
|
context: web
|
|
args:
|
|
# Baked in at build time (static site, not a server) as
|
|
# localhost:8080/8081 -- fetched from the *browser*, which
|
|
# resolves against the host's mapped ports, not the compose
|
|
# network's service DNS names.
|
|
# localhost:8080 works unchanged regardless of which profile is
|
|
# active -- enterprise-api maps the same host port api does when
|
|
# it's the one running (see that service's doc comment).
|
|
VITE_API_BASE_URL: "http://localhost:8080"
|
|
VITE_ALERTING_API_BASE_URL: "http://localhost:8081"
|
|
VITE_ENTERPRISE_AUTH_BASE_URL: "http://localhost:8082"
|
|
container_name: sentry-web
|
|
depends_on:
|
|
# api/enterprise-api optional, same reasoning as alerting's
|
|
# depends_on above -- only one is ever in the active profile set.
|
|
api:
|
|
condition: service_started
|
|
required: false
|
|
enterprise-api:
|
|
condition: service_started
|
|
required: false
|
|
alerting:
|
|
condition: service_started
|
|
enterprise-auth:
|
|
condition: service_started
|
|
ports:
|
|
- "3000:3000"
|
|
|
|
volumes:
|
|
redpanda-data:
|
|
clickhouse-data:
|
|
search-index-data:
|
|
metadata-postgres-data:
|