Phase 4: SSO scaffolding, RBAC enforcement, tenant-scoped dashboards, audit logging, K8s deployment

RBAC (api/internal/authz) is live on /query and /dashboards, backed by a
new enterprise/ module (session issuance, audit logging, RBAC storage,
OIDC/SAML protocol wiring) that core never imports -- only calls over
HTTP. Found and fixed a real cross-tenant vulnerability in dashboards
(no tenant_id filtering at all) while writing the threat model doc.

Two things are explicitly NOT done, documented rather than hidden:
tenant isolation for log data itself (/query still shares one ClickHouse
connection and Tantivy index across every tenant -- RBAC controls who
can query, not what a query can see), and human SSO login (protocol
wiring exists, no HTTP handler calls it yet). See
docs/security/threat-model.md and docs/phase-4-runbook.md.

Also adds deploy/ (Go Operator + Helm chart, validated offline only --
no cluster was reachable in this environment).
This commit is contained in:
2026-08-13 22:16:59 -07:00
parent 9435115ab7
commit 3eb0f4c589
116 changed files with 8589 additions and 126 deletions
+69 -3
View File
@@ -19,19 +19,58 @@ import (
"strings"
"time"
"github.com/sentry/sentry/api/internal/authz"
"github.com/sentry/sentry/api/internal/querylang/executor"
"github.com/sentry/sentry/api/internal/querylang/planner"
)
// AuditLogger is core's extension point for query audit logging --
// deliberately minimal and tenant-agnostic, since core has no concept of
// tenants (see /docs/phase-4-isolation-design.md: that mechanism lives
// entirely in enterprise/). enterprise/internal/audit implements this
// against the real hash-chained, append-only store; a nil AuditLogger
// (the default for a single-tenant deployment without enterprise/
// configured) means no audit logging happens and core behaves exactly
// as it did in Phases 0-3.
//
// Tenant/user identity is deliberately NOT a field on QueryAuditEntry --
// once Phase 4 task 5's auth middleware wraps this handler, it attaches
// that identity to the request's context.Context via
// enterprise/internal/tenant, and LogQuery's ctx parameter is the same
// context the request carried, so an enterprise-side implementation
// reads identity from ctx rather than this interface growing
// tenant-awareness. Per /docs/phase-4-isolation-design.md's audit
// section, this is a fail-open path: a LogQuery error is logged but
// never fails the HTTP response for a routine read query.
type AuditLogger interface {
LogQuery(ctx context.Context, entry QueryAuditEntry) error
}
type QueryAuditEntry struct {
Query string
Language string
RowCount int
Duration time.Duration
Success bool
Error string
}
type Handler struct {
logger *slog.Logger
sqlRunner executor.SQLRunner
search executor.SearchClient
queryTimeout time.Duration
audit AuditLogger
authorizer authz.Authorizer
}
func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search executor.SearchClient, queryTimeout time.Duration) *Handler {
return &Handler{logger: logger, sqlRunner: sqlRunner, search: search, queryTimeout: queryTimeout}
// audit and authorizer may both be nil -- see AuditLogger's doc comment
// and authz.RequireRoleOrService's nil-safety. /query allows RoleViewer
// (human sessions) or the alerting service identity (RoleService) --
// it's the one endpoint /alerting's evaluator legitimately calls, per
// /docs/phase-4-isolation-design.md's alerting service-identity design.
func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search executor.SearchClient, queryTimeout time.Duration, audit AuditLogger, authorizer authz.Authorizer) *Handler {
return &Handler{logger: logger, sqlRunner: sqlRunner, search: search, queryTimeout: queryTimeout, audit: audit, authorizer: authorizer}
}
// RegisterRoutes adds this handler's routes onto a shared mux. Phase 3
@@ -40,7 +79,7 @@ func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search execut
// than by each handler wrapping itself individually -- see
// httpserver.WithCORS.
func (h *Handler) RegisterRoutes(mux *http.ServeMux) {
mux.HandleFunc("POST /query", h.handleQuery)
mux.HandleFunc("POST /query", authz.RequireRoleOrService(h.authorizer, authz.RoleViewer, h.handleQuery))
mux.HandleFunc("GET /healthz", h.handleHealthz)
}
@@ -98,16 +137,43 @@ func (h *Handler) handleQuery(w http.ResponseWriter, r *http.Request) {
ctx, cancel := context.WithTimeout(r.Context(), h.queryTimeout)
defer cancel()
start := time.Now()
result, err := executor.Execute(ctx, plan, h.sqlRunner, h.search)
duration := time.Since(start)
if err != nil {
h.logger.Error("query execution failed", "query", req.Query, "error", err)
h.logAudit(r.Context(), req, 0, duration, err)
writeError(w, http.StatusBadGateway, "query failed: "+err.Error())
return
}
h.logAudit(r.Context(), req, len(result.Rows), duration, nil)
writeJSON(w, queryResponse{Columns: result.Columns, Rows: result.Rows})
}
// logAudit is fail-open by design (see AuditLogger's doc comment): a
// write failure here is logged and otherwise ignored, never surfaced to
// the HTTP caller. Uses r.Context() (the original request context, not
// the query-execution one with its own deadline) so a slow/cancelled
// query's context.WithTimeout expiring doesn't also cancel the audit
// write for it.
func (h *Handler) logAudit(ctx context.Context, req queryRequest, rowCount int, duration time.Duration, execErr error) {
if h.audit == nil {
return
}
entry := QueryAuditEntry{
Query: req.Query, Language: req.Language, RowCount: rowCount,
Duration: duration, Success: execErr == nil,
}
if execErr != nil {
entry.Error = execErr.Error()
}
if err := h.audit.LogQuery(ctx, entry); err != nil {
h.logger.Error("audit log write failed", "error", err)
}
}
func writeJSON(w http.ResponseWriter, v any) {
w.Header().Set("Content-Type", "application/json")
_ = json.NewEncoder(w).Encode(v)
+134 -1
View File
@@ -12,6 +12,7 @@ import (
"testing"
"time"
"github.com/sentry/sentry/api/internal/authz"
"github.com/sentry/sentry/api/internal/querylang/executor"
)
@@ -50,7 +51,21 @@ func newTestHandler(sqlRunner *fakeSQLRunner, search *fakeSearchClient) *Handler
if search == nil {
search = &fakeSearchClient{}
}
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, search, time.Second)
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, search, time.Second, nil, nil)
}
type fakeAuditLogger struct {
entries []QueryAuditEntry
err error
}
func (f *fakeAuditLogger) LogQuery(_ context.Context, entry QueryAuditEntry) error {
f.entries = append(f.entries, entry)
return f.err
}
func newTestHandlerWithAudit(sqlRunner *fakeSQLRunner, audit *fakeAuditLogger) *Handler {
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, &fakeSearchClient{}, time.Second, audit, nil)
}
func newTestMux(h *Handler) *http.ServeMux {
@@ -207,3 +222,121 @@ func TestHandleHealthz(t *testing.T) {
t.Fatalf("status = %d, want 200", rec.Code)
}
}
func TestHandleQueryLogsAuditEntryOnSuccess(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{"host"}, Rows: [][]any{{"h1"}, {"h2"}}}}
audit := &fakeAuditLogger{}
h := newTestHandlerWithAudit(sr, audit)
rec := postQuery(t, h, `{"query": "SELECT host FROM logs"}`)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
}
if len(audit.entries) != 1 {
t.Fatalf("expected 1 audit entry, got %d", len(audit.entries))
}
entry := audit.entries[0]
if entry.Query != "SELECT host FROM logs" || !entry.Success || entry.RowCount != 2 || entry.Error != "" {
t.Fatalf("unexpected audit entry: %+v", entry)
}
}
func TestHandleQueryLogsAuditEntryOnFailure(t *testing.T) {
sr := &fakeSQLRunner{err: errors.New("boom")}
audit := &fakeAuditLogger{}
h := newTestHandlerWithAudit(sr, audit)
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusBadGateway {
t.Fatalf("status = %d, want 502", rec.Code)
}
if len(audit.entries) != 1 {
t.Fatalf("expected 1 audit entry even on failure, got %d", len(audit.entries))
}
entry := audit.entries[0]
if entry.Success || entry.Error == "" {
t.Fatalf("expected a failed audit entry with an error message, got %+v", entry)
}
}
// TestHandleQueryAuditWriteFailureDoesNotFailRequest proves the
// fail-open design: a request still succeeds even when the audit
// logger itself errors -- per queryapi.AuditLogger's doc comment and
// /docs/phase-4-isolation-design.md's audit fail-open/fail-closed policy.
func TestHandleQueryAuditWriteFailureDoesNotFailRequest(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
audit := &fakeAuditLogger{err: errors.New("audit backend unreachable")}
h := newTestHandlerWithAudit(sr, audit)
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200 -- an audit write failure must not fail the request", rec.Code)
}
}
func TestHandleQueryNilAuditLoggerIsNoOp(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
h := newTestHandler(sr, nil) // audit is nil here
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200", rec.Code)
}
}
// fakeAuthorizer resolves every request to a fixed identity/error --
// task 5 wired authz.RequireRoleOrService into RegisterRoutes but every
// existing test above passes a nil authorizer (a deliberate no-op), so
// none of them actually exercise the wiring with a real authorizer
// present. Phase 4 task 8 (adversarial tests) closes that gap: these
// prove /query's authz boundary holds when a real Authorizer is wired
// in, not just that the middleware function works in isolation
// (authz/middleware_test.go already covers that).
type fakeAuthorizer struct {
identity authz.Identity
err error
}
func (f *fakeAuthorizer) Authorize(*http.Request) (authz.Identity, error) {
return f.identity, f.err
}
func newTestHandlerWithAuthorizer(sqlRunner *fakeSQLRunner, authorizer authz.Authorizer) *Handler {
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, &fakeSearchClient{}, time.Second, nil, authorizer)
}
func TestHandleQueryRejectsUnauthenticatedWhenAuthorizerConfigured(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{err: errors.New("no session")})
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusUnauthorized {
t.Fatalf("status = %d, want 401", rec.Code)
}
}
func TestHandleQueryAllowsViewer(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{identity: authz.Identity{TenantID: "acme", Role: authz.RoleViewer}})
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
}
}
// TestHandleQueryAllowsServiceIdentity is the other half of the
// alerting<->api gap's fix (/docs/phase-4-isolation-design.md) --
// /alerting's evaluator must be able to call POST /query with its
// RoleService credential even though it's not a human session.
func TestHandleQueryAllowsServiceIdentity(t *testing.T) {
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{identity: authz.Identity{Role: authz.RoleService}})
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200 -- RoleService must be allowed on /query; body=%s", rec.Code, rec.Body.String())
}
}
@@ -0,0 +1,63 @@
// This file is a checklist, not a passing test suite -- it exists so
// the four adversarial probes /docs/phase-4-isolation-design.md's
// "Verification plan for this design specifically" section names for
// Phase 4 task 8 have a permanent, grep-able home in the test tree,
// even though none of them can run for real yet.
//
// Why they can't run: every one of these probes needs a *per-tenant*
// ClickHouse user/database or Tantivy index to attack -- and none
// exist. api/internal/querylang/executor.SQLRunner/SearchClient (the
// only two interfaces api/internal/queryapi.Handler talks to) carry no
// tenant field at all, confirmed by reading both interfaces; neither
// does proto/sentry/search/v1/search.proto's SearchRequest. See
// /docs/security/threat-model.md's "Read this first" section for the
// full writeup -- there is currently exactly one shared ClickHouse
// connection and one shared Tantivy index for every tenant, so "does
// tenant A's connection leak tenant B's data" has no meaningful
// operational answer yet: there's only one connection.
//
// Each Skip below names precisely what has to exist before that test
// can be written for real (enterprise/internal/tenantprovision,
// enterprise/internal/chrunner, enterprise/internal/searchclient -- all
// still unbuilt, per the Phase 4 task 5 summary). Turning a Skip here
// into a real assertion is the acceptance criterion for those packages,
// not a nice-to-have follow-up.
package queryapi
import "testing"
func TestAdversarial_ClickHouseUserCannotReadOtherTenantDatabaseByFullyQualifiedName(t *testing.T) {
t.Skip("BLOCKED on enterprise/internal/tenantprovision + enterprise/internal/chrunner: " +
"needs two real per-tenant ClickHouse users/databases to attempt " +
"`SELECT * FROM other_tenant_db.logs` against. See " +
"/docs/phase-4-isolation-design.md's verification plan, item 1.")
}
func TestAdversarial_ClickHouseUserCannotReadSystemTables(t *testing.T) {
t.Skip("BLOCKED on enterprise/internal/tenantprovision: needs a real per-tenant " +
"ClickHouse user to attempt `SELECT * FROM system.query_log`, " +
"`system.tables`, `SHOW DATABASES` against, and confirm system.* " +
"access was actually revoked (not just assumed from ClickHouse's " +
"default template -- task 2's finding was that this is " +
"version-dependent and must be checked live, not read from docs). " +
"See /docs/phase-4-isolation-design.md's verification plan, item 2.")
}
func TestAdversarial_TantivySearchExcludesOtherTenantsMatchingResults(t *testing.T) {
t.Skip("BLOCKED on enterprise/internal/searchclient: needs two real " +
"per-tenant Tantivy indices, one seeded with a term, to confirm a " +
"search scoped to the other tenant returns zero hits for that term " +
"even though the term exists in the other index. See " +
"/docs/phase-4-isolation-design.md's verification plan, item 3.")
}
func TestAdversarial_EvaluatorTickMidProvisioningIsRefusedNotServed(t *testing.T) {
t.Skip("BLOCKED on enterprise/internal/tenantprovision's ordered " +
"provisioning state machine (CREATE USER -> GRANT -> mark active): " +
"needs a tenant row that exists but hasn't reached the active gate " +
"yet, and a simulated /alerting evaluator tick against it, to " +
"confirm every tenant-resolution path actually checks tenant " +
"status server-side rather than inferring readiness from ambient " +
"connection success. See /docs/phase-4-isolation-design.md's " +
"verification plan, item 4, and its provisioning-gate requirement.")
}