Phase 4: SSO scaffolding, RBAC enforcement, tenant-scoped dashboards, audit logging, K8s deployment
RBAC (api/internal/authz) is live on /query and /dashboards, backed by a new enterprise/ module (session issuance, audit logging, RBAC storage, OIDC/SAML protocol wiring) that core never imports -- only calls over HTTP. Found and fixed a real cross-tenant vulnerability in dashboards (no tenant_id filtering at all) while writing the threat model doc. Two things are explicitly NOT done, documented rather than hidden: tenant isolation for log data itself (/query still shares one ClickHouse connection and Tantivy index across every tenant -- RBAC controls who can query, not what a query can see), and human SSO login (protocol wiring exists, no HTTP handler calls it yet). See docs/security/threat-model.md and docs/phase-4-runbook.md. Also adds deploy/ (Go Operator + Helm chart, validated offline only -- no cluster was reachable in this environment).
This commit is contained in:
@@ -19,19 +19,58 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/sentry/sentry/api/internal/authz"
|
||||
"github.com/sentry/sentry/api/internal/querylang/executor"
|
||||
"github.com/sentry/sentry/api/internal/querylang/planner"
|
||||
)
|
||||
|
||||
// AuditLogger is core's extension point for query audit logging --
|
||||
// deliberately minimal and tenant-agnostic, since core has no concept of
|
||||
// tenants (see /docs/phase-4-isolation-design.md: that mechanism lives
|
||||
// entirely in enterprise/). enterprise/internal/audit implements this
|
||||
// against the real hash-chained, append-only store; a nil AuditLogger
|
||||
// (the default for a single-tenant deployment without enterprise/
|
||||
// configured) means no audit logging happens and core behaves exactly
|
||||
// as it did in Phases 0-3.
|
||||
//
|
||||
// Tenant/user identity is deliberately NOT a field on QueryAuditEntry --
|
||||
// once Phase 4 task 5's auth middleware wraps this handler, it attaches
|
||||
// that identity to the request's context.Context via
|
||||
// enterprise/internal/tenant, and LogQuery's ctx parameter is the same
|
||||
// context the request carried, so an enterprise-side implementation
|
||||
// reads identity from ctx rather than this interface growing
|
||||
// tenant-awareness. Per /docs/phase-4-isolation-design.md's audit
|
||||
// section, this is a fail-open path: a LogQuery error is logged but
|
||||
// never fails the HTTP response for a routine read query.
|
||||
type AuditLogger interface {
|
||||
LogQuery(ctx context.Context, entry QueryAuditEntry) error
|
||||
}
|
||||
|
||||
type QueryAuditEntry struct {
|
||||
Query string
|
||||
Language string
|
||||
RowCount int
|
||||
Duration time.Duration
|
||||
Success bool
|
||||
Error string
|
||||
}
|
||||
|
||||
type Handler struct {
|
||||
logger *slog.Logger
|
||||
sqlRunner executor.SQLRunner
|
||||
search executor.SearchClient
|
||||
queryTimeout time.Duration
|
||||
audit AuditLogger
|
||||
authorizer authz.Authorizer
|
||||
}
|
||||
|
||||
func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search executor.SearchClient, queryTimeout time.Duration) *Handler {
|
||||
return &Handler{logger: logger, sqlRunner: sqlRunner, search: search, queryTimeout: queryTimeout}
|
||||
// audit and authorizer may both be nil -- see AuditLogger's doc comment
|
||||
// and authz.RequireRoleOrService's nil-safety. /query allows RoleViewer
|
||||
// (human sessions) or the alerting service identity (RoleService) --
|
||||
// it's the one endpoint /alerting's evaluator legitimately calls, per
|
||||
// /docs/phase-4-isolation-design.md's alerting service-identity design.
|
||||
func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search executor.SearchClient, queryTimeout time.Duration, audit AuditLogger, authorizer authz.Authorizer) *Handler {
|
||||
return &Handler{logger: logger, sqlRunner: sqlRunner, search: search, queryTimeout: queryTimeout, audit: audit, authorizer: authorizer}
|
||||
}
|
||||
|
||||
// RegisterRoutes adds this handler's routes onto a shared mux. Phase 3
|
||||
@@ -40,7 +79,7 @@ func NewHandler(logger *slog.Logger, sqlRunner executor.SQLRunner, search execut
|
||||
// than by each handler wrapping itself individually -- see
|
||||
// httpserver.WithCORS.
|
||||
func (h *Handler) RegisterRoutes(mux *http.ServeMux) {
|
||||
mux.HandleFunc("POST /query", h.handleQuery)
|
||||
mux.HandleFunc("POST /query", authz.RequireRoleOrService(h.authorizer, authz.RoleViewer, h.handleQuery))
|
||||
mux.HandleFunc("GET /healthz", h.handleHealthz)
|
||||
}
|
||||
|
||||
@@ -98,16 +137,43 @@ func (h *Handler) handleQuery(w http.ResponseWriter, r *http.Request) {
|
||||
ctx, cancel := context.WithTimeout(r.Context(), h.queryTimeout)
|
||||
defer cancel()
|
||||
|
||||
start := time.Now()
|
||||
result, err := executor.Execute(ctx, plan, h.sqlRunner, h.search)
|
||||
duration := time.Since(start)
|
||||
|
||||
if err != nil {
|
||||
h.logger.Error("query execution failed", "query", req.Query, "error", err)
|
||||
h.logAudit(r.Context(), req, 0, duration, err)
|
||||
writeError(w, http.StatusBadGateway, "query failed: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
h.logAudit(r.Context(), req, len(result.Rows), duration, nil)
|
||||
writeJSON(w, queryResponse{Columns: result.Columns, Rows: result.Rows})
|
||||
}
|
||||
|
||||
// logAudit is fail-open by design (see AuditLogger's doc comment): a
|
||||
// write failure here is logged and otherwise ignored, never surfaced to
|
||||
// the HTTP caller. Uses r.Context() (the original request context, not
|
||||
// the query-execution one with its own deadline) so a slow/cancelled
|
||||
// query's context.WithTimeout expiring doesn't also cancel the audit
|
||||
// write for it.
|
||||
func (h *Handler) logAudit(ctx context.Context, req queryRequest, rowCount int, duration time.Duration, execErr error) {
|
||||
if h.audit == nil {
|
||||
return
|
||||
}
|
||||
entry := QueryAuditEntry{
|
||||
Query: req.Query, Language: req.Language, RowCount: rowCount,
|
||||
Duration: duration, Success: execErr == nil,
|
||||
}
|
||||
if execErr != nil {
|
||||
entry.Error = execErr.Error()
|
||||
}
|
||||
if err := h.audit.LogQuery(ctx, entry); err != nil {
|
||||
h.logger.Error("audit log write failed", "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func writeJSON(w http.ResponseWriter, v any) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_ = json.NewEncoder(w).Encode(v)
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sentry/sentry/api/internal/authz"
|
||||
"github.com/sentry/sentry/api/internal/querylang/executor"
|
||||
)
|
||||
|
||||
@@ -50,7 +51,21 @@ func newTestHandler(sqlRunner *fakeSQLRunner, search *fakeSearchClient) *Handler
|
||||
if search == nil {
|
||||
search = &fakeSearchClient{}
|
||||
}
|
||||
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, search, time.Second)
|
||||
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, search, time.Second, nil, nil)
|
||||
}
|
||||
|
||||
type fakeAuditLogger struct {
|
||||
entries []QueryAuditEntry
|
||||
err error
|
||||
}
|
||||
|
||||
func (f *fakeAuditLogger) LogQuery(_ context.Context, entry QueryAuditEntry) error {
|
||||
f.entries = append(f.entries, entry)
|
||||
return f.err
|
||||
}
|
||||
|
||||
func newTestHandlerWithAudit(sqlRunner *fakeSQLRunner, audit *fakeAuditLogger) *Handler {
|
||||
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, &fakeSearchClient{}, time.Second, audit, nil)
|
||||
}
|
||||
|
||||
func newTestMux(h *Handler) *http.ServeMux {
|
||||
@@ -207,3 +222,121 @@ func TestHandleHealthz(t *testing.T) {
|
||||
t.Fatalf("status = %d, want 200", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleQueryLogsAuditEntryOnSuccess(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{"host"}, Rows: [][]any{{"h1"}, {"h2"}}}}
|
||||
audit := &fakeAuditLogger{}
|
||||
h := newTestHandlerWithAudit(sr, audit)
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT host FROM logs"}`)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
|
||||
}
|
||||
|
||||
if len(audit.entries) != 1 {
|
||||
t.Fatalf("expected 1 audit entry, got %d", len(audit.entries))
|
||||
}
|
||||
entry := audit.entries[0]
|
||||
if entry.Query != "SELECT host FROM logs" || !entry.Success || entry.RowCount != 2 || entry.Error != "" {
|
||||
t.Fatalf("unexpected audit entry: %+v", entry)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleQueryLogsAuditEntryOnFailure(t *testing.T) {
|
||||
sr := &fakeSQLRunner{err: errors.New("boom")}
|
||||
audit := &fakeAuditLogger{}
|
||||
h := newTestHandlerWithAudit(sr, audit)
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusBadGateway {
|
||||
t.Fatalf("status = %d, want 502", rec.Code)
|
||||
}
|
||||
|
||||
if len(audit.entries) != 1 {
|
||||
t.Fatalf("expected 1 audit entry even on failure, got %d", len(audit.entries))
|
||||
}
|
||||
entry := audit.entries[0]
|
||||
if entry.Success || entry.Error == "" {
|
||||
t.Fatalf("expected a failed audit entry with an error message, got %+v", entry)
|
||||
}
|
||||
}
|
||||
|
||||
// TestHandleQueryAuditWriteFailureDoesNotFailRequest proves the
|
||||
// fail-open design: a request still succeeds even when the audit
|
||||
// logger itself errors -- per queryapi.AuditLogger's doc comment and
|
||||
// /docs/phase-4-isolation-design.md's audit fail-open/fail-closed policy.
|
||||
func TestHandleQueryAuditWriteFailureDoesNotFailRequest(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
|
||||
audit := &fakeAuditLogger{err: errors.New("audit backend unreachable")}
|
||||
h := newTestHandlerWithAudit(sr, audit)
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200 -- an audit write failure must not fail the request", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleQueryNilAuditLoggerIsNoOp(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
|
||||
h := newTestHandler(sr, nil) // audit is nil here
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
// fakeAuthorizer resolves every request to a fixed identity/error --
|
||||
// task 5 wired authz.RequireRoleOrService into RegisterRoutes but every
|
||||
// existing test above passes a nil authorizer (a deliberate no-op), so
|
||||
// none of them actually exercise the wiring with a real authorizer
|
||||
// present. Phase 4 task 8 (adversarial tests) closes that gap: these
|
||||
// prove /query's authz boundary holds when a real Authorizer is wired
|
||||
// in, not just that the middleware function works in isolation
|
||||
// (authz/middleware_test.go already covers that).
|
||||
type fakeAuthorizer struct {
|
||||
identity authz.Identity
|
||||
err error
|
||||
}
|
||||
|
||||
func (f *fakeAuthorizer) Authorize(*http.Request) (authz.Identity, error) {
|
||||
return f.identity, f.err
|
||||
}
|
||||
|
||||
func newTestHandlerWithAuthorizer(sqlRunner *fakeSQLRunner, authorizer authz.Authorizer) *Handler {
|
||||
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), sqlRunner, &fakeSearchClient{}, time.Second, nil, authorizer)
|
||||
}
|
||||
|
||||
func TestHandleQueryRejectsUnauthenticatedWhenAuthorizerConfigured(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
|
||||
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{err: errors.New("no session")})
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusUnauthorized {
|
||||
t.Fatalf("status = %d, want 401", rec.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleQueryAllowsViewer(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
|
||||
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{identity: authz.Identity{TenantID: "acme", Role: authz.RoleViewer}})
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestHandleQueryAllowsServiceIdentity is the other half of the
|
||||
// alerting<->api gap's fix (/docs/phase-4-isolation-design.md) --
|
||||
// /alerting's evaluator must be able to call POST /query with its
|
||||
// RoleService credential even though it's not a human session.
|
||||
func TestHandleQueryAllowsServiceIdentity(t *testing.T) {
|
||||
sr := &fakeSQLRunner{result: &executor.Result{Columns: []string{}, Rows: [][]any{}}}
|
||||
h := newTestHandlerWithAuthorizer(sr, &fakeAuthorizer{identity: authz.Identity{Role: authz.RoleService}})
|
||||
|
||||
rec := postQuery(t, h, `{"query": "SELECT 1"}`)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200 -- RoleService must be allowed on /query; body=%s", rec.Code, rec.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
// This file is a checklist, not a passing test suite -- it exists so
|
||||
// the four adversarial probes /docs/phase-4-isolation-design.md's
|
||||
// "Verification plan for this design specifically" section names for
|
||||
// Phase 4 task 8 have a permanent, grep-able home in the test tree,
|
||||
// even though none of them can run for real yet.
|
||||
//
|
||||
// Why they can't run: every one of these probes needs a *per-tenant*
|
||||
// ClickHouse user/database or Tantivy index to attack -- and none
|
||||
// exist. api/internal/querylang/executor.SQLRunner/SearchClient (the
|
||||
// only two interfaces api/internal/queryapi.Handler talks to) carry no
|
||||
// tenant field at all, confirmed by reading both interfaces; neither
|
||||
// does proto/sentry/search/v1/search.proto's SearchRequest. See
|
||||
// /docs/security/threat-model.md's "Read this first" section for the
|
||||
// full writeup -- there is currently exactly one shared ClickHouse
|
||||
// connection and one shared Tantivy index for every tenant, so "does
|
||||
// tenant A's connection leak tenant B's data" has no meaningful
|
||||
// operational answer yet: there's only one connection.
|
||||
//
|
||||
// Each Skip below names precisely what has to exist before that test
|
||||
// can be written for real (enterprise/internal/tenantprovision,
|
||||
// enterprise/internal/chrunner, enterprise/internal/searchclient -- all
|
||||
// still unbuilt, per the Phase 4 task 5 summary). Turning a Skip here
|
||||
// into a real assertion is the acceptance criterion for those packages,
|
||||
// not a nice-to-have follow-up.
|
||||
package queryapi
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestAdversarial_ClickHouseUserCannotReadOtherTenantDatabaseByFullyQualifiedName(t *testing.T) {
|
||||
t.Skip("BLOCKED on enterprise/internal/tenantprovision + enterprise/internal/chrunner: " +
|
||||
"needs two real per-tenant ClickHouse users/databases to attempt " +
|
||||
"`SELECT * FROM other_tenant_db.logs` against. See " +
|
||||
"/docs/phase-4-isolation-design.md's verification plan, item 1.")
|
||||
}
|
||||
|
||||
func TestAdversarial_ClickHouseUserCannotReadSystemTables(t *testing.T) {
|
||||
t.Skip("BLOCKED on enterprise/internal/tenantprovision: needs a real per-tenant " +
|
||||
"ClickHouse user to attempt `SELECT * FROM system.query_log`, " +
|
||||
"`system.tables`, `SHOW DATABASES` against, and confirm system.* " +
|
||||
"access was actually revoked (not just assumed from ClickHouse's " +
|
||||
"default template -- task 2's finding was that this is " +
|
||||
"version-dependent and must be checked live, not read from docs). " +
|
||||
"See /docs/phase-4-isolation-design.md's verification plan, item 2.")
|
||||
}
|
||||
|
||||
func TestAdversarial_TantivySearchExcludesOtherTenantsMatchingResults(t *testing.T) {
|
||||
t.Skip("BLOCKED on enterprise/internal/searchclient: needs two real " +
|
||||
"per-tenant Tantivy indices, one seeded with a term, to confirm a " +
|
||||
"search scoped to the other tenant returns zero hits for that term " +
|
||||
"even though the term exists in the other index. See " +
|
||||
"/docs/phase-4-isolation-design.md's verification plan, item 3.")
|
||||
}
|
||||
|
||||
func TestAdversarial_EvaluatorTickMidProvisioningIsRefusedNotServed(t *testing.T) {
|
||||
t.Skip("BLOCKED on enterprise/internal/tenantprovision's ordered " +
|
||||
"provisioning state machine (CREATE USER -> GRANT -> mark active): " +
|
||||
"needs a tenant row that exists but hasn't reached the active gate " +
|
||||
"yet, and a simulated /alerting evaluator tick against it, to " +
|
||||
"confirm every tenant-resolution path actually checks tenant " +
|
||||
"status server-side rather than inferring readiness from ambient " +
|
||||
"connection success. See /docs/phase-4-isolation-design.md's " +
|
||||
"verification plan, item 4, and its provisioning-gate requirement.")
|
||||
}
|
||||
Reference in New Issue
Block a user