Saved, shareable multi-panel dashboards (table/line/bar/single-stat panels via gridstack + uPlot, global + per-panel time range, JSON export/import) and threshold/absence alert rules with an ok/pending/firing evaluator and webhook/Slack/PagerDuty delivery. - New /metadata component: Postgres control-plane store for dashboards, panels, notification targets, alert rules/state, and delivery log -- see docs/phase-3-dashboard-design.md for why ClickHouse's MergeTree family isn't a fit for this access pattern (needs real row-level locking and read-your-writes consistency). - api/internal/dashboards: dashboard/panel CRUD, pure -- panel query execution stays client-side, reusing the existing /query endpoint. - New /alerting service: rule/target CRUD, a ticker-driven evaluator (claim-then-evaluate concurrency control, transactional-outbox delivery, query errors and threshold zero-rows never coerced into a false transition) and webhook/Slack/PagerDuty delivery with retry/backoff. See docs/phase-3-alerting-design.md for the full state-machine design and the four correctness properties it implements. - web: /dashboards and /alerts UIs; cli: sentryctl dashboards/alerts list/get/apply, seeding a future Terraform provider's JSON contract. - hack/alert-load-test: 500 rules against real ClickHouse data, real measured results in docs/phase-3-runbook.md. Five real bugs found by actually running this against a live stack (documented in the runbook, not just fixed silently): a latent Phase 2 bug where ClickHouse rejected the timestamp format used for earliest=/latest= queries; a "now" literal token injected into query text; a GridStack/uPlot layout-timing race; JS's Date.parse being too lenient to use as a timestamp-detection heuristic; a rule's "enabled" field silently defaulting to false when omitted; and the evaluator's claim-batch-size and worker-pool-concurrency defaulting to the same value, causing 500 concurrently-due rules to take 125s to cycle through instead of the configured 60s.
97 lines
3.4 KiB
Go
97 lines
3.4 KiB
Go
// Package config loads alerting's configuration from environment
|
|
// variables, same convention as /api and /ingest: no config file format.
|
|
package config
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"strconv"
|
|
"time"
|
|
)
|
|
|
|
type Config struct {
|
|
HTTPListenAddr string
|
|
Postgres PostgresConfig
|
|
APIQueryURL string // base URL of /api, e.g. http://api:8080 -- alerting never talks to ClickHouse/Tantivy directly
|
|
CORSAllowedOrigin string
|
|
Evaluator EvaluatorConfig
|
|
}
|
|
|
|
type PostgresConfig struct {
|
|
Addr string
|
|
Database string
|
|
Username string
|
|
Password string
|
|
}
|
|
|
|
type EvaluatorConfig struct {
|
|
TickInterval time.Duration // how often the scheduler checks for due rules
|
|
// ClaimBatchSize and WorkerPoolSize are deliberately separate knobs,
|
|
// not the same number: ClaimBatchSize bounds how many due rules one
|
|
// tick pulls off the queue (needs to be large enough to drain a
|
|
// backlog when many rules share a due time, e.g. right after bulk
|
|
// creation), while WorkerPoolSize bounds concurrent /query calls
|
|
// within that batch. Found by actually running hack/alert-load-test
|
|
// with 500 rules and both numbers defaulted to the same value
|
|
// (20): the evaluator took 125s to cycle through 500 due rules
|
|
// instead of the configured 60s eval_interval_seconds, because each
|
|
// 5s tick could only claim 20 rules regardless of how many more were
|
|
// already due -- see /docs/phase-3-runbook.md's load-test section.
|
|
ClaimBatchSize int
|
|
WorkerPoolSize int
|
|
QueryTimeout time.Duration // per-evaluation POST /query timeout
|
|
}
|
|
|
|
func Load() (Config, error) {
|
|
cfg := Config{
|
|
HTTPListenAddr: getenv("HTTP_LISTEN_ADDR", ":8081"),
|
|
Postgres: PostgresConfig{
|
|
Addr: getenv("POSTGRES_ADDR", "localhost:5432"),
|
|
Database: getenv("POSTGRES_DATABASE", "sentry_metadata"),
|
|
Username: getenv("POSTGRES_USERNAME", "sentry"),
|
|
Password: getenv("POSTGRES_PASSWORD", ""),
|
|
},
|
|
APIQueryURL: getenv("API_QUERY_URL", "http://localhost:8080"),
|
|
// Same "no auth yet" tradeoff as api's CORSAllowedOrigin default --
|
|
// see api/internal/config/config.go's comment, same reasoning here.
|
|
CORSAllowedOrigin: getenv("CORS_ALLOWED_ORIGIN", "*"),
|
|
}
|
|
|
|
tickSec, err := strconv.Atoi(getenv("EVALUATOR_TICK_SECONDS", "5"))
|
|
if err != nil {
|
|
return Config{}, fmt.Errorf("EVALUATOR_TICK_SECONDS: %w", err)
|
|
}
|
|
cfg.Evaluator.TickInterval = time.Duration(tickSec) * time.Second
|
|
|
|
poolSize, err := strconv.Atoi(getenv("EVALUATOR_WORKER_POOL_SIZE", "20"))
|
|
if err != nil {
|
|
return Config{}, fmt.Errorf("EVALUATOR_WORKER_POOL_SIZE: %w", err)
|
|
}
|
|
cfg.Evaluator.WorkerPoolSize = poolSize
|
|
|
|
// Default well above WorkerPoolSize: this is "how many due rules can
|
|
// one tick pull off the queue," not a concurrency limit -- the
|
|
// worker pool below still bounds actual concurrent /query calls
|
|
// regardless of how large a batch gets claimed.
|
|
claimBatchSize, err := strconv.Atoi(getenv("EVALUATOR_CLAIM_BATCH_SIZE", "1000"))
|
|
if err != nil {
|
|
return Config{}, fmt.Errorf("EVALUATOR_CLAIM_BATCH_SIZE: %w", err)
|
|
}
|
|
cfg.Evaluator.ClaimBatchSize = claimBatchSize
|
|
|
|
queryTimeoutSec, err := strconv.Atoi(getenv("EVALUATOR_QUERY_TIMEOUT_SECONDS", "30"))
|
|
if err != nil {
|
|
return Config{}, fmt.Errorf("EVALUATOR_QUERY_TIMEOUT_SECONDS: %w", err)
|
|
}
|
|
cfg.Evaluator.QueryTimeout = time.Duration(queryTimeoutSec) * time.Second
|
|
|
|
return cfg, nil
|
|
}
|
|
|
|
func getenv(key, fallback string) string {
|
|
if v := os.Getenv(key); v != "" {
|
|
return v
|
|
}
|
|
return fallback
|
|
}
|