Phase 1: Windows log collection + full-text search
Extends the agent, ingest, storage, api, and web with Windows Event Log/ETW sourcing and Tantivy-backed free-text search, per the approved Phase 1 plan. - CLAUDE.md: materialized on disk (never existed as a file before) with a new Phase 1 "done looks like" section. - agent: Windows Event Log (EvtSubscribe) and ETW sources, Windows service wrapper (install/uninstall/run-service), both feature- and target_os-gated so Linux builds/tests/clippy stay unaffected. Also fixed two pre-existing Phase 0 clippy gaps (dead-code on default-features-only builds, a type-inference edge case) found while testing every feature combination properly for the first time. UNVERIFIED on real Windows -- no Windows toolchain existed anywhere in the build environment; flagged prominently in three places. - proto/ingest: new record_id field, assigned once server-side in ingest's gRPC front end so ClickHouse and Tantivy agree on the same ID for the same record. - storage: record_id column + bloom filter index, verified against a live ClickHouse. - search: new service, Tantivy index, rskafka consumer as an independent second consumer group on the same Redpanda topic ingest already reads. - api/web: new /search endpoint and page, sharing the query page's result-table shape and component. - hack/windows-fixture: sends realistic Windows-shaped data straight to ingest, so the pipeline's handling of it is verifiable without a Windows host. Verified end-to-end on the live docker-compose stack: the same record_id comes back from both /query and /search for the same log line, including for windows-fixture's synthetic Windows Event Log data. Real bugs found and fixed along the way: api/Dockerfile missing proto/ in its build context, search's logs being completely silent (RUST_LOG gap), and search/target/ missing from .gitignore/.dockerignore.
This commit is contained in:
+15
-3
@@ -4,9 +4,17 @@ Go service sitting between the Rust agent and ClickHouse. Two halves in one
|
||||
binary, selected with `--mode`:
|
||||
|
||||
- **server** — mTLS gRPC front end (`LogIngest.PushBatch`) that agents
|
||||
connect to. Forwards each record, proto-encoded and unchanged, onto
|
||||
Redpanda. Does no normalization — kept thin so agent-facing latency isn't
|
||||
coupled to ClickHouse write performance.
|
||||
connect to. Assigns each record a server-side `record_id` (a UUID,
|
||||
overwriting whatever the agent sent — agents always send it empty) and
|
||||
otherwise forwards records proto-encoded onto Redpanda unchanged. Still
|
||||
kept thin — one field assignment, no real normalization — so agent-
|
||||
facing latency isn't coupled to ClickHouse write performance.
|
||||
`record_id` has to be assigned exactly once, here, rather than
|
||||
independently by each downstream consumer: Phase 1's Tantivy indexer
|
||||
and the ClickHouse writer both read the same Redpanda messages and need
|
||||
to agree on the same ID for the same record to join search hits back to
|
||||
rows — two consumers generating their own IDs would produce mismatched
|
||||
ones for what's supposed to be the same record.
|
||||
- **consumer** — reads back off Redpanda, normalizes into the ClickHouse row
|
||||
shape (`internal/normalize`), and batch-writes via the native protocol
|
||||
driver. Commits Redpanda offsets only after a successful ClickHouse
|
||||
@@ -35,6 +43,10 @@ egress an agent has. See `/docs/architecture.md`.
|
||||
protocol, pure Go (no cgo).
|
||||
- **golang.org/x/sync/errgroup** — used in `cmd/ingest/main.go` to run the
|
||||
server and consumer halves concurrently and propagate the first error.
|
||||
- **github.com/google/uuid** — was already in the dependency graph
|
||||
transitively (via clickhouse-go); promoted to a direct dependency for
|
||||
`record_id` generation in `internal/grpcserver`, so not a new addition
|
||||
to the transitive tree.
|
||||
|
||||
## Configuration
|
||||
|
||||
|
||||
+1
-1
@@ -6,6 +6,7 @@ replace github.com/sentry/sentry/proto => ../proto
|
||||
|
||||
require (
|
||||
github.com/ClickHouse/clickhouse-go/v2 v2.48.0
|
||||
github.com/google/uuid v1.6.0
|
||||
github.com/segmentio/kafka-go v0.4.51
|
||||
github.com/sentry/sentry/proto v0.0.0-00010101000000-000000000000
|
||||
golang.org/x/sync v0.22.0
|
||||
@@ -19,7 +20,6 @@ require (
|
||||
github.com/cespare/xxhash/v2 v2.3.0 // indirect
|
||||
github.com/go-faster/city v1.0.1 // indirect
|
||||
github.com/go-faster/errors v0.7.1 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/klauspost/compress v1.19.1 // indirect
|
||||
github.com/paulmach/orb v0.13.0 // indirect
|
||||
github.com/pierrec/lz4/v4 v4.1.27 // indirect
|
||||
|
||||
@@ -41,14 +41,14 @@ func (w *Writer) Close() error {
|
||||
}
|
||||
|
||||
func (w *Writer) WriteBatch(ctx context.Context, records []*logsv1.LogRecord) error {
|
||||
batch, err := w.conn.PrepareBatch(ctx, "INSERT INTO logs (timestamp, host, service, severity, message, attributes)")
|
||||
batch, err := w.conn.PrepareBatch(ctx, "INSERT INTO logs (timestamp, host, service, severity, message, attributes, record_id)")
|
||||
if err != nil {
|
||||
return fmt.Errorf("preparing batch: %w", err)
|
||||
}
|
||||
|
||||
for _, rec := range records {
|
||||
row := normalize.ToRow(rec)
|
||||
if err := batch.Append(row.Timestamp, row.Host, row.Service, row.Severity, row.Message, row.Attributes); err != nil {
|
||||
if err := batch.Append(row.Timestamp, row.Host, row.Service, row.Severity, row.Message, row.Attributes, row.RecordID); err != nil {
|
||||
return fmt.Errorf("appending row to batch: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
// Package grpcserver implements the agent-facing side of ingest: an mTLS
|
||||
// gRPC server accepting LogIngest.PushBatch calls, which it forwards
|
||||
// unchanged (proto-encoded) onto Redpanda. Normalization into the
|
||||
// ClickHouse row shape happens later, on the consumer side.
|
||||
// gRPC server accepting LogIngest.PushBatch calls. It assigns each record
|
||||
// a stable record_id (see the proto field comment for why this has to
|
||||
// happen exactly once, here, rather than in either downstream consumer)
|
||||
// and otherwise forwards records unchanged onto Redpanda — normalization
|
||||
// into the ClickHouse row shape happens later, on the consumer side.
|
||||
package grpcserver
|
||||
|
||||
import (
|
||||
@@ -10,6 +12,7 @@ import (
|
||||
"log/slog"
|
||||
"net"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/segmentio/kafka-go"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
@@ -76,6 +79,15 @@ func (s *Server) PushBatch(ctx context.Context, req *logsv1.PushBatchRequest) (*
|
||||
|
||||
msgs := make([]kafka.Message, 0, len(req.GetRecords()))
|
||||
for _, rec := range req.GetRecords() {
|
||||
// Assigned here, once, before this record is produced to
|
||||
// Redpanda: the ClickHouse-writer consumer and the Tantivy-
|
||||
// indexer consumer (Phase 1) both read the same Redpanda
|
||||
// messages and need to agree on the same ID for the same
|
||||
// record. Overwrites anything the agent sent (it always sends
|
||||
// empty, per the proto comment, but this is authoritative
|
||||
// regardless).
|
||||
rec.RecordId = uuid.NewString()
|
||||
|
||||
val, err := proto.Marshal(rec)
|
||||
if err != nil {
|
||||
return nil, status.Errorf(codes.InvalidArgument, "marshaling record: %v", err)
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
package grpcserver
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/segmentio/kafka-go"
|
||||
"google.golang.org/protobuf/proto"
|
||||
|
||||
"github.com/sentry/sentry/ingest/internal/config"
|
||||
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
|
||||
)
|
||||
|
||||
type fakeProducer struct {
|
||||
mu sync.Mutex
|
||||
written [][]kafka.Message
|
||||
err error
|
||||
}
|
||||
|
||||
func (f *fakeProducer) WriteBatch(_ context.Context, msgs []kafka.Message) error {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.err != nil {
|
||||
return f.err
|
||||
}
|
||||
batch := make([]kafka.Message, len(msgs))
|
||||
copy(batch, msgs)
|
||||
f.written = append(f.written, batch)
|
||||
return nil
|
||||
}
|
||||
|
||||
func newTestServer(p batchProducer) *Server {
|
||||
return New(slog.New(slog.NewTextHandler(io.Discard, nil)), config.GRPCConfig{}, config.TLSConfig{}, p)
|
||||
}
|
||||
|
||||
func TestPushBatchAssignsRecordID(t *testing.T) {
|
||||
fp := &fakeProducer{}
|
||||
s := newTestServer(fp)
|
||||
|
||||
req := &logsv1.PushBatchRequest{
|
||||
BatchId: "b1",
|
||||
Records: []*logsv1.LogRecord{
|
||||
{Host: "h1", Message: "one"},
|
||||
{Host: "h1", Message: "two"},
|
||||
},
|
||||
}
|
||||
|
||||
resp, err := s.PushBatch(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatalf("PushBatch() error = %v", err)
|
||||
}
|
||||
if resp.GetAccepted() != 2 {
|
||||
t.Fatalf("Accepted = %d, want 2", resp.GetAccepted())
|
||||
}
|
||||
|
||||
fp.mu.Lock()
|
||||
defer fp.mu.Unlock()
|
||||
if len(fp.written) != 1 || len(fp.written[0]) != 2 {
|
||||
t.Fatalf("unexpected written batches: %+v", fp.written)
|
||||
}
|
||||
|
||||
seen := make(map[string]bool)
|
||||
for _, m := range fp.written[0] {
|
||||
var rec logsv1.LogRecord
|
||||
if err := proto.Unmarshal(m.Value, &rec); err != nil {
|
||||
t.Fatalf("unmarshaling produced message: %v", err)
|
||||
}
|
||||
if rec.GetRecordId() == "" {
|
||||
t.Fatalf("record_id was not assigned for message %q", rec.GetMessage())
|
||||
}
|
||||
if seen[rec.GetRecordId()] {
|
||||
t.Fatalf("duplicate record_id %q across records in the same batch", rec.GetRecordId())
|
||||
}
|
||||
seen[rec.GetRecordId()] = true
|
||||
}
|
||||
}
|
||||
|
||||
func TestPushBatchOverwritesAgentSuppliedRecordID(t *testing.T) {
|
||||
fp := &fakeProducer{}
|
||||
s := newTestServer(fp)
|
||||
|
||||
req := &logsv1.PushBatchRequest{
|
||||
Records: []*logsv1.LogRecord{
|
||||
{Host: "h1", Message: "one", RecordId: "agent-supplied-should-be-ignored"},
|
||||
},
|
||||
}
|
||||
|
||||
if _, err := s.PushBatch(context.Background(), req); err != nil {
|
||||
t.Fatalf("PushBatch() error = %v", err)
|
||||
}
|
||||
|
||||
fp.mu.Lock()
|
||||
defer fp.mu.Unlock()
|
||||
var rec logsv1.LogRecord
|
||||
if err := proto.Unmarshal(fp.written[0][0].Value, &rec); err != nil {
|
||||
t.Fatalf("unmarshaling produced message: %v", err)
|
||||
}
|
||||
if rec.GetRecordId() == "agent-supplied-should-be-ignored" {
|
||||
t.Fatal("expected ingest to overwrite any agent-supplied record_id")
|
||||
}
|
||||
if rec.GetRecordId() == "" {
|
||||
t.Fatal("expected a server-assigned record_id")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPushBatchEmptyRecordsIsANoOp(t *testing.T) {
|
||||
fp := &fakeProducer{}
|
||||
s := newTestServer(fp)
|
||||
|
||||
resp, err := s.PushBatch(context.Background(), &logsv1.PushBatchRequest{})
|
||||
if err != nil {
|
||||
t.Fatalf("PushBatch() error = %v", err)
|
||||
}
|
||||
if resp.GetAccepted() != 0 {
|
||||
t.Fatalf("Accepted = %d, want 0", resp.GetAccepted())
|
||||
}
|
||||
|
||||
fp.mu.Lock()
|
||||
defer fp.mu.Unlock()
|
||||
if len(fp.written) != 0 {
|
||||
t.Fatalf("expected no batches written for an empty request, got %d", len(fp.written))
|
||||
}
|
||||
}
|
||||
@@ -9,6 +9,8 @@ package normalize
|
||||
import (
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
|
||||
)
|
||||
|
||||
@@ -19,6 +21,7 @@ type Row struct {
|
||||
Severity string
|
||||
Message string
|
||||
Attributes map[string]string
|
||||
RecordID uuid.UUID
|
||||
}
|
||||
|
||||
func ToRow(rec *logsv1.LogRecord) Row {
|
||||
@@ -26,6 +29,12 @@ func ToRow(rec *logsv1.LogRecord) Row {
|
||||
if attrs == nil {
|
||||
attrs = map[string]string{}
|
||||
}
|
||||
// grpcserver's PushBatch handler always assigns a valid UUID before a
|
||||
// record reaches this point (see its doc comment for why), so a parse
|
||||
// failure here would mean something upstream is bypassing that —
|
||||
// fall back to the nil UUID rather than failing the whole row, same
|
||||
// "never silently drop a record" spirit as the rest of this pipeline.
|
||||
recordID, _ := uuid.Parse(rec.GetRecordId())
|
||||
return Row{
|
||||
Timestamp: time.Unix(0, rec.GetTimestampUnixNano()).UTC(),
|
||||
Host: rec.GetHost(),
|
||||
@@ -33,6 +42,7 @@ func ToRow(rec *logsv1.LogRecord) Row {
|
||||
Severity: severityText(rec.GetSeverity()),
|
||||
Message: rec.GetMessage(),
|
||||
Attributes: attrs,
|
||||
RecordID: recordID,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,10 +4,13 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
|
||||
)
|
||||
|
||||
func TestToRowMapsFieldsAndSeverity(t *testing.T) {
|
||||
id := uuid.New()
|
||||
rec := &logsv1.LogRecord{
|
||||
TimestampUnixNano: time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC).UnixNano(),
|
||||
Host: "host-1",
|
||||
@@ -15,6 +18,7 @@ func TestToRowMapsFieldsAndSeverity(t *testing.T) {
|
||||
Severity: logsv1.Severity_SEVERITY_ERROR,
|
||||
Message: "boom",
|
||||
Attributes: map[string]string{"k": "v"},
|
||||
RecordId: id.String(),
|
||||
}
|
||||
|
||||
row := ToRow(rec)
|
||||
@@ -31,6 +35,17 @@ func TestToRowMapsFieldsAndSeverity(t *testing.T) {
|
||||
if !row.Timestamp.Equal(time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)) {
|
||||
t.Fatalf("unexpected timestamp: %v", row.Timestamp)
|
||||
}
|
||||
if row.RecordID != id {
|
||||
t.Fatalf("RecordID = %v, want %v", row.RecordID, id)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToRowInvalidRecordIDFallsBackToNilUUID(t *testing.T) {
|
||||
rec := &logsv1.LogRecord{Host: "h", Service: "s", Message: "m", RecordId: "not-a-uuid"}
|
||||
row := ToRow(rec)
|
||||
if row.RecordID != uuid.Nil {
|
||||
t.Fatalf("expected nil UUID fallback for an invalid record_id, got %v", row.RecordID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToRowNilAttributesBecomesEmptyMap(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user