Phase 1: Windows log collection + full-text search

Extends the agent, ingest, storage, api, and web with Windows Event
Log/ETW sourcing and Tantivy-backed free-text search, per the approved
Phase 1 plan.

- CLAUDE.md: materialized on disk (never existed as a file before) with
  a new Phase 1 "done looks like" section.
- agent: Windows Event Log (EvtSubscribe) and ETW sources, Windows
  service wrapper (install/uninstall/run-service), both feature- and
  target_os-gated so Linux builds/tests/clippy stay unaffected. Also
  fixed two pre-existing Phase 0 clippy gaps (dead-code on
  default-features-only builds, a type-inference edge case) found while
  testing every feature combination properly for the first time.
  UNVERIFIED on real Windows -- no Windows toolchain existed anywhere in
  the build environment; flagged prominently in three places.
- proto/ingest: new record_id field, assigned once server-side in
  ingest's gRPC front end so ClickHouse and Tantivy agree on the same ID
  for the same record.
- storage: record_id column + bloom filter index, verified against a
  live ClickHouse.
- search: new service, Tantivy index, rskafka consumer as an independent
  second consumer group on the same Redpanda topic ingest already reads.
- api/web: new /search endpoint and page, sharing the query page's
  result-table shape and component.
- hack/windows-fixture: sends realistic Windows-shaped data straight to
  ingest, so the pipeline's handling of it is verifiable without a
  Windows host.

Verified end-to-end on the live docker-compose stack: the same record_id
comes back from both /query and /search for the same log line, including
for windows-fixture's synthetic Windows Event Log data. Real bugs found
and fixed along the way: api/Dockerfile missing proto/ in its build
context, search's logs being completely silent (RUST_LOG gap), and
search/target/ missing from .gitignore/.dockerignore.
This commit is contained in:
2026-08-13 11:27:35 -07:00
parent fe854b1091
commit cd8aa290ca
66 changed files with 6084 additions and 171 deletions
+2 -2
View File
@@ -41,14 +41,14 @@ func (w *Writer) Close() error {
}
func (w *Writer) WriteBatch(ctx context.Context, records []*logsv1.LogRecord) error {
batch, err := w.conn.PrepareBatch(ctx, "INSERT INTO logs (timestamp, host, service, severity, message, attributes)")
batch, err := w.conn.PrepareBatch(ctx, "INSERT INTO logs (timestamp, host, service, severity, message, attributes, record_id)")
if err != nil {
return fmt.Errorf("preparing batch: %w", err)
}
for _, rec := range records {
row := normalize.ToRow(rec)
if err := batch.Append(row.Timestamp, row.Host, row.Service, row.Severity, row.Message, row.Attributes); err != nil {
if err := batch.Append(row.Timestamp, row.Host, row.Service, row.Severity, row.Message, row.Attributes, row.RecordID); err != nil {
return fmt.Errorf("appending row to batch: %w", err)
}
}
+15 -3
View File
@@ -1,7 +1,9 @@
// Package grpcserver implements the agent-facing side of ingest: an mTLS
// gRPC server accepting LogIngest.PushBatch calls, which it forwards
// unchanged (proto-encoded) onto Redpanda. Normalization into the
// ClickHouse row shape happens later, on the consumer side.
// gRPC server accepting LogIngest.PushBatch calls. It assigns each record
// a stable record_id (see the proto field comment for why this has to
// happen exactly once, here, rather than in either downstream consumer)
// and otherwise forwards records unchanged onto Redpanda — normalization
// into the ClickHouse row shape happens later, on the consumer side.
package grpcserver
import (
@@ -10,6 +12,7 @@ import (
"log/slog"
"net"
"github.com/google/uuid"
"github.com/segmentio/kafka-go"
"google.golang.org/grpc"
"google.golang.org/grpc/codes"
@@ -76,6 +79,15 @@ func (s *Server) PushBatch(ctx context.Context, req *logsv1.PushBatchRequest) (*
msgs := make([]kafka.Message, 0, len(req.GetRecords()))
for _, rec := range req.GetRecords() {
// Assigned here, once, before this record is produced to
// Redpanda: the ClickHouse-writer consumer and the Tantivy-
// indexer consumer (Phase 1) both read the same Redpanda
// messages and need to agree on the same ID for the same
// record. Overwrites anything the agent sent (it always sends
// empty, per the proto comment, but this is authoritative
// regardless).
rec.RecordId = uuid.NewString()
val, err := proto.Marshal(rec)
if err != nil {
return nil, status.Errorf(codes.InvalidArgument, "marshaling record: %v", err)
+126
View File
@@ -0,0 +1,126 @@
package grpcserver
import (
"context"
"io"
"log/slog"
"sync"
"testing"
"github.com/segmentio/kafka-go"
"google.golang.org/protobuf/proto"
"github.com/sentry/sentry/ingest/internal/config"
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
)
type fakeProducer struct {
mu sync.Mutex
written [][]kafka.Message
err error
}
func (f *fakeProducer) WriteBatch(_ context.Context, msgs []kafka.Message) error {
f.mu.Lock()
defer f.mu.Unlock()
if f.err != nil {
return f.err
}
batch := make([]kafka.Message, len(msgs))
copy(batch, msgs)
f.written = append(f.written, batch)
return nil
}
func newTestServer(p batchProducer) *Server {
return New(slog.New(slog.NewTextHandler(io.Discard, nil)), config.GRPCConfig{}, config.TLSConfig{}, p)
}
func TestPushBatchAssignsRecordID(t *testing.T) {
fp := &fakeProducer{}
s := newTestServer(fp)
req := &logsv1.PushBatchRequest{
BatchId: "b1",
Records: []*logsv1.LogRecord{
{Host: "h1", Message: "one"},
{Host: "h1", Message: "two"},
},
}
resp, err := s.PushBatch(context.Background(), req)
if err != nil {
t.Fatalf("PushBatch() error = %v", err)
}
if resp.GetAccepted() != 2 {
t.Fatalf("Accepted = %d, want 2", resp.GetAccepted())
}
fp.mu.Lock()
defer fp.mu.Unlock()
if len(fp.written) != 1 || len(fp.written[0]) != 2 {
t.Fatalf("unexpected written batches: %+v", fp.written)
}
seen := make(map[string]bool)
for _, m := range fp.written[0] {
var rec logsv1.LogRecord
if err := proto.Unmarshal(m.Value, &rec); err != nil {
t.Fatalf("unmarshaling produced message: %v", err)
}
if rec.GetRecordId() == "" {
t.Fatalf("record_id was not assigned for message %q", rec.GetMessage())
}
if seen[rec.GetRecordId()] {
t.Fatalf("duplicate record_id %q across records in the same batch", rec.GetRecordId())
}
seen[rec.GetRecordId()] = true
}
}
func TestPushBatchOverwritesAgentSuppliedRecordID(t *testing.T) {
fp := &fakeProducer{}
s := newTestServer(fp)
req := &logsv1.PushBatchRequest{
Records: []*logsv1.LogRecord{
{Host: "h1", Message: "one", RecordId: "agent-supplied-should-be-ignored"},
},
}
if _, err := s.PushBatch(context.Background(), req); err != nil {
t.Fatalf("PushBatch() error = %v", err)
}
fp.mu.Lock()
defer fp.mu.Unlock()
var rec logsv1.LogRecord
if err := proto.Unmarshal(fp.written[0][0].Value, &rec); err != nil {
t.Fatalf("unmarshaling produced message: %v", err)
}
if rec.GetRecordId() == "agent-supplied-should-be-ignored" {
t.Fatal("expected ingest to overwrite any agent-supplied record_id")
}
if rec.GetRecordId() == "" {
t.Fatal("expected a server-assigned record_id")
}
}
func TestPushBatchEmptyRecordsIsANoOp(t *testing.T) {
fp := &fakeProducer{}
s := newTestServer(fp)
resp, err := s.PushBatch(context.Background(), &logsv1.PushBatchRequest{})
if err != nil {
t.Fatalf("PushBatch() error = %v", err)
}
if resp.GetAccepted() != 0 {
t.Fatalf("Accepted = %d, want 0", resp.GetAccepted())
}
fp.mu.Lock()
defer fp.mu.Unlock()
if len(fp.written) != 0 {
t.Fatalf("expected no batches written for an empty request, got %d", len(fp.written))
}
}
+10
View File
@@ -9,6 +9,8 @@ package normalize
import (
"time"
"github.com/google/uuid"
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
)
@@ -19,6 +21,7 @@ type Row struct {
Severity string
Message string
Attributes map[string]string
RecordID uuid.UUID
}
func ToRow(rec *logsv1.LogRecord) Row {
@@ -26,6 +29,12 @@ func ToRow(rec *logsv1.LogRecord) Row {
if attrs == nil {
attrs = map[string]string{}
}
// grpcserver's PushBatch handler always assigns a valid UUID before a
// record reaches this point (see its doc comment for why), so a parse
// failure here would mean something upstream is bypassing that —
// fall back to the nil UUID rather than failing the whole row, same
// "never silently drop a record" spirit as the rest of this pipeline.
recordID, _ := uuid.Parse(rec.GetRecordId())
return Row{
Timestamp: time.Unix(0, rec.GetTimestampUnixNano()).UTC(),
Host: rec.GetHost(),
@@ -33,6 +42,7 @@ func ToRow(rec *logsv1.LogRecord) Row {
Severity: severityText(rec.GetSeverity()),
Message: rec.GetMessage(),
Attributes: attrs,
RecordID: recordID,
}
}
@@ -4,10 +4,13 @@ import (
"testing"
"time"
"github.com/google/uuid"
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
)
func TestToRowMapsFieldsAndSeverity(t *testing.T) {
id := uuid.New()
rec := &logsv1.LogRecord{
TimestampUnixNano: time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC).UnixNano(),
Host: "host-1",
@@ -15,6 +18,7 @@ func TestToRowMapsFieldsAndSeverity(t *testing.T) {
Severity: logsv1.Severity_SEVERITY_ERROR,
Message: "boom",
Attributes: map[string]string{"k": "v"},
RecordId: id.String(),
}
row := ToRow(rec)
@@ -31,6 +35,17 @@ func TestToRowMapsFieldsAndSeverity(t *testing.T) {
if !row.Timestamp.Equal(time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC)) {
t.Fatalf("unexpected timestamp: %v", row.Timestamp)
}
if row.RecordID != id {
t.Fatalf("RecordID = %v, want %v", row.RecordID, id)
}
}
func TestToRowInvalidRecordIDFallsBackToNilUUID(t *testing.T) {
rec := &logsv1.LogRecord{Host: "h", Service: "s", Message: "m", RecordId: "not-a-uuid"}
row := ToRow(rec)
if row.RecordID != uuid.Nil {
t.Fatalf("expected nil UUID fallback for an invalid record_id, got %v", row.RecordID)
}
}
func TestToRowNilAttributesBecomesEmptyMap(t *testing.T) {