Extends the agent, ingest, storage, api, and web with Windows Event Log/ETW sourcing and Tantivy-backed free-text search, per the approved Phase 1 plan. - CLAUDE.md: materialized on disk (never existed as a file before) with a new Phase 1 "done looks like" section. - agent: Windows Event Log (EvtSubscribe) and ETW sources, Windows service wrapper (install/uninstall/run-service), both feature- and target_os-gated so Linux builds/tests/clippy stay unaffected. Also fixed two pre-existing Phase 0 clippy gaps (dead-code on default-features-only builds, a type-inference edge case) found while testing every feature combination properly for the first time. UNVERIFIED on real Windows -- no Windows toolchain existed anywhere in the build environment; flagged prominently in three places. - proto/ingest: new record_id field, assigned once server-side in ingest's gRPC front end so ClickHouse and Tantivy agree on the same ID for the same record. - storage: record_id column + bloom filter index, verified against a live ClickHouse. - search: new service, Tantivy index, rskafka consumer as an independent second consumer group on the same Redpanda topic ingest already reads. - api/web: new /search endpoint and page, sharing the query page's result-table shape and component. - hack/windows-fixture: sends realistic Windows-shaped data straight to ingest, so the pipeline's handling of it is verifiable without a Windows host. Verified end-to-end on the live docker-compose stack: the same record_id comes back from both /query and /search for the same log line, including for windows-fixture's synthetic Windows Event Log data. Real bugs found and fixed along the way: api/Dockerfile missing proto/ in its build context, search's logs being completely silent (RUST_LOG gap), and search/target/ missing from .gitignore/.dockerignore.
109 lines
3.5 KiB
Go
109 lines
3.5 KiB
Go
// Package grpcserver implements the agent-facing side of ingest: an mTLS
|
|
// gRPC server accepting LogIngest.PushBatch calls. It assigns each record
|
|
// a stable record_id (see the proto field comment for why this has to
|
|
// happen exactly once, here, rather than in either downstream consumer)
|
|
// and otherwise forwards records unchanged onto Redpanda — normalization
|
|
// into the ClickHouse row shape happens later, on the consumer side.
|
|
package grpcserver
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"log/slog"
|
|
"net"
|
|
|
|
"github.com/google/uuid"
|
|
"github.com/segmentio/kafka-go"
|
|
"google.golang.org/grpc"
|
|
"google.golang.org/grpc/codes"
|
|
"google.golang.org/grpc/credentials"
|
|
"google.golang.org/grpc/status"
|
|
"google.golang.org/protobuf/proto"
|
|
|
|
"github.com/sentry/sentry/ingest/internal/config"
|
|
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
|
|
)
|
|
|
|
type Server struct {
|
|
logsv1.UnimplementedLogIngestServer
|
|
|
|
logger *slog.Logger
|
|
grpcCfg config.GRPCConfig
|
|
tlsCfg config.TLSConfig
|
|
producer batchProducer
|
|
}
|
|
|
|
// batchProducer is the subset of *producer.Producer this package depends
|
|
// on, so tests can substitute a fake without touching Redpanda.
|
|
type batchProducer interface {
|
|
WriteBatch(ctx context.Context, msgs []kafka.Message) error
|
|
}
|
|
|
|
func New(logger *slog.Logger, grpcCfg config.GRPCConfig, tlsCfg config.TLSConfig, p batchProducer) *Server {
|
|
return &Server{logger: logger, grpcCfg: grpcCfg, tlsCfg: tlsCfg, producer: p}
|
|
}
|
|
|
|
// Run blocks serving gRPC until ctx is canceled, then gracefully stops.
|
|
func (s *Server) Run(ctx context.Context) error {
|
|
tlsConf, err := loadServerTLSConfig(s.tlsCfg)
|
|
if err != nil {
|
|
return fmt.Errorf("loading TLS config: %w", err)
|
|
}
|
|
|
|
lis, err := net.Listen("tcp", s.grpcCfg.ListenAddr)
|
|
if err != nil {
|
|
return fmt.Errorf("listening on %s: %w", s.grpcCfg.ListenAddr, err)
|
|
}
|
|
|
|
grpcSrv := grpc.NewServer(grpc.Creds(credentials.NewTLS(tlsConf)))
|
|
logsv1.RegisterLogIngestServer(grpcSrv, s)
|
|
|
|
s.logger.Info("gRPC server listening", "addr", s.grpcCfg.ListenAddr)
|
|
|
|
errCh := make(chan error, 1)
|
|
go func() { errCh <- grpcSrv.Serve(lis) }()
|
|
|
|
select {
|
|
case <-ctx.Done():
|
|
grpcSrv.GracefulStop()
|
|
return nil
|
|
case err := <-errCh:
|
|
return err
|
|
}
|
|
}
|
|
|
|
func (s *Server) PushBatch(ctx context.Context, req *logsv1.PushBatchRequest) (*logsv1.PushBatchResponse, error) {
|
|
if len(req.GetRecords()) == 0 {
|
|
return &logsv1.PushBatchResponse{Accepted: 0}, nil
|
|
}
|
|
|
|
msgs := make([]kafka.Message, 0, len(req.GetRecords()))
|
|
for _, rec := range req.GetRecords() {
|
|
// Assigned here, once, before this record is produced to
|
|
// Redpanda: the ClickHouse-writer consumer and the Tantivy-
|
|
// indexer consumer (Phase 1) both read the same Redpanda
|
|
// messages and need to agree on the same ID for the same
|
|
// record. Overwrites anything the agent sent (it always sends
|
|
// empty, per the proto comment, but this is authoritative
|
|
// regardless).
|
|
rec.RecordId = uuid.NewString()
|
|
|
|
val, err := proto.Marshal(rec)
|
|
if err != nil {
|
|
return nil, status.Errorf(codes.InvalidArgument, "marshaling record: %v", err)
|
|
}
|
|
msgs = append(msgs, kafka.Message{
|
|
Key: []byte(rec.GetHost()),
|
|
Value: val,
|
|
})
|
|
}
|
|
|
|
if err := s.producer.WriteBatch(ctx, msgs); err != nil {
|
|
s.logger.Error("failed to write batch to redpanda", "batch_id", req.GetBatchId(), "error", err)
|
|
return nil, status.Errorf(codes.Unavailable, "writing to transport: %v", err)
|
|
}
|
|
|
|
s.logger.Debug("batch produced to redpanda", "batch_id", req.GetBatchId(), "records", len(req.GetRecords()))
|
|
return &logsv1.PushBatchResponse{Accepted: uint32(len(req.GetRecords()))}, nil
|
|
}
|