Phase 1: Windows log collection + full-text search

Extends the agent, ingest, storage, api, and web with Windows Event
Log/ETW sourcing and Tantivy-backed free-text search, per the approved
Phase 1 plan.

- CLAUDE.md: materialized on disk (never existed as a file before) with
  a new Phase 1 "done looks like" section.
- agent: Windows Event Log (EvtSubscribe) and ETW sources, Windows
  service wrapper (install/uninstall/run-service), both feature- and
  target_os-gated so Linux builds/tests/clippy stay unaffected. Also
  fixed two pre-existing Phase 0 clippy gaps (dead-code on
  default-features-only builds, a type-inference edge case) found while
  testing every feature combination properly for the first time.
  UNVERIFIED on real Windows -- no Windows toolchain existed anywhere in
  the build environment; flagged prominently in three places.
- proto/ingest: new record_id field, assigned once server-side in
  ingest's gRPC front end so ClickHouse and Tantivy agree on the same ID
  for the same record.
- storage: record_id column + bloom filter index, verified against a
  live ClickHouse.
- search: new service, Tantivy index, rskafka consumer as an independent
  second consumer group on the same Redpanda topic ingest already reads.
- api/web: new /search endpoint and page, sharing the query page's
  result-table shape and component.
- hack/windows-fixture: sends realistic Windows-shaped data straight to
  ingest, so the pipeline's handling of it is verifiable without a
  Windows host.

Verified end-to-end on the live docker-compose stack: the same record_id
comes back from both /query and /search for the same log line, including
for windows-fixture's synthetic Windows Event Log data. Real bugs found
and fixed along the way: api/Dockerfile missing proto/ in its build
context, search's logs being completely silent (RUST_LOG gap), and
search/target/ missing from .gitignore/.dockerignore.
This commit is contained in:
2026-08-13 11:27:35 -07:00
parent fe854b1091
commit cd8aa290ca
66 changed files with 6084 additions and 171 deletions
+15 -16
View File
@@ -1,13 +1,13 @@
// Package queryapi is the Phase 0 query API: a single crude POST /query
// endpoint that takes a raw SQL string, allowlists it to a single SELECT
// statement, and proxies it to ClickHouse. This is a deliberate
// simplification of the pinned "gRPC + REST gateway" control-plane
// pattern (see CLAUDE.md's tech stack table): a plain net/http REST
// handler, not a gRPC service transcoded through grpc-gateway. That
// machinery (proto definitions, googleapis annotations, gateway codegen)
// buys nothing for one crude placeholder endpoint that Phase 2 replaces
// outright with the real SPL-like query layer. Revisit gRPC+gateway when
// /api grows a second real endpoint.
// Package queryapi is Sentry's query API: POST /query (Phase 0, a crude
// raw-SQL passthrough allowlisted to SELECT) and POST /search (Phase 1,
// free-text search via the search service, joined back against
// ClickHouse). This is a deliberate simplification of the pinned "gRPC +
// REST gateway" control-plane pattern (see CLAUDE.md's tech stack table):
// plain net/http REST handlers, not a gRPC service transcoded through
// grpc-gateway. That machinery (proto definitions, googleapis
// annotations, gateway codegen) doesn't buy much for two crude endpoints
// that Phase 2's real SPL-like query layer replaces outright. Revisit
// gRPC+gateway once /api's endpoint count and lifespan justify it.
package queryapi
import (
@@ -28,17 +28,19 @@ type queryExecutor interface {
type Handler struct {
logger *slog.Logger
exec queryExecutor
search searchClient
queryTimeout time.Duration
allowedOrigin string
}
func NewHandler(logger *slog.Logger, exec queryExecutor, queryTimeout time.Duration, allowedOrigin string) *Handler {
return &Handler{logger: logger, exec: exec, queryTimeout: queryTimeout, allowedOrigin: allowedOrigin}
func NewHandler(logger *slog.Logger, exec queryExecutor, search searchClient, queryTimeout time.Duration, allowedOrigin string) *Handler {
return &Handler{logger: logger, exec: exec, search: search, queryTimeout: queryTimeout, allowedOrigin: allowedOrigin}
}
func (h *Handler) Routes() http.Handler {
mux := http.NewServeMux()
mux.HandleFunc("POST /query", h.handleQuery)
mux.HandleFunc("POST /search", h.handleSearch)
mux.HandleFunc("GET /healthz", h.handleHealthz)
return h.withCORS(mux)
}
@@ -99,10 +101,7 @@ func (h *Handler) handleQuery(w http.ResponseWriter, r *http.Request) {
return
}
w.Header().Set("Content-Type", "application/json")
if err := json.NewEncoder(w).Encode(result); err != nil {
h.logger.Error("encoding response", "error", err)
}
writeJSON(w, result)
}
func writeError(w http.ResponseWriter, status int, msg string) {
+17 -1
View File
@@ -27,8 +27,24 @@ func (f *fakeExecutor) Execute(_ context.Context, sql string) (*QueryResult, err
return f.result, nil
}
type fakeSearchClient struct {
recordIDs []string
err error
}
func (f *fakeSearchClient) Search(_ context.Context, _ string, _ uint32) ([]string, error) {
if f.err != nil {
return nil, f.err
}
return f.recordIDs, nil
}
func newTestHandler(exec queryExecutor) *Handler {
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), exec, time.Second, "*")
return newTestHandlerWithSearch(exec, &fakeSearchClient{})
}
func newTestHandlerWithSearch(exec queryExecutor, search searchClient) *Handler {
return NewHandler(slog.New(slog.NewTextHandler(io.Discard, nil)), exec, search, time.Second, "*")
}
func TestHandleQuerySuccess(t *testing.T) {
+96
View File
@@ -0,0 +1,96 @@
package queryapi
import (
"context"
"encoding/json"
"fmt"
"net/http"
"strings"
"github.com/google/uuid"
)
// searchClient is the narrow interface handleSearch depends on, so tests
// can substitute a fake without a real search service. A small gRPC
// adapter in cmd/api satisfies this.
type searchClient interface {
Search(ctx context.Context, query string, limit uint32) ([]string, error)
}
type searchRequest struct {
Query string `json:"query"`
Limit uint32 `json:"limit"`
}
func (h *Handler) handleSearch(w http.ResponseWriter, r *http.Request) {
r.Body = http.MaxBytesReader(w, r.Body, maxBodyBytes)
var req searchRequest
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
writeError(w, http.StatusBadRequest, "invalid JSON body: "+err.Error())
return
}
if strings.TrimSpace(req.Query) == "" {
writeError(w, http.StatusBadRequest, "query must not be empty")
return
}
ctx, cancel := context.WithTimeout(r.Context(), h.queryTimeout)
defer cancel()
recordIDs, err := h.search.Search(ctx, req.Query, req.Limit)
if err != nil {
h.logger.Error("search failed", "error", err)
writeError(w, http.StatusBadGateway, "search failed: "+err.Error())
return
}
if len(recordIDs) == 0 {
writeJSON(w, &QueryResult{Columns: []string{}, Rows: [][]any{}})
return
}
sql, err := recordIDsQuery(recordIDs)
if err != nil {
h.logger.Error("building record_id query", "error", err)
writeError(w, http.StatusBadGateway, "search returned unusable results")
return
}
result, err := h.exec.Execute(ctx, sql)
if err != nil {
h.logger.Error("joining search results against clickhouse failed", "error", err)
writeError(w, http.StatusBadGateway, "query failed: "+err.Error())
return
}
writeJSON(w, result)
}
// recordIDsQuery builds a SELECT ... WHERE record_id IN (...) against the
// IDs the search service returned. Every ID is validated as a real UUID
// before being embedded in the query string -- record_ids come from an
// internal, trusted service (not raw user input), but a UUID that fails
// to parse can't contain SQL-breaking characters either way, so this is
// defense in depth, not a response to a specific threat.
func recordIDsQuery(recordIDs []string) (string, error) {
quoted := make([]string, 0, len(recordIDs))
for _, id := range recordIDs {
if _, err := uuid.Parse(id); err != nil {
continue // skip anything not a valid UUID rather than failing the whole query
}
quoted = append(quoted, "'"+id+"'")
}
if len(quoted) == 0 {
return "", fmt.Errorf("no valid record_ids in search response")
}
return fmt.Sprintf(
"SELECT * FROM logs WHERE record_id IN (%s) ORDER BY timestamp DESC",
strings.Join(quoted, ","),
), nil
}
func writeJSON(w http.ResponseWriter, v any) {
w.Header().Set("Content-Type", "application/json")
_ = json.NewEncoder(w).Encode(v)
}
+125
View File
@@ -0,0 +1,125 @@
package queryapi
import (
"encoding/json"
"errors"
"net/http"
"net/http/httptest"
"strings"
"testing"
)
func TestHandleSearchSuccess(t *testing.T) {
id := "5754b062-ec8b-45b1-b1b8-a50f263adcd3"
fe := &fakeExecutor{result: &QueryResult{
Columns: []string{"message"},
Rows: [][]any{{"hello world"}},
}}
fs := &fakeSearchClient{recordIDs: []string{id}}
h := newTestHandlerWithSearch(fe, fs)
body := strings.NewReader(`{"query": "hello"}`)
req := httptest.NewRequest(http.MethodPost, "/search", body)
rec := httptest.NewRecorder()
h.Routes().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
}
if !strings.Contains(fe.gotSQL, id) {
t.Fatalf("expected the record_id in the generated SQL, got %q", fe.gotSQL)
}
if !strings.Contains(fe.gotSQL, "WHERE record_id IN") {
t.Fatalf("expected an IN clause, got %q", fe.gotSQL)
}
var got QueryResult
if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil {
t.Fatalf("decoding response: %v", err)
}
if len(got.Rows) != 1 {
t.Fatalf("unexpected result: %+v", got)
}
}
func TestHandleSearchRejectsEmptyQuery(t *testing.T) {
fe := &fakeExecutor{}
fs := &fakeSearchClient{}
h := newTestHandlerWithSearch(fe, fs)
body := strings.NewReader(`{"query": " "}`)
req := httptest.NewRequest(http.MethodPost, "/search", body)
rec := httptest.NewRecorder()
h.Routes().ServeHTTP(rec, req)
if rec.Code != http.StatusBadRequest {
t.Fatalf("status = %d, want 400", rec.Code)
}
if fe.gotSQL != "" {
t.Fatal("executor should not have been called for an empty query")
}
}
func TestHandleSearchNoResultsReturnsEmptyNotError(t *testing.T) {
fe := &fakeExecutor{}
fs := &fakeSearchClient{recordIDs: nil}
h := newTestHandlerWithSearch(fe, fs)
body := strings.NewReader(`{"query": "nothing matches this"}`)
req := httptest.NewRequest(http.MethodPost, "/search", body)
rec := httptest.NewRecorder()
h.Routes().ServeHTTP(rec, req)
if rec.Code != http.StatusOK {
t.Fatalf("status = %d, want 200; body=%s", rec.Code, rec.Body.String())
}
if fe.gotSQL != "" {
t.Fatal("executor should not have been called when search returns no IDs")
}
var got QueryResult
if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil {
t.Fatalf("decoding response: %v", err)
}
if len(got.Rows) != 0 {
t.Fatalf("expected empty rows, got %+v", got.Rows)
}
}
func TestHandleSearchServiceErrorReturnsBadGateway(t *testing.T) {
fe := &fakeExecutor{}
fs := &fakeSearchClient{err: errors.New("search service unreachable")}
h := newTestHandlerWithSearch(fe, fs)
body := strings.NewReader(`{"query": "hello"}`)
req := httptest.NewRequest(http.MethodPost, "/search", body)
rec := httptest.NewRecorder()
h.Routes().ServeHTTP(rec, req)
if rec.Code != http.StatusBadGateway {
t.Fatalf("status = %d, want 502", rec.Code)
}
}
func TestRecordIDsQuerySkipsInvalidUUIDs(t *testing.T) {
sql, err := recordIDsQuery([]string{"not-a-uuid", "5754b062-ec8b-45b1-b1b8-a50f263adcd3"})
if err != nil {
t.Fatalf("recordIDsQuery() error = %v", err)
}
if strings.Contains(sql, "not-a-uuid") {
t.Fatalf("expected the invalid UUID to be skipped, got %q", sql)
}
if !strings.Contains(sql, "5754b062-ec8b-45b1-b1b8-a50f263adcd3") {
t.Fatalf("expected the valid UUID to be included, got %q", sql)
}
}
func TestRecordIDsQueryAllInvalidReturnsError(t *testing.T) {
if _, err := recordIDsQuery([]string{"not-a-uuid", "also-not-one"}); err == nil {
t.Fatal("expected an error when no IDs are valid UUIDs")
}
}