Scope log retention deletion to selected hosts, not the whole table

api/logretention no longer deletes wholesale by age alone: a new
GET /logs/retention/hosts lists every host with matching records (plus
any configured retention floor), and preview/delete now require an
explicit, non-empty host list -- there is no "omitted host means every
host" shortcut server-side. Store's count/delete statements are
host-scoped (host IN (...)); Handler.partitionHosts checks the floor
per host instead of one global max, so a floor on one host never
blocks acting on other hosts requested in the same call. A request
that ends up fully or partially blocked still returns 200 with
blocked_hosts explaining why, rather than rejecting the whole call.

Settings' Log retention section is a host picker now: checkboxes with
per-host counts and a "protected Nd" badge where a floor applies,
"select all/none", and a confirm panel that names exactly which hosts
will be affected and which were skipped and why.

Verified live against real ClickHouse/Postgres and in-browser: three
hosts seeded, one protected by a 90-day floor -- a scoped delete
correctly removed the two open hosts' records, left the protected
host's untouched, and the response/UI both named it as skipped. Also
fixed a real spacing bug in the result message caught during that
browser pass (an adjacent {expr}{#if} with no source whitespace
between them rendered with no space either).
This commit is contained in:
2026-08-21 15:32:05 -07:00
parent a20bb5d1c7
commit 087c52a64f
6 changed files with 742 additions and 199 deletions
+77 -22
View File
@@ -1,8 +1,12 @@
// Package logretention lets an owner or admin permanently delete log
// records older than a chosen age -- storage/README.md has flagged "no
// TTL/retention clause yet" since Phase 0; this is the on-demand,
// operator-triggered half of that gap (not an automatic TTL, which is
// a different, engine-driven design nobody asked for here).
// records older than a chosen age, scoped to specific hosts -- deleting
// by age alone (with no way to target which agents' logs) turned out
// to be a real footgun for an operator who only wants to clean up one
// noisy host, not everything; storage/README.md has flagged "no
// TTL/retention clause yet" since Phase 0, this is the on-demand,
// operator-triggered, host-scoped half of that gap (not an automatic
// TTL, which is a different, engine-driven design nobody asked for
// here).
//
// Deliberately scoped to core's single ClickHouse `logs` table, not
// enterprise/'s per-tenant ClickHouse routing
@@ -25,6 +29,8 @@ package logretention
import (
"context"
"fmt"
"strings"
"time"
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
@@ -33,10 +39,10 @@ import (
// Store issues purpose-built, parameterized statements against the
// `logs` table -- deliberately not querylang/executor.ChRunner, whose
// one method (RunSQL) is scoped to arbitrary SELECT statements for the
// query language compiler. This package only ever needs two fixed
// statements (a count and a delete), so keeping them separate avoids
// stretching ChRunner's SELECT-shaped contract to also cover a DML
// mutation.
// query language compiler. This package only ever needs a handful of
// fixed statement shapes (list hosts, count, delete), so keeping them
// separate avoids stretching ChRunner's SELECT-shaped contract to also
// cover a DML mutation.
type Store struct {
conn driver.Conn
}
@@ -45,11 +51,59 @@ func NewStore(conn driver.Conn) *Store {
return &Store{conn: conn}
}
// CountOlderThan reports how many log records are older than cutoff --
// backs the "this will delete N records" preview a caller shows before
// asking for confirmation.
func (s *Store) CountOlderThan(ctx context.Context, cutoff time.Time) (uint64, error) {
row := s.conn.QueryRow(ctx, "SELECT count() FROM logs WHERE timestamp < ?", cutoff)
type HostCount struct {
Host string `json:"host"`
Count uint64 `json:"count"`
}
// HostsOlderThan lists every host with at least one log record older
// than cutoff, along with how many -- backs the host picker a caller
// selects from before previewing/deleting, so the list only ever shows
// hosts that actually have something to act on for the chosen age.
func (s *Store) HostsOlderThan(ctx context.Context, cutoff time.Time) ([]HostCount, error) {
rows, err := s.conn.Query(ctx, `
SELECT host, count() AS n FROM logs WHERE timestamp < ? GROUP BY host ORDER BY n DESC`, cutoff)
if err != nil {
return nil, err
}
defer rows.Close()
var out []HostCount
for rows.Next() {
var hc HostCount
if err := rows.Scan(&hc.Host, &hc.Count); err != nil {
return nil, err
}
out = append(out, hc)
}
return out, rows.Err()
}
// hostPlaceholders builds "?, ?, ..." for n hosts and the matching
// []any argument slice (cutoff first, then each host) -- shared by
// CountOlderThan and DeleteOlderThan since both statements have the
// same "timestamp < ? AND host IN (...)" shape. Callers must never
// pass an empty hosts slice (an empty IN () is invalid SQL, and more
// importantly "no hosts specified" must never silently mean "every
// host" -- see Handler.parseHosts, which rejects that before this is
// ever called).
func hostPlaceholders(cutoff time.Time, hosts []string) (string, []any) {
placeholders := make([]string, len(hosts))
args := make([]any, 0, len(hosts)+1)
args = append(args, cutoff)
for i, h := range hosts {
placeholders[i] = "?"
args = append(args, h)
}
return strings.Join(placeholders, ", "), args
}
// CountOlderThan reports how many log records from any of hosts are
// older than cutoff -- backs the "this will delete N records" preview
// a caller shows before asking for confirmation.
func (s *Store) CountOlderThan(ctx context.Context, cutoff time.Time, hosts []string) (uint64, error) {
ph, args := hostPlaceholders(cutoff, hosts)
row := s.conn.QueryRow(ctx, fmt.Sprintf("SELECT count() FROM logs WHERE timestamp < ? AND host IN (%s)", ph), args...)
var n uint64
if err := row.Scan(&n); err != nil {
return 0, err
@@ -58,13 +112,14 @@ func (s *Store) CountOlderThan(ctx context.Context, cutoff time.Time) (uint64, e
}
// DeleteOlderThan issues a synchronous ClickHouse mutation
// (SETTINGS mutations_sync = 1) deleting every log record older than
// cutoff. Synchronous rather than fire-and-forget so a 200 response
// means the data is actually gone, not just queued -- an owner/admin
// confirming a permanent delete should be able to trust the response.
// This does block for as long as the mutation takes, which could be a
// while against a very large table; a disclosed tradeoff for this
// deployment's homelab/small-scale target, not a hidden one.
func (s *Store) DeleteOlderThan(ctx context.Context, cutoff time.Time) error {
return s.conn.Exec(ctx, "ALTER TABLE logs DELETE WHERE timestamp < ? SETTINGS mutations_sync = 1", cutoff)
// (SETTINGS mutations_sync = 1) deleting every log record from any of
// hosts older than cutoff. Synchronous rather than fire-and-forget so
// a 200 response means the data is actually gone, not just queued -- an
// owner/admin confirming a permanent delete should be able to trust the
// response. This does block for as long as the mutation takes, which
// could be a while against a very large table; a disclosed tradeoff for
// this deployment's homelab/small-scale target, not a hidden one.
func (s *Store) DeleteOlderThan(ctx context.Context, cutoff time.Time, hosts []string) error {
ph, args := hostPlaceholders(cutoff, hosts)
return s.conn.Exec(ctx, fmt.Sprintf("ALTER TABLE logs DELETE WHERE timestamp < ? AND host IN (%s) SETTINGS mutations_sync = 1", ph), args...)
}