Files
cairnobs/enterprise/internal/chwriter/chwriter_test.go
T
jcoffey-dev 2e8ab1ed6a Give chwriter.Registry periodic refresh, matching Tantivy's tracker
Closing search's active-tenant gap last commit surfaced a real asymmetry
by comparison: chwriter.Registry's per-tenant writer map was still a
snapshot built once at enterprise-ingest startup with no refresh at all,
while search's new ActiveTenantTracker refreshes every minute. A tenant
deprovisioned after enterprise-ingest started would keep writing
successfully to ClickHouse until the next restart -- a real, disclosed
staleness gap, not matched by anything on the Tantivy side anymore.

Registry.StartRefreshing spawns a goroutine that re-lists active tenants
every minute (dataSourceRefreshInterval, same interval as search's
tracker) via a new SourceLister callback and reconciles the writer map:
opens a connection for a newly-active tenant, closes and removes one no
longer active. New connections are dialed before taking the write lock,
so a slow/unreachable ClickHouse for one newly-active tenant never
blocks WriteBatch's read lock. A refresh failure (lister error, or one
tenant's connection failing to open) logs and leaves the existing map
untouched for that tick -- the same last-known-good posture
ActiveTenantTracker already uses, so a transient rbacstore/Postgres blip
doesn't evict every other tenant's already-working writer.

WriteBatch now takes a read lock and Close takes a write lock -- the
writer map was safe unsynchronized before only because it was immutable
after New() returned; StartRefreshing makes it mutable at runtime.

enterprise-ingest/main.go extracts the existing rbacstore-row-to-
DataSource adaptation into tenantDataSourceLister, reused for both the
initial synchronous load and StartRefreshing's periodic calls, so the
two can't drift into checking different things.

Verified: the lister-error-keeps-last-known-good path is Docker-free
(same "construct a Registry directly, bypass New" trick the existing
fail-closed tests use). The actual add/remove reconciliation against
real ClickHouse connections (TestRefreshAddsNewlyActiveTenant,
TestRefreshRemovesNoLongerActiveTenant) are skip-gated live-ClickHouse
tests, same CHWRITER_TEST_CLICKHOUSE_ADDR convention as this package's
existing integration tests -- not run against a live database in this
environment.

This closes the last disclosed gap from Phase 4's write-routing work:
both storage engines now share the same one-minute active-tenant
staleness bound instead of one being materially staler than the other.
2026-08-14 23:55:40 -07:00

261 lines
9.9 KiB
Go

// Fail-closed behavior (empty/unknown tenant_id) needs no live
// ClickHouse at all -- Registry.WriteBatch returns before ever touching
// a connection for those cases, so those tests run unconditionally.
// Everything that actually writes data is a real integration test
// against a live ClickHouse (same CHWRITER_TEST_CLICKHOUSE_ADDR
// convention as enterprise/internal/chrunner's own tests), skipped
// unless that's set; run via:
//
// docker run --rm --network sentry_default -v $(pwd)/../../..:/src -w /src/enterprise \
// -e CHWRITER_TEST_CLICKHOUSE_ADDR=clickhouse:9000 \
// -e CHWRITER_TEST_CLICKHOUSE_PASSWORD=sentry-dev-only \
// golang:1.25-alpine go test ./internal/chwriter/... -v
package chwriter
import (
"context"
"errors"
"fmt"
"io"
"log/slog"
"os"
"testing"
chdriver "github.com/ClickHouse/clickhouse-go/v2"
"github.com/google/uuid"
"github.com/sentry/sentry/enterprise/internal/tenantprovision"
"github.com/sentry/sentry/ingest/clickhousewriter"
"github.com/sentry/sentry/ingest/consumer"
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
)
func discardLogger() *slog.Logger {
return slog.New(slog.NewTextHandler(io.Discard, nil))
}
// TestWriteBatchRefusesEmptyTenantID and
// TestWriteBatchRefusesUnknownTenantWithEmptyRegistry construct a
// Registry directly (bypassing New, which would dial ClickHouse) so
// they genuinely run without Docker: WriteBatch's fail-closed checks
// happen before ever touching a real connection, purely a map lookup.
func TestWriteBatchRefusesEmptyTenantID(t *testing.T) {
reg := &Registry{writers: map[string]*clickhousewriter.Writer{}}
err := reg.WriteBatch(context.Background(), []consumer.Record{
{TenantID: "", Record: &logsv1.LogRecord{Message: "untagged"}},
})
if err == nil {
t.Fatal("expected WriteBatch to refuse a record with no tenant_id, not silently drop the tag")
}
}
func TestWriteBatchRefusesUnknownTenantWithEmptyRegistry(t *testing.T) {
reg := &Registry{writers: map[string]*clickhousewriter.Writer{}}
err := reg.WriteBatch(context.Background(), []consumer.Record{
{TenantID: "acme", Record: &logsv1.LogRecord{Message: "m"}},
})
if err == nil {
t.Fatal("expected WriteBatch to refuse a tenant with no entry in the registry")
}
}
// TestRefreshListerErrorLeavesRegistryUnchanged is Docker-free the same
// way the two tests above are: refresh's early-return on a lister error
// happens before anything touches ClickHouse, so this genuinely
// exercises the "keep last-known-good" path -- see refresh's doc
// comment on StartRefreshing.
func TestRefreshListerErrorLeavesRegistryUnchanged(t *testing.T) {
reg := &Registry{writers: map[string]*clickhousewriter.Writer{}}
lister := func(context.Context) ([]DataSource, error) {
return nil, errors.New("rbacstore unreachable")
}
reg.refresh(context.Background(), lister, discardLogger())
// Still refuses -- refresh must not have added a writer for "acme"
// (there's nothing a failed lister call could have legitimately
// learned), and must not have panicked reaching into a nil/partial
// state either.
err := reg.WriteBatch(context.Background(), []consumer.Record{
{TenantID: "acme", Record: &logsv1.LogRecord{Message: "m"}},
})
if err == nil {
t.Fatal("expected WriteBatch to still refuse tenant acme after a failed refresh")
}
}
func testAddr(t *testing.T) string {
t.Helper()
addr := os.Getenv("CHWRITER_TEST_CLICKHOUSE_ADDR")
if addr == "" {
t.Skip("CHWRITER_TEST_CLICKHOUSE_ADDR not set -- skipping live-ClickHouse integration test")
}
return addr
}
func provisionTestTenant(t *testing.T, addr string) (tenantID string, creds tenantprovision.Credentials) {
t.Helper()
admin, err := chdriver.Open(&chdriver.Options{
Addr: []string{addr},
Auth: chdriver.Auth{Database: "default", Username: "default", Password: os.Getenv("CHWRITER_TEST_CLICKHOUSE_PASSWORD")},
})
if err != nil {
t.Fatalf("opening admin connection: %v", err)
}
t.Cleanup(func() { admin.Close() })
tenantID = "cw" + uuid.NewString()[:8]
creds, err = tenantprovision.New(admin).ProvisionClickHouse(context.Background(), tenantID)
if err != nil {
t.Fatalf("provisioning tenant %s: %v", tenantID, err)
}
if err := admin.Exec(context.Background(), fmt.Sprintf(
"CREATE TABLE `%s`.logs (timestamp DateTime64(9), host String, service String, severity String, message String, attributes Map(String, String), record_id UUID) ENGINE = MergeTree ORDER BY timestamp",
tenantID)); err != nil {
t.Fatalf("creating logs table for tenant %s: %v", tenantID, err)
}
return tenantID, creds
}
// TestRegistryWritesEachTenantToItsOwnDatabase is the core adversarial
// probe for the write side, complementing chrunner's own read-side
// version: two tenants, two connections inside one Registry, one
// WriteBatch call mixing records from both, and a direct check (via an
// admin connection, not through Registry) that each tenant's row landed
// only in its own database.
func TestRegistryWritesEachTenantToItsOwnDatabase(t *testing.T) {
addr := testAddr(t)
ctx := context.Background()
tenantA, credsA := provisionTestTenant(t, addr)
tenantB, credsB := provisionTestTenant(t, addr)
reg, err := New(ctx, addr, []DataSource{
{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password},
{TenantID: tenantB, Database: tenantB, Username: credsB.Username, Password: credsB.Password},
})
if err != nil {
t.Fatalf("New: %v", err)
}
defer reg.Close()
err = reg.WriteBatch(ctx, []consumer.Record{
{TenantID: tenantA, Record: &logsv1.LogRecord{Host: "h1", Message: "for-a", RecordId: uuid.NewString()}},
{TenantID: tenantB, Record: &logsv1.LogRecord{Host: "h1", Message: "for-b", RecordId: uuid.NewString()}},
})
if err != nil {
t.Fatalf("WriteBatch: %v", err)
}
admin, err := chdriver.Open(&chdriver.Options{
Addr: []string{addr},
Auth: chdriver.Auth{Database: "default", Username: "default", Password: os.Getenv("CHWRITER_TEST_CLICKHOUSE_PASSWORD")},
})
if err != nil {
t.Fatalf("opening admin connection: %v", err)
}
defer admin.Close()
for tenantID, wantMessage := range map[string]string{tenantA: "for-a", tenantB: "for-b"} {
row := admin.QueryRow(ctx, fmt.Sprintf("SELECT message FROM `%s`.logs", tenantID))
var got string
if err := row.Scan(&got); err != nil {
t.Fatalf("querying %s's logs: %v", tenantID, err)
}
if got != wantMessage {
t.Fatalf("tenant %s's logs.message = %q, want %q", tenantID, got, wantMessage)
}
}
}
func TestRegistryRefusesUnprovisionedTenant(t *testing.T) {
addr := testAddr(t)
ctx := context.Background()
tenantA, credsA := provisionTestTenant(t, addr)
reg, err := New(ctx, addr, []DataSource{
{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password},
})
if err != nil {
t.Fatalf("New: %v", err)
}
defer reg.Close()
err = reg.WriteBatch(ctx, []consumer.Record{
{TenantID: "some-other-tenant-never-provisioned", Record: &logsv1.LogRecord{Host: "h1", Message: "m", RecordId: uuid.NewString()}},
})
if err == nil {
t.Fatal("expected WriteBatch to refuse a tenant with no provisioned connection, not silently drop or misroute it")
}
}
// TestRefreshAddsNewlyActiveTenant is the live counterpart to
// TestRefreshListerErrorLeavesRegistryUnchanged: proves refresh actually
// opens a real, usable connection for a tenant that appears in a later
// lister call but wasn't present at New() time -- the scenario
// StartRefreshing exists to handle (a tenant provisioned after
// enterprise-ingest already started).
func TestRefreshAddsNewlyActiveTenant(t *testing.T) {
addr := testAddr(t)
ctx := context.Background()
tenantA, credsA := provisionTestTenant(t, addr)
reg, err := New(ctx, addr, nil) // starts with zero tenants, same as a cold start before any tenant exists
if err != nil {
t.Fatalf("New: %v", err)
}
defer reg.Close()
if err := reg.WriteBatch(ctx, []consumer.Record{
{TenantID: tenantA, Record: &logsv1.LogRecord{Message: "m", RecordId: uuid.NewString()}},
}); err == nil {
t.Fatal("expected WriteBatch to refuse tenantA before the first refresh has run")
}
lister := func(context.Context) ([]DataSource, error) {
return []DataSource{{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password}}, nil
}
reg.refresh(ctx, lister, discardLogger())
if err := reg.WriteBatch(ctx, []consumer.Record{
{TenantID: tenantA, Record: &logsv1.LogRecord{Host: "h1", Message: "after-refresh", RecordId: uuid.NewString()}},
}); err != nil {
t.Fatalf("expected WriteBatch to succeed for tenantA after refresh added it, got: %v", err)
}
}
// TestRefreshRemovesNoLongerActiveTenant is TestRefreshAddsNewlyActiveTenant's
// mirror image: a tenant present at New() time that a later lister call
// no longer returns (deprovisioned or suspended) must lose its writer,
// not keep writing indefinitely until process restart -- the exact
// staleness gap this whole mechanism exists to close.
func TestRefreshRemovesNoLongerActiveTenant(t *testing.T) {
addr := testAddr(t)
ctx := context.Background()
tenantA, credsA := provisionTestTenant(t, addr)
reg, err := New(ctx, addr, []DataSource{
{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password},
})
if err != nil {
t.Fatalf("New: %v", err)
}
defer reg.Close()
if err := reg.WriteBatch(ctx, []consumer.Record{
{TenantID: tenantA, Record: &logsv1.LogRecord{Host: "h1", Message: "before-removal", RecordId: uuid.NewString()}},
}); err != nil {
t.Fatalf("expected WriteBatch to succeed for tenantA before refresh removes it, got: %v", err)
}
lister := func(context.Context) ([]DataSource, error) {
return nil, nil // tenantA no longer active/provisioned as of this refresh
}
reg.refresh(ctx, lister, discardLogger())
if err := reg.WriteBatch(ctx, []consumer.Record{
{TenantID: tenantA, Record: &logsv1.LogRecord{Message: "after-removal", RecordId: uuid.NewString()}},
}); err == nil {
t.Fatal("expected WriteBatch to refuse tenantA after refresh removed it, not keep writing with a stale connection")
}
}