Give chwriter.Registry periodic refresh, matching Tantivy's tracker
Closing search's active-tenant gap last commit surfaced a real asymmetry by comparison: chwriter.Registry's per-tenant writer map was still a snapshot built once at enterprise-ingest startup with no refresh at all, while search's new ActiveTenantTracker refreshes every minute. A tenant deprovisioned after enterprise-ingest started would keep writing successfully to ClickHouse until the next restart -- a real, disclosed staleness gap, not matched by anything on the Tantivy side anymore. Registry.StartRefreshing spawns a goroutine that re-lists active tenants every minute (dataSourceRefreshInterval, same interval as search's tracker) via a new SourceLister callback and reconciles the writer map: opens a connection for a newly-active tenant, closes and removes one no longer active. New connections are dialed before taking the write lock, so a slow/unreachable ClickHouse for one newly-active tenant never blocks WriteBatch's read lock. A refresh failure (lister error, or one tenant's connection failing to open) logs and leaves the existing map untouched for that tick -- the same last-known-good posture ActiveTenantTracker already uses, so a transient rbacstore/Postgres blip doesn't evict every other tenant's already-working writer. WriteBatch now takes a read lock and Close takes a write lock -- the writer map was safe unsynchronized before only because it was immutable after New() returned; StartRefreshing makes it mutable at runtime. enterprise-ingest/main.go extracts the existing rbacstore-row-to- DataSource adaptation into tenantDataSourceLister, reused for both the initial synchronous load and StartRefreshing's periodic calls, so the two can't drift into checking different things. Verified: the lister-error-keeps-last-known-good path is Docker-free (same "construct a Registry directly, bypass New" trick the existing fail-closed tests use). The actual add/remove reconciliation against real ClickHouse connections (TestRefreshAddsNewlyActiveTenant, TestRefreshRemovesNoLongerActiveTenant) are skip-gated live-ClickHouse tests, same CHWRITER_TEST_CLICKHOUSE_ADDR convention as this package's existing integration tests -- not run against a live database in this environment. This closes the last disclosed gap from Phase 4's write-routing work: both storage engines now share the same one-minute active-tenant staleness bound instead of one being materially staler than the other.
This commit is contained in:
@@ -14,7 +14,10 @@ package chwriter
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
@@ -27,6 +30,10 @@ import (
|
||||
logsv1 "github.com/sentry/sentry/proto/sentry/logs/v1"
|
||||
)
|
||||
|
||||
func discardLogger() *slog.Logger {
|
||||
return slog.New(slog.NewTextHandler(io.Discard, nil))
|
||||
}
|
||||
|
||||
// TestWriteBatchRefusesEmptyTenantID and
|
||||
// TestWriteBatchRefusesUnknownTenantWithEmptyRegistry construct a
|
||||
// Registry directly (bypassing New, which would dial ClickHouse) so
|
||||
@@ -52,6 +59,31 @@ func TestWriteBatchRefusesUnknownTenantWithEmptyRegistry(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRefreshListerErrorLeavesRegistryUnchanged is Docker-free the same
|
||||
// way the two tests above are: refresh's early-return on a lister error
|
||||
// happens before anything touches ClickHouse, so this genuinely
|
||||
// exercises the "keep last-known-good" path -- see refresh's doc
|
||||
// comment on StartRefreshing.
|
||||
func TestRefreshListerErrorLeavesRegistryUnchanged(t *testing.T) {
|
||||
reg := &Registry{writers: map[string]*clickhousewriter.Writer{}}
|
||||
lister := func(context.Context) ([]DataSource, error) {
|
||||
return nil, errors.New("rbacstore unreachable")
|
||||
}
|
||||
|
||||
reg.refresh(context.Background(), lister, discardLogger())
|
||||
|
||||
// Still refuses -- refresh must not have added a writer for "acme"
|
||||
// (there's nothing a failed lister call could have legitimately
|
||||
// learned), and must not have panicked reaching into a nil/partial
|
||||
// state either.
|
||||
err := reg.WriteBatch(context.Background(), []consumer.Record{
|
||||
{TenantID: "acme", Record: &logsv1.LogRecord{Message: "m"}},
|
||||
})
|
||||
if err == nil {
|
||||
t.Fatal("expected WriteBatch to still refuse tenant acme after a failed refresh")
|
||||
}
|
||||
}
|
||||
|
||||
func testAddr(t *testing.T) string {
|
||||
t.Helper()
|
||||
addr := os.Getenv("CHWRITER_TEST_CLICKHOUSE_ADDR")
|
||||
@@ -155,3 +187,74 @@ func TestRegistryRefusesUnprovisionedTenant(t *testing.T) {
|
||||
t.Fatal("expected WriteBatch to refuse a tenant with no provisioned connection, not silently drop or misroute it")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRefreshAddsNewlyActiveTenant is the live counterpart to
|
||||
// TestRefreshListerErrorLeavesRegistryUnchanged: proves refresh actually
|
||||
// opens a real, usable connection for a tenant that appears in a later
|
||||
// lister call but wasn't present at New() time -- the scenario
|
||||
// StartRefreshing exists to handle (a tenant provisioned after
|
||||
// enterprise-ingest already started).
|
||||
func TestRefreshAddsNewlyActiveTenant(t *testing.T) {
|
||||
addr := testAddr(t)
|
||||
ctx := context.Background()
|
||||
tenantA, credsA := provisionTestTenant(t, addr)
|
||||
|
||||
reg, err := New(ctx, addr, nil) // starts with zero tenants, same as a cold start before any tenant exists
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
defer reg.Close()
|
||||
|
||||
if err := reg.WriteBatch(ctx, []consumer.Record{
|
||||
{TenantID: tenantA, Record: &logsv1.LogRecord{Message: "m", RecordId: uuid.NewString()}},
|
||||
}); err == nil {
|
||||
t.Fatal("expected WriteBatch to refuse tenantA before the first refresh has run")
|
||||
}
|
||||
|
||||
lister := func(context.Context) ([]DataSource, error) {
|
||||
return []DataSource{{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password}}, nil
|
||||
}
|
||||
reg.refresh(ctx, lister, discardLogger())
|
||||
|
||||
if err := reg.WriteBatch(ctx, []consumer.Record{
|
||||
{TenantID: tenantA, Record: &logsv1.LogRecord{Host: "h1", Message: "after-refresh", RecordId: uuid.NewString()}},
|
||||
}); err != nil {
|
||||
t.Fatalf("expected WriteBatch to succeed for tenantA after refresh added it, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRefreshRemovesNoLongerActiveTenant is TestRefreshAddsNewlyActiveTenant's
|
||||
// mirror image: a tenant present at New() time that a later lister call
|
||||
// no longer returns (deprovisioned or suspended) must lose its writer,
|
||||
// not keep writing indefinitely until process restart -- the exact
|
||||
// staleness gap this whole mechanism exists to close.
|
||||
func TestRefreshRemovesNoLongerActiveTenant(t *testing.T) {
|
||||
addr := testAddr(t)
|
||||
ctx := context.Background()
|
||||
tenantA, credsA := provisionTestTenant(t, addr)
|
||||
|
||||
reg, err := New(ctx, addr, []DataSource{
|
||||
{TenantID: tenantA, Database: tenantA, Username: credsA.Username, Password: credsA.Password},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
defer reg.Close()
|
||||
|
||||
if err := reg.WriteBatch(ctx, []consumer.Record{
|
||||
{TenantID: tenantA, Record: &logsv1.LogRecord{Host: "h1", Message: "before-removal", RecordId: uuid.NewString()}},
|
||||
}); err != nil {
|
||||
t.Fatalf("expected WriteBatch to succeed for tenantA before refresh removes it, got: %v", err)
|
||||
}
|
||||
|
||||
lister := func(context.Context) ([]DataSource, error) {
|
||||
return nil, nil // tenantA no longer active/provisioned as of this refresh
|
||||
}
|
||||
reg.refresh(ctx, lister, discardLogger())
|
||||
|
||||
if err := reg.WriteBatch(ctx, []consumer.Record{
|
||||
{TenantID: tenantA, Record: &logsv1.LogRecord{Message: "after-removal", RecordId: uuid.NewString()}},
|
||||
}); err == nil {
|
||||
t.Fatal("expected WriteBatch to refuse tenantA after refresh removed it, not keep writing with a stale connection")
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user