logs rows carry a real per-record `service` (nginx, smtp, ufw, ...) -- already true of the schema (storage/migrations/0001) and wire protocol, not something this feature invents. Both the deletion picker and the retention floor now operate on (host, service) pairs instead of whole hosts, so an operator can delete just one noisy log type from an agent without touching everything else it ships, and can protect one service (e.g. keep smtp a year) longer than the rest of that host's default. api/agents.ConfigOverride gains ServiceLogRetentionDays (map[string]int), owner-only to change like LogRetentionDays -- a service listed there overrides the host's LogRetentionDays default for that service only. Agent config page gets a matching "Per-service log retention overrides" add/remove list next to the existing host-level field. api/logretention: Store's count/delete now take []HostService and build a ClickHouse tuple IN ((?,?),...) over (host, service); AgentRetentionStore. FloorsByHost returns each host's default plus its per-service map, with HostFloor.Effective(service) resolving which one applies. preview/delete moved from GET/DELETE-with-query-params to POST-with-JSON-body (a list of targets needs a real body, not a repeated compound query param), and partitionTargets checks the floor per target so one protected service never blocks deleting a different, unprotected one in the same request. Settings' Log retention section is a two-level picker now: each host row (with a "select all services" checkbox and its default floor badge) expands to its services, each with its own count and effective protected-days badge. Verified live against real ClickHouse/Postgres and in-browser: a host with a 7-day default plus a 365-day smtp override -- deleting nginx+ smtp+ufw together correctly removed nginx and ufw, left smtp's 10 records untouched, and confirmed via a follow-up owner delete that bypassing the floor works. Also verified the full click-through (add a service override on the agent page, see it reflected in Settings' picker, select/preview/cancel) and confirmed no regression from the prior host-only version's tests.
454 lines
18 KiB
Go
454 lines
18 KiB
Go
package agents
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"log/slog"
|
|
"net/http"
|
|
"path"
|
|
"strings"
|
|
|
|
"github.com/sentry/sentry/api/authz"
|
|
)
|
|
|
|
// store is the narrow interface Handler depends on -- *Store (store.go)
|
|
// is the production implementation; tests use a fake, same pattern as
|
|
// dashboards.store/queryapi's SQLRunner.
|
|
type store interface {
|
|
List(ctx context.Context, tenantID string) ([]Agent, error)
|
|
Get(ctx context.Context, tenantID, host string) (*Agent, error)
|
|
SetOverride(ctx context.Context, tenantID, host string, override ConfigOverride, updatedBy string) (*Agent, error)
|
|
ClearOverride(ctx context.Context, tenantID, host string) error
|
|
IssueCommand(ctx context.Context, tenantID, host, command, issuedBy string) (*Agent, error)
|
|
}
|
|
|
|
// CommandLogger records an issued lifecycle command into the Phase 4
|
|
// audit trail -- same nil-by-default, fail-open shape as
|
|
// aiapi.InteractionLogger and queryapi.AuditLogger: a single-tenant
|
|
// deployment with no enterprise/ configured just doesn't log these.
|
|
// enterprise/internal/audit supplies the real implementation
|
|
// (event_type = 'agent_command', see
|
|
// metadata/migrations/0039_add_agent_command_event_type.sql) -- this is
|
|
// the one entry point in this package genuinely worth logging even
|
|
// without enterprise/ wired, given "strict RBAC, full audit trail" was
|
|
// the explicit precondition for building lifecycle commands at all (see
|
|
// /docs/agent-management-design.md); it degrades gracefully rather than
|
|
// being required, matching every other optional audit hook in this
|
|
// codebase, but a real deployment should wire it.
|
|
type CommandLogger interface {
|
|
LogCommand(ctx context.Context, entry CommandLogEntry) error
|
|
}
|
|
|
|
type CommandLogEntry struct {
|
|
Host string
|
|
Command string
|
|
IssuedBy string
|
|
}
|
|
|
|
type Handler struct {
|
|
logger *slog.Logger
|
|
store store
|
|
authorizer authz.Authorizer
|
|
commands CommandLogger
|
|
}
|
|
|
|
// commands may be nil -- see CommandLogger's doc comment.
|
|
func NewHandler(logger *slog.Logger, store store, authorizer authz.Authorizer, commands CommandLogger) *Handler {
|
|
return &Handler{logger: logger, store: store, authorizer: authorizer, commands: commands}
|
|
}
|
|
|
|
// RegisterRoutes: viewing inventory is RoleViewer (same bar as viewing
|
|
// a dashboard); editing an agent's remote config is RoleEditor -- an
|
|
// operational-tuning action, not an admin-only one, matching the RBAC
|
|
// matrix's treatment of alert rules/notification targets rather than
|
|
// user/role management. Issuing a lifecycle command is RoleAdmin --
|
|
// stricter than config editing, matching the matrix's treatment of
|
|
// similarly consequential actions (e.g. deleting a notification
|
|
// target) rather than day-to-day tuning.
|
|
func (h *Handler) RegisterRoutes(mux *http.ServeMux) {
|
|
mux.HandleFunc("GET /agents", authz.RequireRole(h.authorizer, authz.RoleViewer, h.handleList))
|
|
mux.HandleFunc("GET /agents/{host}", authz.RequireRole(h.authorizer, authz.RoleViewer, h.handleGet))
|
|
mux.HandleFunc("PUT /agents/{host}/config", authz.RequireRole(h.authorizer, authz.RoleEditor, h.handleSetConfig))
|
|
mux.HandleFunc("DELETE /agents/{host}/config", authz.RequireRole(h.authorizer, authz.RoleEditor, h.handleClearConfig))
|
|
mux.HandleFunc("PUT /agents/{host}/command", authz.RequireRole(h.authorizer, authz.RoleAdmin, h.handleIssueCommand))
|
|
}
|
|
|
|
// tenantID mirrors dashboards.Handler.tenantID exactly -- resolved from
|
|
// the authenticated identity, never from a client-supplied field
|
|
// (there isn't one here to begin with; host alone identifies an agent
|
|
// within a tenant).
|
|
func (h *Handler) tenantID(r *http.Request) string {
|
|
if id, ok := authz.IdentityFromContext(r.Context()); ok && id.TenantID != "" {
|
|
return id.TenantID
|
|
}
|
|
return "default"
|
|
}
|
|
|
|
func (h *Handler) updatedBy(r *http.Request) string {
|
|
if id, ok := authz.IdentityFromContext(r.Context()); ok {
|
|
return id.UserID
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func (h *Handler) handleList(w http.ResponseWriter, r *http.Request) {
|
|
list, err := h.store.List(r.Context(), h.tenantID(r))
|
|
if err != nil {
|
|
h.logger.Error("listing agents", "error", err)
|
|
writeError(w, http.StatusInternalServerError, "listing agents failed")
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, list)
|
|
}
|
|
|
|
func (h *Handler) handleGet(w http.ResponseWriter, r *http.Request) {
|
|
a, err := h.store.Get(r.Context(), h.tenantID(r), r.PathValue("host"))
|
|
if err != nil {
|
|
h.writeStoreErr(w, err, "getting agent")
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, a)
|
|
}
|
|
|
|
// setConfigRequest is deliberately the same shape as ConfigOverride
|
|
// (Handler just decodes straight into it) -- every field optional,
|
|
// unset means "no override for this field." A caller changing just one
|
|
// field (e.g. only heartbeat_interval_ms) must still send the fields
|
|
// they want to KEEP as an override alongside it, since SetOverride
|
|
// replaces the whole stored override -- the web UI's edit form always
|
|
// reads the agent's current DesiredOverride first and PUTs back the
|
|
// full merged set, same pattern any other "edit form that PUTs a whole
|
|
// resource" in this codebase already uses (e.g. dashboards' PUT).
|
|
func (h *Handler) handleSetConfig(w http.ResponseWriter, r *http.Request) {
|
|
var override ConfigOverride
|
|
if !decodeJSON(w, r, &override) {
|
|
return
|
|
}
|
|
if err := validateOverride(override); err != nil {
|
|
writeError(w, http.StatusBadRequest, err.Error())
|
|
return
|
|
}
|
|
|
|
// extra_file_paths is a materially different capability than the
|
|
// rest of this override: every other field tunes an already-running
|
|
// source, but this one tells the agent (which runs as root, with no
|
|
// filesystem sandboxing today -- see the security audit) to read and
|
|
// ship an arbitrary local file. RoleEditor is the right bar for
|
|
// "adjust batch size," not for "grant read access to any file on the
|
|
// host" -- so a request that actually *changes* the set of extra
|
|
// paths (adds or edits one -- shrinking or clearing never needs
|
|
// this, since that only removes capability) requires RoleAdmin,
|
|
// checked here rather than by splitting /agents/{host}/config into
|
|
// two routes with two RegisterRoutes role floors, which would break
|
|
// the "PUT replaces the whole override" contract every field here
|
|
// otherwise shares.
|
|
// A nil authorizer means no RBAC is configured at all (Phase 0-3
|
|
// default-open behavior) -- consistent with RequireRole's own
|
|
// no-op-when-nil posture, this extra gate only applies once an
|
|
// authorizer resolves a real Identity to check.
|
|
if identity, ok := authz.IdentityFromContext(r.Context()); ok && !identity.Role.Satisfies(authz.RoleAdmin) {
|
|
if changesExtraFilePaths(h.currentExtraFilePaths(r.Context(), h.tenantID(r), r.PathValue("host")), override.ExtraFilePaths) {
|
|
writeError(w, http.StatusForbidden, "extra_file_paths requires the admin role")
|
|
return
|
|
}
|
|
}
|
|
|
|
// log_retention_days requires RoleOwner specifically, not just
|
|
// RoleAdmin -- unlike extra_file_paths (an asymmetric "only granting
|
|
// more capability needs the stricter role" check), *any* change here
|
|
// needs the top role, including lowering or clearing an existing
|
|
// value. The whole point of this field is a floor only an owner can
|
|
// override (see api/logretention's doc comment); an admin able to
|
|
// freely lower or remove it would make that floor meaningless.
|
|
if identity, ok := authz.IdentityFromContext(r.Context()); ok && !identity.Role.Satisfies(authz.RoleOwner) {
|
|
if changesLogRetentionDays(h.currentLogRetentionDays(r.Context(), h.tenantID(r), r.PathValue("host")), override.LogRetentionDays) {
|
|
writeError(w, http.StatusForbidden, "log_retention_days requires the owner role")
|
|
return
|
|
}
|
|
}
|
|
|
|
// service_log_retention_days requires RoleOwner too, for exactly the
|
|
// same reason log_retention_days does -- it's the same override-floor
|
|
// mechanism, just keyed per service instead of once for the whole
|
|
// host, so it needs the same "any change, not just raising" gate.
|
|
if identity, ok := authz.IdentityFromContext(r.Context()); ok && !identity.Role.Satisfies(authz.RoleOwner) {
|
|
if changesServiceLogRetentionDays(h.currentServiceLogRetentionDays(r.Context(), h.tenantID(r), r.PathValue("host")), override.ServiceLogRetentionDays) {
|
|
writeError(w, http.StatusForbidden, "service_log_retention_days requires the owner role")
|
|
return
|
|
}
|
|
}
|
|
|
|
a, err := h.store.SetOverride(r.Context(), h.tenantID(r), r.PathValue("host"), override, h.updatedBy(r))
|
|
if err != nil {
|
|
h.writeStoreErr(w, err, "setting agent config")
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, a)
|
|
}
|
|
|
|
type issueCommandRequest struct {
|
|
Command string `json:"command"`
|
|
}
|
|
|
|
// handleIssueCommand queues a one-shot lifecycle command -- see
|
|
// Store.IssueCommand's doc comment for the delivery/clearing semantics.
|
|
// Logs to CommandLogger fail-open (a write failure is logged server-
|
|
// side and otherwise ignored, same posture as aiapi's interaction
|
|
// logging): audit-trail completeness matters, but it shouldn't be able
|
|
// to turn a legitimate restart request into a 500.
|
|
func (h *Handler) handleIssueCommand(w http.ResponseWriter, r *http.Request) {
|
|
var req issueCommandRequest
|
|
if !decodeJSON(w, r, &req) {
|
|
return
|
|
}
|
|
if !validCommand(req.Command) {
|
|
writeError(w, http.StatusBadRequest, `command must be "restart"`)
|
|
return
|
|
}
|
|
|
|
host := r.PathValue("host")
|
|
issuedBy := h.updatedBy(r)
|
|
a, err := h.store.IssueCommand(r.Context(), h.tenantID(r), host, req.Command, issuedBy)
|
|
if err != nil {
|
|
h.writeStoreErr(w, err, "issuing agent command")
|
|
return
|
|
}
|
|
|
|
if h.commands != nil {
|
|
if err := h.commands.LogCommand(r.Context(), CommandLogEntry{Host: host, Command: req.Command, IssuedBy: issuedBy}); err != nil {
|
|
h.logger.Error("logging agent command to audit trail", "host", host, "command", req.Command, "error", err)
|
|
}
|
|
}
|
|
|
|
writeJSON(w, http.StatusOK, a)
|
|
}
|
|
|
|
func (h *Handler) handleClearConfig(w http.ResponseWriter, r *http.Request) {
|
|
if err := h.store.ClearOverride(r.Context(), h.tenantID(r), r.PathValue("host")); err != nil {
|
|
h.writeStoreErr(w, err, "clearing agent config")
|
|
return
|
|
}
|
|
w.WriteHeader(http.StatusNoContent)
|
|
}
|
|
|
|
// validateOverride rejects the two footguns a naive remote-config-edit
|
|
// feature could otherwise ship: a batch/heartbeat interval of 0 would
|
|
// mean "flush constantly"/"heartbeat constantly," hammering ingest and
|
|
// the agent's own CPU for no operator-intended reason -- floors match
|
|
// this codebase's other real floors (alerting's own
|
|
// eval_interval_seconds >= 30, found live during the heartbeat feature
|
|
// this builds on). There is deliberately no validation here for
|
|
// tls/ingest fields, because ConfigOverride has no such fields at all
|
|
// -- ingest connection details are not a remotely-editable dimension of
|
|
// an agent's config, full stop (see /docs/agent-management-design.md's
|
|
// security boundary section).
|
|
func validateOverride(o ConfigOverride) error {
|
|
if o.BatchMaxSize != nil && *o.BatchMaxSize < 1 {
|
|
return errors.New("batch_max_size must be at least 1")
|
|
}
|
|
if o.BatchFlushIntervalMS != nil && *o.BatchFlushIntervalMS < 100 {
|
|
return errors.New("batch_flush_interval_ms must be at least 100")
|
|
}
|
|
if o.HeartbeatIntervalMS != nil && *o.HeartbeatIntervalMS < 5000 {
|
|
return errors.New("heartbeat_interval_ms must be at least 5000 (5s)")
|
|
}
|
|
if len(o.ExtraFilePaths) > 20 {
|
|
return errors.New("extra_file_paths: at most 20 paths")
|
|
}
|
|
for _, p := range o.ExtraFilePaths {
|
|
if err := validateExtraFilePath(p); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
// 3650 days (10 years) matches api/logretention's own
|
|
// maxOlderThanHours bound -- a floor further out than the deletion
|
|
// feature can even reach is meaningless, and rejecting an obviously-
|
|
// wrong input (a stray extra digit) here is cheaper than debugging it
|
|
// later as "why can no one ever delete logs."
|
|
if o.LogRetentionDays != nil && (*o.LogRetentionDays < 1 || *o.LogRetentionDays > 3650) {
|
|
return errors.New("log_retention_days must be between 1 and 3650")
|
|
}
|
|
// 200 services is far beyond any real host's log source variety --
|
|
// exists only to reject a pathological/malformed request, matching
|
|
// extra_file_paths' own cap-for-sanity-not-realistic-use posture.
|
|
if len(o.ServiceLogRetentionDays) > 200 {
|
|
return errors.New("service_log_retention_days: at most 200 services")
|
|
}
|
|
for service, days := range o.ServiceLogRetentionDays {
|
|
if service == "" {
|
|
return errors.New("service_log_retention_days: service name must not be empty")
|
|
}
|
|
if days < 1 || days > 3650 {
|
|
return fmt.Errorf("service_log_retention_days[%q] must be between 1 and 3650", service)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// extraFilePathDenylistPrefixes blocks whole directory trees that are
|
|
// never legitimate log-file locations but very commonly hold sensitive
|
|
// material an agent (which runs as root, unsandboxed, on every host
|
|
// this deployment has been checked against -- see the security audit)
|
|
// can otherwise read: OS credential/config storage, home directories,
|
|
// and kernel/process pseudo-filesystems.
|
|
var extraFilePathDenylistPrefixes = []string{"/etc/", "/root/", "/home/", "/proc/", "/sys/", "/boot/"}
|
|
|
|
// extraFilePathDenylistSubstrings catches credential material that can
|
|
// live outside the directories above too (e.g. a service account's
|
|
// SSH/cloud-credential directory under an app's own working directory,
|
|
// not necessarily /home or /root).
|
|
var extraFilePathDenylistSubstrings = []string{"/.ssh/", "/.gnupg/", "/.aws/", "/.kube/"}
|
|
|
|
// extraFilePathDenylistSuffixes catches specific high-value filenames by
|
|
// name, regardless of directory -- named here because the audit that
|
|
// motivated this check demonstrated /etc/shadow and an SSH private key
|
|
// specifically, and this covers both even outside the prefix-denylisted
|
|
// directories above (e.g. a private key accidentally copied to /opt).
|
|
var extraFilePathDenylistSuffixes = []string{"-key.pem", "id_rsa", "id_ecdsa", "id_ed25519", "id_dsa", "/shadow", "/gshadow"}
|
|
|
|
func validateExtraFilePath(p string) error {
|
|
if p == "" || !strings.HasPrefix(p, "/") {
|
|
return errors.New("extra_file_paths: each path must be a non-empty absolute path")
|
|
}
|
|
if strings.Contains(p, "..") {
|
|
return errors.New(`extra_file_paths: path must not contain ".."`)
|
|
}
|
|
if cleaned := path.Clean(p); cleaned != p {
|
|
return fmt.Errorf("extra_file_paths: %q must be in canonical form (e.g. %q)", p, cleaned)
|
|
}
|
|
for _, prefix := range extraFilePathDenylistPrefixes {
|
|
if p == strings.TrimSuffix(prefix, "/") || strings.HasPrefix(p, prefix) {
|
|
return fmt.Errorf("extra_file_paths: %q is not an allowed path (under denylisted %s)", p, prefix)
|
|
}
|
|
}
|
|
for _, substr := range extraFilePathDenylistSubstrings {
|
|
if strings.Contains(p, substr) {
|
|
return fmt.Errorf("extra_file_paths: %q is not an allowed path", p)
|
|
}
|
|
}
|
|
for _, suffix := range extraFilePathDenylistSuffixes {
|
|
if strings.HasSuffix(p, suffix) {
|
|
return fmt.Errorf("extra_file_paths: %q is not an allowed path", p)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// currentExtraFilePaths reads back the agent's already-stored override
|
|
// (empty/nil if the agent or override doesn't exist yet) so
|
|
// handleSetConfig can tell an addition/change apart from a pure
|
|
// shrink-or-clear -- see changesExtraFilePaths.
|
|
func (h *Handler) currentExtraFilePaths(ctx context.Context, tenantID, host string) []string {
|
|
a, err := h.store.Get(ctx, tenantID, host)
|
|
if err != nil || a.DesiredOverride == nil {
|
|
return nil
|
|
}
|
|
return a.DesiredOverride.ExtraFilePaths
|
|
}
|
|
|
|
// changesExtraFilePaths reports whether desired introduces any path not
|
|
// already present in current -- an addition or an edit, either of which
|
|
// grants the agent read access to something it couldn't read before.
|
|
// Removing paths (desired is a subset of current) is never a capability
|
|
// grant, so that alone never requires the stricter role handleSetConfig
|
|
// applies around this.
|
|
func changesExtraFilePaths(current, desired []string) bool {
|
|
existing := make(map[string]struct{}, len(current))
|
|
for _, p := range current {
|
|
existing[p] = struct{}{}
|
|
}
|
|
for _, p := range desired {
|
|
if _, ok := existing[p]; !ok {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// currentLogRetentionDays mirrors currentExtraFilePaths exactly, for
|
|
// the same reason: handleSetConfig needs the stored value to tell
|
|
// whether this request actually changes it.
|
|
func (h *Handler) currentLogRetentionDays(ctx context.Context, tenantID, host string) *int {
|
|
a, err := h.store.Get(ctx, tenantID, host)
|
|
if err != nil || a.DesiredOverride == nil {
|
|
return nil
|
|
}
|
|
return a.DesiredOverride.LogRetentionDays
|
|
}
|
|
|
|
// changesLogRetentionDays reports whether desired differs from current
|
|
// at all -- unlike changesExtraFilePaths, there is no safe direction
|
|
// here (see handleSetConfig's log_retention_days comment for why even
|
|
// lowering or clearing needs the same gate as raising it).
|
|
func changesLogRetentionDays(current, desired *int) bool {
|
|
if current == nil && desired == nil {
|
|
return false
|
|
}
|
|
if current == nil || desired == nil {
|
|
return true
|
|
}
|
|
return *current != *desired
|
|
}
|
|
|
|
// currentServiceLogRetentionDays mirrors currentLogRetentionDays exactly.
|
|
func (h *Handler) currentServiceLogRetentionDays(ctx context.Context, tenantID, host string) map[string]int {
|
|
a, err := h.store.Get(ctx, tenantID, host)
|
|
if err != nil || a.DesiredOverride == nil {
|
|
return nil
|
|
}
|
|
return a.DesiredOverride.ServiceLogRetentionDays
|
|
}
|
|
|
|
// changesServiceLogRetentionDays reports whether desired differs from
|
|
// current at all -- a full map comparison, same "no safe direction"
|
|
// posture as changesLogRetentionDays.
|
|
func changesServiceLogRetentionDays(current, desired map[string]int) bool {
|
|
if len(current) != len(desired) {
|
|
return true
|
|
}
|
|
for service, days := range desired {
|
|
if cur, ok := current[service]; !ok || cur != days {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func (h *Handler) writeStoreErr(w http.ResponseWriter, err error, action string) {
|
|
if errors.Is(err, ErrNotFound) {
|
|
writeError(w, http.StatusNotFound, "agent not found")
|
|
return
|
|
}
|
|
h.logger.Error(action, "error", err)
|
|
writeError(w, http.StatusInternalServerError, action+" failed")
|
|
}
|
|
|
|
const maxBodyBytes = 1 << 20 // 1 MiB, same cap as queryapi/dashboards
|
|
|
|
func decodeJSON(w http.ResponseWriter, r *http.Request, v any) bool {
|
|
r.Body = http.MaxBytesReader(w, r.Body, maxBodyBytes)
|
|
if err := json.NewDecoder(r.Body).Decode(v); err != nil {
|
|
writeError(w, http.StatusBadRequest, "invalid JSON body: "+err.Error())
|
|
return false
|
|
}
|
|
return true
|
|
}
|
|
|
|
func writeJSON(w http.ResponseWriter, status int, v any) {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
w.WriteHeader(status)
|
|
_ = json.NewEncoder(w).Encode(v)
|
|
}
|
|
|
|
type errorResponse struct {
|
|
Error string `json:"error"`
|
|
}
|
|
|
|
func writeError(w http.ResponseWriter, status int, msg string) {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
w.WriteHeader(status)
|
|
_ = json.NewEncoder(w).Encode(errorResponse{Error: msg})
|
|
}
|