butterrobot/internal/observability/sentry.go
Felipe M. a2e1196953
Some checks failed
CI / goreleaser-lint (push) Successful in 6s
CI / format (push) Successful in 57s
CI / test (push) Successful in 2m59s
CI / lint (push) Successful in 4m9s
CI / build (push) Successful in 7m17s
Release / release (push) Failing after 7m6s
feat: add Sentry observability support
Report errors, panics and optional performance traces to Sentry. Sentry
stays disabled unless SENTRY_DSN is set.

- slog ERROR records become Sentry issues, with the error attribute
  promoted to an exception so issues group by root cause
- queue worker, reminder worker, cache cleanup and HTTP handler panics
  are captured with a stack trace
- events carry platform/plugin/component tags for filtering
- HTTP requests are traced when SENTRY_TRACES_SAMPLE_RATE is above 0;
  /healthz is never traced

Configured credentials are redacted from every outgoing payload. This is
required rather than defensive: the Telegram webhook embeds the bot token
in its URL path, and failed Telegram API calls quote that URL in their
error text, so events would otherwise carry the token in the clear. A new
platform must register its credential in Config.Secrets().

Also moves the module to Go 1.27, refreshes every dependency and pins
golangci-lint v2.13.2. sentry-go 0.48 removed issue creation from its slog
integration, so the ERROR-to-issue conversion lives in
internal/observability/handler.go instead of relying on the SDK; leaving
it to the SDK would have silently downgraded issues to log lines.

Claude-Session: https://claude.ai/code/session_01W7tcpMTEyk9RrHvT7Be5zZ
2026-09-21 14:10:15 +02:00

136 lines
3.7 KiB
Go

// Package observability wires the application into Sentry for error reporting,
// performance tracing and structured log forwarding.
package observability
import (
"context"
"fmt"
"log/slog"
"net/http"
"strings"
"time"
"github.com/getsentry/sentry-go"
sentryhttp "github.com/getsentry/sentry-go/http"
"git.nakama.town/fmartingr/butterrobot/internal/config"
)
// flushTimeout bounds how long shutdown waits for pending Sentry deliveries.
const flushTimeout = 2 * time.Second
// InitSentry configures the global Sentry client. It is a no-op when no DSN is
// configured, which keeps Sentry entirely optional. Secrets are redacted from
// every payload before it leaves the process.
func InitSentry(cfg config.SentryConfig, debug bool, secrets []string) error {
if cfg.DSN == "" {
return nil
}
if err := sentry.Init(clientOptions(cfg, debug, secrets)); err != nil {
return fmt.Errorf("failed to initialize sentry: %w", err)
}
return nil
}
// clientOptions builds the Sentry client configuration.
func clientOptions(cfg config.SentryConfig, debug bool, secrets []string) sentry.ClientOptions {
options := sentry.ClientOptions{
Dsn: cfg.DSN,
Environment: cfg.Environment,
Release: cfg.Release,
Debug: debug,
AttachStacktrace: true,
EnableTracing: cfg.TracesSampleRate > 0,
TracesSampler: tracesSampler(cfg.TracesSampleRate),
}
if scrubber := newScrubber(secrets); scrubber != nil {
options.BeforeSend = func(event *sentry.Event, _ *sentry.EventHint) *sentry.Event {
return scrubber.event(event)
}
options.BeforeSendTransaction = func(event *sentry.Event, _ *sentry.EventHint) *sentry.Event {
return scrubber.event(event)
}
options.BeforeSendLog = scrubber.log
}
return options
}
// Enabled reports whether a Sentry client is configured.
func Enabled() bool {
return sentry.CurrentHub().Client() != nil
}
// Flush waits for buffered Sentry events to be delivered.
func Flush() {
if Enabled() {
sentry.Flush(flushTimeout)
}
}
// WithTags returns a context carrying a Sentry hub scoped with the given tags,
// so anything reported with that context is grouped and filterable by them.
func WithTags(ctx context.Context, tags map[string]string) context.Context {
if !Enabled() {
return ctx
}
hub := sentry.GetHubFromContext(ctx)
if hub == nil {
hub = sentry.CurrentHub()
}
hub = hub.Clone()
hub.Scope().SetTags(tags)
return sentry.SetHubOnContext(ctx, hub)
}
// HTTPMiddleware instruments HTTP handlers with Sentry request context,
// tracing and panic reporting.
func HTTPMiddleware(next http.Handler) http.Handler {
if !Enabled() {
return next
}
return sentryhttp.New(sentryhttp.Options{Repanic: true}).Handle(next)
}
// CapturePanic reports a recovered panic value to Sentry. Delivery is left to
// the background transport: the queue runs a single worker, so blocking here
// would stall message processing on every panic.
func CapturePanic(recovered any) {
if !Enabled() {
return
}
sentry.CurrentHub().Recover(recovered)
}
// RecoverAndCapture is a deferred guard for background goroutines: it reports
// the panic to Sentry, logs it and keeps the process alive.
func RecoverAndCapture(logger *slog.Logger, where string) {
recovered := recover()
if recovered == nil {
return
}
CapturePanic(recovered)
logger.Error("Recovered from panic", "where", where, "error", recovered)
}
// tracesSampler drops health check transactions so they do not dominate the
// performance quota.
func tracesSampler(rate float64) sentry.TracesSampler {
rate = min(max(rate, 0), 1)
return func(ctx sentry.SamplingContext) float64 {
if ctx.Span != nil && strings.HasSuffix(ctx.Span.Name, "/healthz") {
return 0.0
}
return rate
}
}