Some checks failed
Report errors, panics and optional performance traces to Sentry. Sentry stays disabled unless SENTRY_DSN is set. - slog ERROR records become Sentry issues, with the error attribute promoted to an exception so issues group by root cause - queue worker, reminder worker, cache cleanup and HTTP handler panics are captured with a stack trace - events carry platform/plugin/component tags for filtering - HTTP requests are traced when SENTRY_TRACES_SAMPLE_RATE is above 0; /healthz is never traced Configured credentials are redacted from every outgoing payload. This is required rather than defensive: the Telegram webhook embeds the bot token in its URL path, and failed Telegram API calls quote that URL in their error text, so events would otherwise carry the token in the clear. A new platform must register its credential in Config.Secrets(). Also moves the module to Go 1.27, refreshes every dependency and pins golangci-lint v2.13.2. sentry-go 0.48 removed issue creation from its slog integration, so the ERROR-to-issue conversion lives in internal/observability/handler.go instead of relying on the SDK; leaving it to the SDK would have silently downgraded issues to log lines. Claude-Session: https://claude.ai/code/session_01W7tcpMTEyk9RrHvT7Be5zZ
136 lines
3.7 KiB
Go
136 lines
3.7 KiB
Go
// Package observability wires the application into Sentry for error reporting,
|
|
// performance tracing and structured log forwarding.
|
|
package observability
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"log/slog"
|
|
"net/http"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/getsentry/sentry-go"
|
|
sentryhttp "github.com/getsentry/sentry-go/http"
|
|
|
|
"git.nakama.town/fmartingr/butterrobot/internal/config"
|
|
)
|
|
|
|
// flushTimeout bounds how long shutdown waits for pending Sentry deliveries.
|
|
const flushTimeout = 2 * time.Second
|
|
|
|
// InitSentry configures the global Sentry client. It is a no-op when no DSN is
|
|
// configured, which keeps Sentry entirely optional. Secrets are redacted from
|
|
// every payload before it leaves the process.
|
|
func InitSentry(cfg config.SentryConfig, debug bool, secrets []string) error {
|
|
if cfg.DSN == "" {
|
|
return nil
|
|
}
|
|
|
|
if err := sentry.Init(clientOptions(cfg, debug, secrets)); err != nil {
|
|
return fmt.Errorf("failed to initialize sentry: %w", err)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// clientOptions builds the Sentry client configuration.
|
|
func clientOptions(cfg config.SentryConfig, debug bool, secrets []string) sentry.ClientOptions {
|
|
options := sentry.ClientOptions{
|
|
Dsn: cfg.DSN,
|
|
Environment: cfg.Environment,
|
|
Release: cfg.Release,
|
|
Debug: debug,
|
|
AttachStacktrace: true,
|
|
EnableTracing: cfg.TracesSampleRate > 0,
|
|
TracesSampler: tracesSampler(cfg.TracesSampleRate),
|
|
}
|
|
|
|
if scrubber := newScrubber(secrets); scrubber != nil {
|
|
options.BeforeSend = func(event *sentry.Event, _ *sentry.EventHint) *sentry.Event {
|
|
return scrubber.event(event)
|
|
}
|
|
options.BeforeSendTransaction = func(event *sentry.Event, _ *sentry.EventHint) *sentry.Event {
|
|
return scrubber.event(event)
|
|
}
|
|
options.BeforeSendLog = scrubber.log
|
|
}
|
|
|
|
return options
|
|
}
|
|
|
|
// Enabled reports whether a Sentry client is configured.
|
|
func Enabled() bool {
|
|
return sentry.CurrentHub().Client() != nil
|
|
}
|
|
|
|
// Flush waits for buffered Sentry events to be delivered.
|
|
func Flush() {
|
|
if Enabled() {
|
|
sentry.Flush(flushTimeout)
|
|
}
|
|
}
|
|
|
|
// WithTags returns a context carrying a Sentry hub scoped with the given tags,
|
|
// so anything reported with that context is grouped and filterable by them.
|
|
func WithTags(ctx context.Context, tags map[string]string) context.Context {
|
|
if !Enabled() {
|
|
return ctx
|
|
}
|
|
|
|
hub := sentry.GetHubFromContext(ctx)
|
|
if hub == nil {
|
|
hub = sentry.CurrentHub()
|
|
}
|
|
|
|
hub = hub.Clone()
|
|
hub.Scope().SetTags(tags)
|
|
|
|
return sentry.SetHubOnContext(ctx, hub)
|
|
}
|
|
|
|
// HTTPMiddleware instruments HTTP handlers with Sentry request context,
|
|
// tracing and panic reporting.
|
|
func HTTPMiddleware(next http.Handler) http.Handler {
|
|
if !Enabled() {
|
|
return next
|
|
}
|
|
|
|
return sentryhttp.New(sentryhttp.Options{Repanic: true}).Handle(next)
|
|
}
|
|
|
|
// CapturePanic reports a recovered panic value to Sentry. Delivery is left to
|
|
// the background transport: the queue runs a single worker, so blocking here
|
|
// would stall message processing on every panic.
|
|
func CapturePanic(recovered any) {
|
|
if !Enabled() {
|
|
return
|
|
}
|
|
|
|
sentry.CurrentHub().Recover(recovered)
|
|
}
|
|
|
|
// RecoverAndCapture is a deferred guard for background goroutines: it reports
|
|
// the panic to Sentry, logs it and keeps the process alive.
|
|
func RecoverAndCapture(logger *slog.Logger, where string) {
|
|
recovered := recover()
|
|
if recovered == nil {
|
|
return
|
|
}
|
|
|
|
CapturePanic(recovered)
|
|
logger.Error("Recovered from panic", "where", where, "error", recovered)
|
|
}
|
|
|
|
// tracesSampler drops health check transactions so they do not dominate the
|
|
// performance quota.
|
|
func tracesSampler(rate float64) sentry.TracesSampler {
|
|
rate = min(max(rate, 0), 1)
|
|
|
|
return func(ctx sentry.SamplingContext) float64 {
|
|
if ctx.Span != nil && strings.HasSuffix(ctx.Span.Name, "/healthz") {
|
|
return 0.0
|
|
}
|
|
return rate
|
|
}
|
|
}
|