Public source validation / validate (push) Failing after 3m8s
325 lines
12 KiB
Go
325 lines
12 KiB
Go
// Command worker is the ITWorx Pulse background runtime.
|
|
//
|
|
// It runs discovery/reconciliation, alert evaluation, service probes and the
|
|
// notification outbox drain on independent schedules, coordinated with any
|
|
// other worker through database leases. It is strictly observational
|
|
// (ADR-0001): it reads sources and writes Pulse's own state, and never mutates
|
|
// Unraid, Docker, the array or volumes.
|
|
package main
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"log/slog"
|
|
"net/netip"
|
|
"os"
|
|
"os/signal"
|
|
"reflect"
|
|
"strings"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/itworx/pulse/internal/agentsource"
|
|
"github.com/itworx/pulse/internal/agentstore"
|
|
"github.com/itworx/pulse/internal/alert"
|
|
"github.com/itworx/pulse/internal/alertworker"
|
|
"github.com/itworx/pulse/internal/config"
|
|
"github.com/itworx/pulse/internal/container"
|
|
"github.com/itworx/pulse/internal/database"
|
|
"github.com/itworx/pulse/internal/discovery"
|
|
"github.com/itworx/pulse/internal/inventory"
|
|
"github.com/itworx/pulse/internal/metriccatalog"
|
|
"github.com/itworx/pulse/internal/metricquery"
|
|
"github.com/itworx/pulse/internal/notification"
|
|
"github.com/itworx/pulse/internal/observability"
|
|
"github.com/itworx/pulse/internal/probe"
|
|
"github.com/itworx/pulse/internal/prometheus"
|
|
"github.com/itworx/pulse/internal/queryplan"
|
|
"github.com/itworx/pulse/internal/runtimeconfig"
|
|
"github.com/itworx/pulse/internal/servicedefaults"
|
|
"github.com/itworx/pulse/internal/workerruntime"
|
|
"github.com/jackc/pgx/v5/pgxpool"
|
|
)
|
|
|
|
// startupTimeout bounds every blocking call made before the scheduling loop
|
|
// starts, so a slow database cannot hold the process before its first
|
|
// heartbeat.
|
|
const startupTimeout = 15 * time.Second
|
|
|
|
func main() {
|
|
logger := slog.New(slog.NewJSONHandler(os.Stdout, nil))
|
|
if err := run(logger); err != nil {
|
|
logger.Error("pulse worker failed", "error", err)
|
|
os.Exit(1)
|
|
}
|
|
}
|
|
|
|
func run(logger *slog.Logger) error {
|
|
runtimeConfig, err := runtimeconfig.Load("worker")
|
|
if err != nil {
|
|
return err
|
|
}
|
|
application, err := config.LoadWorker()
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if strings.TrimSpace(application.DatabaseURL) == "" {
|
|
return errors.New("PULSE_DATABASE_URL is required: every worker job is database-coordinated")
|
|
}
|
|
ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
|
|
defer stop()
|
|
|
|
startupCtx, cancelStartup := context.WithTimeout(ctx, startupTimeout)
|
|
pool, err := database.NewPool(startupCtx, database.Config{URL: application.DatabaseURL, MaxConns: 8, MinConns: 1})
|
|
if err != nil {
|
|
cancelStartup()
|
|
return err
|
|
}
|
|
defer pool.Close()
|
|
if err := database.Ping(startupCtx, pool); err != nil {
|
|
cancelStartup()
|
|
return err
|
|
}
|
|
cancelStartup()
|
|
|
|
owner := workerOwner()
|
|
metrics := observability.NewRegistry(time.Now().UTC())
|
|
jobs, probeJob, err := buildJobs(application, pool, owner, logger)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
runtime, err := workerruntime.New(workerruntime.Config{
|
|
Owner: owner,
|
|
Tick: workerruntime.DefaultTick,
|
|
HeartbeatFile: runtimeConfig.HeartbeatFile,
|
|
DrainTimeout: runtimeConfig.ShutdownAfter,
|
|
Leases: workerruntime.PostgresLeaseStore{Pool: pool},
|
|
Logger: logger,
|
|
Metrics: metrics,
|
|
}, jobs...)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
logger.Info("pulse worker started",
|
|
"owner", owner, "environment", application.Environment, "jobs", jobNames(jobs),
|
|
"heartbeat_file", runtimeConfig.HeartbeatFile, "shutdown_timeout", runtimeConfig.ShutdownAfter.String(),
|
|
"config", application.String())
|
|
|
|
runErr := runtime.Run(ctx)
|
|
|
|
// Probe execution owns goroutines of its own; give them the same bounded
|
|
// grace as the scheduler before the process exits.
|
|
shutdownCtx, cancelShutdown := context.WithTimeout(context.WithoutCancel(ctx), runtimeConfig.ShutdownAfter)
|
|
defer cancelShutdown()
|
|
if probeJob != nil {
|
|
if err := probeJob.Shutdown(shutdownCtx); err != nil {
|
|
logger.Warn("probe shutdown incomplete", "error", err.Error())
|
|
}
|
|
}
|
|
for _, status := range runtime.Status() {
|
|
logger.Info("worker job final state", "job", status.Name, "component", status.Component,
|
|
"last_status", status.LastStatus, "runs", status.Runs, "failures", status.Failures, "skips", status.Skips)
|
|
}
|
|
logger.Info("pulse worker stopped")
|
|
return runErr
|
|
}
|
|
|
|
// buildJobs wires the repositories each job needs. A capability without its
|
|
// dependencies is scheduled anyway and reports Disabled with a reason, so an
|
|
// unconfigured feature is visible in system status instead of missing.
|
|
func buildJobs(application config.Config, pool *pgxpool.Pool, owner string, logger *slog.Logger) ([]workerruntime.Job, *workerruntime.ProbeJob, error) {
|
|
registry, err := metriccatalog.DefaultRegistry()
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
inventoryRepo, err := inventory.NewRepository(pool)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
discoveryStore, err := discovery.NewPostgresStore(pool, owner)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
notificationRepo, err := notification.NewRepository(pool)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
configureCtx, cancelConfigure := context.WithTimeout(context.Background(), startupTimeout)
|
|
defer cancelConfigure()
|
|
notificationFactories, err := configureWebhookChannel(configureCtx, application, notificationRepo)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
|
|
discoveryJob := workerruntime.DiscoveryJob{
|
|
SourceID: application.ContainerSourceID,
|
|
// Reuse the same bounded agent snapshot transport as the public API. A
|
|
// missing or stale snapshot resolves to Unknown and is skipped without
|
|
// tombstoning inventory; a fresh snapshot drives idempotent reconciliation.
|
|
Provider: containerDiscoveryProvider(pool),
|
|
Aliases: workerruntime.PostgresContainerAliasStore{Pool: pool},
|
|
Inventory: inventoryRepo,
|
|
Runner: discovery.Runner{Store: discoveryStore, MaxAttempts: 2, BaseRetry: 250 * time.Millisecond},
|
|
}
|
|
|
|
evaluator := &workerruntime.AlertEvaluator{
|
|
States: alert.StateRepository{Pool: pool},
|
|
Prior: workerruntime.PostgresAlertStateReader{Pool: pool},
|
|
Versions: alert.Repository{Pool: pool, Registry: registry},
|
|
Notifications: notificationRepo,
|
|
Logger: logger,
|
|
}
|
|
alertJob := workerruntime.AlertEvaluationJob{Reason: "metric_source_not_configured"}
|
|
if application.PrometheusURL != "" {
|
|
source, sourceErr := prometheus.New(application.PrometheusURL, nil, prometheus.Limits{Timeout: application.PrometheusTimeout})
|
|
if sourceErr != nil {
|
|
return nil, nil, sourceErr
|
|
}
|
|
planner := queryplan.NewPlanner(registry, queryplan.Limits{})
|
|
evaluator.Metrics = workerruntime.PrometheusMetricSource{Service: metricquery.NewService(planner, source, nil)}
|
|
worker, workerErr := alertworker.New(alert.Repository{Pool: pool, Registry: registry}, alertworker.PostgresLeaseStore{Pool: pool}, evaluator, alertworker.Config{
|
|
MaxConcurrent: 8, MaxBatch: workerruntime.MaxAlertRules, AttemptTimeout: 15 * time.Second, LeaseTTL: 2 * time.Minute, Owner: owner, Now: time.Now,
|
|
})
|
|
if workerErr != nil {
|
|
return nil, nil, workerErr
|
|
}
|
|
alertJob = workerruntime.AlertEvaluationJob{Worker: &worker, Enabled: true}
|
|
}
|
|
|
|
policy, err := probePolicy(application)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
serviceSummary, err := servicedefaults.Seed(configureCtx, pool, servicedefaults.Options{
|
|
PublicURL: application.PublicURL, OIDCIssuer: application.OIDCIssuer,
|
|
}, policy, probe.NetResolver{})
|
|
if err != nil {
|
|
return nil, nil, fmt.Errorf("configure system service monitoring: %w", err)
|
|
}
|
|
if serviceSummary.Services > 0 {
|
|
logger.Info("system service monitoring configured", "services", serviceSummary.Services,
|
|
"endpoints", serviceSummary.Endpoints, "probes", serviceSummary.Probes, "dependencies", serviceSummary.Dependencies)
|
|
}
|
|
probeJob, err := workerruntime.NewProbeJob(workerruntime.PostgresProbeStore{Pool: pool}, policy, logger)
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
|
|
notificationJob := workerruntime.NotificationDrainJob{
|
|
Store: notificationRepo,
|
|
Channels: notificationRepo,
|
|
Senders: map[string]notification.ChannelSender{},
|
|
Factories: notificationFactories,
|
|
Logger: logger,
|
|
}
|
|
|
|
jobs := workerruntime.Schedule(workerruntime.ScheduleRuns{
|
|
Discovery: discoveryJob.Run,
|
|
AlertEvaluation: alertJob.Run,
|
|
ProbeExecution: probeJob.Run,
|
|
NotificationDrain: notificationJob.Run,
|
|
})
|
|
return jobs, probeJob, nil
|
|
}
|
|
|
|
func containerDiscoveryProvider(pool *pgxpool.Pool) container.Provider {
|
|
return agentsource.ContainerProvider{
|
|
Reader: agentstore.PostgresStore{Pool: pool},
|
|
Windows: agentsource.Windows{},
|
|
}
|
|
}
|
|
|
|
func configureWebhookChannel(ctx context.Context, application config.Config, repository notification.ChannelStore) (map[string]notification.ChannelSenderFactory, error) {
|
|
if repository == nil {
|
|
return nil, notification.ErrUnavailable
|
|
}
|
|
current, getErr := repository.GetChannel(ctx, notification.DefaultWebhookChannelID)
|
|
if application.NotificationWebhookURL == "" {
|
|
if errors.Is(getErr, notification.ErrNotFound) {
|
|
return map[string]notification.ChannelSenderFactory{}, nil
|
|
}
|
|
if getErr != nil {
|
|
return nil, fmt.Errorf("read system webhook channel: %w", getErr)
|
|
}
|
|
if current.Enabled {
|
|
current.Enabled = false
|
|
if _, err := repository.UpdateChannel(ctx, current, current.Revision); err != nil {
|
|
return nil, fmt.Errorf("disable system webhook channel: %w", err)
|
|
}
|
|
}
|
|
return map[string]notification.ChannelSenderFactory{}, nil
|
|
}
|
|
desired := notification.Channel{
|
|
ID: notification.DefaultWebhookChannelID, Name: "Pulse webhook", Type: "webhook", Enabled: true,
|
|
SecretRef: notification.SecretRef{ID: notification.WebhookSecretReference},
|
|
Configuration: map[string]any{"url": application.NotificationWebhookURL, "timeoutSeconds": application.NotificationWebhookTimeout.Seconds()},
|
|
Revision: 1,
|
|
}
|
|
if errors.Is(getErr, notification.ErrNotFound) {
|
|
if _, err := repository.CreateChannel(ctx, desired); err != nil {
|
|
return nil, fmt.Errorf("create system webhook channel: %w", err)
|
|
}
|
|
} else if getErr != nil {
|
|
return nil, fmt.Errorf("read system webhook channel: %w", getErr)
|
|
} else if current.Name != desired.Name || current.Type != desired.Type || !current.Enabled || current.SecretRef != desired.SecretRef || !reflect.DeepEqual(current.Configuration, desired.Configuration) {
|
|
desired.Revision = current.Revision
|
|
if _, err := repository.UpdateChannel(ctx, desired, current.Revision); err != nil {
|
|
return nil, fmt.Errorf("update system webhook channel: %w", err)
|
|
}
|
|
}
|
|
resolver := notification.SecretResolverFunc(func(_ context.Context, ref notification.SecretRef) (string, error) {
|
|
if ref.ID != notification.WebhookSecretReference {
|
|
return "", notification.ErrNotFound
|
|
}
|
|
return application.NotificationWebhookToken, nil
|
|
})
|
|
return map[string]notification.ChannelSenderFactory{
|
|
"webhook": notification.WebhookFactory{
|
|
Secrets: resolver,
|
|
AllowHTTP: application.Environment != config.Production,
|
|
},
|
|
}, nil
|
|
}
|
|
|
|
// probePolicy builds the probe network policy. Only administrator-configured
|
|
// private ranges are added to the allowlist; every other protection in
|
|
// internal/probe/policy.go keeps its default, so link-local, multicast, cloud
|
|
// metadata and unlisted private addresses stay blocked.
|
|
func probePolicy(application config.Config) (probe.NetworkPolicy, error) {
|
|
policy := probe.NetworkPolicy{}
|
|
for _, entry := range application.ProbeAllowedNetworks {
|
|
prefix, err := netip.ParsePrefix(entry)
|
|
if err != nil {
|
|
return probe.NetworkPolicy{}, fmt.Errorf("probe allowlist entry %q is invalid", entry)
|
|
}
|
|
policy.AllowedNetworks = append(policy.AllowedNetworks, prefix)
|
|
}
|
|
if err := policy.Validate(); err != nil {
|
|
return probe.NetworkPolicy{}, err
|
|
}
|
|
return policy, nil
|
|
}
|
|
|
|
// workerOwner identifies this process in job_runs.lease_owner. It contains no
|
|
// secret and stays stable for the lifetime of the process.
|
|
func workerOwner() string {
|
|
host, err := os.Hostname()
|
|
if err != nil || strings.TrimSpace(host) == "" {
|
|
host = "worker"
|
|
}
|
|
owner := fmt.Sprintf("%s/%d", host, os.Getpid())
|
|
if len(owner) > 120 {
|
|
owner = owner[:120]
|
|
}
|
|
return owner
|
|
}
|
|
|
|
func jobNames(jobs []workerruntime.Job) []string {
|
|
names := make([]string, 0, len(jobs))
|
|
for _, job := range jobs {
|
|
names = append(names, job.Name+"@"+job.Interval.String())
|
|
}
|
|
return names
|
|
}
|