feat(coordinator): running state, worker liveness, request-size limits
Polish pass hardening the queue and closing plan gaps. - Task state machine gains `running`: the first heartbeat moves a task from leased to running (migrations 0006/0007 add the enum value and extend the lease-integrity check). verifyLease, ExpireLease, the reaper SQL, and job progress all treat leased and running alike. - Worker liveness: a heartbeat from a registered worker (UUID worker_id) bumps its last_heartbeat_at online; a second background reaper marks workers offline after WORKER_OFFLINE_AFTER of silence (RunReaper generalized to RunPeriodic). - Request-size limits: JSON bodies capped at 1 MiB; dataset/artifact uploads capped at MAX_UPLOAD_BYTES (default 1 GiB) via http.MaxBytesReader. - Tests cover the running transition, liveness + offline reaper (unit over memstore and integration over Postgres).
This commit is contained in:
@@ -72,7 +72,7 @@ func run() error {
|
||||
CreateJob: usecase.NewCreateJob(jobRepo, taskRepo, tx, clk),
|
||||
SubmitDataset: usecase.NewSubmitDataset(blobStore, artifactRepo, jobRepo, taskRepo, tx, clk),
|
||||
ClaimTask: usecase.NewClaimTask(taskRepo, clk, cfg.LeaseDuration),
|
||||
RenewLease: usecase.NewRenewLease(taskRepo, tx, clk, cfg.LeaseDuration),
|
||||
RenewLease: usecase.NewRenewLease(taskRepo, workerRepo, tx, clk, cfg.LeaseDuration),
|
||||
CompleteTask: usecase.NewCompleteTask(taskRepo, jobRepo, artifactRepo, tx, clk),
|
||||
FailTask: usecase.NewFailTask(taskRepo, jobRepo, tx, clk),
|
||||
GetJobStatus: usecase.NewGetJobStatus(jobRepo, taskRepo),
|
||||
@@ -81,19 +81,30 @@ func run() error {
|
||||
GetTaskInput: usecase.NewGetTaskInput(taskRepo, artifactRepo, blobStore),
|
||||
}
|
||||
|
||||
// Background workers are tracked so shutdown can wait for them. Without
|
||||
// this the process would exit while the reaper sat mid-UPDATE, and the
|
||||
// deferred pool.Close() would pull connections out from under it.
|
||||
// Background reapers are tracked so shutdown can wait for them. Without this
|
||||
// the process would exit mid-UPDATE, and the deferred pool.Close() would pull
|
||||
// connections out from under them.
|
||||
expireLeases := usecase.NewExpireLeases(taskRepo, clk)
|
||||
markOffline := usecase.NewMarkWorkersOffline(workerRepo, clk, cfg.WorkerOfflineAfter)
|
||||
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
infra.RunReaper(ctx, log, usecase.NewExpireLeases(taskRepo, clk), cfg.ReaperInterval)
|
||||
}()
|
||||
for _, r := range []struct {
|
||||
name string
|
||||
fn func(context.Context) (int64, error)
|
||||
}{
|
||||
{"reaper requeued expired leases", expireLeases.Execute},
|
||||
{"reaper marked workers offline", markOffline.Execute},
|
||||
} {
|
||||
wg.Add(1)
|
||||
go func(name string, fn func(context.Context) (int64, error)) {
|
||||
defer wg.Done()
|
||||
infra.RunPeriodic(ctx, log, name, cfg.ReaperInterval, fn)
|
||||
}(r.name, r.fn)
|
||||
}
|
||||
|
||||
// pool.Ping backs /health: readiness means the database answers, not just
|
||||
// that the process is alive.
|
||||
api := httptransport.NewServer(useCases, log, cfg.RequestTimeout, cfg.HeartbeatInterval, pool.Ping)
|
||||
api := httptransport.NewServer(useCases, log, cfg.RequestTimeout, cfg.HeartbeatInterval, cfg.MaxUploadBytes, pool.Ping)
|
||||
err = infra.RunServer(ctx, log, cfg.Addr, api.Handler(cfg.Token))
|
||||
|
||||
// Shutdown order matters, and defers alone cannot express it (they run
|
||||
|
||||
Reference in New Issue
Block a user