feat(coordinator): running state, worker liveness, request-size limits

Polish pass hardening the queue and closing plan gaps.

- Task state machine gains `running`: the first heartbeat moves a task from
  leased to running (migrations 0006/0007 add the enum value and extend the
  lease-integrity check). verifyLease, ExpireLease, the reaper SQL, and job
  progress all treat leased and running alike.
- Worker liveness: a heartbeat from a registered worker (UUID worker_id) bumps
  its last_heartbeat_at online; a second background reaper marks workers offline
  after WORKER_OFFLINE_AFTER of silence (RunReaper generalized to RunPeriodic).
- Request-size limits: JSON bodies capped at 1 MiB; dataset/artifact uploads
  capped at MAX_UPLOAD_BYTES (default 1 GiB) via http.MaxBytesReader.
- Tests cover the running transition, liveness + offline reaper (unit over
  memstore and integration over Postgres).
This commit is contained in:
Efremenko Arhip
2026-07-23 17:36:26 +03:00
parent 4fc3c69fdf
commit 3b41455b20
23 changed files with 332 additions and 32 deletions
+21 -10
View File
@@ -72,7 +72,7 @@ func run() error {
CreateJob: usecase.NewCreateJob(jobRepo, taskRepo, tx, clk),
SubmitDataset: usecase.NewSubmitDataset(blobStore, artifactRepo, jobRepo, taskRepo, tx, clk),
ClaimTask: usecase.NewClaimTask(taskRepo, clk, cfg.LeaseDuration),
RenewLease: usecase.NewRenewLease(taskRepo, tx, clk, cfg.LeaseDuration),
RenewLease: usecase.NewRenewLease(taskRepo, workerRepo, tx, clk, cfg.LeaseDuration),
CompleteTask: usecase.NewCompleteTask(taskRepo, jobRepo, artifactRepo, tx, clk),
FailTask: usecase.NewFailTask(taskRepo, jobRepo, tx, clk),
GetJobStatus: usecase.NewGetJobStatus(jobRepo, taskRepo),
@@ -81,19 +81,30 @@ func run() error {
GetTaskInput: usecase.NewGetTaskInput(taskRepo, artifactRepo, blobStore),
}
// Background workers are tracked so shutdown can wait for them. Without
// this the process would exit while the reaper sat mid-UPDATE, and the
// deferred pool.Close() would pull connections out from under it.
// Background reapers are tracked so shutdown can wait for them. Without this
// the process would exit mid-UPDATE, and the deferred pool.Close() would pull
// connections out from under them.
expireLeases := usecase.NewExpireLeases(taskRepo, clk)
markOffline := usecase.NewMarkWorkersOffline(workerRepo, clk, cfg.WorkerOfflineAfter)
var wg sync.WaitGroup
wg.Add(1)
go func() {
defer wg.Done()
infra.RunReaper(ctx, log, usecase.NewExpireLeases(taskRepo, clk), cfg.ReaperInterval)
}()
for _, r := range []struct {
name string
fn func(context.Context) (int64, error)
}{
{"reaper requeued expired leases", expireLeases.Execute},
{"reaper marked workers offline", markOffline.Execute},
} {
wg.Add(1)
go func(name string, fn func(context.Context) (int64, error)) {
defer wg.Done()
infra.RunPeriodic(ctx, log, name, cfg.ReaperInterval, fn)
}(r.name, r.fn)
}
// pool.Ping backs /health: readiness means the database answers, not just
// that the process is alive.
api := httptransport.NewServer(useCases, log, cfg.RequestTimeout, cfg.HeartbeatInterval, pool.Ping)
api := httptransport.NewServer(useCases, log, cfg.RequestTimeout, cfg.HeartbeatInterval, cfg.MaxUploadBytes, pool.Ping)
err = infra.RunServer(ctx, log, cfg.Addr, api.Handler(cfg.Token))
// Shutdown order matters, and defers alone cannot express it (they run