feat(coordinator): scaffold task-queue service in Go
Adds the SciMesh coordinator: a durable task-queue server on PostgreSQL that owns all database access, with workers reaching it over HTTP only. Structured as a modular monolith following Clean Architecture: domain entities and their invariants, no I/O usecase business operations + repository/clock ports transport HTTP handlers, DTOs, auth, error mapping storage PostgreSQL repositories, transactions carried in context infra config, pool, clock, server, lease reaper Dependencies point strictly inward; domain imports nothing from the module. Working: layer wiring, routing, shared-token auth, access logging, request IDs, domain-error to status-code mapping, transactional boundaries, graceful shutdown (HTTP drain -> reaper stop -> pool close), migrations, and a Compose stack starting Postgres -> migrations -> coordinator. The domain is complete and covered by unit tests that need no database: lease ownership, stale attempts, idempotent result replay, retry budgets, and lease expiry. Repository methods are stubs returning ErrNotImplemented (HTTP 501). The SQL for atomic claiming (FOR UPDATE SKIP LOCKED) and for lease expiry is written and ready to wire up. See coordinator/ARCHITECTURE.md for the layer map and a request traced through every layer.
This commit is contained in:
@@ -0,0 +1,51 @@
|
||||
package usecase
|
||||
|
||||
import "github.com/google/uuid"
|
||||
|
||||
// Use-case boundary types. Adapters map their wire formats onto these, so the
|
||||
// HTTP shape can change without touching business code.
|
||||
|
||||
type CreateJobInput struct {
|
||||
Workload string
|
||||
InputURI string
|
||||
Parameters map[string]any
|
||||
Chunks []ChunkInput
|
||||
}
|
||||
|
||||
type ChunkInput struct {
|
||||
ChunkIndex int
|
||||
Workload string
|
||||
InputURI string
|
||||
InputSHA256 string
|
||||
Parameters map[string]any
|
||||
MaxAttempts int
|
||||
}
|
||||
|
||||
type ClaimTaskInput struct {
|
||||
WorkerID string
|
||||
Workloads []string
|
||||
}
|
||||
|
||||
type RenewLeaseInput struct {
|
||||
TaskID uuid.UUID
|
||||
WorkerID string
|
||||
Attempt int
|
||||
}
|
||||
|
||||
type CompleteTaskInput struct {
|
||||
TaskID uuid.UUID
|
||||
WorkerID string
|
||||
Attempt int
|
||||
ResultURI string
|
||||
ResultSHA256 string
|
||||
Metrics map[string]any
|
||||
}
|
||||
|
||||
type FailTaskInput struct {
|
||||
TaskID uuid.UUID
|
||||
WorkerID string
|
||||
Attempt int
|
||||
ErrorCode string
|
||||
ErrorMessage string
|
||||
Retryable bool
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
package usecase
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/domain"
|
||||
)
|
||||
|
||||
// Job operations: the submitter-facing lifecycle of a whole submission.
|
||||
//
|
||||
// CreateJob register a job and fan it out into tasks
|
||||
// GetJobStatus aggregate progress
|
||||
// ListResults completed manifests, ordered for the stitcher
|
||||
// StitchJob merge partial results into the final artifact
|
||||
|
||||
// --- CreateJob -----------------------------------------------------------
|
||||
|
||||
type CreateJob struct {
|
||||
jobs JobRepository
|
||||
tasks TaskRepository
|
||||
tx TxManager
|
||||
clock Clock
|
||||
}
|
||||
|
||||
func NewCreateJob(jobs JobRepository, tasks TaskRepository, tx TxManager, clock Clock) *CreateJob {
|
||||
return &CreateJob{jobs: jobs, tasks: tasks, tx: tx, clock: clock}
|
||||
}
|
||||
|
||||
// Execute builds the job and its tasks, then writes them in one transaction.
|
||||
// The all-or-none guarantee comes from TxManager: a half-created job would
|
||||
// leave chunks no worker could ever complete.
|
||||
func (uc *CreateJob) Execute(ctx context.Context, in CreateJobInput) (*domain.Job, error) {
|
||||
chunks := make([]domain.ChunkSpec, 0, len(in.Chunks))
|
||||
for _, c := range in.Chunks {
|
||||
chunks = append(chunks, domain.ChunkSpec(c))
|
||||
}
|
||||
|
||||
job, tasks, err := domain.NewJobWithTasks(in.Workload, in.InputURI, in.Parameters, chunks, uc.clock.Now())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
err = uc.tx.WithinTx(ctx, func(ctx context.Context) error {
|
||||
if err := uc.jobs.Insert(ctx, job); err != nil {
|
||||
return err
|
||||
}
|
||||
return uc.tasks.InsertBatch(ctx, tasks)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// --- GetJobStatus --------------------------------------------------------
|
||||
|
||||
type GetJobStatus struct {
|
||||
jobs JobRepository
|
||||
tasks TaskRepository
|
||||
}
|
||||
|
||||
func NewGetJobStatus(jobs JobRepository, tasks TaskRepository) *GetJobStatus {
|
||||
return &GetJobStatus{jobs: jobs, tasks: tasks}
|
||||
}
|
||||
|
||||
func (uc *GetJobStatus) Execute(ctx context.Context, jobID uuid.UUID) (domain.JobProgress, error) {
|
||||
job, err := uc.jobs.Get(ctx, jobID)
|
||||
if err != nil {
|
||||
return domain.JobProgress{}, err
|
||||
}
|
||||
counts, err := uc.tasks.CountByStatus(ctx, jobID)
|
||||
if err != nil {
|
||||
return domain.JobProgress{}, err
|
||||
}
|
||||
return progressFrom(*job, counts), nil
|
||||
}
|
||||
|
||||
// --- ListResults ---------------------------------------------------------
|
||||
|
||||
type ListResults struct {
|
||||
tasks TaskRepository
|
||||
}
|
||||
|
||||
func NewListResults(tasks TaskRepository) *ListResults {
|
||||
return &ListResults{tasks: tasks}
|
||||
}
|
||||
|
||||
// Execute preserves chunk_index order: the stitcher merges these into one
|
||||
// artifact, and a non-deterministic order would make the final result depend on
|
||||
// which worker happened to finish first.
|
||||
func (uc *ListResults) Execute(ctx context.Context, jobID uuid.UUID) ([]domain.ResultManifest, error) {
|
||||
tasks, err := uc.tasks.ListCompleted(ctx, jobID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
manifests := make([]domain.ResultManifest, 0, len(tasks))
|
||||
for _, t := range tasks {
|
||||
if t.ResultURI == nil || t.ResultSHA256 == nil {
|
||||
continue // a completed task always carries both; skip defensively
|
||||
}
|
||||
manifests = append(manifests, domain.ResultManifest{
|
||||
TaskID: t.ID,
|
||||
ChunkIndex: t.ChunkIndex,
|
||||
ResultURI: *t.ResultURI,
|
||||
ResultSHA256: *t.ResultSHA256,
|
||||
Metrics: t.Metrics,
|
||||
})
|
||||
}
|
||||
return manifests, nil
|
||||
}
|
||||
|
||||
// --- StitchJob -----------------------------------------------------------
|
||||
|
||||
// StitchJob merges every chunk's partial result into the job's final artifact.
|
||||
// For similarity search that means concatenating each worker's local top-k,
|
||||
// sorting by similarity, and keeping the global top-k — the distributed result
|
||||
// must match what a single local run would produce.
|
||||
type StitchJob struct {
|
||||
results *ListResults
|
||||
}
|
||||
|
||||
func NewStitchJob(results *ListResults) *StitchJob {
|
||||
return &StitchJob{results: results}
|
||||
}
|
||||
|
||||
// Execute returns the URI of the assembled artifact.
|
||||
//
|
||||
// TODO(phase 6): fetch each manifest's CSV, merge, and persist the result.
|
||||
func (uc *StitchJob) Execute(ctx context.Context, jobID uuid.UUID) (string, error) {
|
||||
if _, err := uc.results.Execute(ctx, jobID); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return "", ErrNotImplemented
|
||||
}
|
||||
|
||||
// --- shared helpers ------------------------------------------------------
|
||||
|
||||
// progressFrom turns a status histogram into the domain's progress view.
|
||||
func progressFrom(job domain.Job, counts map[domain.TaskStatus]int) domain.JobProgress {
|
||||
p := domain.JobProgress{
|
||||
Job: job,
|
||||
Pending: counts[domain.TaskPending],
|
||||
Leased: counts[domain.TaskLeased],
|
||||
Done: counts[domain.TaskCompleted],
|
||||
Failed: counts[domain.TaskFailed],
|
||||
}
|
||||
for _, n := range counts {
|
||||
p.Total += n
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// syncJobStatus recomputes a job's status from its task counts and persists it.
|
||||
// Shared by CompleteTask and FailTask so both close a job by the same rule —
|
||||
// the rule itself lives in domain.JobProgress.DeriveStatus.
|
||||
func syncJobStatus(ctx context.Context, jobs JobRepository, tasks TaskRepository,
|
||||
jobID uuid.UUID, now time.Time) error {
|
||||
|
||||
counts, err := tasks.CountByStatus(ctx, jobID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
status := progressFrom(domain.Job{}, counts).DeriveStatus()
|
||||
|
||||
var completedAt *time.Time
|
||||
if status == domain.JobCompleted || status == domain.JobFailed {
|
||||
completedAt = &now
|
||||
}
|
||||
return jobs.UpdateStatus(ctx, jobID, status, completedAt)
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
// Package usecase holds the application's business operations. Each use case is
|
||||
// a small type with its dependencies injected and a single Execute method.
|
||||
//
|
||||
// The interfaces below are *ports*: they are declared here, by the consumer,
|
||||
// and implemented further out in storage/postgres. That is what keeps the
|
||||
// dependency rule intact — usecase never imports storage or transport.
|
||||
package usecase
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/domain"
|
||||
)
|
||||
|
||||
// ClaimFilter narrows which task a worker may be handed.
|
||||
type ClaimFilter struct {
|
||||
Workloads []string // workloads this worker can execute
|
||||
Owner string // worker ID taking the lease
|
||||
Now time.Time
|
||||
LeaseUntil time.Time
|
||||
}
|
||||
|
||||
// TaskRepository persists tasks.
|
||||
//
|
||||
// ClaimNext is deliberately coarse: leasing must be a single atomic statement
|
||||
// (SELECT ... FOR UPDATE SKIP LOCKED + UPDATE), so it cannot be decomposed into
|
||||
// Get+Update without losing the guarantee that one task goes to one worker.
|
||||
type TaskRepository interface {
|
||||
// ClaimNext atomically leases one matching pending task.
|
||||
// Returns (nil, nil) when nothing is available.
|
||||
ClaimNext(ctx context.Context, f ClaimFilter) (*domain.Task, error)
|
||||
|
||||
// GetForUpdate reads a task and locks its row for the enclosing
|
||||
// transaction, so read-modify-write use cases stay serialized.
|
||||
GetForUpdate(ctx context.Context, id uuid.UUID) (*domain.Task, error)
|
||||
|
||||
// Update persists a mutated task, honouring its Version for optimistic
|
||||
// concurrency.
|
||||
Update(ctx context.Context, t *domain.Task) error
|
||||
|
||||
InsertBatch(ctx context.Context, tasks []*domain.Task) error
|
||||
|
||||
// ListCompleted returns completed tasks ordered by chunk_index.
|
||||
ListCompleted(ctx context.Context, jobID uuid.UUID) ([]*domain.Task, error)
|
||||
|
||||
// CountByStatus aggregates a job's tasks for progress reporting.
|
||||
CountByStatus(ctx context.Context, jobID uuid.UUID) (map[domain.TaskStatus]int, error)
|
||||
|
||||
// ExpireLeases applies the lease-expiry rule to every elapsed task and
|
||||
// reports how many were affected.
|
||||
ExpireLeases(ctx context.Context, now time.Time) (int64, error)
|
||||
}
|
||||
|
||||
// JobRepository persists jobs.
|
||||
type JobRepository interface {
|
||||
Insert(ctx context.Context, j *domain.Job) error
|
||||
Get(ctx context.Context, id uuid.UUID) (*domain.Job, error)
|
||||
UpdateStatus(ctx context.Context, id uuid.UUID, status domain.JobStatus, completedAt *time.Time) error
|
||||
}
|
||||
|
||||
// TxManager runs a function inside one database transaction. The transaction
|
||||
// travels in the context, so repositories pick it up without this port ever
|
||||
// mentioning pgx.
|
||||
type TxManager interface {
|
||||
WithinTx(ctx context.Context, fn func(ctx context.Context) error) error
|
||||
}
|
||||
|
||||
// Clock supplies the current time. Injecting it keeps lease and expiry rules
|
||||
// testable without sleeping or freezing the system clock.
|
||||
type Clock interface {
|
||||
Now() time.Time
|
||||
}
|
||||
|
||||
// ErrNotImplemented marks scaffold code with no body yet. Unlike the errors in
|
||||
// domain, it describes the state of this codebase, not a business rule.
|
||||
var ErrNotImplemented = errors.New("not implemented")
|
||||
@@ -0,0 +1,206 @@
|
||||
package usecase
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/domain"
|
||||
)
|
||||
|
||||
// Task operations: the worker-facing lifecycle of a single chunk.
|
||||
//
|
||||
// ClaimTask lease the next available task
|
||||
// RenewLease extend a held lease (heartbeat)
|
||||
// CompleteTask record a successful result
|
||||
// FailTask record a failure
|
||||
// ExpireLeases reclaim leases that elapsed without a heartbeat
|
||||
|
||||
// --- ClaimTask -----------------------------------------------------------
|
||||
|
||||
type ClaimTask struct {
|
||||
tasks TaskRepository
|
||||
clock Clock
|
||||
leaseDuration time.Duration
|
||||
}
|
||||
|
||||
func NewClaimTask(tasks TaskRepository, clock Clock, leaseDuration time.Duration) *ClaimTask {
|
||||
return &ClaimTask{tasks: tasks, clock: clock, leaseDuration: leaseDuration}
|
||||
}
|
||||
|
||||
// Execute reclaims elapsed leases first, then hands out one task.
|
||||
//
|
||||
// Sweeping before claiming matters: otherwise a task abandoned by a dead worker
|
||||
// stays invisible until the reaper's next tick, and a waiting worker is told the
|
||||
// queue is empty while work sits idle.
|
||||
//
|
||||
// This use case is thin by design — the atomicity that makes claiming correct
|
||||
// lives in one SQL statement behind ClaimNext, and splitting it across the layer
|
||||
// boundary would break it.
|
||||
func (uc *ClaimTask) Execute(ctx context.Context, in ClaimTaskInput) (*domain.ClaimedTask, error) {
|
||||
if in.WorkerID == "" {
|
||||
return nil, domain.ErrInvalidInput
|
||||
}
|
||||
now := uc.clock.Now()
|
||||
|
||||
if _, err := uc.tasks.ExpireLeases(ctx, now); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
task, err := uc.tasks.ClaimNext(ctx, ClaimFilter{
|
||||
Workloads: in.Workloads,
|
||||
Owner: in.WorkerID,
|
||||
Now: now,
|
||||
LeaseUntil: now.Add(uc.leaseDuration),
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if task == nil {
|
||||
return nil, nil // empty queue is a normal state, not an error
|
||||
}
|
||||
|
||||
claimed := task.AsClaimed()
|
||||
return &claimed, nil
|
||||
}
|
||||
|
||||
// --- RenewLease ----------------------------------------------------------
|
||||
|
||||
type RenewLease struct {
|
||||
tasks TaskRepository
|
||||
tx TxManager
|
||||
clock Clock
|
||||
leaseDuration time.Duration
|
||||
}
|
||||
|
||||
func NewRenewLease(tasks TaskRepository, tx TxManager, clock Clock, leaseDuration time.Duration) *RenewLease {
|
||||
return &RenewLease{tasks: tasks, tx: tx, clock: clock, leaseDuration: leaseDuration}
|
||||
}
|
||||
|
||||
// Execute is a read-modify-write, so it runs inside a transaction with the row
|
||||
// locked: two concurrent heartbeats must not interleave into a lost update.
|
||||
// Whether the caller may renew at all is decided by the entity, not here.
|
||||
func (uc *RenewLease) Execute(ctx context.Context, in RenewLeaseInput) (*domain.ClaimedTask, error) {
|
||||
var claimed domain.ClaimedTask
|
||||
|
||||
err := uc.tx.WithinTx(ctx, func(ctx context.Context) error {
|
||||
task, err := uc.tasks.GetForUpdate(ctx, in.TaskID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := task.RenewLease(in.WorkerID, in.Attempt, uc.clock.Now().Add(uc.leaseDuration)); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := uc.tasks.Update(ctx, task); err != nil {
|
||||
return err
|
||||
}
|
||||
claimed = task.AsClaimed()
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &claimed, nil
|
||||
}
|
||||
|
||||
// --- CompleteTask --------------------------------------------------------
|
||||
|
||||
type CompleteTask struct {
|
||||
tasks TaskRepository
|
||||
jobs JobRepository
|
||||
tx TxManager
|
||||
clock Clock
|
||||
}
|
||||
|
||||
func NewCompleteTask(tasks TaskRepository, jobs JobRepository, tx TxManager, clock Clock) *CompleteTask {
|
||||
return &CompleteTask{tasks: tasks, jobs: jobs, tx: tx, clock: clock}
|
||||
}
|
||||
|
||||
// Execute applies the result and, when that was the job's last outstanding
|
||||
// task, closes the job in the same transaction — so a caller who sees a
|
||||
// completed task never observes its job still marked running.
|
||||
//
|
||||
// Lease ownership, staleness, and idempotent replays are all decided by
|
||||
// Task.CompleteWith; this use case only orchestrates.
|
||||
func (uc *CompleteTask) Execute(ctx context.Context, in CompleteTaskInput) (*domain.Task, error) {
|
||||
var out *domain.Task
|
||||
|
||||
err := uc.tx.WithinTx(ctx, func(ctx context.Context) error {
|
||||
task, err := uc.tasks.GetForUpdate(ctx, in.TaskID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
now := uc.clock.Now()
|
||||
if err := task.CompleteWith(in.ResultURI, in.ResultSHA256, in.Metrics,
|
||||
in.WorkerID, in.Attempt, now); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := uc.tasks.Update(ctx, task); err != nil {
|
||||
return err
|
||||
}
|
||||
out = task
|
||||
return syncJobStatus(ctx, uc.jobs, uc.tasks, task.JobID, now)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// --- FailTask ------------------------------------------------------------
|
||||
|
||||
type FailTask struct {
|
||||
tasks TaskRepository
|
||||
jobs JobRepository
|
||||
tx TxManager
|
||||
clock Clock
|
||||
}
|
||||
|
||||
func NewFailTask(tasks TaskRepository, jobs JobRepository, tx TxManager, clock Clock) *FailTask {
|
||||
return &FailTask{tasks: tasks, jobs: jobs, tx: tx, clock: clock}
|
||||
}
|
||||
|
||||
// Execute delegates the requeue-or-terminate decision to Task.Fail, then keeps
|
||||
// the parent job's status consistent in the same transaction.
|
||||
func (uc *FailTask) Execute(ctx context.Context, in FailTaskInput) (*domain.Task, error) {
|
||||
var out *domain.Task
|
||||
|
||||
err := uc.tx.WithinTx(ctx, func(ctx context.Context) error {
|
||||
task, err := uc.tasks.GetForUpdate(ctx, in.TaskID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
now := uc.clock.Now()
|
||||
if err := task.Fail(in.WorkerID, in.Attempt, in.ErrorCode, in.ErrorMessage, in.Retryable, now); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := uc.tasks.Update(ctx, task); err != nil {
|
||||
return err
|
||||
}
|
||||
out = task
|
||||
return syncJobStatus(ctx, uc.jobs, uc.tasks, task.JobID, now)
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// --- ExpireLeases --------------------------------------------------------
|
||||
|
||||
type ExpireLeases struct {
|
||||
tasks TaskRepository
|
||||
clock Clock
|
||||
}
|
||||
|
||||
func NewExpireLeases(tasks TaskRepository, clock Clock) *ExpireLeases {
|
||||
return &ExpireLeases{tasks: tasks, clock: clock}
|
||||
}
|
||||
|
||||
// Execute reports how many tasks were reclaimed.
|
||||
//
|
||||
// The sweep is one set-based statement rather than a load-decide-save loop:
|
||||
// several coordinators run it concurrently, and a single atomic UPDATE makes
|
||||
// the duplicate work harmless — the loser simply updates 0 rows.
|
||||
func (uc *ExpireLeases) Execute(ctx context.Context) (int64, error) {
|
||||
return uc.tasks.ExpireLeases(ctx, uc.clock.Now())
|
||||
}
|
||||
Reference in New Issue
Block a user