Add safe artifact previews
This commit is contained in:
@@ -0,0 +1,146 @@
|
||||
package usecase
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/csv"
|
||||
"errors"
|
||||
"io"
|
||||
"mime"
|
||||
"strings"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/domain"
|
||||
)
|
||||
|
||||
// The preview is deliberately a diagnostic aid, never a full artifact
|
||||
// viewer. These limits bound both memory use and storage reads.
|
||||
const (
|
||||
previewMaxRows = 30
|
||||
previewMaxBytes = 64 * 1024
|
||||
)
|
||||
|
||||
// ArtifactPreviewView is the safe, bounded data rendered by the operator UI.
|
||||
// It deliberately contains neither storage keys nor worker-local details.
|
||||
type ArtifactPreviewView struct {
|
||||
JobID string
|
||||
ArtifactID string
|
||||
Filename string
|
||||
Diagnostic bool
|
||||
Previewable bool
|
||||
Reason string
|
||||
Headers []string
|
||||
Rows [][]string
|
||||
Truncated bool
|
||||
RowLimit int
|
||||
ByteLimit int64
|
||||
}
|
||||
|
||||
// PreviewArtifact reads the beginning of a job-scoped CSV result. Partial
|
||||
// results are diagnostic; a final result is available only after the reducer
|
||||
// has persisted it as this job's completed result.
|
||||
type PreviewArtifact struct {
|
||||
read UIReadRepository
|
||||
blobs BlobStore
|
||||
}
|
||||
|
||||
func NewPreviewArtifact(read UIReadRepository, blobs BlobStore) *PreviewArtifact {
|
||||
return &PreviewArtifact{read: read, blobs: blobs}
|
||||
}
|
||||
|
||||
func (p *PreviewArtifact) Execute(ctx context.Context, jobID, artifactID uuid.UUID) (ArtifactPreviewView, error) {
|
||||
job, err := p.read.GetJob(ctx, jobID)
|
||||
if err != nil {
|
||||
return ArtifactPreviewView{}, err
|
||||
}
|
||||
artifacts, err := p.read.ListArtifactsByJob(ctx, jobID)
|
||||
if err != nil {
|
||||
return ArtifactPreviewView{}, err
|
||||
}
|
||||
|
||||
var artifact *domain.Artifact
|
||||
for i := range artifacts {
|
||||
if artifacts[i].ID == artifactID {
|
||||
artifact = &artifacts[i]
|
||||
break
|
||||
}
|
||||
}
|
||||
if artifact == nil || !previewableArtifact(*job, *artifact) {
|
||||
// Use one response for an unknown artifact, another job's artifact, and
|
||||
// an artifact that is not yet public. This avoids leaking its state.
|
||||
return ArtifactPreviewView{}, domain.ErrArtifactNotFound
|
||||
}
|
||||
|
||||
view := ArtifactPreviewView{
|
||||
JobID: jobID.String(),
|
||||
ArtifactID: artifact.ID.String(),
|
||||
Filename: artifact.Filename,
|
||||
Diagnostic: artifact.Kind == domain.ArtifactPartialResult,
|
||||
RowLimit: previewMaxRows,
|
||||
ByteLimit: previewMaxBytes,
|
||||
}
|
||||
if !isCSVArtifact(artifact) {
|
||||
view.Reason = "This artifact is not a CSV file, so it cannot be shown as text here. Download it instead."
|
||||
return view, nil
|
||||
}
|
||||
if artifact.SizeBytes == 0 {
|
||||
view.Reason = "This artifact is empty."
|
||||
return view, nil
|
||||
}
|
||||
|
||||
body, err := p.blobs.Open(ctx, artifact.StorageKey)
|
||||
if err != nil {
|
||||
return ArtifactPreviewView{}, err
|
||||
}
|
||||
defer func() { _ = body.Close() }()
|
||||
|
||||
limited := &io.LimitedReader{R: body, N: previewMaxBytes}
|
||||
reader := csv.NewReader(limited)
|
||||
reader.FieldsPerRecord = -1 // a byte limit may end inside a record
|
||||
|
||||
headers, err := reader.Read()
|
||||
if err != nil {
|
||||
view.Reason = "This artifact could not be read as CSV."
|
||||
return view, nil
|
||||
}
|
||||
view.Headers = append([]string(nil), headers...)
|
||||
view.Rows = make([][]string, 0, previewMaxRows)
|
||||
for len(view.Rows) < previewMaxRows {
|
||||
record, readErr := reader.Read()
|
||||
if readErr != nil {
|
||||
if !errors.Is(readErr, io.EOF) {
|
||||
view.Truncated = true
|
||||
}
|
||||
break
|
||||
}
|
||||
view.Rows = append(view.Rows, append([]string(nil), record...))
|
||||
}
|
||||
|
||||
if artifact.SizeBytes > previewMaxBytes {
|
||||
view.Truncated = true
|
||||
} else if len(view.Rows) == previewMaxRows {
|
||||
if _, readErr := reader.Read(); readErr == nil {
|
||||
view.Truncated = true
|
||||
}
|
||||
}
|
||||
view.Previewable = true
|
||||
return view, nil
|
||||
}
|
||||
|
||||
func previewableArtifact(job domain.Job, artifact domain.Artifact) bool {
|
||||
if artifact.Kind == domain.ArtifactPartialResult {
|
||||
return true
|
||||
}
|
||||
return artifact.Kind == domain.ArtifactFinalResult &&
|
||||
job.Status == domain.JobCompleted &&
|
||||
job.ResultArtifactID != nil &&
|
||||
*job.ResultArtifactID == artifact.ID
|
||||
}
|
||||
|
||||
func isCSVArtifact(artifact *domain.Artifact) bool {
|
||||
mediaType, _, err := mime.ParseMediaType(artifact.ContentType)
|
||||
if err == nil && strings.EqualFold(mediaType, "text/csv") {
|
||||
return true
|
||||
}
|
||||
return strings.HasSuffix(strings.ToLower(artifact.Filename), ".csv")
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
package usecase_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/domain"
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/memstore"
|
||||
"github.com/emil28092005/SciMesh/coordinator/internal/usecase"
|
||||
)
|
||||
|
||||
func newPreviewHarness() (*usecase.PreviewArtifact, *memstore.JobRepo, *memstore.ArtifactRepo, *memstore.BlobStore) {
|
||||
jobs := memstore.NewJobRepo()
|
||||
tasks := memstore.NewTaskRepo()
|
||||
workers := memstore.NewWorkerRepo()
|
||||
artifacts := memstore.NewArtifactRepo()
|
||||
blobs := memstore.NewBlobStore()
|
||||
return usecase.NewPreviewArtifact(memstore.NewUIReadRepo(jobs, tasks, workers, artifacts), blobs), jobs, artifacts, blobs
|
||||
}
|
||||
|
||||
func previewJob(t *testing.T, jobs *memstore.JobRepo, status domain.JobStatus) uuid.UUID {
|
||||
t.Helper()
|
||||
job := &domain.Job{ID: uuid.New(), Workload: "similarity-search", Status: status, CreatedAt: time.Now().UTC()}
|
||||
if err := jobs.Insert(context.Background(), job); err != nil {
|
||||
t.Fatalf("insert preview job: %v", err)
|
||||
}
|
||||
return job.ID
|
||||
}
|
||||
|
||||
func previewArtifact(t *testing.T, artifacts *memstore.ArtifactRepo, blobs *memstore.BlobStore, jobID uuid.UUID, kind domain.ArtifactKind, filename, contentType, contents string) uuid.UUID {
|
||||
t.Helper()
|
||||
id := uuid.New()
|
||||
sha, size, err := blobs.Put(context.Background(), id.String(), strings.NewReader(contents))
|
||||
if err != nil {
|
||||
t.Fatalf("store preview artifact: %v", err)
|
||||
}
|
||||
artifact := &domain.Artifact{ID: id, JobID: jobID, Kind: kind, Filename: filename, StorageKey: id.String(), ContentType: contentType, SizeBytes: size, SHA256: sha, CreatedAt: time.Now().UTC()}
|
||||
if err := artifacts.Insert(context.Background(), artifact); err != nil {
|
||||
t.Fatalf("insert preview artifact: %v", err)
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
func TestPreviewArtifactRendersBoundedCSV(t *testing.T) {
|
||||
preview, jobs, artifacts, blobs := newPreviewHarness()
|
||||
jobID := previewJob(t, jobs, domain.JobRunning)
|
||||
var csv strings.Builder
|
||||
csv.WriteString("chembl_id,score\n")
|
||||
for i := 0; i < 40; i++ {
|
||||
csv.WriteString("CHEMBL" + strconv.Itoa(i) + ",0.9\n")
|
||||
}
|
||||
artifactID := previewArtifact(t, artifacts, blobs, jobID, domain.ArtifactPartialResult, "partial.csv", "text/csv; charset=utf-8", csv.String())
|
||||
|
||||
view, err := preview.Execute(context.Background(), jobID, artifactID)
|
||||
if err != nil {
|
||||
t.Fatalf("preview: %v", err)
|
||||
}
|
||||
if !view.Previewable || !view.Diagnostic || !view.Truncated || len(view.Rows) != 30 || view.Headers[0] != "chembl_id" || view.Rows[0][0] != "CHEMBL0" {
|
||||
t.Fatalf("unexpected preview: %+v", view)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPreviewArtifactCapsStorageReadAndHandlesInvalidCSV(t *testing.T) {
|
||||
preview, jobs, artifacts, blobs := newPreviewHarness()
|
||||
jobID := previewJob(t, jobs, domain.JobRunning)
|
||||
// Fewer than 30 oversized records force the byte cap, rather than the row
|
||||
// cap, to stop parsing.
|
||||
large := "id,value\n" + strings.Repeat("row,"+strings.Repeat("x", 5*1024)+"\n", 20)
|
||||
largeID := previewArtifact(t, artifacts, blobs, jobID, domain.ArtifactPartialResult, "large.csv", "text/csv", large)
|
||||
view, err := preview.Execute(context.Background(), jobID, largeID)
|
||||
if err != nil || !view.Previewable || !view.Truncated || len(view.Rows) > 30 {
|
||||
t.Fatalf("large preview = (%+v, %v)", view, err)
|
||||
}
|
||||
invalidID := previewArtifact(t, artifacts, blobs, jobID, domain.ArtifactPartialResult, "broken.csv", "text/csv", "\"unterminated")
|
||||
invalid, err := preview.Execute(context.Background(), jobID, invalidID)
|
||||
if err != nil || invalid.Previewable || invalid.Reason == "" {
|
||||
t.Fatalf("invalid preview = (%+v, %v)", invalid, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPreviewArtifactRejectsOtherJobsAndNonResults(t *testing.T) {
|
||||
preview, jobs, artifacts, blobs := newPreviewHarness()
|
||||
jobA := previewJob(t, jobs, domain.JobRunning)
|
||||
jobB := previewJob(t, jobs, domain.JobRunning)
|
||||
partialID := previewArtifact(t, artifacts, blobs, jobA, domain.ArtifactPartialResult, "partial.csv", "text/csv", "a,b\n1,2\n")
|
||||
if _, err := preview.Execute(context.Background(), jobB, partialID); !errors.Is(err, domain.ErrArtifactNotFound) {
|
||||
t.Fatalf("cross-job preview error = %v", err)
|
||||
}
|
||||
inputID := previewArtifact(t, artifacts, blobs, jobA, domain.ArtifactInput, "input.csv", "text/csv", "a,b\n1,2\n")
|
||||
if _, err := preview.Execute(context.Background(), jobA, inputID); !errors.Is(err, domain.ErrArtifactNotFound) {
|
||||
t.Fatalf("input preview error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPreviewArtifactExposesOnlyPersistedCompletedFinalResult(t *testing.T) {
|
||||
preview, jobs, artifacts, blobs := newPreviewHarness()
|
||||
jobID := previewJob(t, jobs, domain.JobReducing)
|
||||
finalID := previewArtifact(t, artifacts, blobs, jobID, domain.ArtifactFinalResult, "final.csv", "text/csv", "rank,chembl_id\n1,CHEMBL1\n")
|
||||
if _, err := preview.Execute(context.Background(), jobID, finalID); !errors.Is(err, domain.ErrArtifactNotFound) {
|
||||
t.Fatalf("uncompleted final preview error = %v", err)
|
||||
}
|
||||
if err := jobs.CompleteWithResult(context.Background(), jobID, finalID, time.Now().UTC()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
view, err := preview.Execute(context.Background(), jobID, finalID)
|
||||
if err != nil || !view.Previewable || view.Diagnostic {
|
||||
t.Fatalf("completed final preview = (%+v, %v)", view, err)
|
||||
}
|
||||
if err := jobs.FailReduction(context.Background(), jobID, "reducer_failed", "final result reduction failed", time.Now().UTC()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := preview.Execute(context.Background(), jobID, finalID); !errors.Is(err, domain.ErrArtifactNotFound) {
|
||||
t.Fatalf("failed reducer preview error = %v", err)
|
||||
}
|
||||
}
|
||||
@@ -180,7 +180,7 @@ func (d *Dashboard) JobDetail(ctx context.Context, jobID uuid.UUID) (JobDetailVi
|
||||
}
|
||||
for _, artifact := range artifacts {
|
||||
diagnostic := artifact.Kind == domain.ArtifactPartialResult
|
||||
downloadable := diagnostic || (artifact.Kind == domain.ArtifactFinalResult && out.Status == string(domain.JobCompleted))
|
||||
downloadable := previewableArtifact(*job, artifact)
|
||||
out.Artifacts = append(out.Artifacts, ArtifactCard{ID: artifact.ID.String(), Kind: string(artifact.Kind), Filename: artifact.Filename, SizeBytes: artifact.SizeBytes, SHA256: artifact.SHA256, Downloadable: downloadable, Diagnostic: diagnostic})
|
||||
if artifact.Kind == domain.ArtifactFinalResult && downloadable {
|
||||
out.FinalResultAvailable = true
|
||||
@@ -189,13 +189,21 @@ func (d *Dashboard) JobDetail(ctx context.Context, jobID uuid.UUID) (JobDetailVi
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (d *Dashboard) ArtifactBelongsToJob(ctx context.Context, jobID, artifactID uuid.UUID) (bool, error) {
|
||||
// DownloadableArtifactBelongsToJob applies the same policy used by the UI
|
||||
// projection: partial diagnostics and the persisted final result are public to
|
||||
// the operator; source inputs and shards are not exposed through a guessed UI
|
||||
// URL.
|
||||
func (d *Dashboard) DownloadableArtifactBelongsToJob(ctx context.Context, jobID, artifactID uuid.UUID) (bool, error) {
|
||||
job, err := d.read.GetJob(ctx, jobID)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
artifacts, err := d.read.ListArtifactsByJob(ctx, jobID)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
for _, a := range artifacts {
|
||||
if a.ID == artifactID {
|
||||
if a.ID == artifactID && previewableArtifact(*job, a) {
|
||||
return true, nil
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user