Drive coordinator upload and reduction from the workload catalog

This commit is contained in:
Emil
2026-08-02 17:47:47 +03:00
parent 700a96a259
commit 749396da05
26 changed files with 1473 additions and 184 deletions
+273
View File
@@ -0,0 +1,273 @@
// Package workloads provides the coordinator-side view of the SDK workload
// library. The catalog is generated by `scimesh workload export` and embedded
// into the binary; it is presentation and orchestration metadata, never
// executable code. Every field that drives coordinator behaviour (reduction
// mode, parameter schema, required input columns) is validated at load time
// so a bad export fails fast instead of misbehaving at runtime.
package workloads
import (
"embed"
"encoding/json"
"fmt"
"sort"
)
//go:embed workloads.json
var catalogFile embed.FS
// Catalog is the parsed and validated workload library.
type Catalog struct {
workloads []*Workload
}
// Workload is one entry of the embedded catalog.
type Workload struct {
Name string `json:"name"`
Version string `json:"version"`
Description string `json:"description"`
Capabilities []string `json:"capabilities"`
TrustModes []string `json:"trust_modes"`
Determinism string `json:"determinism"`
Verifier string `json:"verifier"`
Enabled bool `json:"enabled"`
Reduction string `json:"reduction"`
UploadReady bool `json:"upload_ready"`
Parameters map[string]any `json:"parameters_schema"`
UIElements []UIElement `json:"ui_elements"`
Inputs map[string]any `json:"inputs"`
Outputs map[string]any `json:"outputs"`
requiredColumns map[string]bool // derived from input validator configuration
inputMediaTypes map[string]string // port name -> media type
parameterDefaults map[string]any // derived from the schema
}
// UIElement is one workload-declared form control for the "new job" page.
type UIElement struct {
Field string `json:"field"`
Widget string `json:"widget"`
Label string `json:"label"`
Help string `json:"help"`
Placeholder string `json:"placeholder"`
Options []string `json:"options"`
Default any `json:"default"`
Order int `json:"order"`
Group string `json:"group"`
}
type libraryFile struct {
SchemaVersion int `json:"schema_version"`
GeneratedBy string `json:"generated_by"`
Workloads []*Workload `json:"workloads"`
}
// Load reads and validates the embedded catalog.
func Load() (*Catalog, error) {
raw, err := catalogFile.ReadFile("workloads.json")
if err != nil {
return nil, fmt.Errorf("read embedded workload catalog: %w", err)
}
return Parse(raw)
}
// Parse validates and builds a Catalog from catalog JSON bytes.
func Parse(raw []byte) (*Catalog, error) {
var file libraryFile
if err := json.Unmarshal(raw, &file); err != nil {
return nil, fmt.Errorf("parse workload catalog: %w", err)
}
if file.SchemaVersion != 2 {
return nil, fmt.Errorf("workload catalog schema_version must be 2, got %d", file.SchemaVersion)
}
if len(file.Workloads) == 0 {
return nil, fmt.Errorf("workload catalog contains no workloads")
}
catalog := &Catalog{}
names := make(map[string]bool, len(file.Workloads))
for _, workload := range file.Workloads {
if err := validateWorkload(workload); err != nil {
return nil, err
}
if names[workload.Name] {
return nil, fmt.Errorf("workload catalog lists %q more than once", workload.Name)
}
names[workload.Name] = true
catalog.workloads = append(catalog.workloads, workload)
}
sort.Slice(catalog.workloads, func(i, j int) bool {
return catalog.workloads[i].Name < catalog.workloads[j].Name
})
return catalog, nil
}
func validateWorkload(workload *Workload) error {
if workload.Name == "" || workload.Version == "" {
return fmt.Errorf("workload catalog entry must have a name and version")
}
switch workload.Reduction {
case "top-k", "ordered-concat":
default:
return fmt.Errorf("workload %q declares unknown reduction %q", workload.Name, workload.Reduction)
}
if err := validateSchema(workload.Name, workload.Parameters); err != nil {
return err
}
workload.parameterDefaults = schemaDefaults(workload.Parameters)
workload.requiredColumns = map[string]bool{}
for portName, port := range workload.Inputs {
config, ok := port.(map[string]any)
if !ok {
continue
}
validator, _ := config["validator_configuration"].(map[string]any)
columns, _ := validator["required_columns"].([]any)
for _, column := range columns {
text, ok := column.(string)
if ok {
workload.requiredColumns[text] = true
}
}
if mediaType, ok := config["media_type"].(string); ok {
if workload.inputMediaTypes == nil {
workload.inputMediaTypes = map[string]string{}
}
workload.inputMediaTypes[portName] = mediaType
}
}
for _, element := range workload.UIElements {
if err := validateUIElement(workload, element); err != nil {
return err
}
}
return nil
}
func validateUIElement(workload *Workload, element UIElement) error {
switch element.Widget {
case "text", "textarea", "number", "select", "checkbox":
default:
return fmt.Errorf("workload %q ui element %q has unknown widget %q", workload.Name, element.Field, element.Widget)
}
if element.Widget == "select" && len(element.Options) == 0 {
return fmt.Errorf("workload %q ui element %q is a select without options", workload.Name, element.Field)
}
properties, ok := workload.Parameters["properties"].(map[string]any)
if !ok {
return nil
}
property, declared := properties[element.Field].(map[string]any)
if !declared {
return fmt.Errorf("workload %q ui element %q does not name a declared parameter", workload.Name, element.Field)
}
if schemaType(property) == "boolean" && element.Widget != "checkbox" {
return fmt.Errorf("workload %q ui element %q must use the checkbox widget for a boolean parameter", workload.Name, element.Field)
}
return nil
}
// Enabled returns the enabled workloads, sorted by name.
func (c *Catalog) Enabled() []*Workload {
result := make([]*Workload, 0, len(c.workloads))
for _, workload := range c.workloads {
if workload.Enabled {
result = append(result, workload)
}
}
return result
}
// ByName returns the workload with the given name, or nil.
func (c *Catalog) ByName(name string) *Workload {
for _, workload := range c.workloads {
if workload.Name == name {
return workload
}
}
return nil
}
// ValidateParameters checks job parameters against the workload schema. It
// enforces the strict subset of JSON Schema used by the SDK manifests: object
// shape, types, required, enum, numeric bounds, string lengths, and oneOf.
func (c *Catalog) ValidateParameters(name string, parameters map[string]any) error {
workload := c.ByName(name)
if workload == nil {
return fmt.Errorf("unknown workload %q", name)
}
if !workload.Enabled {
return fmt.Errorf("workload %q is not enabled", name)
}
return validateParameters(name, workload.Parameters, parameters)
}
// UploadReady reports whether the workload can be driven from a single
// uploaded dataset file. Workloads that need planner-produced inputs (such as
// the graph block-pair shards) declare upload_ready=false.
func (c *Catalog) UploadReady(name string) bool {
workload := c.ByName(name)
if workload == nil {
return false
}
return workload.UploadReady
}
// RequiredColumns reports every column the workload's input port requires.
func (c *Catalog) RequiredColumns(name string) map[string]bool {
workload := c.ByName(name)
if workload == nil {
return nil
}
return workload.requiredColumns
}
// InputMediaType returns the declared media type of the named input port.
func (c *Catalog) InputMediaType(name, port string) string {
workload := c.ByName(name)
if workload == nil {
return ""
}
return workload.inputMediaTypes[port]
}
// InputPortNames returns the sorted input port names of the workload.
func (c *Catalog) InputPortNames(name string) []string {
workload := c.ByName(name)
if workload == nil {
return nil
}
names := make([]string, 0, len(workload.Inputs))
for port := range workload.Inputs {
names = append(names, port)
}
sort.Strings(names)
return names
}
// Reduction returns the reduction mode for the workload, or "" if unknown.
func (c *Catalog) Reduction(name string) string {
workload := c.ByName(name)
if workload == nil {
return ""
}
return workload.Reduction
}
// ParameterDefaults returns the schema-declared defaults for the workload.
func (c *Catalog) ParameterDefaults(name string) map[string]any {
workload := c.ByName(name)
if workload == nil {
return nil
}
return workload.parameterDefaults
}
// ParameterSchema returns the schema property for a workload parameter, and
// whether the parameter exists.
func (w *Workload) PropertySchema(field string) (map[string]any, bool) {
properties, ok := w.Parameters["properties"].(map[string]any)
if !ok {
return nil, false
}
property, ok := properties[field].(map[string]any)
return property, ok
}
+367
View File
@@ -0,0 +1,367 @@
package workloads
import (
"fmt"
"math"
"sort"
)
// validateSchema checks the strict JSON Schema subset used by SDK manifests.
// It mirrors what the SDK registry enforces on the Python side: an object
// schema with additionalProperties=false, typed properties, and the keyword
// subset the coordinator understands (type, enum, required, minimum/maximum,
// minLength/maxLength, oneOf, not).
func validateSchema(workloadName string, schema map[string]any) error {
if schema == nil {
return fmt.Errorf("workload %q has no parameter schema", workloadName)
}
if err := validateSchemaNode(workloadName+".parameters_schema", schema); err != nil {
return err
}
if schemaType(schema) != "object" {
return fmt.Errorf("workload %q parameter schema must be an object schema", workloadName)
}
if additional, ok := schema["additionalProperties"].(bool); !ok || additional {
return fmt.Errorf("workload %q parameter schema must set additionalProperties=false", workloadName)
}
properties, ok := schema["properties"].(map[string]any)
if !ok {
return fmt.Errorf("workload %q parameter schema must declare properties", workloadName)
}
for name, property := range properties {
child, ok := property.(map[string]any)
if !ok {
return fmt.Errorf("workload %q parameter %q must be a schema object", workloadName, name)
}
if err := validateSchemaNode(name, child); err != nil {
return err
}
}
return nil
}
func validateSchemaNode(field string, node map[string]any) error {
for keyword := range node {
switch keyword {
case "type", "enum", "required", "minimum", "maximum", "exclusiveMinimum",
"exclusiveMaximum", "minLength", "maxLength", "properties",
"additionalProperties", "oneOf", "not", "default", "description",
"items", "minItems", "maxItems":
default:
return fmt.Errorf("%s uses unsupported JSON Schema keyword %q", field, keyword)
}
}
if rawType, ok := node["type"]; ok {
schemaType, ok := rawType.(string)
if !ok {
return fmt.Errorf("%s type must be a string", field)
}
switch schemaType {
case "string", "number", "integer", "boolean", "object", "array":
default:
return fmt.Errorf("%s has unknown type %q", field, schemaType)
}
}
for _, keyword := range []string{"minimum", "maximum", "exclusiveMinimum", "exclusiveMaximum"} {
if value, ok := node[keyword]; ok {
if number, ok := value.(float64); !ok || math.IsNaN(number) || math.IsInf(number, 0) {
return fmt.Errorf("%s %s must be a finite number", field, keyword)
}
}
}
for _, keyword := range []string{"minLength", "maxLength", "minItems", "maxItems"} {
if value, ok := node[keyword]; ok {
if number, ok := value.(float64); !ok || number < 0 || number != math.Trunc(number) {
return fmt.Errorf("%s %s must be a non-negative integer", field, keyword)
}
}
}
if required, ok := node["required"]; ok {
entries, ok := required.([]any)
if !ok {
return fmt.Errorf("%s required must be an array of strings", field)
}
for _, entry := range entries {
if _, ok := entry.(string); !ok {
return fmt.Errorf("%s required must be an array of strings", field)
}
}
}
if enums, ok := node["enum"]; ok {
entries, ok := enums.([]any)
if !ok || len(entries) == 0 {
return fmt.Errorf("%s enum must be a non-empty array", field)
}
}
if oneOf, ok := node["oneOf"]; ok {
entries, ok := oneOf.([]any)
if !ok || len(entries) == 0 {
return fmt.Errorf("%s oneOf must be a non-empty array", field)
}
for index, entry := range entries {
child, ok := entry.(map[string]any)
if !ok {
return fmt.Errorf("%s oneOf[%d] must be a schema object", field, index)
}
if err := validateSchemaNode(fmt.Sprintf("%s.oneOf[%d]", field, index), child); err != nil {
return err
}
}
}
if child, ok := node["not"]; ok {
not, ok := child.(map[string]any)
if !ok {
return fmt.Errorf("%s not must be a schema object", field)
}
if err := validateSchemaNode(field+".not", not); err != nil {
return err
}
}
if items, ok := node["items"]; ok {
child, ok := items.(map[string]any)
if !ok {
return fmt.Errorf("%s items must be a schema object", field)
}
if err := validateSchemaNode(field+".items", child); err != nil {
return err
}
}
if properties, ok := node["properties"]; ok {
entries, ok := properties.(map[string]any)
if !ok {
return fmt.Errorf("%s properties must be an object", field)
}
for name, property := range entries {
child, ok := property.(map[string]any)
if !ok {
return fmt.Errorf("%s property %q must be a schema object", field, name)
}
if err := validateSchemaNode(field+"."+name, child); err != nil {
return err
}
}
}
return nil
}
func schemaType(node map[string]any) string {
rawType, _ := node["type"].(string)
return rawType
}
// validateParameters checks values against the strict schema subset.
func validateParameters(workloadName string, schema map[string]any, parameters map[string]any) error {
properties, _ := schema["properties"].(map[string]any)
for name := range parameters {
if _, declared := properties[name]; !declared {
return fmt.Errorf("workload %q does not accept parameter %q", workloadName, name)
}
}
required, _ := schema["required"].([]any)
for _, name := range required {
field, _ := name.(string)
if _, present := parameters[field]; !present {
return fmt.Errorf("workload %q requires parameter %q", workloadName, field)
}
}
for name, value := range parameters {
property, declared := properties[name].(map[string]any)
if !declared {
continue
}
if err := validateProperty(workloadName+"."+name, property, value); err != nil {
return err
}
}
if oneOf, ok := schema["oneOf"].([]any); ok && len(oneOf) > 0 {
if err := validateOneOf(workloadName, oneOf, parameters); err != nil {
return err
}
}
return nil
}
func validateProperty(field string, property map[string]any, value any) error {
if enums, ok := property["enum"].([]any); ok {
for _, candidate := range enums {
if valuesEqual(candidate, value) {
return nil
}
}
return fmt.Errorf("%s must be one of the declared enum values", field)
}
switch schemaType(property) {
case "string":
text, ok := value.(string)
if !ok {
return fmt.Errorf("%s must be a string", field)
}
if minimum, ok := lengthBound(property["minLength"]); ok && len([]rune(text)) < minimum {
return fmt.Errorf("%s is shorter than the minimum length", field)
}
if maximum, ok := lengthBound(property["maxLength"]); ok && len([]rune(text)) > maximum {
return fmt.Errorf("%s exceeds the maximum length", field)
}
case "number", "integer":
number, ok := asFloat(value)
if !ok {
return fmt.Errorf("%s must be a number", field)
}
if schemaType(property) == "integer" && number != math.Trunc(number) {
return fmt.Errorf("%s must be an integer", field)
}
if minimum, ok := numberBound(property["minimum"]); ok && number < minimum {
return fmt.Errorf("%s is below the minimum", field)
}
if maximum, ok := numberBound(property["maximum"]); ok && number > maximum {
return fmt.Errorf("%s exceeds the maximum", field)
}
case "boolean":
if _, ok := value.(bool); !ok {
return fmt.Errorf("%s must be a boolean", field)
}
case "object":
child, ok := value.(map[string]any)
if !ok {
return fmt.Errorf("%s must be an object", field)
}
properties, _ := property["properties"].(map[string]any)
for name := range child {
if _, declared := properties[name]; !declared {
return fmt.Errorf("%s has undeclared field %q", field, name)
}
}
case "array":
items, ok := value.([]any)
if !ok {
return fmt.Errorf("%s must be an array", field)
}
if itemSchema, ok := property["items"].(map[string]any); ok {
for index, item := range items {
if err := validateProperty(fmt.Sprintf("%s[%d]", field, index), itemSchema, item); err != nil {
return err
}
}
}
case "":
// No type keyword: enum-only properties are handled above.
return fmt.Errorf("%s has no JSON Schema type", field)
}
return nil
}
func validateOneOf(workloadName string, oneOf []any, parameters map[string]any) error {
satisfied := 0
for _, candidate := range oneOf {
option, ok := candidate.(map[string]any)
if !ok {
continue
}
if optionSatisfied(option, parameters) {
satisfied++
}
}
if satisfied != 1 {
return fmt.Errorf("workload %q requires exactly one of the declared parameter alternatives", workloadName)
}
return nil
}
func optionSatisfied(option map[string]any, parameters map[string]any) bool {
if required, ok := option["required"].([]any); ok {
for _, name := range required {
field, _ := name.(string)
if _, present := parameters[field]; !present {
return false
}
}
}
if not, ok := option["not"].(map[string]any); ok {
if required, ok := not["required"].([]any); ok {
for _, name := range required {
field, _ := name.(string)
if _, present := parameters[field]; present {
return false
}
}
}
}
return true
}
func lengthBound(value any) (int, bool) {
number, ok := value.(float64)
if !ok || number != math.Trunc(number) {
return 0, false
}
return int(number), true
}
func numberBound(value any) (float64, bool) {
number, ok := asFloat(value)
if !ok || math.IsNaN(number) || math.IsInf(number, 0) {
return 0, false
}
return number, true
}
func asFloat(value any) (float64, bool) {
switch v := value.(type) {
case float64:
return v, true
case float32:
return float64(v), true
case int:
return float64(v), true
case int32:
return float64(v), true
case int64:
return float64(v), true
}
return 0, false
}
func valuesEqual(left, right any) bool {
switch l := left.(type) {
case float64:
r, ok := right.(float64)
return ok && l == r
case string:
r, ok := right.(string)
return ok && l == r
case bool:
r, ok := right.(bool)
return ok && l == r
case nil:
return right == nil
}
return false
}
// schemaDefaults collects the declared default for each property. The UI uses
// these to pre-fill controls that have no workload-declared UI default.
func schemaDefaults(schema map[string]any) map[string]any {
properties, _ := schema["properties"].(map[string]any)
defaults := map[string]any{}
for name, property := range properties {
child, ok := property.(map[string]any)
if !ok {
continue
}
if value, present := child["default"]; present {
defaults[name] = value
}
}
return defaults
}
// SortedFields returns the sorted declared parameter names.
func SortedFields(schema map[string]any) []string {
properties, _ := schema["properties"].(map[string]any)
names := make([]string, 0, len(properties))
for name := range properties {
names = append(names, name)
}
sort.Strings(names)
return names
}
@@ -0,0 +1,602 @@
{
"generated_by": "scimesh workload export",
"schema_version": 2,
"workloads": [
{
"capabilities": [
"descriptor-batch"
],
"description": "Compute a pinned set of RDKit 2D descriptors, one canonical CSV row per input molecule, in deterministic input order.",
"determinism": "byte_exact",
"enabled": true,
"inputs": {
"input": {
"allow_nested_collections": false,
"canonicalizer": "scimesh-tsv-v1",
"encoding": "utf-8",
"max_bytes": 10737418240,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/tab-separated-values",
"privacy_class": "project",
"ref": "molecule-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"required_columns": [
"canonical_smiles",
"chembl_id"
]
}
}
},
"name": "descriptor-batch",
"outputs": {
"result": {
"allow_nested_collections": false,
"canonicalizer": "descriptor-table-v1",
"encoding": "utf-8",
"max_bytes": 107374182400,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/csv",
"privacy_class": "project",
"ref": "descriptor-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"columns": [
"chembl_id",
"canonical_smiles",
"ExactMolWt",
"MolWt",
"HeavyAtomMolWt",
"HeavyAtomCount",
"NumHDonors",
"NumHAcceptors",
"NumRotatableBonds",
"NumHeteroatoms",
"NumRadicalElectrons",
"NumValenceElectrons",
"FractionCSP3",
"RingCount",
"NumAromaticRings",
"NumSaturatedRings",
"NumAliphaticRings",
"NumAromaticHeterocycles",
"NumSaturatedHeterocycles",
"NumAliphaticHeterocycles",
"NumAromaticCarbocycles",
"NumSaturatedCarbocycles",
"NumAliphaticCarbocycles",
"TPSA",
"LabuteASA",
"MolLogP",
"MolMR",
"BalabanJ",
"BertzCT",
"HallKierAlpha",
"Kappa1",
"Kappa2",
"Kappa3",
"Chi0",
"Chi1",
"Chi0n",
"Chi1n",
"Chi2n",
"Chi3n",
"Chi4n",
"Chi0v",
"Chi1v",
"Chi2v",
"Chi3v",
"Chi4v",
"PEOE_VSA1",
"PEOE_VSA2",
"PEOE_VSA3",
"PEOE_VSA4",
"PEOE_VSA5",
"PEOE_VSA6",
"PEOE_VSA7",
"PEOE_VSA8",
"PEOE_VSA9",
"PEOE_VSA10",
"PEOE_VSA11",
"PEOE_VSA12",
"PEOE_VSA13",
"PEOE_VSA14",
"SMR_VSA1",
"SMR_VSA2",
"SMR_VSA3",
"SMR_VSA4",
"SMR_VSA5",
"SMR_VSA6",
"SMR_VSA7",
"SMR_VSA8",
"SMR_VSA9",
"SMR_VSA10",
"SlogP_VSA1",
"SlogP_VSA2",
"SlogP_VSA3",
"SlogP_VSA4",
"SlogP_VSA5",
"SlogP_VSA6",
"SlogP_VSA7",
"SlogP_VSA8",
"SlogP_VSA9",
"SlogP_VSA10",
"SlogP_VSA11",
"SlogP_VSA12",
"NHOHCount",
"NOCount"
]
}
}
},
"parameters_schema": {
"additionalProperties": false,
"properties": {
"skip_invalid": {
"default": true,
"description": "Skip rows with invalid SMILES instead of failing",
"type": "boolean"
}
},
"type": "object"
},
"reduction": "ordered-concat",
"trust_modes": [
"trusted",
"untrusted_quorum"
],
"ui_elements": [
{
"default": true,
"field": "skip_invalid",
"group": "",
"help": "Skip rows with invalid SMILES instead of failing the shard.",
"label": "Skip invalid molecules",
"options": [],
"order": 1,
"placeholder": "",
"widget": "checkbox"
}
],
"upload_ready": true,
"verifier": "exact-artifact@1",
"version": "1.0.0"
},
{
"capabilities": [
"molwt-filter"
],
"description": "Filter molecules by exact RDKit molecular weight, one canonical CSV row per kept input molecule, in deterministic input order.",
"determinism": "byte_exact",
"enabled": true,
"inputs": {
"input": {
"allow_nested_collections": false,
"canonicalizer": "scimesh-tsv-v1",
"encoding": "utf-8",
"max_bytes": 10737418240,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/tab-separated-values",
"privacy_class": "project",
"ref": "molecule-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"required_columns": [
"canonical_smiles",
"chembl_id"
]
}
}
},
"name": "molwt-filter",
"outputs": {
"result": {
"allow_nested_collections": false,
"canonicalizer": "molwt-filtered-table-v1",
"encoding": "utf-8",
"max_bytes": 107374182400,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/csv",
"privacy_class": "project",
"ref": "molwt-filtered-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"columns": [
"chembl_id",
"canonical_smiles",
"molwt"
]
}
}
},
"parameters_schema": {
"additionalProperties": false,
"properties": {
"max_molwt": {
"description": "Keep molecules with MolWt <= this value",
"minimum": 0,
"type": "number"
},
"min_molwt": {
"description": "Keep molecules with MolWt >= this value",
"minimum": 0,
"type": "number"
},
"skip_invalid": {
"default": true,
"description": "Skip rows with invalid SMILES instead of failing",
"type": "boolean"
}
},
"type": "object"
},
"reduction": "ordered-concat",
"trust_modes": [
"trusted",
"untrusted_quorum"
],
"ui_elements": [
{
"default": null,
"field": "min_molwt",
"group": "",
"help": "Keep molecules with MolWt at least this value. Optional.",
"label": "Minimum molecular weight",
"options": [],
"order": 1,
"placeholder": "e.g. 100",
"widget": "number"
},
{
"default": null,
"field": "max_molwt",
"group": "",
"help": "Keep molecules with MolWt at most this value. Optional.",
"label": "Maximum molecular weight",
"options": [],
"order": 2,
"placeholder": "e.g. 600",
"widget": "number"
},
{
"default": true,
"field": "skip_invalid",
"group": "",
"help": "Skip rows with invalid SMILES instead of failing the shard.",
"label": "Skip invalid molecules",
"options": [],
"order": 3,
"placeholder": "",
"widget": "checkbox"
}
],
"upload_ready": true,
"verifier": "exact-artifact@1",
"version": "1.0.0"
},
{
"capabilities": [
"similarity-graph"
],
"description": "Exact sparse Tanimoto similarity graph over deterministic block pairs with a duplicate-safe, coverage-checked merge.",
"determinism": "byte_exact",
"enabled": true,
"inputs": {
"input": {
"allow_nested_collections": false,
"canonicalizer": "scimesh-tsv-v1",
"encoding": "utf-8",
"max_bytes": 10737418240,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/tab-separated-values",
"privacy_class": "project",
"ref": "molecule-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"required_columns": [
"canonical_smiles",
"chembl_id"
]
}
}
},
"name": "similarity-graph",
"outputs": {
"result": {
"allow_nested_collections": false,
"canonicalizer": "similarity-edge-table-v1",
"encoding": "utf-8",
"max_bytes": 107374182400,
"max_dimensions": [],
"max_records": 1000000000,
"media_type": "text/csv",
"privacy_class": "project",
"ref": "similarity-edge-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"columns": [
"source_id",
"target_id",
"similarity"
]
}
}
},
"parameters_schema": {
"additionalProperties": false,
"properties": {
"block_size": {
"minimum": 1,
"type": "integer"
},
"max_rows": {
"minimum": 1,
"type": "integer"
},
"threshold": {
"maximum": 1,
"minimum": 0,
"type": "number"
},
"threshold_direction": {
"enum": [
"greater",
"less"
]
}
},
"required": [
"threshold"
],
"type": "object"
},
"reduction": "ordered-concat",
"trust_modes": [
"trusted",
"untrusted_quorum"
],
"ui_elements": [
{
"default": null,
"field": "threshold",
"group": "",
"help": "Minimum (greater) or maximum (less) edge similarity. Required.",
"label": "Similarity threshold",
"options": [],
"order": 1,
"placeholder": "",
"widget": "number"
},
{
"default": "greater",
"field": "threshold_direction",
"group": "",
"help": "Whether to keep edges above (greater) or below (less) the threshold.",
"label": "Direction",
"options": [
"greater",
"less"
],
"order": 2,
"placeholder": "",
"widget": "select"
},
{
"default": 100,
"field": "block_size",
"group": "",
"help": "Deterministic block size for pair sharding.",
"label": "Block size",
"options": [],
"order": 3,
"placeholder": "",
"widget": "number"
}
],
"upload_ready": false,
"verifier": "exact-artifact@1",
"version": "1.0.0"
},
{
"capabilities": [
"similarity-search"
],
"description": "Exact top-k Tanimoto molecular similarity search over deterministic TSV shards with a bounded merge.",
"determinism": "byte_exact",
"enabled": true,
"inputs": {
"input": {
"allow_nested_collections": false,
"canonicalizer": "scimesh-tsv-v1",
"encoding": "utf-8",
"max_bytes": 10737418240,
"max_dimensions": [],
"max_records": 100000000,
"media_type": "text/tab-separated-values",
"privacy_class": "project",
"ref": "molecule-table@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"required_columns": [
"canonical_smiles",
"chembl_id"
]
}
}
},
"name": "similarity-search",
"outputs": {
"result": {
"allow_nested_collections": false,
"canonicalizer": "scimesh-search-result-v1",
"encoding": "utf-8",
"max_bytes": 1073741824,
"max_dimensions": [],
"max_records": 100000,
"media_type": "text/csv",
"privacy_class": "project",
"ref": "similarity-search-result@1",
"retention_class": "durable",
"streaming": false,
"validator": "delimited-table@1",
"validator_configuration": {
"columns": [
"rank",
"chembl_id",
"canonical_smiles",
"similarity"
]
}
}
},
"parameters_schema": {
"additionalProperties": false,
"oneOf": [
{
"not": {
"required": [
"query_smiles"
]
},
"required": [
"query_id"
]
},
{
"not": {
"required": [
"query_id"
]
},
"required": [
"query_smiles"
]
}
],
"properties": {
"max_rows": {
"minimum": 1,
"type": "integer"
},
"progress_every": {
"minimum": 0,
"type": "integer"
},
"query_id": {
"maxLength": 200,
"minLength": 1,
"type": "string"
},
"query_smiles": {
"maxLength": 200,
"minLength": 1,
"type": "string"
},
"threshold": {
"maximum": 1,
"minimum": 0,
"type": "number"
},
"threshold_direction": {
"enum": [
"greater",
"less"
]
},
"top_k": {
"minimum": 1,
"type": "integer"
}
},
"type": "object"
},
"reduction": "top-k",
"trust_modes": [
"trusted",
"untrusted_quorum"
],
"ui_elements": [
{
"default": null,
"field": "query_id",
"group": "",
"help": "ChEMBL id of the query molecule. Provide exactly one of id or SMILES.",
"label": "Query molecule id",
"options": [],
"order": 1,
"placeholder": "",
"widget": "text"
},
{
"default": null,
"field": "query_smiles",
"group": "",
"help": "SMILES of the query molecule. Provide exactly one of id or SMILES.",
"label": "Query molecule SMILES",
"options": [],
"order": 2,
"placeholder": "",
"widget": "text"
},
{
"default": 20,
"field": "top_k",
"group": "",
"help": "Number of most similar molecules to keep per shard (global merge keeps the best of these).",
"label": "Top k",
"options": [],
"order": 3,
"placeholder": "",
"widget": "number"
},
{
"default": "greater",
"field": "threshold_direction",
"group": "",
"help": "Keep molecules with similarity greater or less than the threshold.",
"label": "Direction",
"options": [
"greater",
"less"
],
"order": 4,
"placeholder": "",
"widget": "select"
},
{
"default": null,
"field": "threshold",
"group": "",
"help": "Optional similarity bound: results are filtered to this direction.",
"label": "Similarity threshold",
"options": [],
"order": 5,
"placeholder": "e.g. 0.8",
"widget": "number"
}
],
"upload_ready": true,
"verifier": "exact-artifact@1",
"version": "1.0.0"
}
]
}