Skip to content

Models

Create a custom model
client.Models.New(ctx, body) (*InferenceModel, error)
POST/v5/models
List custom models
client.Models.List(ctx, query) (*CursorPage[InferenceModel], error)
GET/v5/models
Update a custom model
client.Models.Update(ctx, modelID, body) (*InferenceModel, error)
PATCH/v5/models/{model_id}
Delete a custom model
client.Models.Delete(ctx, modelID) (*ModelDeleteResponse, error)
DELETE/v5/models/{model_id}
Get a custom model
client.Models.Get(ctx, modelID) (*InferenceModel, error)
GET/v5/models/{model_id}
ModelsExpand Collapse
type InferenceModel struct{…}
ID string

The unique identifier of the entity.

CreatedAt Time

The date and time when the entity was created in ISO format.

formatdate-time
CreatedByIdentityType InferenceModelCreatedByIdentityType

The type of identity that created the entity.

One of the following:
const InferenceModelCreatedByIdentityTypeUser InferenceModelCreatedByIdentityType = "user"
const InferenceModelCreatedByIdentityTypeServiceAccount InferenceModelCreatedByIdentityType = "service_account"
CreatedByUserID string

The user who originally created the entity.

One of the following:
const InferenceModelTypeGeneric InferenceModelType = "generic"
const InferenceModelTypeCompletion InferenceModelType = "completion"
const InferenceModelTypeChatCompletion InferenceModelType = "chat_completion"
One of the following:
const InferenceModelVendorOpenAI InferenceModelVendor = "openai"
const InferenceModelVendorCohere InferenceModelVendor = "cohere"
const InferenceModelVendorVertexAI InferenceModelVendor = "vertex_ai"
const InferenceModelVendorAnthropic InferenceModelVendor = "anthropic"
const InferenceModelVendorAzure InferenceModelVendor = "azure"
const InferenceModelVendorGemini InferenceModelVendor = "gemini"
const InferenceModelVendorLaunch InferenceModelVendor = "launch"
const InferenceModelVendorLlmengine InferenceModelVendor = "llmengine"
const InferenceModelVendorModelZoo InferenceModelVendor = "model_zoo"
const InferenceModelVendorBedrock InferenceModelVendor = "bedrock"
const InferenceModelVendorXai InferenceModelVendor = "xai"
const InferenceModelVendorFireworksAI InferenceModelVendor = "fireworks_ai"
Name string
Status InferenceModelStatus
One of the following:
const InferenceModelStatusFailed InferenceModelStatus = "failed"
const InferenceModelStatusReady InferenceModelStatus = "ready"
const InferenceModelStatusDeploying InferenceModelStatus = "deploying"
const InferenceModelStatusDeploymentTimeout InferenceModelStatus = "deployment_timeout"
ModelAvailability InferenceModelAvailabilityOptional
One of the following:
const InferenceModelAvailabilityUnknown InferenceModelAvailability = "unknown"
const InferenceModelAvailabilityAvailable InferenceModelAvailability = "available"
const InferenceModelAvailabilityUnavailable InferenceModelAvailability = "unavailable"
ModelMetadata map[string, any]Optional
Object InferenceModelObjectOptional
StatusReason stringOptional
VendorConfiguration InferenceModelVendorConfigurationUnionOptional
One of the following:
type LaunchVendorConfiguration struct{…}
ModelImage LaunchVendorConfigurationModelImage
Command []string
Registry string
Repository string
Tag string
EnvVars map[string, any]Optional
HealthcheckRoute stringOptional
PredictRoute stringOptional
ReadinessDelay int64Optional
RequestSchema map[string, any]Optional
ResponseSchema map[string, any]Optional
StreamingCommand []stringOptional
StreamingPredictRoute stringOptional
ModelInfra LaunchVendorConfigurationModelInfra
CPUs LaunchVendorConfigurationModelInfraCPUsUnionOptional
One of the following:
string
int64
EndpointType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraEndpointTypeAsync LaunchVendorConfigurationModelInfraEndpointType = "async"
const LaunchVendorConfigurationModelInfraEndpointTypeSync LaunchVendorConfigurationModelInfraEndpointType = "sync"
const LaunchVendorConfigurationModelInfraEndpointTypeStreaming LaunchVendorConfigurationModelInfraEndpointType = "streaming"
GPUType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaTeslaT4 LaunchVendorConfigurationModelInfraGPUType = "nvidia-tesla-t4"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA10 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a10"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100e LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100e"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_1g20gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-1g20gb"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_3g40gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-3g40gb"
GPUs int64Optional
HighPriority boolOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
PerWorker int64Optional
PublicInference boolOptional
Storage stringOptional
type LlmEngineVendorConfiguration struct{…}
Model string
ChatTemplateOverride stringOptional
CheckpointPath stringOptional
CPUs int64Optional
DefaultCallbackURL stringOptional
EndpointType stringOptional
GPUType stringOptional
GPUs int64Optional
HighPriority boolOptional
InferenceFramework stringOptional
InferenceFrameworkImageTag stringOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
NodesPerWorker int64Optional
NumShards int64Optional
PerWorker int64Optional
PostInferenceHooks []stringOptional
PublicInference boolOptional
Quantize stringOptional
Source stringOptional
Storage stringOptional
type InferenceModelAvailability string
One of the following:
const InferenceModelAvailabilityUnknown InferenceModelAvailability = "unknown"
const InferenceModelAvailabilityAvailable InferenceModelAvailability = "available"
const InferenceModelAvailabilityUnavailable InferenceModelAvailability = "unavailable"
type InferenceModelType string
One of the following:
const InferenceModelTypeGeneric InferenceModelType = "generic"
const InferenceModelTypeCompletion InferenceModelType = "completion"
const InferenceModelTypeChatCompletion InferenceModelType = "chat_completion"
type LaunchVendorConfiguration struct{…}
ModelImage LaunchVendorConfigurationModelImage
Command []string
Registry string
Repository string
Tag string
EnvVars map[string, any]Optional
HealthcheckRoute stringOptional
PredictRoute stringOptional
ReadinessDelay int64Optional
RequestSchema map[string, any]Optional
ResponseSchema map[string, any]Optional
StreamingCommand []stringOptional
StreamingPredictRoute stringOptional
ModelInfra LaunchVendorConfigurationModelInfra
CPUs LaunchVendorConfigurationModelInfraCPUsUnionOptional
One of the following:
string
int64
EndpointType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraEndpointTypeAsync LaunchVendorConfigurationModelInfraEndpointType = "async"
const LaunchVendorConfigurationModelInfraEndpointTypeSync LaunchVendorConfigurationModelInfraEndpointType = "sync"
const LaunchVendorConfigurationModelInfraEndpointTypeStreaming LaunchVendorConfigurationModelInfraEndpointType = "streaming"
GPUType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaTeslaT4 LaunchVendorConfigurationModelInfraGPUType = "nvidia-tesla-t4"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA10 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a10"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100e LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100e"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_1g20gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-1g20gb"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_3g40gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-3g40gb"
GPUs int64Optional
HighPriority boolOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
PerWorker int64Optional
PublicInference boolOptional
Storage stringOptional
type LlmEngineVendorConfiguration struct{…}
Model string
ChatTemplateOverride stringOptional
CheckpointPath stringOptional
CPUs int64Optional
DefaultCallbackURL stringOptional
EndpointType stringOptional
GPUType stringOptional
GPUs int64Optional
HighPriority boolOptional
InferenceFramework stringOptional
InferenceFrameworkImageTag stringOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
NodesPerWorker int64Optional
NumShards int64Optional
PerWorker int64Optional
PostInferenceHooks []stringOptional
PublicInference boolOptional
Quantize stringOptional
Source stringOptional
Storage stringOptional