Skip to content

List custom models

client.Models.List(ctx, query) (*CursorPage[InferenceModel], error)
GET/v5/models

List the custom model records registered in your account.

Returns a paginated list of the model records managed through this API — models your account deploys through the launch or llmengine serving vendors — optionally filtered by name and by model vendor, and scoped to the caller’s account. This is different from GET /v5/chat/completions/models, which lists the models available to invoke for chat completions; this endpoint returns the managed records along with their deployment status, not the catalog of callable completion models.

ParametersExpand Collapse
query ModelListParams
EndingBefore param.Field[string]Optional
Limit param.Field[int64]Optional
maximum10000
minimum1
ModelVendor param.Field[InferenceModelVendor]Optional
Name param.Field[string]Optional
SortBy param.Field[string]Optional
SortOrder param.Field[SortOrder]Optional
StartingAfter param.Field[string]Optional
ReturnsExpand Collapse
type InferenceModel struct{…}
ID string

The unique identifier of the entity.

CreatedAt Time

The date and time when the entity was created in ISO format.

formatdate-time
CreatedByIdentityType InferenceModelCreatedByIdentityType

The type of identity that created the entity.

One of the following:
const InferenceModelCreatedByIdentityTypeUser InferenceModelCreatedByIdentityType = "user"
const InferenceModelCreatedByIdentityTypeServiceAccount InferenceModelCreatedByIdentityType = "service_account"
CreatedByUserID string

The user who originally created the entity.

One of the following:
const InferenceModelTypeGeneric InferenceModelType = "generic"
const InferenceModelTypeCompletion InferenceModelType = "completion"
const InferenceModelTypeChatCompletion InferenceModelType = "chat_completion"
One of the following:
const InferenceModelVendorOpenAI InferenceModelVendor = "openai"
const InferenceModelVendorCohere InferenceModelVendor = "cohere"
const InferenceModelVendorVertexAI InferenceModelVendor = "vertex_ai"
const InferenceModelVendorAnthropic InferenceModelVendor = "anthropic"
const InferenceModelVendorAzure InferenceModelVendor = "azure"
const InferenceModelVendorGemini InferenceModelVendor = "gemini"
const InferenceModelVendorLaunch InferenceModelVendor = "launch"
const InferenceModelVendorLlmengine InferenceModelVendor = "llmengine"
const InferenceModelVendorModelZoo InferenceModelVendor = "model_zoo"
const InferenceModelVendorBedrock InferenceModelVendor = "bedrock"
const InferenceModelVendorXai InferenceModelVendor = "xai"
const InferenceModelVendorFireworksAI InferenceModelVendor = "fireworks_ai"
Name string
Status InferenceModelStatus
One of the following:
const InferenceModelStatusFailed InferenceModelStatus = "failed"
const InferenceModelStatusReady InferenceModelStatus = "ready"
const InferenceModelStatusDeploying InferenceModelStatus = "deploying"
const InferenceModelStatusDeploymentTimeout InferenceModelStatus = "deployment_timeout"
ModelAvailability InferenceModelAvailabilityOptional
One of the following:
const InferenceModelAvailabilityUnknown InferenceModelAvailability = "unknown"
const InferenceModelAvailabilityAvailable InferenceModelAvailability = "available"
const InferenceModelAvailabilityUnavailable InferenceModelAvailability = "unavailable"
ModelMetadata map[string, any]Optional
Object InferenceModelObjectOptional
StatusReason stringOptional
VendorConfiguration InferenceModelVendorConfigurationUnionOptional
One of the following:
type LaunchVendorConfiguration struct{…}
ModelImage LaunchVendorConfigurationModelImage
Command []string
Registry string
Repository string
Tag string
EnvVars map[string, any]Optional
HealthcheckRoute stringOptional
PredictRoute stringOptional
ReadinessDelay int64Optional
RequestSchema map[string, any]Optional
ResponseSchema map[string, any]Optional
StreamingCommand []stringOptional
StreamingPredictRoute stringOptional
ModelInfra LaunchVendorConfigurationModelInfra
CPUs LaunchVendorConfigurationModelInfraCPUsUnionOptional
One of the following:
string
int64
EndpointType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraEndpointTypeAsync LaunchVendorConfigurationModelInfraEndpointType = "async"
const LaunchVendorConfigurationModelInfraEndpointTypeSync LaunchVendorConfigurationModelInfraEndpointType = "sync"
const LaunchVendorConfigurationModelInfraEndpointTypeStreaming LaunchVendorConfigurationModelInfraEndpointType = "streaming"
GPUType stringOptional
One of the following:
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaTeslaT4 LaunchVendorConfigurationModelInfraGPUType = "nvidia-tesla-t4"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA10 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a10"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaAmpereA100e LaunchVendorConfigurationModelInfraGPUType = "nvidia-ampere-a100e"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100 LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_1g20gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-1g20gb"
const LaunchVendorConfigurationModelInfraGPUTypeNvidiaHopperH100_3g40gb LaunchVendorConfigurationModelInfraGPUType = "nvidia-hopper-h100-3g40gb"
GPUs int64Optional
HighPriority boolOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
PerWorker int64Optional
PublicInference boolOptional
Storage stringOptional
type LlmEngineVendorConfiguration struct{…}
Model string
ChatTemplateOverride stringOptional
CheckpointPath stringOptional
CPUs int64Optional
DefaultCallbackURL stringOptional
EndpointType stringOptional
GPUType stringOptional
GPUs int64Optional
HighPriority boolOptional
InferenceFramework stringOptional
InferenceFrameworkImageTag stringOptional
Labels map[string, string]Optional
MaxWorkers int64Optional
Memory stringOptional
MinWorkers int64Optional
NodesPerWorker int64Optional
NumShards int64Optional
PerWorker int64Optional
PostInferenceHooks []stringOptional
PublicInference boolOptional
Quantize stringOptional
Source stringOptional
Storage stringOptional

List custom models

package main

import (
  "context"
  "fmt"

  "github.com/scaleapi/sgp-dev-go"
  "github.com/scaleapi/sgp-dev-go/option"
)

func main() {
  client := sgpdev.NewClient(
    option.WithAPIKey("My API Key"),
    option.WithAccountID("My Account ID"),
  )
  page, err := client.Models.List(context.TODO(), sgpdev.ModelListParams{

  })
  if err != nil {
    panic(err.Error())
  }
  fmt.Printf("%+v\n", page)
}
{
  "has_more": true,
  "items": [
    {
      "id": "id",
      "created_at": "2019-12-27T18:11:19.117Z",
      "created_by_identity_type": "user",
      "created_by_user_id": "created_by_user_id",
      "model_type": "generic",
      "model_vendor": "openai",
      "name": "name",
      "status": "failed",
      "model_availability": "unknown",
      "model_metadata": {
        "foo": "bar"
      },
      "object": "model",
      "status_reason": "status_reason",
      "vendor_configuration": {
        "model_image": {
          "command": [
            "string"
          ],
          "registry": "registry",
          "repository": "repository",
          "tag": "tag",
          "env_vars": {
            "foo": "bar"
          },
          "healthcheck_route": "healthcheck_route",
          "predict_route": "predict_route",
          "readiness_delay": 0,
          "request_schema": {
            "foo": "bar"
          },
          "response_schema": {
            "foo": "bar"
          },
          "streaming_command": [
            "string"
          ],
          "streaming_predict_route": "streaming_predict_route"
        },
        "model_infra": {
          "cpus": "string",
          "endpoint_type": "async",
          "gpu_type": "nvidia-tesla-t4",
          "gpus": 0,
          "high_priority": true,
          "labels": {
            "foo": "string"
          },
          "max_workers": 0,
          "memory": "memory",
          "min_workers": 0,
          "per_worker": 0,
          "public_inference": true,
          "storage": "storage"
        }
      }
    }
  ],
  "total": 0,
  "limit": 0,
  "object": "list"
}
Returns Examples
{
  "has_more": true,
  "items": [
    {
      "id": "id",
      "created_at": "2019-12-27T18:11:19.117Z",
      "created_by_identity_type": "user",
      "created_by_user_id": "created_by_user_id",
      "model_type": "generic",
      "model_vendor": "openai",
      "name": "name",
      "status": "failed",
      "model_availability": "unknown",
      "model_metadata": {
        "foo": "bar"
      },
      "object": "model",
      "status_reason": "status_reason",
      "vendor_configuration": {
        "model_image": {
          "command": [
            "string"
          ],
          "registry": "registry",
          "repository": "repository",
          "tag": "tag",
          "env_vars": {
            "foo": "bar"
          },
          "healthcheck_route": "healthcheck_route",
          "predict_route": "predict_route",
          "readiness_delay": 0,
          "request_schema": {
            "foo": "bar"
          },
          "response_schema": {
            "foo": "bar"
          },
          "streaming_command": [
            "string"
          ],
          "streaming_predict_route": "streaming_predict_route"
        },
        "model_infra": {
          "cpus": "string",
          "endpoint_type": "async",
          "gpu_type": "nvidia-tesla-t4",
          "gpus": 0,
          "high_priority": true,
          "labels": {
            "foo": "string"
          },
          "max_workers": 0,
          "memory": "memory",
          "min_workers": 0,
          "per_worker": 0,
          "public_inference": true,
          "storage": "storage"
        }
      }
    }
  ],
  "total": 0,
  "limit": 0,
  "object": "list"
}