List custom models
models.list(ModelListParams**kwargs) -> SyncCursorPage[InferenceModel]
GET/v5/models
List the custom model records registered in your account.
Returns a paginated list of the model records managed through this API — models your account deploys through the launch or llmengine serving vendors — optionally filtered by name and by model vendor, and scoped to the caller’s account. This is different from GET /v5/chat/completions/models, which lists the models available to invoke for chat completions; this endpoint returns the managed records along with their deployment status, not the catalog of callable completion models.
List custom models
import os
from scale_gp_beta import SGPClient
client = SGPClient(
api_key=os.environ.get("SGP_API_KEY"), # This is the default and can be omitted
)
page = client.models.list()
page = page.items[0]
print(page.id){
"has_more": true,
"items": [
{
"id": "id",
"created_at": "2019-12-27T18:11:19.117Z",
"created_by_identity_type": "user",
"created_by_user_id": "created_by_user_id",
"model_type": "generic",
"model_vendor": "openai",
"name": "name",
"status": "failed",
"model_availability": "unknown",
"model_metadata": {
"foo": "bar"
},
"object": "model",
"status_reason": "status_reason",
"vendor_configuration": {
"model_image": {
"command": [
"string"
],
"registry": "registry",
"repository": "repository",
"tag": "tag",
"env_vars": {
"foo": "bar"
},
"healthcheck_route": "healthcheck_route",
"predict_route": "predict_route",
"readiness_delay": 0,
"request_schema": {
"foo": "bar"
},
"response_schema": {
"foo": "bar"
},
"streaming_command": [
"string"
],
"streaming_predict_route": "streaming_predict_route"
},
"model_infra": {
"cpus": "string",
"endpoint_type": "async",
"gpu_type": "nvidia-tesla-t4",
"gpus": 0,
"high_priority": true,
"labels": {
"foo": "string"
},
"max_workers": 0,
"memory": "memory",
"min_workers": 0,
"per_worker": 0,
"public_inference": true,
"storage": "storage"
}
}
}
],
"total": 0,
"limit": 0,
"object": "list"
}Returns Examples
{
"has_more": true,
"items": [
{
"id": "id",
"created_at": "2019-12-27T18:11:19.117Z",
"created_by_identity_type": "user",
"created_by_user_id": "created_by_user_id",
"model_type": "generic",
"model_vendor": "openai",
"name": "name",
"status": "failed",
"model_availability": "unknown",
"model_metadata": {
"foo": "bar"
},
"object": "model",
"status_reason": "status_reason",
"vendor_configuration": {
"model_image": {
"command": [
"string"
],
"registry": "registry",
"repository": "repository",
"tag": "tag",
"env_vars": {
"foo": "bar"
},
"healthcheck_route": "healthcheck_route",
"predict_route": "predict_route",
"readiness_delay": 0,
"request_schema": {
"foo": "bar"
},
"response_schema": {
"foo": "bar"
},
"streaming_command": [
"string"
],
"streaming_predict_route": "streaming_predict_route"
},
"model_infra": {
"cpus": "string",
"endpoint_type": "async",
"gpu_type": "nvidia-tesla-t4",
"gpus": 0,
"high_priority": true,
"labels": {
"foo": "string"
},
"max_workers": 0,
"memory": "memory",
"min_workers": 0,
"per_worker": 0,
"public_inference": true,
"storage": "storage"
}
}
}
],
"total": 0,
"limit": 0,
"object": "list"
}