Update a custom model
Update a custom model record; vendor-configuration changes are applied asynchronously by redeploying the model.
This supports three kinds of update: changing model metadata only, renaming the model, and changing the vendor configuration. A vendor-configuration change is asynchronous — it puts the model back into a deploying status, records an update job, and starts a Temporal workflow to redeploy, so the new configuration is not live when this returns; metadata-only and rename changes take effect immediately. The vendor configuration supplied must match the model’s own vendor (launch or llmengine), and only those two vendors are supported. A model that is currently deploying cannot be modified and the request fails until deployment finishes. When renaming with on_conflict set to swap, the name is exchanged with an existing model of the same name and vendor instead of failing on the uniqueness constraint.
Update a custom model
curl https://api.egp.scale.com/v5/models/$MODEL_ID \
-X PATCH \
-H 'Content-Type: application/json' \
-H "x-api-key: $SGP_API_KEY" \
-d '{
"model_metadata": {
"foo": "bar"
}
}'{
"id": "id",
"created_at": "2019-12-27T18:11:19.117Z",
"created_by_identity_type": "user",
"created_by_user_id": "created_by_user_id",
"model_type": "generic",
"model_vendor": "openai",
"name": "name",
"status": "failed",
"model_availability": "unknown",
"model_metadata": {
"foo": "bar"
},
"object": "model",
"status_reason": "status_reason",
"vendor_configuration": {
"model_image": {
"command": [
"string"
],
"registry": "registry",
"repository": "repository",
"tag": "tag",
"env_vars": {
"foo": "bar"
},
"healthcheck_route": "healthcheck_route",
"predict_route": "predict_route",
"readiness_delay": 0,
"request_schema": {
"foo": "bar"
},
"response_schema": {
"foo": "bar"
},
"streaming_command": [
"string"
],
"streaming_predict_route": "streaming_predict_route"
},
"model_infra": {
"cpus": "string",
"endpoint_type": "async",
"gpu_type": "nvidia-tesla-t4",
"gpus": 0,
"high_priority": true,
"labels": {
"foo": "string"
},
"max_workers": 0,
"memory": "memory",
"min_workers": 0,
"per_worker": 0,
"public_inference": true,
"storage": "storage"
}
}
}Returns Examples
{
"id": "id",
"created_at": "2019-12-27T18:11:19.117Z",
"created_by_identity_type": "user",
"created_by_user_id": "created_by_user_id",
"model_type": "generic",
"model_vendor": "openai",
"name": "name",
"status": "failed",
"model_availability": "unknown",
"model_metadata": {
"foo": "bar"
},
"object": "model",
"status_reason": "status_reason",
"vendor_configuration": {
"model_image": {
"command": [
"string"
],
"registry": "registry",
"repository": "repository",
"tag": "tag",
"env_vars": {
"foo": "bar"
},
"healthcheck_route": "healthcheck_route",
"predict_route": "predict_route",
"readiness_delay": 0,
"request_schema": {
"foo": "bar"
},
"response_schema": {
"foo": "bar"
},
"streaming_command": [
"string"
],
"streaming_predict_route": "streaming_predict_route"
},
"model_infra": {
"cpus": "string",
"endpoint_type": "async",
"gpu_type": "nvidia-tesla-t4",
"gpus": 0,
"high_priority": true,
"labels": {
"foo": "string"
},
"max_workers": 0,
"memory": "memory",
"min_workers": 0,
"per_worker": 0,
"public_inference": true,
"storage": "storage"
}
}
}