Retire model version
Retire a fine-tuned version for good, by its number, such as router@3, or its id: it no longer answers or runs anything, and it no longer counts against the account’s storage. A name or an alias such as router@candidate is refused, since a retry could find another version behind it. Its record stays, and GET /v1/models/router lists it as retired. Its files are deleted after the days in limits.retired_days of GET /v1/decisionone, or sooner: when retiring leaves the account’s retired versions’ files taking more than its storage allowance, the oldest go then, apart from any retired in the last few minutes. Never while a fine-tune trains from it. A version in production is refused: take it out of production first. Retiring again changes nothing.
curl --request POST \
--url https://console.sqwish.ai/v1/models/{model_id}/retire \
--header 'Authorization: Bearer <token>'import requests
url = "https://console.sqwish.ai/v1/models/{model_id}/retire"
headers = {"Authorization": "Bearer <token>"}
response = requests.post(url, headers=headers)
print(response.text)const options = {method: 'POST', headers: {Authorization: 'Bearer <token>'}};
fetch('https://console.sqwish.ai/v1/models/{model_id}/retire', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"id": "<string>",
"status": "<string>",
"evaluation_key": "<string>",
"name": "<string>",
"description": "<string>",
"family": "<string>",
"size_order": 123,
"kind": "<string>",
"lora": "<string>",
"metrics": {},
"max_prompt_tokens": 123,
"max_questions": 123,
"calibration": "<string>",
"availability_reason": "<string>",
"available": true,
"access_requested": true,
"adapter_capacity": {
"max_rank": 123,
"slots": 123
},
"created_at": 123,
"retired_at": 123,
"job_id": "<string>",
"model_name": "<string>",
"version": 123,
"parent": "<string>",
"residency": {
"state": "hot",
"pinned": true
},
"state": "loaded",
"decision_price_nano": "<string>",
"base": "<string>",
"evaluation_set": {
"dataset": {
"id": "<string>",
"name": "<string>",
"deleted_at": 123
},
"training_examples": 123,
"validation_examples": 123,
"remembered_examples": 123
},
"request_latency": {
"median_ms": 123,
"samples": 123,
"median_is_lower_bound": true,
"start": "<string>",
"end": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"retryable": true,
"request_id": "<string>",
"details": [
{
"path": [
"<string>"
],
"message": "<string>",
"type": "<string>"
}
]
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"retryable": true,
"request_id": "<string>",
"details": [
{
"path": [
"<string>"
],
"message": "<string>",
"type": "<string>"
}
]
}
}Authorizations
Create an API key in the console. Required when accounts are enabled.
Path Parameters
Response
Successful Response
A base, an early-access catalogue entry, or a trained adapter.
Why it can't answer now: no_engine (its pool has no engine right now), not_served (this server doesn't serve it), or, for an adapter, base_mismatch, engine_unconfigured or engine_changed.
Show child attributes
Show child attributes
When its owner retired it. A retired version answers and runs nothing.
Show child attributes
Show child attributes
loaded, loading, asleep Exact integer string of nano-USD per billed input token from the current billing policy. Zero means free; null means unpriced.
An adapter's base model id: the base whose prices it pays.
Show child attributes
Show child attributes
Show child attributes
Show child attributes
curl --request POST \
--url https://console.sqwish.ai/v1/models/{model_id}/retire \
--header 'Authorization: Bearer <token>'import requests
url = "https://console.sqwish.ai/v1/models/{model_id}/retire"
headers = {"Authorization": "Bearer <token>"}
response = requests.post(url, headers=headers)
print(response.text)const options = {method: 'POST', headers: {Authorization: 'Bearer <token>'}};
fetch('https://console.sqwish.ai/v1/models/{model_id}/retire', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"id": "<string>",
"status": "<string>",
"evaluation_key": "<string>",
"name": "<string>",
"description": "<string>",
"family": "<string>",
"size_order": 123,
"kind": "<string>",
"lora": "<string>",
"metrics": {},
"max_prompt_tokens": 123,
"max_questions": 123,
"calibration": "<string>",
"availability_reason": "<string>",
"available": true,
"access_requested": true,
"adapter_capacity": {
"max_rank": 123,
"slots": 123
},
"created_at": 123,
"retired_at": 123,
"job_id": "<string>",
"model_name": "<string>",
"version": 123,
"parent": "<string>",
"residency": {
"state": "hot",
"pinned": true
},
"state": "loaded",
"decision_price_nano": "<string>",
"base": "<string>",
"evaluation_set": {
"dataset": {
"id": "<string>",
"name": "<string>",
"deleted_at": 123
},
"training_examples": 123,
"validation_examples": 123,
"remembered_examples": 123
},
"request_latency": {
"median_ms": 123,
"samples": 123,
"median_is_lower_bound": true,
"start": "<string>",
"end": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"retryable": true,
"request_id": "<string>",
"details": [
{
"path": [
"<string>"
],
"message": "<string>",
"type": "<string>"
}
]
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"retryable": true,
"request_id": "<string>",
"details": [
{
"path": [
"<string>"
],
"message": "<string>",
"type": "<string>"
}
]
}
}