Create chat completion
curl --request POST \
--url https://api.protege.sh/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"task": "<string>",
"model": "<string>",
"messages": [
{}
],
"stream": true,
"max_tokens": 123,
"temperature": 123,
"top_p": 123,
"stop": {},
"response_format": {},
"tools": [
{}
],
"tool_choice": {},
"seed": 123,
"user": "<string>",
"metadata": {}
}
'import requests
url = "https://api.protege.sh/v1/chat/completions"
payload = {
"task": "<string>",
"model": "<string>",
"messages": [{}],
"stream": True,
"max_tokens": 123,
"temperature": 123,
"top_p": 123,
"stop": {},
"response_format": {},
"tools": [{}],
"tool_choice": {},
"seed": 123,
"user": "<string>",
"metadata": {}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
task: '<string>',
model: '<string>',
messages: [{}],
stream: true,
max_tokens: 123,
temperature: 123,
top_p: 123,
stop: {},
response_format: {},
tools: [{}],
tool_choice: {},
seed: 123,
user: '<string>',
metadata: {}
})
};
fetch('https://api.protege.sh/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.protege.sh/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'task' => '<string>',
'model' => '<string>',
'messages' => [
[
]
],
'stream' => true,
'max_tokens' => 123,
'temperature' => 123,
'top_p' => 123,
'stop' => [
],
'response_format' => [
],
'tools' => [
[
]
],
'tool_choice' => [
],
'seed' => 123,
'user' => '<string>',
'metadata' => [
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.protege.sh/v1/chat/completions"
payload := strings.NewReader("{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.protege.sh/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.protege.sh/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"model": "<string>",
"choices": [
{
"index": 123,
"message": {},
"finish_reason": "<string>"
}
],
"usage": {}
}Reference
Create chat completion
POST /v1/chat/completions. The OpenAI chat completions body, plus task.
POST
/
v1
/
chat
/
completions
Create chat completion
curl --request POST \
--url https://api.protege.sh/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"task": "<string>",
"model": "<string>",
"messages": [
{}
],
"stream": true,
"max_tokens": 123,
"temperature": 123,
"top_p": 123,
"stop": {},
"response_format": {},
"tools": [
{}
],
"tool_choice": {},
"seed": 123,
"user": "<string>",
"metadata": {}
}
'import requests
url = "https://api.protege.sh/v1/chat/completions"
payload = {
"task": "<string>",
"model": "<string>",
"messages": [{}],
"stream": True,
"max_tokens": 123,
"temperature": 123,
"top_p": 123,
"stop": {},
"response_format": {},
"tools": [{}],
"tool_choice": {},
"seed": 123,
"user": "<string>",
"metadata": {}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
task: '<string>',
model: '<string>',
messages: [{}],
stream: true,
max_tokens: 123,
temperature: 123,
top_p: 123,
stop: {},
response_format: {},
tools: [{}],
tool_choice: {},
seed: 123,
user: '<string>',
metadata: {}
})
};
fetch('https://api.protege.sh/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.protege.sh/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'task' => '<string>',
'model' => '<string>',
'messages' => [
[
]
],
'stream' => true,
'max_tokens' => 123,
'temperature' => 123,
'top_p' => 123,
'stop' => [
],
'response_format' => [
],
'tools' => [
[
]
],
'tool_choice' => [
],
'seed' => 123,
'user' => '<string>',
'metadata' => [
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.protege.sh/v1/chat/completions"
payload := strings.NewReader("{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.protege.sh/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.protege.sh/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"task\": \"<string>\",\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": {},\n \"response_format\": {},\n \"tools\": [\n {}\n ],\n \"tool_choice\": {},\n \"seed\": 123,\n \"user\": \"<string>\",\n \"metadata\": {}\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"model": "<string>",
"choices": [
{
"index": 123,
"message": {},
"finish_reason": "<string>"
}
],
"usage": {}
}Creates a model response for a conversation. This is the only inference endpoint:
every workload goes through it, distinguished by its scope.
Log the first two. Seeing
Scope and rate-limit headers arrive with the response head, before the first
chunk.
Protégé parameters
string
Alias for the workload name, for clients where a body field is easier than a
header. Must match
^[a-z0-9_-]{1,63}$.x-protege-workload wins when both are present. Omitting both attributes the
call to the main workload of the default project. See
Projects and workloads.string
default:"default"
Project slug,
^[a-z0-9-]{1,63}$. Provisioned on first use.string
default:"main"
Workload name,
^[a-z0-9_-]{1,63}$. Provisioned on first use.string
required
A model id from the catalog, for example
deepseek-v4-flash. The
provider-pinned form (alibaba/qwen-flash) is accepted and normalises to the
same canonical id, which is what comes back in the response.Standard parameters
array
required
The conversation so far. Each message has a
role of system, user,
assistant or tool, and content.integer
Upper bound on generated tokens. Output bills at several times input on most
models, so this is the highest-leverage cost control you have before any
routing decision.
number
default:"1"
Sampling temperature between 0 and 2. Lower is more deterministic.
number
default:"1"
Nucleus sampling. Set this or
temperature, not both.string | array
Up to four sequences that halt generation.
object
Set
{"type": "json_object"} to constrain output to valid JSON. Schema-shaped
output is usually cheaper and more reliably scored than free text, which is why
it often clears an eval on a smaller model.array
Tool definitions the model may call, in OpenAI function-calling format.
string | object
auto, none, required, or a specific tool.integer
Best-effort determinism for repeated identical requests.
string
Stable end-user or tenant identifier. Use this to separate tenants within one
task rather than encoding the tenant into the task name.
object
Up to 16 string key-value pairs echoed back on the response. Useful for
carrying your own request or trace IDs.
Request
curl https://api.protege.sh/v1/chat/completions \
-H "Authorization: Bearer $PROTEGE_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"task": "invoice_extraction",
"model": "deepseek-v4-flash",
"messages": [
{"role": "system", "content": "Return only the total, as a number."},
{"role": "user", "content": "INVOICE #4412 ... TOTAL 1,284.00 USD"}
],
"max_tokens": 16,
"response_format": {"type": "json_object"}
}'
from openai import OpenAI
client = OpenAI(
base_url="https://api.protege.sh/v1",
api_key=os.environ["PROTEGE_API_KEY"],
)
resp = client.chat.completions.create(
model="deepseek-v4-flash",
messages=[
{"role": "system", "content": "Return only the total, as a number."},
{"role": "user", "content": invoice_text},
],
max_tokens=16,
extra_body={"task": "invoice_extraction"},
)
const resp = await client.chat.completions.create({
model: "deepseek-v4-flash",
messages: [
{ role: "system", content: "Return only the total, as a number." },
{ role: "user", content: invoiceText },
],
max_tokens: 16,
// @ts-expect-error - `task` is a Protégé extension
task: "invoice_extraction",
});
Response
string
Unique identifier for the completion.
string
Always
chat.completion.string
The model that actually served the call. With
model: "deepseek-v4-flash" this is the
resolved route, not the string you sent.array
object
prompt_tokens, completion_tokens and total_tokens.Response headers
| Header | Meaning |
|---|---|
x-protege-project | The project the call was attributed to |
x-protege-workload | The workload it was attributed to |
x-request-id | Identifier to quote when reporting a problem |
x-ratelimit-limit-requests | Requests permitted in the window |
x-ratelimit-remaining-requests | Requests left |
x-ratelimit-reset-requests | Seconds until reset |
default and main when you expected otherwise means
the header was dropped or task was stripped by your SDK.
Response
{
"id": "chatcmpl-9f2a7c31",
"object": "chat.completion",
"created": 1786531200,
"model": "deepseek-v4-flash",
"choices": [
{
"index": 0,
"message": { "role": "assistant", "content": "{\"total\": 1284.00}" },
"finish_reason": "stop"
}
],
"usage": { "prompt_tokens": 812, "completion_tokens": 11, "total_tokens": 823 }
}
Streaming
Setstream: true to receive chat.completion.chunk events as server-sent
events, terminated by data: [DONE].
curl https://api.protege.sh/v1/chat/completions \
-H "Authorization: Bearer $PROTEGE_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"task": "support_draft_reply",
"model": "deepseek-v4-flash",
"stream": true,
"messages": [{"role": "user", "content": "Draft a reply about a late order."}]
}'
Errors
Status codes, error shapes, and which are worth retrying.