curl --request POST \
--url https://api.egp.scale.com/v5/completions \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"model": "<string>",
"prompt": "<string>",
"stream": true,
"best_of": 123,
"echo": true,
"frequency_penalty": 123,
"logit_bias": {},
"logprobs": 123,
"max_tokens": 123,
"n": 123,
"presence_penalty": 123,
"seed": 123,
"stop": "<string>",
"stream_options": {},
"suffix": "<string>",
"temperature": 123,
"top_p": 123,
"user": "<string>"
}
'import requests
url = "https://api.egp.scale.com/v5/completions"
payload = {
"model": "<string>",
"prompt": "<string>",
"stream": True,
"best_of": 123,
"echo": True,
"frequency_penalty": 123,
"logit_bias": {},
"logprobs": 123,
"max_tokens": 123,
"n": 123,
"presence_penalty": 123,
"seed": 123,
"stop": "<string>",
"stream_options": {},
"suffix": "<string>",
"temperature": 123,
"top_p": 123,
"user": "<string>"
}
headers = {
"x-api-key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
prompt: '<string>',
stream: true,
best_of: 123,
echo: true,
frequency_penalty: 123,
logit_bias: {},
logprobs: 123,
max_tokens: 123,
n: 123,
presence_penalty: 123,
seed: 123,
stop: '<string>',
stream_options: {},
suffix: '<string>',
temperature: 123,
top_p: 123,
user: '<string>'
})
};
fetch('https://api.egp.scale.com/v5/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.egp.scale.com/v5/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'prompt' => '<string>',
'stream' => true,
'best_of' => 123,
'echo' => true,
'frequency_penalty' => 123,
'logit_bias' => [
],
'logprobs' => 123,
'max_tokens' => 123,
'n' => 123,
'presence_penalty' => 123,
'seed' => 123,
'stop' => '<string>',
'stream_options' => [
],
'suffix' => '<string>',
'temperature' => 123,
'top_p' => 123,
'user' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.egp.scale.com/v5/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.egp.scale.com/v5/completions")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.egp.scale.com/v5/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"choices": [
{
"finish_reason": "stop",
"index": 123,
"text": "<string>",
"logprobs": {
"text_offset": [
123
],
"token_logprobs": [
123
],
"tokens": [
"<string>"
],
"top_logprobs": [
{}
]
}
}
],
"created": 123,
"model": "<string>",
"object": "text_completion",
"system_fingerprint": "<string>",
"usage": {
"completion_tokens": 123,
"prompt_tokens": 123,
"total_tokens": 123,
"completion_tokens_details": {
"accepted_prediction_tokens": 123,
"audio_tokens": 123,
"reasoning_tokens": 123,
"rejected_prediction_tokens": 123
},
"prompt_tokens_details": {
"audio_tokens": 123,
"cached_tokens": 123
}
}
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Generate legacy text completion from prompt
Generates a legacy text completion from a raw prompt (a string or list of strings).
Use this endpoint for non-chat, prompt-in/text-out inference using the OpenAI text-completion
contract; use /v5/chat/completions when you have a structured messages array, /v5/responses for
the OpenAI Responses API, and /v5/inference for payloads that follow no OpenAI schema. The model is
selected from model given as vendor/name and routed to the matching per-vendor gateway. When
stream is set the response is delivered as server-sent events; otherwise a single text_completion
object is returned. Token usage is recorded for the account, read from the final chunk on streaming
responses.
curl --request POST \
--url https://api.egp.scale.com/v5/completions \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"model": "<string>",
"prompt": "<string>",
"stream": true,
"best_of": 123,
"echo": true,
"frequency_penalty": 123,
"logit_bias": {},
"logprobs": 123,
"max_tokens": 123,
"n": 123,
"presence_penalty": 123,
"seed": 123,
"stop": "<string>",
"stream_options": {},
"suffix": "<string>",
"temperature": 123,
"top_p": 123,
"user": "<string>"
}
'import requests
url = "https://api.egp.scale.com/v5/completions"
payload = {
"model": "<string>",
"prompt": "<string>",
"stream": True,
"best_of": 123,
"echo": True,
"frequency_penalty": 123,
"logit_bias": {},
"logprobs": 123,
"max_tokens": 123,
"n": 123,
"presence_penalty": 123,
"seed": 123,
"stop": "<string>",
"stream_options": {},
"suffix": "<string>",
"temperature": 123,
"top_p": 123,
"user": "<string>"
}
headers = {
"x-api-key": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
prompt: '<string>',
stream: true,
best_of: 123,
echo: true,
frequency_penalty: 123,
logit_bias: {},
logprobs: 123,
max_tokens: 123,
n: 123,
presence_penalty: 123,
seed: 123,
stop: '<string>',
stream_options: {},
suffix: '<string>',
temperature: 123,
top_p: 123,
user: '<string>'
})
};
fetch('https://api.egp.scale.com/v5/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.egp.scale.com/v5/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'prompt' => '<string>',
'stream' => true,
'best_of' => 123,
'echo' => true,
'frequency_penalty' => 123,
'logit_bias' => [
],
'logprobs' => 123,
'max_tokens' => 123,
'n' => 123,
'presence_penalty' => 123,
'seed' => 123,
'stop' => '<string>',
'stream_options' => [
],
'suffix' => '<string>',
'temperature' => 123,
'top_p' => 123,
'user' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.egp.scale.com/v5/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.egp.scale.com/v5/completions")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.egp.scale.com/v5/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"prompt\": \"<string>\",\n \"stream\": true,\n \"best_of\": 123,\n \"echo\": true,\n \"frequency_penalty\": 123,\n \"logit_bias\": {},\n \"logprobs\": 123,\n \"max_tokens\": 123,\n \"n\": 123,\n \"presence_penalty\": 123,\n \"seed\": 123,\n \"stop\": \"<string>\",\n \"stream_options\": {},\n \"suffix\": \"<string>\",\n \"temperature\": 123,\n \"top_p\": 123,\n \"user\": \"<string>\"\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"choices": [
{
"finish_reason": "stop",
"index": 123,
"text": "<string>",
"logprobs": {
"text_offset": [
123
],
"token_logprobs": [
123
],
"tokens": [
"<string>"
],
"top_logprobs": [
{}
]
}
}
],
"created": 123,
"model": "<string>",
"object": "text_completion",
"system_fingerprint": "<string>",
"usage": {
"completion_tokens": 123,
"prompt_tokens": 123,
"total_tokens": 123,
"completion_tokens_details": {
"accepted_prediction_tokens": 123,
"audio_tokens": 123,
"reasoning_tokens": 123,
"rejected_prediction_tokens": 123
},
"prompt_tokens_details": {
"audio_tokens": 123,
"cached_tokens": 123
}
}
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}Authorizations
Headers
Body
model specified as model_vendor/model, for example openai/gpt-4o
The prompt to generate completions for, encoded as a string
Whether to stream back partial progress. If set, tokens will be sent as data-only server-sent events.
Generates best_of completions server-side and returns the best one. Must be greater than n when used together.
Echo back the prompt in addition to the completion
Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text.
Modify the likelihood of specified tokens appearing in the completion. Maps tokens to bias values from -100 to 100.
Show child attributes
Show child attributes
Include log probabilities of the most likely tokens. Maximum value is 5.
The maximum number of tokens that can be generated in the completion.
How many completions to generate for each prompt.
Number between -2.0 and 2.0. Positive values penalize new tokens based on their presence in the text so far.
If specified, attempts to generate deterministic samples. Determinism is not guaranteed.
Up to 4 sequences where the API will stop generating further tokens.
Options for streaming response. Only set this when stream is True.
The suffix that comes after a completion of inserted text. Only supported for gpt-3.5-turbo-instruct.
Sampling temperature between 0 and 2. Higher values make output more random, lower more focused.
Alternative to temperature. Consider only tokens with top_p probability mass. Range 0-1.
A unique identifier representing your end-user, which can help OpenAI monitor and detect abuse.
Response
Successful Response
Show child attributes
Show child attributes
"text_completion"Usage statistics for the completion request.
Show child attributes
Show child attributes

