OpenAI Responses API Compatible Endpoint
curl --request POST \
--url https://api.nugen.in/api/v3/inference/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "model_01kmqm4nrn9fw6r",
"input": "<string>",
"instructions": "<string>",
"max_output_tokens": 123,
"temperature": 1,
"top_p": 123,
"stream": false,
"metadata": {},
"updated_at": "2023-11-07T05:31:56Z"
}
'import requests
url = "https://api.nugen.in/api/v3/inference/responses"
payload = {
"model": "model_01kmqm4nrn9fw6r",
"input": "<string>",
"instructions": "<string>",
"max_output_tokens": 123,
"temperature": 1,
"top_p": 123,
"stream": False,
"metadata": {},
"updated_at": "2023-11-07T05:31:56Z"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'model_01kmqm4nrn9fw6r',
input: '<string>',
instructions: '<string>',
max_output_tokens: 123,
temperature: 1,
top_p: 123,
stream: false,
metadata: {},
updated_at: '2023-11-07T05:31:56Z'
})
};
fetch('https://api.nugen.in/api/v3/inference/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.nugen.in/api/v3/inference/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'model_01kmqm4nrn9fw6r',
'input' => '<string>',
'instructions' => '<string>',
'max_output_tokens' => 123,
'temperature' => 1,
'top_p' => 123,
'stream' => false,
'metadata' => [
],
'updated_at' => '2023-11-07T05:31:56Z'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.nugen.in/api/v3/inference/responses"
payload := strings.NewReader("{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.nugen.in/api/v3/inference/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.nugen.in/api/v3/inference/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}"
response = http.request(request)
puts response.read_body{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Inference
OpenAI Responses API Compatible Endpoint
OpenAI responses.create-compatible endpoint.
Accepts the same shape as openai.responses.create and returns a Response object.
Uses the LiteLLM Python SDK for provider-agnostic routing.
Request Body:
model: Model ID (required)input: User input — a plain string or array of{"role": ..., "content": ...}dictsinstructions(optional): System-level instructionsmax_output_tokens(optional): Maximum tokens to generatetemperature(optional): Sampling temperature 0–2 (default: 1.0)top_p(optional): Nucleus sampling parameterstream(optional): Stream partial output as SSE (default: false)
Example Request:
POST /api/v3/inference/responses
Headers: {"Authorization": "Bearer <api_key>"}
{
"model": "Qwen/Qwen2.5-0.5B-Instruct",
"input": "What is the capital of France?",
"instructions": "You are a helpful assistant.",
"max_output_tokens": 200
}
Example Response:
{
"id": "resp_abc123",
"object": "response",
"created_at": 1704123600,
"model": "Qwen/Qwen2.5-0.5B-Instruct",
"output": [
{
"type": "message",
"role": "assistant",
"content": [{"type": "output_text", "text": "The capital of France is Paris."}]
}
],
"usage": {"input_tokens": 18, "output_tokens": 9, "total_tokens": 27}
}
POST
/
api
/
v3
/
inference
/
responses
OpenAI Responses API Compatible Endpoint
curl --request POST \
--url https://api.nugen.in/api/v3/inference/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "model_01kmqm4nrn9fw6r",
"input": "<string>",
"instructions": "<string>",
"max_output_tokens": 123,
"temperature": 1,
"top_p": 123,
"stream": false,
"metadata": {},
"updated_at": "2023-11-07T05:31:56Z"
}
'import requests
url = "https://api.nugen.in/api/v3/inference/responses"
payload = {
"model": "model_01kmqm4nrn9fw6r",
"input": "<string>",
"instructions": "<string>",
"max_output_tokens": 123,
"temperature": 1,
"top_p": 123,
"stream": False,
"metadata": {},
"updated_at": "2023-11-07T05:31:56Z"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'model_01kmqm4nrn9fw6r',
input: '<string>',
instructions: '<string>',
max_output_tokens: 123,
temperature: 1,
top_p: 123,
stream: false,
metadata: {},
updated_at: '2023-11-07T05:31:56Z'
})
};
fetch('https://api.nugen.in/api/v3/inference/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.nugen.in/api/v3/inference/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'model_01kmqm4nrn9fw6r',
'input' => '<string>',
'instructions' => '<string>',
'max_output_tokens' => 123,
'temperature' => 1,
'top_p' => 123,
'stream' => false,
'metadata' => [
],
'updated_at' => '2023-11-07T05:31:56Z'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.nugen.in/api/v3/inference/responses"
payload := strings.NewReader("{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.nugen.in/api/v3/inference/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.nugen.in/api/v3/inference/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"model_01kmqm4nrn9fw6r\",\n \"input\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 123,\n \"temperature\": 1,\n \"top_p\": 123,\n \"stream\": false,\n \"metadata\": {},\n \"updated_at\": \"2023-11-07T05:31:56Z\"\n}"
response = http.request(request)
puts response.read_body{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Body
application/json
Model ID to use for the response
Example:
"model_01kmqm4nrn9fw6r"
Input text or array of message dicts
System-level instructions (maps to system message)
Maximum tokens to generate
Sampling temperature
Required range:
0 <= x <= 2Nucleus sampling parameter
Whether to stream back partial progress
Optional metadata, ignored by the model
Response
Successful Response
Was this page helpful?