Create chat completion
curl --request POST \
--url https://api.hcompany.ai/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"messages": [
{}
],
"structured_outputs": {},
"chat_template_kwargs": {},
"reasoning_effort": "<string>",
"tools": [
{}
],
"tool_choice": "<string>",
"stream": true,
"max_tokens": 123,
"temperature": 123
}
'import requests
url = "https://api.hcompany.ai/v1/chat/completions"
payload = {
"model": "<string>",
"messages": [{}],
"structured_outputs": {},
"chat_template_kwargs": {},
"reasoning_effort": "<string>",
"tools": [{}],
"tool_choice": "<string>",
"stream": True,
"max_tokens": 123,
"temperature": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
messages: [{}],
structured_outputs: {},
chat_template_kwargs: {},
reasoning_effort: '<string>',
tools: [{}],
tool_choice: '<string>',
stream: true,
max_tokens: 123,
temperature: 123
})
};
fetch('https://api.hcompany.ai/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hcompany.ai/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'messages' => [
[
]
],
'structured_outputs' => [
],
'chat_template_kwargs' => [
],
'reasoning_effort' => '<string>',
'tools' => [
[
]
],
'tool_choice' => '<string>',
'stream' => true,
'max_tokens' => 123,
'temperature' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hcompany.ai/v1/chat/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hcompany.ai/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hcompany.ai/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}"
response = http.request(request)
puts response.read_body{
"choices[].message.content": "<string>",
"choices[].message.reasoning": "<string>",
"choices[].message.tool_calls": [
{}
],
"choices[].finish_reason": "<string>",
"usage": {}
}Endpoints
Create chat completion
OpenAI-compatible chat completion with Holo-specific structured outputs and reasoning.
POST
/
v1
/
chat
/
completions
Create chat completion
curl --request POST \
--url https://api.hcompany.ai/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"messages": [
{}
],
"structured_outputs": {},
"chat_template_kwargs": {},
"reasoning_effort": "<string>",
"tools": [
{}
],
"tool_choice": "<string>",
"stream": true,
"max_tokens": 123,
"temperature": 123
}
'import requests
url = "https://api.hcompany.ai/v1/chat/completions"
payload = {
"model": "<string>",
"messages": [{}],
"structured_outputs": {},
"chat_template_kwargs": {},
"reasoning_effort": "<string>",
"tools": [{}],
"tool_choice": "<string>",
"stream": True,
"max_tokens": 123,
"temperature": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
messages: [{}],
structured_outputs: {},
chat_template_kwargs: {},
reasoning_effort: '<string>',
tools: [{}],
tool_choice: '<string>',
stream: true,
max_tokens: 123,
temperature: 123
})
};
fetch('https://api.hcompany.ai/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.hcompany.ai/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'messages' => [
[
]
],
'structured_outputs' => [
],
'chat_template_kwargs' => [
],
'reasoning_effort' => '<string>',
'tools' => [
[
]
],
'tool_choice' => '<string>',
'stream' => true,
'max_tokens' => 123,
'temperature' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.hcompany.ai/v1/chat/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.hcompany.ai/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.hcompany.ai/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"messages\": [\n {}\n ],\n \"structured_outputs\": {},\n \"chat_template_kwargs\": {},\n \"reasoning_effort\": \"<string>\",\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<string>\",\n \"stream\": true,\n \"max_tokens\": 123,\n \"temperature\": 123\n}"
response = http.request(request)
puts response.read_body{
"choices[].message.content": "<string>",
"choices[].message.reasoning": "<string>",
"choices[].message.tool_calls": [
{}
],
"choices[].finish_reason": "<string>",
"usage": {}
}The single inference endpoint. It is OpenAI-compatible: the official OpenAI clients work as-is with
base_url pointed at https://api.hcompany.ai/v1/. Holo-specific behavior (structured outputs, the reasoning toggle) is controlled by extra body fields documented below.
Returns a chat completion object, or a stream of chunk objects when stream is true.
Body parameters
array
required
The conversation so far. Standard OpenAI message objects (
role, content); content can be a string or an array of text and image_url parts. Images accept HTTPS URLs or base64 data URIs (JPEG, PNG, WebP), up to 5 per request.object
Holo-specific. Constrain the response, at the decoding level, to a JSON object matching a schema: pass
{"json": <JSON Schema>}. The object is returned in message.content. Use this for the structured-output agent loop and element localization.object
Holo-specific.
{"enable_thinking": bool} toggles the reasoning channel. Use true for agent loops (Holo plans before acting), false for single-shot calls like grounding and OCR.string
How much the model plans before acting:
"low", "medium", or "high". "medium" is a sensible default for agent loops.array
OpenAI-style function declarations for native function calling. Supported by every model except
holo3-122b-a10b. A model supports it when tools is in its supported_features in GET /v1/models. Set tool_choice: "required" so the model acts on every step, and do not mix with structured_outputs.string
Standard OpenAI semantics. Use
"required" in function-calling agent loops.boolean
default:"false"
Stream the response as server-sent chunk events. Reasoning tokens arrive in
delta.reasoning, content in delta.content.integer
Output cap for this request. The hard ceiling is the model’s
max_output_length in GET /v1/models, 8,192 tokens on the Holo3 models.number
Sampling temperature. Use
0.0 for deterministic single-shot calls (localization, OCR); 0.8 works well in agent loops. Also supported: top_p, top_k, stop, frequency_penalty, presence_penalty, seed.Response
string
The action or answer: the constrained JSON object (structured-output mode) or the assistant text.
null when the model responded with tool_calls only.string
The thinking trace, present when thinking is enabled. Read it for visibility; never feed it back. Not carried between turns, see Reasoning.
array
Present in native function-calling mode only. Each call carries an
id and a function object with name and a JSON-encoded arguments string.string
stop, length (hit max_tokens or the model ceiling), or tool_calls.object
prompt_tokens, completion_tokens, total_tokens for the request. prompt_tokens_details.cached_tokens is the part of the prompt served from the cache, see Prompt caching. prompt_tokens_details is null when nothing was.Prompt caching
When a prompt starts with the same tokens as a recent request to the same model, such as a fixed system prompt or the growing history of an agent loop, the server reuses its work on that part. There is nothing to enable. The reused part is reported asusage.prompt_tokens_details.cached_tokens and billed at the cached input price. The rest of the prompt and all output are billed at the listed rates. Each model’s cached input price is on the Models page and in GET /v1/models as pricing.input_cache_read.
Caching is best effort. Keep the stable part of the prompt at the start, and expect the count to vary between requests.
Examples
import os
from openai import OpenAI
client = OpenAI(
base_url="https://api.hcompany.ai/v1/",
api_key=os.environ["HAI_API_KEY"],
)
resp = client.chat.completions.create(
model="holo4-27b",
messages=[{"role": "user", "content": "In one sentence, what is a computer-use agent?"}],
reasoning_effort="medium",
extra_body={"chat_template_kwargs": {"enable_thinking": True}},
)
print(resp.choices[0].message.content)
import OpenAI from "openai";
const client = new OpenAI({
baseURL: "https://api.hcompany.ai/v1/",
apiKey: process.env.HAI_API_KEY,
});
const resp = await client.chat.completions.create({
model: "holo4-27b",
messages: [{ role: "user", content: "In one sentence, what is a computer-use agent?" }],
reasoning_effort: "medium",
...({ chat_template_kwargs: { enable_thinking: true } } as any),
});
console.log(resp.choices[0].message.content);
curl https://api.hcompany.ai/v1/chat/completions \
-H "Authorization: Bearer $HAI_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "holo4-27b",
"messages": [{"role": "user", "content": "In one sentence, what is a computer-use agent?"}],
"reasoning_effort": "medium",
"chat_template_kwargs": {"enable_thinking": true}
}'
Streaming
stream = client.chat.completions.create(
model="holo4-27b",
messages=[{"role": "user", "content": "In one sentence, what is a computer-use agent?"}],
stream=True,
)
for chunk in stream:
delta = chunk.choices[0].delta
if delta.content:
print(delta.content, end="", flush=True)
const stream = await client.chat.completions.create({
model: "holo4-27b",
messages: [{ role: "user", content: "In one sentence, what is a computer-use agent?" }],
stream: true,
});
for await (const chunk of stream) {
const delta = chunk.choices[0]?.delta;
if (delta?.content) process.stdout.write(delta.content);
}
Was this page helpful?